mirror of
https://github.com/neondatabase/neon.git
synced 2026-08-14 10:11:33 +00:00
Compare commits
25 Commits
| Author | SHA1 | Date | |
|---|---|---|---|
| 10774a5132 | |||
| bb18c9958a | |||
| e764fbf9f9 | |||
| d37f7a0dd2 | |||
| 20137d9588 | |||
| 634be4f4e0 | |||
| d340cf3721 | |||
| 1741edf933 | |||
| 269e20aeab | |||
| 91435006bd | |||
| b263510866 | |||
| e418fc6dc3 | |||
| 434eaadbe3 | |||
| 6fb7edf494 | |||
| 505aa242ac | |||
| 1c516906e7 | |||
| 7d7cd8375c | |||
| c92b7543b5 | |||
| dbf88cf2d7 | |||
| f1db87ac36 | |||
| 3f9defbfb4 | |||
| c7143dbde6 | |||
| cbf9a40889 | |||
| 10aba174c9 | |||
| ab2ea8cfa5 |
Generated
+11
@@ -2617,6 +2617,16 @@ dependencies = [
|
|||||||
"windows-sys 0.45.0",
|
"windows-sys 0.45.0",
|
||||||
]
|
]
|
||||||
|
|
||||||
|
[[package]]
|
||||||
|
name = "pbkdf2"
|
||||||
|
version = "0.12.1"
|
||||||
|
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||||
|
checksum = "f0ca0b5a68607598bf3bad68f32227a8164f6254833f84eafaac409cd6746c31"
|
||||||
|
dependencies = [
|
||||||
|
"digest",
|
||||||
|
"hmac",
|
||||||
|
]
|
||||||
|
|
||||||
[[package]]
|
[[package]]
|
||||||
name = "peeking_take_while"
|
name = "peeking_take_while"
|
||||||
version = "0.1.2"
|
version = "0.1.2"
|
||||||
@@ -3010,6 +3020,7 @@ dependencies = [
|
|||||||
"once_cell",
|
"once_cell",
|
||||||
"opentelemetry",
|
"opentelemetry",
|
||||||
"parking_lot 0.12.1",
|
"parking_lot 0.12.1",
|
||||||
|
"pbkdf2",
|
||||||
"pin-project-lite",
|
"pin-project-lite",
|
||||||
"postgres-native-tls",
|
"postgres-native-tls",
|
||||||
"postgres_backend",
|
"postgres_backend",
|
||||||
|
|||||||
@@ -86,6 +86,7 @@ opentelemetry = "0.18.0"
|
|||||||
opentelemetry-otlp = { version = "0.11.0", default_features=false, features = ["http-proto", "trace", "http", "reqwest-client"] }
|
opentelemetry-otlp = { version = "0.11.0", default_features=false, features = ["http-proto", "trace", "http", "reqwest-client"] }
|
||||||
opentelemetry-semantic-conventions = "0.10.0"
|
opentelemetry-semantic-conventions = "0.10.0"
|
||||||
parking_lot = "0.12"
|
parking_lot = "0.12"
|
||||||
|
pbkdf2 = "0.12.1"
|
||||||
pin-project-lite = "0.2"
|
pin-project-lite = "0.2"
|
||||||
prometheus = {version = "0.13", default_features=false, features = ["process"]} # removes protobuf dependency
|
prometheus = {version = "0.13", default_features=false, features = ["process"]} # removes protobuf dependency
|
||||||
prost = "0.11"
|
prost = "0.11"
|
||||||
|
|||||||
+22
-2
@@ -189,8 +189,8 @@ RUN wget https://github.com/df7cb/postgresql-unit/archive/refs/tags/7.7.tar.gz -
|
|||||||
FROM build-deps AS vector-pg-build
|
FROM build-deps AS vector-pg-build
|
||||||
COPY --from=pg-build /usr/local/pgsql/ /usr/local/pgsql/
|
COPY --from=pg-build /usr/local/pgsql/ /usr/local/pgsql/
|
||||||
|
|
||||||
RUN wget https://github.com/pgvector/pgvector/archive/refs/tags/v0.4.0.tar.gz -O pgvector.tar.gz && \
|
RUN wget https://github.com/pgvector/pgvector/archive/refs/tags/v0.4.4.tar.gz -O pgvector.tar.gz && \
|
||||||
echo "b76cf84ddad452cc880a6c8c661d137ddd8679c000a16332f4f03ecf6e10bcc8 pgvector.tar.gz" | sha256sum --check && \
|
echo "1cb70a63f8928e396474796c22a20be9f7285a8a013009deb8152445b61b72e6 pgvector.tar.gz" | sha256sum --check && \
|
||||||
mkdir pgvector-src && cd pgvector-src && tar xvzf ../pgvector.tar.gz --strip-components=1 -C . && \
|
mkdir pgvector-src && cd pgvector-src && tar xvzf ../pgvector.tar.gz --strip-components=1 -C . && \
|
||||||
make -j $(getconf _NPROCESSORS_ONLN) PG_CONFIG=/usr/local/pgsql/bin/pg_config && \
|
make -j $(getconf _NPROCESSORS_ONLN) PG_CONFIG=/usr/local/pgsql/bin/pg_config && \
|
||||||
make -j $(getconf _NPROCESSORS_ONLN) install PG_CONFIG=/usr/local/pgsql/bin/pg_config && \
|
make -j $(getconf _NPROCESSORS_ONLN) install PG_CONFIG=/usr/local/pgsql/bin/pg_config && \
|
||||||
@@ -515,6 +515,25 @@ RUN wget https://github.com/ChenHuajun/pg_roaringbitmap/archive/refs/tags/v0.5.4
|
|||||||
make -j $(getconf _NPROCESSORS_ONLN) install && \
|
make -j $(getconf _NPROCESSORS_ONLN) install && \
|
||||||
echo 'trusted = true' >> /usr/local/pgsql/share/extension/roaringbitmap.control
|
echo 'trusted = true' >> /usr/local/pgsql/share/extension/roaringbitmap.control
|
||||||
|
|
||||||
|
#########################################################################################
|
||||||
|
#
|
||||||
|
# Layer "pg-embedding-pg-build"
|
||||||
|
# compile pg_embedding extension
|
||||||
|
#
|
||||||
|
#########################################################################################
|
||||||
|
FROM build-deps AS pg-embedding-pg-build
|
||||||
|
COPY --from=pg-build /usr/local/pgsql/ /usr/local/pgsql/
|
||||||
|
|
||||||
|
ENV PATH "/usr/local/pgsql/bin/:$PATH"
|
||||||
|
# 2465f831ea1f8d49c1d74f8959adb7fc277d70cd made on 05/07/2023
|
||||||
|
# There is no release tag yet
|
||||||
|
RUN wget https://github.com/neondatabase/pg_embedding/archive/2465f831ea1f8d49c1d74f8959adb7fc277d70cd.tar.gz -O pg_embedding.tar.gz && \
|
||||||
|
echo "047af2b1f664a1e6e37867bd4eeaf5934fa27d6ba3d6c4461efa388ddf7cd1d5 pg_embedding.tar.gz" | sha256sum --check && \
|
||||||
|
mkdir pg_embedding-src && cd pg_embedding-src && tar xvzf ../pg_embedding.tar.gz --strip-components=1 -C . && \
|
||||||
|
make -j $(getconf _NPROCESSORS_ONLN) && \
|
||||||
|
make -j $(getconf _NPROCESSORS_ONLN) install && \
|
||||||
|
echo 'trusted = true' >> /usr/local/pgsql/share/extension/embedding.control
|
||||||
|
|
||||||
#########################################################################################
|
#########################################################################################
|
||||||
#
|
#
|
||||||
# Layer "pg-anon-pg-build"
|
# Layer "pg-anon-pg-build"
|
||||||
@@ -671,6 +690,7 @@ COPY --from=pg-pgx-ulid-build /usr/local/pgsql/ /usr/local/pgsql/
|
|||||||
COPY --from=rdkit-pg-build /usr/local/pgsql/ /usr/local/pgsql/
|
COPY --from=rdkit-pg-build /usr/local/pgsql/ /usr/local/pgsql/
|
||||||
COPY --from=pg-uuidv7-pg-build /usr/local/pgsql/ /usr/local/pgsql/
|
COPY --from=pg-uuidv7-pg-build /usr/local/pgsql/ /usr/local/pgsql/
|
||||||
COPY --from=pg-roaringbitmap-pg-build /usr/local/pgsql/ /usr/local/pgsql/
|
COPY --from=pg-roaringbitmap-pg-build /usr/local/pgsql/ /usr/local/pgsql/
|
||||||
|
COPY --from=pg-embedding-pg-build /usr/local/pgsql/ /usr/local/pgsql/
|
||||||
COPY pgxn/ pgxn/
|
COPY pgxn/ pgxn/
|
||||||
|
|
||||||
RUN make -j $(getconf _NPROCESSORS_ONLN) \
|
RUN make -j $(getconf _NPROCESSORS_ONLN) \
|
||||||
|
|||||||
@@ -516,9 +516,9 @@ impl ComputeNode {
|
|||||||
self.prepare_pgdata(&compute_state)?;
|
self.prepare_pgdata(&compute_state)?;
|
||||||
|
|
||||||
let start_time = Utc::now();
|
let start_time = Utc::now();
|
||||||
|
|
||||||
let pg = self.start_postgres(pspec.storage_auth_token.clone())?;
|
let pg = self.start_postgres(pspec.storage_auth_token.clone())?;
|
||||||
|
|
||||||
|
let config_time = Utc::now();
|
||||||
if pspec.spec.mode == ComputeMode::Primary && !pspec.spec.skip_pg_catalog_updates {
|
if pspec.spec.mode == ComputeMode::Primary && !pspec.spec.skip_pg_catalog_updates {
|
||||||
self.apply_config(&compute_state)?;
|
self.apply_config(&compute_state)?;
|
||||||
}
|
}
|
||||||
@@ -526,11 +526,16 @@ impl ComputeNode {
|
|||||||
let startup_end_time = Utc::now();
|
let startup_end_time = Utc::now();
|
||||||
{
|
{
|
||||||
let mut state = self.state.lock().unwrap();
|
let mut state = self.state.lock().unwrap();
|
||||||
state.metrics.config_ms = startup_end_time
|
state.metrics.start_postgres_ms = config_time
|
||||||
.signed_duration_since(start_time)
|
.signed_duration_since(start_time)
|
||||||
.to_std()
|
.to_std()
|
||||||
.unwrap()
|
.unwrap()
|
||||||
.as_millis() as u64;
|
.as_millis() as u64;
|
||||||
|
state.metrics.config_ms = startup_end_time
|
||||||
|
.signed_duration_since(config_time)
|
||||||
|
.to_std()
|
||||||
|
.unwrap()
|
||||||
|
.as_millis() as u64;
|
||||||
state.metrics.total_startup_ms = startup_end_time
|
state.metrics.total_startup_ms = startup_end_time
|
||||||
.signed_duration_since(compute_state.start_time)
|
.signed_duration_since(compute_state.start_time)
|
||||||
.to_std()
|
.to_std()
|
||||||
|
|||||||
@@ -189,7 +189,7 @@ services:
|
|||||||
- "/bin/bash"
|
- "/bin/bash"
|
||||||
- "-c"
|
- "-c"
|
||||||
command:
|
command:
|
||||||
- "until pg_isready -h compute -p 55433 ; do
|
- "until pg_isready -h compute -p 55433 -U cloud_admin ; do
|
||||||
echo 'Waiting to start compute...' && sleep 1;
|
echo 'Waiting to start compute...' && sleep 1;
|
||||||
done"
|
done"
|
||||||
depends_on:
|
depends_on:
|
||||||
|
|||||||
@@ -48,6 +48,7 @@ Creating docker-compose_storage_broker_1 ... done
|
|||||||
2. connect compute node
|
2. connect compute node
|
||||||
```
|
```
|
||||||
$ echo "localhost:55433:postgres:cloud_admin:cloud_admin" >> ~/.pgpass
|
$ echo "localhost:55433:postgres:cloud_admin:cloud_admin" >> ~/.pgpass
|
||||||
|
$ chmod 600 ~/.pgpass
|
||||||
$ psql -h localhost -p 55433 -U cloud_admin
|
$ psql -h localhost -p 55433 -U cloud_admin
|
||||||
postgres=# CREATE TABLE t(key int primary key, value text);
|
postgres=# CREATE TABLE t(key int primary key, value text);
|
||||||
CREATE TABLE
|
CREATE TABLE
|
||||||
|
|||||||
@@ -71,6 +71,7 @@ pub struct ComputeMetrics {
|
|||||||
pub wait_for_spec_ms: u64,
|
pub wait_for_spec_ms: u64,
|
||||||
pub sync_safekeepers_ms: u64,
|
pub sync_safekeepers_ms: u64,
|
||||||
pub basebackup_ms: u64,
|
pub basebackup_ms: u64,
|
||||||
|
pub start_postgres_ms: u64,
|
||||||
pub config_ms: u64,
|
pub config_ms: u64,
|
||||||
pub total_startup_ms: u64,
|
pub total_startup_ms: u64,
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -57,9 +57,9 @@ pub fn slru_may_delete_clogsegment(segpage: u32, cutoff_page: u32) -> bool {
|
|||||||
// Multixact utils
|
// Multixact utils
|
||||||
|
|
||||||
pub fn mx_offset_to_flags_offset(xid: MultiXactId) -> usize {
|
pub fn mx_offset_to_flags_offset(xid: MultiXactId) -> usize {
|
||||||
((xid / pg_constants::MULTIXACT_MEMBERS_PER_MEMBERGROUP as u32) as u16
|
((xid / pg_constants::MULTIXACT_MEMBERS_PER_MEMBERGROUP as u32)
|
||||||
% pg_constants::MULTIXACT_MEMBERGROUPS_PER_PAGE
|
% pg_constants::MULTIXACT_MEMBERGROUPS_PER_PAGE as u32
|
||||||
* pg_constants::MULTIXACT_MEMBERGROUP_SIZE) as usize
|
* pg_constants::MULTIXACT_MEMBERGROUP_SIZE as u32) as usize
|
||||||
}
|
}
|
||||||
|
|
||||||
pub fn mx_offset_to_flags_bitshift(xid: MultiXactId) -> u16 {
|
pub fn mx_offset_to_flags_bitshift(xid: MultiXactId) -> u16 {
|
||||||
|
|||||||
@@ -234,14 +234,18 @@ pub async fn collect_metrics_iteration(
|
|||||||
// Note that this metric is calculated in a separate bgworker
|
// Note that this metric is calculated in a separate bgworker
|
||||||
// Here we only use cached value, which may lag behind the real latest one
|
// Here we only use cached value, which may lag behind the real latest one
|
||||||
let tenant_synthetic_size = tenant.get_cached_synthetic_size();
|
let tenant_synthetic_size = tenant.get_cached_synthetic_size();
|
||||||
current_metrics.push((
|
|
||||||
PageserverConsumptionMetricsKey {
|
if tenant_synthetic_size != 0 {
|
||||||
tenant_id,
|
// only send non-zeroes because otherwise these show up as errors in logs
|
||||||
timeline_id: None,
|
current_metrics.push((
|
||||||
metric: SYNTHETIC_STORAGE_SIZE,
|
PageserverConsumptionMetricsKey {
|
||||||
},
|
tenant_id,
|
||||||
tenant_synthetic_size,
|
timeline_id: None,
|
||||||
));
|
metric: SYNTHETIC_STORAGE_SIZE,
|
||||||
|
},
|
||||||
|
tenant_synthetic_size,
|
||||||
|
));
|
||||||
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
// Filter metrics, unless we want to send all metrics, including cached ones.
|
// Filter metrics, unless we want to send all metrics, including cached ones.
|
||||||
|
|||||||
@@ -110,7 +110,6 @@ pub fn launch_disk_usage_global_eviction_task(
|
|||||||
|
|
||||||
disk_usage_eviction_task(&state, task_config, storage, &conf.tenants_path(), cancel)
|
disk_usage_eviction_task(&state, task_config, storage, &conf.tenants_path(), cancel)
|
||||||
.await;
|
.await;
|
||||||
info!("disk usage based eviction task finishing");
|
|
||||||
Ok(())
|
Ok(())
|
||||||
},
|
},
|
||||||
);
|
);
|
||||||
@@ -126,13 +125,16 @@ async fn disk_usage_eviction_task(
|
|||||||
tenants_dir: &Path,
|
tenants_dir: &Path,
|
||||||
cancel: CancellationToken,
|
cancel: CancellationToken,
|
||||||
) {
|
) {
|
||||||
|
scopeguard::defer! {
|
||||||
|
info!("disk usage based eviction task finishing");
|
||||||
|
};
|
||||||
|
|
||||||
use crate::tenant::tasks::random_init_delay;
|
use crate::tenant::tasks::random_init_delay;
|
||||||
{
|
{
|
||||||
if random_init_delay(task_config.period, &cancel)
|
if random_init_delay(task_config.period, &cancel)
|
||||||
.await
|
.await
|
||||||
.is_err()
|
.is_err()
|
||||||
{
|
{
|
||||||
info!("shutting down");
|
|
||||||
return;
|
return;
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
@@ -167,7 +169,6 @@ async fn disk_usage_eviction_task(
|
|||||||
tokio::select! {
|
tokio::select! {
|
||||||
_ = tokio::time::sleep_until(sleep_until) => {},
|
_ = tokio::time::sleep_until(sleep_until) => {},
|
||||||
_ = cancel.cancelled() => {
|
_ = cancel.cancelled() => {
|
||||||
info!("shutting down");
|
|
||||||
break
|
break
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
@@ -314,7 +315,7 @@ pub async fn disk_usage_eviction_task_iteration_impl<U: Usage>(
|
|||||||
partition,
|
partition,
|
||||||
candidate.layer.get_tenant_id(),
|
candidate.layer.get_tenant_id(),
|
||||||
candidate.layer.get_timeline_id(),
|
candidate.layer.get_timeline_id(),
|
||||||
candidate.layer.filename().file_name(),
|
candidate.layer,
|
||||||
);
|
);
|
||||||
}
|
}
|
||||||
|
|
||||||
|
|||||||
+119
-8
@@ -1,9 +1,9 @@
|
|||||||
use metrics::metric_vec_duration::DurationResultObserver;
|
use metrics::metric_vec_duration::DurationResultObserver;
|
||||||
use metrics::{
|
use metrics::{
|
||||||
register_counter_vec, register_histogram, register_histogram_vec, register_int_counter,
|
register_counter_vec, register_histogram, register_histogram_vec, register_int_counter,
|
||||||
register_int_counter_vec, register_int_gauge, register_int_gauge_vec, register_uint_gauge_vec,
|
register_int_counter_vec, register_int_gauge, register_int_gauge_vec, register_uint_gauge,
|
||||||
Counter, CounterVec, Histogram, HistogramVec, IntCounter, IntCounterVec, IntGauge, IntGaugeVec,
|
register_uint_gauge_vec, Counter, CounterVec, Histogram, HistogramVec, IntCounter,
|
||||||
UIntGauge, UIntGaugeVec,
|
IntCounterVec, IntGauge, IntGaugeVec, UIntGauge, UIntGaugeVec,
|
||||||
};
|
};
|
||||||
use once_cell::sync::Lazy;
|
use once_cell::sync::Lazy;
|
||||||
use pageserver_api::models::TenantState;
|
use pageserver_api::models::TenantState;
|
||||||
@@ -130,6 +130,122 @@ pub static MATERIALIZED_PAGE_CACHE_HIT: Lazy<IntCounter> = Lazy::new(|| {
|
|||||||
.expect("failed to define a metric")
|
.expect("failed to define a metric")
|
||||||
});
|
});
|
||||||
|
|
||||||
|
pub struct PageCacheMetrics {
|
||||||
|
pub read_accesses_materialized_page: IntCounter,
|
||||||
|
pub read_accesses_ephemeral: IntCounter,
|
||||||
|
pub read_accesses_immutable: IntCounter,
|
||||||
|
|
||||||
|
pub read_hits_ephemeral: IntCounter,
|
||||||
|
pub read_hits_immutable: IntCounter,
|
||||||
|
pub read_hits_materialized_page_exact: IntCounter,
|
||||||
|
pub read_hits_materialized_page_older_lsn: IntCounter,
|
||||||
|
}
|
||||||
|
|
||||||
|
static PAGE_CACHE_READ_HITS: Lazy<IntCounterVec> = Lazy::new(|| {
|
||||||
|
register_int_counter_vec!(
|
||||||
|
"pageserver_page_cache_read_hits_total",
|
||||||
|
"Number of read accesses to the page cache that hit",
|
||||||
|
&["key_kind", "hit_kind"]
|
||||||
|
)
|
||||||
|
.expect("failed to define a metric")
|
||||||
|
});
|
||||||
|
|
||||||
|
static PAGE_CACHE_READ_ACCESSES: Lazy<IntCounterVec> = Lazy::new(|| {
|
||||||
|
register_int_counter_vec!(
|
||||||
|
"pageserver_page_cache_read_accesses_total",
|
||||||
|
"Number of read accesses to the page cache",
|
||||||
|
&["key_kind"]
|
||||||
|
)
|
||||||
|
.expect("failed to define a metric")
|
||||||
|
});
|
||||||
|
|
||||||
|
pub static PAGE_CACHE: Lazy<PageCacheMetrics> = Lazy::new(|| PageCacheMetrics {
|
||||||
|
read_accesses_materialized_page: {
|
||||||
|
PAGE_CACHE_READ_ACCESSES
|
||||||
|
.get_metric_with_label_values(&["materialized_page"])
|
||||||
|
.unwrap()
|
||||||
|
},
|
||||||
|
|
||||||
|
read_accesses_ephemeral: {
|
||||||
|
PAGE_CACHE_READ_ACCESSES
|
||||||
|
.get_metric_with_label_values(&["ephemeral"])
|
||||||
|
.unwrap()
|
||||||
|
},
|
||||||
|
|
||||||
|
read_accesses_immutable: {
|
||||||
|
PAGE_CACHE_READ_ACCESSES
|
||||||
|
.get_metric_with_label_values(&["immutable"])
|
||||||
|
.unwrap()
|
||||||
|
},
|
||||||
|
|
||||||
|
read_hits_ephemeral: {
|
||||||
|
PAGE_CACHE_READ_HITS
|
||||||
|
.get_metric_with_label_values(&["ephemeral", "-"])
|
||||||
|
.unwrap()
|
||||||
|
},
|
||||||
|
|
||||||
|
read_hits_immutable: {
|
||||||
|
PAGE_CACHE_READ_HITS
|
||||||
|
.get_metric_with_label_values(&["immutable", "-"])
|
||||||
|
.unwrap()
|
||||||
|
},
|
||||||
|
|
||||||
|
read_hits_materialized_page_exact: {
|
||||||
|
PAGE_CACHE_READ_HITS
|
||||||
|
.get_metric_with_label_values(&["materialized_page", "exact"])
|
||||||
|
.unwrap()
|
||||||
|
},
|
||||||
|
|
||||||
|
read_hits_materialized_page_older_lsn: {
|
||||||
|
PAGE_CACHE_READ_HITS
|
||||||
|
.get_metric_with_label_values(&["materialized_page", "older_lsn"])
|
||||||
|
.unwrap()
|
||||||
|
},
|
||||||
|
});
|
||||||
|
|
||||||
|
pub struct PageCacheSizeMetrics {
|
||||||
|
pub max_bytes: UIntGauge,
|
||||||
|
|
||||||
|
pub current_bytes_ephemeral: UIntGauge,
|
||||||
|
pub current_bytes_immutable: UIntGauge,
|
||||||
|
pub current_bytes_materialized_page: UIntGauge,
|
||||||
|
}
|
||||||
|
|
||||||
|
static PAGE_CACHE_SIZE_CURRENT_BYTES: Lazy<UIntGaugeVec> = Lazy::new(|| {
|
||||||
|
register_uint_gauge_vec!(
|
||||||
|
"pageserver_page_cache_size_current_bytes",
|
||||||
|
"Current size of the page cache in bytes, by key kind",
|
||||||
|
&["key_kind"]
|
||||||
|
)
|
||||||
|
.expect("failed to define a metric")
|
||||||
|
});
|
||||||
|
|
||||||
|
pub static PAGE_CACHE_SIZE: Lazy<PageCacheSizeMetrics> = Lazy::new(|| PageCacheSizeMetrics {
|
||||||
|
max_bytes: {
|
||||||
|
register_uint_gauge!(
|
||||||
|
"pageserver_page_cache_size_max_bytes",
|
||||||
|
"Maximum size of the page cache in bytes"
|
||||||
|
)
|
||||||
|
.expect("failed to define a metric")
|
||||||
|
},
|
||||||
|
|
||||||
|
current_bytes_ephemeral: {
|
||||||
|
PAGE_CACHE_SIZE_CURRENT_BYTES
|
||||||
|
.get_metric_with_label_values(&["ephemeral"])
|
||||||
|
.unwrap()
|
||||||
|
},
|
||||||
|
current_bytes_immutable: {
|
||||||
|
PAGE_CACHE_SIZE_CURRENT_BYTES
|
||||||
|
.get_metric_with_label_values(&["immutable"])
|
||||||
|
.unwrap()
|
||||||
|
},
|
||||||
|
current_bytes_materialized_page: {
|
||||||
|
PAGE_CACHE_SIZE_CURRENT_BYTES
|
||||||
|
.get_metric_with_label_values(&["materialized_page"])
|
||||||
|
.unwrap()
|
||||||
|
},
|
||||||
|
});
|
||||||
|
|
||||||
static WAIT_LSN_TIME: Lazy<HistogramVec> = Lazy::new(|| {
|
static WAIT_LSN_TIME: Lazy<HistogramVec> = Lazy::new(|| {
|
||||||
register_histogram_vec!(
|
register_histogram_vec!(
|
||||||
"pageserver_wait_lsn_seconds",
|
"pageserver_wait_lsn_seconds",
|
||||||
@@ -968,7 +1084,6 @@ impl RemoteTimelineClientMetrics {
|
|||||||
op_kind: &RemoteOpKind,
|
op_kind: &RemoteOpKind,
|
||||||
status: &'static str,
|
status: &'static str,
|
||||||
) -> Histogram {
|
) -> Histogram {
|
||||||
// XXX would be nice to have an upgradable RwLock
|
|
||||||
let mut guard = self.remote_operation_time.lock().unwrap();
|
let mut guard = self.remote_operation_time.lock().unwrap();
|
||||||
let key = (file_kind.as_str(), op_kind.as_str(), status);
|
let key = (file_kind.as_str(), op_kind.as_str(), status);
|
||||||
let metric = guard.entry(key).or_insert_with(move || {
|
let metric = guard.entry(key).or_insert_with(move || {
|
||||||
@@ -990,7 +1105,6 @@ impl RemoteTimelineClientMetrics {
|
|||||||
file_kind: &RemoteOpFileKind,
|
file_kind: &RemoteOpFileKind,
|
||||||
op_kind: &RemoteOpKind,
|
op_kind: &RemoteOpKind,
|
||||||
) -> IntGauge {
|
) -> IntGauge {
|
||||||
// XXX would be nice to have an upgradable RwLock
|
|
||||||
let mut guard = self.calls_unfinished_gauge.lock().unwrap();
|
let mut guard = self.calls_unfinished_gauge.lock().unwrap();
|
||||||
let key = (file_kind.as_str(), op_kind.as_str());
|
let key = (file_kind.as_str(), op_kind.as_str());
|
||||||
let metric = guard.entry(key).or_insert_with(move || {
|
let metric = guard.entry(key).or_insert_with(move || {
|
||||||
@@ -1011,7 +1125,6 @@ impl RemoteTimelineClientMetrics {
|
|||||||
file_kind: &RemoteOpFileKind,
|
file_kind: &RemoteOpFileKind,
|
||||||
op_kind: &RemoteOpKind,
|
op_kind: &RemoteOpKind,
|
||||||
) -> Histogram {
|
) -> Histogram {
|
||||||
// XXX would be nice to have an upgradable RwLock
|
|
||||||
let mut guard = self.calls_started_hist.lock().unwrap();
|
let mut guard = self.calls_started_hist.lock().unwrap();
|
||||||
let key = (file_kind.as_str(), op_kind.as_str());
|
let key = (file_kind.as_str(), op_kind.as_str());
|
||||||
let metric = guard.entry(key).or_insert_with(move || {
|
let metric = guard.entry(key).or_insert_with(move || {
|
||||||
@@ -1032,7 +1145,6 @@ impl RemoteTimelineClientMetrics {
|
|||||||
file_kind: &RemoteOpFileKind,
|
file_kind: &RemoteOpFileKind,
|
||||||
op_kind: &RemoteOpKind,
|
op_kind: &RemoteOpKind,
|
||||||
) -> IntCounter {
|
) -> IntCounter {
|
||||||
// XXX would be nice to have an upgradable RwLock
|
|
||||||
let mut guard = self.bytes_started_counter.lock().unwrap();
|
let mut guard = self.bytes_started_counter.lock().unwrap();
|
||||||
let key = (file_kind.as_str(), op_kind.as_str());
|
let key = (file_kind.as_str(), op_kind.as_str());
|
||||||
let metric = guard.entry(key).or_insert_with(move || {
|
let metric = guard.entry(key).or_insert_with(move || {
|
||||||
@@ -1053,7 +1165,6 @@ impl RemoteTimelineClientMetrics {
|
|||||||
file_kind: &RemoteOpFileKind,
|
file_kind: &RemoteOpFileKind,
|
||||||
op_kind: &RemoteOpKind,
|
op_kind: &RemoteOpKind,
|
||||||
) -> IntCounter {
|
) -> IntCounter {
|
||||||
// XXX would be nice to have an upgradable RwLock
|
|
||||||
let mut guard = self.bytes_finished_counter.lock().unwrap();
|
let mut guard = self.bytes_finished_counter.lock().unwrap();
|
||||||
let key = (file_kind.as_str(), op_kind.as_str());
|
let key = (file_kind.as_str(), op_kind.as_str());
|
||||||
let metric = guard.entry(key).or_insert_with(move || {
|
let metric = guard.entry(key).or_insert_with(move || {
|
||||||
|
|||||||
@@ -53,8 +53,8 @@ use utils::{
|
|||||||
lsn::Lsn,
|
lsn::Lsn,
|
||||||
};
|
};
|
||||||
|
|
||||||
use crate::repository::Key;
|
|
||||||
use crate::tenant::writeback_ephemeral_file;
|
use crate::tenant::writeback_ephemeral_file;
|
||||||
|
use crate::{metrics::PageCacheSizeMetrics, repository::Key};
|
||||||
|
|
||||||
static PAGE_CACHE: OnceCell<PageCache> = OnceCell::new();
|
static PAGE_CACHE: OnceCell<PageCache> = OnceCell::new();
|
||||||
const TEST_PAGE_CACHE_SIZE: usize = 50;
|
const TEST_PAGE_CACHE_SIZE: usize = 50;
|
||||||
@@ -187,6 +187,8 @@ pub struct PageCache {
|
|||||||
/// Index of the next candidate to evict, for the Clock replacement algorithm.
|
/// Index of the next candidate to evict, for the Clock replacement algorithm.
|
||||||
/// This is interpreted modulo the page cache size.
|
/// This is interpreted modulo the page cache size.
|
||||||
next_evict_slot: AtomicUsize,
|
next_evict_slot: AtomicUsize,
|
||||||
|
|
||||||
|
size_metrics: &'static PageCacheSizeMetrics,
|
||||||
}
|
}
|
||||||
|
|
||||||
///
|
///
|
||||||
@@ -313,6 +315,10 @@ impl PageCache {
|
|||||||
key: &Key,
|
key: &Key,
|
||||||
lsn: Lsn,
|
lsn: Lsn,
|
||||||
) -> Option<(Lsn, PageReadGuard)> {
|
) -> Option<(Lsn, PageReadGuard)> {
|
||||||
|
crate::metrics::PAGE_CACHE
|
||||||
|
.read_accesses_materialized_page
|
||||||
|
.inc();
|
||||||
|
|
||||||
let mut cache_key = CacheKey::MaterializedPage {
|
let mut cache_key = CacheKey::MaterializedPage {
|
||||||
hash_key: MaterializedPageHashKey {
|
hash_key: MaterializedPageHashKey {
|
||||||
tenant_id,
|
tenant_id,
|
||||||
@@ -323,8 +329,21 @@ impl PageCache {
|
|||||||
};
|
};
|
||||||
|
|
||||||
if let Some(guard) = self.try_lock_for_read(&mut cache_key) {
|
if let Some(guard) = self.try_lock_for_read(&mut cache_key) {
|
||||||
if let CacheKey::MaterializedPage { hash_key: _, lsn } = cache_key {
|
if let CacheKey::MaterializedPage {
|
||||||
Some((lsn, guard))
|
hash_key: _,
|
||||||
|
lsn: available_lsn,
|
||||||
|
} = cache_key
|
||||||
|
{
|
||||||
|
if available_lsn == lsn {
|
||||||
|
crate::metrics::PAGE_CACHE
|
||||||
|
.read_hits_materialized_page_exact
|
||||||
|
.inc();
|
||||||
|
} else {
|
||||||
|
crate::metrics::PAGE_CACHE
|
||||||
|
.read_hits_materialized_page_older_lsn
|
||||||
|
.inc();
|
||||||
|
}
|
||||||
|
Some((available_lsn, guard))
|
||||||
} else {
|
} else {
|
||||||
panic!("unexpected key type in slot");
|
panic!("unexpected key type in slot");
|
||||||
}
|
}
|
||||||
@@ -499,11 +518,31 @@ impl PageCache {
|
|||||||
/// ```
|
/// ```
|
||||||
///
|
///
|
||||||
fn lock_for_read(&self, cache_key: &mut CacheKey) -> anyhow::Result<ReadBufResult> {
|
fn lock_for_read(&self, cache_key: &mut CacheKey) -> anyhow::Result<ReadBufResult> {
|
||||||
|
let (read_access, hit) = match cache_key {
|
||||||
|
CacheKey::MaterializedPage { .. } => {
|
||||||
|
unreachable!("Materialized pages use lookup_materialized_page")
|
||||||
|
}
|
||||||
|
CacheKey::EphemeralPage { .. } => (
|
||||||
|
&crate::metrics::PAGE_CACHE.read_accesses_ephemeral,
|
||||||
|
&crate::metrics::PAGE_CACHE.read_hits_ephemeral,
|
||||||
|
),
|
||||||
|
CacheKey::ImmutableFilePage { .. } => (
|
||||||
|
&crate::metrics::PAGE_CACHE.read_accesses_immutable,
|
||||||
|
&crate::metrics::PAGE_CACHE.read_hits_immutable,
|
||||||
|
),
|
||||||
|
};
|
||||||
|
read_access.inc();
|
||||||
|
|
||||||
|
let mut is_first_iteration = true;
|
||||||
loop {
|
loop {
|
||||||
// First check if the key already exists in the cache.
|
// First check if the key already exists in the cache.
|
||||||
if let Some(read_guard) = self.try_lock_for_read(cache_key) {
|
if let Some(read_guard) = self.try_lock_for_read(cache_key) {
|
||||||
|
if is_first_iteration {
|
||||||
|
hit.inc();
|
||||||
|
}
|
||||||
return Ok(ReadBufResult::Found(read_guard));
|
return Ok(ReadBufResult::Found(read_guard));
|
||||||
}
|
}
|
||||||
|
is_first_iteration = false;
|
||||||
|
|
||||||
// Not found. Find a victim buffer
|
// Not found. Find a victim buffer
|
||||||
let (slot_idx, mut inner) =
|
let (slot_idx, mut inner) =
|
||||||
@@ -681,6 +720,9 @@ impl PageCache {
|
|||||||
|
|
||||||
if let Ok(version_idx) = versions.binary_search_by_key(old_lsn, |v| v.lsn) {
|
if let Ok(version_idx) = versions.binary_search_by_key(old_lsn, |v| v.lsn) {
|
||||||
versions.remove(version_idx);
|
versions.remove(version_idx);
|
||||||
|
self.size_metrics
|
||||||
|
.current_bytes_materialized_page
|
||||||
|
.sub_page_sz(1);
|
||||||
if versions.is_empty() {
|
if versions.is_empty() {
|
||||||
old_entry.remove_entry();
|
old_entry.remove_entry();
|
||||||
}
|
}
|
||||||
@@ -693,11 +735,13 @@ impl PageCache {
|
|||||||
let mut map = self.ephemeral_page_map.write().unwrap();
|
let mut map = self.ephemeral_page_map.write().unwrap();
|
||||||
map.remove(&(*file_id, *blkno))
|
map.remove(&(*file_id, *blkno))
|
||||||
.expect("could not find old key in mapping");
|
.expect("could not find old key in mapping");
|
||||||
|
self.size_metrics.current_bytes_ephemeral.sub_page_sz(1);
|
||||||
}
|
}
|
||||||
CacheKey::ImmutableFilePage { file_id, blkno } => {
|
CacheKey::ImmutableFilePage { file_id, blkno } => {
|
||||||
let mut map = self.immutable_page_map.write().unwrap();
|
let mut map = self.immutable_page_map.write().unwrap();
|
||||||
map.remove(&(*file_id, *blkno))
|
map.remove(&(*file_id, *blkno))
|
||||||
.expect("could not find old key in mapping");
|
.expect("could not find old key in mapping");
|
||||||
|
self.size_metrics.current_bytes_immutable.sub_page_sz(1);
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
@@ -725,6 +769,9 @@ impl PageCache {
|
|||||||
slot_idx,
|
slot_idx,
|
||||||
},
|
},
|
||||||
);
|
);
|
||||||
|
self.size_metrics
|
||||||
|
.current_bytes_materialized_page
|
||||||
|
.add_page_sz(1);
|
||||||
None
|
None
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
@@ -735,6 +782,7 @@ impl PageCache {
|
|||||||
Entry::Occupied(entry) => Some(*entry.get()),
|
Entry::Occupied(entry) => Some(*entry.get()),
|
||||||
Entry::Vacant(entry) => {
|
Entry::Vacant(entry) => {
|
||||||
entry.insert(slot_idx);
|
entry.insert(slot_idx);
|
||||||
|
self.size_metrics.current_bytes_ephemeral.add_page_sz(1);
|
||||||
None
|
None
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
@@ -745,6 +793,7 @@ impl PageCache {
|
|||||||
Entry::Occupied(entry) => Some(*entry.get()),
|
Entry::Occupied(entry) => Some(*entry.get()),
|
||||||
Entry::Vacant(entry) => {
|
Entry::Vacant(entry) => {
|
||||||
entry.insert(slot_idx);
|
entry.insert(slot_idx);
|
||||||
|
self.size_metrics.current_bytes_immutable.add_page_sz(1);
|
||||||
None
|
None
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
@@ -844,6 +893,12 @@ impl PageCache {
|
|||||||
|
|
||||||
let page_buffer = Box::leak(vec![0u8; num_pages * PAGE_SZ].into_boxed_slice());
|
let page_buffer = Box::leak(vec![0u8; num_pages * PAGE_SZ].into_boxed_slice());
|
||||||
|
|
||||||
|
let size_metrics = &crate::metrics::PAGE_CACHE_SIZE;
|
||||||
|
size_metrics.max_bytes.set_page_sz(num_pages);
|
||||||
|
size_metrics.current_bytes_ephemeral.set_page_sz(0);
|
||||||
|
size_metrics.current_bytes_immutable.set_page_sz(0);
|
||||||
|
size_metrics.current_bytes_materialized_page.set_page_sz(0);
|
||||||
|
|
||||||
let slots = page_buffer
|
let slots = page_buffer
|
||||||
.chunks_exact_mut(PAGE_SZ)
|
.chunks_exact_mut(PAGE_SZ)
|
||||||
.map(|chunk| {
|
.map(|chunk| {
|
||||||
@@ -866,6 +921,30 @@ impl PageCache {
|
|||||||
immutable_page_map: Default::default(),
|
immutable_page_map: Default::default(),
|
||||||
slots,
|
slots,
|
||||||
next_evict_slot: AtomicUsize::new(0),
|
next_evict_slot: AtomicUsize::new(0),
|
||||||
|
size_metrics,
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
trait PageSzBytesMetric {
|
||||||
|
fn set_page_sz(&self, count: usize);
|
||||||
|
fn add_page_sz(&self, count: usize);
|
||||||
|
fn sub_page_sz(&self, count: usize);
|
||||||
|
}
|
||||||
|
|
||||||
|
#[inline(always)]
|
||||||
|
fn count_times_page_sz(count: usize) -> u64 {
|
||||||
|
u64::try_from(count).unwrap() * u64::try_from(PAGE_SZ).unwrap()
|
||||||
|
}
|
||||||
|
|
||||||
|
impl PageSzBytesMetric for metrics::UIntGauge {
|
||||||
|
fn set_page_sz(&self, count: usize) {
|
||||||
|
self.set(count_times_page_sz(count));
|
||||||
|
}
|
||||||
|
fn add_page_sz(&self, count: usize) {
|
||||||
|
self.add(count_times_page_sz(count));
|
||||||
|
}
|
||||||
|
fn sub_page_sz(&self, count: usize) {
|
||||||
|
self.sub(count_times_page_sz(count));
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|||||||
+19
-234
@@ -11,7 +11,7 @@
|
|||||||
//! parent timeline, and the last LSN that has been written to disk.
|
//! parent timeline, and the last LSN that has been written to disk.
|
||||||
//!
|
//!
|
||||||
|
|
||||||
use anyhow::{bail, ensure, Context};
|
use anyhow::{bail, Context};
|
||||||
use futures::FutureExt;
|
use futures::FutureExt;
|
||||||
use pageserver_api::models::TimelineState;
|
use pageserver_api::models::TimelineState;
|
||||||
use remote_storage::DownloadError;
|
use remote_storage::DownloadError;
|
||||||
@@ -49,6 +49,8 @@ use std::time::{Duration, Instant};
|
|||||||
use self::config::TenantConf;
|
use self::config::TenantConf;
|
||||||
use self::metadata::TimelineMetadata;
|
use self::metadata::TimelineMetadata;
|
||||||
use self::remote_timeline_client::RemoteTimelineClient;
|
use self::remote_timeline_client::RemoteTimelineClient;
|
||||||
|
use self::timeline::uninit::TimelineUninitMark;
|
||||||
|
use self::timeline::uninit::UninitializedTimeline;
|
||||||
use self::timeline::EvictionTaskTenantState;
|
use self::timeline::EvictionTaskTenantState;
|
||||||
use crate::config::PageServerConf;
|
use crate::config::PageServerConf;
|
||||||
use crate::context::{DownloadBehavior, RequestContext};
|
use crate::context::{DownloadBehavior, RequestContext};
|
||||||
@@ -68,6 +70,7 @@ use crate::tenant::storage_layer::ImageLayer;
|
|||||||
use crate::tenant::storage_layer::Layer;
|
use crate::tenant::storage_layer::Layer;
|
||||||
use crate::InitializationOrder;
|
use crate::InitializationOrder;
|
||||||
|
|
||||||
|
use crate::tenant::timeline::uninit::cleanup_timeline_directory;
|
||||||
use crate::virtual_file::VirtualFile;
|
use crate::virtual_file::VirtualFile;
|
||||||
use crate::walredo::PostgresRedoManager;
|
use crate::walredo::PostgresRedoManager;
|
||||||
use crate::walredo::WalRedoManager;
|
use crate::walredo::WalRedoManager;
|
||||||
@@ -87,6 +90,7 @@ pub mod disk_btree;
|
|||||||
pub(crate) mod ephemeral_file;
|
pub(crate) mod ephemeral_file;
|
||||||
pub mod layer_map;
|
pub mod layer_map;
|
||||||
pub mod manifest;
|
pub mod manifest;
|
||||||
|
mod span;
|
||||||
|
|
||||||
pub mod metadata;
|
pub mod metadata;
|
||||||
mod par_fsync;
|
mod par_fsync;
|
||||||
@@ -102,7 +106,7 @@ mod timeline;
|
|||||||
|
|
||||||
pub mod size;
|
pub mod size;
|
||||||
|
|
||||||
pub(crate) use timeline::debug_assert_current_span_has_tenant_and_timeline_id;
|
pub(crate) use timeline::span::debug_assert_current_span_has_tenant_and_timeline_id;
|
||||||
pub use timeline::{
|
pub use timeline::{
|
||||||
LocalLayerInfoForDiskUsageEviction, LogicalSizeCalculationCause, PageReconstructError, Timeline,
|
LocalLayerInfoForDiskUsageEviction, LogicalSizeCalculationCause, PageReconstructError, Timeline,
|
||||||
};
|
};
|
||||||
@@ -161,200 +165,6 @@ pub struct Tenant {
|
|||||||
eviction_task_tenant_state: tokio::sync::Mutex<EvictionTaskTenantState>,
|
eviction_task_tenant_state: tokio::sync::Mutex<EvictionTaskTenantState>,
|
||||||
}
|
}
|
||||||
|
|
||||||
/// A timeline with some of its files on disk, being initialized.
|
|
||||||
/// This struct ensures the atomicity of the timeline init: it's either properly created and inserted into pageserver's memory, or
|
|
||||||
/// its local files are removed. In the worst case of a crash, an uninit mark file is left behind, which causes the directory
|
|
||||||
/// to be removed on next restart.
|
|
||||||
///
|
|
||||||
/// The caller is responsible for proper timeline data filling before the final init.
|
|
||||||
#[must_use]
|
|
||||||
pub struct UninitializedTimeline<'t> {
|
|
||||||
owning_tenant: &'t Tenant,
|
|
||||||
timeline_id: TimelineId,
|
|
||||||
raw_timeline: Option<(Arc<Timeline>, TimelineUninitMark)>,
|
|
||||||
}
|
|
||||||
|
|
||||||
/// An uninit mark file, created along the timeline dir to ensure the timeline either gets fully initialized and loaded into pageserver's memory,
|
|
||||||
/// or gets removed eventually.
|
|
||||||
///
|
|
||||||
/// XXX: it's important to create it near the timeline dir, not inside it to ensure timeline dir gets removed first.
|
|
||||||
#[must_use]
|
|
||||||
struct TimelineUninitMark {
|
|
||||||
uninit_mark_deleted: bool,
|
|
||||||
uninit_mark_path: PathBuf,
|
|
||||||
timeline_path: PathBuf,
|
|
||||||
}
|
|
||||||
|
|
||||||
impl UninitializedTimeline<'_> {
|
|
||||||
/// Finish timeline creation: insert it into the Tenant's timelines map and remove the
|
|
||||||
/// uninit mark file.
|
|
||||||
///
|
|
||||||
/// This function launches the flush loop if not already done.
|
|
||||||
///
|
|
||||||
/// The caller is responsible for activating the timeline (function `.activate()`).
|
|
||||||
fn finish_creation(mut self) -> anyhow::Result<Arc<Timeline>> {
|
|
||||||
let timeline_id = self.timeline_id;
|
|
||||||
let tenant_id = self.owning_tenant.tenant_id;
|
|
||||||
|
|
||||||
let (new_timeline, uninit_mark) = self.raw_timeline.take().with_context(|| {
|
|
||||||
format!("No timeline for initalization found for {tenant_id}/{timeline_id}")
|
|
||||||
})?;
|
|
||||||
|
|
||||||
// Check that the caller initialized disk_consistent_lsn
|
|
||||||
let new_disk_consistent_lsn = new_timeline.get_disk_consistent_lsn();
|
|
||||||
ensure!(
|
|
||||||
new_disk_consistent_lsn.is_valid(),
|
|
||||||
"new timeline {tenant_id}/{timeline_id} has invalid disk_consistent_lsn"
|
|
||||||
);
|
|
||||||
|
|
||||||
let mut timelines = self.owning_tenant.timelines.lock().unwrap();
|
|
||||||
match timelines.entry(timeline_id) {
|
|
||||||
Entry::Occupied(_) => anyhow::bail!(
|
|
||||||
"Found freshly initialized timeline {tenant_id}/{timeline_id} in the tenant map"
|
|
||||||
),
|
|
||||||
Entry::Vacant(v) => {
|
|
||||||
uninit_mark.remove_uninit_mark().with_context(|| {
|
|
||||||
format!(
|
|
||||||
"Failed to remove uninit mark file for timeline {tenant_id}/{timeline_id}"
|
|
||||||
)
|
|
||||||
})?;
|
|
||||||
v.insert(Arc::clone(&new_timeline));
|
|
||||||
|
|
||||||
new_timeline.maybe_spawn_flush_loop();
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
Ok(new_timeline)
|
|
||||||
}
|
|
||||||
|
|
||||||
/// Prepares timeline data by loading it from the basebackup archive.
|
|
||||||
pub async fn import_basebackup_from_tar(
|
|
||||||
self,
|
|
||||||
copyin_read: &mut (impl tokio::io::AsyncRead + Send + Sync + Unpin),
|
|
||||||
base_lsn: Lsn,
|
|
||||||
broker_client: storage_broker::BrokerClientChannel,
|
|
||||||
ctx: &RequestContext,
|
|
||||||
) -> anyhow::Result<Arc<Timeline>> {
|
|
||||||
let raw_timeline = self.raw_timeline()?;
|
|
||||||
|
|
||||||
import_datadir::import_basebackup_from_tar(raw_timeline, copyin_read, base_lsn, ctx)
|
|
||||||
.await
|
|
||||||
.context("Failed to import basebackup")?;
|
|
||||||
|
|
||||||
// Flush the new layer files to disk, before we make the timeline as available to
|
|
||||||
// the outside world.
|
|
||||||
//
|
|
||||||
// Flush loop needs to be spawned in order to be able to flush.
|
|
||||||
raw_timeline.maybe_spawn_flush_loop();
|
|
||||||
|
|
||||||
fail::fail_point!("before-checkpoint-new-timeline", |_| {
|
|
||||||
bail!("failpoint before-checkpoint-new-timeline");
|
|
||||||
});
|
|
||||||
|
|
||||||
raw_timeline
|
|
||||||
.freeze_and_flush()
|
|
||||||
.await
|
|
||||||
.context("Failed to flush after basebackup import")?;
|
|
||||||
|
|
||||||
// All the data has been imported. Insert the Timeline into the tenant's timelines
|
|
||||||
// map and remove the uninit mark file.
|
|
||||||
let tl = self.finish_creation()?;
|
|
||||||
tl.activate(broker_client, None, ctx);
|
|
||||||
Ok(tl)
|
|
||||||
}
|
|
||||||
|
|
||||||
fn raw_timeline(&self) -> anyhow::Result<&Arc<Timeline>> {
|
|
||||||
Ok(&self
|
|
||||||
.raw_timeline
|
|
||||||
.as_ref()
|
|
||||||
.with_context(|| {
|
|
||||||
format!(
|
|
||||||
"No raw timeline {}/{} found",
|
|
||||||
self.owning_tenant.tenant_id, self.timeline_id
|
|
||||||
)
|
|
||||||
})?
|
|
||||||
.0)
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
impl Drop for UninitializedTimeline<'_> {
|
|
||||||
fn drop(&mut self) {
|
|
||||||
if let Some((_, uninit_mark)) = self.raw_timeline.take() {
|
|
||||||
let _entered = info_span!("drop_uninitialized_timeline", tenant = %self.owning_tenant.tenant_id, timeline = %self.timeline_id).entered();
|
|
||||||
error!("Timeline got dropped without initializing, cleaning its files");
|
|
||||||
cleanup_timeline_directory(uninit_mark);
|
|
||||||
}
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
fn cleanup_timeline_directory(uninit_mark: TimelineUninitMark) {
|
|
||||||
let timeline_path = &uninit_mark.timeline_path;
|
|
||||||
match ignore_absent_files(|| fs::remove_dir_all(timeline_path)) {
|
|
||||||
Ok(()) => {
|
|
||||||
info!("Timeline dir {timeline_path:?} removed successfully, removing the uninit mark")
|
|
||||||
}
|
|
||||||
Err(e) => {
|
|
||||||
error!("Failed to clean up uninitialized timeline directory {timeline_path:?}: {e:?}")
|
|
||||||
}
|
|
||||||
}
|
|
||||||
drop(uninit_mark); // mark handles its deletion on drop, gets retained if timeline dir exists
|
|
||||||
}
|
|
||||||
|
|
||||||
impl TimelineUninitMark {
|
|
||||||
fn new(uninit_mark_path: PathBuf, timeline_path: PathBuf) -> Self {
|
|
||||||
Self {
|
|
||||||
uninit_mark_deleted: false,
|
|
||||||
uninit_mark_path,
|
|
||||||
timeline_path,
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
fn remove_uninit_mark(mut self) -> anyhow::Result<()> {
|
|
||||||
if !self.uninit_mark_deleted {
|
|
||||||
self.delete_mark_file_if_present()?;
|
|
||||||
}
|
|
||||||
|
|
||||||
Ok(())
|
|
||||||
}
|
|
||||||
|
|
||||||
fn delete_mark_file_if_present(&mut self) -> anyhow::Result<()> {
|
|
||||||
let uninit_mark_file = &self.uninit_mark_path;
|
|
||||||
let uninit_mark_parent = uninit_mark_file
|
|
||||||
.parent()
|
|
||||||
.with_context(|| format!("Uninit mark file {uninit_mark_file:?} has no parent"))?;
|
|
||||||
ignore_absent_files(|| fs::remove_file(uninit_mark_file)).with_context(|| {
|
|
||||||
format!("Failed to remove uninit mark file at path {uninit_mark_file:?}")
|
|
||||||
})?;
|
|
||||||
crashsafe::fsync(uninit_mark_parent).context("Failed to fsync uninit mark parent")?;
|
|
||||||
self.uninit_mark_deleted = true;
|
|
||||||
|
|
||||||
Ok(())
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
impl Drop for TimelineUninitMark {
|
|
||||||
fn drop(&mut self) {
|
|
||||||
if !self.uninit_mark_deleted {
|
|
||||||
if self.timeline_path.exists() {
|
|
||||||
error!(
|
|
||||||
"Uninit mark {} is not removed, timeline {} stays uninitialized",
|
|
||||||
self.uninit_mark_path.display(),
|
|
||||||
self.timeline_path.display()
|
|
||||||
)
|
|
||||||
} else {
|
|
||||||
// unblock later timeline creation attempts
|
|
||||||
warn!(
|
|
||||||
"Removing intermediate uninit mark file {}",
|
|
||||||
self.uninit_mark_path.display()
|
|
||||||
);
|
|
||||||
if let Err(e) = self.delete_mark_file_if_present() {
|
|
||||||
error!("Failed to remove the uninit mark file: {e}")
|
|
||||||
}
|
|
||||||
}
|
|
||||||
}
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
// We should not blindly overwrite local metadata with remote one.
|
// We should not blindly overwrite local metadata with remote one.
|
||||||
// For example, consider the following case:
|
// For example, consider the following case:
|
||||||
// Image layer is flushed to disk as a new delta layer, we update local metadata and start upload task but after that
|
// Image layer is flushed to disk as a new delta layer, we update local metadata and start upload task but after that
|
||||||
@@ -695,7 +505,7 @@ impl Tenant {
|
|||||||
/// No background tasks are started as part of this routine.
|
/// No background tasks are started as part of this routine.
|
||||||
///
|
///
|
||||||
async fn attach(self: &Arc<Tenant>, ctx: &RequestContext) -> anyhow::Result<()> {
|
async fn attach(self: &Arc<Tenant>, ctx: &RequestContext) -> anyhow::Result<()> {
|
||||||
debug_assert_current_span_has_tenant_id();
|
span::debug_assert_current_span_has_tenant_id();
|
||||||
|
|
||||||
let marker_file = self.conf.tenant_attaching_mark_file_path(&self.tenant_id);
|
let marker_file = self.conf.tenant_attaching_mark_file_path(&self.tenant_id);
|
||||||
if !tokio::fs::try_exists(&marker_file)
|
if !tokio::fs::try_exists(&marker_file)
|
||||||
@@ -833,7 +643,7 @@ impl Tenant {
|
|||||||
remote_client: RemoteTimelineClient,
|
remote_client: RemoteTimelineClient,
|
||||||
ctx: &RequestContext,
|
ctx: &RequestContext,
|
||||||
) -> anyhow::Result<()> {
|
) -> anyhow::Result<()> {
|
||||||
debug_assert_current_span_has_tenant_id();
|
span::debug_assert_current_span_has_tenant_id();
|
||||||
|
|
||||||
info!("downloading index file for timeline {}", timeline_id);
|
info!("downloading index file for timeline {}", timeline_id);
|
||||||
tokio::fs::create_dir_all(self.conf.timeline_path(&timeline_id, &self.tenant_id))
|
tokio::fs::create_dir_all(self.conf.timeline_path(&timeline_id, &self.tenant_id))
|
||||||
@@ -912,7 +722,7 @@ impl Tenant {
|
|||||||
init_order: Option<InitializationOrder>,
|
init_order: Option<InitializationOrder>,
|
||||||
ctx: &RequestContext,
|
ctx: &RequestContext,
|
||||||
) -> Arc<Tenant> {
|
) -> Arc<Tenant> {
|
||||||
debug_assert_current_span_has_tenant_id();
|
span::debug_assert_current_span_has_tenant_id();
|
||||||
|
|
||||||
let tenant_conf = match Self::load_tenant_config(conf, tenant_id) {
|
let tenant_conf = match Self::load_tenant_config(conf, tenant_id) {
|
||||||
Ok(conf) => conf,
|
Ok(conf) => conf,
|
||||||
@@ -1098,7 +908,7 @@ impl Tenant {
|
|||||||
init_order: Option<&InitializationOrder>,
|
init_order: Option<&InitializationOrder>,
|
||||||
ctx: &RequestContext,
|
ctx: &RequestContext,
|
||||||
) -> anyhow::Result<()> {
|
) -> anyhow::Result<()> {
|
||||||
debug_assert_current_span_has_tenant_id();
|
span::debug_assert_current_span_has_tenant_id();
|
||||||
|
|
||||||
debug!("loading tenant task");
|
debug!("loading tenant task");
|
||||||
|
|
||||||
@@ -1144,7 +954,7 @@ impl Tenant {
|
|||||||
init_order: Option<&InitializationOrder>,
|
init_order: Option<&InitializationOrder>,
|
||||||
ctx: &RequestContext,
|
ctx: &RequestContext,
|
||||||
) -> anyhow::Result<()> {
|
) -> anyhow::Result<()> {
|
||||||
debug_assert_current_span_has_tenant_id();
|
span::debug_assert_current_span_has_tenant_id();
|
||||||
|
|
||||||
let remote_client = self.remote_storage.as_ref().map(|remote_storage| {
|
let remote_client = self.remote_storage.as_ref().map(|remote_storage| {
|
||||||
RemoteTimelineClient::new(
|
RemoteTimelineClient::new(
|
||||||
@@ -1735,7 +1545,7 @@ impl Tenant {
|
|||||||
timeline_id: TimelineId,
|
timeline_id: TimelineId,
|
||||||
_ctx: &RequestContext,
|
_ctx: &RequestContext,
|
||||||
) -> Result<(), DeleteTimelineError> {
|
) -> Result<(), DeleteTimelineError> {
|
||||||
timeline::debug_assert_current_span_has_tenant_and_timeline_id();
|
debug_assert_current_span_has_tenant_and_timeline_id();
|
||||||
|
|
||||||
// Transition the timeline into TimelineState::Stopping.
|
// Transition the timeline into TimelineState::Stopping.
|
||||||
// This should prevent new operations from starting.
|
// This should prevent new operations from starting.
|
||||||
@@ -1899,7 +1709,7 @@ impl Tenant {
|
|||||||
background_jobs_can_start: Option<&completion::Barrier>,
|
background_jobs_can_start: Option<&completion::Barrier>,
|
||||||
ctx: &RequestContext,
|
ctx: &RequestContext,
|
||||||
) {
|
) {
|
||||||
debug_assert_current_span_has_tenant_id();
|
span::debug_assert_current_span_has_tenant_id();
|
||||||
|
|
||||||
let mut activating = false;
|
let mut activating = false;
|
||||||
self.state.send_modify(|current_state| {
|
self.state.send_modify(|current_state| {
|
||||||
@@ -1970,7 +1780,7 @@ impl Tenant {
|
|||||||
///
|
///
|
||||||
/// This will attempt to shutdown even if tenant is broken.
|
/// This will attempt to shutdown even if tenant is broken.
|
||||||
pub(crate) async fn shutdown(&self, freeze_and_flush: bool) -> Result<(), ShutdownError> {
|
pub(crate) async fn shutdown(&self, freeze_and_flush: bool) -> Result<(), ShutdownError> {
|
||||||
debug_assert_current_span_has_tenant_id();
|
span::debug_assert_current_span_has_tenant_id();
|
||||||
// Set tenant (and its timlines) to Stoppping state.
|
// Set tenant (and its timlines) to Stoppping state.
|
||||||
//
|
//
|
||||||
// Since we can only transition into Stopping state after activation is complete,
|
// Since we can only transition into Stopping state after activation is complete,
|
||||||
@@ -3012,11 +2822,11 @@ impl Tenant {
|
|||||||
|
|
||||||
debug!("Successfully created initial files for timeline {tenant_id}/{new_timeline_id}");
|
debug!("Successfully created initial files for timeline {tenant_id}/{new_timeline_id}");
|
||||||
|
|
||||||
Ok(UninitializedTimeline {
|
Ok(UninitializedTimeline::new(
|
||||||
owning_tenant: self,
|
self,
|
||||||
timeline_id: new_timeline_id,
|
new_timeline_id,
|
||||||
raw_timeline: Some((timeline_struct, uninit_mark)),
|
Some((timeline_struct, uninit_mark)),
|
||||||
})
|
))
|
||||||
}
|
}
|
||||||
|
|
||||||
fn create_timeline_files(
|
fn create_timeline_files(
|
||||||
@@ -4571,28 +4381,3 @@ mod tests {
|
|||||||
Ok(())
|
Ok(())
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
#[cfg(not(debug_assertions))]
|
|
||||||
#[inline]
|
|
||||||
pub(crate) fn debug_assert_current_span_has_tenant_id() {}
|
|
||||||
|
|
||||||
#[cfg(debug_assertions)]
|
|
||||||
pub static TENANT_ID_EXTRACTOR: once_cell::sync::Lazy<
|
|
||||||
utils::tracing_span_assert::MultiNameExtractor<2>,
|
|
||||||
> = once_cell::sync::Lazy::new(|| {
|
|
||||||
utils::tracing_span_assert::MultiNameExtractor::new("TenantId", ["tenant_id", "tenant"])
|
|
||||||
});
|
|
||||||
|
|
||||||
#[cfg(debug_assertions)]
|
|
||||||
#[inline]
|
|
||||||
pub(crate) fn debug_assert_current_span_has_tenant_id() {
|
|
||||||
use utils::tracing_span_assert;
|
|
||||||
|
|
||||||
match tracing_span_assert::check_fields_present([&*TENANT_ID_EXTRACTOR]) {
|
|
||||||
Ok(()) => (),
|
|
||||||
Err(missing) => panic!(
|
|
||||||
"missing extractors: {:?}",
|
|
||||||
missing.into_iter().map(|e| e.name()).collect::<Vec<_>>()
|
|
||||||
),
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|||||||
@@ -608,10 +608,7 @@ impl RemoteTimelineClient {
|
|||||||
self.calls_unfinished_metric_begin(&op);
|
self.calls_unfinished_metric_begin(&op);
|
||||||
upload_queue.queued_operations.push_back(op);
|
upload_queue.queued_operations.push_back(op);
|
||||||
|
|
||||||
info!(
|
info!("scheduled layer file upload {layer_file_name}");
|
||||||
"scheduled layer file upload {}",
|
|
||||||
layer_file_name.file_name()
|
|
||||||
);
|
|
||||||
|
|
||||||
// Launch the task immediately, if possible
|
// Launch the task immediately, if possible
|
||||||
self.launch_queued_tasks(upload_queue);
|
self.launch_queued_tasks(upload_queue);
|
||||||
@@ -664,7 +661,7 @@ impl RemoteTimelineClient {
|
|||||||
});
|
});
|
||||||
self.calls_unfinished_metric_begin(&op);
|
self.calls_unfinished_metric_begin(&op);
|
||||||
upload_queue.queued_operations.push_back(op);
|
upload_queue.queued_operations.push_back(op);
|
||||||
info!("scheduled layer file deletion {}", name.file_name());
|
info!("scheduled layer file deletion {name}");
|
||||||
}
|
}
|
||||||
|
|
||||||
// Launch the tasks immediately, if possible
|
// Launch the tasks immediately, if possible
|
||||||
@@ -828,7 +825,7 @@ impl RemoteTimelineClient {
|
|||||||
.queued_operations
|
.queued_operations
|
||||||
.push_back(op);
|
.push_back(op);
|
||||||
|
|
||||||
info!("scheduled layer file deletion {}", name.file_name());
|
info!("scheduled layer file deletion {name}");
|
||||||
deletions_queued += 1;
|
deletions_queued += 1;
|
||||||
}
|
}
|
||||||
|
|
||||||
|
|||||||
@@ -16,7 +16,7 @@ use tracing::{info, warn};
|
|||||||
|
|
||||||
use crate::config::PageServerConf;
|
use crate::config::PageServerConf;
|
||||||
use crate::tenant::storage_layer::LayerFileName;
|
use crate::tenant::storage_layer::LayerFileName;
|
||||||
use crate::tenant::timeline::debug_assert_current_span_has_tenant_and_timeline_id;
|
use crate::tenant::timeline::span::debug_assert_current_span_has_tenant_and_timeline_id;
|
||||||
use crate::{exponential_backoff, DEFAULT_BASE_BACKOFF_SECONDS, DEFAULT_MAX_BACKOFF_SECONDS};
|
use crate::{exponential_backoff, DEFAULT_BASE_BACKOFF_SECONDS, DEFAULT_MAX_BACKOFF_SECONDS};
|
||||||
use remote_storage::{DownloadError, GenericRemoteStorage};
|
use remote_storage::{DownloadError, GenericRemoteStorage};
|
||||||
use utils::crashsafe::path_with_suffix_extension;
|
use utils::crashsafe::path_with_suffix_extension;
|
||||||
|
|||||||
@@ -0,0 +1,20 @@
|
|||||||
|
#[cfg(debug_assertions)]
|
||||||
|
use utils::tracing_span_assert::{check_fields_present, MultiNameExtractor};
|
||||||
|
|
||||||
|
#[cfg(not(debug_assertions))]
|
||||||
|
pub(crate) fn debug_assert_current_span_has_tenant_id() {}
|
||||||
|
|
||||||
|
#[cfg(debug_assertions)]
|
||||||
|
pub(crate) static TENANT_ID_EXTRACTOR: once_cell::sync::Lazy<MultiNameExtractor<2>> =
|
||||||
|
once_cell::sync::Lazy::new(|| MultiNameExtractor::new("TenantId", ["tenant_id", "tenant"]));
|
||||||
|
|
||||||
|
#[cfg(debug_assertions)]
|
||||||
|
#[track_caller]
|
||||||
|
pub(crate) fn debug_assert_current_span_has_tenant_id() {
|
||||||
|
if let Err(missing) = check_fields_present([&*TENANT_ID_EXTRACTOR]) {
|
||||||
|
panic!(
|
||||||
|
"missing extractors: {:?}",
|
||||||
|
missing.into_iter().map(|e| e.name()).collect::<Vec<_>>()
|
||||||
|
)
|
||||||
|
}
|
||||||
|
}
|
||||||
@@ -335,7 +335,7 @@ impl LayerAccessStats {
|
|||||||
/// All layers should implement a minimal `std::fmt::Debug` without tenant or
|
/// All layers should implement a minimal `std::fmt::Debug` without tenant or
|
||||||
/// timeline names, because those are known in the context of which the layers
|
/// timeline names, because those are known in the context of which the layers
|
||||||
/// are used in (timeline).
|
/// are used in (timeline).
|
||||||
pub trait Layer: std::fmt::Debug + Send + Sync {
|
pub trait Layer: std::fmt::Debug + std::fmt::Display + Send + Sync {
|
||||||
/// Range of keys that this layer covers
|
/// Range of keys that this layer covers
|
||||||
fn get_key_range(&self) -> Range<Key>;
|
fn get_key_range(&self) -> Range<Key>;
|
||||||
|
|
||||||
@@ -373,9 +373,6 @@ pub trait Layer: std::fmt::Debug + Send + Sync {
|
|||||||
ctx: &RequestContext,
|
ctx: &RequestContext,
|
||||||
) -> Result<ValueReconstructResult>;
|
) -> Result<ValueReconstructResult>;
|
||||||
|
|
||||||
/// A short ID string that uniquely identifies the given layer within a [`LayerMap`].
|
|
||||||
fn short_id(&self) -> String;
|
|
||||||
|
|
||||||
/// Dump summary of the contents of the layer to stdout
|
/// Dump summary of the contents of the layer to stdout
|
||||||
fn dump(&self, verbose: bool, ctx: &RequestContext) -> Result<()>;
|
fn dump(&self, verbose: bool, ctx: &RequestContext) -> Result<()>;
|
||||||
}
|
}
|
||||||
@@ -512,10 +509,12 @@ pub mod tests {
|
|||||||
fn is_incremental(&self) -> bool {
|
fn is_incremental(&self) -> bool {
|
||||||
self.layer_desc().is_incremental
|
self.layer_desc().is_incremental
|
||||||
}
|
}
|
||||||
|
}
|
||||||
|
|
||||||
/// Boilerplate to implement the Layer trait, always use layer_desc for persistent layers.
|
/// Boilerplate to implement the Layer trait, always use layer_desc for persistent layers.
|
||||||
fn short_id(&self) -> String {
|
impl std::fmt::Display for LayerDescriptor {
|
||||||
self.layer_desc().short_id()
|
fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result {
|
||||||
|
write!(f, "{}", self.layer_desc().short_id())
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
|||||||
@@ -222,13 +222,14 @@ impl Layer for DeltaLayer {
|
|||||||
/// debugging function to print out the contents of the layer
|
/// debugging function to print out the contents of the layer
|
||||||
fn dump(&self, verbose: bool, ctx: &RequestContext) -> Result<()> {
|
fn dump(&self, verbose: bool, ctx: &RequestContext) -> Result<()> {
|
||||||
println!(
|
println!(
|
||||||
"----- delta layer for ten {} tli {} keys {}-{} lsn {}-{} ----",
|
"----- delta layer for ten {} tli {} keys {}-{} lsn {}-{} size {} ----",
|
||||||
self.desc.tenant_id,
|
self.desc.tenant_id,
|
||||||
self.desc.timeline_id,
|
self.desc.timeline_id,
|
||||||
self.desc.key_range.start,
|
self.desc.key_range.start,
|
||||||
self.desc.key_range.end,
|
self.desc.key_range.end,
|
||||||
self.desc.lsn_range.start,
|
self.desc.lsn_range.start,
|
||||||
self.desc.lsn_range.end
|
self.desc.lsn_range.end,
|
||||||
|
self.desc.file_size,
|
||||||
);
|
);
|
||||||
|
|
||||||
if !verbose {
|
if !verbose {
|
||||||
@@ -394,10 +395,11 @@ impl Layer for DeltaLayer {
|
|||||||
fn is_incremental(&self) -> bool {
|
fn is_incremental(&self) -> bool {
|
||||||
self.layer_desc().is_incremental
|
self.layer_desc().is_incremental
|
||||||
}
|
}
|
||||||
|
}
|
||||||
/// Boilerplate to implement the Layer trait, always use layer_desc for persistent layers.
|
/// Boilerplate to implement the Layer trait, always use layer_desc for persistent layers.
|
||||||
fn short_id(&self) -> String {
|
impl std::fmt::Display for DeltaLayer {
|
||||||
self.layer_desc().short_id()
|
fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result {
|
||||||
|
write!(f, "{}", self.layer_desc().short_id())
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
|||||||
@@ -210,9 +210,15 @@ pub enum LayerFileName {
|
|||||||
|
|
||||||
impl LayerFileName {
|
impl LayerFileName {
|
||||||
pub fn file_name(&self) -> String {
|
pub fn file_name(&self) -> String {
|
||||||
|
self.to_string()
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
impl fmt::Display for LayerFileName {
|
||||||
|
fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result {
|
||||||
match self {
|
match self {
|
||||||
Self::Image(fname) => fname.to_string(),
|
Self::Image(fname) => write!(f, "{fname}"),
|
||||||
Self::Delta(fname) => fname.to_string(),
|
Self::Delta(fname) => write!(f, "{fname}"),
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -153,12 +153,14 @@ impl Layer for ImageLayer {
|
|||||||
/// debugging function to print out the contents of the layer
|
/// debugging function to print out the contents of the layer
|
||||||
fn dump(&self, verbose: bool, ctx: &RequestContext) -> Result<()> {
|
fn dump(&self, verbose: bool, ctx: &RequestContext) -> Result<()> {
|
||||||
println!(
|
println!(
|
||||||
"----- image layer for ten {} tli {} key {}-{} at {} ----",
|
"----- image layer for ten {} tli {} key {}-{} at {} is_incremental {} size {} ----",
|
||||||
self.desc.tenant_id,
|
self.desc.tenant_id,
|
||||||
self.desc.timeline_id,
|
self.desc.timeline_id,
|
||||||
self.desc.key_range.start,
|
self.desc.key_range.start,
|
||||||
self.desc.key_range.end,
|
self.desc.key_range.end,
|
||||||
self.lsn
|
self.lsn,
|
||||||
|
self.desc.is_incremental,
|
||||||
|
self.desc.file_size
|
||||||
);
|
);
|
||||||
|
|
||||||
if !verbose {
|
if !verbose {
|
||||||
@@ -230,10 +232,12 @@ impl Layer for ImageLayer {
|
|||||||
fn is_incremental(&self) -> bool {
|
fn is_incremental(&self) -> bool {
|
||||||
self.layer_desc().is_incremental
|
self.layer_desc().is_incremental
|
||||||
}
|
}
|
||||||
|
}
|
||||||
|
|
||||||
/// Boilerplate to implement the Layer trait, always use layer_desc for persistent layers.
|
/// Boilerplate to implement the Layer trait, always use layer_desc for persistent layers.
|
||||||
fn short_id(&self) -> String {
|
impl std::fmt::Display for ImageLayer {
|
||||||
self.layer_desc().short_id()
|
fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result {
|
||||||
|
write!(f, "{}", self.layer_desc().short_id())
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
|||||||
@@ -131,13 +131,6 @@ impl Layer for InMemoryLayer {
|
|||||||
true
|
true
|
||||||
}
|
}
|
||||||
|
|
||||||
fn short_id(&self) -> String {
|
|
||||||
let inner = self.inner.read().unwrap();
|
|
||||||
|
|
||||||
let end_lsn = inner.end_lsn.unwrap_or(Lsn(u64::MAX));
|
|
||||||
format!("inmem-{:016X}-{:016X}", self.start_lsn.0, end_lsn.0)
|
|
||||||
}
|
|
||||||
|
|
||||||
/// debugging function to print out the contents of the layer
|
/// debugging function to print out the contents of the layer
|
||||||
fn dump(&self, verbose: bool, _ctx: &RequestContext) -> Result<()> {
|
fn dump(&self, verbose: bool, _ctx: &RequestContext) -> Result<()> {
|
||||||
let inner = self.inner.read().unwrap();
|
let inner = self.inner.read().unwrap();
|
||||||
@@ -240,6 +233,15 @@ impl Layer for InMemoryLayer {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
impl std::fmt::Display for InMemoryLayer {
|
||||||
|
fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result {
|
||||||
|
let inner = self.inner.read().unwrap();
|
||||||
|
|
||||||
|
let end_lsn = inner.end_lsn.unwrap_or(Lsn(u64::MAX));
|
||||||
|
write!(f, "inmem-{:016X}-{:016X}", self.start_lsn.0, end_lsn.0)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
impl InMemoryLayer {
|
impl InMemoryLayer {
|
||||||
///
|
///
|
||||||
/// Get layer size on the disk
|
/// Get layer size on the disk
|
||||||
|
|||||||
@@ -1,4 +1,5 @@
|
|||||||
use anyhow::Result;
|
use anyhow::Result;
|
||||||
|
use core::fmt::Display;
|
||||||
use std::ops::Range;
|
use std::ops::Range;
|
||||||
use utils::{
|
use utils::{
|
||||||
id::{TenantId, TimelineId},
|
id::{TenantId, TimelineId},
|
||||||
@@ -48,8 +49,8 @@ impl PersistentLayerDesc {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
pub fn short_id(&self) -> String {
|
pub fn short_id(&self) -> impl Display {
|
||||||
self.filename().file_name()
|
self.filename()
|
||||||
}
|
}
|
||||||
|
|
||||||
#[cfg(test)]
|
#[cfg(test)]
|
||||||
@@ -173,13 +174,16 @@ impl PersistentLayerDesc {
|
|||||||
|
|
||||||
pub fn dump(&self, _verbose: bool, _ctx: &RequestContext) -> Result<()> {
|
pub fn dump(&self, _verbose: bool, _ctx: &RequestContext) -> Result<()> {
|
||||||
println!(
|
println!(
|
||||||
"----- layer for ten {} tli {} keys {}-{} lsn {}-{} ----",
|
"----- layer for ten {} tli {} keys {}-{} lsn {}-{} is_delta {} is_incremental {} size {} ----",
|
||||||
self.tenant_id,
|
self.tenant_id,
|
||||||
self.timeline_id,
|
self.timeline_id,
|
||||||
self.key_range.start,
|
self.key_range.start,
|
||||||
self.key_range.end,
|
self.key_range.end,
|
||||||
self.lsn_range.start,
|
self.lsn_range.start,
|
||||||
self.lsn_range.end
|
self.lsn_range.end,
|
||||||
|
self.is_delta,
|
||||||
|
self.is_incremental,
|
||||||
|
self.file_size,
|
||||||
);
|
);
|
||||||
|
|
||||||
Ok(())
|
Ok(())
|
||||||
|
|||||||
@@ -71,22 +71,22 @@ impl Layer for RemoteLayer {
|
|||||||
_reconstruct_state: &mut ValueReconstructState,
|
_reconstruct_state: &mut ValueReconstructState,
|
||||||
_ctx: &RequestContext,
|
_ctx: &RequestContext,
|
||||||
) -> Result<ValueReconstructResult> {
|
) -> Result<ValueReconstructResult> {
|
||||||
bail!(
|
bail!("layer {self} needs to be downloaded");
|
||||||
"layer {} needs to be downloaded",
|
|
||||||
self.filename().file_name()
|
|
||||||
);
|
|
||||||
}
|
}
|
||||||
|
|
||||||
/// debugging function to print out the contents of the layer
|
/// debugging function to print out the contents of the layer
|
||||||
fn dump(&self, _verbose: bool, _ctx: &RequestContext) -> Result<()> {
|
fn dump(&self, _verbose: bool, _ctx: &RequestContext) -> Result<()> {
|
||||||
println!(
|
println!(
|
||||||
"----- remote layer for ten {} tli {} keys {}-{} lsn {}-{} ----",
|
"----- remote layer for ten {} tli {} keys {}-{} lsn {}-{} is_delta {} is_incremental {} size {} ----",
|
||||||
self.desc.tenant_id,
|
self.desc.tenant_id,
|
||||||
self.desc.timeline_id,
|
self.desc.timeline_id,
|
||||||
self.desc.key_range.start,
|
self.desc.key_range.start,
|
||||||
self.desc.key_range.end,
|
self.desc.key_range.end,
|
||||||
self.desc.lsn_range.start,
|
self.desc.lsn_range.start,
|
||||||
self.desc.lsn_range.end
|
self.desc.lsn_range.end,
|
||||||
|
self.desc.is_delta,
|
||||||
|
self.desc.is_incremental,
|
||||||
|
self.desc.file_size,
|
||||||
);
|
);
|
||||||
|
|
||||||
Ok(())
|
Ok(())
|
||||||
@@ -106,10 +106,12 @@ impl Layer for RemoteLayer {
|
|||||||
fn is_incremental(&self) -> bool {
|
fn is_incremental(&self) -> bool {
|
||||||
self.layer_desc().is_incremental
|
self.layer_desc().is_incremental
|
||||||
}
|
}
|
||||||
|
}
|
||||||
|
|
||||||
/// Boilerplate to implement the Layer trait, always use layer_desc for persistent layers.
|
/// Boilerplate to implement the Layer trait, always use layer_desc for persistent layers.
|
||||||
fn short_id(&self) -> String {
|
impl std::fmt::Display for RemoteLayer {
|
||||||
self.layer_desc().short_id()
|
fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result {
|
||||||
|
write!(f, "{}", self.layer_desc().short_id())
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
|||||||
@@ -1,6 +1,9 @@
|
|||||||
//!
|
//!
|
||||||
|
|
||||||
mod eviction_task;
|
mod eviction_task;
|
||||||
|
mod logical_size;
|
||||||
|
pub mod span;
|
||||||
|
pub mod uninit;
|
||||||
mod walreceiver;
|
mod walreceiver;
|
||||||
|
|
||||||
use anyhow::{anyhow, bail, ensure, Context, Result};
|
use anyhow::{anyhow, bail, ensure, Context, Result};
|
||||||
@@ -8,7 +11,6 @@ use bytes::Bytes;
|
|||||||
use fail::fail_point;
|
use fail::fail_point;
|
||||||
use futures::StreamExt;
|
use futures::StreamExt;
|
||||||
use itertools::Itertools;
|
use itertools::Itertools;
|
||||||
use once_cell::sync::OnceCell;
|
|
||||||
use pageserver_api::models::{
|
use pageserver_api::models::{
|
||||||
DownloadRemoteLayersTaskInfo, DownloadRemoteLayersTaskSpawnRequest,
|
DownloadRemoteLayersTaskInfo, DownloadRemoteLayersTaskSpawnRequest,
|
||||||
DownloadRemoteLayersTaskState, LayerMapInfo, LayerResidenceEventReason, LayerResidenceStatus,
|
DownloadRemoteLayersTaskState, LayerMapInfo, LayerResidenceEventReason, LayerResidenceStatus,
|
||||||
@@ -17,7 +19,7 @@ use pageserver_api::models::{
|
|||||||
use remote_storage::GenericRemoteStorage;
|
use remote_storage::GenericRemoteStorage;
|
||||||
use serde_with::serde_as;
|
use serde_with::serde_as;
|
||||||
use storage_broker::BrokerClientChannel;
|
use storage_broker::BrokerClientChannel;
|
||||||
use tokio::sync::{oneshot, watch, Semaphore, TryAcquireError};
|
use tokio::sync::{oneshot, watch, TryAcquireError};
|
||||||
use tokio_util::sync::CancellationToken;
|
use tokio_util::sync::CancellationToken;
|
||||||
use tracing::*;
|
use tracing::*;
|
||||||
use utils::id::TenantTimelineId;
|
use utils::id::TenantTimelineId;
|
||||||
@@ -28,7 +30,7 @@ use std::fs;
|
|||||||
use std::ops::{Deref, Range};
|
use std::ops::{Deref, Range};
|
||||||
use std::path::{Path, PathBuf};
|
use std::path::{Path, PathBuf};
|
||||||
use std::pin::pin;
|
use std::pin::pin;
|
||||||
use std::sync::atomic::{AtomicI64, Ordering as AtomicOrdering};
|
use std::sync::atomic::Ordering as AtomicOrdering;
|
||||||
use std::sync::{Arc, Mutex, RwLock, Weak};
|
use std::sync::{Arc, Mutex, RwLock, Weak};
|
||||||
use std::time::{Duration, Instant, SystemTime};
|
use std::time::{Duration, Instant, SystemTime};
|
||||||
|
|
||||||
@@ -38,6 +40,7 @@ use crate::tenant::storage_layer::{
|
|||||||
DeltaFileName, DeltaLayerWriter, ImageFileName, ImageLayerWriter, InMemoryLayer,
|
DeltaFileName, DeltaLayerWriter, ImageFileName, ImageLayerWriter, InMemoryLayer,
|
||||||
LayerAccessStats, LayerFileName, RemoteLayer,
|
LayerAccessStats, LayerFileName, RemoteLayer,
|
||||||
};
|
};
|
||||||
|
use crate::tenant::timeline::logical_size::CurrentLogicalSize;
|
||||||
use crate::tenant::{
|
use crate::tenant::{
|
||||||
ephemeral_file::is_ephemeral_file,
|
ephemeral_file::is_ephemeral_file,
|
||||||
layer_map::{LayerMap, SearchResult},
|
layer_map::{LayerMap, SearchResult},
|
||||||
@@ -79,6 +82,7 @@ use crate::{is_temporary, task_mgr};
|
|||||||
|
|
||||||
pub(super) use self::eviction_task::EvictionTaskTenantState;
|
pub(super) use self::eviction_task::EvictionTaskTenantState;
|
||||||
use self::eviction_task::EvictionTaskTimelineState;
|
use self::eviction_task::EvictionTaskTimelineState;
|
||||||
|
use self::logical_size::LogicalSize;
|
||||||
use self::walreceiver::{WalReceiver, WalReceiverConf};
|
use self::walreceiver::{WalReceiver, WalReceiverConf};
|
||||||
|
|
||||||
use super::config::TenantConf;
|
use super::config::TenantConf;
|
||||||
@@ -128,7 +132,7 @@ impl LayerFileManager {
|
|||||||
// A layer's descriptor is present in the LayerMap => the LayerFileManager contains a layer for the descriptor.
|
// A layer's descriptor is present in the LayerMap => the LayerFileManager contains a layer for the descriptor.
|
||||||
self.0
|
self.0
|
||||||
.get(&desc.key())
|
.get(&desc.key())
|
||||||
.with_context(|| format!("get layer from desc: {}", desc.filename().file_name()))
|
.with_context(|| format!("get layer from desc: {}", desc.filename()))
|
||||||
.expect("not found")
|
.expect("not found")
|
||||||
.clone()
|
.clone()
|
||||||
}
|
}
|
||||||
@@ -365,126 +369,6 @@ pub struct Timeline {
|
|||||||
initial_logical_size_attempt: Mutex<Option<completion::Completion>>,
|
initial_logical_size_attempt: Mutex<Option<completion::Completion>>,
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Internal structure to hold all data needed for logical size calculation.
|
|
||||||
///
|
|
||||||
/// Calculation consists of two stages:
|
|
||||||
///
|
|
||||||
/// 1. Initial size calculation. That might take a long time, because it requires
|
|
||||||
/// reading all layers containing relation sizes at `initial_part_end`.
|
|
||||||
///
|
|
||||||
/// 2. Collecting an incremental part and adding that to the initial size.
|
|
||||||
/// Increments are appended on walreceiver writing new timeline data,
|
|
||||||
/// which result in increase or decrease of the logical size.
|
|
||||||
struct LogicalSize {
|
|
||||||
/// Size, potentially slow to compute. Calculating this might require reading multiple
|
|
||||||
/// layers, and even ancestor's layers.
|
|
||||||
///
|
|
||||||
/// NOTE: size at a given LSN is constant, but after a restart we will calculate
|
|
||||||
/// the initial size at a different LSN.
|
|
||||||
initial_logical_size: OnceCell<u64>,
|
|
||||||
|
|
||||||
/// Semaphore to track ongoing calculation of `initial_logical_size`.
|
|
||||||
initial_size_computation: Arc<tokio::sync::Semaphore>,
|
|
||||||
|
|
||||||
/// Latest Lsn that has its size uncalculated, could be absent for freshly created timelines.
|
|
||||||
initial_part_end: Option<Lsn>,
|
|
||||||
|
|
||||||
/// All other size changes after startup, combined together.
|
|
||||||
///
|
|
||||||
/// Size shouldn't ever be negative, but this is signed for two reasons:
|
|
||||||
///
|
|
||||||
/// 1. If we initialized the "baseline" size lazily, while we already
|
|
||||||
/// process incoming WAL, the incoming WAL records could decrement the
|
|
||||||
/// variable and temporarily make it negative. (This is just future-proofing;
|
|
||||||
/// the initialization is currently not done lazily.)
|
|
||||||
///
|
|
||||||
/// 2. If there is a bug and we e.g. forget to increment it in some cases
|
|
||||||
/// when size grows, but remember to decrement it when it shrinks again, the
|
|
||||||
/// variable could go negative. In that case, it seems better to at least
|
|
||||||
/// try to keep tracking it, rather than clamp or overflow it. Note that
|
|
||||||
/// get_current_logical_size() will clamp the returned value to zero if it's
|
|
||||||
/// negative, and log an error. Could set it permanently to zero or some
|
|
||||||
/// special value to indicate "broken" instead, but this will do for now.
|
|
||||||
///
|
|
||||||
/// Note that we also expose a copy of this value as a prometheus metric,
|
|
||||||
/// see `current_logical_size_gauge`. Use the `update_current_logical_size`
|
|
||||||
/// to modify this, it will also keep the prometheus metric in sync.
|
|
||||||
size_added_after_initial: AtomicI64,
|
|
||||||
}
|
|
||||||
|
|
||||||
/// Normalized current size, that the data in pageserver occupies.
|
|
||||||
#[derive(Debug, Clone, Copy)]
|
|
||||||
enum CurrentLogicalSize {
|
|
||||||
/// The size is not yet calculated to the end, this is an intermediate result,
|
|
||||||
/// constructed from walreceiver increments and normalized: logical data could delete some objects, hence be negative,
|
|
||||||
/// yet total logical size cannot be below 0.
|
|
||||||
Approximate(u64),
|
|
||||||
// Fully calculated logical size, only other future walreceiver increments are changing it, and those changes are
|
|
||||||
// available for observation without any calculations.
|
|
||||||
Exact(u64),
|
|
||||||
}
|
|
||||||
|
|
||||||
impl CurrentLogicalSize {
|
|
||||||
fn size(&self) -> u64 {
|
|
||||||
*match self {
|
|
||||||
Self::Approximate(size) => size,
|
|
||||||
Self::Exact(size) => size,
|
|
||||||
}
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
impl LogicalSize {
|
|
||||||
fn empty_initial() -> Self {
|
|
||||||
Self {
|
|
||||||
initial_logical_size: OnceCell::with_value(0),
|
|
||||||
// initial_logical_size already computed, so, don't admit any calculations
|
|
||||||
initial_size_computation: Arc::new(Semaphore::new(0)),
|
|
||||||
initial_part_end: None,
|
|
||||||
size_added_after_initial: AtomicI64::new(0),
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
fn deferred_initial(compute_to: Lsn) -> Self {
|
|
||||||
Self {
|
|
||||||
initial_logical_size: OnceCell::new(),
|
|
||||||
initial_size_computation: Arc::new(Semaphore::new(1)),
|
|
||||||
initial_part_end: Some(compute_to),
|
|
||||||
size_added_after_initial: AtomicI64::new(0),
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
fn current_size(&self) -> anyhow::Result<CurrentLogicalSize> {
|
|
||||||
let size_increment: i64 = self.size_added_after_initial.load(AtomicOrdering::Acquire);
|
|
||||||
// ^^^ keep this type explicit so that the casts in this function break if
|
|
||||||
// we change the type.
|
|
||||||
match self.initial_logical_size.get() {
|
|
||||||
Some(initial_size) => {
|
|
||||||
initial_size.checked_add_signed(size_increment)
|
|
||||||
.with_context(|| format!("Overflow during logical size calculation, initial_size: {initial_size}, size_increment: {size_increment}"))
|
|
||||||
.map(CurrentLogicalSize::Exact)
|
|
||||||
}
|
|
||||||
None => {
|
|
||||||
let non_negative_size_increment = u64::try_from(size_increment).unwrap_or(0);
|
|
||||||
Ok(CurrentLogicalSize::Approximate(non_negative_size_increment))
|
|
||||||
}
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
fn increment_size(&self, delta: i64) {
|
|
||||||
self.size_added_after_initial
|
|
||||||
.fetch_add(delta, AtomicOrdering::SeqCst);
|
|
||||||
}
|
|
||||||
|
|
||||||
/// Make the value computed by initial logical size computation
|
|
||||||
/// available for re-use. This doesn't contain the incremental part.
|
|
||||||
fn initialized_size(&self, lsn: Lsn) -> Option<u64> {
|
|
||||||
match self.initial_part_end {
|
|
||||||
Some(v) if v == lsn => self.initial_logical_size.get().copied(),
|
|
||||||
_ => None,
|
|
||||||
}
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
pub struct WalReceiverInfo {
|
pub struct WalReceiverInfo {
|
||||||
pub wal_source_connconf: PgConnectionConfig,
|
pub wal_source_connconf: PgConnectionConfig,
|
||||||
pub last_received_msg_lsn: Lsn,
|
pub last_received_msg_lsn: Lsn,
|
||||||
@@ -1381,9 +1265,9 @@ impl Timeline {
|
|||||||
.read()
|
.read()
|
||||||
.unwrap()
|
.unwrap()
|
||||||
.observe(delta);
|
.observe(delta);
|
||||||
info!(layer=%local_layer.short_id(), residence_millis=delta.as_millis(), "evicted layer after known residence period");
|
info!(layer=%local_layer, residence_millis=delta.as_millis(), "evicted layer after known residence period");
|
||||||
} else {
|
} else {
|
||||||
info!(layer=%local_layer.short_id(), "evicted layer after unknown residence period");
|
info!(layer=%local_layer, "evicted layer after unknown residence period");
|
||||||
}
|
}
|
||||||
|
|
||||||
true
|
true
|
||||||
@@ -2239,7 +2123,7 @@ impl Timeline {
|
|||||||
ctx: &RequestContext,
|
ctx: &RequestContext,
|
||||||
cancel: CancellationToken,
|
cancel: CancellationToken,
|
||||||
) -> Result<u64, CalculateLogicalSizeError> {
|
) -> Result<u64, CalculateLogicalSizeError> {
|
||||||
debug_assert_current_span_has_tenant_and_timeline_id();
|
span::debug_assert_current_span_has_tenant_and_timeline_id();
|
||||||
|
|
||||||
let mut timeline_state_updates = self.subscribe_for_state_updates();
|
let mut timeline_state_updates = self.subscribe_for_state_updates();
|
||||||
let self_calculation = Arc::clone(self);
|
let self_calculation = Arc::clone(self);
|
||||||
@@ -2462,11 +2346,7 @@ impl TraversalLayerExt for Arc<dyn PersistentLayer> {
|
|||||||
format!("{}", local_path.display())
|
format!("{}", local_path.display())
|
||||||
}
|
}
|
||||||
None => {
|
None => {
|
||||||
format!(
|
format!("remote {}/{self}", self.get_timeline_id())
|
||||||
"remote {}/{}",
|
|
||||||
self.get_timeline_id(),
|
|
||||||
self.filename().file_name()
|
|
||||||
)
|
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
@@ -2474,11 +2354,7 @@ impl TraversalLayerExt for Arc<dyn PersistentLayer> {
|
|||||||
|
|
||||||
impl TraversalLayerExt for Arc<InMemoryLayer> {
|
impl TraversalLayerExt for Arc<InMemoryLayer> {
|
||||||
fn traversal_id(&self) -> TraversalId {
|
fn traversal_id(&self) -> TraversalId {
|
||||||
format!(
|
format!("timeline {} in-memory {self}", self.get_timeline_id())
|
||||||
"timeline {} in-memory {}",
|
|
||||||
self.get_timeline_id(),
|
|
||||||
self.short_id()
|
|
||||||
)
|
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -2998,7 +2874,7 @@ impl Timeline {
|
|||||||
}
|
}
|
||||||
|
|
||||||
/// Flush one frozen in-memory layer to disk, as a new delta layer.
|
/// Flush one frozen in-memory layer to disk, as a new delta layer.
|
||||||
#[instrument(skip_all, fields(tenant_id=%self.tenant_id, timeline_id=%self.timeline_id, layer=%frozen_layer.short_id()))]
|
#[instrument(skip_all, fields(tenant_id=%self.tenant_id, timeline_id=%self.timeline_id, layer=%frozen_layer))]
|
||||||
async fn flush_frozen_layer(
|
async fn flush_frozen_layer(
|
||||||
self: &Arc<Self>,
|
self: &Arc<Self>,
|
||||||
frozen_layer: Arc<InMemoryLayer>,
|
frozen_layer: Arc<InMemoryLayer>,
|
||||||
@@ -3677,7 +3553,7 @@ impl Timeline {
|
|||||||
let remotes = deltas_to_compact
|
let remotes = deltas_to_compact
|
||||||
.iter()
|
.iter()
|
||||||
.filter(|l| l.is_remote_layer())
|
.filter(|l| l.is_remote_layer())
|
||||||
.inspect(|l| info!("compact requires download of {}", l.filename().file_name()))
|
.inspect(|l| info!("compact requires download of {l}"))
|
||||||
.map(|l| {
|
.map(|l| {
|
||||||
l.clone()
|
l.clone()
|
||||||
.downcast_remote_layer()
|
.downcast_remote_layer()
|
||||||
@@ -3701,7 +3577,7 @@ impl Timeline {
|
|||||||
);
|
);
|
||||||
|
|
||||||
for l in deltas_to_compact.iter() {
|
for l in deltas_to_compact.iter() {
|
||||||
info!("compact includes {}", l.filename().file_name());
|
info!("compact includes {l}");
|
||||||
}
|
}
|
||||||
|
|
||||||
// We don't need the original list of layers anymore. Drop it so that
|
// We don't need the original list of layers anymore. Drop it so that
|
||||||
@@ -4316,8 +4192,8 @@ impl Timeline {
|
|||||||
if l.get_lsn_range().end > horizon_cutoff {
|
if l.get_lsn_range().end > horizon_cutoff {
|
||||||
debug!(
|
debug!(
|
||||||
"keeping {} because it's newer than horizon_cutoff {}",
|
"keeping {} because it's newer than horizon_cutoff {}",
|
||||||
l.filename().file_name(),
|
l.filename(),
|
||||||
horizon_cutoff
|
horizon_cutoff,
|
||||||
);
|
);
|
||||||
result.layers_needed_by_cutoff += 1;
|
result.layers_needed_by_cutoff += 1;
|
||||||
continue 'outer;
|
continue 'outer;
|
||||||
@@ -4327,8 +4203,8 @@ impl Timeline {
|
|||||||
if l.get_lsn_range().end > pitr_cutoff {
|
if l.get_lsn_range().end > pitr_cutoff {
|
||||||
debug!(
|
debug!(
|
||||||
"keeping {} because it's newer than pitr_cutoff {}",
|
"keeping {} because it's newer than pitr_cutoff {}",
|
||||||
l.filename().file_name(),
|
l.filename(),
|
||||||
pitr_cutoff
|
pitr_cutoff,
|
||||||
);
|
);
|
||||||
result.layers_needed_by_pitr += 1;
|
result.layers_needed_by_pitr += 1;
|
||||||
continue 'outer;
|
continue 'outer;
|
||||||
@@ -4346,7 +4222,7 @@ impl Timeline {
|
|||||||
if &l.get_lsn_range().start <= retain_lsn {
|
if &l.get_lsn_range().start <= retain_lsn {
|
||||||
debug!(
|
debug!(
|
||||||
"keeping {} because it's still might be referenced by child branch forked at {} is_dropped: xx is_incremental: {}",
|
"keeping {} because it's still might be referenced by child branch forked at {} is_dropped: xx is_incremental: {}",
|
||||||
l.filename().file_name(),
|
l.filename(),
|
||||||
retain_lsn,
|
retain_lsn,
|
||||||
l.is_incremental(),
|
l.is_incremental(),
|
||||||
);
|
);
|
||||||
@@ -4377,10 +4253,7 @@ impl Timeline {
|
|||||||
if !layers
|
if !layers
|
||||||
.image_layer_exists(&l.get_key_range(), &(l.get_lsn_range().end..new_gc_cutoff))?
|
.image_layer_exists(&l.get_key_range(), &(l.get_lsn_range().end..new_gc_cutoff))?
|
||||||
{
|
{
|
||||||
debug!(
|
debug!("keeping {} because it is the latest layer", l.filename());
|
||||||
"keeping {} because it is the latest layer",
|
|
||||||
l.filename().file_name()
|
|
||||||
);
|
|
||||||
// Collect delta key ranges that need image layers to allow garbage
|
// Collect delta key ranges that need image layers to allow garbage
|
||||||
// collecting the layers.
|
// collecting the layers.
|
||||||
// It is not so obvious whether we need to propagate information only about
|
// It is not so obvious whether we need to propagate information only about
|
||||||
@@ -4397,7 +4270,7 @@ impl Timeline {
|
|||||||
// We didn't find any reason to keep this file, so remove it.
|
// We didn't find any reason to keep this file, so remove it.
|
||||||
debug!(
|
debug!(
|
||||||
"garbage collecting {} is_dropped: xx is_incremental: {}",
|
"garbage collecting {} is_dropped: xx is_incremental: {}",
|
||||||
l.filename().file_name(),
|
l.filename(),
|
||||||
l.is_incremental(),
|
l.is_incremental(),
|
||||||
);
|
);
|
||||||
layers_to_remove.push(Arc::clone(&l));
|
layers_to_remove.push(Arc::clone(&l));
|
||||||
@@ -4551,12 +4424,12 @@ impl Timeline {
|
|||||||
/// If the caller has a deadline or needs a timeout, they can simply stop polling:
|
/// If the caller has a deadline or needs a timeout, they can simply stop polling:
|
||||||
/// we're **cancellation-safe** because the download happens in a separate task_mgr task.
|
/// we're **cancellation-safe** because the download happens in a separate task_mgr task.
|
||||||
/// So, the current download attempt will run to completion even if we stop polling.
|
/// So, the current download attempt will run to completion even if we stop polling.
|
||||||
#[instrument(skip_all, fields(layer=%remote_layer.short_id()))]
|
#[instrument(skip_all, fields(layer=%remote_layer))]
|
||||||
pub async fn download_remote_layer(
|
pub async fn download_remote_layer(
|
||||||
&self,
|
&self,
|
||||||
remote_layer: Arc<RemoteLayer>,
|
remote_layer: Arc<RemoteLayer>,
|
||||||
) -> anyhow::Result<()> {
|
) -> anyhow::Result<()> {
|
||||||
debug_assert_current_span_has_tenant_and_timeline_id();
|
span::debug_assert_current_span_has_tenant_and_timeline_id();
|
||||||
|
|
||||||
use std::sync::atomic::Ordering::Relaxed;
|
use std::sync::atomic::Ordering::Relaxed;
|
||||||
|
|
||||||
@@ -4589,7 +4462,7 @@ impl Timeline {
|
|||||||
TaskKind::RemoteDownloadTask,
|
TaskKind::RemoteDownloadTask,
|
||||||
Some(self.tenant_id),
|
Some(self.tenant_id),
|
||||||
Some(self.timeline_id),
|
Some(self.timeline_id),
|
||||||
&format!("download layer {}", remote_layer.short_id()),
|
&format!("download layer {}", remote_layer),
|
||||||
false,
|
false,
|
||||||
async move {
|
async move {
|
||||||
let remote_client = self_clone.remote_client.as_ref().unwrap();
|
let remote_client = self_clone.remote_client.as_ref().unwrap();
|
||||||
@@ -4865,15 +4738,12 @@ impl Timeline {
|
|||||||
continue;
|
continue;
|
||||||
}
|
}
|
||||||
|
|
||||||
let last_activity_ts = l
|
let last_activity_ts = l.access_stats().latest_activity().unwrap_or_else(|| {
|
||||||
.access_stats()
|
// We only use this fallback if there's an implementation error.
|
||||||
.latest_activity()
|
// `latest_activity` already does rate-limited warn!() log.
|
||||||
.unwrap_or_else(|| {
|
debug!(layer=%l, "last_activity returns None, using SystemTime::now");
|
||||||
// We only use this fallback if there's an implementation error.
|
SystemTime::now()
|
||||||
// `latest_activity` already does rate-limited warn!() log.
|
});
|
||||||
debug!(layer=%l.filename().file_name(), "last_activity returns None, using SystemTime::now");
|
|
||||||
SystemTime::now()
|
|
||||||
});
|
|
||||||
|
|
||||||
resident_layers.push(LocalLayerInfoForDiskUsageEviction {
|
resident_layers.push(LocalLayerInfoForDiskUsageEviction {
|
||||||
layer: l,
|
layer: l,
|
||||||
@@ -4993,33 +4863,6 @@ fn rename_to_backup(path: &Path) -> anyhow::Result<()> {
|
|||||||
bail!("couldn't find an unused backup number for {:?}", path)
|
bail!("couldn't find an unused backup number for {:?}", path)
|
||||||
}
|
}
|
||||||
|
|
||||||
#[cfg(not(debug_assertions))]
|
|
||||||
#[inline]
|
|
||||||
pub(crate) fn debug_assert_current_span_has_tenant_and_timeline_id() {}
|
|
||||||
|
|
||||||
#[cfg(debug_assertions)]
|
|
||||||
#[inline]
|
|
||||||
pub(crate) fn debug_assert_current_span_has_tenant_and_timeline_id() {
|
|
||||||
use utils::tracing_span_assert;
|
|
||||||
|
|
||||||
pub static TIMELINE_ID_EXTRACTOR: once_cell::sync::Lazy<
|
|
||||||
tracing_span_assert::MultiNameExtractor<2>,
|
|
||||||
> = once_cell::sync::Lazy::new(|| {
|
|
||||||
tracing_span_assert::MultiNameExtractor::new("TimelineId", ["timeline_id", "timeline"])
|
|
||||||
});
|
|
||||||
|
|
||||||
match tracing_span_assert::check_fields_present([
|
|
||||||
&*super::TENANT_ID_EXTRACTOR,
|
|
||||||
&*TIMELINE_ID_EXTRACTOR,
|
|
||||||
]) {
|
|
||||||
Ok(()) => (),
|
|
||||||
Err(missing) => panic!(
|
|
||||||
"missing extractors: {:?}",
|
|
||||||
missing.into_iter().map(|e| e.name()).collect::<Vec<_>>()
|
|
||||||
),
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
/// Similar to `Arc::ptr_eq`, but only compares the object pointers, not vtables.
|
/// Similar to `Arc::ptr_eq`, but only compares the object pointers, not vtables.
|
||||||
///
|
///
|
||||||
/// Returns `true` if the two `Arc` point to the same layer, false otherwise.
|
/// Returns `true` if the two `Arc` point to the same layer, false otherwise.
|
||||||
|
|||||||
@@ -70,7 +70,6 @@ impl Timeline {
|
|||||||
};
|
};
|
||||||
|
|
||||||
self_clone.eviction_task(cancel).await;
|
self_clone.eviction_task(cancel).await;
|
||||||
info!("eviction task finishing");
|
|
||||||
Ok(())
|
Ok(())
|
||||||
},
|
},
|
||||||
);
|
);
|
||||||
@@ -78,6 +77,9 @@ impl Timeline {
|
|||||||
|
|
||||||
#[instrument(skip_all, fields(tenant_id = %self.tenant_id, timeline_id = %self.timeline_id))]
|
#[instrument(skip_all, fields(tenant_id = %self.tenant_id, timeline_id = %self.timeline_id))]
|
||||||
async fn eviction_task(self: Arc<Self>, cancel: CancellationToken) {
|
async fn eviction_task(self: Arc<Self>, cancel: CancellationToken) {
|
||||||
|
scopeguard::defer! {
|
||||||
|
info!("eviction task finishing");
|
||||||
|
}
|
||||||
use crate::tenant::tasks::random_init_delay;
|
use crate::tenant::tasks::random_init_delay;
|
||||||
{
|
{
|
||||||
let policy = self.get_eviction_policy();
|
let policy = self.get_eviction_policy();
|
||||||
@@ -86,7 +88,6 @@ impl Timeline {
|
|||||||
EvictionPolicy::NoEviction => Duration::from_secs(10),
|
EvictionPolicy::NoEviction => Duration::from_secs(10),
|
||||||
};
|
};
|
||||||
if random_init_delay(period, &cancel).await.is_err() {
|
if random_init_delay(period, &cancel).await.is_err() {
|
||||||
info!("shutting down");
|
|
||||||
return;
|
return;
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
@@ -101,7 +102,6 @@ impl Timeline {
|
|||||||
ControlFlow::Continue(sleep_until) => {
|
ControlFlow::Continue(sleep_until) => {
|
||||||
tokio::select! {
|
tokio::select! {
|
||||||
_ = cancel.cancelled() => {
|
_ = cancel.cancelled() => {
|
||||||
info!("shutting down");
|
|
||||||
break;
|
break;
|
||||||
}
|
}
|
||||||
_ = tokio::time::sleep_until(sleep_until) => { }
|
_ = tokio::time::sleep_until(sleep_until) => { }
|
||||||
@@ -209,7 +209,7 @@ impl Timeline {
|
|||||||
let last_activity_ts = hist_layer.access_stats().latest_activity().unwrap_or_else(|| {
|
let last_activity_ts = hist_layer.access_stats().latest_activity().unwrap_or_else(|| {
|
||||||
// We only use this fallback if there's an implementation error.
|
// We only use this fallback if there's an implementation error.
|
||||||
// `latest_activity` already does rate-limited warn!() log.
|
// `latest_activity` already does rate-limited warn!() log.
|
||||||
debug!(layer=%hist_layer.filename().file_name(), "last_activity returns None, using SystemTime::now");
|
debug!(layer=%hist_layer, "last_activity returns None, using SystemTime::now");
|
||||||
SystemTime::now()
|
SystemTime::now()
|
||||||
});
|
});
|
||||||
|
|
||||||
|
|||||||
@@ -0,0 +1,128 @@
|
|||||||
|
use anyhow::Context;
|
||||||
|
use once_cell::sync::OnceCell;
|
||||||
|
|
||||||
|
use tokio::sync::Semaphore;
|
||||||
|
use utils::lsn::Lsn;
|
||||||
|
|
||||||
|
use std::sync::atomic::{AtomicI64, Ordering as AtomicOrdering};
|
||||||
|
use std::sync::Arc;
|
||||||
|
|
||||||
|
/// Internal structure to hold all data needed for logical size calculation.
|
||||||
|
///
|
||||||
|
/// Calculation consists of two stages:
|
||||||
|
///
|
||||||
|
/// 1. Initial size calculation. That might take a long time, because it requires
|
||||||
|
/// reading all layers containing relation sizes at `initial_part_end`.
|
||||||
|
///
|
||||||
|
/// 2. Collecting an incremental part and adding that to the initial size.
|
||||||
|
/// Increments are appended on walreceiver writing new timeline data,
|
||||||
|
/// which result in increase or decrease of the logical size.
|
||||||
|
pub(super) struct LogicalSize {
|
||||||
|
/// Size, potentially slow to compute. Calculating this might require reading multiple
|
||||||
|
/// layers, and even ancestor's layers.
|
||||||
|
///
|
||||||
|
/// NOTE: size at a given LSN is constant, but after a restart we will calculate
|
||||||
|
/// the initial size at a different LSN.
|
||||||
|
pub initial_logical_size: OnceCell<u64>,
|
||||||
|
|
||||||
|
/// Semaphore to track ongoing calculation of `initial_logical_size`.
|
||||||
|
pub initial_size_computation: Arc<tokio::sync::Semaphore>,
|
||||||
|
|
||||||
|
/// Latest Lsn that has its size uncalculated, could be absent for freshly created timelines.
|
||||||
|
pub initial_part_end: Option<Lsn>,
|
||||||
|
|
||||||
|
/// All other size changes after startup, combined together.
|
||||||
|
///
|
||||||
|
/// Size shouldn't ever be negative, but this is signed for two reasons:
|
||||||
|
///
|
||||||
|
/// 1. If we initialized the "baseline" size lazily, while we already
|
||||||
|
/// process incoming WAL, the incoming WAL records could decrement the
|
||||||
|
/// variable and temporarily make it negative. (This is just future-proofing;
|
||||||
|
/// the initialization is currently not done lazily.)
|
||||||
|
///
|
||||||
|
/// 2. If there is a bug and we e.g. forget to increment it in some cases
|
||||||
|
/// when size grows, but remember to decrement it when it shrinks again, the
|
||||||
|
/// variable could go negative. In that case, it seems better to at least
|
||||||
|
/// try to keep tracking it, rather than clamp or overflow it. Note that
|
||||||
|
/// get_current_logical_size() will clamp the returned value to zero if it's
|
||||||
|
/// negative, and log an error. Could set it permanently to zero or some
|
||||||
|
/// special value to indicate "broken" instead, but this will do for now.
|
||||||
|
///
|
||||||
|
/// Note that we also expose a copy of this value as a prometheus metric,
|
||||||
|
/// see `current_logical_size_gauge`. Use the `update_current_logical_size`
|
||||||
|
/// to modify this, it will also keep the prometheus metric in sync.
|
||||||
|
pub size_added_after_initial: AtomicI64,
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Normalized current size, that the data in pageserver occupies.
|
||||||
|
#[derive(Debug, Clone, Copy)]
|
||||||
|
pub(super) enum CurrentLogicalSize {
|
||||||
|
/// The size is not yet calculated to the end, this is an intermediate result,
|
||||||
|
/// constructed from walreceiver increments and normalized: logical data could delete some objects, hence be negative,
|
||||||
|
/// yet total logical size cannot be below 0.
|
||||||
|
Approximate(u64),
|
||||||
|
// Fully calculated logical size, only other future walreceiver increments are changing it, and those changes are
|
||||||
|
// available for observation without any calculations.
|
||||||
|
Exact(u64),
|
||||||
|
}
|
||||||
|
|
||||||
|
impl CurrentLogicalSize {
|
||||||
|
pub(super) fn size(&self) -> u64 {
|
||||||
|
*match self {
|
||||||
|
Self::Approximate(size) => size,
|
||||||
|
Self::Exact(size) => size,
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
impl LogicalSize {
|
||||||
|
pub(super) fn empty_initial() -> Self {
|
||||||
|
Self {
|
||||||
|
initial_logical_size: OnceCell::with_value(0),
|
||||||
|
// initial_logical_size already computed, so, don't admit any calculations
|
||||||
|
initial_size_computation: Arc::new(Semaphore::new(0)),
|
||||||
|
initial_part_end: None,
|
||||||
|
size_added_after_initial: AtomicI64::new(0),
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
pub(super) fn deferred_initial(compute_to: Lsn) -> Self {
|
||||||
|
Self {
|
||||||
|
initial_logical_size: OnceCell::new(),
|
||||||
|
initial_size_computation: Arc::new(Semaphore::new(1)),
|
||||||
|
initial_part_end: Some(compute_to),
|
||||||
|
size_added_after_initial: AtomicI64::new(0),
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
pub(super) fn current_size(&self) -> anyhow::Result<CurrentLogicalSize> {
|
||||||
|
let size_increment: i64 = self.size_added_after_initial.load(AtomicOrdering::Acquire);
|
||||||
|
// ^^^ keep this type explicit so that the casts in this function break if
|
||||||
|
// we change the type.
|
||||||
|
match self.initial_logical_size.get() {
|
||||||
|
Some(initial_size) => {
|
||||||
|
initial_size.checked_add_signed(size_increment)
|
||||||
|
.with_context(|| format!("Overflow during logical size calculation, initial_size: {initial_size}, size_increment: {size_increment}"))
|
||||||
|
.map(CurrentLogicalSize::Exact)
|
||||||
|
}
|
||||||
|
None => {
|
||||||
|
let non_negative_size_increment = u64::try_from(size_increment).unwrap_or(0);
|
||||||
|
Ok(CurrentLogicalSize::Approximate(non_negative_size_increment))
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
pub(super) fn increment_size(&self, delta: i64) {
|
||||||
|
self.size_added_after_initial
|
||||||
|
.fetch_add(delta, AtomicOrdering::SeqCst);
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Make the value computed by initial logical size computation
|
||||||
|
/// available for re-use. This doesn't contain the incremental part.
|
||||||
|
pub(super) fn initialized_size(&self, lsn: Lsn) -> Option<u64> {
|
||||||
|
match self.initial_part_end {
|
||||||
|
Some(v) if v == lsn => self.initial_logical_size.get().copied(),
|
||||||
|
_ => None,
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
@@ -0,0 +1,25 @@
|
|||||||
|
#[cfg(debug_assertions)]
|
||||||
|
use utils::tracing_span_assert::{check_fields_present, Extractor, MultiNameExtractor};
|
||||||
|
|
||||||
|
#[cfg(not(debug_assertions))]
|
||||||
|
pub(crate) fn debug_assert_current_span_has_tenant_and_timeline_id() {}
|
||||||
|
|
||||||
|
#[cfg(debug_assertions)]
|
||||||
|
#[track_caller]
|
||||||
|
pub(crate) fn debug_assert_current_span_has_tenant_and_timeline_id() {
|
||||||
|
static TIMELINE_ID_EXTRACTOR: once_cell::sync::Lazy<MultiNameExtractor<2>> =
|
||||||
|
once_cell::sync::Lazy::new(|| {
|
||||||
|
MultiNameExtractor::new("TimelineId", ["timeline_id", "timeline"])
|
||||||
|
});
|
||||||
|
|
||||||
|
let fields: [&dyn Extractor; 2] = [
|
||||||
|
&*crate::tenant::span::TENANT_ID_EXTRACTOR,
|
||||||
|
&*TIMELINE_ID_EXTRACTOR,
|
||||||
|
];
|
||||||
|
if let Err(missing) = check_fields_present(fields) {
|
||||||
|
panic!(
|
||||||
|
"missing extractors: {:?}",
|
||||||
|
missing.into_iter().map(|e| e.name()).collect::<Vec<_>>()
|
||||||
|
)
|
||||||
|
}
|
||||||
|
}
|
||||||
@@ -0,0 +1,219 @@
|
|||||||
|
use std::{collections::hash_map::Entry, fs, path::PathBuf, sync::Arc};
|
||||||
|
|
||||||
|
use anyhow::Context;
|
||||||
|
use tracing::{error, info, info_span, warn};
|
||||||
|
use utils::{crashsafe, id::TimelineId, lsn::Lsn};
|
||||||
|
|
||||||
|
use crate::{
|
||||||
|
context::RequestContext,
|
||||||
|
import_datadir,
|
||||||
|
tenant::{ignore_absent_files, Tenant},
|
||||||
|
};
|
||||||
|
|
||||||
|
use super::Timeline;
|
||||||
|
|
||||||
|
/// A timeline with some of its files on disk, being initialized.
|
||||||
|
/// This struct ensures the atomicity of the timeline init: it's either properly created and inserted into pageserver's memory, or
|
||||||
|
/// its local files are removed. In the worst case of a crash, an uninit mark file is left behind, which causes the directory
|
||||||
|
/// to be removed on next restart.
|
||||||
|
///
|
||||||
|
/// The caller is responsible for proper timeline data filling before the final init.
|
||||||
|
#[must_use]
|
||||||
|
pub struct UninitializedTimeline<'t> {
|
||||||
|
pub(crate) owning_tenant: &'t Tenant,
|
||||||
|
timeline_id: TimelineId,
|
||||||
|
raw_timeline: Option<(Arc<Timeline>, TimelineUninitMark)>,
|
||||||
|
}
|
||||||
|
|
||||||
|
impl<'t> UninitializedTimeline<'t> {
|
||||||
|
pub(crate) fn new(
|
||||||
|
owning_tenant: &'t Tenant,
|
||||||
|
timeline_id: TimelineId,
|
||||||
|
raw_timeline: Option<(Arc<Timeline>, TimelineUninitMark)>,
|
||||||
|
) -> Self {
|
||||||
|
Self {
|
||||||
|
owning_tenant,
|
||||||
|
timeline_id,
|
||||||
|
raw_timeline,
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Finish timeline creation: insert it into the Tenant's timelines map and remove the
|
||||||
|
/// uninit mark file.
|
||||||
|
///
|
||||||
|
/// This function launches the flush loop if not already done.
|
||||||
|
///
|
||||||
|
/// The caller is responsible for activating the timeline (function `.activate()`).
|
||||||
|
pub(crate) fn finish_creation(mut self) -> anyhow::Result<Arc<Timeline>> {
|
||||||
|
let timeline_id = self.timeline_id;
|
||||||
|
let tenant_id = self.owning_tenant.tenant_id;
|
||||||
|
|
||||||
|
let (new_timeline, uninit_mark) = self.raw_timeline.take().with_context(|| {
|
||||||
|
format!("No timeline for initalization found for {tenant_id}/{timeline_id}")
|
||||||
|
})?;
|
||||||
|
|
||||||
|
// Check that the caller initialized disk_consistent_lsn
|
||||||
|
let new_disk_consistent_lsn = new_timeline.get_disk_consistent_lsn();
|
||||||
|
anyhow::ensure!(
|
||||||
|
new_disk_consistent_lsn.is_valid(),
|
||||||
|
"new timeline {tenant_id}/{timeline_id} has invalid disk_consistent_lsn"
|
||||||
|
);
|
||||||
|
|
||||||
|
let mut timelines = self.owning_tenant.timelines.lock().unwrap();
|
||||||
|
match timelines.entry(timeline_id) {
|
||||||
|
Entry::Occupied(_) => anyhow::bail!(
|
||||||
|
"Found freshly initialized timeline {tenant_id}/{timeline_id} in the tenant map"
|
||||||
|
),
|
||||||
|
Entry::Vacant(v) => {
|
||||||
|
uninit_mark.remove_uninit_mark().with_context(|| {
|
||||||
|
format!(
|
||||||
|
"Failed to remove uninit mark file for timeline {tenant_id}/{timeline_id}"
|
||||||
|
)
|
||||||
|
})?;
|
||||||
|
v.insert(Arc::clone(&new_timeline));
|
||||||
|
|
||||||
|
new_timeline.maybe_spawn_flush_loop();
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
Ok(new_timeline)
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Prepares timeline data by loading it from the basebackup archive.
|
||||||
|
pub(crate) async fn import_basebackup_from_tar(
|
||||||
|
self,
|
||||||
|
copyin_read: &mut (impl tokio::io::AsyncRead + Send + Sync + Unpin),
|
||||||
|
base_lsn: Lsn,
|
||||||
|
broker_client: storage_broker::BrokerClientChannel,
|
||||||
|
ctx: &RequestContext,
|
||||||
|
) -> anyhow::Result<Arc<Timeline>> {
|
||||||
|
let raw_timeline = self.raw_timeline()?;
|
||||||
|
|
||||||
|
import_datadir::import_basebackup_from_tar(raw_timeline, copyin_read, base_lsn, ctx)
|
||||||
|
.await
|
||||||
|
.context("Failed to import basebackup")?;
|
||||||
|
|
||||||
|
// Flush the new layer files to disk, before we make the timeline as available to
|
||||||
|
// the outside world.
|
||||||
|
//
|
||||||
|
// Flush loop needs to be spawned in order to be able to flush.
|
||||||
|
raw_timeline.maybe_spawn_flush_loop();
|
||||||
|
|
||||||
|
fail::fail_point!("before-checkpoint-new-timeline", |_| {
|
||||||
|
anyhow::bail!("failpoint before-checkpoint-new-timeline");
|
||||||
|
});
|
||||||
|
|
||||||
|
raw_timeline
|
||||||
|
.freeze_and_flush()
|
||||||
|
.await
|
||||||
|
.context("Failed to flush after basebackup import")?;
|
||||||
|
|
||||||
|
// All the data has been imported. Insert the Timeline into the tenant's timelines
|
||||||
|
// map and remove the uninit mark file.
|
||||||
|
let tl = self.finish_creation()?;
|
||||||
|
tl.activate(broker_client, None, ctx);
|
||||||
|
Ok(tl)
|
||||||
|
}
|
||||||
|
|
||||||
|
pub(crate) fn raw_timeline(&self) -> anyhow::Result<&Arc<Timeline>> {
|
||||||
|
Ok(&self
|
||||||
|
.raw_timeline
|
||||||
|
.as_ref()
|
||||||
|
.with_context(|| {
|
||||||
|
format!(
|
||||||
|
"No raw timeline {}/{} found",
|
||||||
|
self.owning_tenant.tenant_id, self.timeline_id
|
||||||
|
)
|
||||||
|
})?
|
||||||
|
.0)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
impl Drop for UninitializedTimeline<'_> {
|
||||||
|
fn drop(&mut self) {
|
||||||
|
if let Some((_, uninit_mark)) = self.raw_timeline.take() {
|
||||||
|
let _entered = info_span!("drop_uninitialized_timeline", tenant = %self.owning_tenant.tenant_id, timeline = %self.timeline_id).entered();
|
||||||
|
error!("Timeline got dropped without initializing, cleaning its files");
|
||||||
|
cleanup_timeline_directory(uninit_mark);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
pub(crate) fn cleanup_timeline_directory(uninit_mark: TimelineUninitMark) {
|
||||||
|
let timeline_path = &uninit_mark.timeline_path;
|
||||||
|
match ignore_absent_files(|| fs::remove_dir_all(timeline_path)) {
|
||||||
|
Ok(()) => {
|
||||||
|
info!("Timeline dir {timeline_path:?} removed successfully, removing the uninit mark")
|
||||||
|
}
|
||||||
|
Err(e) => {
|
||||||
|
error!("Failed to clean up uninitialized timeline directory {timeline_path:?}: {e:?}")
|
||||||
|
}
|
||||||
|
}
|
||||||
|
drop(uninit_mark); // mark handles its deletion on drop, gets retained if timeline dir exists
|
||||||
|
}
|
||||||
|
|
||||||
|
/// An uninit mark file, created along the timeline dir to ensure the timeline either gets fully initialized and loaded into pageserver's memory,
|
||||||
|
/// or gets removed eventually.
|
||||||
|
///
|
||||||
|
/// XXX: it's important to create it near the timeline dir, not inside it to ensure timeline dir gets removed first.
|
||||||
|
#[must_use]
|
||||||
|
pub(crate) struct TimelineUninitMark {
|
||||||
|
uninit_mark_deleted: bool,
|
||||||
|
uninit_mark_path: PathBuf,
|
||||||
|
pub(crate) timeline_path: PathBuf,
|
||||||
|
}
|
||||||
|
|
||||||
|
impl TimelineUninitMark {
|
||||||
|
pub(crate) fn new(uninit_mark_path: PathBuf, timeline_path: PathBuf) -> Self {
|
||||||
|
Self {
|
||||||
|
uninit_mark_deleted: false,
|
||||||
|
uninit_mark_path,
|
||||||
|
timeline_path,
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
fn remove_uninit_mark(mut self) -> anyhow::Result<()> {
|
||||||
|
if !self.uninit_mark_deleted {
|
||||||
|
self.delete_mark_file_if_present()?;
|
||||||
|
}
|
||||||
|
|
||||||
|
Ok(())
|
||||||
|
}
|
||||||
|
|
||||||
|
fn delete_mark_file_if_present(&mut self) -> anyhow::Result<()> {
|
||||||
|
let uninit_mark_file = &self.uninit_mark_path;
|
||||||
|
let uninit_mark_parent = uninit_mark_file
|
||||||
|
.parent()
|
||||||
|
.with_context(|| format!("Uninit mark file {uninit_mark_file:?} has no parent"))?;
|
||||||
|
ignore_absent_files(|| fs::remove_file(uninit_mark_file)).with_context(|| {
|
||||||
|
format!("Failed to remove uninit mark file at path {uninit_mark_file:?}")
|
||||||
|
})?;
|
||||||
|
crashsafe::fsync(uninit_mark_parent).context("Failed to fsync uninit mark parent")?;
|
||||||
|
self.uninit_mark_deleted = true;
|
||||||
|
|
||||||
|
Ok(())
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
impl Drop for TimelineUninitMark {
|
||||||
|
fn drop(&mut self) {
|
||||||
|
if !self.uninit_mark_deleted {
|
||||||
|
if self.timeline_path.exists() {
|
||||||
|
error!(
|
||||||
|
"Uninit mark {} is not removed, timeline {} stays uninitialized",
|
||||||
|
self.uninit_mark_path.display(),
|
||||||
|
self.timeline_path.display()
|
||||||
|
)
|
||||||
|
} else {
|
||||||
|
// unblock later timeline creation attempts
|
||||||
|
warn!(
|
||||||
|
"Removing intermediate uninit mark file {}",
|
||||||
|
self.uninit_mark_path.display()
|
||||||
|
);
|
||||||
|
if let Err(e) = self.delete_mark_file_if_present() {
|
||||||
|
error!("Failed to remove the uninit mark file: {e}")
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
@@ -71,6 +71,8 @@ pub(super) async fn handle_walreceiver_connection(
|
|||||||
ctx: RequestContext,
|
ctx: RequestContext,
|
||||||
node: NodeId,
|
node: NodeId,
|
||||||
) -> anyhow::Result<()> {
|
) -> anyhow::Result<()> {
|
||||||
|
debug_assert_current_span_has_tenant_and_timeline_id();
|
||||||
|
|
||||||
WALRECEIVER_STARTED_CONNECTIONS.inc();
|
WALRECEIVER_STARTED_CONNECTIONS.inc();
|
||||||
|
|
||||||
// Connect to the database in replication mode.
|
// Connect to the database in replication mode.
|
||||||
@@ -140,6 +142,9 @@ pub(super) async fn handle_walreceiver_connection(
|
|||||||
}
|
}
|
||||||
Ok(())
|
Ok(())
|
||||||
}
|
}
|
||||||
|
// Enrich the log lines emitted by this closure with meaningful context.
|
||||||
|
// TODO: technically, this task outlives the surrounding function, so, the
|
||||||
|
// spans won't be properly nested.
|
||||||
.instrument(tracing::info_span!("poller")),
|
.instrument(tracing::info_span!("poller")),
|
||||||
);
|
);
|
||||||
|
|
||||||
|
|||||||
@@ -302,15 +302,6 @@ impl VirtualFile {
|
|||||||
.observe_closure_duration(|| self.open_options.open(&self.path))?;
|
.observe_closure_duration(|| self.open_options.open(&self.path))?;
|
||||||
|
|
||||||
// Perform the requested operation on it
|
// Perform the requested operation on it
|
||||||
//
|
|
||||||
// TODO: We could downgrade the locks to read mode before calling
|
|
||||||
// 'func', to allow a little bit more concurrency, but the standard
|
|
||||||
// library RwLock doesn't allow downgrading without releasing the lock,
|
|
||||||
// and that doesn't seem worth the trouble.
|
|
||||||
//
|
|
||||||
// XXX: `parking_lot::RwLock` can enable such downgrades, yet its implementation is fair and
|
|
||||||
// may deadlock on subsequent read calls.
|
|
||||||
// Simply replacing all `RwLock` in project causes deadlocks, so use it sparingly.
|
|
||||||
let result = STORAGE_IO_TIME
|
let result = STORAGE_IO_TIME
|
||||||
.with_label_values(&[op, &self.tenant_id, &self.timeline_id])
|
.with_label_values(&[op, &self.tenant_id, &self.timeline_id])
|
||||||
.observe_closure_duration(|| func(&file));
|
.observe_closure_duration(|| func(&file));
|
||||||
|
|||||||
@@ -122,6 +122,43 @@ hnsw_populate(HierarchicalNSW* hnsw, Relation indexRel, Relation heapRel)
|
|||||||
true, true, hnsw_build_callback, (void *) hnsw, NULL);
|
true, true, hnsw_build_callback, (void *) hnsw, NULL);
|
||||||
}
|
}
|
||||||
|
|
||||||
|
#ifdef __APPLE__
|
||||||
|
|
||||||
|
#include <sys/types.h>
|
||||||
|
#include <sys/sysctl.h>
|
||||||
|
|
||||||
|
static void
|
||||||
|
hnsw_check_available_memory(Size requested)
|
||||||
|
{
|
||||||
|
size_t total;
|
||||||
|
if (sysctlbyname("hw.memsize", NULL, &total, NULL, 0) < 0)
|
||||||
|
elog(ERROR, "Failed to get amount of RAM: %m");
|
||||||
|
|
||||||
|
if ((Size)NBuffers*BLCKSZ + requested >= total)
|
||||||
|
elog(ERROR, "HNSW index requeries %ld bytes while only %ld are available",
|
||||||
|
requested, total - (Size)NBuffers*BLCKSZ);
|
||||||
|
}
|
||||||
|
|
||||||
|
#else
|
||||||
|
|
||||||
|
#include <sys/sysinfo.h>
|
||||||
|
|
||||||
|
static void
|
||||||
|
hnsw_check_available_memory(Size requested)
|
||||||
|
{
|
||||||
|
struct sysinfo si;
|
||||||
|
Size total;
|
||||||
|
if (sysinfo(&si) < 0)
|
||||||
|
elog(ERROR, "Failed to get amount of RAM: %m");
|
||||||
|
|
||||||
|
total = si.totalram*si.mem_unit;
|
||||||
|
if ((Size)NBuffers*BLCKSZ + requested >= total)
|
||||||
|
elog(ERROR, "HNSW index requeries %ld bytes while only %ld are available",
|
||||||
|
requested, total - (Size)NBuffers*BLCKSZ);
|
||||||
|
}
|
||||||
|
|
||||||
|
#endif
|
||||||
|
|
||||||
static HierarchicalNSW*
|
static HierarchicalNSW*
|
||||||
hnsw_get_index(Relation indexRel, Relation heapRel)
|
hnsw_get_index(Relation indexRel, Relation heapRel)
|
||||||
{
|
{
|
||||||
@@ -156,6 +193,8 @@ hnsw_get_index(Relation indexRel, Relation heapRel)
|
|||||||
size_data_per_element = size_links_level0 + data_size + sizeof(label_t);
|
size_data_per_element = size_links_level0 + data_size + sizeof(label_t);
|
||||||
shmem_size = hnsw_sizeof() + maxelements * size_data_per_element;
|
shmem_size = hnsw_sizeof() + maxelements * size_data_per_element;
|
||||||
|
|
||||||
|
hnsw_check_available_memory(shmem_size);
|
||||||
|
|
||||||
/* first try to attach to existed index */
|
/* first try to attach to existed index */
|
||||||
if (!dsm_impl_op(DSM_OP_ATTACH, handle, 0, &impl_private,
|
if (!dsm_impl_op(DSM_OP_ATTACH, handle, 0, &impl_private,
|
||||||
&mapped_address, &mapped_size, DEBUG1))
|
&mapped_address, &mapped_size, DEBUG1))
|
||||||
@@ -541,6 +580,7 @@ l2_distance(PG_FUNCTION_ARGS)
|
|||||||
errmsg("different array dimensions %d and %d", a_dim, b_dim)));
|
errmsg("different array dimensions %d and %d", a_dim, b_dim)));
|
||||||
}
|
}
|
||||||
|
|
||||||
|
#pragma clang loop vectorize(enable)
|
||||||
for (int i = 0; i < a_dim; i++)
|
for (int i = 0; i < a_dim; i++)
|
||||||
{
|
{
|
||||||
diff = ax[i] - bx[i];
|
diff = ax[i] - bx[i];
|
||||||
|
|||||||
@@ -1,4 +1,4 @@
|
|||||||
comment = 'hNsw index'
|
comment = 'hnsw index'
|
||||||
default_version = '0.1.0'
|
default_version = '0.1.0'
|
||||||
module_pathname = '$libdir/hnsw'
|
module_pathname = '$libdir/hnsw'
|
||||||
relocatable = true
|
relocatable = true
|
||||||
|
|||||||
@@ -223,6 +223,7 @@ dist_t fstdistfunc_scalar(const coord_t *x, const coord_t *y, size_t n)
|
|||||||
{
|
{
|
||||||
dist_t distance = 0.0;
|
dist_t distance = 0.0;
|
||||||
|
|
||||||
|
#pragma clang loop vectorize(enable)
|
||||||
for (size_t i = 0; i < n; i++)
|
for (size_t i = 0; i < n; i++)
|
||||||
{
|
{
|
||||||
dist_t diff = x[i] - y[i];
|
dist_t diff = x[i] - y[i];
|
||||||
|
|||||||
+62
-55
@@ -34,7 +34,6 @@
|
|||||||
|
|
||||||
#define PageStoreTrace DEBUG5
|
#define PageStoreTrace DEBUG5
|
||||||
|
|
||||||
#define MAX_RECONNECT_ATTEMPTS 5
|
|
||||||
#define RECONNECT_INTERVAL_USEC 1000000
|
#define RECONNECT_INTERVAL_USEC 1000000
|
||||||
|
|
||||||
bool connected = false;
|
bool connected = false;
|
||||||
@@ -55,13 +54,15 @@ int32 max_cluster_size;
|
|||||||
char *page_server_connstring;
|
char *page_server_connstring;
|
||||||
char *neon_auth_token;
|
char *neon_auth_token;
|
||||||
|
|
||||||
int n_unflushed_requests = 0;
|
|
||||||
int flush_every_n_requests = 8;
|
|
||||||
int readahead_buffer_size = 128;
|
int readahead_buffer_size = 128;
|
||||||
|
int flush_every_n_requests = 8;
|
||||||
|
|
||||||
|
int n_reconnect_attempts = 0;
|
||||||
|
int max_reconnect_attempts = 60;
|
||||||
|
|
||||||
bool (*old_redo_read_buffer_filter) (XLogReaderState *record, uint8 block_id) = NULL;
|
bool (*old_redo_read_buffer_filter) (XLogReaderState *record, uint8 block_id) = NULL;
|
||||||
|
|
||||||
static void pageserver_flush(void);
|
static bool pageserver_flush(void);
|
||||||
|
|
||||||
static bool
|
static bool
|
||||||
pageserver_connect(int elevel)
|
pageserver_connect(int elevel)
|
||||||
@@ -232,16 +233,17 @@ pageserver_disconnect(void)
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
static void
|
static bool
|
||||||
pageserver_send(NeonRequest * request)
|
pageserver_send(NeonRequest * request)
|
||||||
{
|
{
|
||||||
StringInfoData req_buff;
|
StringInfoData req_buff;
|
||||||
int n_reconnect_attempts = 0;
|
|
||||||
|
|
||||||
/* If the connection was lost for some reason, reconnect */
|
/* If the connection was lost for some reason, reconnect */
|
||||||
if (connected && PQstatus(pageserver_conn) == CONNECTION_BAD)
|
if (connected && PQstatus(pageserver_conn) == CONNECTION_BAD)
|
||||||
|
{
|
||||||
|
neon_log(LOG, "pageserver_send disconnect bad connection");
|
||||||
pageserver_disconnect();
|
pageserver_disconnect();
|
||||||
|
}
|
||||||
|
|
||||||
req_buff = nm_pack_request(request);
|
req_buff = nm_pack_request(request);
|
||||||
|
|
||||||
@@ -252,53 +254,36 @@ pageserver_send(NeonRequest * request)
|
|||||||
* See https://github.com/neondatabase/neon/issues/1138
|
* See https://github.com/neondatabase/neon/issues/1138
|
||||||
* So try to reestablish connection in case of failure.
|
* So try to reestablish connection in case of failure.
|
||||||
*/
|
*/
|
||||||
while (true)
|
if (!connected)
|
||||||
{
|
{
|
||||||
if (!connected)
|
while (!pageserver_connect(n_reconnect_attempts < max_reconnect_attempts ? LOG : ERROR))
|
||||||
{
|
{
|
||||||
if (!pageserver_connect(n_reconnect_attempts < MAX_RECONNECT_ATTEMPTS ? LOG : ERROR))
|
n_reconnect_attempts += 1;
|
||||||
{
|
pg_usleep(RECONNECT_INTERVAL_USEC);
|
||||||
n_reconnect_attempts += 1;
|
|
||||||
pg_usleep(RECONNECT_INTERVAL_USEC);
|
|
||||||
continue;
|
|
||||||
}
|
|
||||||
}
|
}
|
||||||
|
n_reconnect_attempts = 0;
|
||||||
|
}
|
||||||
|
|
||||||
/*
|
/*
|
||||||
* Send request.
|
* Send request.
|
||||||
*
|
*
|
||||||
* In principle, this could block if the output buffer is full, and we
|
* In principle, this could block if the output buffer is full, and we
|
||||||
* should use async mode and check for interrupts while waiting. In
|
* should use async mode and check for interrupts while waiting. In
|
||||||
* practice, our requests are small enough to always fit in the output and
|
* practice, our requests are small enough to always fit in the output and
|
||||||
* TCP buffer.
|
* TCP buffer.
|
||||||
*/
|
*/
|
||||||
if (PQputCopyData(pageserver_conn, req_buff.data, req_buff.len) <= 0)
|
if (PQputCopyData(pageserver_conn, req_buff.data, req_buff.len) <= 0)
|
||||||
{
|
{
|
||||||
char *msg = pchomp(PQerrorMessage(pageserver_conn));
|
char *msg = pchomp(PQerrorMessage(pageserver_conn));
|
||||||
if (n_reconnect_attempts < MAX_RECONNECT_ATTEMPTS)
|
pageserver_disconnect();
|
||||||
{
|
neon_log(LOG, "pageserver_send disconnect because failed to send page request (try to reconnect): %s", msg);
|
||||||
neon_log(LOG, "failed to send page request (try to reconnect): %s", msg);
|
pfree(msg);
|
||||||
if (n_reconnect_attempts != 0) /* do not sleep before first reconnect attempt, assuming that pageserver is already restarted */
|
pfree(req_buff.data);
|
||||||
pg_usleep(RECONNECT_INTERVAL_USEC);
|
return false;
|
||||||
n_reconnect_attempts += 1;
|
|
||||||
continue;
|
|
||||||
}
|
|
||||||
else
|
|
||||||
{
|
|
||||||
pageserver_disconnect();
|
|
||||||
neon_log(ERROR, "failed to send page request: %s", msg);
|
|
||||||
}
|
|
||||||
}
|
|
||||||
break;
|
|
||||||
}
|
}
|
||||||
|
|
||||||
pfree(req_buff.data);
|
pfree(req_buff.data);
|
||||||
|
|
||||||
n_unflushed_requests++;
|
|
||||||
|
|
||||||
if (flush_every_n_requests > 0 && n_unflushed_requests >= flush_every_n_requests)
|
|
||||||
pageserver_flush();
|
|
||||||
|
|
||||||
if (message_level_is_interesting(PageStoreTrace))
|
if (message_level_is_interesting(PageStoreTrace))
|
||||||
{
|
{
|
||||||
char *msg = nm_to_string((NeonMessage *) request);
|
char *msg = nm_to_string((NeonMessage *) request);
|
||||||
@@ -306,6 +291,7 @@ pageserver_send(NeonRequest * request)
|
|||||||
neon_log(PageStoreTrace, "sent request: %s", msg);
|
neon_log(PageStoreTrace, "sent request: %s", msg);
|
||||||
pfree(msg);
|
pfree(msg);
|
||||||
}
|
}
|
||||||
|
return true;
|
||||||
}
|
}
|
||||||
|
|
||||||
static NeonResponse *
|
static NeonResponse *
|
||||||
@@ -340,16 +326,25 @@ pageserver_receive(void)
|
|||||||
}
|
}
|
||||||
else if (rc == -1)
|
else if (rc == -1)
|
||||||
{
|
{
|
||||||
|
neon_log(LOG, "pageserver_receive disconnect because call_PQgetCopyData returns -1: %s", pchomp(PQerrorMessage(pageserver_conn)));
|
||||||
pageserver_disconnect();
|
pageserver_disconnect();
|
||||||
resp = NULL;
|
resp = NULL;
|
||||||
}
|
}
|
||||||
else if (rc == -2)
|
else if (rc == -2)
|
||||||
neon_log(ERROR, "could not read COPY data: %s", pchomp(PQerrorMessage(pageserver_conn)));
|
{
|
||||||
|
char* msg = pchomp(PQerrorMessage(pageserver_conn));
|
||||||
|
pageserver_disconnect();
|
||||||
|
neon_log(ERROR, "pageserver_receive disconnect because could not read COPY data: %s", msg);
|
||||||
|
}
|
||||||
else
|
else
|
||||||
neon_log(ERROR, "unexpected PQgetCopyData return value: %d", rc);
|
{
|
||||||
|
pageserver_disconnect();
|
||||||
|
neon_log(ERROR, "pageserver_receive disconnect because unexpected PQgetCopyData return value: %d", rc);
|
||||||
|
}
|
||||||
}
|
}
|
||||||
PG_CATCH();
|
PG_CATCH();
|
||||||
{
|
{
|
||||||
|
neon_log(LOG, "pageserver_receive disconnect due to caught exception");
|
||||||
pageserver_disconnect();
|
pageserver_disconnect();
|
||||||
PG_RE_THROW();
|
PG_RE_THROW();
|
||||||
}
|
}
|
||||||
@@ -359,21 +354,25 @@ pageserver_receive(void)
|
|||||||
}
|
}
|
||||||
|
|
||||||
|
|
||||||
static void
|
static bool
|
||||||
pageserver_flush(void)
|
pageserver_flush(void)
|
||||||
{
|
{
|
||||||
if (!connected)
|
if (!connected)
|
||||||
{
|
{
|
||||||
neon_log(WARNING, "Tried to flush while disconnected");
|
neon_log(WARNING, "Tried to flush while disconnected");
|
||||||
}
|
}
|
||||||
else if (PQflush(pageserver_conn))
|
else
|
||||||
{
|
{
|
||||||
char *msg = pchomp(PQerrorMessage(pageserver_conn));
|
if (PQflush(pageserver_conn))
|
||||||
|
{
|
||||||
pageserver_disconnect();
|
char *msg = pchomp(PQerrorMessage(pageserver_conn));
|
||||||
neon_log(ERROR, "failed to flush page requests: %s", msg);
|
pageserver_disconnect();
|
||||||
|
neon_log(LOG, "pageserver_flush disconnect because failed to flush page requests: %s", msg);
|
||||||
|
pfree(msg);
|
||||||
|
return false;
|
||||||
|
}
|
||||||
}
|
}
|
||||||
n_unflushed_requests = 0;
|
return true;
|
||||||
}
|
}
|
||||||
|
|
||||||
page_server_api api = {
|
page_server_api api = {
|
||||||
@@ -439,6 +438,14 @@ pg_init_libpagestore(void)
|
|||||||
PGC_USERSET,
|
PGC_USERSET,
|
||||||
0, /* no flags required */
|
0, /* no flags required */
|
||||||
NULL, NULL, NULL);
|
NULL, NULL, NULL);
|
||||||
|
DefineCustomIntVariable("neon.max_reconnect_attempts",
|
||||||
|
"Maximal attempts to reconnect to pages server (with 1 second timeout)",
|
||||||
|
NULL,
|
||||||
|
&max_reconnect_attempts,
|
||||||
|
10, 0, INT_MAX,
|
||||||
|
PGC_USERSET,
|
||||||
|
0,
|
||||||
|
NULL, NULL, NULL);
|
||||||
DefineCustomIntVariable("neon.readahead_buffer_size",
|
DefineCustomIntVariable("neon.readahead_buffer_size",
|
||||||
"number of prefetches to buffer",
|
"number of prefetches to buffer",
|
||||||
"This buffer is used to hold and manage prefetched "
|
"This buffer is used to hold and manage prefetched "
|
||||||
|
|||||||
@@ -145,9 +145,9 @@ extern char *nm_to_string(NeonMessage * msg);
|
|||||||
|
|
||||||
typedef struct
|
typedef struct
|
||||||
{
|
{
|
||||||
void (*send) (NeonRequest * request);
|
bool (*send) (NeonRequest * request);
|
||||||
NeonResponse *(*receive) (void);
|
NeonResponse *(*receive) (void);
|
||||||
void (*flush) (void);
|
bool (*flush) (void);
|
||||||
} page_server_api;
|
} page_server_api;
|
||||||
|
|
||||||
extern void prefetch_on_ps_disconnect(void);
|
extern void prefetch_on_ps_disconnect(void);
|
||||||
|
|||||||
+24
-18
@@ -489,7 +489,8 @@ prefetch_wait_for(uint64 ring_index)
|
|||||||
if (MyPState->ring_flush <= ring_index &&
|
if (MyPState->ring_flush <= ring_index &&
|
||||||
MyPState->ring_unused > MyPState->ring_flush)
|
MyPState->ring_unused > MyPState->ring_flush)
|
||||||
{
|
{
|
||||||
page_server->flush();
|
if (!page_server->flush())
|
||||||
|
return false;
|
||||||
MyPState->ring_flush = MyPState->ring_unused;
|
MyPState->ring_flush = MyPState->ring_unused;
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -666,7 +667,7 @@ prefetch_do_request(PrefetchRequest *slot, bool *force_latest, XLogRecPtr *force
|
|||||||
* smaller than the current WAL insert/redo pointer, which is already
|
* smaller than the current WAL insert/redo pointer, which is already
|
||||||
* larger than this prefetch_lsn. So in any case, that would
|
* larger than this prefetch_lsn. So in any case, that would
|
||||||
* invalidate this cache.
|
* invalidate this cache.
|
||||||
*
|
*
|
||||||
* The best LSN to use for effective_request_lsn would be
|
* The best LSN to use for effective_request_lsn would be
|
||||||
* XLogCtl->Insert.RedoRecPtr, but that's expensive to access.
|
* XLogCtl->Insert.RedoRecPtr, but that's expensive to access.
|
||||||
*/
|
*/
|
||||||
@@ -677,7 +678,8 @@ prefetch_do_request(PrefetchRequest *slot, bool *force_latest, XLogRecPtr *force
|
|||||||
|
|
||||||
Assert(slot->response == NULL);
|
Assert(slot->response == NULL);
|
||||||
Assert(slot->my_ring_index == MyPState->ring_unused);
|
Assert(slot->my_ring_index == MyPState->ring_unused);
|
||||||
page_server->send((NeonRequest *) &request);
|
|
||||||
|
while (!page_server->send((NeonRequest *) &request));
|
||||||
|
|
||||||
/* update prefetch state */
|
/* update prefetch state */
|
||||||
MyPState->n_requests_inflight += 1;
|
MyPState->n_requests_inflight += 1;
|
||||||
@@ -687,6 +689,7 @@ prefetch_do_request(PrefetchRequest *slot, bool *force_latest, XLogRecPtr *force
|
|||||||
/* update slot state */
|
/* update slot state */
|
||||||
slot->status = PRFS_REQUESTED;
|
slot->status = PRFS_REQUESTED;
|
||||||
|
|
||||||
|
|
||||||
prfh_insert(MyPState->prf_hash, slot, &found);
|
prfh_insert(MyPState->prf_hash, slot, &found);
|
||||||
Assert(!found);
|
Assert(!found);
|
||||||
}
|
}
|
||||||
@@ -743,6 +746,7 @@ prefetch_register_buffer(BufferTag tag, bool *force_latest, XLogRecPtr *force_ls
|
|||||||
prefetch_set_unused(ring_index);
|
prefetch_set_unused(ring_index);
|
||||||
entry = NULL;
|
entry = NULL;
|
||||||
}
|
}
|
||||||
|
|
||||||
}
|
}
|
||||||
/* if we don't want the latest version, only accept requests with the exact same LSN */
|
/* if we don't want the latest version, only accept requests with the exact same LSN */
|
||||||
else
|
else
|
||||||
@@ -756,20 +760,23 @@ prefetch_register_buffer(BufferTag tag, bool *force_latest, XLogRecPtr *force_ls
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
/*
|
if (entry != NULL)
|
||||||
* We received a prefetch for a page that was recently read and
|
|
||||||
* removed from the buffers. Remove that request from the buffers.
|
|
||||||
*/
|
|
||||||
else if (slot->status == PRFS_TAG_REMAINS)
|
|
||||||
{
|
{
|
||||||
prefetch_set_unused(ring_index);
|
/*
|
||||||
entry = NULL;
|
* We received a prefetch for a page that was recently read and
|
||||||
}
|
* removed from the buffers. Remove that request from the buffers.
|
||||||
else
|
*/
|
||||||
{
|
if (slot->status == PRFS_TAG_REMAINS)
|
||||||
/* The buffered request is good enough, return that index */
|
{
|
||||||
pgBufferUsage.prefetch.duplicates++;
|
prefetch_set_unused(ring_index);
|
||||||
return ring_index;
|
entry = NULL;
|
||||||
|
}
|
||||||
|
else
|
||||||
|
{
|
||||||
|
/* The buffered request is good enough, return that index */
|
||||||
|
pgBufferUsage.prefetch.duplicates++;
|
||||||
|
return ring_index;
|
||||||
|
}
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -859,8 +866,7 @@ page_server_request(void const *req)
|
|||||||
{
|
{
|
||||||
NeonResponse* resp;
|
NeonResponse* resp;
|
||||||
do {
|
do {
|
||||||
page_server->send((NeonRequest *) req);
|
while (!page_server->send((NeonRequest *) req) || !page_server->flush());
|
||||||
page_server->flush();
|
|
||||||
MyPState->ring_flush = MyPState->ring_unused;
|
MyPState->ring_flush = MyPState->ring_unused;
|
||||||
consume_prefetch_responses();
|
consume_prefetch_responses();
|
||||||
resp = page_server->receive();
|
resp = page_server->receive();
|
||||||
|
|||||||
@@ -29,6 +29,7 @@ metrics.workspace = true
|
|||||||
once_cell.workspace = true
|
once_cell.workspace = true
|
||||||
opentelemetry.workspace = true
|
opentelemetry.workspace = true
|
||||||
parking_lot.workspace = true
|
parking_lot.workspace = true
|
||||||
|
pbkdf2.workspace = true
|
||||||
pin-project-lite.workspace = true
|
pin-project-lite.workspace = true
|
||||||
postgres_backend.workspace = true
|
postgres_backend.workspace = true
|
||||||
pq_proto.workspace = true
|
pq_proto.workspace = true
|
||||||
|
|||||||
@@ -136,18 +136,17 @@ impl Default for ConnCfg {
|
|||||||
|
|
||||||
impl ConnCfg {
|
impl ConnCfg {
|
||||||
/// Establish a raw TCP connection to the compute node.
|
/// Establish a raw TCP connection to the compute node.
|
||||||
async fn connect_raw(&self) -> io::Result<(SocketAddr, TcpStream, &str)> {
|
async fn connect_raw(&self, timeout: Duration) -> io::Result<(SocketAddr, TcpStream, &str)> {
|
||||||
use tokio_postgres::config::Host;
|
use tokio_postgres::config::Host;
|
||||||
|
|
||||||
// wrap TcpStream::connect with timeout
|
// wrap TcpStream::connect with timeout
|
||||||
let connect_with_timeout = |host, port| {
|
let connect_with_timeout = |host, port| {
|
||||||
let connection_timeout = Duration::from_millis(10000);
|
tokio::time::timeout(timeout, TcpStream::connect((host, port))).map(
|
||||||
tokio::time::timeout(connection_timeout, TcpStream::connect((host, port))).map(
|
|
||||||
move |res| match res {
|
move |res| match res {
|
||||||
Ok(tcpstream_connect_res) => tcpstream_connect_res,
|
Ok(tcpstream_connect_res) => tcpstream_connect_res,
|
||||||
Err(_) => Err(io::Error::new(
|
Err(_) => Err(io::Error::new(
|
||||||
io::ErrorKind::TimedOut,
|
io::ErrorKind::TimedOut,
|
||||||
format!("exceeded connection timeout {connection_timeout:?}"),
|
format!("exceeded connection timeout {timeout:?}"),
|
||||||
)),
|
)),
|
||||||
},
|
},
|
||||||
)
|
)
|
||||||
@@ -223,8 +222,9 @@ impl ConnCfg {
|
|||||||
async fn do_connect(
|
async fn do_connect(
|
||||||
&self,
|
&self,
|
||||||
allow_self_signed_compute: bool,
|
allow_self_signed_compute: bool,
|
||||||
|
timeout: Duration,
|
||||||
) -> Result<PostgresConnection, ConnectionError> {
|
) -> Result<PostgresConnection, ConnectionError> {
|
||||||
let (socket_addr, stream, host) = self.connect_raw().await?;
|
let (socket_addr, stream, host) = self.connect_raw(timeout).await?;
|
||||||
|
|
||||||
let tls_connector = native_tls::TlsConnector::builder()
|
let tls_connector = native_tls::TlsConnector::builder()
|
||||||
.danger_accept_invalid_certs(allow_self_signed_compute)
|
.danger_accept_invalid_certs(allow_self_signed_compute)
|
||||||
@@ -264,8 +264,9 @@ impl ConnCfg {
|
|||||||
pub async fn connect(
|
pub async fn connect(
|
||||||
&self,
|
&self,
|
||||||
allow_self_signed_compute: bool,
|
allow_self_signed_compute: bool,
|
||||||
|
timeout: Duration,
|
||||||
) -> Result<PostgresConnection, ConnectionError> {
|
) -> Result<PostgresConnection, ConnectionError> {
|
||||||
self.do_connect(allow_self_signed_compute)
|
self.do_connect(allow_self_signed_compute, timeout)
|
||||||
.inspect_err(|err| {
|
.inspect_err(|err| {
|
||||||
// Immediately log the error we have at our disposal.
|
// Immediately log the error we have at our disposal.
|
||||||
error!("couldn't connect to compute node: {err}");
|
error!("couldn't connect to compute node: {err}");
|
||||||
|
|||||||
+1
-1
@@ -212,7 +212,7 @@ pub struct CacheOptions {
|
|||||||
|
|
||||||
impl CacheOptions {
|
impl CacheOptions {
|
||||||
/// Default options for [`crate::auth::caches::NodeInfoCache`].
|
/// Default options for [`crate::auth::caches::NodeInfoCache`].
|
||||||
pub const DEFAULT_OPTIONS_NODE_INFO: &str = "size=4000,ttl=5m";
|
pub const DEFAULT_OPTIONS_NODE_INFO: &str = "size=4000,ttl=4m";
|
||||||
|
|
||||||
/// Parse cache options passed via cmdline.
|
/// Parse cache options passed via cmdline.
|
||||||
/// Example: [`Self::DEFAULT_OPTIONS_NODE_INFO`].
|
/// Example: [`Self::DEFAULT_OPTIONS_NODE_INFO`].
|
||||||
|
|||||||
@@ -2,6 +2,7 @@
|
|||||||
//! Other modules should use stuff from this module instead of
|
//! Other modules should use stuff from this module instead of
|
||||||
//! directly relying on deps like `reqwest` (think loose coupling).
|
//! directly relying on deps like `reqwest` (think loose coupling).
|
||||||
|
|
||||||
|
pub mod conn_pool;
|
||||||
pub mod server;
|
pub mod server;
|
||||||
pub mod sql_over_http;
|
pub mod sql_over_http;
|
||||||
pub mod websocket;
|
pub mod websocket;
|
||||||
|
|||||||
@@ -0,0 +1,278 @@
|
|||||||
|
use parking_lot::Mutex;
|
||||||
|
use pq_proto::StartupMessageParams;
|
||||||
|
use std::fmt;
|
||||||
|
use std::{collections::HashMap, sync::Arc};
|
||||||
|
|
||||||
|
use futures::TryFutureExt;
|
||||||
|
|
||||||
|
use crate::config;
|
||||||
|
use crate::{auth, console};
|
||||||
|
|
||||||
|
use super::sql_over_http::MAX_RESPONSE_SIZE;
|
||||||
|
|
||||||
|
use crate::proxy::invalidate_cache;
|
||||||
|
use crate::proxy::NUM_RETRIES_WAKE_COMPUTE;
|
||||||
|
|
||||||
|
use tracing::error;
|
||||||
|
use tracing::info;
|
||||||
|
|
||||||
|
pub const APP_NAME: &str = "sql_over_http";
|
||||||
|
const MAX_CONNS_PER_ENDPOINT: usize = 20;
|
||||||
|
|
||||||
|
#[derive(Debug)]
|
||||||
|
pub struct ConnInfo {
|
||||||
|
pub username: String,
|
||||||
|
pub dbname: String,
|
||||||
|
pub hostname: String,
|
||||||
|
pub password: String,
|
||||||
|
}
|
||||||
|
|
||||||
|
impl ConnInfo {
|
||||||
|
// hm, change to hasher to avoid cloning?
|
||||||
|
pub fn db_and_user(&self) -> (String, String) {
|
||||||
|
(self.dbname.clone(), self.username.clone())
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
impl fmt::Display for ConnInfo {
|
||||||
|
// use custom display to avoid logging password
|
||||||
|
fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result {
|
||||||
|
write!(f, "{}@{}/{}", self.username, self.hostname, self.dbname)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
struct ConnPoolEntry {
|
||||||
|
conn: tokio_postgres::Client,
|
||||||
|
_last_access: std::time::Instant,
|
||||||
|
}
|
||||||
|
|
||||||
|
// Per-endpoint connection pool, (dbname, username) -> Vec<ConnPoolEntry>
|
||||||
|
// Number of open connections is limited by the `max_conns_per_endpoint`.
|
||||||
|
pub struct EndpointConnPool {
|
||||||
|
pools: HashMap<(String, String), Vec<ConnPoolEntry>>,
|
||||||
|
total_conns: usize,
|
||||||
|
}
|
||||||
|
|
||||||
|
pub struct GlobalConnPool {
|
||||||
|
// endpoint -> per-endpoint connection pool
|
||||||
|
//
|
||||||
|
// That should be a fairly conteded map, so return reference to the per-endpoint
|
||||||
|
// pool as early as possible and release the lock.
|
||||||
|
global_pool: Mutex<HashMap<String, Arc<Mutex<EndpointConnPool>>>>,
|
||||||
|
|
||||||
|
// Maximum number of connections per one endpoint.
|
||||||
|
// Can mix different (dbname, username) connections.
|
||||||
|
// When running out of free slots for a particular endpoint,
|
||||||
|
// falls back to opening a new connection for each request.
|
||||||
|
max_conns_per_endpoint: usize,
|
||||||
|
|
||||||
|
proxy_config: &'static crate::config::ProxyConfig,
|
||||||
|
}
|
||||||
|
|
||||||
|
impl GlobalConnPool {
|
||||||
|
pub fn new(config: &'static crate::config::ProxyConfig) -> Arc<Self> {
|
||||||
|
Arc::new(Self {
|
||||||
|
global_pool: Mutex::new(HashMap::new()),
|
||||||
|
max_conns_per_endpoint: MAX_CONNS_PER_ENDPOINT,
|
||||||
|
proxy_config: config,
|
||||||
|
})
|
||||||
|
}
|
||||||
|
|
||||||
|
pub async fn get(
|
||||||
|
&self,
|
||||||
|
conn_info: &ConnInfo,
|
||||||
|
force_new: bool,
|
||||||
|
) -> anyhow::Result<tokio_postgres::Client> {
|
||||||
|
let mut client: Option<tokio_postgres::Client> = None;
|
||||||
|
|
||||||
|
if !force_new {
|
||||||
|
let pool = self.get_endpoint_pool(&conn_info.hostname).await;
|
||||||
|
|
||||||
|
// find a pool entry by (dbname, username) if exists
|
||||||
|
let mut pool = pool.lock();
|
||||||
|
let pool_entries = pool.pools.get_mut(&conn_info.db_and_user());
|
||||||
|
if let Some(pool_entries) = pool_entries {
|
||||||
|
if let Some(entry) = pool_entries.pop() {
|
||||||
|
client = Some(entry.conn);
|
||||||
|
pool.total_conns -= 1;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// ok return cached connection if found and establish a new one otherwise
|
||||||
|
if let Some(client) = client {
|
||||||
|
if client.is_closed() {
|
||||||
|
info!("pool: cached connection '{conn_info}' is closed, opening a new one");
|
||||||
|
connect_to_compute(self.proxy_config, conn_info).await
|
||||||
|
} else {
|
||||||
|
info!("pool: reusing connection '{conn_info}'");
|
||||||
|
Ok(client)
|
||||||
|
}
|
||||||
|
} else {
|
||||||
|
info!("pool: opening a new connection '{conn_info}'");
|
||||||
|
connect_to_compute(self.proxy_config, conn_info).await
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
pub async fn put(
|
||||||
|
&self,
|
||||||
|
conn_info: &ConnInfo,
|
||||||
|
client: tokio_postgres::Client,
|
||||||
|
) -> anyhow::Result<()> {
|
||||||
|
let pool = self.get_endpoint_pool(&conn_info.hostname).await;
|
||||||
|
|
||||||
|
// return connection to the pool
|
||||||
|
let mut total_conns;
|
||||||
|
let mut returned = false;
|
||||||
|
let mut per_db_size = 0;
|
||||||
|
{
|
||||||
|
let mut pool = pool.lock();
|
||||||
|
total_conns = pool.total_conns;
|
||||||
|
|
||||||
|
let pool_entries: &mut Vec<ConnPoolEntry> = pool
|
||||||
|
.pools
|
||||||
|
.entry(conn_info.db_and_user())
|
||||||
|
.or_insert_with(|| Vec::with_capacity(1));
|
||||||
|
if total_conns < self.max_conns_per_endpoint {
|
||||||
|
pool_entries.push(ConnPoolEntry {
|
||||||
|
conn: client,
|
||||||
|
_last_access: std::time::Instant::now(),
|
||||||
|
});
|
||||||
|
|
||||||
|
total_conns += 1;
|
||||||
|
returned = true;
|
||||||
|
per_db_size = pool_entries.len();
|
||||||
|
|
||||||
|
pool.total_conns += 1;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// do logging outside of the mutex
|
||||||
|
if returned {
|
||||||
|
info!("pool: returning connection '{conn_info}' back to the pool, total_conns={total_conns}, for this (db, user)={per_db_size}");
|
||||||
|
} else {
|
||||||
|
info!("pool: throwing away connection '{conn_info}' because pool is full, total_conns={total_conns}");
|
||||||
|
}
|
||||||
|
|
||||||
|
Ok(())
|
||||||
|
}
|
||||||
|
|
||||||
|
async fn get_endpoint_pool(&self, endpoint: &String) -> Arc<Mutex<EndpointConnPool>> {
|
||||||
|
// find or create a pool for this endpoint
|
||||||
|
let mut created = false;
|
||||||
|
let mut global_pool = self.global_pool.lock();
|
||||||
|
let pool = global_pool
|
||||||
|
.entry(endpoint.clone())
|
||||||
|
.or_insert_with(|| {
|
||||||
|
created = true;
|
||||||
|
Arc::new(Mutex::new(EndpointConnPool {
|
||||||
|
pools: HashMap::new(),
|
||||||
|
total_conns: 0,
|
||||||
|
}))
|
||||||
|
})
|
||||||
|
.clone();
|
||||||
|
let global_pool_size = global_pool.len();
|
||||||
|
drop(global_pool);
|
||||||
|
|
||||||
|
// log new global pool size
|
||||||
|
if created {
|
||||||
|
info!(
|
||||||
|
"pool: created new pool for '{endpoint}', global pool size now {global_pool_size}"
|
||||||
|
);
|
||||||
|
}
|
||||||
|
|
||||||
|
pool
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
//
|
||||||
|
// Wake up the destination if needed. Code here is a bit involved because
|
||||||
|
// we reuse the code from the usual proxy and we need to prepare few structures
|
||||||
|
// that this code expects.
|
||||||
|
//
|
||||||
|
async fn connect_to_compute(
|
||||||
|
config: &config::ProxyConfig,
|
||||||
|
conn_info: &ConnInfo,
|
||||||
|
) -> anyhow::Result<tokio_postgres::Client> {
|
||||||
|
let tls = config.tls_config.as_ref();
|
||||||
|
let common_names = tls.and_then(|tls| tls.common_names.clone());
|
||||||
|
|
||||||
|
let credential_params = StartupMessageParams::new([
|
||||||
|
("user", &conn_info.username),
|
||||||
|
("database", &conn_info.dbname),
|
||||||
|
("application_name", APP_NAME),
|
||||||
|
]);
|
||||||
|
|
||||||
|
let creds = config
|
||||||
|
.auth_backend
|
||||||
|
.as_ref()
|
||||||
|
.map(|_| {
|
||||||
|
auth::ClientCredentials::parse(
|
||||||
|
&credential_params,
|
||||||
|
Some(&conn_info.hostname),
|
||||||
|
common_names,
|
||||||
|
)
|
||||||
|
})
|
||||||
|
.transpose()?;
|
||||||
|
let extra = console::ConsoleReqExtra {
|
||||||
|
session_id: uuid::Uuid::new_v4(),
|
||||||
|
application_name: Some(APP_NAME),
|
||||||
|
};
|
||||||
|
|
||||||
|
let node_info = &mut creds.wake_compute(&extra).await?.expect("msg");
|
||||||
|
|
||||||
|
// This code is a copy of `connect_to_compute` from `src/proxy.rs` with
|
||||||
|
// the difference that it uses `tokio_postgres` for the connection.
|
||||||
|
let mut num_retries: usize = NUM_RETRIES_WAKE_COMPUTE;
|
||||||
|
loop {
|
||||||
|
match connect_to_compute_once(node_info, conn_info).await {
|
||||||
|
Err(e) if num_retries > 0 => {
|
||||||
|
info!("compute node's state has changed; requesting a wake-up");
|
||||||
|
match creds.wake_compute(&extra).await? {
|
||||||
|
// Update `node_info` and try one more time.
|
||||||
|
Some(new) => {
|
||||||
|
*node_info = new;
|
||||||
|
}
|
||||||
|
// Link auth doesn't work that way, so we just exit.
|
||||||
|
None => return Err(e),
|
||||||
|
}
|
||||||
|
}
|
||||||
|
other => return other,
|
||||||
|
}
|
||||||
|
|
||||||
|
num_retries -= 1;
|
||||||
|
info!("retrying after wake-up ({num_retries} attempts left)");
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
async fn connect_to_compute_once(
|
||||||
|
node_info: &console::CachedNodeInfo,
|
||||||
|
conn_info: &ConnInfo,
|
||||||
|
) -> anyhow::Result<tokio_postgres::Client> {
|
||||||
|
let mut config = (*node_info.config).clone();
|
||||||
|
|
||||||
|
let (client, connection) = config
|
||||||
|
.user(&conn_info.username)
|
||||||
|
.password(&conn_info.password)
|
||||||
|
.dbname(&conn_info.dbname)
|
||||||
|
.max_backend_message_size(MAX_RESPONSE_SIZE)
|
||||||
|
.connect(tokio_postgres::NoTls)
|
||||||
|
.inspect_err(|e: &tokio_postgres::Error| {
|
||||||
|
error!(
|
||||||
|
"failed to connect to compute node hosts={:?} ports={:?}: {}",
|
||||||
|
node_info.config.get_hosts(),
|
||||||
|
node_info.config.get_ports(),
|
||||||
|
e
|
||||||
|
);
|
||||||
|
invalidate_cache(node_info)
|
||||||
|
})
|
||||||
|
.await?;
|
||||||
|
|
||||||
|
tokio::spawn(async move {
|
||||||
|
if let Err(e) = connection.await {
|
||||||
|
error!("connection error: {}", e);
|
||||||
|
}
|
||||||
|
});
|
||||||
|
|
||||||
|
Ok(client)
|
||||||
|
}
|
||||||
+19
-112
@@ -1,25 +1,21 @@
|
|||||||
|
use std::sync::Arc;
|
||||||
|
|
||||||
use futures::pin_mut;
|
use futures::pin_mut;
|
||||||
use futures::StreamExt;
|
use futures::StreamExt;
|
||||||
use futures::TryFutureExt;
|
|
||||||
use hyper::body::HttpBody;
|
use hyper::body::HttpBody;
|
||||||
use hyper::http::HeaderName;
|
use hyper::http::HeaderName;
|
||||||
use hyper::http::HeaderValue;
|
use hyper::http::HeaderValue;
|
||||||
use hyper::{Body, HeaderMap, Request};
|
use hyper::{Body, HeaderMap, Request};
|
||||||
use pq_proto::StartupMessageParams;
|
|
||||||
use serde_json::json;
|
use serde_json::json;
|
||||||
use serde_json::Map;
|
use serde_json::Map;
|
||||||
use serde_json::Value;
|
use serde_json::Value;
|
||||||
use tokio_postgres::types::Kind;
|
use tokio_postgres::types::Kind;
|
||||||
use tokio_postgres::types::Type;
|
use tokio_postgres::types::Type;
|
||||||
use tokio_postgres::Row;
|
use tokio_postgres::Row;
|
||||||
use tracing::error;
|
|
||||||
use tracing::info;
|
|
||||||
use tracing::instrument;
|
|
||||||
use url::Url;
|
use url::Url;
|
||||||
|
|
||||||
use crate::proxy::invalidate_cache;
|
use super::conn_pool::ConnInfo;
|
||||||
use crate::proxy::NUM_RETRIES_WAKE_COMPUTE;
|
use super::conn_pool::GlobalConnPool;
|
||||||
use crate::{auth, config::ProxyConfig, console};
|
|
||||||
|
|
||||||
#[derive(serde::Deserialize)]
|
#[derive(serde::Deserialize)]
|
||||||
struct QueryData {
|
struct QueryData {
|
||||||
@@ -27,12 +23,13 @@ struct QueryData {
|
|||||||
params: Vec<serde_json::Value>,
|
params: Vec<serde_json::Value>,
|
||||||
}
|
}
|
||||||
|
|
||||||
const APP_NAME: &str = "sql_over_http";
|
pub const MAX_RESPONSE_SIZE: usize = 1024 * 1024; // 1 MB
|
||||||
const MAX_RESPONSE_SIZE: usize = 1024 * 1024; // 1 MB
|
|
||||||
const MAX_REQUEST_SIZE: u64 = 1024 * 1024; // 1 MB
|
const MAX_REQUEST_SIZE: u64 = 1024 * 1024; // 1 MB
|
||||||
|
|
||||||
static RAW_TEXT_OUTPUT: HeaderName = HeaderName::from_static("neon-raw-text-output");
|
static RAW_TEXT_OUTPUT: HeaderName = HeaderName::from_static("neon-raw-text-output");
|
||||||
static ARRAY_MODE: HeaderName = HeaderName::from_static("neon-array-mode");
|
static ARRAY_MODE: HeaderName = HeaderName::from_static("neon-array-mode");
|
||||||
|
static ALLOW_POOL: HeaderName = HeaderName::from_static("neon-pool-opt-in");
|
||||||
|
|
||||||
static HEADER_VALUE_TRUE: HeaderValue = HeaderValue::from_static("true");
|
static HEADER_VALUE_TRUE: HeaderValue = HeaderValue::from_static("true");
|
||||||
|
|
||||||
//
|
//
|
||||||
@@ -96,13 +93,6 @@ fn json_array_to_pg_array(value: &Value) -> Result<Option<String>, serde_json::E
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
struct ConnInfo {
|
|
||||||
username: String,
|
|
||||||
dbname: String,
|
|
||||||
hostname: String,
|
|
||||||
password: String,
|
|
||||||
}
|
|
||||||
|
|
||||||
fn get_conn_info(
|
fn get_conn_info(
|
||||||
headers: &HeaderMap,
|
headers: &HeaderMap,
|
||||||
sni_hostname: Option<String>,
|
sni_hostname: Option<String>,
|
||||||
@@ -169,50 +159,23 @@ fn get_conn_info(
|
|||||||
|
|
||||||
// TODO: return different http error codes
|
// TODO: return different http error codes
|
||||||
pub async fn handle(
|
pub async fn handle(
|
||||||
config: &'static ProxyConfig,
|
|
||||||
request: Request<Body>,
|
request: Request<Body>,
|
||||||
sni_hostname: Option<String>,
|
sni_hostname: Option<String>,
|
||||||
|
conn_pool: Arc<GlobalConnPool>,
|
||||||
) -> anyhow::Result<Value> {
|
) -> anyhow::Result<Value> {
|
||||||
//
|
//
|
||||||
// Determine the destination and connection params
|
// Determine the destination and connection params
|
||||||
//
|
//
|
||||||
let headers = request.headers();
|
let headers = request.headers();
|
||||||
let conn_info = get_conn_info(headers, sni_hostname)?;
|
let conn_info = get_conn_info(headers, sni_hostname)?;
|
||||||
let credential_params = StartupMessageParams::new([
|
|
||||||
("user", &conn_info.username),
|
|
||||||
("database", &conn_info.dbname),
|
|
||||||
("application_name", APP_NAME),
|
|
||||||
]);
|
|
||||||
|
|
||||||
// Determine the output options. Default behaviour is 'false'. Anything that is not
|
// Determine the output options. Default behaviour is 'false'. Anything that is not
|
||||||
// strictly 'true' assumed to be false.
|
// strictly 'true' assumed to be false.
|
||||||
let raw_output = headers.get(&RAW_TEXT_OUTPUT) == Some(&HEADER_VALUE_TRUE);
|
let raw_output = headers.get(&RAW_TEXT_OUTPUT) == Some(&HEADER_VALUE_TRUE);
|
||||||
let array_mode = headers.get(&ARRAY_MODE) == Some(&HEADER_VALUE_TRUE);
|
let array_mode = headers.get(&ARRAY_MODE) == Some(&HEADER_VALUE_TRUE);
|
||||||
|
|
||||||
//
|
// Allow connection pooling only if explicitly requested
|
||||||
// Wake up the destination if needed. Code here is a bit involved because
|
let allow_pool = headers.get(&ALLOW_POOL) == Some(&HEADER_VALUE_TRUE);
|
||||||
// we reuse the code from the usual proxy and we need to prepare few structures
|
|
||||||
// that this code expects.
|
|
||||||
//
|
|
||||||
let tls = config.tls_config.as_ref();
|
|
||||||
let common_names = tls.and_then(|tls| tls.common_names.clone());
|
|
||||||
let creds = config
|
|
||||||
.auth_backend
|
|
||||||
.as_ref()
|
|
||||||
.map(|_| {
|
|
||||||
auth::ClientCredentials::parse(
|
|
||||||
&credential_params,
|
|
||||||
Some(&conn_info.hostname),
|
|
||||||
common_names,
|
|
||||||
)
|
|
||||||
})
|
|
||||||
.transpose()?;
|
|
||||||
let extra = console::ConsoleReqExtra {
|
|
||||||
session_id: uuid::Uuid::new_v4(),
|
|
||||||
application_name: Some(APP_NAME),
|
|
||||||
};
|
|
||||||
|
|
||||||
let mut node_info = creds.wake_compute(&extra).await?.expect("msg");
|
|
||||||
|
|
||||||
let request_content_length = match request.body().size_hint().upper() {
|
let request_content_length = match request.body().size_hint().upper() {
|
||||||
Some(v) => v,
|
Some(v) => v,
|
||||||
@@ -235,7 +198,8 @@ pub async fn handle(
|
|||||||
//
|
//
|
||||||
// Now execute the query and return the result
|
// Now execute the query and return the result
|
||||||
//
|
//
|
||||||
let client = connect_to_compute(&mut node_info, &extra, &creds, &conn_info).await?;
|
let client = conn_pool.get(&conn_info, !allow_pool).await?;
|
||||||
|
|
||||||
let row_stream = client.query_raw_txt(query, query_params).await?;
|
let row_stream = client.query_raw_txt(query, query_params).await?;
|
||||||
|
|
||||||
// Manually drain the stream into a vector to leave row_stream hanging
|
// Manually drain the stream into a vector to leave row_stream hanging
|
||||||
@@ -292,6 +256,13 @@ pub async fn handle(
|
|||||||
.map(|row| pg_text_row_to_json(row, raw_output, array_mode))
|
.map(|row| pg_text_row_to_json(row, raw_output, array_mode))
|
||||||
.collect::<Result<Vec<_>, _>>()?;
|
.collect::<Result<Vec<_>, _>>()?;
|
||||||
|
|
||||||
|
if allow_pool {
|
||||||
|
// return connection to the pool
|
||||||
|
tokio::task::spawn(async move {
|
||||||
|
let _ = conn_pool.put(&conn_info, client).await;
|
||||||
|
});
|
||||||
|
}
|
||||||
|
|
||||||
// resulting JSON format is based on the format of node-postgres result
|
// resulting JSON format is based on the format of node-postgres result
|
||||||
Ok(json!({
|
Ok(json!({
|
||||||
"command": command_tag_name,
|
"command": command_tag_name,
|
||||||
@@ -302,70 +273,6 @@ pub async fn handle(
|
|||||||
}))
|
}))
|
||||||
}
|
}
|
||||||
|
|
||||||
/// This function is a copy of `connect_to_compute` from `src/proxy.rs` with
|
|
||||||
/// the difference that it uses `tokio_postgres` for the connection.
|
|
||||||
#[instrument(skip_all)]
|
|
||||||
async fn connect_to_compute(
|
|
||||||
node_info: &mut console::CachedNodeInfo,
|
|
||||||
extra: &console::ConsoleReqExtra<'_>,
|
|
||||||
creds: &auth::BackendType<'_, auth::ClientCredentials<'_>>,
|
|
||||||
conn_info: &ConnInfo,
|
|
||||||
) -> anyhow::Result<tokio_postgres::Client> {
|
|
||||||
let mut num_retries: usize = NUM_RETRIES_WAKE_COMPUTE;
|
|
||||||
|
|
||||||
loop {
|
|
||||||
match connect_to_compute_once(node_info, conn_info).await {
|
|
||||||
Err(e) if num_retries > 0 => {
|
|
||||||
info!("compute node's state has changed; requesting a wake-up");
|
|
||||||
match creds.wake_compute(extra).await? {
|
|
||||||
// Update `node_info` and try one more time.
|
|
||||||
Some(new) => {
|
|
||||||
*node_info = new;
|
|
||||||
}
|
|
||||||
// Link auth doesn't work that way, so we just exit.
|
|
||||||
None => return Err(e),
|
|
||||||
}
|
|
||||||
}
|
|
||||||
other => return other,
|
|
||||||
}
|
|
||||||
|
|
||||||
num_retries -= 1;
|
|
||||||
info!("retrying after wake-up ({num_retries} attempts left)");
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
async fn connect_to_compute_once(
|
|
||||||
node_info: &console::CachedNodeInfo,
|
|
||||||
conn_info: &ConnInfo,
|
|
||||||
) -> anyhow::Result<tokio_postgres::Client> {
|
|
||||||
let mut config = (*node_info.config).clone();
|
|
||||||
|
|
||||||
let (client, connection) = config
|
|
||||||
.user(&conn_info.username)
|
|
||||||
.password(&conn_info.password)
|
|
||||||
.dbname(&conn_info.dbname)
|
|
||||||
.max_backend_message_size(MAX_RESPONSE_SIZE)
|
|
||||||
.connect(tokio_postgres::NoTls)
|
|
||||||
.inspect_err(|e: &tokio_postgres::Error| {
|
|
||||||
error!(
|
|
||||||
"failed to connect to compute node hosts={:?} ports={:?}: {}",
|
|
||||||
node_info.config.get_hosts(),
|
|
||||||
node_info.config.get_ports(),
|
|
||||||
e
|
|
||||||
);
|
|
||||||
invalidate_cache(node_info)
|
|
||||||
})
|
|
||||||
.await?;
|
|
||||||
|
|
||||||
tokio::spawn(async move {
|
|
||||||
if let Err(e) = connection.await {
|
|
||||||
error!("connection error: {}", e);
|
|
||||||
}
|
|
||||||
});
|
|
||||||
|
|
||||||
Ok(client)
|
|
||||||
}
|
|
||||||
|
|
||||||
//
|
//
|
||||||
// Convert postgres row with text-encoded values to JSON object
|
// Convert postgres row with text-encoded values to JSON object
|
||||||
//
|
//
|
||||||
|
|||||||
@@ -35,7 +35,7 @@ use utils::http::{error::ApiError, json::json_response};
|
|||||||
// Tracking issue: https://github.com/rust-lang/rust/issues/98407.
|
// Tracking issue: https://github.com/rust-lang/rust/issues/98407.
|
||||||
use sync_wrapper::SyncWrapper;
|
use sync_wrapper::SyncWrapper;
|
||||||
|
|
||||||
use super::sql_over_http;
|
use super::{conn_pool::GlobalConnPool, sql_over_http};
|
||||||
|
|
||||||
pin_project! {
|
pin_project! {
|
||||||
/// This is a wrapper around a [`WebSocketStream`] that
|
/// This is a wrapper around a [`WebSocketStream`] that
|
||||||
@@ -164,6 +164,7 @@ async fn serve_websocket(
|
|||||||
async fn ws_handler(
|
async fn ws_handler(
|
||||||
mut request: Request<Body>,
|
mut request: Request<Body>,
|
||||||
config: &'static ProxyConfig,
|
config: &'static ProxyConfig,
|
||||||
|
conn_pool: Arc<GlobalConnPool>,
|
||||||
cancel_map: Arc<CancelMap>,
|
cancel_map: Arc<CancelMap>,
|
||||||
session_id: uuid::Uuid,
|
session_id: uuid::Uuid,
|
||||||
sni_hostname: Option<String>,
|
sni_hostname: Option<String>,
|
||||||
@@ -192,7 +193,7 @@ async fn ws_handler(
|
|||||||
// TODO: that deserves a refactor as now this function also handles http json client besides websockets.
|
// TODO: that deserves a refactor as now this function also handles http json client besides websockets.
|
||||||
// Right now I don't want to blow up sql-over-http patch with file renames and do that as a follow up instead.
|
// Right now I don't want to blow up sql-over-http patch with file renames and do that as a follow up instead.
|
||||||
} else if request.uri().path() == "/sql" && request.method() == Method::POST {
|
} else if request.uri().path() == "/sql" && request.method() == Method::POST {
|
||||||
let result = sql_over_http::handle(config, request, sni_hostname)
|
let result = sql_over_http::handle(request, sni_hostname, conn_pool)
|
||||||
.instrument(info_span!("sql-over-http"))
|
.instrument(info_span!("sql-over-http"))
|
||||||
.await;
|
.await;
|
||||||
let status_code = match result {
|
let status_code = match result {
|
||||||
@@ -234,6 +235,8 @@ pub async fn task_main(
|
|||||||
info!("websocket server has shut down");
|
info!("websocket server has shut down");
|
||||||
}
|
}
|
||||||
|
|
||||||
|
let conn_pool: Arc<GlobalConnPool> = GlobalConnPool::new(config);
|
||||||
|
|
||||||
let tls_config = config.tls_config.as_ref().map(|cfg| cfg.to_server_config());
|
let tls_config = config.tls_config.as_ref().map(|cfg| cfg.to_server_config());
|
||||||
let tls_acceptor: tokio_rustls::TlsAcceptor = match tls_config {
|
let tls_acceptor: tokio_rustls::TlsAcceptor = match tls_config {
|
||||||
Some(config) => config.into(),
|
Some(config) => config.into(),
|
||||||
@@ -258,15 +261,18 @@ pub async fn task_main(
|
|||||||
let make_svc =
|
let make_svc =
|
||||||
hyper::service::make_service_fn(|stream: &tokio_rustls::server::TlsStream<AddrStream>| {
|
hyper::service::make_service_fn(|stream: &tokio_rustls::server::TlsStream<AddrStream>| {
|
||||||
let sni_name = stream.get_ref().1.sni_hostname().map(|s| s.to_string());
|
let sni_name = stream.get_ref().1.sni_hostname().map(|s| s.to_string());
|
||||||
|
let conn_pool = conn_pool.clone();
|
||||||
|
|
||||||
async move {
|
async move {
|
||||||
Ok::<_, Infallible>(hyper::service::service_fn(move |req: Request<Body>| {
|
Ok::<_, Infallible>(hyper::service::service_fn(move |req: Request<Body>| {
|
||||||
let sni_name = sni_name.clone();
|
let sni_name = sni_name.clone();
|
||||||
|
let conn_pool = conn_pool.clone();
|
||||||
|
|
||||||
async move {
|
async move {
|
||||||
let cancel_map = Arc::new(CancelMap::default());
|
let cancel_map = Arc::new(CancelMap::default());
|
||||||
let session_id = uuid::Uuid::new_v4();
|
let session_id = uuid::Uuid::new_v4();
|
||||||
|
|
||||||
ws_handler(req, config, cancel_map, session_id, sni_name)
|
ws_handler(req, config, conn_pool, cancel_map, session_id, sni_name)
|
||||||
.instrument(info_span!(
|
.instrument(info_span!(
|
||||||
"ws-client",
|
"ws-client",
|
||||||
session = format_args!("{session_id}")
|
session = format_args!("{session_id}")
|
||||||
|
|||||||
+27
-3
@@ -16,7 +16,10 @@ use metrics::{register_int_counter, register_int_counter_vec, IntCounter, IntCou
|
|||||||
use once_cell::sync::Lazy;
|
use once_cell::sync::Lazy;
|
||||||
use pq_proto::{BeMessage as Be, FeStartupPacket, StartupMessageParams};
|
use pq_proto::{BeMessage as Be, FeStartupPacket, StartupMessageParams};
|
||||||
use std::sync::Arc;
|
use std::sync::Arc;
|
||||||
use tokio::io::{AsyncRead, AsyncWrite, AsyncWriteExt};
|
use tokio::{
|
||||||
|
io::{AsyncRead, AsyncWrite, AsyncWriteExt},
|
||||||
|
time,
|
||||||
|
};
|
||||||
use tokio_util::sync::CancellationToken;
|
use tokio_util::sync::CancellationToken;
|
||||||
use tracing::{error, info, warn};
|
use tracing::{error, info, warn};
|
||||||
use utils::measured_stream::MeasuredStream;
|
use utils::measured_stream::MeasuredStream;
|
||||||
@@ -305,12 +308,13 @@ pub fn invalidate_cache(node_info: &console::CachedNodeInfo) {
|
|||||||
#[tracing::instrument(name = "connect_once", skip_all)]
|
#[tracing::instrument(name = "connect_once", skip_all)]
|
||||||
async fn connect_to_compute_once(
|
async fn connect_to_compute_once(
|
||||||
node_info: &console::CachedNodeInfo,
|
node_info: &console::CachedNodeInfo,
|
||||||
|
timeout: time::Duration,
|
||||||
) -> Result<PostgresConnection, compute::ConnectionError> {
|
) -> Result<PostgresConnection, compute::ConnectionError> {
|
||||||
let allow_self_signed_compute = node_info.allow_self_signed_compute;
|
let allow_self_signed_compute = node_info.allow_self_signed_compute;
|
||||||
|
|
||||||
node_info
|
node_info
|
||||||
.config
|
.config
|
||||||
.connect(allow_self_signed_compute)
|
.connect(allow_self_signed_compute, timeout)
|
||||||
.inspect_err(|_: &compute::ConnectionError| invalidate_cache(node_info))
|
.inspect_err(|_: &compute::ConnectionError| invalidate_cache(node_info))
|
||||||
.await
|
.await
|
||||||
}
|
}
|
||||||
@@ -328,7 +332,27 @@ async fn connect_to_compute(
|
|||||||
loop {
|
loop {
|
||||||
// Apply startup params to the (possibly, cached) compute node info.
|
// Apply startup params to the (possibly, cached) compute node info.
|
||||||
node_info.config.set_startup_params(params);
|
node_info.config.set_startup_params(params);
|
||||||
match connect_to_compute_once(node_info).await {
|
|
||||||
|
// Set a shorter timeout for the initial connection attempt.
|
||||||
|
//
|
||||||
|
// In case we try to connect to an outdated address that is no longer valid, the
|
||||||
|
// default behavior of Kubernetes is to drop the packets, causing us to wait for
|
||||||
|
// the entire timeout period. We want to fail fast in such cases.
|
||||||
|
//
|
||||||
|
// A specific case to consider is when we have cached compute node information
|
||||||
|
// with a 4-minute TTL (Time To Live), but the user has executed a `/suspend` API
|
||||||
|
// call, resulting in the nonexistence of the compute node.
|
||||||
|
//
|
||||||
|
// We only use caching in case of scram proxy backed by the console, so reduce
|
||||||
|
// the timeout only in that case.
|
||||||
|
let is_scram_proxy = matches!(creds, auth::BackendType::Console(_, _));
|
||||||
|
let timeout = if is_scram_proxy && num_retries == NUM_RETRIES_WAKE_COMPUTE {
|
||||||
|
time::Duration::from_secs(2)
|
||||||
|
} else {
|
||||||
|
time::Duration::from_secs(10)
|
||||||
|
};
|
||||||
|
|
||||||
|
match connect_to_compute_once(node_info, timeout).await {
|
||||||
Err(e) if num_retries > 0 => {
|
Err(e) if num_retries > 0 => {
|
||||||
info!("compute node's state has changed; requesting a wake-up");
|
info!("compute node's state has changed; requesting a wake-up");
|
||||||
match creds.wake_compute(extra).map_err(io_error).await? {
|
match creds.wake_compute(extra).map_err(io_error).await? {
|
||||||
|
|||||||
+64
-7
@@ -45,17 +45,74 @@ fn hmac_sha256<'a>(key: &[u8], parts: impl IntoIterator<Item = &'a [u8]>) -> [u8
|
|||||||
let mut mac = Hmac::<Sha256>::new_from_slice(key).expect("bad key size");
|
let mut mac = Hmac::<Sha256>::new_from_slice(key).expect("bad key size");
|
||||||
parts.into_iter().for_each(|s| mac.update(s));
|
parts.into_iter().for_each(|s| mac.update(s));
|
||||||
|
|
||||||
// TODO: maybe newer `hmac` et al already migrated to regular arrays?
|
mac.finalize().into_bytes().into()
|
||||||
let mut result = [0u8; 32];
|
|
||||||
result.copy_from_slice(mac.finalize().into_bytes().as_slice());
|
|
||||||
result
|
|
||||||
}
|
}
|
||||||
|
|
||||||
fn sha256<'a>(parts: impl IntoIterator<Item = &'a [u8]>) -> [u8; 32] {
|
fn sha256<'a>(parts: impl IntoIterator<Item = &'a [u8]>) -> [u8; 32] {
|
||||||
let mut hasher = Sha256::new();
|
let mut hasher = Sha256::new();
|
||||||
parts.into_iter().for_each(|s| hasher.update(s));
|
parts.into_iter().for_each(|s| hasher.update(s));
|
||||||
|
|
||||||
let mut result = [0u8; 32];
|
hasher.finalize().into()
|
||||||
result.copy_from_slice(hasher.finalize().as_slice());
|
}
|
||||||
result
|
|
||||||
|
#[cfg(test)]
|
||||||
|
mod tests {
|
||||||
|
use crate::sasl::{Mechanism, Step};
|
||||||
|
|
||||||
|
use super::{password::SaltedPassword, Exchange, ServerSecret};
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn happy_path() {
|
||||||
|
let iterations = 4096;
|
||||||
|
let salt_base64 = "QSXCR+Q6sek8bf92";
|
||||||
|
let pw = SaltedPassword::new(
|
||||||
|
b"pencil",
|
||||||
|
base64::decode(salt_base64).unwrap().as_slice(),
|
||||||
|
iterations,
|
||||||
|
);
|
||||||
|
|
||||||
|
let secret = ServerSecret {
|
||||||
|
iterations,
|
||||||
|
salt_base64: salt_base64.to_owned(),
|
||||||
|
stored_key: pw.client_key().sha256(),
|
||||||
|
server_key: pw.server_key(),
|
||||||
|
doomed: false,
|
||||||
|
};
|
||||||
|
const NONCE: [u8; 18] = [
|
||||||
|
1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12, 13, 14, 15, 16, 17, 18,
|
||||||
|
];
|
||||||
|
let mut exchange = Exchange::new(&secret, || NONCE, None);
|
||||||
|
|
||||||
|
let client_first = "n,,n=user,r=rOprNGfwEbeRWgbNEkqO";
|
||||||
|
let client_final = "c=biws,r=rOprNGfwEbeRWgbNEkqOAQIDBAUGBwgJCgsMDQ4PEBES,p=rw1r5Kph5ThxmaUBC2GAQ6MfXbPnNkFiTIvdb/Rear0=";
|
||||||
|
let server_first =
|
||||||
|
"r=rOprNGfwEbeRWgbNEkqOAQIDBAUGBwgJCgsMDQ4PEBES,s=QSXCR+Q6sek8bf92,i=4096";
|
||||||
|
let server_final = "v=qtUDIofVnIhM7tKn93EQUUt5vgMOldcDVu1HC+OH0o0=";
|
||||||
|
|
||||||
|
exchange = match exchange.exchange(client_first).unwrap() {
|
||||||
|
Step::Continue(exchange, message) => {
|
||||||
|
assert_eq!(message, server_first);
|
||||||
|
exchange
|
||||||
|
}
|
||||||
|
Step::Success(_, _) => panic!("expected continue, got success"),
|
||||||
|
Step::Failure(f) => panic!("{f}"),
|
||||||
|
};
|
||||||
|
|
||||||
|
let key = match exchange.exchange(client_final).unwrap() {
|
||||||
|
Step::Success(key, message) => {
|
||||||
|
assert_eq!(message, server_final);
|
||||||
|
key
|
||||||
|
}
|
||||||
|
Step::Continue(_, _) => panic!("expected success, got continue"),
|
||||||
|
Step::Failure(f) => panic!("{f}"),
|
||||||
|
};
|
||||||
|
|
||||||
|
assert_eq!(
|
||||||
|
key.as_bytes(),
|
||||||
|
[
|
||||||
|
74, 103, 1, 132, 12, 31, 200, 48, 28, 54, 82, 232, 207, 12, 138, 189, 40, 32, 134,
|
||||||
|
27, 125, 170, 232, 35, 171, 167, 166, 41, 70, 228, 182, 112,
|
||||||
|
]
|
||||||
|
);
|
||||||
|
}
|
||||||
}
|
}
|
||||||
|
|||||||
+39
-13
@@ -14,19 +14,7 @@ impl SaltedPassword {
|
|||||||
/// See `scram-common.c : scram_SaltedPassword` for details.
|
/// See `scram-common.c : scram_SaltedPassword` for details.
|
||||||
/// Further reading: <https://datatracker.ietf.org/doc/html/rfc2898> (see `PBKDF2`).
|
/// Further reading: <https://datatracker.ietf.org/doc/html/rfc2898> (see `PBKDF2`).
|
||||||
pub fn new(password: &[u8], salt: &[u8], iterations: u32) -> SaltedPassword {
|
pub fn new(password: &[u8], salt: &[u8], iterations: u32) -> SaltedPassword {
|
||||||
let one = 1_u32.to_be_bytes(); // magic
|
pbkdf2::pbkdf2_hmac_array::<sha2::Sha256, 32>(password, salt, iterations).into()
|
||||||
|
|
||||||
let mut current = super::hmac_sha256(password, [salt, &one]);
|
|
||||||
let mut result = current;
|
|
||||||
for _ in 1..iterations {
|
|
||||||
current = super::hmac_sha256(password, [current.as_ref()]);
|
|
||||||
// TODO: result = current.zip(result).map(|(x, y)| x ^ y), issue #80094
|
|
||||||
for (i, x) in current.iter().enumerate() {
|
|
||||||
result[i] ^= x;
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
result.into()
|
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Derive `ClientKey` from a salted hashed password.
|
/// Derive `ClientKey` from a salted hashed password.
|
||||||
@@ -46,3 +34,41 @@ impl From<[u8; SALTED_PASSWORD_LEN]> for SaltedPassword {
|
|||||||
Self { bytes }
|
Self { bytes }
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
#[cfg(test)]
|
||||||
|
mod tests {
|
||||||
|
use super::SaltedPassword;
|
||||||
|
|
||||||
|
fn legacy_pbkdf2_impl(password: &[u8], salt: &[u8], iterations: u32) -> SaltedPassword {
|
||||||
|
let one = 1_u32.to_be_bytes(); // magic
|
||||||
|
|
||||||
|
let mut current = super::super::hmac_sha256(password, [salt, &one]);
|
||||||
|
let mut result = current;
|
||||||
|
for _ in 1..iterations {
|
||||||
|
current = super::super::hmac_sha256(password, [current.as_ref()]);
|
||||||
|
// TODO: result = current.zip(result).map(|(x, y)| x ^ y), issue #80094
|
||||||
|
for (i, x) in current.iter().enumerate() {
|
||||||
|
result[i] ^= x;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
result.into()
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn pbkdf2() {
|
||||||
|
let password = "a-very-secure-password";
|
||||||
|
let salt = "such-a-random-salt";
|
||||||
|
let iterations = 4096;
|
||||||
|
let output = [
|
||||||
|
203, 18, 206, 81, 4, 154, 193, 100, 147, 41, 211, 217, 177, 203, 69, 210, 194, 211,
|
||||||
|
101, 1, 248, 156, 96, 0, 8, 223, 30, 87, 158, 41, 20, 42,
|
||||||
|
];
|
||||||
|
|
||||||
|
let actual = SaltedPassword::new(password.as_bytes(), salt.as_bytes(), iterations);
|
||||||
|
let expected = legacy_pbkdf2_impl(password.as_bytes(), salt.as_bytes(), iterations);
|
||||||
|
|
||||||
|
assert_eq!(actual.bytes, output);
|
||||||
|
assert_eq!(actual.bytes, expected.bytes);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|||||||
@@ -191,6 +191,12 @@ impl Storage for FileStorage {
|
|||||||
control_partial_path.display()
|
control_partial_path.display()
|
||||||
)
|
)
|
||||||
})?;
|
})?;
|
||||||
|
control_partial.flush().await.with_context(|| {
|
||||||
|
format!(
|
||||||
|
"failed to flush safekeeper state into control file at: {}",
|
||||||
|
control_partial_path.display()
|
||||||
|
)
|
||||||
|
})?;
|
||||||
|
|
||||||
// fsync the file
|
// fsync the file
|
||||||
if !self.conf.no_sync {
|
if !self.conf.no_sync {
|
||||||
|
|||||||
@@ -188,6 +188,7 @@ async fn pull_timeline(status: TimelineStatus, host: String) -> Result<Response>
|
|||||||
let mut response = client.get(&http_url).send().await?;
|
let mut response = client.get(&http_url).send().await?;
|
||||||
while let Some(chunk) = response.chunk().await? {
|
while let Some(chunk) = response.chunk().await? {
|
||||||
file.write_all(&chunk).await?;
|
file.write_all(&chunk).await?;
|
||||||
|
file.flush().await?;
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
|||||||
@@ -403,16 +403,18 @@ impl SafekeeperPostgresHandler {
|
|||||||
};
|
};
|
||||||
|
|
||||||
// take the latest commit_lsn if don't have stop_pos
|
// take the latest commit_lsn if don't have stop_pos
|
||||||
let mut end_pos = stop_pos.unwrap_or(*commit_lsn_watch_rx.borrow());
|
let end_pos = stop_pos.unwrap_or(*commit_lsn_watch_rx.borrow());
|
||||||
|
|
||||||
if end_pos < start_pos {
|
if end_pos < start_pos {
|
||||||
warn!("start_pos {} is ahead of end_pos {}", start_pos, end_pos);
|
warn!(
|
||||||
end_pos = start_pos;
|
"requested start_pos {} is ahead of available WAL end_pos {}",
|
||||||
|
start_pos, end_pos
|
||||||
|
);
|
||||||
}
|
}
|
||||||
|
|
||||||
info!(
|
info!(
|
||||||
"starting streaming from {:?} till {:?}",
|
"starting streaming from {:?} till {:?}, available WAL ends at {}",
|
||||||
start_pos, stop_pos
|
start_pos, stop_pos, end_pos
|
||||||
);
|
);
|
||||||
|
|
||||||
// switch to copy
|
// switch to copy
|
||||||
@@ -547,12 +549,14 @@ impl<IO: AsyncRead + AsyncWrite + Unpin> WalSender<'_, IO> {
|
|||||||
self.end_pos = *self.commit_lsn_watch_rx.borrow();
|
self.end_pos = *self.commit_lsn_watch_rx.borrow();
|
||||||
if self.end_pos > self.start_pos {
|
if self.end_pos > self.start_pos {
|
||||||
// We have something to send.
|
// We have something to send.
|
||||||
|
trace!("got end_pos {:?}, streaming", self.end_pos);
|
||||||
return Ok(());
|
return Ok(());
|
||||||
}
|
}
|
||||||
|
|
||||||
// Wait for WAL to appear, now self.end_pos == self.start_pos.
|
// Wait for WAL to appear, now self.end_pos == self.start_pos.
|
||||||
if let Some(lsn) = wait_for_lsn(&mut self.commit_lsn_watch_rx, self.start_pos).await? {
|
if let Some(lsn) = wait_for_lsn(&mut self.commit_lsn_watch_rx, self.start_pos).await? {
|
||||||
self.end_pos = lsn;
|
self.end_pos = lsn;
|
||||||
|
trace!("got end_pos {:?}, streaming", self.end_pos);
|
||||||
return Ok(());
|
return Ok(());
|
||||||
}
|
}
|
||||||
|
|
||||||
|
|||||||
@@ -248,6 +248,10 @@ impl PhysicalStorage {
|
|||||||
};
|
};
|
||||||
|
|
||||||
file.write_all(buf).await?;
|
file.write_all(buf).await?;
|
||||||
|
// Note: flush just ensures write above reaches the OS (this is not
|
||||||
|
// needed in case of sync IO as Write::write there calls directly write
|
||||||
|
// syscall, but needed in case of async). It does *not* fsyncs the file.
|
||||||
|
file.flush().await?;
|
||||||
|
|
||||||
if xlogoff + buf.len() == self.wal_seg_size {
|
if xlogoff + buf.len() == self.wal_seg_size {
|
||||||
// If we reached the end of a WAL segment, flush and close it.
|
// If we reached the end of a WAL segment, flush and close it.
|
||||||
@@ -716,6 +720,7 @@ async fn write_zeroes(file: &mut File, mut count: usize) -> Result<()> {
|
|||||||
count -= XLOG_BLCKSZ;
|
count -= XLOG_BLCKSZ;
|
||||||
}
|
}
|
||||||
file.write_all(&ZERO_BLOCK[0..count]).await?;
|
file.write_all(&ZERO_BLOCK[0..count]).await?;
|
||||||
|
file.flush().await?;
|
||||||
Ok(())
|
Ok(())
|
||||||
}
|
}
|
||||||
|
|
||||||
|
|||||||
@@ -59,6 +59,10 @@ PAGESERVER_GLOBAL_METRICS: Tuple[str, ...] = (
|
|||||||
"libmetrics_tracing_event_count_total",
|
"libmetrics_tracing_event_count_total",
|
||||||
"pageserver_materialized_cache_hits_total",
|
"pageserver_materialized_cache_hits_total",
|
||||||
"pageserver_materialized_cache_hits_direct_total",
|
"pageserver_materialized_cache_hits_direct_total",
|
||||||
|
"pageserver_page_cache_read_hits_total",
|
||||||
|
"pageserver_page_cache_read_accesses_total",
|
||||||
|
"pageserver_page_cache_size_current_bytes",
|
||||||
|
"pageserver_page_cache_size_max_bytes",
|
||||||
"pageserver_getpage_reconstruct_seconds_bucket",
|
"pageserver_getpage_reconstruct_seconds_bucket",
|
||||||
"pageserver_getpage_reconstruct_seconds_count",
|
"pageserver_getpage_reconstruct_seconds_count",
|
||||||
"pageserver_getpage_reconstruct_seconds_sum",
|
"pageserver_getpage_reconstruct_seconds_sum",
|
||||||
|
|||||||
@@ -52,6 +52,7 @@ def test_startup_simple(neon_env_builder: NeonEnvBuilder, zenbenchmark: NeonBenc
|
|||||||
"wait_for_spec_ms": f"{i}_wait_for_spec",
|
"wait_for_spec_ms": f"{i}_wait_for_spec",
|
||||||
"sync_safekeepers_ms": f"{i}_sync_safekeepers",
|
"sync_safekeepers_ms": f"{i}_sync_safekeepers",
|
||||||
"basebackup_ms": f"{i}_basebackup",
|
"basebackup_ms": f"{i}_basebackup",
|
||||||
|
"start_postgres_ms": f"{i}_start_postgres",
|
||||||
"config_ms": f"{i}_config",
|
"config_ms": f"{i}_config",
|
||||||
"total_startup_ms": f"{i}_total_startup",
|
"total_startup_ms": f"{i}_total_startup",
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -1,3 +1,7 @@
|
|||||||
|
import random
|
||||||
|
import threading
|
||||||
|
from threading import Thread
|
||||||
|
|
||||||
from fixtures.log_helper import log
|
from fixtures.log_helper import log
|
||||||
from fixtures.neon_fixtures import NeonEnv, check_restored_datadir_content
|
from fixtures.neon_fixtures import NeonEnv, check_restored_datadir_content
|
||||||
from fixtures.utils import query_scalar
|
from fixtures.utils import query_scalar
|
||||||
@@ -15,11 +19,17 @@ def test_multixact(neon_simple_env: NeonEnv, test_output_dir):
|
|||||||
endpoint = env.endpoints.create_start("test_multixact")
|
endpoint = env.endpoints.create_start("test_multixact")
|
||||||
|
|
||||||
log.info("postgres is running on 'test_multixact' branch")
|
log.info("postgres is running on 'test_multixact' branch")
|
||||||
|
|
||||||
|
n_records = 100
|
||||||
|
n_threads = 5
|
||||||
|
n_iters = 1000
|
||||||
|
n_restarts = 10
|
||||||
|
|
||||||
cur = endpoint.connect().cursor()
|
cur = endpoint.connect().cursor()
|
||||||
cur.execute(
|
cur.execute(
|
||||||
"""
|
f"""
|
||||||
CREATE TABLE t1(i int primary key);
|
CREATE TABLE t1(pk int primary key, val integer);
|
||||||
INSERT INTO t1 select * from generate_series(1, 100);
|
INSERT INTO t1 values (generate_series(1, {n_records}), 0);
|
||||||
"""
|
"""
|
||||||
)
|
)
|
||||||
|
|
||||||
@@ -28,26 +38,32 @@ def test_multixact(neon_simple_env: NeonEnv, test_output_dir):
|
|||||||
)
|
)
|
||||||
|
|
||||||
# Lock entries using parallel connections in a round-robin fashion.
|
# Lock entries using parallel connections in a round-robin fashion.
|
||||||
nclients = 20
|
def do_updates():
|
||||||
connections = []
|
|
||||||
for i in range(nclients):
|
|
||||||
# Do not turn on autocommit. We want to hold the key-share locks.
|
|
||||||
conn = endpoint.connect(autocommit=False)
|
conn = endpoint.connect(autocommit=False)
|
||||||
connections.append(conn)
|
for i in range(n_iters):
|
||||||
|
pk = random.randrange(1, n_records)
|
||||||
|
conn.cursor().execute(f"update t1 set val=val+1 where pk={pk}")
|
||||||
|
conn.cursor().execute("select * from t1 for key share")
|
||||||
|
conn.commit()
|
||||||
|
conn.close()
|
||||||
|
|
||||||
# On each iteration, we commit the previous transaction on a connection,
|
for iter in range(n_restarts):
|
||||||
# and issue antoher select. Each SELECT generates a new multixact that
|
threads: List[threading.Thread] = []
|
||||||
# includes the new XID, and the XIDs of all the other parallel transactions.
|
for i in range(n_threads):
|
||||||
# This generates enough traffic on both multixact offsets and members SLRUs
|
threads.append(threading.Thread(target=do_updates, args=(), daemon=False))
|
||||||
# to cross page boundaries.
|
threads[-1].start()
|
||||||
for i in range(5000):
|
|
||||||
conn = connections[i % nclients]
|
|
||||||
conn.commit()
|
|
||||||
conn.cursor().execute("select * from t1 for key share")
|
|
||||||
|
|
||||||
# We have multixacts now. We can close the connections.
|
for thread in threads:
|
||||||
for c in connections:
|
thread.join()
|
||||||
c.close()
|
|
||||||
|
# Restart endpoint
|
||||||
|
endpoint.stop()
|
||||||
|
endpoint.start()
|
||||||
|
|
||||||
|
conn = endpoint.connect()
|
||||||
|
cur = conn.cursor()
|
||||||
|
cur.execute("select count(*) from t1")
|
||||||
|
assert cur.fetchone() == (n_records,)
|
||||||
|
|
||||||
# force wal flush
|
# force wal flush
|
||||||
cur.execute("checkpoint")
|
cur.execute("checkpoint")
|
||||||
@@ -74,6 +90,3 @@ def test_multixact(neon_simple_env: NeonEnv, test_output_dir):
|
|||||||
|
|
||||||
# Check that we restored pg_controlfile correctly
|
# Check that we restored pg_controlfile correctly
|
||||||
assert next_multixact_id_new == next_multixact_id
|
assert next_multixact_id_new == next_multixact_id
|
||||||
|
|
||||||
# Check that we can restore the content of the datadir correctly
|
|
||||||
check_restored_datadir_content(test_output_dir, env, endpoint)
|
|
||||||
|
|||||||
Reference in New Issue
Block a user