mirror of
https://github.com/GreptimeTeam/greptimedb.git
synced 2026-10-03 10:35:35 +00:00
* feat(log-store): add the enqueued acknowledgement mode to the object store WAL Add `ack_mode` (`durable` by default, or `enqueued`) and the backlog thresholds `max_unpersisted_bytes` and `max_unpersisted_age` to the object store WAL config, validated by the datanode and the store. In the `enqueued` mode `append_batch` returns on admission with the entry ids assigned and the object is created in the background. At a backlog threshold the next append is held back until an upload completes. A transient create failure is repeated under the same sequence with the same bytes; any conflicting object poisons the store. `stop` uploads the backlog, or returns the error that dropped it once stop began. `obsolete` clamps the watermark to the durable entry id, and an id the store handed out needs no sequence floor. Add `LogStore::wait_durable` with a default that returns at once. The object store WAL answers it once the region is durable and indexed through the entry id, and fails it after a backlog was lost. Signed-off-by: jeremyhi <fengjiachun@gmail.com> * fix(log-store): poison an enqueued store on permanent create failures Repeat a failed create in the enqueued mode only when the storage error is retryable; any other storage error poisons the store. A conflicting object poisons an enqueued store without reading its epoch, so a failed header read cannot turn the conflict into a retry. A durability wait for an id above the highest id the store handed out now waits for the handed-out ids of the region below it instead of returning at once. Signed-off-by: jeremyhi <fengjiachun@gmail.com> * fix(log-store): poison on permanent create failures after stop begins A create that fails with a storage error that is not retryable poisons an enqueued store even after stop began; only a transient failure drops the backlog without poisoning. Durability waiters whose callers stopped waiting are pruned before a new waiter is queued. The backlog age test no longer depends on a follow-up append finishing within the threshold. Signed-off-by: jeremyhi <fengjiachun@gmail.com> * fix(log-store): answer a durability wait once no earlier entry is pending A durability wait now returns once the region holds no entry at or below the target that is handed out but not durable, instead of waiting for the largest id handed out to the region. A later object of the region that is still being created no longer holds back a wait whose target it does not cover. Signed-off-by: jeremyhi <fengjiachun@gmail.com> * test(log-store): order the pending durability wait check after the actor Signed-off-by: jeremyhi <fengjiachun@gmail.com> * docs(store-api): state that the default wait_durable keeps each store's guarantee The default `LogStore::wait_durable` returns at once, which keeps each log store's own acknowledgement guarantee; Raft Engine with `sync_write = false` acknowledges before its periodic sync, so the documentation no longer claims that every entry id a caller holds is durable. The object store WAL configuration test now also serializes the new options and reads them back. Signed-off-by: jeremyhi <fengjiachun@gmail.com> * docs(log-store): limit the acknowledgement guarantees to the durable mode Signed-off-by: jeremyhi <fengjiachun@gmail.com> * fix(log-store): repeat enqueued creates that a retry layer marks persistent An object store wrapped in the OpenDAL retry layer reports a temporary error that outlasted its retries as persistent rather than temporary. The enqueued mode now repeats a create after any storage error that is not permanent, so a transient outage behind the retry layer no longer poisons the store and drops the acknowledged backlog; after stop began such a failure still drops the backlog without poisoning. Signed-off-by: jeremyhi <fengjiachun@gmail.com> * test(log-store): cover a persistent create failure after stop begins Signed-off-by: jeremyhi <fengjiachun@gmail.com> --------- Signed-off-by: jeremyhi <fengjiachun@gmail.com>
886 lines
33 KiB
TOML
886 lines
33 KiB
TOML
## The datanode identifier and should be unique in the cluster.
|
|
## @toml2docs:none-default
|
|
node_id = 42
|
|
|
|
## The default column prefix for auto-created time index, value, and native histogram columns.
|
|
## Legacy OTLP summary columns keep their historical `greptime_` prefix.
|
|
## @toml2docs:none-default
|
|
default_column_prefix = "greptime"
|
|
|
|
## Start services after regions have obtained leases.
|
|
## It will block the datanode start if it can't receive leases in the heartbeat from metasrv.
|
|
require_lease_before_startup = false
|
|
|
|
## Initialize all regions in the background during the startup.
|
|
## By default, it provides services after all regions have been initialized.
|
|
init_regions_in_background = false
|
|
|
|
## Parallelism of initializing regions.
|
|
init_regions_parallelism = 16
|
|
|
|
## The maximum concurrent queries allowed to be executed. Zero means unlimited.
|
|
max_concurrent_queries = 0
|
|
|
|
## Timeout to acquire a permit from the concurrent query limiter when `max_concurrent_queries` is reached.
|
|
concurrent_query_limiter_timeout = "100ms"
|
|
|
|
## Enable telemetry to collect anonymous usage data. Enabled by default.
|
|
#+ enable_telemetry = true
|
|
|
|
## The HTTP server options.
|
|
[http]
|
|
## The address to bind the HTTP server.
|
|
addr = "127.0.0.1:4000"
|
|
## HTTP request timeout. Set to 0 to disable timeout.
|
|
timeout = "0s"
|
|
## HTTP request body limit.
|
|
## The following units are supported: `B`, `KB`, `KiB`, `MB`, `MiB`, `GB`, `GiB`, `TB`, `TiB`, `PB`, `PiB`.
|
|
## Set to 0 to disable limit.
|
|
body_limit = "64MB"
|
|
|
|
## The gRPC server options.
|
|
[grpc]
|
|
## The address to bind the gRPC server.
|
|
bind_addr = "127.0.0.1:3001"
|
|
## The address advertised to the metasrv, and used for connections from outside the host.
|
|
## If left empty or unset, the server will automatically use the IP address of the first network interface
|
|
## on the host, with the same port number as the one specified in `grpc.bind_addr`.
|
|
server_addr = "127.0.0.1:3001"
|
|
## The number of server worker threads.
|
|
runtime_size = 8
|
|
## The maximum receive message size for gRPC server.
|
|
max_recv_message_size = "512MB"
|
|
## The maximum send message size for gRPC server.
|
|
max_send_message_size = "512MB"
|
|
## Compression mode for datanode side Arrow IPC service. Available options:
|
|
## - `none`: disable all compression
|
|
## - `transport`: only enable gRPC transport compression (zstd)
|
|
## - `arrow_ipc`: only enable Arrow IPC compression (lz4)
|
|
## - `all`: enable all compression.
|
|
## Default to `none`
|
|
flight_compression = "arrow_ipc"
|
|
|
|
## gRPC server TLS options, see `mysql.tls` section.
|
|
[grpc.tls]
|
|
## TLS mode.
|
|
mode = "disable"
|
|
|
|
## Certificate file path.
|
|
## @toml2docs:none-default
|
|
cert_path = ""
|
|
|
|
## Private key file path.
|
|
## @toml2docs:none-default
|
|
key_path = ""
|
|
|
|
## Watch for Certificate and key file change and auto reload.
|
|
## For now, gRPC tls config does not support auto reload.
|
|
watch = false
|
|
|
|
## The runtime options.
|
|
#+ [runtime]
|
|
## The number of threads to execute the runtime for global read operations.
|
|
#+ global_rt_size = 8
|
|
## The number of threads to execute compact operations.
|
|
#+ compact_rt_size = 4
|
|
## The maximum number of blocking threads for compact operations.
|
|
## Defaults to max(num_cpus / 2, 2).
|
|
#+ compact_rt_max_blocking_threads = 4
|
|
## The number of threads to execute datanode query operations.
|
|
## Defaults to max(num_cpus - 1, 2).
|
|
#+ query_rt_size = 7
|
|
## The number of threads to execute datanode ingestion operations.
|
|
#+ ingest_rt_size = 8
|
|
|
|
## Experimental weighted, work-conserving query/write task scheduler.
|
|
#+ [runtime.experimental_workload_scheduler]
|
|
## Enable when concurrent queries and writes interfere with each other—for example, when long-running queries increase ingestion latency.
|
|
## The weights set their relative runtime shares while both are backlogged. Disabled by default.
|
|
#+ enable = false
|
|
## Relative query share while both query and write workloads are backlogged.
|
|
#+ query_weight = 2
|
|
## Relative write share while both query and write workloads are backlogged.
|
|
#+ write_weight = 8
|
|
## Number of polls between scheduler fairness samples. Must be greater than zero.
|
|
#+ sample_every_polls = 16
|
|
|
|
## The metasrv client options.
|
|
[meta_client]
|
|
## The addresses of the metasrv.
|
|
metasrv_addrs = ["127.0.0.1:3002"]
|
|
|
|
## Operation timeout.
|
|
timeout = "3s"
|
|
|
|
## DDL timeout.
|
|
ddl_timeout = "10s"
|
|
|
|
## Connect server timeout.
|
|
connect_timeout = "1s"
|
|
|
|
## `TCP_NODELAY` option for accepted connections.
|
|
tcp_nodelay = true
|
|
|
|
## The configuration about the cache of the metadata.
|
|
metadata_cache_max_capacity = 100000
|
|
|
|
## TTL of the metadata cache.
|
|
metadata_cache_ttl = "10m"
|
|
|
|
# TTI of the metadata cache.
|
|
metadata_cache_tti = "5m"
|
|
|
|
## The WAL options.
|
|
[wal]
|
|
## The provider of the WAL.
|
|
## - `raft_engine`: the wal is stored in the local file system by raft-engine.
|
|
## - `kafka`: it's remote wal that data is stored in Kafka.
|
|
## - `noop`: it's a no-op WAL provider that does not store any WAL data.<br/>**Notes: any unflushed data will be lost when the datanode is shutdown.**
|
|
## - `experimental_object_store`: the wal is stored as objects in an object store.<br/>**Notes: experimental and not supported yet.**
|
|
provider = "raft_engine"
|
|
|
|
## The directory to store the WAL files.
|
|
## **It's only used when the provider is `raft_engine`**.
|
|
## @toml2docs:none-default
|
|
dir = "./greptimedb_data/wal"
|
|
|
|
## The size of the WAL segment file.
|
|
## **It's only used when the provider is `raft_engine`**.
|
|
file_size = "128MB"
|
|
|
|
## The threshold of the WAL size to trigger a purge.
|
|
## **It's only used when the provider is `raft_engine`**.
|
|
purge_threshold = "1GB"
|
|
|
|
## The interval to trigger a purge.
|
|
## **It's only used when the provider is `raft_engine`**.
|
|
purge_interval = "1m"
|
|
|
|
## The read batch size.
|
|
## **It's only used when the provider is `raft_engine`**.
|
|
read_batch_size = 128
|
|
|
|
## Whether to use sync write.
|
|
## **It's only used when the provider is `raft_engine`**.
|
|
sync_write = false
|
|
|
|
## Whether to reuse logically truncated log files.
|
|
## **It's only used when the provider is `raft_engine`**.
|
|
enable_log_recycle = true
|
|
|
|
## Whether to pre-create log files on start up.
|
|
## **It's only used when the provider is `raft_engine`**.
|
|
prefill_log_files = false
|
|
|
|
## Duration for fsyncing log files.
|
|
## **It's only used when the provider is `raft_engine`**.
|
|
sync_period = "5s"
|
|
|
|
## Parallelism during WAL recovery.
|
|
recovery_parallelism = 2
|
|
|
|
## The Kafka broker endpoints.
|
|
## **It's only used when the provider is `kafka`**.
|
|
broker_endpoints = ["127.0.0.1:9092"]
|
|
|
|
## The connect timeout for kafka client.
|
|
## **It's only used when the provider is `kafka`**.
|
|
#+ connect_timeout = "3s"
|
|
|
|
## The total request timeout for kafka client.
|
|
## **It's only used when the provider is `kafka`**.
|
|
#+ timeout = "5s"
|
|
|
|
## The max size of a single producer batch.
|
|
## Warning: Kafka has a default limit of 1MB per message in a topic.
|
|
## Defaults to `8MB` when the provider is `experimental_object_store`.
|
|
## **It's only used when the provider is `kafka` or `experimental_object_store`**.
|
|
max_batch_bytes = "1MB"
|
|
|
|
## The consumer wait timeout.
|
|
## **It's only used when the provider is `kafka`**.
|
|
consumer_wait_timeout = "100ms"
|
|
|
|
## Whether to enable WAL index creation.
|
|
## **It's only used when the provider is `kafka`**.
|
|
create_index = false
|
|
|
|
## The interval for dumping WAL indexes.
|
|
## **It's only used when the provider is `kafka`**.
|
|
dump_index_interval = "60s"
|
|
|
|
## Ignore missing entries during read WAL.
|
|
## **It's only used when the provider is `kafka`**.
|
|
##
|
|
## This option ensures that when Kafka messages are deleted, the system
|
|
## can still successfully replay memtable data without throwing an
|
|
## out-of-range error.
|
|
## However, enabling this option might lead to unexpected data loss,
|
|
## as the system will skip over missing entries instead of treating
|
|
## them as critical errors.
|
|
overwrite_entry_start_id = false
|
|
|
|
## The name of the storage provider that holds the WAL objects, an empty name selects the default object store.
|
|
## **It's only used when the provider is `experimental_object_store`**.
|
|
#+ storage_provider = ""
|
|
|
|
## The path prefix of the WAL objects inside the storage provider.
|
|
## The objects are written under `<prefix>/datanodes/<node_id>/epochs/<generation>`, which is derived from this prefix.
|
|
## **It's only used when the provider is `experimental_object_store`**.
|
|
#+ prefix = "wal"
|
|
|
|
## The interval of flushing buffered entries to the object store, at least `10ms`, defaults to `100ms`.
|
|
## Each non-empty timer-triggered flush creates one object, and a batch that reaches `max_batch_bytes` is flushed immediately, so under sustained load the batch seals on size and the interval no longer matters.
|
|
## When writes are sparse, a shorter interval lowers the acknowledgement latency of `durable` appends and raises the number of object requests: timer-triggered sealing creates at most one object per interval per node, and in `enqueued` mode the backlog thresholds can seal earlier. An `enqueued` append does not wait for its batch to seal once it is admitted, but the backlog thresholds can delay admission.
|
|
## **It's only used when the provider is `experimental_object_store`**.
|
|
#+ flush_interval = "100ms"
|
|
|
|
## When an append to the object store WAL returns.
|
|
## - `durable`: an append returns after the object holding its entries is durable (the default).
|
|
## - `enqueued`: an append returns once its entries are admitted and their ids are assigned, and the object is created in the background. A crash loses the unpersisted backlog.
|
|
## **It's only used when the provider is `experimental_object_store`**.
|
|
#+ ack_mode = "durable"
|
|
|
|
## The size of the unpersisted backlog at which new appends stall until an upload completes, in `enqueued` mode.
|
|
## Nothing is dropped and nothing is rejected; the threshold does not bound what a crash during an outage can lose.
|
|
## **It's only used when the provider is `experimental_object_store` and `ack_mode` is `enqueued`**.
|
|
#+ max_unpersisted_bytes = "64MB"
|
|
|
|
## The age of the oldest unpersisted entry at which new appends stall until an upload completes, in `enqueued` mode.
|
|
## **It's only used when the provider is `experimental_object_store` and `ack_mode` is `enqueued`**.
|
|
#+ max_unpersisted_age = "8s"
|
|
|
|
## What a read does with a segment that still does not decode after a second fetch, because its checksum does not match or its content disagrees with its footer entry.
|
|
## - `skip`: the segment is skipped and recorded as a WAL hole of its region, a metric is incremented and a warning is logged; the other regions of the object are unaffected (the default).
|
|
## - `fail`: the read fails, so the region does not open.
|
|
## **It's only used when the provider is `experimental_object_store`**.
|
|
#+ on_corrupted_segment = "skip"
|
|
|
|
# The Kafka SASL configuration.
|
|
# **It's only used when the provider is `kafka`**.
|
|
# Available SASL mechanisms:
|
|
# - `PLAIN`
|
|
# - `SCRAM-SHA-256`
|
|
# - `SCRAM-SHA-512`
|
|
# [wal.sasl]
|
|
# type = "SCRAM-SHA-512"
|
|
# username = "user_kafka"
|
|
# password = "secret"
|
|
|
|
# The Kafka TLS configuration.
|
|
# **It's only used when the provider is `kafka`**.
|
|
# [wal.tls]
|
|
# server_ca_cert_path = "/path/to/server_cert"
|
|
# client_cert_path = "/path/to/client_cert"
|
|
# client_key_path = "/path/to/key"
|
|
|
|
# Example of using S3 as the storage.
|
|
# [storage]
|
|
# type = "S3"
|
|
# bucket = "greptimedb"
|
|
# root = "data"
|
|
# access_key_id = "test"
|
|
# secret_access_key = "123456"
|
|
# endpoint = "https://s3.amazonaws.com"
|
|
# region = "us-west-2"
|
|
# enable_virtual_host_style = false
|
|
# disable_ec2_metadata = false
|
|
|
|
# Example of using Oss as the storage.
|
|
# [storage]
|
|
# type = "Oss"
|
|
# bucket = "greptimedb"
|
|
# root = "data"
|
|
# access_key_id = "test"
|
|
# access_key_secret = "123456"
|
|
# endpoint = "https://oss-cn-hangzhou.aliyuncs.com"
|
|
|
|
# Example of using Azblob as the storage.
|
|
# [storage]
|
|
# type = "Azblob"
|
|
# container = "greptimedb"
|
|
# root = "data"
|
|
# account_name = "test"
|
|
# account_key = "123456"
|
|
# endpoint = "https://greptimedb.blob.core.windows.net"
|
|
# sas_token = ""
|
|
|
|
# Example of using Gcs as the storage.
|
|
# [storage]
|
|
# type = "Gcs"
|
|
# bucket = "greptimedb"
|
|
# root = "data"
|
|
# scope = "test"
|
|
# credential_path = "123456"
|
|
# credential = "base64-credential"
|
|
# endpoint = "https://storage.googleapis.com"
|
|
|
|
# Example of using HDFS as the storage with the native Rust client.
|
|
# [storage]
|
|
# type = "Hdfs"
|
|
# root = "/greptimedb"
|
|
# name_node = "hdfs://127.0.0.1:9000"
|
|
# options = { "dfs.client.block.write.replace-datanode-on-failure.enable" = "true" }
|
|
|
|
## The query engine options.
|
|
[query]
|
|
## Parallelism of the query engine.
|
|
## Default to 0, which means the number of CPU cores.
|
|
parallelism = 0
|
|
|
|
## Memory pool size for query execution operators (aggregation, sorting, join).
|
|
## Supports absolute size (e.g., "2GB", "4GB") or percentage of system memory (e.g., "20%").
|
|
## Setting it to 0 disables the limit (unbounded, default behavior).
|
|
## When this limit is reached, queries will fail with ResourceExhausted error.
|
|
## NOTE: This does NOT limit memory used by table scans.
|
|
memory_pool_size = "50%"
|
|
|
|
## Experimental memory pool allocation policy:
|
|
## - "greedy" (default): first-come-first-served allocation; preserves current behavior.
|
|
## - "fair": divides available memory among spillable operators and may spill earlier.
|
|
## Only effective when `memory_pool_size` is bounded (>0).
|
|
#+ experimental_memory_pool_policy = "greedy"
|
|
|
|
# --- Experimental: DataFusion spill-to-disk controls ---
|
|
## Spill mode:
|
|
## - "default": preserve DataFusion built-in OS temp directory (default).
|
|
## - "custom": explicitly configure spill path, quota, and compression.
|
|
## - "disabled": explicitly disable disk spilling.
|
|
## Set this to "custom" before using the path/quota/compression keys below.
|
|
#+ experimental_spill_mode = "default"
|
|
## Spill directory path. Ignored unless mode is "custom".
|
|
## @toml2docs:none-default
|
|
#+ experimental_spill_path = "/path/to/spill"
|
|
## Maximum total size of spill directory (default: "1GiB").
|
|
## Ignored unless mode is "custom".
|
|
#+ experimental_spill_max_temp_directory_size = "1GiB"
|
|
## Compression for spilled data files: "uncompressed" (default), "lz4_frame", "zstd".
|
|
## Ignored unless mode is "custom".
|
|
#+ experimental_spill_compression = "uncompressed"
|
|
# --- End spill-to-disk ---
|
|
|
|
## The data storage options.
|
|
[storage]
|
|
## The working home directory.
|
|
data_home = "./greptimedb_data"
|
|
|
|
## The storage type used to store the data.
|
|
## - `File`: the data is stored in the local file system.
|
|
## - `S3`: the data is stored in the S3 object storage.
|
|
## - `Gcs`: the data is stored in the Google Cloud Storage.
|
|
## - `Azblob`: the data is stored in the Azure Blob Storage.
|
|
## - `Oss`: the data is stored in the Aliyun OSS.
|
|
## - `Hdfs`: the data is stored in the Hadoop Distributed File System.
|
|
type = "File"
|
|
|
|
## The S3 bucket name.
|
|
## **It's only used when the storage type is `S3`, `Oss` and `Gcs`**.
|
|
## @toml2docs:none-default
|
|
bucket = "greptimedb"
|
|
|
|
## The directory or object prefix under which data is stored.
|
|
## **It's only used when the storage type is `S3`, `Oss`, `Gcs`, `Azblob` and `Hdfs`**.
|
|
## @toml2docs:none-default
|
|
root = "greptimedb"
|
|
|
|
## The HDFS NameNode URI, for example, `hdfs://127.0.0.1:9000`.
|
|
## **It's only used when the storage type is `Hdfs`**.
|
|
## @toml2docs:none-default
|
|
name_node = "hdfs://127.0.0.1:9000"
|
|
|
|
## Additional options passed to the native HDFS client.
|
|
## **It's only used when the storage type is `Hdfs`**.
|
|
## @toml2docs:none-default
|
|
options = {}
|
|
|
|
## The access key id of the aws account.
|
|
## It's **highly recommended** to use AWS IAM roles instead of hardcoding the access key id and secret key.
|
|
## **It's only used when the storage type is `S3` and `Oss`**.
|
|
## @toml2docs:none-default
|
|
access_key_id = "test"
|
|
|
|
## The secret access key of the aws account.
|
|
## It's **highly recommended** to use AWS IAM roles instead of hardcoding the access key id and secret key.
|
|
## **It's only used when the storage type is `S3`**.
|
|
## @toml2docs:none-default
|
|
secret_access_key = "test"
|
|
|
|
## The secret access key of the aliyun account.
|
|
## **It's only used when the storage type is `Oss`**.
|
|
## @toml2docs:none-default
|
|
access_key_secret = "test"
|
|
|
|
## The account key of the azure account.
|
|
## **It's only used when the storage type is `Azblob`**.
|
|
## @toml2docs:none-default
|
|
account_name = "test"
|
|
|
|
## The account key of the azure account.
|
|
## **It's only used when the storage type is `Azblob`**.
|
|
## @toml2docs:none-default
|
|
account_key = "test"
|
|
|
|
## The scope of the google cloud storage.
|
|
## **It's only used when the storage type is `Gcs`**.
|
|
## @toml2docs:none-default
|
|
scope = "test"
|
|
|
|
## The credential path of the google cloud storage.
|
|
## **It's only used when the storage type is `Gcs`**.
|
|
## @toml2docs:none-default
|
|
credential_path = "test"
|
|
|
|
## The credential of the google cloud storage.
|
|
## **It's only used when the storage type is `Gcs`**.
|
|
## @toml2docs:none-default
|
|
credential = "base64-credential"
|
|
|
|
## The container of the azure account.
|
|
## **It's only used when the storage type is `Azblob`**.
|
|
## @toml2docs:none-default
|
|
container = "greptimedb"
|
|
|
|
## The sas token of the azure account.
|
|
## **It's only used when the storage type is `Azblob`**.
|
|
## @toml2docs:none-default
|
|
sas_token = ""
|
|
|
|
## The endpoint of the S3 service.
|
|
## **It's only used when the storage type is `S3`, `Oss`, `Gcs` and `Azblob`**.
|
|
## @toml2docs:none-default
|
|
endpoint = "https://s3.amazonaws.com"
|
|
|
|
## The region of the S3 service.
|
|
## **It's only used when the storage type is `S3`, `Oss`, `Gcs` and `Azblob`**.
|
|
## @toml2docs:none-default
|
|
region = "us-west-2"
|
|
|
|
## The http client options to the storage.
|
|
## **It's only used when the storage type is `S3`, `Oss`, `Gcs` and `Azblob`**.
|
|
[storage.http_client]
|
|
|
|
## The maximum idle connection per host allowed in the pool.
|
|
pool_max_idle_per_host = 1024
|
|
|
|
## The timeout for only the connect phase of a http client.
|
|
connect_timeout = "30s"
|
|
|
|
## The total request timeout, applied from when the request starts connecting until the response body has finished.
|
|
## Also considered a total deadline.
|
|
timeout = "30s"
|
|
|
|
## The timeout for idle sockets being kept-alive.
|
|
pool_idle_timeout = "90s"
|
|
|
|
## To skip the ssl verification
|
|
## **Security Notice**: Setting `skip_ssl_validation = true` disables certificate verification, making connections vulnerable to man-in-the-middle attacks. Only use this in development or trusted private networks.
|
|
skip_ssl_validation = false
|
|
|
|
# Custom storage options
|
|
# [[storage.providers]]
|
|
# name = "S3"
|
|
# type = "S3"
|
|
# bucket = "greptimedb"
|
|
# root = "data"
|
|
# access_key_id = "test"
|
|
# secret_access_key = "123456"
|
|
# endpoint = "https://s3.amazonaws.com"
|
|
# region = "us-west-2"
|
|
# [[storage.providers]]
|
|
# name = "Gcs"
|
|
# type = "Gcs"
|
|
# bucket = "greptimedb"
|
|
# root = "data"
|
|
# scope = "test"
|
|
# credential_path = "123456"
|
|
# credential = "base64-credential"
|
|
# endpoint = "https://storage.googleapis.com"
|
|
|
|
## The region engine options. You can configure multiple region engines.
|
|
## Each engine type (mito, file, metric) may appear only once; duplicates cause startup to fail.
|
|
[[region_engine]]
|
|
|
|
## The Mito engine options.
|
|
[region_engine.mito]
|
|
|
|
## Number of region workers.
|
|
#+ num_workers = 8
|
|
|
|
## Request channel size of each worker.
|
|
worker_channel_size = 128
|
|
|
|
## Max batch size for a worker to handle requests.
|
|
worker_request_batch_size = 64
|
|
|
|
## Number of meta action updated to trigger a new checkpoint for the manifest.
|
|
manifest_checkpoint_distance = 10
|
|
|
|
|
|
## Number of removed files to keep in manifest's `removed_files` field before also
|
|
## remove them from `removed_files`. Mostly for debugging purpose.
|
|
## If set to 0, it will only use `keep_removed_file_ttl` to decide when to remove files
|
|
## from `removed_files` field.
|
|
experimental_manifest_keep_removed_file_count = 256
|
|
|
|
## How long to keep removed files in the `removed_files` field of manifest
|
|
## after they are removed from manifest.
|
|
## files will only be removed from `removed_files` field
|
|
## if both `keep_removed_file_count` and `keep_removed_file_ttl` is reached.
|
|
experimental_manifest_keep_removed_file_ttl = "1h"
|
|
|
|
## Whether to compress manifest and checkpoint file by gzip (default false).
|
|
compress_manifest = false
|
|
|
|
## Under development; do not enable. Whether to enable series indexes.
|
|
## Indexes are stored on the local filesystem under `{data_home}/series_index`.
|
|
#+ experimental_enable_series_index = false
|
|
|
|
## Approximate series and range index size limit in open regions, shared across workers.
|
|
## Workers periodically refresh usage and skip maintenance when full. In-flight reconciliation
|
|
## can exceed the limit. Closed-region files, temporary output, catalogs, and old snapshots
|
|
## retained by readers are not counted.
|
|
## Minimum: 1KiB. Takes effect on restart.
|
|
#+ experimental_series_index_max_size = "5GiB"
|
|
|
|
## Whether to build and query range indexes when series indexes are enabled.
|
|
## Obsolete range-index metadata and files are still cleaned up when disabled.
|
|
#+ experimental_enable_range_index = false
|
|
|
|
## Interval between series-index maintenance runs. Zero uses the default of 5 min.
|
|
#+ experimental_series_index_maintenance_interval = "5m"
|
|
|
|
## Requested minimum series-index bucket width (default: 5 days), rounded up to
|
|
## an exact multiple of each region's compaction time window.
|
|
#+ experimental_series_index_bucket_width = "5days"
|
|
|
|
## Max number of running background flush jobs (default: 1/2 of cpu cores).
|
|
## @toml2docs:none-default="Auto"
|
|
#+ max_background_flushes = 4
|
|
|
|
## Max number of running background compaction jobs (default: 1/4 of cpu cores).
|
|
## @toml2docs:none-default="Auto"
|
|
#+ max_background_compactions = 2
|
|
|
|
## Max number of running background purge jobs (default: number of cpu cores).
|
|
## @toml2docs:none-default="Auto"
|
|
#+ max_background_purges = 8
|
|
|
|
## Memory budget for compaction tasks.
|
|
## Supports absolute size (e.g., "2GiB", "512MB") or percentage of system memory (e.g., "50%").
|
|
## Setting it to 0 or "unlimited" disables the limit.
|
|
## @toml2docs:none-default="0"
|
|
#+ experimental_compaction_memory_limit = "0"
|
|
|
|
## Behavior when compaction cannot acquire memory from the budget.
|
|
## Options: "wait" (default, 10s), "wait(<duration>)", "fail"
|
|
## @toml2docs:none-default="wait"
|
|
#+ experimental_compaction_on_exhausted = "wait"
|
|
|
|
## Interval to auto flush a region if it has not flushed yet.
|
|
auto_flush_interval = "10m"
|
|
|
|
## Global write buffer size for all regions. If not set, it's default to 1/8 of OS memory with a max limitation of 1GB.
|
|
## @toml2docs:none-default="Auto"
|
|
#+ global_write_buffer_size = "1GB"
|
|
|
|
## Global write buffer size threshold to reject write requests. If not set, it's default to 2 times of `global_write_buffer_size`
|
|
## @toml2docs:none-default="Auto"
|
|
#+ global_write_buffer_reject_size = "2GB"
|
|
|
|
## Default write buffer size for each region. Regions stall at this size and reject writes at twice this size. Setting it to 0 disables both limits unless the table specifies `write_buffer_size`.
|
|
#+ default_region_write_buffer_size = "0"
|
|
|
|
## Cache size for SST metadata. Setting it to 0 to disable the cache.
|
|
## If not set, it's default to 1/8 of OS memory with a max limitation of 512MB.
|
|
## @toml2docs:none-default="Auto"
|
|
#+ sst_meta_cache_size = "512MB"
|
|
|
|
## Cache size for vectors and arrow arrays. Setting it to 0 to disable the cache.
|
|
## If not set, it's default to 1/16 of OS memory with a max limitation of 512MB.
|
|
## @toml2docs:none-default="Auto"
|
|
#+ vector_cache_size = "512MB"
|
|
|
|
## Cache size for pages of SST row groups. Setting it to 0 to disable the cache.
|
|
## If not set, it's default to 1/8 of OS memory.
|
|
## @toml2docs:none-default="Auto"
|
|
#+ page_cache_size = "512MB"
|
|
|
|
## Cache size for time series selector (e.g. `last_value()`). Setting it to 0 to disable the cache.
|
|
## If not set, it's default to 1/16 of OS memory with a max limitation of 512MB.
|
|
## @toml2docs:none-default="Auto"
|
|
#+ selector_result_cache_size = "512MB"
|
|
|
|
## Cache size for flat range scan results. Setting it to 0 to disable the cache.
|
|
## If not set, it's default to 1/16 of OS memory with a max limitation of 512MB.
|
|
## @toml2docs:none-default="Auto"
|
|
#+ range_result_cache_size = "512MB"
|
|
|
|
## Cache size for prefilter results. Setting it to 0 to disable the cache.
|
|
## If not set, it's default to 1/32 of OS memory with a max limitation of 128MB.
|
|
## @toml2docs:none-default="Auto"
|
|
#+ prefilter_result_cache_size = "128MB"
|
|
|
|
## Whether to enable the write cache, it's enabled by default when using object storage. It is recommended to enable it when using object storage for better performance.
|
|
enable_write_cache = false
|
|
|
|
## File system path for write cache, defaults to `{data_home}`.
|
|
write_cache_path = ""
|
|
|
|
## Capacity for write cache. If your disk space is sufficient, it is recommended to set it larger.
|
|
write_cache_size = "5GiB"
|
|
|
|
## TTL for write cache.
|
|
## @toml2docs:none-default
|
|
write_cache_ttl = "8h"
|
|
|
|
## Preload index (puffin) files into cache on region open (default: true).
|
|
## When enabled, index files are loaded into the write cache during region initialization,
|
|
## which can improve query performance at the cost of longer startup times.
|
|
preload_index_cache = true
|
|
|
|
## Percentage of write cache capacity allocated for index (puffin) files (default: 20).
|
|
## The remaining capacity is used for data (parquet) files.
|
|
## Must be between 0 and 100 (exclusive). For example, with a 5GiB write cache and 20% allocation,
|
|
## 1GiB is reserved for index files and 4GiB for data files.
|
|
index_cache_percent = 20
|
|
|
|
## Enable refilling cache on read operations (default: true).
|
|
## When disabled, cache refilling on read won't happen.
|
|
enable_refill_cache_on_read = true
|
|
|
|
## Capacity for manifest cache (default: 256MB).
|
|
manifest_cache_size = "256MB"
|
|
|
|
## Buffer size for SST writing.
|
|
sst_write_buffer_size = "8MB"
|
|
|
|
## Maximum number of SST files to scan concurrently.
|
|
max_concurrent_scan_files = 384
|
|
|
|
## Whether to allow stale WAL entries read during replay.
|
|
allow_stale_entries = false
|
|
|
|
## Memory limit for table scans across all queries.
|
|
## Supports absolute size (e.g., "2GB") or percentage of system memory (e.g., "20%").
|
|
## Setting it to 0 or "unlimited" disables the limit.
|
|
scan_memory_limit = "unlimited"
|
|
## Controls what happens when a scan cannot get memory immediately.
|
|
## "fail" (default) fails fast and is the recommended option for most users.
|
|
## "wait" / "wait(<duration>)" waits for memory to become available. This is mainly
|
|
## for advanced tuning in bursty workloads where temporary contention is common and
|
|
## higher latency is acceptable.
|
|
## "wait" means "wait(10s)", not unlimited waiting.
|
|
scan_memory_on_exhausted = "fail"
|
|
|
|
## Minimum time interval between two compactions.
|
|
## To align with the old behavior, the default value is 0 (no restrictions).
|
|
min_compaction_interval = "0m"
|
|
|
|
## Whether to allow to schedule a compaction after a successful region edit.
|
|
##
|
|
## Setting this to "true" is a necessary but not sufficient condition for scheduling compaction after a region edit.
|
|
## Other constraints, such as "min_compaction_interval", may still prevent compaction from being scheduled.
|
|
## Setting this to "false", however, guarantees that compaction will not be scheduled after a region edit.
|
|
schedule_compaction_after_edit = true
|
|
|
|
## Whether to enable flat format as the default SST format.
|
|
default_flat_format = true
|
|
|
|
## Whether to enable the experimental two-phase mode for eligible metric series scans.
|
|
experimental_series_scan_v2 = true
|
|
|
|
## The options for index in Mito engine.
|
|
[region_engine.mito.index]
|
|
|
|
## Auxiliary directory path for the index in filesystem, used to store intermediate files for
|
|
## creating the index and staging files for searching the index, defaults to `{data_home}/index_intermediate`.
|
|
## The default name for this directory is `index_intermediate` for backward compatibility.
|
|
##
|
|
## This path contains two subdirectories:
|
|
## - `__intm`: for storing intermediate files used during creating index.
|
|
## - `staging`: for storing staging files used during searching index.
|
|
aux_path = ""
|
|
|
|
## The max capacity of the staging directory.
|
|
staging_size = "2GB"
|
|
|
|
## The TTL of the staging directory.
|
|
## Defaults to 7 days.
|
|
## Setting it to "0s" to disable TTL.
|
|
staging_ttl = "7d"
|
|
|
|
## Cache size for inverted index metadata.
|
|
metadata_cache_size = "64MiB"
|
|
|
|
## Cache size for inverted index content.
|
|
content_cache_size = "128MiB"
|
|
|
|
## Page size for inverted index content cache.
|
|
content_cache_page_size = "64KiB"
|
|
|
|
## Cache size for index result.
|
|
result_cache_size = "128MiB"
|
|
|
|
## The options for inverted index in Mito engine.
|
|
[region_engine.mito.inverted_index]
|
|
|
|
## Whether to create the index on flush.
|
|
## - `auto`: automatically (default)
|
|
## - `disable`: never
|
|
create_on_flush = "auto"
|
|
|
|
## Whether to create the index on compaction.
|
|
## - `auto`: automatically (default)
|
|
## - `disable`: never
|
|
create_on_compaction = "auto"
|
|
|
|
## Whether to apply the index on query
|
|
## - `auto`: automatically (default)
|
|
## - `disable`: never
|
|
apply_on_query = "auto"
|
|
|
|
## Memory threshold for performing an external sort during index creation.
|
|
## - `auto`: automatically determine the threshold based on the system memory size (default)
|
|
## - `unlimited`: no memory limit
|
|
## - `[size]` e.g. `64MB`: fixed memory threshold
|
|
mem_threshold_on_create = "auto"
|
|
|
|
## Deprecated, use `region_engine.mito.index.aux_path` instead.
|
|
intermediate_path = ""
|
|
|
|
## The options for full-text index in Mito engine.
|
|
[region_engine.mito.fulltext_index]
|
|
|
|
## Whether to create the index on flush.
|
|
## - `auto`: automatically (default)
|
|
## - `disable`: never
|
|
create_on_flush = "auto"
|
|
|
|
## Whether to create the index on compaction.
|
|
## - `auto`: automatically (default)
|
|
## - `disable`: never
|
|
create_on_compaction = "auto"
|
|
|
|
## Whether to apply the index on query
|
|
## - `auto`: automatically (default)
|
|
## - `disable`: never
|
|
apply_on_query = "auto"
|
|
|
|
## Memory threshold for index creation.
|
|
## - `auto`: automatically determine the threshold based on the system memory size (default)
|
|
## - `unlimited`: no memory limit
|
|
## - `[size]` e.g. `64MB`: fixed memory threshold
|
|
mem_threshold_on_create = "auto"
|
|
|
|
## The options for bloom filter index in Mito engine.
|
|
[region_engine.mito.bloom_filter_index]
|
|
|
|
## Whether to create the index on flush.
|
|
## - `auto`: automatically (default)
|
|
## - `disable`: never
|
|
create_on_flush = "auto"
|
|
|
|
## Whether to create the index on compaction.
|
|
## - `auto`: automatically (default)
|
|
## - `disable`: never
|
|
create_on_compaction = "auto"
|
|
|
|
## Whether to apply the index on query
|
|
## - `auto`: automatically (default)
|
|
## - `disable`: never
|
|
apply_on_query = "auto"
|
|
|
|
## Memory threshold for the index creation.
|
|
## - `auto`: automatically determine the threshold based on the system memory size (default)
|
|
## - `unlimited`: no memory limit
|
|
## - `[size]` e.g. `64MB`: fixed memory threshold
|
|
mem_threshold_on_create = "auto"
|
|
|
|
[region_engine.mito.gc]
|
|
## Whether GC is enabled. Need to be the same with metasrv's `gc.enable` or unexpected behavior will occur
|
|
enable = false
|
|
|
|
## Lingering time before deleting files.
|
|
## Should be long enough to allow long running queries to finish.
|
|
## If set to None, then unused files will be deleted immediately.
|
|
lingering_time = "1h"
|
|
|
|
## Lingering time before deleting unknown files (files with undetermined expel time).
|
|
## Only applies during full file listing GC.
|
|
## This uses the object's last-modified timestamp as a heuristic (strict less-than comparison);
|
|
## do not configure this value too small in production to avoid deleting pre-manifest files
|
|
## from in-progress compaction or flush.
|
|
## For active/open regions, an unknown file is deleted only if its object last-modified time exceeds this TTL.
|
|
## If the object store does not provide a last-modified timestamp, the file is conservatively kept.
|
|
## For dropped regions, unknown files are deleted immediately.
|
|
unknown_file_lingering_time = "1d"
|
|
|
|
[[region_engine]]
|
|
## Enable the file engine.
|
|
[region_engine.file]
|
|
|
|
[[region_engine]]
|
|
## Metric engine options.
|
|
[region_engine.metric]
|
|
|
|
## The logging options.
|
|
[logging]
|
|
## The directory to store the log files. If set to empty, logs will not be written to files.
|
|
dir = "./greptimedb_data/logs"
|
|
|
|
## The log level. Can be `info`/`debug`/`warn`/`error`.
|
|
## @toml2docs:none-default
|
|
level = "info"
|
|
|
|
## Enable OTLP tracing.
|
|
enable_otlp_tracing = false
|
|
|
|
## The OTLP tracing endpoint.
|
|
otlp_endpoint = "http://localhost:4318/v1/traces"
|
|
|
|
## Whether to append logs to stdout.
|
|
append_stdout = true
|
|
|
|
## Whether to write logs to files in `dir`.
|
|
enable_file_logging = true
|
|
|
|
## The log format. Can be `text`/`json`.
|
|
log_format = "text"
|
|
|
|
## The maximum amount of log files.
|
|
max_log_files = 720
|
|
|
|
## The maximum total size of managed log files in `dir`.
|
|
## Old closed log files are removed before writing when necessary. Active files may exceed it. Set to `0B` to disable.
|
|
max_log_dir_size = "0B"
|
|
|
|
## The OTLP tracing export protocol. Can be `grpc`/`http`.
|
|
otlp_export_protocol = "http"
|
|
|
|
## Additional OTLP headers, only valid when using OTLP http
|
|
[logging.otlp_headers]
|
|
## @toml2docs:none-default
|
|
#Authorization = "Bearer my-token"
|
|
## @toml2docs:none-default
|
|
#Database = "My database"
|
|
|
|
## The percentage of tracing will be sampled and exported.
|
|
## Valid range `[0, 1]`, 1 means all traces are sampled, 0 means all traces are not sampled, the default value is 1.
|
|
## ratio > 1 are treated as 1. Fractions < 0 are treated as 0
|
|
[logging.tracing_sample_ratio]
|
|
default_ratio = 1.0
|
|
|
|
## The tracing options. Only effect when compiled with `tokio-console` feature.
|
|
#+ [tracing]
|
|
## The tokio console address.
|
|
## @toml2docs:none-default
|
|
#+ tokio_console_addr = "127.0.0.1"
|
|
|
|
## The memory options.
|
|
[memory]
|
|
## Whether to enable heap profiling activation during startup.
|
|
## When enabled, heap profiling will be activated if the `MALLOC_CONF` environment variable
|
|
## is set to "prof:true,prof_active:false". The official image adds this env variable.
|
|
## Default is true.
|
|
enable_heap_profiling = true
|