chore: Revert "feat: prune in each partition"

This reverts commit 3f9bf48161.
feat: tokio dump
2025-12-23 06:30:05 +00:00 · 2024-11-08 23:57:18 +08:00 · 2024-11-08 23:08:47 +08:00 · 2024-11-08 21:31:03 +08:00 · 2024-11-08 20:35:40 +08:00 · 2024-11-08 19:09:08 +08:00
706 changed files with 38431 additions and 13676 deletions
--- a/.github/actions/build-dev-builder-images/action.yml
+++ b/.github/actions/build-dev-builder-images/action.yml
@@ -50,7 +50,7 @@ runs:
          BUILDX_MULTI_PLATFORM_BUILD=all \
          IMAGE_REGISTRY=${{ inputs.dockerhub-image-registry }} \
          IMAGE_NAMESPACE=${{ inputs.dockerhub-image-namespace }} \
-          IMAGE_TAG=${{ inputs.version }}
+          DEV_BUILDER_IMAGE_TAG=${{ inputs.version }}

    - name: Build and push dev-builder-centos image
      shell: bash
@@ -61,7 +61,7 @@ runs:
          BUILDX_MULTI_PLATFORM_BUILD=amd64 \
          IMAGE_REGISTRY=${{ inputs.dockerhub-image-registry }} \
          IMAGE_NAMESPACE=${{ inputs.dockerhub-image-namespace }} \
-          IMAGE_TAG=${{ inputs.version }}
+          DEV_BUILDER_IMAGE_TAG=${{ inputs.version }}

    - name: Build and push dev-builder-android image # Only build image for amd64 platform.
      shell: bash
@@ -71,6 +71,6 @@ runs:
          BASE_IMAGE=android \
          IMAGE_REGISTRY=${{ inputs.dockerhub-image-registry }} \
          IMAGE_NAMESPACE=${{ inputs.dockerhub-image-namespace }} \
-          IMAGE_TAG=${{ inputs.version }} && \
+          DEV_BUILDER_IMAGE_TAG=${{ inputs.version }} && \

        docker push ${{ inputs.dockerhub-image-registry }}/${{ inputs.dockerhub-image-namespace }}/dev-builder-android:${{ inputs.version }}
--- a/.github/actions/build-windows-artifacts/action.yml
+++ b/.github/actions/build-windows-artifacts/action.yml
@@ -40,7 +40,7 @@ runs:

    - name: Install PyArrow Package
      shell: pwsh
-      run: pip install pyarrow
+      run: pip install pyarrow numpy

    - name: Install WSL distribution
      uses: Vampire/setup-wsl@v2
--- a/.github/actions/setup-etcd-cluster/action.yml
+++ b/.github/actions/setup-etcd-cluster/action.yml
@@ -18,7 +18,7 @@ runs:
        --set replicaCount=${{ inputs.etcd-replicas }} \
        --set resources.requests.cpu=50m \
        --set resources.requests.memory=128Mi \
-        --set resources.limits.cpu=1000m \
+        --set resources.limits.cpu=1500m \
        --set resources.limits.memory=2Gi \
        --set auth.rbac.create=false \
        --set auth.rbac.token.enabled=false \
--- a/.github/workflows/develop.yml
+++ b/.github/workflows/develop.yml
@@ -269,6 +269,13 @@ jobs:
      - name: Install cargo-gc-bin
        shell: bash
        run: cargo install cargo-gc-bin
+      - name: Check aws-lc-sys will not build
+        shell: bash
+        run: |
+             if cargo tree -i aws-lc-sys -e features | grep -q aws-lc-sys; then
+               echo "Found aws-lc-sys, which has compilation problems on older gcc versions. Please replace it with ring until its building experience improves."
+               exit 1
+             fi
      - name: Build greptime bianry
        shell: bash
        # `cargo gc` will invoke `cargo build` with specified args
@@ -429,12 +436,25 @@ jobs:
    timeout-minutes: 60
    strategy:
      matrix:
-        target: ["fuzz_migrate_mito_regions", "fuzz_failover_mito_regions", "fuzz_failover_metric_regions"]
+        target: ["fuzz_migrate_mito_regions", "fuzz_migrate_metric_regions", "fuzz_failover_mito_regions", "fuzz_failover_metric_regions"]
        mode:
          - name: "Remote WAL"
            minio: true
            kafka: true
            values: "with-remote-wal.yaml"
+        include:
+          - target: "fuzz_migrate_mito_regions"
+            mode:
+              name: "Local WAL"
+              minio: true
+              kafka: false
+              values: "with-minio.yaml"
+          - target: "fuzz_migrate_metric_regions"
+            mode:
+              name: "Local WAL"
+              minio: true
+              kafka: false
+              values: "with-minio.yaml"
    steps:
      - name: Remove unused software
        run: |
@@ -523,7 +543,7 @@ jobs:
        with:
          image-registry: localhost:5001
          values-filename: ${{ matrix.mode.values }}
-          enable-region-failover: true
+          enable-region-failover: ${{ matrix.mode.kafka }}
      - name: Port forward (mysql)
        run: |
          kubectl port-forward service/my-greptimedb-frontend 4002:4002 -n my-greptimedb&
@@ -674,7 +694,7 @@ jobs:
        with:
          python-version: '3.10'
      - name: Install PyArrow Package
-        run: pip install pyarrow
+        run: pip install pyarrow numpy
      - name: Setup etcd server
        working-directory: tests-integration/fixtures/etcd
        run: docker compose -f docker-compose-standalone.yml up -d --wait
--- a/.github/workflows/nightly-ci.yml
+++ b/.github/workflows/nightly-ci.yml
@@ -92,7 +92,7 @@ jobs:
        with:
          python-version: "3.10"
      - name: Install PyArrow Package
-        run: pip install pyarrow
+        run: pip install pyarrow numpy
      - name: Install WSL distribution
        uses: Vampire/setup-wsl@v2
        with:
--- a/Cargo.lock
+++ b/Cargo.lock
--- a/Cargo.toml
+++ b/Cargo.toml
@@ -20,6 +20,7 @@ members = [
    "src/common/mem-prof",
    "src/common/meta",
    "src/common/plugins",
+    "src/common/pprof",
    "src/common/procedure",
    "src/common/procedure-test",
    "src/common/query",
@@ -64,7 +65,7 @@ members = [
 resolver = "2"

 [workspace.package]
-version = "0.9.3"
+version = "0.9.5"
 edition = "2021"
 license = "Apache-2.0"

@@ -90,7 +91,7 @@ aquamarine = "0.3"
 arrow = { version = "51.0.0", features = ["prettyprint"] }
 arrow-array = { version = "51.0.0", default-features = false, features = ["chrono-tz"] }
 arrow-flight = "51.0"
-arrow-ipc = { version = "51.0.0", default-features = false, features = ["lz4"] }
+arrow-ipc = { version = "51.0.0", default-features = false, features = ["lz4", "zstd"] }
 arrow-schema = { version = "51.0", features = ["serde"] }
 async-stream = "0.3"
 async-trait = "0.1"
@@ -99,7 +100,7 @@ base64 = "0.21"
 bigdecimal = "0.4.2"
 bitflags = "2.4.1"
 bytemuck = "1.12"
-bytes = { version = "1.5", features = ["serde"] }
+bytes = { version = "1.7", features = ["serde"] }
 chrono = { version = "0.4", features = ["serde"] }
 clap = { version = "4.4", features = ["derive"] }
 config = "0.13.0"
@@ -120,12 +121,13 @@ etcd-client = { version = "0.13" }
 fst = "0.4.7"
 futures = "0.3"
 futures-util = "0.3"
-greptime-proto = { git = "https://github.com/GreptimeTeam/greptime-proto.git", rev = "c437b55725b7f5224fe9d46db21072b4a682ee4b" }
+greptime-proto = { git = "https://github.com/GreptimeTeam/greptime-proto.git", rev = "255f87a3318ace3f88a67f76995a0e14910983f4" }
 humantime = "2.1"
 humantime-serde = "1.1"
 itertools = "0.10"
+jsonb = { git = "https://github.com/databendlabs/jsonb.git", rev = "46ad50fc71cf75afbf98eec455f7892a6387c1fc", default-features = false }
 lazy_static = "1.4"
-meter-core = { git = "https://github.com/GreptimeTeam/greptime-meter.git", rev = "80eb97c24c88af4dd9a86f8bbaf50e741d4eb8cd" }
+meter-core = { git = "https://github.com/GreptimeTeam/greptime-meter.git", rev = "a10facb353b41460eeb98578868ebf19c2084fac" }
 mockall = "0.11.4"
 moka = "0.12"
 notify = "6.1"
@@ -135,15 +137,19 @@ opentelemetry-proto = { version = "0.5", features = [
    "gen-tonic",
    "metrics",
    "trace",
+    "with-serde",
+    "logs",
 ] }
+parking_lot = "0.12"
 parquet = { version = "51.0.0", default-features = false, features = ["arrow", "async", "object_store"] }
 paste = "1.0"
 pin-project = "1.0"
 prometheus = { version = "0.13.3", features = ["process"] }
-promql-parser = { version = "0.4" }
+promql-parser = { version = "0.4.3", features = ["ser"] }
 prost = "0.12"
 raft-engine = { version = "0.4.1", default-features = false }
 rand = "0.8"
+ratelimit = "0.9"
 regex = "1.8"
 regex-automata = { version = "0.4" }
 reqwest = { version = "0.12", default-features = false, features = [
@@ -163,7 +169,8 @@ schemars = "0.8"
 serde = { version = "1.0", features = ["derive"] }
 serde_json = { version = "1.0", features = ["float_roundtrip"] }
 serde_with = "3"
-shadow-rs = "0.31"
+shadow-rs = "0.35"
+similar-asserts = "1.6.0"
 smallvec = { version = "1", features = ["serde"] }
 snafu = "0.8"
 sysinfo = "0.30"
@@ -173,13 +180,16 @@ sqlparser = { git = "https://github.com/GreptimeTeam/sqlparser-rs.git", rev = "5
 ] }
 strum = { version = "0.25", features = ["derive"] }
 tempfile = "3"
-tokio = { version = "1.36", features = ["full"] }
+tokio = { version = "1.40", features = ["full"] }
 tokio-postgres = "0.7"
 tokio-stream = { version = "0.1" }
 tokio-util = { version = "0.7", features = ["io-util", "compat"] }
 toml = "0.8.8"
 tonic = { version = "0.11", features = ["tls", "gzip", "zstd"] }
 tower = { version = "0.4" }
+tracing-appender = "0.2"
+tracing-subscriber = { version = "0.3", features = ["env-filter", "json", "fmt"] }
+typetag = "0.2"
 uuid = { version = "1.7", features = ["serde", "v4", "fast-rng"] }
 zstd = "0.13"

@@ -205,6 +215,7 @@ common-macro = { path = "src/common/macro" }
 common-mem-prof = { path = "src/common/mem-prof" }
 common-meta = { path = "src/common/meta" }
 common-plugins = { path = "src/common/plugins" }
+common-pprof = { path = "src/common/pprof" }
 common-procedure = { path = "src/common/procedure" }
 common-procedure-test = { path = "src/common/procedure-test" }
 common-query = { path = "src/common/query" }
@@ -242,18 +253,29 @@ store-api = { path = "src/store-api" }
 substrait = { path = "src/common/substrait" }
 table = { path = "src/table" }

+[patch.crates-io]
+# change all rustls dependencies to use our fork to default to `ring` to make it "just work"
+hyper-rustls = { git = "https://github.com/GreptimeTeam/hyper-rustls" }
+rustls = { git = "https://github.com/GreptimeTeam/rustls" }
+tokio-rustls = { git = "https://github.com/GreptimeTeam/tokio-rustls" }
+# This is commented, since we are not using aws-lc-sys, if we need to use it, we need to uncomment this line or use a release after this commit, or it wouldn't compile with gcc < 8.1
+# see https://github.com/aws/aws-lc-rs/pull/526
+# aws-lc-sys = { git ="https://github.com/aws/aws-lc-rs", rev = "556558441e3494af4b156ae95ebc07ebc2fd38aa" }
+
 [workspace.dependencies.meter-macros]
 git = "https://github.com/GreptimeTeam/greptime-meter.git"
-rev = "80eb97c24c88af4dd9a86f8bbaf50e741d4eb8cd"
+rev = "a10facb353b41460eeb98578868ebf19c2084fac"

 [profile.release]
-debug = 1
+# debug = 1
+split-debuginfo = "off"

 [profile.nightly]
 inherits = "release"
-strip = "debuginfo"
+split-debuginfo = "off"
+# strip = "debuginfo"
 lto = "thin"
-debug = false
+# debug = false
 incremental = false

 [profile.ci]
--- a/4
+++ b/4
@@ -8,7 +8,7 @@ CARGO_BUILD_OPTS := --locked
 IMAGE_REGISTRY ?= docker.io
 IMAGE_NAMESPACE ?= greptime
 IMAGE_TAG ?= latest
-DEV_BUILDER_IMAGE_TAG ?= 2024-06-06-b4b105ad-20240827021230
+DEV_BUILDER_IMAGE_TAG ?= 2024-10-19-a5c00e85-20241024184445
 BUILDX_MULTI_PLATFORM_BUILD ?= false
 BUILDX_BUILDER_NAME ?= gtbuilder
 BASE_IMAGE ?= ubuntu
@@ -221,7 +221,7 @@ config-docs: ## Generate configuration documentation from toml files.
 	docker run --rm \
    -v ${PWD}:/greptimedb \
    -w /greptimedb/config \
-    toml2docs/toml2docs:v0.1.1 \
+    toml2docs/toml2docs:v0.1.3 \
    -p '##' \
    -t ./config-docs-template.md \
    -o ./config.md
--- a/README.md
+++ b/README.md
@@ -74,7 +74,7 @@ Our core developers have been building time-series data platforms for years. Bas

 * **Compatible with InfluxDB, Prometheus and more protocols**

-  Widely adopted database protocols and APIs, including MySQL, PostgreSQL, and Prometheus Remote Storage, etc. [Read more](https://docs.greptime.com/user-guide/clients/overview).
+  Widely adopted database protocols and APIs, including MySQL, PostgreSQL, and Prometheus Remote Storage, etc. [Read more](https://docs.greptime.com/user-guide/protocols/overview).

 ## Try GreptimeDB

--- a/config/config.md
+++ b/config/config.md
@@ -14,9 +14,10 @@
 | --- | -----| ------- | ----------- |
 | `mode` | String | `standalone` | The running mode of the datanode. It can be `standalone` or `distributed`. |
 | `enable_telemetry` | Bool | `true` | Enable telemetry to collect anonymous usage data. |
-| `default_timezone` | String | `None` | The default timezone of the server. |
+| `default_timezone` | String | Unset | The default timezone of the server. |
 | `init_regions_in_background` | Bool | `false` | Initialize all regions in the background during the startup.<br/>By default, it provides services after all regions have been initialized. |
 | `init_regions_parallelism` | Integer | `16` | Parallelism of initializing regions. |
+| `max_concurrent_queries` | Integer | `0` | The maximum current queries allowed to be executed. Zero means unlimited. |
 | `runtime` | -- | -- | The runtime options. |
 | `runtime.global_rt_size` | Integer | `8` | The number of threads to execute the runtime for global read operations. |
 | `runtime.compact_rt_size` | Integer | `4` | The number of threads to execute the runtime for global write operations. |
@@ -29,8 +30,8 @@
 | `grpc.runtime_size` | Integer | `8` | The number of server worker threads. |
 | `grpc.tls` | -- | -- | gRPC server TLS options, see `mysql.tls` section. |
 | `grpc.tls.mode` | String | `disable` | TLS mode. |
-| `grpc.tls.cert_path` | String | `None` | Certificate file path. |
-| `grpc.tls.key_path` | String | `None` | Private key file path. |
+| `grpc.tls.cert_path` | String | Unset | Certificate file path. |
+| `grpc.tls.key_path` | String | Unset | Private key file path. |
 | `grpc.tls.watch` | Bool | `false` | Watch for Certificate and key file change and auto reload.<br/>For now, gRPC tls config does not support auto reload. |
 | `mysql` | -- | -- | MySQL server options. |
 | `mysql.enable` | Bool | `true` | Whether to enable. |
@@ -38,8 +39,8 @@
 | `mysql.runtime_size` | Integer | `2` | The number of server worker threads. |
 | `mysql.tls` | -- | -- | -- |
 | `mysql.tls.mode` | String | `disable` | TLS mode, refer to https://www.postgresql.org/docs/current/libpq-ssl.html<br/>- `disable` (default value)<br/>- `prefer`<br/>- `require`<br/>- `verify-ca`<br/>- `verify-full` |
-| `mysql.tls.cert_path` | String | `None` | Certificate file path. |
-| `mysql.tls.key_path` | String | `None` | Private key file path. |
+| `mysql.tls.cert_path` | String | Unset | Certificate file path. |
+| `mysql.tls.key_path` | String | Unset | Private key file path. |
 | `mysql.tls.watch` | Bool | `false` | Watch for Certificate and key file change and auto reload |
 | `postgres` | -- | -- | PostgresSQL server options. |
 | `postgres.enable` | Bool | `true` | Whether to enable |
@@ -47,8 +48,8 @@
 | `postgres.runtime_size` | Integer | `2` | The number of server worker threads. |
 | `postgres.tls` | -- | -- | PostgresSQL server TLS options, see `mysql.tls` section. |
 | `postgres.tls.mode` | String | `disable` | TLS mode. |
-| `postgres.tls.cert_path` | String | `None` | Certificate file path. |
-| `postgres.tls.key_path` | String | `None` | Private key file path. |
+| `postgres.tls.cert_path` | String | Unset | Certificate file path. |
+| `postgres.tls.key_path` | String | Unset | Private key file path. |
 | `postgres.tls.watch` | Bool | `false` | Watch for Certificate and key file change and auto reload |
 | `opentsdb` | -- | -- | OpenTSDB protocol options. |
 | `opentsdb.enable` | Bool | `true` | Whether to enable OpenTSDB put in HTTP API. |
@@ -59,7 +60,7 @@
 | `prom_store.with_metric_engine` | Bool | `true` | Whether to store the data from Prometheus remote write in metric engine. |
 | `wal` | -- | -- | The WAL options. |
 | `wal.provider` | String | `raft_engine` | The provider of the WAL.<br/>- `raft_engine`: the wal is stored in the local file system by raft-engine.<br/>- `kafka`: it's remote wal that data is stored in Kafka. |
-| `wal.dir` | String | `None` | The directory to store the WAL files.<br/>**It's only used when the provider is `raft_engine`**. |
+| `wal.dir` | String | Unset | The directory to store the WAL files.<br/>**It's only used when the provider is `raft_engine`**. |
 | `wal.file_size` | String | `256MB` | The size of the WAL segment file.<br/>**It's only used when the provider is `raft_engine`**. |
 | `wal.purge_threshold` | String | `4GB` | The threshold of the WAL size to trigger a flush.<br/>**It's only used when the provider is `raft_engine`**. |
 | `wal.purge_interval` | String | `10m` | The interval to trigger a flush.<br/>**It's only used when the provider is `raft_engine`**. |
@@ -68,6 +69,7 @@
 | `wal.enable_log_recycle` | Bool | `true` | Whether to reuse logically truncated log files.<br/>**It's only used when the provider is `raft_engine`**. |
 | `wal.prefill_log_files` | Bool | `false` | Whether to pre-create log files on start up.<br/>**It's only used when the provider is `raft_engine`**. |
 | `wal.sync_period` | String | `10s` | Duration for fsyncing log files.<br/>**It's only used when the provider is `raft_engine`**. |
+| `wal.recovery_parallelism` | Integer | `2` | Parallelism during WAL recovery. |
 | `wal.broker_endpoints` | Array | -- | The Kafka broker endpoints.<br/>**It's only used when the provider is `kafka`**. |
 | `wal.auto_create_topics` | Bool | `true` | Automatically create topics for WAL.<br/>Set to `true` to automatically create topics for WAL.<br/>Otherwise, use topics named `topic_name_prefix_[0..num_topics)` |
 | `wal.num_topics` | Integer | `64` | Number of topics.<br/>**It's only used when the provider is `kafka`**. |
@@ -81,6 +83,7 @@
 | `wal.backoff_max` | String | `10s` | The maximum backoff delay.<br/>**It's only used when the provider is `kafka`**. |
 | `wal.backoff_base` | Integer | `2` | The exponential backoff rate, i.e. next backoff = base * current backoff.<br/>**It's only used when the provider is `kafka`**. |
 | `wal.backoff_deadline` | String | `5mins` | The deadline of retries.<br/>**It's only used when the provider is `kafka`**. |
+| `wal.overwrite_entry_start_id` | Bool | `false` | Ignore missing entries during read WAL.<br/>**It's only used when the provider is `kafka`**.<br/><br/>This option ensures that when Kafka messages are deleted, the system<br/>can still successfully replay memtable data without throwing an<br/>out-of-range error.<br/>However, enabling this option might lead to unexpected data loss,<br/>as the system will skip over missing entries instead of treating<br/>them as critical errors. |
 | `metadata_store` | -- | -- | Metadata storage options. |
 | `metadata_store.file_size` | String | `256MB` | Kv file size in bytes. |
 | `metadata_store.purge_threshold` | String | `4GB` | Kv purge threshold. |
@@ -90,22 +93,22 @@
 | `storage` | -- | -- | The data storage options. |
 | `storage.data_home` | String | `/tmp/greptimedb/` | The working home directory. |
 | `storage.type` | String | `File` | The storage type used to store the data.<br/>- `File`: the data is stored in the local file system.<br/>- `S3`: the data is stored in the S3 object storage.<br/>- `Gcs`: the data is stored in the Google Cloud Storage.<br/>- `Azblob`: the data is stored in the Azure Blob Storage.<br/>- `Oss`: the data is stored in the Aliyun OSS. |
-| `storage.cache_path` | String | `None` | Cache configuration for object storage such as 'S3' etc.<br/>The local file cache directory. |
-| `storage.cache_capacity` | String | `None` | The local file cache capacity in bytes. |
-| `storage.bucket` | String | `None` | The S3 bucket name.<br/>**It's only used when the storage type is `S3`, `Oss` and `Gcs`**. |
-| `storage.root` | String | `None` | The S3 data will be stored in the specified prefix, for example, `s3://${bucket}/${root}`.<br/>**It's only used when the storage type is `S3`, `Oss` and `Azblob`**. |
-| `storage.access_key_id` | String | `None` | The access key id of the aws account.<br/>It's **highly recommended** to use AWS IAM roles instead of hardcoding the access key id and secret key.<br/>**It's only used when the storage type is `S3` and `Oss`**. |
-| `storage.secret_access_key` | String | `None` | The secret access key of the aws account.<br/>It's **highly recommended** to use AWS IAM roles instead of hardcoding the access key id and secret key.<br/>**It's only used when the storage type is `S3`**. |
-| `storage.access_key_secret` | String | `None` | The secret access key of the aliyun account.<br/>**It's only used when the storage type is `Oss`**. |
-| `storage.account_name` | String | `None` | The account key of the azure account.<br/>**It's only used when the storage type is `Azblob`**. |
-| `storage.account_key` | String | `None` | The account key of the azure account.<br/>**It's only used when the storage type is `Azblob`**. |
-| `storage.scope` | String | `None` | The scope of the google cloud storage.<br/>**It's only used when the storage type is `Gcs`**. |
-| `storage.credential_path` | String | `None` | The credential path of the google cloud storage.<br/>**It's only used when the storage type is `Gcs`**. |
-| `storage.credential` | String | `None` | The credential of the google cloud storage.<br/>**It's only used when the storage type is `Gcs`**. |
-| `storage.container` | String | `None` | The container of the azure account.<br/>**It's only used when the storage type is `Azblob`**. |
-| `storage.sas_token` | String | `None` | The sas token of the azure account.<br/>**It's only used when the storage type is `Azblob`**. |
-| `storage.endpoint` | String | `None` | The endpoint of the S3 service.<br/>**It's only used when the storage type is `S3`, `Oss`, `Gcs` and `Azblob`**. |
-| `storage.region` | String | `None` | The region of the S3 service.<br/>**It's only used when the storage type is `S3`, `Oss`, `Gcs` and `Azblob`**. |
+| `storage.cache_path` | String | Unset | Cache configuration for object storage such as 'S3' etc.<br/>The local file cache directory. |
+| `storage.cache_capacity` | String | Unset | The local file cache capacity in bytes. |
+| `storage.bucket` | String | Unset | The S3 bucket name.<br/>**It's only used when the storage type is `S3`, `Oss` and `Gcs`**. |
+| `storage.root` | String | Unset | The S3 data will be stored in the specified prefix, for example, `s3://${bucket}/${root}`.<br/>**It's only used when the storage type is `S3`, `Oss` and `Azblob`**. |
+| `storage.access_key_id` | String | Unset | The access key id of the aws account.<br/>It's **highly recommended** to use AWS IAM roles instead of hardcoding the access key id and secret key.<br/>**It's only used when the storage type is `S3` and `Oss`**. |
+| `storage.secret_access_key` | String | Unset | The secret access key of the aws account.<br/>It's **highly recommended** to use AWS IAM roles instead of hardcoding the access key id and secret key.<br/>**It's only used when the storage type is `S3`**. |
+| `storage.access_key_secret` | String | Unset | The secret access key of the aliyun account.<br/>**It's only used when the storage type is `Oss`**. |
+| `storage.account_name` | String | Unset | The account key of the azure account.<br/>**It's only used when the storage type is `Azblob`**. |
+| `storage.account_key` | String | Unset | The account key of the azure account.<br/>**It's only used when the storage type is `Azblob`**. |
+| `storage.scope` | String | Unset | The scope of the google cloud storage.<br/>**It's only used when the storage type is `Gcs`**. |
+| `storage.credential_path` | String | Unset | The credential path of the google cloud storage.<br/>**It's only used when the storage type is `Gcs`**. |
+| `storage.credential` | String | Unset | The credential of the google cloud storage.<br/>**It's only used when the storage type is `Gcs`**. |
+| `storage.container` | String | Unset | The container of the azure account.<br/>**It's only used when the storage type is `Azblob`**. |
+| `storage.sas_token` | String | Unset | The sas token of the azure account.<br/>**It's only used when the storage type is `Azblob`**. |
+| `storage.endpoint` | String | Unset | The endpoint of the S3 service.<br/>**It's only used when the storage type is `S3`, `Oss`, `Gcs` and `Azblob`**. |
+| `storage.region` | String | Unset | The region of the S3 service.<br/>**It's only used when the storage type is `S3`, `Oss`, `Gcs` and `Azblob`**. |
 | `[[region_engine]]` | -- | -- | The region engine options. You can configure multiple region engines. |
 | `region_engine.mito` | -- | -- | The Mito engine options. |
 | `region_engine.mito.num_workers` | Integer | `8` | Number of region workers. |
@@ -113,18 +116,20 @@
 | `region_engine.mito.worker_request_batch_size` | Integer | `64` | Max batch size for a worker to handle requests. |
 | `region_engine.mito.manifest_checkpoint_distance` | Integer | `10` | Number of meta action updated to trigger a new checkpoint for the manifest. |
 | `region_engine.mito.compress_manifest` | Bool | `false` | Whether to compress manifest and checkpoint file by gzip (default false). |
-| `region_engine.mito.max_background_jobs` | Integer | `4` | Max number of running background jobs |
+| `region_engine.mito.max_background_flushes` | Integer | Auto | Max number of running background flush jobs (default: 1/2 of cpu cores). |
+| `region_engine.mito.max_background_compactions` | Integer | Auto | Max number of running background compaction jobs (default: 1/4 of cpu cores). |
+| `region_engine.mito.max_background_purges` | Integer | Auto | Max number of running background purge jobs (default: number of cpu cores). |
 | `region_engine.mito.auto_flush_interval` | String | `1h` | Interval to auto flush a region if it has not flushed yet. |
-| `region_engine.mito.global_write_buffer_size` | String | `1GB` | Global write buffer size for all regions. If not set, it's default to 1/8 of OS memory with a max limitation of 1GB. |
-| `region_engine.mito.global_write_buffer_reject_size` | String | `2GB` | Global write buffer size threshold to reject write requests. If not set, it's default to 2 times of `global_write_buffer_size` |
-| `region_engine.mito.sst_meta_cache_size` | String | `128MB` | Cache size for SST metadata. Setting it to 0 to disable the cache.<br/>If not set, it's default to 1/32 of OS memory with a max limitation of 128MB. |
-| `region_engine.mito.vector_cache_size` | String | `512MB` | Cache size for vectors and arrow arrays. Setting it to 0 to disable the cache.<br/>If not set, it's default to 1/16 of OS memory with a max limitation of 512MB. |
-| `region_engine.mito.page_cache_size` | String | `512MB` | Cache size for pages of SST row groups. Setting it to 0 to disable the cache.<br/>If not set, it's default to 1/8 of OS memory. |
-| `region_engine.mito.selector_result_cache_size` | String | `512MB` | Cache size for time series selector (e.g. `last_value()`). Setting it to 0 to disable the cache.<br/>If not set, it's default to 1/16 of OS memory with a max limitation of 512MB. |
+| `region_engine.mito.global_write_buffer_size` | String | Auto | Global write buffer size for all regions. If not set, it's default to 1/8 of OS memory with a max limitation of 1GB. |
+| `region_engine.mito.global_write_buffer_reject_size` | String | Auto | Global write buffer size threshold to reject write requests. If not set, it's default to 2 times of `global_write_buffer_size`. |
+| `region_engine.mito.sst_meta_cache_size` | String | Auto | Cache size for SST metadata. Setting it to 0 to disable the cache.<br/>If not set, it's default to 1/32 of OS memory with a max limitation of 128MB. |
+| `region_engine.mito.vector_cache_size` | String | Auto | Cache size for vectors and arrow arrays. Setting it to 0 to disable the cache.<br/>If not set, it's default to 1/16 of OS memory with a max limitation of 512MB. |
+| `region_engine.mito.page_cache_size` | String | Auto | Cache size for pages of SST row groups. Setting it to 0 to disable the cache.<br/>If not set, it's default to 1/8 of OS memory. |
+| `region_engine.mito.selector_result_cache_size` | String | Auto | Cache size for time series selector (e.g. `last_value()`). Setting it to 0 to disable the cache.<br/>If not set, it's default to 1/16 of OS memory with a max limitation of 512MB. |
 | `region_engine.mito.enable_experimental_write_cache` | Bool | `false` | Whether to enable the experimental write cache. |
 | `region_engine.mito.experimental_write_cache_path` | String | `""` | File system path for write cache, defaults to `{data_home}/write_cache`. |
 | `region_engine.mito.experimental_write_cache_size` | String | `512MB` | Capacity for write cache. |
-| `region_engine.mito.experimental_write_cache_ttl` | String | `None` | TTL for write cache. |
+| `region_engine.mito.experimental_write_cache_ttl` | String | Unset | TTL for write cache. |
 | `region_engine.mito.sst_write_buffer_size` | String | `8MB` | Buffer size for SST writing. |
 | `region_engine.mito.scan_parallelism` | Integer | `0` | Parallelism to scan a region (default: 1/4 of cpu cores).<br/>- `0`: using the default value (1/4 of cpu cores).<br/>- `1`: scan in current thread.<br/>- `n`: scan in parallelism n. |
 | `region_engine.mito.parallel_scan_channel_size` | Integer | `32` | Capacity of the channel to send data from parallel scan tasks to the main task. |
@@ -154,23 +159,28 @@
 | `region_engine.file` | -- | -- | Enable the file engine. |
 | `logging` | -- | -- | The logging options. |
 | `logging.dir` | String | `/tmp/greptimedb/logs` | The directory to store the log files. If set to empty, logs will not be written to files. |
-| `logging.level` | String | `None` | The log level. Can be `info`/`debug`/`warn`/`error`. |
+| `logging.level` | String | Unset | The log level. Can be `info`/`debug`/`warn`/`error`. |
 | `logging.enable_otlp_tracing` | Bool | `false` | Enable OTLP tracing. |
 | `logging.otlp_endpoint` | String | `http://localhost:4317` | The OTLP tracing endpoint. |
 | `logging.append_stdout` | Bool | `true` | Whether to append logs to stdout. |
 | `logging.log_format` | String | `text` | The log format. Can be `text`/`json`. |
+| `logging.max_log_files` | Integer | `720` | The maximum amount of log files. |
 | `logging.tracing_sample_ratio` | -- | -- | The percentage of tracing will be sampled and exported.<br/>Valid range `[0, 1]`, 1 means all traces are sampled, 0 means all traces are not sampled, the default value is 1.<br/>ratio > 1 are treated as 1. Fractions < 0 are treated as 0 |
 | `logging.tracing_sample_ratio.default_ratio` | Float | `1.0` | -- |
+| `logging.slow_query` | -- | -- | The slow query log options. |
+| `logging.slow_query.enable` | Bool | `false` | Whether to enable slow query log. |
+| `logging.slow_query.threshold` | String | Unset | The threshold of slow query. |
+| `logging.slow_query.sample_ratio` | Float | Unset | The sampling ratio of slow query log. The value should be in the range of (0, 1]. |
 | `export_metrics` | -- | -- | The datanode can export its metrics and send to Prometheus compatible service (e.g. send to `greptimedb` itself) from remote-write API.<br/>This is only used for `greptimedb` to export its own metrics internally. It's different from prometheus scrape. |
 | `export_metrics.enable` | Bool | `false` | whether enable export metrics. |
 | `export_metrics.write_interval` | String | `30s` | The interval of export metrics. |
-| `export_metrics.self_import` | -- | -- | For `standalone` mode, `self_import` is recommend to collect metrics generated by itself<br/>You must create the database before enabling it. |
-| `export_metrics.self_import.db` | String | `None` | -- |
+| `export_metrics.self_import` | -- | -- | For `standalone` mode, `self_import` is recommended to collect metrics generated by itself<br/>You must create the database before enabling it. |
+| `export_metrics.self_import.db` | String | Unset | -- |
 | `export_metrics.remote_write` | -- | -- | -- |
 | `export_metrics.remote_write.url` | String | `""` | The url the metrics send to. The url example can be: `http://127.0.0.1:4000/v1/prometheus/write?db=greptime_metrics`. |
 | `export_metrics.remote_write.headers` | InlineTable | -- | HTTP headers of Prometheus remote-write carry. |
 | `tracing` | -- | -- | The tracing options. Only effect when compiled with `tokio-console` feature. |
-| `tracing.tokio_console_addr` | String | `None` | The tokio console address. |
+| `tracing.tokio_console_addr` | String | Unset | The tokio console address. |


 ## Distributed Mode
@@ -179,7 +189,7 @@

 | Key | Type | Default | Descriptions |
 | --- | -----| ------- | ----------- |
-| `default_timezone` | String | `None` | The default timezone of the server. |
+| `default_timezone` | String | Unset | The default timezone of the server. |
 | `runtime` | -- | -- | The runtime options. |
 | `runtime.global_rt_size` | Integer | `8` | The number of threads to execute the runtime for global read operations. |
 | `runtime.compact_rt_size` | Integer | `4` | The number of threads to execute the runtime for global write operations. |
@@ -196,8 +206,8 @@
 | `grpc.runtime_size` | Integer | `8` | The number of server worker threads. |
 | `grpc.tls` | -- | -- | gRPC server TLS options, see `mysql.tls` section. |
 | `grpc.tls.mode` | String | `disable` | TLS mode. |
-| `grpc.tls.cert_path` | String | `None` | Certificate file path. |
-| `grpc.tls.key_path` | String | `None` | Private key file path. |
+| `grpc.tls.cert_path` | String | Unset | Certificate file path. |
+| `grpc.tls.key_path` | String | Unset | Private key file path. |
 | `grpc.tls.watch` | Bool | `false` | Watch for Certificate and key file change and auto reload.<br/>For now, gRPC tls config does not support auto reload. |
 | `mysql` | -- | -- | MySQL server options. |
 | `mysql.enable` | Bool | `true` | Whether to enable. |
@@ -205,8 +215,8 @@
 | `mysql.runtime_size` | Integer | `2` | The number of server worker threads. |
 | `mysql.tls` | -- | -- | -- |
 | `mysql.tls.mode` | String | `disable` | TLS mode, refer to https://www.postgresql.org/docs/current/libpq-ssl.html<br/>- `disable` (default value)<br/>- `prefer`<br/>- `require`<br/>- `verify-ca`<br/>- `verify-full` |
-| `mysql.tls.cert_path` | String | `None` | Certificate file path. |
-| `mysql.tls.key_path` | String | `None` | Private key file path. |
+| `mysql.tls.cert_path` | String | Unset | Certificate file path. |
+| `mysql.tls.key_path` | String | Unset | Private key file path. |
 | `mysql.tls.watch` | Bool | `false` | Watch for Certificate and key file change and auto reload |
 | `postgres` | -- | -- | PostgresSQL server options. |
 | `postgres.enable` | Bool | `true` | Whether to enable |
@@ -214,8 +224,8 @@
 | `postgres.runtime_size` | Integer | `2` | The number of server worker threads. |
 | `postgres.tls` | -- | -- | PostgresSQL server TLS options, see `mysql.tls` section. |
 | `postgres.tls.mode` | String | `disable` | TLS mode. |
-| `postgres.tls.cert_path` | String | `None` | Certificate file path. |
-| `postgres.tls.key_path` | String | `None` | Private key file path. |
+| `postgres.tls.cert_path` | String | Unset | Certificate file path. |
+| `postgres.tls.key_path` | String | Unset | Private key file path. |
 | `postgres.tls.watch` | Bool | `false` | Watch for Certificate and key file change and auto reload |
 | `opentsdb` | -- | -- | OpenTSDB protocol options. |
 | `opentsdb.enable` | Bool | `true` | Whether to enable OpenTSDB put in HTTP API. |
@@ -240,23 +250,28 @@
 | `datanode.client.tcp_nodelay` | Bool | `true` | -- |
 | `logging` | -- | -- | The logging options. |
 | `logging.dir` | String | `/tmp/greptimedb/logs` | The directory to store the log files. If set to empty, logs will not be written to files. |
-| `logging.level` | String | `None` | The log level. Can be `info`/`debug`/`warn`/`error`. |
+| `logging.level` | String | Unset | The log level. Can be `info`/`debug`/`warn`/`error`. |
 | `logging.enable_otlp_tracing` | Bool | `false` | Enable OTLP tracing. |
 | `logging.otlp_endpoint` | String | `http://localhost:4317` | The OTLP tracing endpoint. |
 | `logging.append_stdout` | Bool | `true` | Whether to append logs to stdout. |
 | `logging.log_format` | String | `text` | The log format. Can be `text`/`json`. |
+| `logging.max_log_files` | Integer | `720` | The maximum amount of log files. |
 | `logging.tracing_sample_ratio` | -- | -- | The percentage of tracing will be sampled and exported.<br/>Valid range `[0, 1]`, 1 means all traces are sampled, 0 means all traces are not sampled, the default value is 1.<br/>ratio > 1 are treated as 1. Fractions < 0 are treated as 0 |
 | `logging.tracing_sample_ratio.default_ratio` | Float | `1.0` | -- |
+| `logging.slow_query` | -- | -- | The slow query log options. |
+| `logging.slow_query.enable` | Bool | `false` | Whether to enable slow query log. |
+| `logging.slow_query.threshold` | String | Unset | The threshold of slow query. |
+| `logging.slow_query.sample_ratio` | Float | Unset | The sampling ratio of slow query log. The value should be in the range of (0, 1]. |
 | `export_metrics` | -- | -- | The datanode can export its metrics and send to Prometheus compatible service (e.g. send to `greptimedb` itself) from remote-write API.<br/>This is only used for `greptimedb` to export its own metrics internally. It's different from prometheus scrape. |
 | `export_metrics.enable` | Bool | `false` | whether enable export metrics. |
 | `export_metrics.write_interval` | String | `30s` | The interval of export metrics. |
 | `export_metrics.self_import` | -- | -- | For `standalone` mode, `self_import` is recommend to collect metrics generated by itself<br/>You must create the database before enabling it. |
-| `export_metrics.self_import.db` | String | `None` | -- |
+| `export_metrics.self_import.db` | String | Unset | -- |
 | `export_metrics.remote_write` | -- | -- | -- |
 | `export_metrics.remote_write.url` | String | `""` | The url the metrics send to. The url example can be: `http://127.0.0.1:4000/v1/prometheus/write?db=greptime_metrics`. |
 | `export_metrics.remote_write.headers` | InlineTable | -- | HTTP headers of Prometheus remote-write carry. |
 | `tracing` | -- | -- | The tracing options. Only effect when compiled with `tokio-console` feature. |
-| `tracing.tokio_console_addr` | String | `None` | The tokio console address. |
+| `tracing.tokio_console_addr` | String | Unset | The tokio console address. |


 ### Metasrv
@@ -305,23 +320,28 @@
 | `wal.backoff_deadline` | String | `5mins` | Stop reconnecting if the total wait time reaches the deadline. If this config is missing, the reconnecting won't terminate. |
 | `logging` | -- | -- | The logging options. |
 | `logging.dir` | String | `/tmp/greptimedb/logs` | The directory to store the log files. If set to empty, logs will not be written to files. |
-| `logging.level` | String | `None` | The log level. Can be `info`/`debug`/`warn`/`error`. |
+| `logging.level` | String | Unset | The log level. Can be `info`/`debug`/`warn`/`error`. |
 | `logging.enable_otlp_tracing` | Bool | `false` | Enable OTLP tracing. |
 | `logging.otlp_endpoint` | String | `http://localhost:4317` | The OTLP tracing endpoint. |
 | `logging.append_stdout` | Bool | `true` | Whether to append logs to stdout. |
 | `logging.log_format` | String | `text` | The log format. Can be `text`/`json`. |
+| `logging.max_log_files` | Integer | `720` | The maximum amount of log files. |
 | `logging.tracing_sample_ratio` | -- | -- | The percentage of tracing will be sampled and exported.<br/>Valid range `[0, 1]`, 1 means all traces are sampled, 0 means all traces are not sampled, the default value is 1.<br/>ratio > 1 are treated as 1. Fractions < 0 are treated as 0 |
 | `logging.tracing_sample_ratio.default_ratio` | Float | `1.0` | -- |
+| `logging.slow_query` | -- | -- | The slow query log options. |
+| `logging.slow_query.enable` | Bool | `false` | Whether to enable slow query log. |
+| `logging.slow_query.threshold` | String | Unset | The threshold of slow query. |
+| `logging.slow_query.sample_ratio` | Float | Unset | The sampling ratio of slow query log. The value should be in the range of (0, 1]. |
 | `export_metrics` | -- | -- | The datanode can export its metrics and send to Prometheus compatible service (e.g. send to `greptimedb` itself) from remote-write API.<br/>This is only used for `greptimedb` to export its own metrics internally. It's different from prometheus scrape. |
 | `export_metrics.enable` | Bool | `false` | whether enable export metrics. |
 | `export_metrics.write_interval` | String | `30s` | The interval of export metrics. |
 | `export_metrics.self_import` | -- | -- | For `standalone` mode, `self_import` is recommend to collect metrics generated by itself<br/>You must create the database before enabling it. |
-| `export_metrics.self_import.db` | String | `None` | -- |
+| `export_metrics.self_import.db` | String | Unset | -- |
 | `export_metrics.remote_write` | -- | -- | -- |
 | `export_metrics.remote_write.url` | String | `""` | The url the metrics send to. The url example can be: `http://127.0.0.1:4000/v1/prometheus/write?db=greptime_metrics`. |
 | `export_metrics.remote_write.headers` | InlineTable | -- | HTTP headers of Prometheus remote-write carry. |
 | `tracing` | -- | -- | The tracing options. Only effect when compiled with `tokio-console` feature. |
-| `tracing.tokio_console_addr` | String | `None` | The tokio console address. |
+| `tracing.tokio_console_addr` | String | Unset | The tokio console address. |


 ### Datanode
@@ -329,16 +349,17 @@
 | Key | Type | Default | Descriptions |
 | --- | -----| ------- | ----------- |
 | `mode` | String | `standalone` | The running mode of the datanode. It can be `standalone` or `distributed`. |
-| `node_id` | Integer | `None` | The datanode identifier and should be unique in the cluster. |
+| `node_id` | Integer | Unset | The datanode identifier and should be unique in the cluster. |
 | `require_lease_before_startup` | Bool | `false` | Start services after regions have obtained leases.<br/>It will block the datanode start if it can't receive leases in the heartbeat from metasrv. |
 | `init_regions_in_background` | Bool | `false` | Initialize all regions in the background during the startup.<br/>By default, it provides services after all regions have been initialized. |
 | `enable_telemetry` | Bool | `true` | Enable telemetry to collect anonymous usage data. |
 | `init_regions_parallelism` | Integer | `16` | Parallelism of initializing regions. |
-| `rpc_addr` | String | `None` | Deprecated, use `grpc.addr` instead. |
-| `rpc_hostname` | String | `None` | Deprecated, use `grpc.hostname` instead. |
-| `rpc_runtime_size` | Integer | `None` | Deprecated, use `grpc.runtime_size` instead. |
-| `rpc_max_recv_message_size` | String | `None` | Deprecated, use `grpc.rpc_max_recv_message_size` instead. |
-| `rpc_max_send_message_size` | String | `None` | Deprecated, use `grpc.rpc_max_send_message_size` instead. |
+| `max_concurrent_queries` | Integer | `0` | The maximum current queries allowed to be executed. Zero means unlimited. |
+| `rpc_addr` | String | Unset | Deprecated, use `grpc.addr` instead. |
+| `rpc_hostname` | String | Unset | Deprecated, use `grpc.hostname` instead. |
+| `rpc_runtime_size` | Integer | Unset | Deprecated, use `grpc.runtime_size` instead. |
+| `rpc_max_recv_message_size` | String | Unset | Deprecated, use `grpc.rpc_max_recv_message_size` instead. |
+| `rpc_max_send_message_size` | String | Unset | Deprecated, use `grpc.rpc_max_send_message_size` instead. |
 | `http` | -- | -- | The HTTP server options. |
 | `http.addr` | String | `127.0.0.1:4000` | The address to bind the HTTP server. |
 | `http.timeout` | String | `30s` | HTTP request timeout. Set to 0 to disable timeout. |
@@ -351,8 +372,8 @@
 | `grpc.max_send_message_size` | String | `512MB` | The maximum send message size for gRPC server. |
 | `grpc.tls` | -- | -- | gRPC server TLS options, see `mysql.tls` section. |
 | `grpc.tls.mode` | String | `disable` | TLS mode. |
-| `grpc.tls.cert_path` | String | `None` | Certificate file path. |
-| `grpc.tls.key_path` | String | `None` | Private key file path. |
+| `grpc.tls.cert_path` | String | Unset | Certificate file path. |
+| `grpc.tls.key_path` | String | Unset | Private key file path. |
 | `grpc.tls.watch` | Bool | `false` | Watch for Certificate and key file change and auto reload.<br/>For now, gRPC tls config does not support auto reload. |
 | `runtime` | -- | -- | The runtime options. |
 | `runtime.global_rt_size` | Integer | `8` | The number of threads to execute the runtime for global read operations. |
@@ -372,7 +393,7 @@
 | `meta_client.metadata_cache_tti` | String | `5m` | -- |
 | `wal` | -- | -- | The WAL options. |
 | `wal.provider` | String | `raft_engine` | The provider of the WAL.<br/>- `raft_engine`: the wal is stored in the local file system by raft-engine.<br/>- `kafka`: it's remote wal that data is stored in Kafka. |
-| `wal.dir` | String | `None` | The directory to store the WAL files.<br/>**It's only used when the provider is `raft_engine`**. |
+| `wal.dir` | String | Unset | The directory to store the WAL files.<br/>**It's only used when the provider is `raft_engine`**. |
 | `wal.file_size` | String | `256MB` | The size of the WAL segment file.<br/>**It's only used when the provider is `raft_engine`**. |
 | `wal.purge_threshold` | String | `4GB` | The threshold of the WAL size to trigger a flush.<br/>**It's only used when the provider is `raft_engine`**. |
 | `wal.purge_interval` | String | `10m` | The interval to trigger a flush.<br/>**It's only used when the provider is `raft_engine`**. |
@@ -381,6 +402,7 @@
 | `wal.enable_log_recycle` | Bool | `true` | Whether to reuse logically truncated log files.<br/>**It's only used when the provider is `raft_engine`**. |
 | `wal.prefill_log_files` | Bool | `false` | Whether to pre-create log files on start up.<br/>**It's only used when the provider is `raft_engine`**. |
 | `wal.sync_period` | String | `10s` | Duration for fsyncing log files.<br/>**It's only used when the provider is `raft_engine`**. |
+| `wal.recovery_parallelism` | Integer | `2` | Parallelism during WAL recovery. |
 | `wal.broker_endpoints` | Array | -- | The Kafka broker endpoints.<br/>**It's only used when the provider is `kafka`**. |
 | `wal.max_batch_bytes` | String | `1MB` | The max size of a single producer batch.<br/>Warning: Kafka has a default limit of 1MB per message in a topic.<br/>**It's only used when the provider is `kafka`**. |
 | `wal.consumer_wait_timeout` | String | `100ms` | The consumer wait timeout.<br/>**It's only used when the provider is `kafka`**. |
@@ -390,25 +412,26 @@
 | `wal.backoff_deadline` | String | `5mins` | The deadline of retries.<br/>**It's only used when the provider is `kafka`**. |
 | `wal.create_index` | Bool | `true` | Whether to enable WAL index creation.<br/>**It's only used when the provider is `kafka`**. |
 | `wal.dump_index_interval` | String | `60s` | The interval for dumping WAL indexes.<br/>**It's only used when the provider is `kafka`**. |
+| `wal.overwrite_entry_start_id` | Bool | `false` | Ignore missing entries during read WAL.<br/>**It's only used when the provider is `kafka`**.<br/><br/>This option ensures that when Kafka messages are deleted, the system<br/>can still successfully replay memtable data without throwing an<br/>out-of-range error.<br/>However, enabling this option might lead to unexpected data loss,<br/>as the system will skip over missing entries instead of treating<br/>them as critical errors. |
 | `storage` | -- | -- | The data storage options. |
 | `storage.data_home` | String | `/tmp/greptimedb/` | The working home directory. |
 | `storage.type` | String | `File` | The storage type used to store the data.<br/>- `File`: the data is stored in the local file system.<br/>- `S3`: the data is stored in the S3 object storage.<br/>- `Gcs`: the data is stored in the Google Cloud Storage.<br/>- `Azblob`: the data is stored in the Azure Blob Storage.<br/>- `Oss`: the data is stored in the Aliyun OSS. |
-| `storage.cache_path` | String | `None` | Cache configuration for object storage such as 'S3' etc.<br/>The local file cache directory. |
-| `storage.cache_capacity` | String | `None` | The local file cache capacity in bytes. |
-| `storage.bucket` | String | `None` | The S3 bucket name.<br/>**It's only used when the storage type is `S3`, `Oss` and `Gcs`**. |
-| `storage.root` | String | `None` | The S3 data will be stored in the specified prefix, for example, `s3://${bucket}/${root}`.<br/>**It's only used when the storage type is `S3`, `Oss` and `Azblob`**. |
-| `storage.access_key_id` | String | `None` | The access key id of the aws account.<br/>It's **highly recommended** to use AWS IAM roles instead of hardcoding the access key id and secret key.<br/>**It's only used when the storage type is `S3` and `Oss`**. |
-| `storage.secret_access_key` | String | `None` | The secret access key of the aws account.<br/>It's **highly recommended** to use AWS IAM roles instead of hardcoding the access key id and secret key.<br/>**It's only used when the storage type is `S3`**. |
-| `storage.access_key_secret` | String | `None` | The secret access key of the aliyun account.<br/>**It's only used when the storage type is `Oss`**. |
-| `storage.account_name` | String | `None` | The account key of the azure account.<br/>**It's only used when the storage type is `Azblob`**. |
-| `storage.account_key` | String | `None` | The account key of the azure account.<br/>**It's only used when the storage type is `Azblob`**. |
-| `storage.scope` | String | `None` | The scope of the google cloud storage.<br/>**It's only used when the storage type is `Gcs`**. |
-| `storage.credential_path` | String | `None` | The credential path of the google cloud storage.<br/>**It's only used when the storage type is `Gcs`**. |
-| `storage.credential` | String | `None` | The credential of the google cloud storage.<br/>**It's only used when the storage type is `Gcs`**. |
-| `storage.container` | String | `None` | The container of the azure account.<br/>**It's only used when the storage type is `Azblob`**. |
-| `storage.sas_token` | String | `None` | The sas token of the azure account.<br/>**It's only used when the storage type is `Azblob`**. |
-| `storage.endpoint` | String | `None` | The endpoint of the S3 service.<br/>**It's only used when the storage type is `S3`, `Oss`, `Gcs` and `Azblob`**. |
-| `storage.region` | String | `None` | The region of the S3 service.<br/>**It's only used when the storage type is `S3`, `Oss`, `Gcs` and `Azblob`**. |
+| `storage.cache_path` | String | Unset | Cache configuration for object storage such as 'S3' etc.<br/>The local file cache directory. |
+| `storage.cache_capacity` | String | Unset | The local file cache capacity in bytes. |
+| `storage.bucket` | String | Unset | The S3 bucket name.<br/>**It's only used when the storage type is `S3`, `Oss` and `Gcs`**. |
+| `storage.root` | String | Unset | The S3 data will be stored in the specified prefix, for example, `s3://${bucket}/${root}`.<br/>**It's only used when the storage type is `S3`, `Oss` and `Azblob`**. |
+| `storage.access_key_id` | String | Unset | The access key id of the aws account.<br/>It's **highly recommended** to use AWS IAM roles instead of hardcoding the access key id and secret key.<br/>**It's only used when the storage type is `S3` and `Oss`**. |
+| `storage.secret_access_key` | String | Unset | The secret access key of the aws account.<br/>It's **highly recommended** to use AWS IAM roles instead of hardcoding the access key id and secret key.<br/>**It's only used when the storage type is `S3`**. |
+| `storage.access_key_secret` | String | Unset | The secret access key of the aliyun account.<br/>**It's only used when the storage type is `Oss`**. |
+| `storage.account_name` | String | Unset | The account key of the azure account.<br/>**It's only used when the storage type is `Azblob`**. |
+| `storage.account_key` | String | Unset | The account key of the azure account.<br/>**It's only used when the storage type is `Azblob`**. |
+| `storage.scope` | String | Unset | The scope of the google cloud storage.<br/>**It's only used when the storage type is `Gcs`**. |
+| `storage.credential_path` | String | Unset | The credential path of the google cloud storage.<br/>**It's only used when the storage type is `Gcs`**. |
+| `storage.credential` | String | Unset | The credential of the google cloud storage.<br/>**It's only used when the storage type is `Gcs`**. |
+| `storage.container` | String | Unset | The container of the azure account.<br/>**It's only used when the storage type is `Azblob`**. |
+| `storage.sas_token` | String | Unset | The sas token of the azure account.<br/>**It's only used when the storage type is `Azblob`**. |
+| `storage.endpoint` | String | Unset | The endpoint of the S3 service.<br/>**It's only used when the storage type is `S3`, `Oss`, `Gcs` and `Azblob`**. |
+| `storage.region` | String | Unset | The region of the S3 service.<br/>**It's only used when the storage type is `S3`, `Oss`, `Gcs` and `Azblob`**. |
 | `[[region_engine]]` | -- | -- | The region engine options. You can configure multiple region engines. |
 | `region_engine.mito` | -- | -- | The Mito engine options. |
 | `region_engine.mito.num_workers` | Integer | `8` | Number of region workers. |
@@ -416,18 +439,20 @@
 | `region_engine.mito.worker_request_batch_size` | Integer | `64` | Max batch size for a worker to handle requests. |
 | `region_engine.mito.manifest_checkpoint_distance` | Integer | `10` | Number of meta action updated to trigger a new checkpoint for the manifest. |
 | `region_engine.mito.compress_manifest` | Bool | `false` | Whether to compress manifest and checkpoint file by gzip (default false). |
-| `region_engine.mito.max_background_jobs` | Integer | `4` | Max number of running background jobs |
+| `region_engine.mito.max_background_flushes` | Integer | Auto | Max number of running background flush jobs (default: 1/2 of cpu cores). |
+| `region_engine.mito.max_background_compactions` | Integer | Auto | Max number of running background compaction jobs (default: 1/4 of cpu cores). |
+| `region_engine.mito.max_background_purges` | Integer | Auto | Max number of running background purge jobs (default: number of cpu cores). |
 | `region_engine.mito.auto_flush_interval` | String | `1h` | Interval to auto flush a region if it has not flushed yet. |
-| `region_engine.mito.global_write_buffer_size` | String | `1GB` | Global write buffer size for all regions. If not set, it's default to 1/8 of OS memory with a max limitation of 1GB. |
-| `region_engine.mito.global_write_buffer_reject_size` | String | `2GB` | Global write buffer size threshold to reject write requests. If not set, it's default to 2 times of `global_write_buffer_size` |
-| `region_engine.mito.sst_meta_cache_size` | String | `128MB` | Cache size for SST metadata. Setting it to 0 to disable the cache.<br/>If not set, it's default to 1/32 of OS memory with a max limitation of 128MB. |
-| `region_engine.mito.vector_cache_size` | String | `512MB` | Cache size for vectors and arrow arrays. Setting it to 0 to disable the cache.<br/>If not set, it's default to 1/16 of OS memory with a max limitation of 512MB. |
-| `region_engine.mito.page_cache_size` | String | `512MB` | Cache size for pages of SST row groups. Setting it to 0 to disable the cache.<br/>If not set, it's default to 1/8 of OS memory. |
-| `region_engine.mito.selector_result_cache_size` | String | `512MB` | Cache size for time series selector (e.g. `last_value()`). Setting it to 0 to disable the cache.<br/>If not set, it's default to 1/16 of OS memory with a max limitation of 512MB. |
+| `region_engine.mito.global_write_buffer_size` | String | Auto | Global write buffer size for all regions. If not set, it's default to 1/8 of OS memory with a max limitation of 1GB. |
+| `region_engine.mito.global_write_buffer_reject_size` | String | Auto | Global write buffer size threshold to reject write requests. If not set, it's default to 2 times of `global_write_buffer_size` |
+| `region_engine.mito.sst_meta_cache_size` | String | Auto | Cache size for SST metadata. Setting it to 0 to disable the cache.<br/>If not set, it's default to 1/32 of OS memory with a max limitation of 128MB. |
+| `region_engine.mito.vector_cache_size` | String | Auto | Cache size for vectors and arrow arrays. Setting it to 0 to disable the cache.<br/>If not set, it's default to 1/16 of OS memory with a max limitation of 512MB. |
+| `region_engine.mito.page_cache_size` | String | Auto | Cache size for pages of SST row groups. Setting it to 0 to disable the cache.<br/>If not set, it's default to 1/8 of OS memory. |
+| `region_engine.mito.selector_result_cache_size` | String | Auto | Cache size for time series selector (e.g. `last_value()`). Setting it to 0 to disable the cache.<br/>If not set, it's default to 1/16 of OS memory with a max limitation of 512MB. |
 | `region_engine.mito.enable_experimental_write_cache` | Bool | `false` | Whether to enable the experimental write cache. |
 | `region_engine.mito.experimental_write_cache_path` | String | `""` | File system path for write cache, defaults to `{data_home}/write_cache`. |
 | `region_engine.mito.experimental_write_cache_size` | String | `512MB` | Capacity for write cache. |
-| `region_engine.mito.experimental_write_cache_ttl` | String | `None` | TTL for write cache. |
+| `region_engine.mito.experimental_write_cache_ttl` | String | Unset | TTL for write cache. |
 | `region_engine.mito.sst_write_buffer_size` | String | `8MB` | Buffer size for SST writing. |
 | `region_engine.mito.scan_parallelism` | Integer | `0` | Parallelism to scan a region (default: 1/4 of cpu cores).<br/>- `0`: using the default value (1/4 of cpu cores).<br/>- `1`: scan in current thread.<br/>- `n`: scan in parallelism n. |
 | `region_engine.mito.parallel_scan_channel_size` | Integer | `32` | Capacity of the channel to send data from parallel scan tasks to the main task. |
@@ -455,23 +480,28 @@
 | `region_engine.file` | -- | -- | Enable the file engine. |
 | `logging` | -- | -- | The logging options. |
 | `logging.dir` | String | `/tmp/greptimedb/logs` | The directory to store the log files. If set to empty, logs will not be written to files. |
-| `logging.level` | String | `None` | The log level. Can be `info`/`debug`/`warn`/`error`. |
+| `logging.level` | String | Unset | The log level. Can be `info`/`debug`/`warn`/`error`. |
 | `logging.enable_otlp_tracing` | Bool | `false` | Enable OTLP tracing. |
 | `logging.otlp_endpoint` | String | `http://localhost:4317` | The OTLP tracing endpoint. |
 | `logging.append_stdout` | Bool | `true` | Whether to append logs to stdout. |
 | `logging.log_format` | String | `text` | The log format. Can be `text`/`json`. |
+| `logging.max_log_files` | Integer | `720` | The maximum amount of log files. |
 | `logging.tracing_sample_ratio` | -- | -- | The percentage of tracing will be sampled and exported.<br/>Valid range `[0, 1]`, 1 means all traces are sampled, 0 means all traces are not sampled, the default value is 1.<br/>ratio > 1 are treated as 1. Fractions < 0 are treated as 0 |
 | `logging.tracing_sample_ratio.default_ratio` | Float | `1.0` | -- |
+| `logging.slow_query` | -- | -- | The slow query log options. |
+| `logging.slow_query.enable` | Bool | `false` | Whether to enable slow query log. |
+| `logging.slow_query.threshold` | String | Unset | The threshold of slow query. |
+| `logging.slow_query.sample_ratio` | Float | Unset | The sampling ratio of slow query log. The value should be in the range of (0, 1]. |
 | `export_metrics` | -- | -- | The datanode can export its metrics and send to Prometheus compatible service (e.g. send to `greptimedb` itself) from remote-write API.<br/>This is only used for `greptimedb` to export its own metrics internally. It's different from prometheus scrape. |
 | `export_metrics.enable` | Bool | `false` | whether enable export metrics. |
 | `export_metrics.write_interval` | String | `30s` | The interval of export metrics. |
 | `export_metrics.self_import` | -- | -- | For `standalone` mode, `self_import` is recommend to collect metrics generated by itself<br/>You must create the database before enabling it. |
-| `export_metrics.self_import.db` | String | `None` | -- |
+| `export_metrics.self_import.db` | String | Unset | -- |
 | `export_metrics.remote_write` | -- | -- | -- |
 | `export_metrics.remote_write.url` | String | `""` | The url the metrics send to. The url example can be: `http://127.0.0.1:4000/v1/prometheus/write?db=greptime_metrics`. |
 | `export_metrics.remote_write.headers` | InlineTable | -- | HTTP headers of Prometheus remote-write carry. |
 | `tracing` | -- | -- | The tracing options. Only effect when compiled with `tokio-console` feature. |
-| `tracing.tokio_console_addr` | String | `None` | The tokio console address. |
+| `tracing.tokio_console_addr` | String | Unset | The tokio console address. |


 ### Flownode
@@ -479,7 +509,7 @@
 | Key | Type | Default | Descriptions |
 | --- | -----| ------- | ----------- |
 | `mode` | String | `distributed` | The running mode of the flownode. It can be `standalone` or `distributed`. |
-| `node_id` | Integer | `None` | The flownode identifier and should be unique in the cluster. |
+| `node_id` | Integer | Unset | The flownode identifier and should be unique in the cluster. |
 | `grpc` | -- | -- | The gRPC server options. |
 | `grpc.addr` | String | `127.0.0.1:6800` | The address to bind the gRPC server. |
 | `grpc.hostname` | String | `127.0.0.1` | The hostname advertised to the metasrv,<br/>and used for connections from outside the host |
@@ -501,12 +531,17 @@
 | `heartbeat.retry_interval` | String | `3s` | Interval for retrying to send heartbeat messages to the metasrv. |
 | `logging` | -- | -- | The logging options. |
 | `logging.dir` | String | `/tmp/greptimedb/logs` | The directory to store the log files. If set to empty, logs will not be written to files. |
-| `logging.level` | String | `None` | The log level. Can be `info`/`debug`/`warn`/`error`. |
+| `logging.level` | String | Unset | The log level. Can be `info`/`debug`/`warn`/`error`. |
 | `logging.enable_otlp_tracing` | Bool | `false` | Enable OTLP tracing. |
 | `logging.otlp_endpoint` | String | `http://localhost:4317` | The OTLP tracing endpoint. |
 | `logging.append_stdout` | Bool | `true` | Whether to append logs to stdout. |
 | `logging.log_format` | String | `text` | The log format. Can be `text`/`json`. |
+| `logging.max_log_files` | Integer | `720` | The maximum amount of log files. |
 | `logging.tracing_sample_ratio` | -- | -- | The percentage of tracing will be sampled and exported.<br/>Valid range `[0, 1]`, 1 means all traces are sampled, 0 means all traces are not sampled, the default value is 1.<br/>ratio > 1 are treated as 1. Fractions < 0 are treated as 0 |
 | `logging.tracing_sample_ratio.default_ratio` | Float | `1.0` | -- |
+| `logging.slow_query` | -- | -- | The slow query log options. |
+| `logging.slow_query.enable` | Bool | `false` | Whether to enable slow query log. |
+| `logging.slow_query.threshold` | String | Unset | The threshold of slow query. |
+| `logging.slow_query.sample_ratio` | Float | Unset | The sampling ratio of slow query log. The value should be in the range of (0, 1]. |
 | `tracing` | -- | -- | The tracing options. Only effect when compiled with `tokio-console` feature. |
-| `tracing.tokio_console_addr` | String | `None` | The tokio console address. |
+| `tracing.tokio_console_addr` | String | Unset | The tokio console address. |
--- a/config/datanode.example.toml
+++ b/config/datanode.example.toml
@@ -2,7 +2,7 @@
 mode = "standalone"

 ## The datanode identifier and should be unique in the cluster.
-## +toml2docs:none-default
+## @toml2docs:none-default
 node_id = 42

 ## Start services after regions have obtained leases.
@@ -19,24 +19,27 @@ enable_telemetry = true
 ## Parallelism of initializing regions.
 init_regions_parallelism = 16

+## The maximum current queries allowed to be executed. Zero means unlimited.
+max_concurrent_queries = 0
+
 ## Deprecated, use `grpc.addr` instead.
-## +toml2docs:none-default
+## @toml2docs:none-default
 rpc_addr = "127.0.0.1:3001"

 ## Deprecated, use `grpc.hostname` instead.
-## +toml2docs:none-default
+## @toml2docs:none-default
 rpc_hostname = "127.0.0.1"

 ## Deprecated, use `grpc.runtime_size` instead.
-## +toml2docs:none-default
+## @toml2docs:none-default
 rpc_runtime_size = 8

 ## Deprecated, use `grpc.rpc_max_recv_message_size` instead.
-## +toml2docs:none-default
+## @toml2docs:none-default
 rpc_max_recv_message_size = "512MB"

 ## Deprecated, use `grpc.rpc_max_send_message_size` instead.
-## +toml2docs:none-default
+## @toml2docs:none-default
 rpc_max_send_message_size = "512MB"


@@ -71,11 +74,11 @@ max_send_message_size = "512MB"
 mode = "disable"

 ## Certificate file path.
-## +toml2docs:none-default
+## @toml2docs:none-default
 cert_path = ""

 ## Private key file path.
-## +toml2docs:none-default
+## @toml2docs:none-default
 key_path = ""

 ## Watch for Certificate and key file change and auto reload.
@@ -83,11 +86,11 @@ key_path = ""
 watch = false

 ## The runtime options.
-[runtime]
+#+ [runtime]
 ## The number of threads to execute the runtime for global read operations.
-global_rt_size = 8
+#+ global_rt_size = 8
 ## The number of threads to execute the runtime for global write operations.
-compact_rt_size = 4
+#+ compact_rt_size = 4

 ## The heartbeat options.
 [heartbeat]
@@ -135,7 +138,7 @@ provider = "raft_engine"

 ## The directory to store the WAL files.
 ## **It's only used when the provider is `raft_engine`**.
-## +toml2docs:none-default
+## @toml2docs:none-default
 dir = "/tmp/greptimedb/wal"

 ## The size of the WAL segment file.
@@ -170,6 +173,9 @@ prefill_log_files = false
 ## **It's only used when the provider is `raft_engine`**.
 sync_period = "10s"

+## Parallelism during WAL recovery.
+recovery_parallelism = 2
+
 ## The Kafka broker endpoints.
 ## **It's only used when the provider is `kafka`**.
 broker_endpoints = ["127.0.0.1:9092"]
@@ -207,6 +213,17 @@ create_index = true
 ## **It's only used when the provider is `kafka`**.
 dump_index_interval = "60s"

+## Ignore missing entries during read WAL.
+## **It's only used when the provider is `kafka`**.
+##
+## This option ensures that when Kafka messages are deleted, the system
+## can still successfully replay memtable data without throwing an
+## out-of-range error.
+## However, enabling this option might lead to unexpected data loss,
+## as the system will skip over missing entries instead of treating
+## them as critical errors.
+overwrite_entry_start_id = false
+
 # The Kafka SASL configuration.
 # **It's only used when the provider is `kafka`**.
 # Available SASL mechanisms:
@@ -279,83 +296,83 @@ type = "File"

 ## Cache configuration for object storage such as 'S3' etc.
 ## The local file cache directory.
-## +toml2docs:none-default
+## @toml2docs:none-default
 cache_path = "/path/local_cache"

 ## The local file cache capacity in bytes.
-## +toml2docs:none-default
+## @toml2docs:none-default
 cache_capacity = "256MB"

 ## The S3 bucket name.
 ## **It's only used when the storage type is `S3`, `Oss` and `Gcs`**.
-## +toml2docs:none-default
+## @toml2docs:none-default
 bucket = "greptimedb"

 ## The S3 data will be stored in the specified prefix, for example, `s3://${bucket}/${root}`.
 ## **It's only used when the storage type is `S3`, `Oss` and `Azblob`**.
-## +toml2docs:none-default
+## @toml2docs:none-default
 root = "greptimedb"

 ## The access key id of the aws account.
 ## It's **highly recommended** to use AWS IAM roles instead of hardcoding the access key id and secret key.
 ## **It's only used when the storage type is `S3` and `Oss`**.
-## +toml2docs:none-default
+## @toml2docs:none-default
 access_key_id = "test"

 ## The secret access key of the aws account.
 ## It's **highly recommended** to use AWS IAM roles instead of hardcoding the access key id and secret key.
 ## **It's only used when the storage type is `S3`**.
-## +toml2docs:none-default
+## @toml2docs:none-default
 secret_access_key = "test"

 ## The secret access key of the aliyun account.
 ## **It's only used when the storage type is `Oss`**.
-## +toml2docs:none-default
+## @toml2docs:none-default
 access_key_secret = "test"

 ## The account key of the azure account.
 ## **It's only used when the storage type is `Azblob`**.
-## +toml2docs:none-default
+## @toml2docs:none-default
 account_name = "test"

 ## The account key of the azure account.
 ## **It's only used when the storage type is `Azblob`**.
-## +toml2docs:none-default
+## @toml2docs:none-default
 account_key = "test"

 ## The scope of the google cloud storage.
 ## **It's only used when the storage type is `Gcs`**.
-## +toml2docs:none-default
+## @toml2docs:none-default
 scope = "test"

 ## The credential path of the google cloud storage.
 ## **It's only used when the storage type is `Gcs`**.
-## +toml2docs:none-default
+## @toml2docs:none-default
 credential_path = "test"

 ## The credential of the google cloud storage.
 ## **It's only used when the storage type is `Gcs`**.
-## +toml2docs:none-default
+## @toml2docs:none-default
 credential = "base64-credential"

 ## The container of the azure account.
 ## **It's only used when the storage type is `Azblob`**.
-## +toml2docs:none-default
+## @toml2docs:none-default
 container = "greptimedb"

 ## The sas token of the azure account.
 ## **It's only used when the storage type is `Azblob`**.
-## +toml2docs:none-default
+## @toml2docs:none-default
 sas_token = ""

 ## The endpoint of the S3 service.
 ## **It's only used when the storage type is `S3`, `Oss`, `Gcs` and `Azblob`**.
-## +toml2docs:none-default
+## @toml2docs:none-default
 endpoint = "https://s3.amazonaws.com"

 ## The region of the S3 service.
 ## **It's only used when the storage type is `S3`, `Oss`, `Gcs` and `Azblob`**.
-## +toml2docs:none-default
+## @toml2docs:none-default
 region = "us-west-2"

 # Custom storage options
@@ -385,7 +402,7 @@ region = "us-west-2"
 [region_engine.mito]

 ## Number of region workers.
-num_workers = 8
+#+ num_workers = 8

 ## Request channel size of each worker.
 worker_channel_size = 128
@@ -399,33 +416,48 @@ manifest_checkpoint_distance = 10
 ## Whether to compress manifest and checkpoint file by gzip (default false).
 compress_manifest = false

-## Max number of running background jobs
-max_background_jobs = 4
+## Max number of running background flush jobs (default: 1/2 of cpu cores).
+## @toml2docs:none-default="Auto"
+#+ max_background_flushes = 4
+
+## Max number of running background compaction jobs (default: 1/4 of cpu cores).
+## @toml2docs:none-default="Auto"
+#+ max_background_compactions = 2
+
+## Max number of running background purge jobs (default: number of cpu cores).
+## @toml2docs:none-default="Auto"
+#+ max_background_purges = 8

 ## Interval to auto flush a region if it has not flushed yet.
 auto_flush_interval = "1h"

 ## Global write buffer size for all regions. If not set, it's default to 1/8 of OS memory with a max limitation of 1GB.
-global_write_buffer_size = "1GB"
+## @toml2docs:none-default="Auto"
+#+ global_write_buffer_size = "1GB"

 ## Global write buffer size threshold to reject write requests. If not set, it's default to 2 times of `global_write_buffer_size`
-global_write_buffer_reject_size = "2GB"
+## @toml2docs:none-default="Auto"
+#+ global_write_buffer_reject_size = "2GB"

 ## Cache size for SST metadata. Setting it to 0 to disable the cache.
 ## If not set, it's default to 1/32 of OS memory with a max limitation of 128MB.
-sst_meta_cache_size = "128MB"
+## @toml2docs:none-default="Auto"
+#+ sst_meta_cache_size = "128MB"

 ## Cache size for vectors and arrow arrays. Setting it to 0 to disable the cache.
 ## If not set, it's default to 1/16 of OS memory with a max limitation of 512MB.
-vector_cache_size = "512MB"
+## @toml2docs:none-default="Auto"
+#+ vector_cache_size = "512MB"

 ## Cache size for pages of SST row groups. Setting it to 0 to disable the cache.
 ## If not set, it's default to 1/8 of OS memory.
-page_cache_size = "512MB"
+## @toml2docs:none-default="Auto"
+#+ page_cache_size = "512MB"

 ## Cache size for time series selector (e.g. `last_value()`). Setting it to 0 to disable the cache.
 ## If not set, it's default to 1/16 of OS memory with a max limitation of 512MB.
-selector_result_cache_size = "512MB"
+## @toml2docs:none-default="Auto"
+#+ selector_result_cache_size = "512MB"

 ## Whether to enable the experimental write cache.
 enable_experimental_write_cache = false
@@ -437,7 +469,7 @@ experimental_write_cache_path = ""
 experimental_write_cache_size = "512MB"

 ## TTL for write cache.
-## +toml2docs:none-default
+## @toml2docs:none-default
 experimental_write_cache_ttl = "8h"

 ## Buffer size for SST writing.
@@ -553,7 +585,7 @@ fork_dictionary_bytes = "1GiB"
 dir = "/tmp/greptimedb/logs"

 ## The log level. Can be `info`/`debug`/`warn`/`error`.
-## +toml2docs:none-default
+## @toml2docs:none-default
 level = "info"

 ## Enable OTLP tracing.
@@ -568,12 +600,28 @@ append_stdout = true
 ## The log format. Can be `text`/`json`.
 log_format = "text"

+## The maximum amount of log files.
+max_log_files = 720
+
 ## The percentage of tracing will be sampled and exported.
 ## Valid range `[0, 1]`, 1 means all traces are sampled, 0 means all traces are not sampled, the default value is 1.
 ## ratio > 1 are treated as 1. Fractions < 0 are treated as 0
 [logging.tracing_sample_ratio]
 default_ratio = 1.0

+## The slow query log options.
+[logging.slow_query]
+## Whether to enable slow query log.
+enable = false
+
+## The threshold of slow query.
+## @toml2docs:none-default
+threshold = "10s"
+
+## The sampling ratio of slow query log. The value should be in the range of (0, 1].
+## @toml2docs:none-default
+sample_ratio = 1.0
+
 ## The datanode can export its metrics and send to Prometheus compatible service (e.g. send to `greptimedb` itself) from remote-write API.
 ## This is only used for `greptimedb` to export its own metrics internally. It's different from prometheus scrape.
 [export_metrics]
@@ -587,7 +635,7 @@ write_interval = "30s"
 ## For `standalone` mode, `self_import` is recommend to collect metrics generated by itself
 ## You must create the database before enabling it.
 [export_metrics.self_import]
-## +toml2docs:none-default
+## @toml2docs:none-default
 db = "greptime_metrics"

 [export_metrics.remote_write]
@@ -598,7 +646,7 @@ url = ""
 headers = { }

 ## The tracing options. Only effect when compiled with `tokio-console` feature.
-[tracing]
+#+ [tracing]
 ## The tokio console address.
-## +toml2docs:none-default
-tokio_console_addr = "127.0.0.1"
+## @toml2docs:none-default
+#+ tokio_console_addr = "127.0.0.1"
--- a/config/flownode.example.toml
+++ b/config/flownode.example.toml
@@ -2,7 +2,7 @@
 mode = "distributed"

 ## The flownode identifier and should be unique in the cluster.
-## +toml2docs:none-default
+## @toml2docs:none-default
 node_id = 14

 ## The gRPC server options.
@@ -63,7 +63,7 @@ retry_interval = "3s"
 dir = "/tmp/greptimedb/logs"

 ## The log level. Can be `info`/`debug`/`warn`/`error`.
-## +toml2docs:none-default
+## @toml2docs:none-default
 level = "info"

 ## Enable OTLP tracing.
@@ -78,15 +78,31 @@ append_stdout = true
 ## The log format. Can be `text`/`json`.
 log_format = "text"

+## The maximum amount of log files.
+max_log_files = 720
+
 ## The percentage of tracing will be sampled and exported.
 ## Valid range `[0, 1]`, 1 means all traces are sampled, 0 means all traces are not sampled, the default value is 1.
 ## ratio > 1 are treated as 1. Fractions < 0 are treated as 0
 [logging.tracing_sample_ratio]
 default_ratio = 1.0

-## The tracing options. Only effect when compiled with `tokio-console` feature.
-[tracing]
-## The tokio console address.
-## +toml2docs:none-default
-tokio_console_addr = "127.0.0.1"
+## The slow query log options.
+[logging.slow_query]
+## Whether to enable slow query log.
+enable = false
+
+## The threshold of slow query.
+## @toml2docs:none-default
+threshold = "10s"
+
+## The sampling ratio of slow query log. The value should be in the range of (0, 1].
+## @toml2docs:none-default
+sample_ratio = 1.0
+
+## The tracing options. Only effect when compiled with `tokio-console` feature.
+#+ [tracing]
+## The tokio console address.
+## @toml2docs:none-default
+#+ tokio_console_addr = "127.0.0.1"

--- a/config/frontend.example.toml
+++ b/config/frontend.example.toml
@@ -1,13 +1,13 @@
 ## The default timezone of the server.
-## +toml2docs:none-default
+## @toml2docs:none-default
 default_timezone = "UTC"

 ## The runtime options.
-[runtime]
+#+ [runtime]
 ## The number of threads to execute the runtime for global read operations.
-global_rt_size = 8
+#+ global_rt_size = 8
 ## The number of threads to execute the runtime for global write operations.
-compact_rt_size = 4
+#+ compact_rt_size = 4

 ## The heartbeat options.
 [heartbeat]
@@ -44,11 +44,11 @@ runtime_size = 8
 mode = "disable"

 ## Certificate file path.
-## +toml2docs:none-default
+## @toml2docs:none-default
 cert_path = ""

 ## Private key file path.
-## +toml2docs:none-default
+## @toml2docs:none-default
 key_path = ""

 ## Watch for Certificate and key file change and auto reload.
@@ -76,11 +76,11 @@ runtime_size = 2
 mode = "disable"

 ## Certificate file path.
-## +toml2docs:none-default
+## @toml2docs:none-default
 cert_path = ""

 ## Private key file path.
-## +toml2docs:none-default
+## @toml2docs:none-default
 key_path = ""

 ## Watch for Certificate and key file change and auto reload
@@ -101,11 +101,11 @@ runtime_size = 2
 mode = "disable"

 ## Certificate file path.
-## +toml2docs:none-default
+## @toml2docs:none-default
 cert_path = ""

 ## Private key file path.
-## +toml2docs:none-default
+## @toml2docs:none-default
 key_path = ""

 ## Watch for Certificate and key file change and auto reload
@@ -170,7 +170,7 @@ tcp_nodelay = true
 dir = "/tmp/greptimedb/logs"

 ## The log level. Can be `info`/`debug`/`warn`/`error`.
-## +toml2docs:none-default
+## @toml2docs:none-default
 level = "info"

 ## Enable OTLP tracing.
@@ -185,12 +185,28 @@ append_stdout = true
 ## The log format. Can be `text`/`json`.
 log_format = "text"

+## The maximum amount of log files.
+max_log_files = 720
+
 ## The percentage of tracing will be sampled and exported.
 ## Valid range `[0, 1]`, 1 means all traces are sampled, 0 means all traces are not sampled, the default value is 1.
 ## ratio > 1 are treated as 1. Fractions < 0 are treated as 0
 [logging.tracing_sample_ratio]
 default_ratio = 1.0

+## The slow query log options.
+[logging.slow_query]
+## Whether to enable slow query log.
+enable = false
+
+## The threshold of slow query.
+## @toml2docs:none-default
+threshold = "10s"
+
+## The sampling ratio of slow query log. The value should be in the range of (0, 1].
+## @toml2docs:none-default
+sample_ratio = 1.0
+
 ## The datanode can export its metrics and send to Prometheus compatible service (e.g. send to `greptimedb` itself) from remote-write API.
 ## This is only used for `greptimedb` to export its own metrics internally. It's different from prometheus scrape.
 [export_metrics]
@@ -204,7 +220,7 @@ write_interval = "30s"
 ## For `standalone` mode, `self_import` is recommend to collect metrics generated by itself
 ## You must create the database before enabling it.
 [export_metrics.self_import]
-## +toml2docs:none-default
+## @toml2docs:none-default
 db = "greptime_metrics"

 [export_metrics.remote_write]
@@ -215,7 +231,7 @@ url = ""
 headers = { }

 ## The tracing options. Only effect when compiled with `tokio-console` feature.
-[tracing]
+#+ [tracing]
 ## The tokio console address.
-## +toml2docs:none-default
-tokio_console_addr = "127.0.0.1"
+## @toml2docs:none-default
+#+ tokio_console_addr = "127.0.0.1"
--- a/config/metasrv.example.toml
+++ b/config/metasrv.example.toml
@@ -36,11 +36,11 @@ enable_region_failover = false
 backend = "EtcdStore"

 ## The runtime options.
-[runtime]
+#+ [runtime]
 ## The number of threads to execute the runtime for global read operations.
-global_rt_size = 8
+#+ global_rt_size = 8
 ## The number of threads to execute the runtime for global write operations.
-compact_rt_size = 4
+#+ compact_rt_size = 4

 ## Procedure storage options.
 [procedure]
@@ -157,7 +157,7 @@ backoff_deadline = "5mins"
 dir = "/tmp/greptimedb/logs"

 ## The log level. Can be `info`/`debug`/`warn`/`error`.
-## +toml2docs:none-default
+## @toml2docs:none-default
 level = "info"

 ## Enable OTLP tracing.
@@ -172,12 +172,28 @@ append_stdout = true
 ## The log format. Can be `text`/`json`.
 log_format = "text"

+## The maximum amount of log files.
+max_log_files = 720
+
 ## The percentage of tracing will be sampled and exported.
 ## Valid range `[0, 1]`, 1 means all traces are sampled, 0 means all traces are not sampled, the default value is 1.
 ## ratio > 1 are treated as 1. Fractions < 0 are treated as 0
 [logging.tracing_sample_ratio]
 default_ratio = 1.0

+## The slow query log options.
+[logging.slow_query]
+## Whether to enable slow query log.
+enable = false
+
+## The threshold of slow query.
+## @toml2docs:none-default
+threshold = "10s"
+
+## The sampling ratio of slow query log. The value should be in the range of (0, 1].
+## @toml2docs:none-default
+sample_ratio = 1.0
+
 ## The datanode can export its metrics and send to Prometheus compatible service (e.g. send to `greptimedb` itself) from remote-write API.
 ## This is only used for `greptimedb` to export its own metrics internally. It's different from prometheus scrape.
 [export_metrics]
@@ -191,7 +207,7 @@ write_interval = "30s"
 ## For `standalone` mode, `self_import` is recommend to collect metrics generated by itself
 ## You must create the database before enabling it.
 [export_metrics.self_import]
-## +toml2docs:none-default
+## @toml2docs:none-default
 db = "greptime_metrics"

 [export_metrics.remote_write]
@@ -202,7 +218,7 @@ url = ""
 headers = { }

 ## The tracing options. Only effect when compiled with `tokio-console` feature.
-[tracing]
+#+ [tracing]
 ## The tokio console address.
-## +toml2docs:none-default
-tokio_console_addr = "127.0.0.1"
+## @toml2docs:none-default
+#+ tokio_console_addr = "127.0.0.1"
--- a/config/standalone.example.toml
+++ b/config/standalone.example.toml
@@ -5,7 +5,7 @@ mode = "standalone"
 enable_telemetry = true

 ## The default timezone of the server.
-## +toml2docs:none-default
+## @toml2docs:none-default
 default_timezone = "UTC"

 ## Initialize all regions in the background during the startup.
@@ -15,12 +15,15 @@ init_regions_in_background = false
 ## Parallelism of initializing regions.
 init_regions_parallelism = 16

+## The maximum current queries allowed to be executed. Zero means unlimited.
+max_concurrent_queries = 0
+
 ## The runtime options.
-[runtime]
+#+ [runtime]
 ## The number of threads to execute the runtime for global read operations.
-global_rt_size = 8
+#+ global_rt_size = 8
 ## The number of threads to execute the runtime for global write operations.
-compact_rt_size = 4
+#+ compact_rt_size = 4

 ## The HTTP server options.
 [http]
@@ -46,11 +49,11 @@ runtime_size = 8
 mode = "disable"

 ## Certificate file path.
-## +toml2docs:none-default
+## @toml2docs:none-default
 cert_path = ""

 ## Private key file path.
-## +toml2docs:none-default
+## @toml2docs:none-default
 key_path = ""

 ## Watch for Certificate and key file change and auto reload.
@@ -78,11 +81,11 @@ runtime_size = 2
 mode = "disable"

 ## Certificate file path.
-## +toml2docs:none-default
+## @toml2docs:none-default
 cert_path = ""

 ## Private key file path.
-## +toml2docs:none-default
+## @toml2docs:none-default
 key_path = ""

 ## Watch for Certificate and key file change and auto reload
@@ -103,11 +106,11 @@ runtime_size = 2
 mode = "disable"

 ## Certificate file path.
-## +toml2docs:none-default
+## @toml2docs:none-default
 cert_path = ""

 ## Private key file path.
-## +toml2docs:none-default
+## @toml2docs:none-default
 key_path = ""

 ## Watch for Certificate and key file change and auto reload
@@ -139,7 +142,7 @@ provider = "raft_engine"

 ## The directory to store the WAL files.
 ## **It's only used when the provider is `raft_engine`**.
-## +toml2docs:none-default
+## @toml2docs:none-default
 dir = "/tmp/greptimedb/wal"

 ## The size of the WAL segment file.
@@ -174,6 +177,9 @@ prefill_log_files = false
 ## **It's only used when the provider is `raft_engine`**.
 sync_period = "10s"

+## Parallelism during WAL recovery.
+recovery_parallelism = 2
+
 ## The Kafka broker endpoints.
 ## **It's only used when the provider is `kafka`**.
 broker_endpoints = ["127.0.0.1:9092"]
@@ -231,6 +237,17 @@ backoff_base = 2
 ## **It's only used when the provider is `kafka`**.
 backoff_deadline = "5mins"

+## Ignore missing entries during read WAL.
+## **It's only used when the provider is `kafka`**.
+##
+## This option ensures that when Kafka messages are deleted, the system
+## can still successfully replay memtable data without throwing an
+## out-of-range error.
+## However, enabling this option might lead to unexpected data loss,
+## as the system will skip over missing entries instead of treating
+## them as critical errors.
+overwrite_entry_start_id = false
+
 # The Kafka SASL configuration.
 # **It's only used when the provider is `kafka`**.
 # Available SASL mechanisms:
@@ -317,83 +334,83 @@ type = "File"

 ## Cache configuration for object storage such as 'S3' etc.
 ## The local file cache directory.
-## +toml2docs:none-default
+## @toml2docs:none-default
 cache_path = "/path/local_cache"

 ## The local file cache capacity in bytes.
-## +toml2docs:none-default
+## @toml2docs:none-default
 cache_capacity = "256MB"

 ## The S3 bucket name.
 ## **It's only used when the storage type is `S3`, `Oss` and `Gcs`**.
-## +toml2docs:none-default
+## @toml2docs:none-default
 bucket = "greptimedb"

 ## The S3 data will be stored in the specified prefix, for example, `s3://${bucket}/${root}`.
 ## **It's only used when the storage type is `S3`, `Oss` and `Azblob`**.
-## +toml2docs:none-default
+## @toml2docs:none-default
 root = "greptimedb"

 ## The access key id of the aws account.
 ## It's **highly recommended** to use AWS IAM roles instead of hardcoding the access key id and secret key.
 ## **It's only used when the storage type is `S3` and `Oss`**.
-## +toml2docs:none-default
+## @toml2docs:none-default
 access_key_id = "test"

 ## The secret access key of the aws account.
 ## It's **highly recommended** to use AWS IAM roles instead of hardcoding the access key id and secret key.
 ## **It's only used when the storage type is `S3`**.
-## +toml2docs:none-default
+## @toml2docs:none-default
 secret_access_key = "test"

 ## The secret access key of the aliyun account.
 ## **It's only used when the storage type is `Oss`**.
-## +toml2docs:none-default
+## @toml2docs:none-default
 access_key_secret = "test"

 ## The account key of the azure account.
 ## **It's only used when the storage type is `Azblob`**.
-## +toml2docs:none-default
+## @toml2docs:none-default
 account_name = "test"

 ## The account key of the azure account.
 ## **It's only used when the storage type is `Azblob`**.
-## +toml2docs:none-default
+## @toml2docs:none-default
 account_key = "test"

 ## The scope of the google cloud storage.
 ## **It's only used when the storage type is `Gcs`**.
-## +toml2docs:none-default
+## @toml2docs:none-default
 scope = "test"

 ## The credential path of the google cloud storage.
 ## **It's only used when the storage type is `Gcs`**.
-## +toml2docs:none-default
+## @toml2docs:none-default
 credential_path = "test"

 ## The credential of the google cloud storage.
 ## **It's only used when the storage type is `Gcs`**.
-## +toml2docs:none-default
+## @toml2docs:none-default
 credential = "base64-credential"

 ## The container of the azure account.
 ## **It's only used when the storage type is `Azblob`**.
-## +toml2docs:none-default
+## @toml2docs:none-default
 container = "greptimedb"

 ## The sas token of the azure account.
 ## **It's only used when the storage type is `Azblob`**.
-## +toml2docs:none-default
+## @toml2docs:none-default
 sas_token = ""

 ## The endpoint of the S3 service.
 ## **It's only used when the storage type is `S3`, `Oss`, `Gcs` and `Azblob`**.
-## +toml2docs:none-default
+## @toml2docs:none-default
 endpoint = "https://s3.amazonaws.com"

 ## The region of the S3 service.
 ## **It's only used when the storage type is `S3`, `Oss`, `Gcs` and `Azblob`**.
-## +toml2docs:none-default
+## @toml2docs:none-default
 region = "us-west-2"

 # Custom storage options
@@ -423,7 +440,7 @@ region = "us-west-2"
 [region_engine.mito]

 ## Number of region workers.
-num_workers = 8
+#+ num_workers = 8

 ## Request channel size of each worker.
 worker_channel_size = 128
@@ -437,33 +454,48 @@ manifest_checkpoint_distance = 10
 ## Whether to compress manifest and checkpoint file by gzip (default false).
 compress_manifest = false

-## Max number of running background jobs
-max_background_jobs = 4
+## Max number of running background flush jobs (default: 1/2 of cpu cores).
+## @toml2docs:none-default="Auto"
+#+ max_background_flushes = 4
+
+## Max number of running background compaction jobs (default: 1/4 of cpu cores).
+## @toml2docs:none-default="Auto"
+#+ max_background_compactions = 2
+
+## Max number of running background purge jobs (default: number of cpu cores).
+## @toml2docs:none-default="Auto"
+#+ max_background_purges = 8

 ## Interval to auto flush a region if it has not flushed yet.
 auto_flush_interval = "1h"

 ## Global write buffer size for all regions. If not set, it's default to 1/8 of OS memory with a max limitation of 1GB.
-global_write_buffer_size = "1GB"
+## @toml2docs:none-default="Auto"
+#+ global_write_buffer_size = "1GB"

-## Global write buffer size threshold to reject write requests. If not set, it's default to 2 times of `global_write_buffer_size`
-global_write_buffer_reject_size = "2GB"
+## Global write buffer size threshold to reject write requests. If not set, it's default to 2 times of `global_write_buffer_size`.
+## @toml2docs:none-default="Auto"
+#+ global_write_buffer_reject_size = "2GB"

 ## Cache size for SST metadata. Setting it to 0 to disable the cache.
 ## If not set, it's default to 1/32 of OS memory with a max limitation of 128MB.
-sst_meta_cache_size = "128MB"
+## @toml2docs:none-default="Auto"
+#+ sst_meta_cache_size = "128MB"

 ## Cache size for vectors and arrow arrays. Setting it to 0 to disable the cache.
 ## If not set, it's default to 1/16 of OS memory with a max limitation of 512MB.
-vector_cache_size = "512MB"
+## @toml2docs:none-default="Auto"
+#+ vector_cache_size = "512MB"

 ## Cache size for pages of SST row groups. Setting it to 0 to disable the cache.
 ## If not set, it's default to 1/8 of OS memory.
-page_cache_size = "512MB"
+## @toml2docs:none-default="Auto"
+#+ page_cache_size = "512MB"

 ## Cache size for time series selector (e.g. `last_value()`). Setting it to 0 to disable the cache.
 ## If not set, it's default to 1/16 of OS memory with a max limitation of 512MB.
-selector_result_cache_size = "512MB"
+## @toml2docs:none-default="Auto"
+#+ selector_result_cache_size = "512MB"

 ## Whether to enable the experimental write cache.
 enable_experimental_write_cache = false
@@ -475,7 +507,7 @@ experimental_write_cache_path = ""
 experimental_write_cache_size = "512MB"

 ## TTL for write cache.
-## +toml2docs:none-default
+## @toml2docs:none-default
 experimental_write_cache_ttl = "8h"

 ## Buffer size for SST writing.
@@ -597,7 +629,7 @@ fork_dictionary_bytes = "1GiB"
 dir = "/tmp/greptimedb/logs"

 ## The log level. Can be `info`/`debug`/`warn`/`error`.
-## +toml2docs:none-default
+## @toml2docs:none-default
 level = "info"

 ## Enable OTLP tracing.
@@ -612,12 +644,28 @@ append_stdout = true
 ## The log format. Can be `text`/`json`.
 log_format = "text"

+## The maximum amount of log files.
+max_log_files = 720
+
 ## The percentage of tracing will be sampled and exported.
 ## Valid range `[0, 1]`, 1 means all traces are sampled, 0 means all traces are not sampled, the default value is 1.
 ## ratio > 1 are treated as 1. Fractions < 0 are treated as 0
 [logging.tracing_sample_ratio]
 default_ratio = 1.0

+## The slow query log options.
+[logging.slow_query]
+## Whether to enable slow query log.
+enable = false
+
+## The threshold of slow query.
+## @toml2docs:none-default
+threshold = "10s"
+
+## The sampling ratio of slow query log. The value should be in the range of (0, 1].
+## @toml2docs:none-default
+sample_ratio = 1.0
+
 ## The datanode can export its metrics and send to Prometheus compatible service (e.g. send to `greptimedb` itself) from remote-write API.
 ## This is only used for `greptimedb` to export its own metrics internally. It's different from prometheus scrape.
 [export_metrics]
@@ -628,10 +676,10 @@ enable = false
 ## The interval of export metrics.
 write_interval = "30s"

-## For `standalone` mode, `self_import` is recommend to collect metrics generated by itself
+## For `standalone` mode, `self_import` is recommended to collect metrics generated by itself
 ## You must create the database before enabling it.
 [export_metrics.self_import]
-## +toml2docs:none-default
+## @toml2docs:none-default
 db = "greptime_metrics"

 [export_metrics.remote_write]
@@ -642,7 +690,7 @@ url = ""
 headers = { }

 ## The tracing options. Only effect when compiled with `tokio-console` feature.
-[tracing]
+#+ [tracing]
 ## The tokio console address.
-## +toml2docs:none-default
-tokio_console_addr = "127.0.0.1"
+## @toml2docs:none-default
+#+ tokio_console_addr = "127.0.0.1"
--- a/docker/dev-builder/binstall/pull_binstall.sh
+++ b/docker/dev-builder/binstall/pull_binstall.sh
@@ -0,0 +1,50 @@
+#!/bin/bash
+
+set -euxo pipefail
+
+cd "$(mktemp -d)"
+# Fix version to v1.6.6, this is different than the latest version in original install script in
+# https://raw.githubusercontent.com/cargo-bins/cargo-binstall/main/install-from-binstall-release.sh
+base_url="https://github.com/cargo-bins/cargo-binstall/releases/download/v1.6.6/cargo-binstall-"
+
+os="$(uname -s)"
+if [ "$os" == "Darwin" ]; then
+    url="${base_url}universal-apple-darwin.zip"
+    curl -LO --proto '=https' --tlsv1.2 -sSf "$url"
+    unzip cargo-binstall-universal-apple-darwin.zip
+elif [ "$os" == "Linux" ]; then
+    machine="$(uname -m)"
+    if [ "$machine" == "armv7l" ]; then
+        machine="armv7"
+    fi
+    target="${machine}-unknown-linux-musl"
+    if [ "$machine" == "armv7" ]; then
+        target="${target}eabihf"
+    fi
+
+    url="${base_url}${target}.tgz"
+    curl -L --proto '=https' --tlsv1.2 -sSf "$url" | tar -xvzf -
+elif [ "${OS-}" = "Windows_NT" ]; then
+    machine="$(uname -m)"
+    target="${machine}-pc-windows-msvc"
+    url="${base_url}${target}.zip"
+    curl -LO --proto '=https' --tlsv1.2 -sSf "$url"
+    unzip "cargo-binstall-${target}.zip"
+else
+    echo "Unsupported OS ${os}"
+    exit 1
+fi
+
+./cargo-binstall -y --force cargo-binstall
+
+CARGO_HOME="${CARGO_HOME:-$HOME/.cargo}"
+
+if ! [[ ":$PATH:" == *":$CARGO_HOME/bin:"* ]]; then
+    if [ -n "${CI:-}" ] && [ -n "${GITHUB_PATH:-}" ]; then
+        echo "$CARGO_HOME/bin" >> "$GITHUB_PATH"
+    else
+        echo
+        printf "\033[0;31mYour path is missing %s, you might want to add it.\033[0m\n" "$CARGO_HOME/bin"
+        echo
+    fi
+fi
--- a/docker/dev-builder/centos/Dockerfile
+++ b/docker/dev-builder/centos/Dockerfile
@@ -32,7 +32,9 @@ RUN rustup toolchain install ${RUST_TOOLCHAIN}

 # Install cargo-binstall with a specific version to adapt the current rust toolchain.
 # Note: if we use the latest version, we may encounter the following `use of unstable library feature 'io_error_downcast'` error.
-RUN cargo install cargo-binstall --version 1.6.6 --locked
+# compile from source take too long, so we use the precompiled binary instead
+COPY $DOCKER_BUILD_ROOT/docker/dev-builder/binstall/pull_binstall.sh /usr/local/bin/pull_binstall.sh
+RUN chmod +x /usr/local/bin/pull_binstall.sh && /usr/local/bin/pull_binstall.sh

 # Install nextest.
 RUN cargo binstall cargo-nextest --no-confirm
--- a/docker/dev-builder/ubuntu/Dockerfile
+++ b/docker/dev-builder/ubuntu/Dockerfile
@@ -24,6 +24,15 @@ RUN apt-get update && DEBIAN_FRONTEND=noninteractive apt-get install -y \
    python3.10 \
    python3.10-dev

+# https://github.com/GreptimeTeam/greptimedb/actions/runs/10935485852/job/30357457188#step:3:7106
+# `aws-lc-sys` require gcc >= 10.3.0 to work, hence alias to use gcc-10
+RUN apt-get remove -y gcc-9 g++-9 cpp-9 && \
+    apt-get install -y gcc-10 g++-10 cpp-10 make cmake && \
+    ln -sf /usr/bin/gcc-10 /usr/bin/gcc && ln -sf /usr/bin/g++-10 /usr/bin/g++ && \
+    ln -sf /usr/bin/gcc-10 /usr/bin/cc && \
+    ln -sf /usr/bin/g++-10 /usr/bin/cpp && ln -sf /usr/bin/g++-10 /usr/bin/c++ && \
+    cc --version && gcc --version && g++ --version && cpp --version && c++ --version
+
 # Remove Python 3.8 and install pip.
 RUN apt-get -y purge python3.8 && \
    apt-get -y autoremove && \
@@ -57,7 +66,9 @@ RUN rustup toolchain install ${RUST_TOOLCHAIN}

 # Install cargo-binstall with a specific version to adapt the current rust toolchain.
 # Note: if we use the latest version, we may encounter the following `use of unstable library feature 'io_error_downcast'` error.
-RUN cargo install cargo-binstall --version 1.6.6 --locked
+# compile from source take too long, so we use the precompiled binary instead
+COPY $DOCKER_BUILD_ROOT/docker/dev-builder/binstall/pull_binstall.sh /usr/local/bin/pull_binstall.sh
+RUN chmod +x /usr/local/bin/pull_binstall.sh && /usr/local/bin/pull_binstall.sh

 # Install nextest.
 RUN cargo binstall cargo-nextest --no-confirm
--- a/docs/benchmarks/log/README.md
+++ b/docs/benchmarks/log/README.md
@@ -48,4 +48,4 @@ Please refer to [SQL query](./query.sql) for GreptimeDB and Clickhouse, and [que

 ## Addition
 - You can tune GreptimeDB's configuration to get better performance.
- You can setup GreptimeDB to use S3 as storage, see [here](https://docs.greptime.com/user-guide/operations/configuration/#storage-options).
+- You can setup GreptimeDB to use S3 as storage, see [here](https://docs.greptime.com/user-guide/deployments/configuration#storage-options).
--- a/docs/how-to/how-to-change-log-level-on-the-fly.md
+++ b/docs/how-to/how-to-change-log-level-on-the-fly.md
@@ -0,0 +1,16 @@
+# Change Log Level on the Fly
+
+## HTTP API
+
+example:
+```bash
+curl --data "trace;flow=debug" 127.0.0.1:4000/debug/log_level
+```
+And database will reply with something like:
+```bash
+Log Level changed from Some("info") to "trace;flow=debug"%
+```
+
+The data is a string in the format of `global_level;module1=level1;module2=level2;...` that follow the same rule of `RUST_LOG`. 
+
+The module is the module name of the log, and the level is the log level. The log level can be one of the following: `trace`, `debug`, `info`, `warn`, `error`, `off`(case insensitive).
--- a/src/servers/src/http/pprof/README.md
+++ b/src/servers/src/http/pprof/README.md
@@ -1,15 +1,9 @@
 # Profiling CPU

-## Build GreptimeDB with `pprof` feature
-
-```bash
-cargo build --features=pprof
-```
-
 ## HTTP API
 Sample at 99 Hertz, for 5 seconds, output report in [protobuf format](https://github.com/google/pprof/blob/master/proto/profile.proto).
 ```bash
-curl -s '0:4000/v1/prof/cpu' > /tmp/pprof.out
+curl -s '0:4000/debug/prof/cpu' > /tmp/pprof.out
 ```

 Then you can use `pprof` command with the protobuf file.
@@ -19,10 +13,10 @@ go tool pprof -top /tmp/pprof.out

 Sample at 99 Hertz, for 60 seconds, output report in flamegraph format.
 ```bash
-curl -s '0:4000/v1/prof/cpu?seconds=60&output=flamegraph' > /tmp/pprof.svg
+curl -s '0:4000/debug/prof/cpu?seconds=60&output=flamegraph' > /tmp/pprof.svg
 ```

 Sample at 49 Hertz, for 10 seconds, output report in text format.
 ```bash
-curl -s '0:4000/v1/prof/cpu?seconds=10&frequency=49&output=text' > /tmp/pprof.txt
+curl -s '0:4000/debug/prof/cpu?seconds=10&frequency=49&output=text' > /tmp/pprof.txt
 ```
--- a/docs/how-to/how-to-profile-memory.md
+++ b/docs/how-to/how-to-profile-memory.md
@@ -12,16 +12,10 @@ brew install jemalloc
 sudo apt install libjemalloc-dev
 ```

-### [flamegraph](https://github.com/brendangregg/FlameGraph) 
+### [flamegraph](https://github.com/brendangregg/FlameGraph)

 ```bash
-curl https://raw.githubusercontent.com/brendangregg/FlameGraph/master/flamegraph.pl > ./flamegraph.pl 
-```
-
-### Build GreptimeDB with `mem-prof` feature.
-
-```bash
-cargo build --features=mem-prof
+curl https://raw.githubusercontent.com/brendangregg/FlameGraph/master/flamegraph.pl > ./flamegraph.pl
 ```

 ## Profiling
@@ -35,7 +29,7 @@ MALLOC_CONF=prof:true,lg_prof_interval:28 ./target/debug/greptime standalone sta
 Dump memory profiling data through HTTP API:

 ```bash
-curl localhost:4000/v1/prof/mem > greptime.hprof
+curl localhost:4000/debug/prof/mem > greptime.hprof
 ```

 You can periodically dump profiling data and compare them to find the delta memory usage.
@@ -45,6 +39,9 @@ You can periodically dump profiling data and compare them to find the delta memo
 To create flamegraph according to dumped profiling data:

 ```bash
-jeprof --svg <path_to_greptimedb_binary> --base=<baseline_prof> <profile_data> > output.svg
-```
+sudo apt install -y libjemalloc-dev

+jeprof <path_to_greptime_binary> <profile_data> --collapse | ./flamegraph.pl > mem-prof.svg
+
+jeprof <path_to_greptime_binary> --base <baseline_prof> <profile_data> --collapse | ./flamegraph.pl > output.svg
+```
--- a/docs/logo-text-padding-dark.png
+++ b/docs/logo-text-padding-dark.png
--- a/docs/logo-text-padding.png
+++ b/docs/logo-text-padding.png
--- a/docs/rfcs/2024-08-06-json-datatype.md
+++ b/docs/rfcs/2024-08-06-json-datatype.md
@@ -0,0 +1,197 @@
+---
+Feature Name: Json Datatype
+Tracking Issue: https://github.com/GreptimeTeam/greptimedb/issues/4230
+Date: 2024-8-6
+Author: "Yuhan Wang <profsyb@gmail.com>"
+---
+
+# Summary
+This RFC proposes a method for storing and querying JSON data in the database.
+
+# Motivation
+JSON is widely used across various scenarios. Direct support for writing and querying JSON can significantly enhance the database's flexibility.
+
+# Details
+
+## Storage and Query
+
+GreptimeDB's type system is built on Arrow/DataFusion, where each data type in GreptimeDB corresponds to a data type in Arrow/DataFusion. The proposed JSON type will be implemented on top of the existing `Binary` type, leveraging the current `datatype::value::Value` and `datatype::vectors::BinaryVector` implementations, utilizing the JSONB format as the encoding of JSON data. JSON data is stored and processed similarly to binary data within the storage layer and query engine.
+
+This approach brings problems when dealing with insertions and queries of JSON columns.
+
+## Insertion
+
+Users commonly write JSON data as strings. Thus we need to make conversions between string and JSONB. There are 2 ways to do this:
+
+1. MySQL and PostgreSQL servers provide auto-conversions between strings and JSONB. When a string is inserted into a JSON column, the server will try to parse the string as JSON and convert it to JSONB. The non-JSON strings will be rejected.
+
+2. A function `parse_json` is provided to convert string to JSONB. If the string is not a valid JSON string, the function will return an error.
+
+For example, in MySQL client:
+```SQL
+CREATE TABLE IF NOT EXISTS test (
+    ts TIMESTAMP TIME INDEX,
+    a INT,
+    b JSON
+);
+
+INSERT INTO test VALUES(
+    0,
+    0,
+    '{
+        "name": "jHl2oDDnPc1i2OzlP5Y",
+        "timestamp": "2024-07-25T04:33:11.369386Z",
+        "attributes": { "event_attributes": 48.28667 }
+    }'
+);
+
+INSERT INTO test VALUES(
+    0,
+    0,
+    parse_json('{
+        "name": "jHl2oDDnPc1i2OzlP5Y",
+        "timestamp": "2024-07-25T04:33:11.369386Z",
+        "attributes": { "event_attributes": 48.28667 }
+    }')
+);
+```
+Are both valid.
+
+The dataflow of the insertion process is as follows:
+```
+Insert JSON strings directly through client:
+                                   Parse                       Insert
+        String(Serialized JSON)┌──────────┐Arrow Binary(JSONB)┌──────┐Arrow Binary(JSONB)
+ Client ---------------------->│  Server  │------------------>│ Mito │------------------> Storage
+                               └──────────┘                   └──────┘
+        (Server identifies JSON type and performs auto-conversion)
+
+Insert JSON strings through parse_json function:
+                                                                   Parse                     Insert
+        String(Serialized JSON)┌──────────┐String(Serialized JSON)┌─────┐Arrow Binary(JSONB)┌──────┐Arrow Binary(JSONB)
+ Client ---------------------->│  Server  │---------------------->│ UDF │------------------>│ Mito │------------------> Storage
+                               └──────────┘                       └─────┘                   └──────┘
+                                            (Conversion is performed by UDF inside Query Engine)
+```
+
+Servers identify JSON column through column schema and perform auto-conversions. But when using prepared statements and binding parameters, the corresponding cached plans in datafusion generated by prepared statements cannot identify JSON columns. Under this circumstance, the servers identify JSON columns through the given parameters and perform auto-conversions.
+
+The following is an example of inserting JSON data through prepared statements:
+```Rust
+sqlx::query(
+    "create table test(ts timestamp time index, j json)",
+)
+.execute(&pool)
+.await
+.unwrap();
+
+let json = serde_json::json!({
+    "code": 200,
+    "success": true,
+    "payload": {
+        "features": [
+            "serde",
+            "json"
+        ],
+        "homepage": null
+    }
+});
+
+// Valid, can identify serde_json::Value as JSON type
+sqlx::query("insert into test values($1, $2)")
+    .bind(i)
+    .bind(json)
+    .execute(&pool)
+    .await
+    .unwrap();
+
+// Invalid, cannot identify String as JSON type
+sqlx::query("insert into test values($1, $2)")
+    .bind(i)
+    .bind(json.to_string())
+    .execute(&pool)
+    .await
+    .unwrap();
+```
+
+## Query
+
+Correspondingly, users prefer to display JSON data as strings. Thus we need to make conversions between JSON data and strings before presenting JSON data. There are also 2 ways to do this: auto-conversions on MySQL and PostgreSQL servers, and function `json_to_string`.
+
+For example, in MySQL client:
+```SQL
+SELECT b FROM test;
+
+SELECT json_to_string(b) FROM test;
+```
+Will both return the JSON as human-readable strings.
+
+Specifically, to perform auto-conversions, we attach a message to JSON data in the `metadata` of `Field` in Arrow/Datafusion schema when scanning a JSON column. Frontend servers could identify JSON data and convert it to strings.
+
+The dataflow of the query process is as follows:
+```
+Query directly through client:
+                                  Decode                            Scan
+        String(Serialized JSON)┌──────────┐Arrow Binary(JSONB)┌──────────────┐Arrow Binary(JSONB)
+ Client <----------------------│  Server  │<------------------│ Query Engine │<----------------- Storage
+                               └──────────┘                   └──────────────┘
+(Server identifies JSON type and performs auto-conversion based on column metadata)
+
+Query through json_to_string function:
+                                                                   Scan & Decode
+        String(Serialized JSON)┌──────────┐String(Serialized JSON)┌──────────────┐Arrow Binary(JSONB)
+ Client <----------------------│  Server  │<----------------------│ Query Engine │<----------------- Storage
+                               └──────────┘                       └──────────────┘
+                                                 (Conversion is performed by UDF inside Query Engine)
+
+```
+
+However, if a function uses JSON type as its return type, the metadata method mentioned above is not applicable. Thus the functions of JSON type should specify the return type explicitly instead of returning a JSON type, such as `json_get_int` and `json_get_float` which return corresponding data of `INT` and `FLOAT` type respectively.
+
+## Functions
+Similar to the common JSON type, JSON data can be queried with functions.
+
+For example:
+```SQL
+CREATE TABLE IF NOT EXISTS test (
+    ts TIMESTAMP TIME INDEX,
+    a INT,
+    b JSON
+);
+
+INSERT INTO test VALUES(
+    0,
+    0,
+    '{
+        "name": "jHl2oDDnPc1i2OzlP5Y",
+        "timestamp": "2024-07-25T04:33:11.369386Z",
+        "attributes": { "event_attributes": 48.28667 }
+    }'
+);
+
+SELECT json_get_string(b, 'name') FROM test;
+---------------------+
+| b.name              |
+---------------------+
+| jHl2oDDnPc1i2OzlP5Y |
+---------------------+
+
+SELECT json_get_float(b, 'attributes.event_attributes') FROM test;
+--------------------------------+
+| b.attributes.event_attributes  |
+--------------------------------+
+| 48.28667                       |
+--------------------------------+
+
+```
+And more functions can be added in the future.
+
+# Drawbacks
+
+As a general purpose JSON data type, JSONB may not be as efficient as specialized data types for specific scenarios.
+
+The auto-conversion mechanism is not supported in all scenarios. We need to find workarounds for these scenarios.
+
+# Alternatives
+
+Extract and flatten JSON schema to store in a structured format through pipeline. For nested data, we can provide nested types like `STRUCT` or `ARRAY`.
--- a/grafana/greptimedb.json
+++ b/grafana/greptimedb.json
@@ -409,7 +409,39 @@
      "fieldConfig": {
        "defaults": {
          "color": {
-            "mode": "thresholds"
+            "mode": "palette-classic"
+          },
+          "custom": {
+            "axisBorderShow": false,
+            "axisCenteredZero": false,
+            "axisColorMode": "text",
+            "axisLabel": "",
+            "axisPlacement": "auto",
+            "barAlignment": 0,
+            "drawStyle": "line",
+            "fillOpacity": 0,
+            "gradientMode": "none",
+            "hideFrom": {
+              "legend": false,
+              "tooltip": false,
+              "viz": false
+            },
+            "insertNulls": false,
+            "lineInterpolation": "linear",
+            "lineWidth": 1,
+            "pointSize": 5,
+            "scaleDistribution": {
+              "type": "linear"
+            },
+            "showPoints": "auto",
+            "spanNulls": false,
+            "stacking": {
+              "group": "A",
+              "mode": "none"
+            },
+            "thresholdsStyle": {
+              "mode": "off"
+            }
          },
          "fieldMinMax": false,
          "mappings": [],
@@ -438,18 +470,16 @@
      },
      "id": 27,
      "options": {
-        "colorMode": "value",
-        "graphMode": "area",
-        "justifyMode": "auto",
-        "orientation": "auto",
-        "reduceOptions": {
-          "calcs": ["lastNotNull"],
-          "fields": "",
-          "values": false
+        "legend": {
+          "calcs": [],
+          "displayMode": "list",
+          "placement": "bottom",
+          "showLegend": true
        },
-        "text": {},
-        "textMode": "auto",
-        "wideLayout": true
+        "tooltip": {
+          "mode": "single",
+          "sort": "none"
+        }
      },
      "pluginVersion": "10.2.3",
      "targets": [
@@ -467,7 +497,7 @@
        }
      ],
      "title": "CPU",
-      "type": "stat"
+      "type": "timeseries"
    },
    {
      "datasource": {
@@ -477,7 +507,39 @@
      "fieldConfig": {
        "defaults": {
          "color": {
-            "mode": "thresholds"
+            "mode": "palette-classic"
+          },
+          "custom": {
+            "axisBorderShow": false,
+            "axisCenteredZero": false,
+            "axisColorMode": "text",
+            "axisLabel": "",
+            "axisPlacement": "auto",
+            "barAlignment": 0,
+            "drawStyle": "line",
+            "fillOpacity": 0,
+            "gradientMode": "none",
+            "hideFrom": {
+              "legend": false,
+              "tooltip": false,
+              "viz": false
+            },
+            "insertNulls": false,
+            "lineInterpolation": "linear",
+            "lineWidth": 1,
+            "pointSize": 5,
+            "scaleDistribution": {
+              "type": "linear"
+            },
+            "showPoints": "auto",
+            "spanNulls": false,
+            "stacking": {
+              "group": "A",
+              "mode": "none"
+            },
+            "thresholdsStyle": {
+              "mode": "off"
+            }
          },
          "decimals": 0,
          "fieldMinMax": false,
@@ -503,18 +565,16 @@
      },
      "id": 28,
      "options": {
-        "colorMode": "value",
-        "graphMode": "area",
-        "justifyMode": "auto",
-        "orientation": "auto",
-        "reduceOptions": {
-          "calcs": ["lastNotNull"],
-          "fields": "",
-          "values": false
+        "legend": {
+          "calcs": [],
+          "displayMode": "list",
+          "placement": "bottom",
+          "showLegend": true
        },
-        "text": {},
-        "textMode": "auto",
-        "wideLayout": true
+        "tooltip": {
+          "mode": "single",
+          "sort": "none"
+        }
      },
      "pluginVersion": "10.2.3",
      "targets": [
@@ -532,7 +592,7 @@
        }
      ],
      "title": "Memory",
-      "type": "stat"
+      "type": "timeseries"
    },
    {
      "collapsed": false,
@@ -3335,6 +3395,6 @@
  "timezone": "",
  "title": "GreptimeDB",
  "uid": "e7097237-669b-4f8d-b751-13067afbfb68",
-  "version": 15,
+  "version": 16,
  "weekStart": ""
 }
--- a/rust-toolchain.toml
+++ b/rust-toolchain.toml
@@ -1,3 +1,2 @@
 [toolchain]
-channel = "nightly-2024-06-06"
-
+channel = "nightly-2024-10-19"
--- a/src/api/src/helper.rs
+++ b/src/api/src/helper.rs
@@ -17,10 +17,11 @@ use std::sync::Arc;
 use common_base::BitVec;
 use common_decimal::decimal128::{DECIMAL128_DEFAULT_SCALE, DECIMAL128_MAX_PRECISION};
 use common_decimal::Decimal128;
-use common_time::interval::IntervalUnit;
 use common_time::time::Time;
 use common_time::timestamp::TimeUnit;
-use common_time::{Date, DateTime, Interval, Timestamp};
+use common_time::{
+    Date, DateTime, IntervalDayTime, IntervalMonthDayNano, IntervalYearMonth, Timestamp,
+};
 use datatypes::prelude::{ConcreteDataType, ValueRef};
 use datatypes::scalars::ScalarVector;
 use datatypes::types::{
@@ -42,7 +43,8 @@ use greptime_proto::v1::greptime_request::Request;
 use greptime_proto::v1::query_request::Query;
 use greptime_proto::v1::value::ValueData;
 use greptime_proto::v1::{
-    ColumnDataTypeExtension, DdlRequest, DecimalTypeExtension, QueryRequest, Row, SemanticType,
+    ColumnDataTypeExtension, DdlRequest, DecimalTypeExtension, JsonTypeExtension, QueryRequest,
+    Row, SemanticType,
 };
 use paste::paste;
 use snafu::prelude::*;
@@ -103,7 +105,18 @@ impl From<ColumnDataTypeWrapper> for ConcreteDataType {
            ColumnDataType::Uint64 => ConcreteDataType::uint64_datatype(),
            ColumnDataType::Float32 => ConcreteDataType::float32_datatype(),
            ColumnDataType::Float64 => ConcreteDataType::float64_datatype(),
-            ColumnDataType::Binary => ConcreteDataType::binary_datatype(),
+            ColumnDataType::Binary => {
+                if let Some(TypeExt::JsonType(_)) = datatype_wrapper
+                    .datatype_ext
+                    .as_ref()
+                    .and_then(|datatype_ext| datatype_ext.type_ext.as_ref())
+                {
+                    ConcreteDataType::json_datatype()
+                } else {
+                    ConcreteDataType::binary_datatype()
+                }
+            }
+            ColumnDataType::Json => ConcreteDataType::json_datatype(),
            ColumnDataType::String => ConcreteDataType::string_datatype(),
            ColumnDataType::Date => ConcreteDataType::date_datatype(),
            ColumnDataType::Datetime => ConcreteDataType::datetime_datatype(),
@@ -236,7 +249,7 @@ impl TryFrom<ConcreteDataType> for ColumnDataTypeWrapper {
            ConcreteDataType::UInt64(_) => ColumnDataType::Uint64,
            ConcreteDataType::Float32(_) => ColumnDataType::Float32,
            ConcreteDataType::Float64(_) => ColumnDataType::Float64,
-            ConcreteDataType::Binary(_) => ColumnDataType::Binary,
+            ConcreteDataType::Binary(_) | ConcreteDataType::Json(_) => ColumnDataType::Binary,
            ConcreteDataType::String(_) => ColumnDataType::String,
            ConcreteDataType::Date(_) => ColumnDataType::Date,
            ConcreteDataType::DateTime(_) => ColumnDataType::Datetime,
@@ -276,6 +289,16 @@ impl TryFrom<ConcreteDataType> for ColumnDataTypeWrapper {
                        })),
                    })
            }
+            ColumnDataType::Binary => {
+                if datatype == ConcreteDataType::json_datatype() {
+                    // Json is the same as  binary in proto. The extension marks the binary in proto is actually a json.
+                    Some(ColumnDataTypeExtension {
+                        type_ext: Some(TypeExt::JsonType(JsonTypeExtension::JsonBinary.into())),
+                    })
+                } else {
+                    None
+                }
+            }
            _ => None,
        };
        Ok(Self {
@@ -395,6 +418,10 @@ pub fn values_with_capacity(datatype: ColumnDataType, capacity: usize) -> Values
            decimal128_values: Vec::with_capacity(capacity),
            ..Default::default()
        },
+        ColumnDataType::Json => Values {
+            string_values: Vec::with_capacity(capacity),
+            ..Default::default()
+        },
    }
 }

@@ -435,13 +462,11 @@ pub fn push_vals(column: &mut Column, origin_count: usize, vector: VectorRef) {
            TimeUnit::Microsecond => values.time_microsecond_values.push(val.value()),
            TimeUnit::Nanosecond => values.time_nanosecond_values.push(val.value()),
        },
-        Value::Interval(val) => match val.unit() {
-            IntervalUnit::YearMonth => values.interval_year_month_values.push(val.to_i32()),
-            IntervalUnit::DayTime => values.interval_day_time_values.push(val.to_i64()),
-            IntervalUnit::MonthDayNano => values
-                .interval_month_day_nano_values
-                .push(convert_i128_to_interval(val.to_i128())),
-        },
+        Value::IntervalYearMonth(val) => values.interval_year_month_values.push(val.to_i32()),
+        Value::IntervalDayTime(val) => values.interval_day_time_values.push(val.to_i64()),
+        Value::IntervalMonthDayNano(val) => values
+            .interval_month_day_nano_values
+            .push(convert_month_day_nano_to_pb(val)),
        Value::Decimal128(val) => values.decimal128_values.push(convert_to_pb_decimal128(val)),
        Value::List(_) | Value::Duration(_) => unreachable!(),
    });
@@ -486,14 +511,12 @@ fn ddl_request_type(request: &DdlRequest) -> &'static str {
    }
 }

-/// Converts an i128 value to google protobuf type [IntervalMonthDayNano].
-pub fn convert_i128_to_interval(v: i128) -> v1::IntervalMonthDayNano {
-    let interval = Interval::from_i128(v);
-    let (months, days, nanoseconds) = interval.to_month_day_nano();
+/// Converts an interval to google protobuf type [IntervalMonthDayNano].
+pub fn convert_month_day_nano_to_pb(v: IntervalMonthDayNano) -> v1::IntervalMonthDayNano {
    v1::IntervalMonthDayNano {
-        months,
-        days,
-        nanoseconds,
+        months: v.months,
+        days: v.days,
+        nanoseconds: v.nanoseconds,
    }
 }

@@ -541,11 +564,15 @@ pub fn pb_value_to_value_ref<'a>(
        ValueData::TimeMillisecondValue(t) => ValueRef::Time(Time::new_millisecond(*t)),
        ValueData::TimeMicrosecondValue(t) => ValueRef::Time(Time::new_microsecond(*t)),
        ValueData::TimeNanosecondValue(t) => ValueRef::Time(Time::new_nanosecond(*t)),
-        ValueData::IntervalYearMonthValue(v) => ValueRef::Interval(Interval::from_i32(*v)),
-        ValueData::IntervalDayTimeValue(v) => ValueRef::Interval(Interval::from_i64(*v)),
+        ValueData::IntervalYearMonthValue(v) => {
+            ValueRef::IntervalYearMonth(IntervalYearMonth::from_i32(*v))
+        }
+        ValueData::IntervalDayTimeValue(v) => {
+            ValueRef::IntervalDayTime(IntervalDayTime::from_i64(*v))
+        }
        ValueData::IntervalMonthDayNanoValue(v) => {
-            let interval = Interval::from_month_day_nano(v.months, v.days, v.nanoseconds);
-            ValueRef::Interval(interval)
+            let interval = IntervalMonthDayNano::new(v.months, v.days, v.nanoseconds);
+            ValueRef::IntervalMonthDayNano(interval)
        }
        ValueData::Decimal128Value(v) => {
            // get precision and scale from datatype_extension
@@ -636,7 +663,7 @@ pub fn pb_values_to_vector_ref(data_type: &ConcreteDataType, values: Values) ->
            IntervalType::MonthDayNano(_) => {
                Arc::new(IntervalMonthDayNanoVector::from_iter_values(
                    values.interval_month_day_nano_values.iter().map(|x| {
-                        Interval::from_month_day_nano(x.months, x.days, x.nanoseconds).to_i128()
+                        IntervalMonthDayNano::new(x.months, x.days, x.nanoseconds).to_i128()
                    }),
                ))
            }
@@ -649,7 +676,8 @@ pub fn pb_values_to_vector_ref(data_type: &ConcreteDataType, values: Values) ->
        ConcreteDataType::Null(_)
        | ConcreteDataType::List(_)
        | ConcreteDataType::Dictionary(_)
-        | ConcreteDataType::Duration(_) => {
+        | ConcreteDataType::Duration(_)
+        | ConcreteDataType::Json(_) => {
            unreachable!()
        }
    }
@@ -780,18 +808,18 @@ pub fn pb_values_to_values(data_type: &ConcreteDataType, values: Values) -> Vec<
        ConcreteDataType::Interval(IntervalType::YearMonth(_)) => values
            .interval_year_month_values
            .into_iter()
-            .map(|v| Value::Interval(Interval::from_i32(v)))
+            .map(|v| Value::IntervalYearMonth(IntervalYearMonth::from_i32(v)))
            .collect(),
        ConcreteDataType::Interval(IntervalType::DayTime(_)) => values
            .interval_day_time_values
            .into_iter()
-            .map(|v| Value::Interval(Interval::from_i64(v)))
+            .map(|v| Value::IntervalDayTime(IntervalDayTime::from_i64(v)))
            .collect(),
        ConcreteDataType::Interval(IntervalType::MonthDayNano(_)) => values
            .interval_month_day_nano_values
            .into_iter()
            .map(|v| {
-                Value::Interval(Interval::from_month_day_nano(
+                Value::IntervalMonthDayNano(IntervalMonthDayNano::new(
                    v.months,
                    v.days,
                    v.nanoseconds,
@@ -813,7 +841,8 @@ pub fn pb_values_to_values(data_type: &ConcreteDataType, values: Values) -> Vec<
        ConcreteDataType::Null(_)
        | ConcreteDataType::List(_)
        | ConcreteDataType::Dictionary(_)
-        | ConcreteDataType::Duration(_) => {
+        | ConcreteDataType::Duration(_)
+        | ConcreteDataType::Json(_) => {
            unreachable!()
        }
    }
@@ -831,7 +860,13 @@ pub fn is_column_type_value_eq(
    expect_type: &ConcreteDataType,
 ) -> bool {
    ColumnDataTypeWrapper::try_new(type_value, type_extension)
-        .map(|wrapper| ConcreteDataType::from(wrapper) == *expect_type)
+        .map(|wrapper| {
+            let datatype = ConcreteDataType::from(wrapper);
+            (datatype == *expect_type)
+            // Json type leverage binary type in pb, so this is valid.
+                || (datatype == ConcreteDataType::binary_datatype()
+                    && *expect_type == ConcreteDataType::json_datatype())
+        })
        .unwrap_or(false)
 }

@@ -912,18 +947,16 @@ pub fn to_proto_value(value: Value) -> Option<v1::Value> {
                value_data: Some(ValueData::TimeNanosecondValue(v.value())),
            },
        },
-        Value::Interval(v) => match v.unit() {
-            IntervalUnit::YearMonth => v1::Value {
-                value_data: Some(ValueData::IntervalYearMonthValue(v.to_i32())),
-            },
-            IntervalUnit::DayTime => v1::Value {
-                value_data: Some(ValueData::IntervalDayTimeValue(v.to_i64())),
-            },
-            IntervalUnit::MonthDayNano => v1::Value {
-                value_data: Some(ValueData::IntervalMonthDayNanoValue(
-                    convert_i128_to_interval(v.to_i128()),
-                )),
-            },
+        Value::IntervalYearMonth(v) => v1::Value {
+            value_data: Some(ValueData::IntervalYearMonthValue(v.to_i32())),
+        },
+        Value::IntervalDayTime(v) => v1::Value {
+            value_data: Some(ValueData::IntervalDayTimeValue(v.to_i64())),
+        },
+        Value::IntervalMonthDayNano(v) => v1::Value {
+            value_data: Some(ValueData::IntervalMonthDayNanoValue(
+                convert_month_day_nano_to_pb(v),
+            )),
        },
        Value::Decimal128(v) => v1::Value {
            value_data: Some(ValueData::Decimal128Value(convert_to_pb_decimal128(v))),
@@ -1015,13 +1048,11 @@ pub fn value_to_grpc_value(value: Value) -> GrpcValue {
                TimeUnit::Microsecond => ValueData::TimeMicrosecondValue(v.value()),
                TimeUnit::Nanosecond => ValueData::TimeNanosecondValue(v.value()),
            }),
-            Value::Interval(v) => Some(match v.unit() {
-                IntervalUnit::YearMonth => ValueData::IntervalYearMonthValue(v.to_i32()),
-                IntervalUnit::DayTime => ValueData::IntervalDayTimeValue(v.to_i64()),
-                IntervalUnit::MonthDayNano => {
-                    ValueData::IntervalMonthDayNanoValue(convert_i128_to_interval(v.to_i128()))
-                }
-            }),
+            Value::IntervalYearMonth(v) => Some(ValueData::IntervalYearMonthValue(v.to_i32())),
+            Value::IntervalDayTime(v) => Some(ValueData::IntervalDayTimeValue(v.to_i64())),
+            Value::IntervalMonthDayNano(v) => Some(ValueData::IntervalMonthDayNanoValue(
+                convert_month_day_nano_to_pb(v),
+            )),
            Value::Decimal128(v) => Some(ValueData::Decimal128Value(convert_to_pb_decimal128(v))),
            Value::List(_) | Value::Duration(_) => unreachable!(),
        },
@@ -1032,6 +1063,7 @@ pub fn value_to_grpc_value(value: Value) -> GrpcValue {
 mod tests {
    use std::sync::Arc;

+    use common_time::interval::IntervalUnit;
    use datatypes::types::{
        Int32Type, IntervalDayTimeType, IntervalMonthDayNanoType, IntervalYearMonthType,
        TimeMillisecondType, TimeSecondType, TimestampMillisecondType, TimestampSecondType,
@@ -1477,11 +1509,11 @@ mod tests {

    #[test]
    fn test_convert_i128_to_interval() {
-        let i128_val = 3000;
-        let interval = convert_i128_to_interval(i128_val);
+        let i128_val = 3;
+        let interval = convert_month_day_nano_to_pb(IntervalMonthDayNano::from_i128(i128_val));
        assert_eq!(interval.months, 0);
        assert_eq!(interval.days, 0);
-        assert_eq!(interval.nanoseconds, 3000);
+        assert_eq!(interval.nanoseconds, 3);
    }

    #[test]
@@ -1561,9 +1593,9 @@ mod tests {
            },
        );
        let expect = vec![
-            Value::Interval(Interval::from_year_month(1_i32)),
-            Value::Interval(Interval::from_year_month(2_i32)),
-            Value::Interval(Interval::from_year_month(3_i32)),
+            Value::IntervalYearMonth(IntervalYearMonth::new(1_i32)),
+            Value::IntervalYearMonth(IntervalYearMonth::new(2_i32)),
+            Value::IntervalYearMonth(IntervalYearMonth::new(3_i32)),
        ];
        assert_eq!(expect, actual);

@@ -1576,9 +1608,9 @@ mod tests {
            },
        );
        let expect = vec![
-            Value::Interval(Interval::from_i64(1_i64)),
-            Value::Interval(Interval::from_i64(2_i64)),
-            Value::Interval(Interval::from_i64(3_i64)),
+            Value::IntervalDayTime(IntervalDayTime::from_i64(1_i64)),
+            Value::IntervalDayTime(IntervalDayTime::from_i64(2_i64)),
+            Value::IntervalDayTime(IntervalDayTime::from_i64(3_i64)),
        ];
        assert_eq!(expect, actual);

@@ -1607,9 +1639,9 @@ mod tests {
            },
        );
        let expect = vec![
-            Value::Interval(Interval::from_month_day_nano(1, 2, 3)),
-            Value::Interval(Interval::from_month_day_nano(5, 6, 7)),
-            Value::Interval(Interval::from_month_day_nano(9, 10, 11)),
+            Value::IntervalMonthDayNano(IntervalMonthDayNano::new(1, 2, 3)),
+            Value::IntervalMonthDayNano(IntervalMonthDayNano::new(5, 6, 7)),
+            Value::IntervalMonthDayNano(IntervalMonthDayNano::new(9, 10, 11)),
        ];
        assert_eq!(expect, actual);
    }
--- a/src/api/src/region.rs
+++ b/src/api/src/region.rs
@@ -21,14 +21,14 @@ use greptime_proto::v1::region::RegionResponse as RegionResponseV1;
 #[derive(Debug)]
 pub struct RegionResponse {
    pub affected_rows: AffectedRows,
-    pub extension: HashMap<String, Vec<u8>>,
+    pub extensions: HashMap<String, Vec<u8>>,
 }

 impl RegionResponse {
    pub fn from_region_response(region_response: RegionResponseV1) -> Self {
        Self {
            affected_rows: region_response.affected_rows as _,
-            extension: region_response.extension,
+            extensions: region_response.extensions,
        }
    }

@@ -36,7 +36,7 @@ impl RegionResponse {
    pub fn new(affected_rows: AffectedRows) -> Self {
        Self {
            affected_rows,
-            extension: Default::default(),
+            extensions: Default::default(),
        }
    }
 }
--- a/src/auth/src/common.rs
+++ b/src/auth/src/common.rs
@@ -75,6 +75,16 @@ pub enum Password<'a> {
    PgMD5(HashedPassword<'a>, Salt<'a>),
 }

+impl Password<'_> {
+    pub fn r#type(&self) -> &str {
+        match self {
+            Password::PlainText(_) => "plain_text",
+            Password::MysqlNativePassword(_, _) => "mysql_native_password",
+            Password::PgMD5(_, _) => "pg_md5",
+        }
+    }
+}
+
 pub fn auth_mysql(
    auth_data: HashedPassword,
    salt: Salt,
--- a/src/auth/src/error.rs
+++ b/src/auth/src/error.rs
@@ -89,7 +89,7 @@ impl ErrorExt for Error {
            Error::FileWatch { .. } => StatusCode::InvalidArguments,
            Error::InternalState { .. } => StatusCode::Unexpected,
            Error::Io { .. } => StatusCode::StorageUnavailable,
-            Error::AuthBackend { .. } => StatusCode::Internal,
+            Error::AuthBackend { source, .. } => source.status_code(),

            Error::UserNotFound { .. } => StatusCode::UserNotFound,
            Error::UnsupportedPasswordType { .. } => StatusCode::UnsupportedPasswordType,
--- a/src/auth/src/user_provider.rs
+++ b/src/auth/src/user_provider.rs
@@ -57,6 +57,11 @@ pub trait UserProvider: Send + Sync {
        self.authorize(catalog, schema, &user_info).await?;
        Ok(user_info)
    }
+
+    /// Returns whether this user provider implementation is backed by an external system.
+    fn external(&self) -> bool {
+        false
+    }
 }

 fn load_credential_from_file(filepath: &str) -> Result<Option<HashMap<String, Vec<u8>>>> {
--- a/src/auth/src/user_provider/static_user_provider.rs
+++ b/src/auth/src/user_provider/static_user_provider.rs
@@ -33,7 +33,7 @@ impl StaticUserProvider {
            value: value.to_string(),
            msg: "StaticUserProviderOption must be in format `<option>:<value>`",
        })?;
-        return match mode {
+        match mode {
            "file" => {
                let users = load_credential_from_file(content)?
                    .context(InvalidConfigSnafu {
@@ -58,7 +58,7 @@ impl StaticUserProvider {
                msg: "StaticUserProviderOption must be in format `file:<path>` or `cmd:<values>`",
            }
                .fail(),
-        };
+        }
    }
 }

--- a/src/catalog/Cargo.toml
+++ b/src/catalog/Cargo.toml
@@ -22,8 +22,10 @@ common-config.workspace = true
 common-error.workspace = true
 common-macro.workspace = true
 common-meta.workspace = true
+common-procedure.workspace = true
 common-query.workspace = true
 common-recordbatch.workspace = true
+common-runtime.workspace = true
 common-telemetry.workspace = true
 common-time.workspace = true
 common-version.workspace = true
@@ -48,6 +50,7 @@ sql.workspace = true
 store-api.workspace = true
 table.workspace = true
 tokio.workspace = true
+tokio-stream = "0.1"

 [dev-dependencies]
 cache.workspace = true
--- a/src/catalog/src/error.rs
+++ b/src/catalog/src/error.rs
@@ -50,13 +50,20 @@ pub enum Error {
        source: BoxedError,
    },

-    #[snafu(display("Failed to list nodes in cluster: {source}"))]
+    #[snafu(display("Failed to list nodes in cluster"))]
    ListNodes {
        #[snafu(implicit)]
        location: Location,
        source: BoxedError,
    },

+    #[snafu(display("Failed to region stats in cluster"))]
+    ListRegionStats {
+        #[snafu(implicit)]
+        location: Location,
+        source: BoxedError,
+    },
+
    #[snafu(display("Failed to list flows in catalog {catalog}"))]
    ListFlows {
        #[snafu(implicit)]
@@ -82,6 +89,32 @@ pub enum Error {
        location: Location,
    },

+    #[snafu(display("Failed to get information extension client"))]
+    GetInformationExtension {
+        #[snafu(implicit)]
+        location: Location,
+    },
+
+    #[snafu(display("Failed to list procedures"))]
+    ListProcedures {
+        #[snafu(implicit)]
+        location: Location,
+        source: BoxedError,
+    },
+
+    #[snafu(display("Procedure id not found"))]
+    ProcedureIdNotFound {
+        #[snafu(implicit)]
+        location: Location,
+    },
+
+    #[snafu(display("convert proto data error"))]
+    ConvertProtoData {
+        #[snafu(implicit)]
+        location: Location,
+        source: BoxedError,
+    },
+
    #[snafu(display("Failed to re-compile script due to internal error"))]
    CompileScriptInternal {
        #[snafu(implicit)]
@@ -266,7 +299,9 @@ impl ErrorExt for Error {
            | Error::FindRegionRoutes { .. }
            | Error::CacheNotFound { .. }
            | Error::CastManager { .. }
-            | Error::Json { .. } => StatusCode::Unexpected,
+            | Error::Json { .. }
+            | Error::GetInformationExtension { .. }
+            | Error::ProcedureIdNotFound { .. } => StatusCode::Unexpected,

            Error::ViewPlanColumnsChanged { .. } => StatusCode::InvalidArguments,

@@ -283,7 +318,10 @@ impl ErrorExt for Error {
            | Error::ListNodes { source, .. }
            | Error::ListSchemas { source, .. }
            | Error::ListTables { source, .. }
-            | Error::ListFlows { source, .. } => source.status_code(),
+            | Error::ListFlows { source, .. }
+            | Error::ListProcedures { source, .. }
+            | Error::ListRegionStats { source, .. }
+            | Error::ConvertProtoData { source, .. } => source.status_code(),

            Error::CreateTable { source, .. } => source.status_code(),

--- a/src/catalog/src/kvbackend/manager.rs
+++ b/src/catalog/src/kvbackend/manager.rs
@@ -21,7 +21,6 @@ use common_catalog::consts::{
    DEFAULT_CATALOG_NAME, DEFAULT_SCHEMA_NAME, INFORMATION_SCHEMA_NAME, NUMBERS_TABLE_ID,
    PG_CATALOG_NAME,
 };
-use common_config::Mode;
 use common_error::ext::BoxedError;
 use common_meta::cache::{LayeredCacheRegistryRef, ViewInfoCacheRef};
 use common_meta::key::catalog_name::CatalogNameKey;
@@ -31,22 +30,25 @@ use common_meta::key::table_info::TableInfoValue;
 use common_meta::key::table_name::TableNameKey;
 use common_meta::key::{TableMetadataManager, TableMetadataManagerRef};
 use common_meta::kv_backend::KvBackendRef;
+use common_procedure::ProcedureManagerRef;
 use futures_util::stream::BoxStream;
 use futures_util::{StreamExt, TryStreamExt};
-use meta_client::client::MetaClient;
 use moka::sync::Cache;
 use partition::manager::{PartitionRuleManager, PartitionRuleManagerRef};
+use session::context::{Channel, QueryContext};
 use snafu::prelude::*;
 use table::dist_table::DistTable;
 use table::table::numbers::{NumbersTable, NUMBERS_TABLE_NAME};
 use table::table_name::TableName;
 use table::TableRef;
+use tokio::sync::Semaphore;
+use tokio_stream::wrappers::ReceiverStream;

 use crate::error::{
    CacheNotFoundSnafu, GetTableCacheSnafu, InvalidTableInfoInCatalogSnafu, ListCatalogsSnafu,
    ListSchemasSnafu, ListTablesSnafu, Result, TableMetadataManagerSnafu,
 };
-use crate::information_schema::InformationSchemaProvider;
+use crate::information_schema::{InformationExtensionRef, InformationSchemaProvider};
 use crate::kvbackend::TableCacheRef;
 use crate::system_schema::pg_catalog::PGCatalogProvider;
 use crate::system_schema::SystemSchemaProvider;
@@ -59,27 +61,31 @@ use crate::CatalogManager;
 /// comes from `SystemCatalog`, which is static and read-only.
 #[derive(Clone)]
 pub struct KvBackendCatalogManager {
-    mode: Mode,
-    meta_client: Option<Arc<MetaClient>>,
+    /// Provides the extension methods for the `information_schema` tables
+    information_extension: InformationExtensionRef,
+    /// Manages partition rules.
    partition_manager: PartitionRuleManagerRef,
+    /// Manages table metadata.
    table_metadata_manager: TableMetadataManagerRef,
    /// A sub-CatalogManager that handles system tables
    system_catalog: SystemCatalog,
+    /// Cache registry for all caches.
    cache_registry: LayeredCacheRegistryRef,
+    /// Only available in `Standalone` mode.
+    procedure_manager: Option<ProcedureManagerRef>,
 }

 const CATALOG_CACHE_MAX_CAPACITY: u64 = 128;

 impl KvBackendCatalogManager {
    pub fn new(
-        mode: Mode,
-        meta_client: Option<Arc<MetaClient>>,
+        information_extension: InformationExtensionRef,
        backend: KvBackendRef,
        cache_registry: LayeredCacheRegistryRef,
+        procedure_manager: Option<ProcedureManagerRef>,
    ) -> Arc<Self> {
        Arc::new_cyclic(|me| Self {
-            mode,
-            meta_client,
+            information_extension,
            partition_manager: Arc::new(PartitionRuleManager::new(
                backend.clone(),
                cache_registry
@@ -103,23 +109,19 @@ impl KvBackendCatalogManager {
                backend,
            },
            cache_registry,
+            procedure_manager,
        })
    }

-    /// Returns the server running mode.
-    pub fn running_mode(&self) -> &Mode {
-        &self.mode
-    }
-
    pub fn view_info_cache(&self) -> Result<ViewInfoCacheRef> {
        self.cache_registry.get().context(CacheNotFoundSnafu {
            name: "view_info_cache",
        })
    }

-    /// Returns the `[MetaClient]`.
-    pub fn meta_client(&self) -> Option<Arc<MetaClient>> {
-        self.meta_client.clone()
+    /// Returns the [`InformationExtension`].
+    pub fn information_extension(&self) -> InformationExtensionRef {
+        self.information_extension.clone()
    }

    pub fn partition_manager(&self) -> PartitionRuleManagerRef {
@@ -129,6 +131,10 @@ impl KvBackendCatalogManager {
    pub fn table_metadata_manager_ref(&self) -> &TableMetadataManagerRef {
        &self.table_metadata_manager
    }
+
+    pub fn procedure_manager(&self) -> Option<ProcedureManagerRef> {
+        self.procedure_manager.clone()
+    }
 }

 #[async_trait::async_trait]
@@ -152,7 +158,11 @@ impl CatalogManager for KvBackendCatalogManager {
        Ok(keys)
    }

-    async fn schema_names(&self, catalog: &str) -> Result<Vec<String>> {
+    async fn schema_names(
+        &self,
+        catalog: &str,
+        query_ctx: Option<&QueryContext>,
+    ) -> Result<Vec<String>> {
        let stream = self
            .table_metadata_manager
            .schema_manager()
@@ -163,27 +173,29 @@ impl CatalogManager for KvBackendCatalogManager {
            .map_err(BoxedError::new)
            .context(ListSchemasSnafu { catalog })?;

-        keys.extend(self.system_catalog.schema_names());
+        keys.extend(self.system_catalog.schema_names(query_ctx));

        Ok(keys.into_iter().collect())
    }

-    async fn table_names(&self, catalog: &str, schema: &str) -> Result<Vec<String>> {
-        let stream = self
+    async fn table_names(
+        &self,
+        catalog: &str,
+        schema: &str,
+        query_ctx: Option<&QueryContext>,
+    ) -> Result<Vec<String>> {
+        let mut tables = self
            .table_metadata_manager
            .table_name_manager()
-            .tables(catalog, schema);
-        let mut tables = stream
+            .tables(catalog, schema)
+            .map_ok(|(table_name, _)| table_name)
            .try_collect::<Vec<_>>()
            .await
            .map_err(BoxedError::new)
-            .context(ListTablesSnafu { catalog, schema })?
-            .into_iter()
-            .map(|(k, _)| k)
-            .collect::<Vec<_>>();
-        tables.extend_from_slice(&self.system_catalog.table_names(schema));
+            .context(ListTablesSnafu { catalog, schema })?;

-        Ok(tables.into_iter().collect())
+        tables.extend(self.system_catalog.table_names(schema, query_ctx));
+        Ok(tables)
    }

    async fn catalog_exists(&self, catalog: &str) -> Result<bool> {
@@ -194,8 +206,13 @@ impl CatalogManager for KvBackendCatalogManager {
            .context(TableMetadataManagerSnafu)
    }

-    async fn schema_exists(&self, catalog: &str, schema: &str) -> Result<bool> {
-        if self.system_catalog.schema_exists(schema) {
+    async fn schema_exists(
+        &self,
+        catalog: &str,
+        schema: &str,
+        query_ctx: Option<&QueryContext>,
+    ) -> Result<bool> {
+        if self.system_catalog.schema_exists(schema, query_ctx) {
            return Ok(true);
        }

@@ -206,8 +223,14 @@ impl CatalogManager for KvBackendCatalogManager {
            .context(TableMetadataManagerSnafu)
    }

-    async fn table_exists(&self, catalog: &str, schema: &str, table: &str) -> Result<bool> {
-        if self.system_catalog.table_exists(schema, table) {
+    async fn table_exists(
+        &self,
+        catalog: &str,
+        schema: &str,
+        table: &str,
+        query_ctx: Option<&QueryContext>,
+    ) -> Result<bool> {
+        if self.system_catalog.table_exists(schema, table, query_ctx) {
            return Ok(true);
        }

@@ -225,10 +248,12 @@ impl CatalogManager for KvBackendCatalogManager {
        catalog_name: &str,
        schema_name: &str,
        table_name: &str,
+        query_ctx: Option<&QueryContext>,
    ) -> Result<Option<TableRef>> {
-        if let Some(table) = self
-            .system_catalog
-            .table(catalog_name, schema_name, table_name)
+        let channel = query_ctx.map_or(Channel::Unknown, |ctx| ctx.channel());
+        if let Some(table) =
+            self.system_catalog
+                .table(catalog_name, schema_name, table_name, query_ctx)
        {
            return Ok(Some(table));
        }
@@ -236,58 +261,112 @@ impl CatalogManager for KvBackendCatalogManager {
        let table_cache: TableCacheRef = self.cache_registry.get().context(CacheNotFoundSnafu {
            name: "table_cache",
        })?;
-
-        table_cache
+        if let Some(table) = table_cache
            .get_by_ref(&TableName {
                catalog_name: catalog_name.to_string(),
                schema_name: schema_name.to_string(),
                table_name: table_name.to_string(),
            })
            .await
-            .context(GetTableCacheSnafu)
+            .context(GetTableCacheSnafu)?
+        {
+            return Ok(Some(table));
+        }
+
+        if channel == Channel::Postgres {
+            // falldown to pg_catalog
+            if let Some(table) =
+                self.system_catalog
+                    .table(catalog_name, PG_CATALOG_NAME, table_name, query_ctx)
+            {
+                return Ok(Some(table));
+            }
+        }
+
+        return Ok(None);
    }

-    fn tables<'a>(&'a self, catalog: &'a str, schema: &'a str) -> BoxStream<'a, Result<TableRef>> {
+    fn tables<'a>(
+        &'a self,
+        catalog: &'a str,
+        schema: &'a str,
+        query_ctx: Option<&'a QueryContext>,
+    ) -> BoxStream<'a, Result<TableRef>> {
        let sys_tables = try_stream!({
            // System tables
-            let sys_table_names = self.system_catalog.table_names(schema);
+            let sys_table_names = self.system_catalog.table_names(schema, query_ctx);
            for table_name in sys_table_names {
-                if let Some(table) = self.system_catalog.table(catalog, schema, &table_name) {
+                if let Some(table) =
+                    self.system_catalog
+                        .table(catalog, schema, &table_name, query_ctx)
+                {
                    yield table;
                }
            }
        });

-        let table_id_stream = self
-            .table_metadata_manager
-            .table_name_manager()
-            .tables(catalog, schema)
-            .map_ok(|(_, v)| v.table_id());
        const BATCH_SIZE: usize = 128;
-        let user_tables = try_stream!({
+        const CONCURRENCY: usize = 8;
+
+        let (tx, rx) = tokio::sync::mpsc::channel(64);
+        let metadata_manager = self.table_metadata_manager.clone();
+        let catalog = catalog.to_string();
+        let schema = schema.to_string();
+        let semaphore = Arc::new(Semaphore::new(CONCURRENCY));
+
+        common_runtime::spawn_global(async move {
+            let table_id_stream = metadata_manager
+                .table_name_manager()
+                .tables(&catalog, &schema)
+                .map_ok(|(_, v)| v.table_id());
            // Split table ids into chunks
            let mut table_id_chunks = table_id_stream.ready_chunks(BATCH_SIZE);

            while let Some(table_ids) = table_id_chunks.next().await {
-                let table_ids = table_ids
+                let table_ids = match table_ids
                    .into_iter()
                    .collect::<std::result::Result<Vec<_>, _>>()
                    .map_err(BoxedError::new)
-                    .context(ListTablesSnafu { catalog, schema })?;
+                    .context(ListTablesSnafu {
+                        catalog: &catalog,
+                        schema: &schema,
+                    }) {
+                    Ok(table_ids) => table_ids,
+                    Err(e) => {
+                        let _ = tx.send(Err(e)).await;
+                        return;
+                    }
+                };

-                let table_info_values = self
-                    .table_metadata_manager
-                    .table_info_manager()
-                    .batch_get(&table_ids)
-                    .await
-                    .context(TableMetadataManagerSnafu)?;
+                let metadata_manager = metadata_manager.clone();
+                let tx = tx.clone();
+                let semaphore = semaphore.clone();
+                common_runtime::spawn_global(async move {
+                    // we don't explicitly close the semaphore so just ignore the potential error.
+                    let _ = semaphore.acquire().await;
+                    let table_info_values = match metadata_manager
+                        .table_info_manager()
+                        .batch_get(&table_ids)
+                        .await
+                        .context(TableMetadataManagerSnafu)
+                    {
+                        Ok(table_info_values) => table_info_values,
+                        Err(e) => {
+                            let _ = tx.send(Err(e)).await;
+                            return;
+                        }
+                    };

-                for table_info_value in table_info_values.into_values() {
-                    yield build_table(table_info_value)?;
-                }
+                    for table in table_info_values.into_values().map(build_table) {
+                        if tx.send(table).await.is_err() {
+                            return;
+                        }
+                    }
+                });
            }
        });

+        let user_tables = ReceiverStream::new(rx);
        Box::pin(sys_tables.chain(user_tables))
    }
 }
@@ -320,18 +399,27 @@ struct SystemCatalog {
 }

 impl SystemCatalog {
-    // TODO(j0hn50n133): remove the duplicated hard-coded table names logic
-    fn schema_names(&self) -> Vec<String> {
-        vec![
-            INFORMATION_SCHEMA_NAME.to_string(),
-            PG_CATALOG_NAME.to_string(),
-        ]
+    fn schema_names(&self, query_ctx: Option<&QueryContext>) -> Vec<String> {
+        let channel = query_ctx.map_or(Channel::Unknown, |ctx| ctx.channel());
+        match channel {
+            // pg_catalog only visible under postgres protocol
+            Channel::Postgres => vec![
+                INFORMATION_SCHEMA_NAME.to_string(),
+                PG_CATALOG_NAME.to_string(),
+            ],
+            _ => {
+                vec![INFORMATION_SCHEMA_NAME.to_string()]
+            }
+        }
    }

-    fn table_names(&self, schema: &str) -> Vec<String> {
+    fn table_names(&self, schema: &str, query_ctx: Option<&QueryContext>) -> Vec<String> {
+        let channel = query_ctx.map_or(Channel::Unknown, |ctx| ctx.channel());
        match schema {
            INFORMATION_SCHEMA_NAME => self.information_schema_provider.table_names(),
-            PG_CATALOG_NAME => self.pg_catalog_provider.table_names(),
+            PG_CATALOG_NAME if channel == Channel::Postgres => {
+                self.pg_catalog_provider.table_names()
+            }
            DEFAULT_SCHEMA_NAME => {
                vec![NUMBERS_TABLE_NAME.to_string()]
            }
@@ -339,23 +427,35 @@ impl SystemCatalog {
        }
    }

-    fn schema_exists(&self, schema: &str) -> bool {
-        schema == INFORMATION_SCHEMA_NAME || schema == PG_CATALOG_NAME
+    fn schema_exists(&self, schema: &str, query_ctx: Option<&QueryContext>) -> bool {
+        let channel = query_ctx.map_or(Channel::Unknown, |ctx| ctx.channel());
+        match channel {
+            Channel::Postgres => schema == PG_CATALOG_NAME || schema == INFORMATION_SCHEMA_NAME,
+            _ => schema == INFORMATION_SCHEMA_NAME,
+        }
    }

-    fn table_exists(&self, schema: &str, table: &str) -> bool {
+    fn table_exists(&self, schema: &str, table: &str, query_ctx: Option<&QueryContext>) -> bool {
+        let channel = query_ctx.map_or(Channel::Unknown, |ctx| ctx.channel());
        if schema == INFORMATION_SCHEMA_NAME {
            self.information_schema_provider.table(table).is_some()
        } else if schema == DEFAULT_SCHEMA_NAME {
            table == NUMBERS_TABLE_NAME
-        } else if schema == PG_CATALOG_NAME {
+        } else if schema == PG_CATALOG_NAME && channel == Channel::Postgres {
            self.pg_catalog_provider.table(table).is_some()
        } else {
            false
        }
    }

-    fn table(&self, catalog: &str, schema: &str, table_name: &str) -> Option<TableRef> {
+    fn table(
+        &self,
+        catalog: &str,
+        schema: &str,
+        table_name: &str,
+        query_ctx: Option<&QueryContext>,
+    ) -> Option<TableRef> {
+        let channel = query_ctx.map_or(Channel::Unknown, |ctx| ctx.channel());
        if schema == INFORMATION_SCHEMA_NAME {
            let information_schema_provider =
                self.catalog_cache.get_with_by_ref(catalog, move || {
@@ -366,7 +466,7 @@ impl SystemCatalog {
                    ))
                });
            information_schema_provider.table(table_name)
-        } else if schema == PG_CATALOG_NAME {
+        } else if schema == PG_CATALOG_NAME && channel == Channel::Postgres {
            if catalog == DEFAULT_CATALOG_NAME {
                self.pg_catalog_provider.table(table_name)
            } else {
--- a/src/catalog/src/lib.rs
+++ b/src/catalog/src/lib.rs
@@ -20,8 +20,10 @@ use std::fmt::{Debug, Formatter};
 use std::sync::Arc;

 use api::v1::CreateTableExpr;
+use common_catalog::consts::{INFORMATION_SCHEMA_NAME, PG_CATALOG_NAME};
 use futures::future::BoxFuture;
 use futures_util::stream::BoxStream;
+use session::context::QueryContext;
 use table::metadata::TableId;
 use table::TableRef;

@@ -44,15 +46,35 @@ pub trait CatalogManager: Send + Sync {

    async fn catalog_names(&self) -> Result<Vec<String>>;

-    async fn schema_names(&self, catalog: &str) -> Result<Vec<String>>;
+    async fn schema_names(
+        &self,
+        catalog: &str,
+        query_ctx: Option<&QueryContext>,
+    ) -> Result<Vec<String>>;

-    async fn table_names(&self, catalog: &str, schema: &str) -> Result<Vec<String>>;
+    async fn table_names(
+        &self,
+        catalog: &str,
+        schema: &str,
+        query_ctx: Option<&QueryContext>,
+    ) -> Result<Vec<String>>;

    async fn catalog_exists(&self, catalog: &str) -> Result<bool>;

-    async fn schema_exists(&self, catalog: &str, schema: &str) -> Result<bool>;
+    async fn schema_exists(
+        &self,
+        catalog: &str,
+        schema: &str,
+        query_ctx: Option<&QueryContext>,
+    ) -> Result<bool>;

-    async fn table_exists(&self, catalog: &str, schema: &str, table: &str) -> Result<bool>;
+    async fn table_exists(
+        &self,
+        catalog: &str,
+        schema: &str,
+        table: &str,
+        query_ctx: Option<&QueryContext>,
+    ) -> Result<bool>;

    /// Returns the table by catalog, schema and table name.
    async fn table(
@@ -60,10 +82,25 @@ pub trait CatalogManager: Send + Sync {
        catalog: &str,
        schema: &str,
        table_name: &str,
+        query_ctx: Option<&QueryContext>,
    ) -> Result<Option<TableRef>>;

    /// Returns all tables with a stream by catalog and schema.
-    fn tables<'a>(&'a self, catalog: &'a str, schema: &'a str) -> BoxStream<'a, Result<TableRef>>;
+    fn tables<'a>(
+        &'a self,
+        catalog: &'a str,
+        schema: &'a str,
+        query_ctx: Option<&'a QueryContext>,
+    ) -> BoxStream<'a, Result<TableRef>>;
+
+    /// Check if `schema` is a reserved schema name
+    fn is_reserved_schema_name(&self, schema: &str) -> bool {
+        // We have to check whether a schema name is reserved before create schema.
+        // We need this rather than use schema_exists directly because `pg_catalog` is
+        // only visible via postgres protocol. So if we don't check, a mysql client may
+        // create a schema named `pg_catalog` which is somehow malformed.
+        schema == INFORMATION_SCHEMA_NAME || schema == PG_CATALOG_NAME
+    }
 }

 pub type CatalogManagerRef = Arc<dyn CatalogManager>;
--- a/src/catalog/src/memory/manager.rs
+++ b/src/catalog/src/memory/manager.rs
@@ -26,6 +26,7 @@ use common_catalog::consts::{
 use common_meta::key::flow::FlowMetadataManager;
 use common_meta::kv_backend::memory::MemoryKvBackend;
 use futures_util::stream::BoxStream;
+use session::context::QueryContext;
 use snafu::OptionExt;
 use table::TableRef;

@@ -53,7 +54,11 @@ impl CatalogManager for MemoryCatalogManager {
        Ok(self.catalogs.read().unwrap().keys().cloned().collect())
    }

-    async fn schema_names(&self, catalog: &str) -> Result<Vec<String>> {
+    async fn schema_names(
+        &self,
+        catalog: &str,
+        _query_ctx: Option<&QueryContext>,
+    ) -> Result<Vec<String>> {
        Ok(self
            .catalogs
            .read()
@@ -67,7 +72,12 @@ impl CatalogManager for MemoryCatalogManager {
            .collect())
    }

-    async fn table_names(&self, catalog: &str, schema: &str) -> Result<Vec<String>> {
+    async fn table_names(
+        &self,
+        catalog: &str,
+        schema: &str,
+        _query_ctx: Option<&QueryContext>,
+    ) -> Result<Vec<String>> {
        Ok(self
            .catalogs
            .read()
@@ -87,11 +97,22 @@ impl CatalogManager for MemoryCatalogManager {
        self.catalog_exist_sync(catalog)
    }

-    async fn schema_exists(&self, catalog: &str, schema: &str) -> Result<bool> {
+    async fn schema_exists(
+        &self,
+        catalog: &str,
+        schema: &str,
+        _query_ctx: Option<&QueryContext>,
+    ) -> Result<bool> {
        self.schema_exist_sync(catalog, schema)
    }

-    async fn table_exists(&self, catalog: &str, schema: &str, table: &str) -> Result<bool> {
+    async fn table_exists(
+        &self,
+        catalog: &str,
+        schema: &str,
+        table: &str,
+        _query_ctx: Option<&QueryContext>,
+    ) -> Result<bool> {
        let catalogs = self.catalogs.read().unwrap();
        Ok(catalogs
            .get(catalog)
@@ -108,6 +129,7 @@ impl CatalogManager for MemoryCatalogManager {
        catalog: &str,
        schema: &str,
        table_name: &str,
+        _query_ctx: Option<&QueryContext>,
    ) -> Result<Option<TableRef>> {
        let result = try {
            self.catalogs
@@ -121,7 +143,12 @@ impl CatalogManager for MemoryCatalogManager {
        Ok(result)
    }

-    fn tables<'a>(&'a self, catalog: &'a str, schema: &'a str) -> BoxStream<'a, Result<TableRef>> {
+    fn tables<'a>(
+        &'a self,
+        catalog: &'a str,
+        schema: &'a str,
+        _query_ctx: Option<&QueryContext>,
+    ) -> BoxStream<'a, Result<TableRef>> {
        let catalogs = self.catalogs.read().unwrap();

        let Some(schemas) = catalogs.get(catalog) else {
@@ -371,11 +398,12 @@ mod tests {
                DEFAULT_CATALOG_NAME,
                DEFAULT_SCHEMA_NAME,
                NUMBERS_TABLE_NAME,
+                None,
            )
            .await
            .unwrap()
            .unwrap();
-        let stream = catalog_list.tables(DEFAULT_CATALOG_NAME, DEFAULT_SCHEMA_NAME);
+        let stream = catalog_list.tables(DEFAULT_CATALOG_NAME, DEFAULT_SCHEMA_NAME, None);
        let tables = stream.try_collect::<Vec<_>>().await.unwrap();
        assert_eq!(tables.len(), 1);
        assert_eq!(
@@ -384,7 +412,12 @@ mod tests {
        );

        assert!(catalog_list
-            .table(DEFAULT_CATALOG_NAME, DEFAULT_SCHEMA_NAME, "not_exists")
+            .table(
+                DEFAULT_CATALOG_NAME,
+                DEFAULT_SCHEMA_NAME,
+                "not_exists",
+                None
+            )
            .await
            .unwrap()
            .is_none());
@@ -411,7 +444,7 @@ mod tests {
        };
        catalog.register_table_sync(register_table_req).unwrap();
        assert!(catalog
-            .table(DEFAULT_CATALOG_NAME, DEFAULT_SCHEMA_NAME, table_name)
+            .table(DEFAULT_CATALOG_NAME, DEFAULT_SCHEMA_NAME, table_name, None)
            .await
            .unwrap()
            .is_some());
@@ -423,7 +456,7 @@ mod tests {
        };
        catalog.deregister_table_sync(deregister_table_req).unwrap();
        assert!(catalog
-            .table(DEFAULT_CATALOG_NAME, DEFAULT_SCHEMA_NAME, table_name)
+            .table(DEFAULT_CATALOG_NAME, DEFAULT_SCHEMA_NAME, table_name, None)
            .await
            .unwrap()
            .is_none());
--- a/src/catalog/src/system_schema/information_schema.rs
+++ b/src/catalog/src/system_schema/information_schema.rs
@@ -18,7 +18,9 @@ pub mod flows;
 mod information_memory_table;
 pub mod key_column_usage;
 mod partitions;
+mod procedure_info;
 mod region_peers;
+mod region_statistics;
 mod runtime_metrics;
 pub mod schemata;
 mod table_constraints;
@@ -30,7 +32,11 @@ use std::collections::HashMap;
 use std::sync::{Arc, Weak};

 use common_catalog::consts::{self, DEFAULT_CATALOG_NAME, INFORMATION_SCHEMA_NAME};
+use common_error::ext::ErrorExt;
+use common_meta::cluster::NodeInfo;
+use common_meta::datanode::RegionStat;
 use common_meta::key::flow::FlowMetadataManager;
+use common_procedure::ProcedureInfo;
 use common_recordbatch::SendableRecordBatchStream;
 use datatypes::schema::SchemaRef;
 use lazy_static::lazy_static;
@@ -43,7 +49,7 @@ use views::InformationSchemaViews;

 use self::columns::InformationSchemaColumns;
 use super::{SystemSchemaProviderInner, SystemTable, SystemTableRef};
-use crate::error::Result;
+use crate::error::{Error, Result};
 use crate::system_schema::information_schema::cluster_info::InformationSchemaClusterInfo;
 use crate::system_schema::information_schema::flows::InformationSchemaFlows;
 use crate::system_schema::information_schema::information_memory_table::get_schema_columns;
@@ -188,6 +194,16 @@ impl SystemSchemaProviderInner for InformationSchemaProvider {
                self.catalog_name.clone(),
                self.flow_metadata_manager.clone(),
            )) as _),
+            PROCEDURE_INFO => Some(
+                Arc::new(procedure_info::InformationSchemaProcedureInfo::new(
+                    self.catalog_manager.clone(),
+                )) as _,
+            ),
+            REGION_STATISTICS => Some(Arc::new(
+                region_statistics::InformationSchemaRegionStatistics::new(
+                    self.catalog_manager.clone(),
+                ),
+            ) as _),
            _ => None,
        }
    }
@@ -235,6 +251,14 @@ impl InformationSchemaProvider {
                CLUSTER_INFO.to_string(),
                self.build_table(CLUSTER_INFO).unwrap(),
            );
+            tables.insert(
+                PROCEDURE_INFO.to_string(),
+                self.build_table(PROCEDURE_INFO).unwrap(),
+            );
+            tables.insert(
+                REGION_STATISTICS.to_string(),
+                self.build_table(REGION_STATISTICS).unwrap(),
+            );
        }

        tables.insert(TABLES.to_string(), self.build_table(TABLES).unwrap());
@@ -250,7 +274,6 @@ impl InformationSchemaProvider {
            self.build_table(TABLE_CONSTRAINTS).unwrap(),
        );
        tables.insert(FLOWS.to_string(), self.build_table(FLOWS).unwrap());
-
        // Add memory tables
        for name in MEMORY_TABLES.iter() {
            tables.insert((*name).to_string(), self.build_table(name).expect(name));
@@ -299,3 +322,39 @@ where
        InformationTable::to_stream(self, request)
    }
 }
+
+pub type InformationExtensionRef = Arc<dyn InformationExtension<Error = Error> + Send + Sync>;
+
+/// The `InformationExtension` trait provides the extension methods for the `information_schema` tables.
+#[async_trait::async_trait]
+pub trait InformationExtension {
+    type Error: ErrorExt;
+
+    /// Gets the nodes information.
+    async fn nodes(&self) -> std::result::Result<Vec<NodeInfo>, Self::Error>;
+
+    /// Gets the procedures information.
+    async fn procedures(&self) -> std::result::Result<Vec<(String, ProcedureInfo)>, Self::Error>;
+
+    /// Gets the region statistics.
+    async fn region_stats(&self) -> std::result::Result<Vec<RegionStat>, Self::Error>;
+}
+
+pub struct NoopInformationExtension;
+
+#[async_trait::async_trait]
+impl InformationExtension for NoopInformationExtension {
+    type Error = Error;
+
+    async fn nodes(&self) -> std::result::Result<Vec<NodeInfo>, Self::Error> {
+        Ok(vec![])
+    }
+
+    async fn procedures(&self) -> std::result::Result<Vec<(String, ProcedureInfo)>, Self::Error> {
+        Ok(vec![])
+    }
+
+    async fn region_stats(&self) -> std::result::Result<Vec<RegionStat>, Self::Error> {
+        Ok(vec![])
+    }
+}
--- a/src/catalog/src/system_schema/information_schema/cluster_info.rs
+++ b/src/catalog/src/system_schema/information_schema/cluster_info.rs
@@ -17,13 +17,10 @@ use std::time::Duration;

 use arrow_schema::SchemaRef as ArrowSchemaRef;
 use common_catalog::consts::INFORMATION_SCHEMA_CLUSTER_INFO_TABLE_ID;
-use common_config::Mode;
 use common_error::ext::BoxedError;
-use common_meta::cluster::{ClusterInfo, NodeInfo, NodeStatus};
-use common_meta::peer::Peer;
+use common_meta::cluster::NodeInfo;
 use common_recordbatch::adapter::RecordBatchStreamAdapter;
 use common_recordbatch::{RecordBatch, SendableRecordBatchStream};
-use common_telemetry::warn;
 use common_time::timestamp::Timestamp;
 use datafusion::execution::TaskContext;
 use datafusion::physical_plan::stream::RecordBatchStreamAdapter as DfRecordBatchStreamAdapter;
@@ -40,7 +37,7 @@ use snafu::ResultExt;
 use store_api::storage::{ScanRequest, TableId};

 use super::CLUSTER_INFO;
-use crate::error::{CreateRecordBatchSnafu, InternalSnafu, ListNodesSnafu, Result};
+use crate::error::{CreateRecordBatchSnafu, InternalSnafu, Result};
 use crate::system_schema::information_schema::{InformationTable, Predicates};
 use crate::system_schema::utils;
 use crate::CatalogManager;
@@ -70,7 +67,6 @@ const INIT_CAPACITY: usize = 42;
 pub(super) struct InformationSchemaClusterInfo {
    schema: SchemaRef,
    catalog_manager: Weak<dyn CatalogManager>,
-    start_time_ms: u64,
 }

 impl InformationSchemaClusterInfo {
@@ -78,7 +74,6 @@ impl InformationSchemaClusterInfo {
        Self {
            schema: Self::schema(),
            catalog_manager,
-            start_time_ms: common_time::util::current_time_millis() as u64,
        }
    }

@@ -100,11 +95,7 @@ impl InformationSchemaClusterInfo {
    }

    fn builder(&self) -> InformationSchemaClusterInfoBuilder {
-        InformationSchemaClusterInfoBuilder::new(
-            self.schema.clone(),
-            self.catalog_manager.clone(),
-            self.start_time_ms,
-        )
+        InformationSchemaClusterInfoBuilder::new(self.schema.clone(), self.catalog_manager.clone())
    }
 }

@@ -144,7 +135,6 @@ impl InformationTable for InformationSchemaClusterInfo {

 struct InformationSchemaClusterInfoBuilder {
    schema: SchemaRef,
-    start_time_ms: u64,
    catalog_manager: Weak<dyn CatalogManager>,

    peer_ids: Int64VectorBuilder,
@@ -158,11 +148,7 @@ struct InformationSchemaClusterInfoBuilder {
 }

 impl InformationSchemaClusterInfoBuilder {
-    fn new(
-        schema: SchemaRef,
-        catalog_manager: Weak<dyn CatalogManager>,
-        start_time_ms: u64,
-    ) -> Self {
+    fn new(schema: SchemaRef, catalog_manager: Weak<dyn CatalogManager>) -> Self {
        Self {
            schema,
            catalog_manager,
@@ -174,56 +160,17 @@ impl InformationSchemaClusterInfoBuilder {
            start_times: TimestampMillisecondVectorBuilder::with_capacity(INIT_CAPACITY),
            uptimes: StringVectorBuilder::with_capacity(INIT_CAPACITY),
            active_times: StringVectorBuilder::with_capacity(INIT_CAPACITY),
-            start_time_ms,
        }
    }

    /// Construct the `information_schema.cluster_info` virtual table
    async fn make_cluster_info(&mut self, request: Option<ScanRequest>) -> Result<RecordBatch> {
        let predicates = Predicates::from_scan_request(&request);
-        let mode = utils::running_mode(&self.catalog_manager)?.unwrap_or(Mode::Standalone);
-
-        match mode {
-            Mode::Standalone => {
-                let build_info = common_version::build_info();
-
-                self.add_node_info(
-                    &predicates,
-                    NodeInfo {
-                        // For the standalone:
-                        // - id always 0
-                        // - empty string for peer_addr
-                        peer: Peer {
-                            id: 0,
-                            addr: "".to_string(),
-                        },
-                        last_activity_ts: -1,
-                        status: NodeStatus::Standalone,
-                        version: build_info.version.to_string(),
-                        git_commit: build_info.commit_short.to_string(),
-                        // Use `self.start_time_ms` instead.
-                        // It's not precise but enough.
-                        start_time_ms: self.start_time_ms,
-                    },
-                );
-            }
-            Mode::Distributed => {
-                if let Some(meta_client) = utils::meta_client(&self.catalog_manager)? {
-                    let node_infos = meta_client
-                        .list_nodes(None)
-                        .await
-                        .map_err(BoxedError::new)
-                        .context(ListNodesSnafu)?;
-
-                    for node_info in node_infos {
-                        self.add_node_info(&predicates, node_info);
-                    }
-                } else {
-                    warn!("Could not find meta client in distributed mode.");
-                }
-            }
+        let information_extension = utils::information_extension(&self.catalog_manager)?;
+        let node_infos = information_extension.nodes().await?;
+        for node_info in node_infos {
+            self.add_node_info(&predicates, node_info);
        }
-
        self.finish()
    }

--- a/src/catalog/src/system_schema/information_schema/columns.rs
+++ b/src/catalog/src/system_schema/information_schema/columns.rs
@@ -257,8 +257,8 @@ impl InformationSchemaColumnsBuilder {
            .context(UpgradeWeakCatalogManagerRefSnafu)?;
        let predicates = Predicates::from_scan_request(&request);

-        for schema_name in catalog_manager.schema_names(&catalog_name).await? {
-            let mut stream = catalog_manager.tables(&catalog_name, &schema_name);
+        for schema_name in catalog_manager.schema_names(&catalog_name, None).await? {
+            let mut stream = catalog_manager.tables(&catalog_name, &schema_name, None);

            while let Some(table) = stream.try_next().await? {
                let keys = &table.table_info().meta.primary_key_indices;
--- a/src/catalog/src/system_schema/information_schema/key_column_usage.rs
+++ b/src/catalog/src/system_schema/information_schema/key_column_usage.rs
@@ -212,8 +212,8 @@ impl InformationSchemaKeyColumnUsageBuilder {
            .context(UpgradeWeakCatalogManagerRefSnafu)?;
        let predicates = Predicates::from_scan_request(&request);

-        for schema_name in catalog_manager.schema_names(&catalog_name).await? {
-            let mut stream = catalog_manager.tables(&catalog_name, &schema_name);
+        for schema_name in catalog_manager.schema_names(&catalog_name, None).await? {
+            let mut stream = catalog_manager.tables(&catalog_name, &schema_name, None);

            while let Some(table) = stream.try_next().await? {
                let mut primary_constraints = vec![];
--- a/src/catalog/src/system_schema/information_schema/partitions.rs
+++ b/src/catalog/src/system_schema/information_schema/partitions.rs
@@ -240,9 +240,9 @@ impl InformationSchemaPartitionsBuilder {

        let predicates = Predicates::from_scan_request(&request);

-        for schema_name in catalog_manager.schema_names(&catalog_name).await? {
+        for schema_name in catalog_manager.schema_names(&catalog_name, None).await? {
            let table_info_stream = catalog_manager
-                .tables(&catalog_name, &schema_name)
+                .tables(&catalog_name, &schema_name, None)
                .try_filter_map(|t| async move {
                    let table_info = t.table_info();
                    if table_info.table_type == TableType::Temporary {
--- a/src/catalog/src/system_schema/information_schema/procedure_info.rs
+++ b/src/catalog/src/system_schema/information_schema/procedure_info.rs
@@ -0,0 +1,241 @@
+// Copyright 2023 Greptime Team
+//
+// Licensed under the Apache License, Version 2.0 (the "License");
+// you may not use this file except in compliance with the License.
+// You may obtain a copy of the License at
+//
+//     http://www.apache.org/licenses/LICENSE-2.0
+//
+// Unless required by applicable law or agreed to in writing, software
+// distributed under the License is distributed on an "AS IS" BASIS,
+// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+// See the License for the specific language governing permissions and
+// limitations under the License.
+
+use std::sync::{Arc, Weak};
+
+use arrow_schema::SchemaRef as ArrowSchemaRef;
+use common_catalog::consts::INFORMATION_SCHEMA_PROCEDURE_INFO_TABLE_ID;
+use common_error::ext::BoxedError;
+use common_procedure::ProcedureInfo;
+use common_recordbatch::adapter::RecordBatchStreamAdapter;
+use common_recordbatch::{RecordBatch, SendableRecordBatchStream};
+use common_time::timestamp::Timestamp;
+use datafusion::execution::TaskContext;
+use datafusion::physical_plan::stream::RecordBatchStreamAdapter as DfRecordBatchStreamAdapter;
+use datafusion::physical_plan::streaming::PartitionStream as DfPartitionStream;
+use datafusion::physical_plan::SendableRecordBatchStream as DfSendableRecordBatchStream;
+use datatypes::prelude::{ConcreteDataType, ScalarVectorBuilder, VectorRef};
+use datatypes::schema::{ColumnSchema, Schema, SchemaRef};
+use datatypes::timestamp::TimestampMillisecond;
+use datatypes::value::Value;
+use datatypes::vectors::{StringVectorBuilder, TimestampMillisecondVectorBuilder};
+use snafu::ResultExt;
+use store_api::storage::{ScanRequest, TableId};
+
+use super::PROCEDURE_INFO;
+use crate::error::{CreateRecordBatchSnafu, InternalSnafu, Result};
+use crate::system_schema::information_schema::{InformationTable, Predicates};
+use crate::system_schema::utils;
+use crate::CatalogManager;
+
+const PROCEDURE_ID: &str = "procedure_id";
+const PROCEDURE_TYPE: &str = "procedure_type";
+const START_TIME: &str = "start_time";
+const END_TIME: &str = "end_time";
+const STATUS: &str = "status";
+const LOCK_KEYS: &str = "lock_keys";
+
+const INIT_CAPACITY: usize = 42;
+
+/// The `PROCEDURE_INFO` table provides information about the current procedure information of the cluster.
+///
+/// - `procedure_id`: the unique identifier of the procedure.
+/// - `procedure_name`: the name of the procedure.
+/// - `start_time`: the starting execution time of the procedure.
+/// - `end_time`: the ending execution time of the procedure.
+/// - `status`: the status of the procedure.
+/// - `lock_keys`: the lock keys of the procedure.
+///
+pub(super) struct InformationSchemaProcedureInfo {
+    schema: SchemaRef,
+    catalog_manager: Weak<dyn CatalogManager>,
+}
+
+impl InformationSchemaProcedureInfo {
+    pub(super) fn new(catalog_manager: Weak<dyn CatalogManager>) -> Self {
+        Self {
+            schema: Self::schema(),
+            catalog_manager,
+        }
+    }
+
+    pub(crate) fn schema() -> SchemaRef {
+        Arc::new(Schema::new(vec![
+            ColumnSchema::new(PROCEDURE_ID, ConcreteDataType::string_datatype(), false),
+            ColumnSchema::new(PROCEDURE_TYPE, ConcreteDataType::string_datatype(), false),
+            ColumnSchema::new(
+                START_TIME,
+                ConcreteDataType::timestamp_millisecond_datatype(),
+                true,
+            ),
+            ColumnSchema::new(
+                END_TIME,
+                ConcreteDataType::timestamp_millisecond_datatype(),
+                true,
+            ),
+            ColumnSchema::new(STATUS, ConcreteDataType::string_datatype(), false),
+            ColumnSchema::new(LOCK_KEYS, ConcreteDataType::string_datatype(), true),
+        ]))
+    }
+
+    fn builder(&self) -> InformationSchemaProcedureInfoBuilder {
+        InformationSchemaProcedureInfoBuilder::new(
+            self.schema.clone(),
+            self.catalog_manager.clone(),
+        )
+    }
+}
+
+impl InformationTable for InformationSchemaProcedureInfo {
+    fn table_id(&self) -> TableId {
+        INFORMATION_SCHEMA_PROCEDURE_INFO_TABLE_ID
+    }
+
+    fn table_name(&self) -> &'static str {
+        PROCEDURE_INFO
+    }
+
+    fn schema(&self) -> SchemaRef {
+        self.schema.clone()
+    }
+
+    fn to_stream(&self, request: ScanRequest) -> Result<SendableRecordBatchStream> {
+        let schema = self.schema.arrow_schema().clone();
+        let mut builder = self.builder();
+        let stream = Box::pin(DfRecordBatchStreamAdapter::new(
+            schema,
+            futures::stream::once(async move {
+                builder
+                    .make_procedure_info(Some(request))
+                    .await
+                    .map(|x| x.into_df_record_batch())
+                    .map_err(Into::into)
+            }),
+        ));
+        Ok(Box::pin(
+            RecordBatchStreamAdapter::try_new(stream)
+                .map_err(BoxedError::new)
+                .context(InternalSnafu)?,
+        ))
+    }
+}
+
+struct InformationSchemaProcedureInfoBuilder {
+    schema: SchemaRef,
+    catalog_manager: Weak<dyn CatalogManager>,
+
+    procedure_ids: StringVectorBuilder,
+    procedure_types: StringVectorBuilder,
+    start_times: TimestampMillisecondVectorBuilder,
+    end_times: TimestampMillisecondVectorBuilder,
+    statuses: StringVectorBuilder,
+    lock_keys: StringVectorBuilder,
+}
+
+impl InformationSchemaProcedureInfoBuilder {
+    fn new(schema: SchemaRef, catalog_manager: Weak<dyn CatalogManager>) -> Self {
+        Self {
+            schema,
+            catalog_manager,
+            procedure_ids: StringVectorBuilder::with_capacity(INIT_CAPACITY),
+            procedure_types: StringVectorBuilder::with_capacity(INIT_CAPACITY),
+            start_times: TimestampMillisecondVectorBuilder::with_capacity(INIT_CAPACITY),
+            end_times: TimestampMillisecondVectorBuilder::with_capacity(INIT_CAPACITY),
+            statuses: StringVectorBuilder::with_capacity(INIT_CAPACITY),
+            lock_keys: StringVectorBuilder::with_capacity(INIT_CAPACITY),
+        }
+    }
+
+    /// Construct the `information_schema.procedure_info` virtual table
+    async fn make_procedure_info(&mut self, request: Option<ScanRequest>) -> Result<RecordBatch> {
+        let predicates = Predicates::from_scan_request(&request);
+        let information_extension = utils::information_extension(&self.catalog_manager)?;
+        let procedures = information_extension.procedures().await?;
+        for (status, procedure_info) in procedures {
+            self.add_procedure(&predicates, status, procedure_info);
+        }
+        self.finish()
+    }
+
+    fn add_procedure(
+        &mut self,
+        predicates: &Predicates,
+        status: String,
+        procedure_info: ProcedureInfo,
+    ) {
+        let ProcedureInfo {
+            id,
+            type_name,
+            start_time_ms,
+            end_time_ms,
+            lock_keys,
+            ..
+        } = procedure_info;
+        let pid = id.to_string();
+        let start_time = TimestampMillisecond(Timestamp::new_millisecond(start_time_ms));
+        let end_time = TimestampMillisecond(Timestamp::new_millisecond(end_time_ms));
+        let lock_keys = lock_keys.join(",");
+
+        let row = [
+            (PROCEDURE_ID, &Value::from(pid.clone())),
+            (PROCEDURE_TYPE, &Value::from(type_name.clone())),
+            (START_TIME, &Value::from(start_time)),
+            (END_TIME, &Value::from(end_time)),
+            (STATUS, &Value::from(status.clone())),
+            (LOCK_KEYS, &Value::from(lock_keys.clone())),
+        ];
+        if !predicates.eval(&row) {
+            return;
+        }
+        self.procedure_ids.push(Some(&pid));
+        self.procedure_types.push(Some(&type_name));
+        self.start_times.push(Some(start_time));
+        self.end_times.push(Some(end_time));
+        self.statuses.push(Some(&status));
+        self.lock_keys.push(Some(&lock_keys));
+    }
+
+    fn finish(&mut self) -> Result<RecordBatch> {
+        let columns: Vec<VectorRef> = vec![
+            Arc::new(self.procedure_ids.finish()),
+            Arc::new(self.procedure_types.finish()),
+            Arc::new(self.start_times.finish()),
+            Arc::new(self.end_times.finish()),
+            Arc::new(self.statuses.finish()),
+            Arc::new(self.lock_keys.finish()),
+        ];
+        RecordBatch::new(self.schema.clone(), columns).context(CreateRecordBatchSnafu)
+    }
+}
+
+impl DfPartitionStream for InformationSchemaProcedureInfo {
+    fn schema(&self) -> &ArrowSchemaRef {
+        self.schema.arrow_schema()
+    }
+
+    fn execute(&self, _: Arc<TaskContext>) -> DfSendableRecordBatchStream {
+        let schema = self.schema.arrow_schema().clone();
+        let mut builder = self.builder();
+        Box::pin(DfRecordBatchStreamAdapter::new(
+            schema,
+            futures::stream::once(async move {
+                builder
+                    .make_procedure_info(None)
+                    .await
+                    .map(|x| x.into_df_record_batch())
+                    .map_err(Into::into)
+            }),
+        ))
+    }
+}
--- a/src/catalog/src/system_schema/information_schema/region_peers.rs
+++ b/src/catalog/src/system_schema/information_schema/region_peers.rs
@@ -176,9 +176,9 @@ impl InformationSchemaRegionPeersBuilder {

        let predicates = Predicates::from_scan_request(&request);

-        for schema_name in catalog_manager.schema_names(&catalog_name).await? {
+        for schema_name in catalog_manager.schema_names(&catalog_name, None).await? {
            let table_id_stream = catalog_manager
-                .tables(&catalog_name, &schema_name)
+                .tables(&catalog_name, &schema_name, None)
                .try_filter_map(|t| async move {
                    let table_info = t.table_info();
                    if table_info.table_type == TableType::Temporary {
@@ -224,8 +224,8 @@ impl InformationSchemaRegionPeersBuilder {
            let region_id = RegionId::new(table_id, route.region.id.region_number()).as_u64();
            let peer_id = route.leader_peer.clone().map(|p| p.id);
            let peer_addr = route.leader_peer.clone().map(|p| p.addr);
-            let status = if let Some(status) = route.leader_status {
-                Some(status.as_ref().to_string())
+            let state = if let Some(state) = route.leader_state {
+                Some(state.as_ref().to_string())
            } else {
                // Alive by default
                Some("ALIVE".to_string())
@@ -242,7 +242,7 @@ impl InformationSchemaRegionPeersBuilder {
            self.peer_ids.push(peer_id);
            self.peer_addrs.push(peer_addr.as_deref());
            self.is_leaders.push(Some("Yes"));
-            self.statuses.push(status.as_deref());
+            self.statuses.push(state.as_deref());
            self.down_seconds
                .push(route.leader_down_millis().map(|m| m / 1000));
        }
--- a/src/catalog/src/system_schema/information_schema/region_statistics.rs
+++ b/src/catalog/src/system_schema/information_schema/region_statistics.rs
@@ -0,0 +1,261 @@
+// Copyright 2023 Greptime Team
+//
+// Licensed under the Apache License, Version 2.0 (the "License");
+// you may not use this file except in compliance with the License.
+// You may obtain a copy of the License at
+//
+//     http://www.apache.org/licenses/LICENSE-2.0
+//
+// Unless required by applicable law or agreed to in writing, software
+// distributed under the License is distributed on an "AS IS" BASIS,
+// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+// See the License for the specific language governing permissions and
+// limitations under the License.
+
+use std::sync::{Arc, Weak};
+
+use arrow_schema::SchemaRef as ArrowSchemaRef;
+use common_catalog::consts::INFORMATION_SCHEMA_REGION_STATISTICS_TABLE_ID;
+use common_error::ext::BoxedError;
+use common_meta::datanode::RegionStat;
+use common_recordbatch::adapter::RecordBatchStreamAdapter;
+use common_recordbatch::{DfSendableRecordBatchStream, RecordBatch, SendableRecordBatchStream};
+use datafusion::execution::TaskContext;
+use datafusion::physical_plan::stream::RecordBatchStreamAdapter as DfRecordBatchStreamAdapter;
+use datafusion::physical_plan::streaming::PartitionStream as DfPartitionStream;
+use datatypes::prelude::{ConcreteDataType, ScalarVectorBuilder, VectorRef};
+use datatypes::schema::{ColumnSchema, Schema, SchemaRef};
+use datatypes::value::Value;
+use datatypes::vectors::{StringVectorBuilder, UInt32VectorBuilder, UInt64VectorBuilder};
+use snafu::ResultExt;
+use store_api::storage::{ScanRequest, TableId};
+
+use super::{InformationTable, REGION_STATISTICS};
+use crate::error::{CreateRecordBatchSnafu, InternalSnafu, Result};
+use crate::information_schema::Predicates;
+use crate::system_schema::utils;
+use crate::CatalogManager;
+
+const REGION_ID: &str = "region_id";
+const TABLE_ID: &str = "table_id";
+const REGION_NUMBER: &str = "region_number";
+const REGION_ROWS: &str = "region_rows";
+const DISK_SIZE: &str = "disk_size";
+const MEMTABLE_SIZE: &str = "memtable_size";
+const MANIFEST_SIZE: &str = "manifest_size";
+const SST_SIZE: &str = "sst_size";
+const INDEX_SIZE: &str = "index_size";
+const ENGINE: &str = "engine";
+const REGION_ROLE: &str = "region_role";
+
+const INIT_CAPACITY: usize = 42;
+
+/// The `REGION_STATISTICS` table provides information about the region statistics. Including fields:
+///
+/// - `region_id`: The region id.
+/// - `table_id`: The table id.
+/// - `region_number`: The region number.
+/// - `region_rows`: The number of rows in region.
+/// - `memtable_size`: The memtable size in bytes.
+/// - `disk_size`: The approximate disk size in bytes.
+/// - `manifest_size`: The manifest size in bytes.
+/// - `sst_size`: The sst data files size in bytes.
+/// - `index_size`: The sst index files size in bytes.
+/// - `engine`: The engine type.
+/// - `region_role`: The region role.
+///
+pub(super) struct InformationSchemaRegionStatistics {
+    schema: SchemaRef,
+    catalog_manager: Weak<dyn CatalogManager>,
+}
+
+impl InformationSchemaRegionStatistics {
+    pub(super) fn new(catalog_manager: Weak<dyn CatalogManager>) -> Self {
+        Self {
+            schema: Self::schema(),
+            catalog_manager,
+        }
+    }
+
+    pub(crate) fn schema() -> SchemaRef {
+        Arc::new(Schema::new(vec![
+            ColumnSchema::new(REGION_ID, ConcreteDataType::uint64_datatype(), false),
+            ColumnSchema::new(TABLE_ID, ConcreteDataType::uint32_datatype(), false),
+            ColumnSchema::new(REGION_NUMBER, ConcreteDataType::uint32_datatype(), false),
+            ColumnSchema::new(REGION_ROWS, ConcreteDataType::uint64_datatype(), true),
+            ColumnSchema::new(DISK_SIZE, ConcreteDataType::uint64_datatype(), true),
+            ColumnSchema::new(MEMTABLE_SIZE, ConcreteDataType::uint64_datatype(), true),
+            ColumnSchema::new(MANIFEST_SIZE, ConcreteDataType::uint64_datatype(), true),
+            ColumnSchema::new(SST_SIZE, ConcreteDataType::uint64_datatype(), true),
+            ColumnSchema::new(INDEX_SIZE, ConcreteDataType::uint64_datatype(), true),
+            ColumnSchema::new(ENGINE, ConcreteDataType::string_datatype(), true),
+            ColumnSchema::new(REGION_ROLE, ConcreteDataType::string_datatype(), true),
+        ]))
+    }
+
+    fn builder(&self) -> InformationSchemaRegionStatisticsBuilder {
+        InformationSchemaRegionStatisticsBuilder::new(
+            self.schema.clone(),
+            self.catalog_manager.clone(),
+        )
+    }
+}
+
+impl InformationTable for InformationSchemaRegionStatistics {
+    fn table_id(&self) -> TableId {
+        INFORMATION_SCHEMA_REGION_STATISTICS_TABLE_ID
+    }
+
+    fn table_name(&self) -> &'static str {
+        REGION_STATISTICS
+    }
+
+    fn schema(&self) -> SchemaRef {
+        self.schema.clone()
+    }
+
+    fn to_stream(&self, request: ScanRequest) -> Result<SendableRecordBatchStream> {
+        let schema = self.schema.arrow_schema().clone();
+        let mut builder = self.builder();
+
+        let stream = Box::pin(DfRecordBatchStreamAdapter::new(
+            schema,
+            futures::stream::once(async move {
+                builder
+                    .make_region_statistics(Some(request))
+                    .await
+                    .map(|x| x.into_df_record_batch())
+                    .map_err(Into::into)
+            }),
+        ));
+
+        Ok(Box::pin(
+            RecordBatchStreamAdapter::try_new(stream)
+                .map_err(BoxedError::new)
+                .context(InternalSnafu)?,
+        ))
+    }
+}
+
+struct InformationSchemaRegionStatisticsBuilder {
+    schema: SchemaRef,
+    catalog_manager: Weak<dyn CatalogManager>,
+
+    region_ids: UInt64VectorBuilder,
+    table_ids: UInt32VectorBuilder,
+    region_numbers: UInt32VectorBuilder,
+    region_rows: UInt64VectorBuilder,
+    disk_sizes: UInt64VectorBuilder,
+    memtable_sizes: UInt64VectorBuilder,
+    manifest_sizes: UInt64VectorBuilder,
+    sst_sizes: UInt64VectorBuilder,
+    index_sizes: UInt64VectorBuilder,
+    engines: StringVectorBuilder,
+    region_roles: StringVectorBuilder,
+}
+
+impl InformationSchemaRegionStatisticsBuilder {
+    fn new(schema: SchemaRef, catalog_manager: Weak<dyn CatalogManager>) -> Self {
+        Self {
+            schema,
+            catalog_manager,
+            region_ids: UInt64VectorBuilder::with_capacity(INIT_CAPACITY),
+            table_ids: UInt32VectorBuilder::with_capacity(INIT_CAPACITY),
+            region_numbers: UInt32VectorBuilder::with_capacity(INIT_CAPACITY),
+            region_rows: UInt64VectorBuilder::with_capacity(INIT_CAPACITY),
+            disk_sizes: UInt64VectorBuilder::with_capacity(INIT_CAPACITY),
+            memtable_sizes: UInt64VectorBuilder::with_capacity(INIT_CAPACITY),
+            manifest_sizes: UInt64VectorBuilder::with_capacity(INIT_CAPACITY),
+            sst_sizes: UInt64VectorBuilder::with_capacity(INIT_CAPACITY),
+            index_sizes: UInt64VectorBuilder::with_capacity(INIT_CAPACITY),
+            engines: StringVectorBuilder::with_capacity(INIT_CAPACITY),
+            region_roles: StringVectorBuilder::with_capacity(INIT_CAPACITY),
+        }
+    }
+
+    /// Construct a new `InformationSchemaRegionStatistics` from the collected data.
+    async fn make_region_statistics(
+        &mut self,
+        request: Option<ScanRequest>,
+    ) -> Result<RecordBatch> {
+        let predicates = Predicates::from_scan_request(&request);
+        let information_extension = utils::information_extension(&self.catalog_manager)?;
+        let region_stats = information_extension.region_stats().await?;
+        for region_stat in region_stats {
+            self.add_region_statistic(&predicates, region_stat);
+        }
+        self.finish()
+    }
+
+    fn add_region_statistic(&mut self, predicate: &Predicates, region_stat: RegionStat) {
+        let row = [
+            (REGION_ID, &Value::from(region_stat.id.as_u64())),
+            (TABLE_ID, &Value::from(region_stat.id.table_id())),
+            (REGION_NUMBER, &Value::from(region_stat.id.region_number())),
+            (REGION_ROWS, &Value::from(region_stat.num_rows)),
+            (DISK_SIZE, &Value::from(region_stat.approximate_bytes)),
+            (MEMTABLE_SIZE, &Value::from(region_stat.memtable_size)),
+            (MANIFEST_SIZE, &Value::from(region_stat.manifest_size)),
+            (SST_SIZE, &Value::from(region_stat.sst_size)),
+            (INDEX_SIZE, &Value::from(region_stat.index_size)),
+            (ENGINE, &Value::from(region_stat.engine.as_str())),
+            (REGION_ROLE, &Value::from(region_stat.role.to_string())),
+        ];
+
+        if !predicate.eval(&row) {
+            return;
+        }
+
+        self.region_ids.push(Some(region_stat.id.as_u64()));
+        self.table_ids.push(Some(region_stat.id.table_id()));
+        self.region_numbers
+            .push(Some(region_stat.id.region_number()));
+        self.region_rows.push(Some(region_stat.num_rows));
+        self.disk_sizes.push(Some(region_stat.approximate_bytes));
+        self.memtable_sizes.push(Some(region_stat.memtable_size));
+        self.manifest_sizes.push(Some(region_stat.manifest_size));
+        self.sst_sizes.push(Some(region_stat.sst_size));
+        self.index_sizes.push(Some(region_stat.index_size));
+        self.engines.push(Some(&region_stat.engine));
+        self.region_roles.push(Some(&region_stat.role.to_string()));
+    }
+
+    fn finish(&mut self) -> Result<RecordBatch> {
+        let columns: Vec<VectorRef> = vec![
+            Arc::new(self.region_ids.finish()),
+            Arc::new(self.table_ids.finish()),
+            Arc::new(self.region_numbers.finish()),
+            Arc::new(self.region_rows.finish()),
+            Arc::new(self.disk_sizes.finish()),
+            Arc::new(self.memtable_sizes.finish()),
+            Arc::new(self.manifest_sizes.finish()),
+            Arc::new(self.sst_sizes.finish()),
+            Arc::new(self.index_sizes.finish()),
+            Arc::new(self.engines.finish()),
+            Arc::new(self.region_roles.finish()),
+        ];
+
+        RecordBatch::new(self.schema.clone(), columns).context(CreateRecordBatchSnafu)
+    }
+}
+
+impl DfPartitionStream for InformationSchemaRegionStatistics {
+    fn schema(&self) -> &ArrowSchemaRef {
+        self.schema.arrow_schema()
+    }
+
+    fn execute(&self, _: Arc<TaskContext>) -> DfSendableRecordBatchStream {
+        let schema = self.schema.arrow_schema().clone();
+        let mut builder = self.builder();
+        Box::pin(DfRecordBatchStreamAdapter::new(
+            schema,
+            futures::stream::once(async move {
+                builder
+                    .make_region_statistics(None)
+                    .await
+                    .map(|x| x.into_df_record_batch())
+                    .map_err(Into::into)
+            }),
+        ))
+    }
+}
--- a/src/catalog/src/system_schema/information_schema/schemata.rs
+++ b/src/catalog/src/system_schema/information_schema/schemata.rs
@@ -171,7 +171,7 @@ impl InformationSchemaSchemataBuilder {
        let table_metadata_manager = utils::table_meta_manager(&self.catalog_manager)?;
        let predicates = Predicates::from_scan_request(&request);

-        for schema_name in catalog_manager.schema_names(&catalog_name).await? {
+        for schema_name in catalog_manager.schema_names(&catalog_name, None).await? {
            let opts = if let Some(table_metadata_manager) = &table_metadata_manager {
                table_metadata_manager
                    .schema_manager()
--- a/src/catalog/src/system_schema/information_schema/table_constraints.rs
+++ b/src/catalog/src/system_schema/information_schema/table_constraints.rs
@@ -176,8 +176,8 @@ impl InformationSchemaTableConstraintsBuilder {
            .context(UpgradeWeakCatalogManagerRefSnafu)?;
        let predicates = Predicates::from_scan_request(&request);

-        for schema_name in catalog_manager.schema_names(&catalog_name).await? {
-            let mut stream = catalog_manager.tables(&catalog_name, &schema_name);
+        for schema_name in catalog_manager.schema_names(&catalog_name, None).await? {
+            let mut stream = catalog_manager.tables(&catalog_name, &schema_name, None);

            while let Some(table) = stream.try_next().await? {
                let keys = &table.table_info().meta.primary_key_indices;
--- a/src/catalog/src/system_schema/information_schema/table_names.rs
+++ b/src/catalog/src/system_schema/information_schema/table_names.rs
@@ -12,7 +12,7 @@
 // See the License for the specific language governing permissions and
 // limitations under the License.

-/// All table names in `information_schema`.
+//! All table names in `information_schema`.

 pub const TABLES: &str = "tables";
 pub const COLUMNS: &str = "columns";
@@ -45,3 +45,5 @@ pub const TABLE_CONSTRAINTS: &str = "table_constraints";
 pub const CLUSTER_INFO: &str = "cluster_info";
 pub const VIEWS: &str = "views";
 pub const FLOWS: &str = "flows";
+pub const PROCEDURE_INFO: &str = "procedure_info";
+pub const REGION_STATISTICS: &str = "region_statistics";
--- a/src/catalog/src/system_schema/information_schema/tables.rs
+++ b/src/catalog/src/system_schema/information_schema/tables.rs
@@ -234,8 +234,8 @@ impl InformationSchemaTablesBuilder {
            .context(UpgradeWeakCatalogManagerRefSnafu)?;
        let predicates = Predicates::from_scan_request(&request);

-        for schema_name in catalog_manager.schema_names(&catalog_name).await? {
-            let mut stream = catalog_manager.tables(&catalog_name, &schema_name);
+        for schema_name in catalog_manager.schema_names(&catalog_name, None).await? {
+            let mut stream = catalog_manager.tables(&catalog_name, &schema_name, None);

            while let Some(table) = stream.try_next().await? {
                let table_info = table.table_info();
--- a/src/catalog/src/system_schema/information_schema/views.rs
+++ b/src/catalog/src/system_schema/information_schema/views.rs
@@ -192,8 +192,8 @@ impl InformationSchemaViewsBuilder {
            .context(CastManagerSnafu)?
            .view_info_cache()?;

-        for schema_name in catalog_manager.schema_names(&catalog_name).await? {
-            let mut stream = catalog_manager.tables(&catalog_name, &schema_name);
+        for schema_name in catalog_manager.schema_names(&catalog_name, None).await? {
+            let mut stream = catalog_manager.tables(&catalog_name, &schema_name, None);

            while let Some(table) = stream.try_next().await? {
                let table_info = table.table_info();
--- a/src/catalog/src/system_schema/memory_table.rs
+++ b/src/catalog/src/system_schema/memory_table.rs
@@ -74,7 +74,7 @@ impl MemoryTableBuilder {
    /// Construct the `information_schema.{table_name}` virtual table
    pub async fn memory_records(&mut self) -> Result<RecordBatch> {
        if self.columns.is_empty() {
-            RecordBatch::new_empty(self.schema.clone()).context(CreateRecordBatchSnafu)
+            Ok(RecordBatch::new_empty(self.schema.clone()))
        } else {
            RecordBatch::new(self.schema.clone(), std::mem::take(&mut self.columns))
                .context(CreateRecordBatchSnafu)
--- a/src/catalog/src/system_schema/pg_catalog.rs
+++ b/src/catalog/src/system_schema/pg_catalog.rs
@@ -18,15 +18,16 @@ mod pg_namespace;
 mod table_names;

 use std::collections::HashMap;
-use std::sync::{Arc, Weak};
+use std::sync::{Arc, LazyLock, Weak};

-use common_catalog::consts::{self, PG_CATALOG_NAME};
+use common_catalog::consts::{self, DEFAULT_CATALOG_NAME, DEFAULT_SCHEMA_NAME, PG_CATALOG_NAME};
 use datatypes::schema::ColumnSchema;
 use lazy_static::lazy_static;
 use paste::paste;
 use pg_catalog_memory_table::get_schema_columns;
 use pg_class::PGClass;
 use pg_namespace::PGNamespace;
+use session::context::{Channel, QueryContext};
 use table::TableRef;
 pub use table_names::*;

@@ -142,3 +143,12 @@ impl SystemSchemaProviderInner for PGCatalogProvider {
        &self.catalog_name
    }
 }
+
+/// Provide query context to call the [`CatalogManager`]'s method.
+static PG_QUERY_CTX: LazyLock<QueryContext> = LazyLock::new(|| {
+    QueryContext::with_channel(DEFAULT_CATALOG_NAME, DEFAULT_SCHEMA_NAME, Channel::Postgres)
+});
+
+fn query_ctx() -> Option<&'static QueryContext> {
+    Some(&PG_QUERY_CTX)
+}
--- a/src/catalog/src/system_schema/pg_catalog/pg_class.rs
+++ b/src/catalog/src/system_schema/pg_catalog/pg_class.rs
@@ -32,7 +32,7 @@ use store_api::storage::ScanRequest;
 use table::metadata::TableType;

 use super::pg_namespace::oid_map::PGNamespaceOidMapRef;
-use super::{OID_COLUMN_NAME, PG_CLASS};
+use super::{query_ctx, OID_COLUMN_NAME, PG_CLASS};
 use crate::error::{
    CreateRecordBatchSnafu, InternalSnafu, Result, UpgradeWeakCatalogManagerRefSnafu,
 };
@@ -202,8 +202,11 @@ impl PGClassBuilder {
            .upgrade()
            .context(UpgradeWeakCatalogManagerRefSnafu)?;
        let predicates = Predicates::from_scan_request(&request);
-        for schema_name in catalog_manager.schema_names(&catalog_name).await? {
-            let mut stream = catalog_manager.tables(&catalog_name, &schema_name);
+        for schema_name in catalog_manager
+            .schema_names(&catalog_name, query_ctx())
+            .await?
+        {
+            let mut stream = catalog_manager.tables(&catalog_name, &schema_name, query_ctx());
            while let Some(table) = stream.try_next().await? {
                let table_info = table.table_info();
                self.add_class(
--- a/src/catalog/src/system_schema/pg_catalog/pg_namespace.rs
+++ b/src/catalog/src/system_schema/pg_catalog/pg_namespace.rs
@@ -12,6 +12,9 @@
 // See the License for the specific language governing permissions and
 // limitations under the License.

+//! The `pg_catalog.pg_namespace` table implementation.
+//! namespace is a schema in greptime
+
 pub(super) mod oid_map;

 use std::sync::{Arc, Weak};
@@ -31,7 +34,7 @@ use datatypes::vectors::{StringVectorBuilder, UInt32VectorBuilder, VectorRef};
 use snafu::{OptionExt, ResultExt};
 use store_api::storage::ScanRequest;

-use super::{PGNamespaceOidMapRef, OID_COLUMN_NAME, PG_NAMESPACE};
+use super::{query_ctx, PGNamespaceOidMapRef, OID_COLUMN_NAME, PG_NAMESPACE};
 use crate::error::{
    CreateRecordBatchSnafu, InternalSnafu, Result, UpgradeWeakCatalogManagerRefSnafu,
 };
@@ -40,9 +43,6 @@ use crate::system_schema::utils::tables::{string_column, u32_column};
 use crate::system_schema::SystemTable;
 use crate::CatalogManager;

-/// The `pg_catalog.pg_namespace` table implementation.
-/// namespace is a schema in greptime
-
 const NSPNAME: &str = "nspname";
 const INIT_CAPACITY: usize = 42;

@@ -180,7 +180,10 @@ impl PGNamespaceBuilder {
            .upgrade()
            .context(UpgradeWeakCatalogManagerRefSnafu)?;
        let predicates = Predicates::from_scan_request(&request);
-        for schema_name in catalog_manager.schema_names(&catalog_name).await? {
+        for schema_name in catalog_manager
+            .schema_names(&catalog_name, query_ctx())
+            .await?
+        {
            self.add_namespace(&predicates, &schema_name);
        }
        self.finish()
--- a/src/catalog/src/system_schema/utils.rs
+++ b/src/catalog/src/system_schema/utils.rs
@@ -12,47 +12,33 @@
 // See the License for the specific language governing permissions and
 // limitations under the License.

-pub mod tables;
+use std::sync::Weak;

-use std::sync::{Arc, Weak};
-
-use common_config::Mode;
 use common_meta::key::TableMetadataManagerRef;
-use meta_client::client::MetaClient;
 use snafu::OptionExt;

-use crate::error::{Result, UpgradeWeakCatalogManagerRefSnafu};
+use crate::error::{GetInformationExtensionSnafu, Result, UpgradeWeakCatalogManagerRefSnafu};
+use crate::information_schema::InformationExtensionRef;
 use crate::kvbackend::KvBackendCatalogManager;
 use crate::CatalogManager;

-/// Try to get the server running mode from `[CatalogManager]` weak reference.
-pub fn running_mode(catalog_manager: &Weak<dyn CatalogManager>) -> Result<Option<Mode>> {
+pub mod tables;
+
+/// Try to get the `[InformationExtension]` from `[CatalogManager]` weak reference.
+pub fn information_extension(
+    catalog_manager: &Weak<dyn CatalogManager>,
+) -> Result<InformationExtensionRef> {
    let catalog_manager = catalog_manager
        .upgrade()
        .context(UpgradeWeakCatalogManagerRefSnafu)?;

-    Ok(catalog_manager
+    let information_extension = catalog_manager
        .as_any()
        .downcast_ref::<KvBackendCatalogManager>()
-        .map(|manager| manager.running_mode())
-        .copied())
-}
+        .map(|manager| manager.information_extension())
+        .context(GetInformationExtensionSnafu)?;

-/// Try to get the `[MetaClient]` from `[CatalogManager]` weak reference.
-pub fn meta_client(catalog_manager: &Weak<dyn CatalogManager>) -> Result<Option<Arc<MetaClient>>> {
-    let catalog_manager = catalog_manager
-        .upgrade()
-        .context(UpgradeWeakCatalogManagerRefSnafu)?;
-
-    let meta_client = match catalog_manager
-        .as_any()
-        .downcast_ref::<KvBackendCatalogManager>()
-    {
-        None => None,
-        Some(manager) => manager.meta_client(),
-    };
-
-    Ok(meta_client)
+    Ok(information_extension)
 }

 /// Try to get the `[TableMetadataManagerRef]` from `[CatalogManager]` weak reference.
--- a/src/catalog/src/table_source.rs
+++ b/src/catalog/src/table_source.rs
@@ -23,7 +23,7 @@ use datafusion::datasource::view::ViewTable;
 use datafusion::datasource::{provider_as_source, TableProvider};
 use datafusion::logical_expr::TableSource;
 use itertools::Itertools;
-use session::context::QueryContext;
+use session::context::QueryContextRef;
 use snafu::{ensure, OptionExt, ResultExt};
 use table::metadata::TableType;
 use table::table::adapter::DfTableProviderAdapter;
@@ -45,6 +45,7 @@ pub struct DfTableSourceProvider {
    disallow_cross_catalog_query: bool,
    default_catalog: String,
    default_schema: String,
+    query_ctx: QueryContextRef,
    plan_decoder: SubstraitPlanDecoderRef,
    enable_ident_normalization: bool,
 }
@@ -53,7 +54,7 @@ impl DfTableSourceProvider {
    pub fn new(
        catalog_manager: CatalogManagerRef,
        disallow_cross_catalog_query: bool,
-        query_ctx: &QueryContext,
+        query_ctx: QueryContextRef,
        plan_decoder: SubstraitPlanDecoderRef,
        enable_ident_normalization: bool,
    ) -> Self {
@@ -63,6 +64,7 @@ impl DfTableSourceProvider {
            resolved_tables: HashMap::new(),
            default_catalog: query_ctx.current_catalog().to_owned(),
            default_schema: query_ctx.current_schema(),
+            query_ctx,
            plan_decoder,
            enable_ident_normalization,
        }
@@ -71,8 +73,7 @@ impl DfTableSourceProvider {
    pub fn resolve_table_ref(&self, table_ref: TableReference) -> Result<ResolvedTableReference> {
        if self.disallow_cross_catalog_query {
            match &table_ref {
-                TableReference::Bare { .. } => (),
-                TableReference::Partial { .. } => {}
+                TableReference::Bare { .. } | TableReference::Partial { .. } => {}
                TableReference::Full {
                    catalog, schema, ..
                } => {
@@ -107,7 +108,7 @@ impl DfTableSourceProvider {

        let table = self
            .catalog_manager
-            .table(catalog_name, schema_name, table_name)
+            .table(catalog_name, schema_name, table_name, Some(&self.query_ctx))
            .await?
            .with_context(|| TableNotExistSnafu {
                table: format_full_table_name(catalog_name, schema_name, table_name),
@@ -210,12 +211,12 @@ mod tests {

    #[test]
    fn test_validate_table_ref() {
-        let query_ctx = &QueryContext::with("greptime", "public");
+        let query_ctx = Arc::new(QueryContext::with("greptime", "public"));

        let table_provider = DfTableSourceProvider::new(
            MemoryCatalogManager::with_default_setup(),
            true,
-            query_ctx,
+            query_ctx.clone(),
            DummyDecoder::arc(),
            true,
        );
@@ -258,7 +259,6 @@ mod tests {

    use arrow::datatypes::{DataType, Field, Schema, SchemaRef};
    use cache::{build_fundamental_cache_registry, with_default_composite_cache_registry};
-    use common_config::Mode;
    use common_meta::cache::{CacheRegistryBuilder, LayeredCacheRegistryBuilder};
    use common_meta::key::TableMetadataManager;
    use common_meta::kv_backend::memory::MemoryKvBackend;
@@ -268,6 +268,8 @@ mod tests {
    use datafusion::logical_expr::builder::LogicalTableSource;
    use datafusion::logical_expr::{col, lit, LogicalPlan, LogicalPlanBuilder};

+    use crate::information_schema::NoopInformationExtension;
+
    struct MockDecoder;
    impl MockDecoder {
        pub fn arc() -> Arc<Self> {
@@ -308,7 +310,7 @@ mod tests {

    #[tokio::test]
    async fn test_resolve_view() {
-        let query_ctx = &QueryContext::with("greptime", "public");
+        let query_ctx = Arc::new(QueryContext::with("greptime", "public"));
        let backend = Arc::new(MemoryKvBackend::default());
        let layered_cache_builder = LayeredCacheRegistryBuilder::default()
            .add_cache_registry(CacheRegistryBuilder::default().build());
@@ -322,10 +324,10 @@ mod tests {
        );

        let catalog_manager = KvBackendCatalogManager::new(
-            Mode::Standalone,
-            None,
+            Arc::new(NoopInformationExtension),
            backend.clone(),
            layered_cache_registry,
+            None,
        );
        let table_metadata_manager = TableMetadataManager::new(backend);
        let mut view_info = common_meta::key::test_utils::new_test_table_info(1024, vec![]);
@@ -344,8 +346,13 @@ mod tests {
            .await
            .unwrap();

-        let mut table_provider =
-            DfTableSourceProvider::new(catalog_manager, true, query_ctx, MockDecoder::arc(), true);
+        let mut table_provider = DfTableSourceProvider::new(
+            catalog_manager,
+            true,
+            query_ctx.clone(),
+            MockDecoder::arc(),
+            true,
+        );

        // View not found
        let table_ref = TableReference::bare("not_exists_view");
--- a/src/catalog/src/table_source/dummy_catalog.rs
+++ b/src/catalog/src/table_source/dummy_catalog.rs
@@ -112,7 +112,7 @@ impl SchemaProvider for DummySchemaProvider {
    async fn table(&self, name: &str) -> datafusion::error::Result<Option<Arc<dyn TableProvider>>> {
        let table = self
            .catalog_manager
-            .table(&self.catalog_name, &self.schema_name, name)
+            .table(&self.catalog_name, &self.schema_name, name, None)
            .await?
            .with_context(|| TableNotExistSnafu {
                table: format_full_table_name(&self.catalog_name, &self.schema_name, name),
--- a/src/client/Cargo.toml
+++ b/src/client/Cargo.toml
@@ -28,7 +28,7 @@ enum_dispatch = "0.3"
 futures-util.workspace = true
 lazy_static.workspace = true
 moka = { workspace = true, features = ["future"] }
-parking_lot = "0.12"
+parking_lot.workspace = true
 prometheus.workspace = true
 prost.workspace = true
 query.workspace = true
@@ -45,7 +45,6 @@ common-grpc-expr.workspace = true
 datanode.workspace = true
 derive-new = "0.5"
 tracing = "0.1"
-tracing-subscriber = { version = "0.3", features = ["env-filter"] }

 [dev-dependencies.substrait_proto]
 package = "substrait"
--- a/src/cmd/Cargo.toml
+++ b/src/cmd/Cargo.toml
@@ -10,7 +10,7 @@ name = "greptime"
 path = "src/bin/greptime.rs"

 [features]
-default = ["python"]
+default = ["python", "servers/pprof", "servers/mem-prof"]
 tokio-console = ["common-telemetry/tokio-console"]
 python = ["frontend/python"]

@@ -70,6 +70,7 @@ serde.workspace = true
 serde_json.workspace = true
 servers.workspace = true
 session.workspace = true
+similar-asserts.workspace = true
 snafu.workspace = true
 store-api.workspace = true
 substrait.workspace = true
@@ -77,7 +78,7 @@ table.workspace = true
 tokio.workspace = true
 toml.workspace = true
 tonic.workspace = true
-tracing-appender = "0.2"
+tracing-appender.workspace = true

 [target.'cfg(not(windows))'.dependencies]
 tikv-jemallocator = "0.6"
--- a/src/cmd/src/bin/greptime.rs
+++ b/src/cmd/src/bin/greptime.rs
@@ -15,10 +15,11 @@
 #![doc = include_str!("../../../../README.md")]

 use clap::{Parser, Subcommand};
-use cmd::error::Result;
+use cmd::error::{InitTlsProviderSnafu, Result};
 use cmd::options::GlobalOptions;
 use cmd::{cli, datanode, flownode, frontend, metasrv, standalone, App};
 use common_version::version;
+use servers::install_ring_crypto_provider;

 #[derive(Parser)]
 #[command(name = "greptime", author, version, long_version = version(), about)]
@@ -94,6 +95,7 @@ async fn main() -> Result<()> {

 async fn main_body() -> Result<()> {
    setup_human_panic();
+    install_ring_crypto_provider().map_err(|msg| InitTlsProviderSnafu { msg }.build())?;
    start(Command::parse()).await
 }

--- a/src/cmd/src/cli/bench.rs
+++ b/src/cmd/src/cli/bench.rs
@@ -158,7 +158,7 @@ fn create_region_routes(regions: Vec<RegionNumber>) -> Vec<RegionRoute> {
                addr: String::new(),
            }),
            follower_peers: vec![],
-            leader_status: None,
+            leader_state: None,
            leader_down_since: None,
        });
    }
--- a/src/cmd/src/cli/repl.rs
+++ b/src/cmd/src/cli/repl.rs
@@ -35,7 +35,6 @@ use either::Either;
 use meta_client::client::MetaClientBuilder;
 use query::datafusion::DatafusionQueryEngine;
 use query::parser::QueryLanguageParser;
-use query::plan::LogicalPlan;
 use query::query_engine::{DefaultSerializer, QueryEngineState};
 use query::QueryEngine;
 use rustyline::error::ReadlineError;
@@ -47,12 +46,12 @@ use substrait::{DFLogicalSubstraitConvertor, SubstraitPlan};
 use crate::cli::cmd::ReplCommand;
 use crate::cli::helper::RustylineHelper;
 use crate::cli::AttachCommand;
-use crate::error;
 use crate::error::{
    CollectRecordBatchesSnafu, ParseSqlSnafu, PlanStatementSnafu, PrettyPrintRecordBatchesSnafu,
    ReadlineSnafu, ReplCreationSnafu, RequestDatabaseSnafu, Result, StartMetaClientSnafu,
    SubstraitEncodeLogicalPlanSnafu,
 };
+use crate::{error, DistributedInformationExtension};

 /// Captures the state of the repl, gathers commands and executes them one by one
 pub struct Repl {
@@ -175,11 +174,11 @@ impl Repl {

            let plan = query_engine
                .planner()
-                .plan(stmt, query_ctx.clone())
+                .plan(&stmt, query_ctx.clone())
                .await
                .context(PlanStatementSnafu)?;

-            let LogicalPlan::DfPlan(plan) = query_engine
+            let plan = query_engine
                .optimize(&query_engine.engine_context(query_ctx), &plan)
                .context(PlanStatementSnafu)?;

@@ -276,11 +275,12 @@ async fn create_query_engine(meta_addr: &str) -> Result<DatafusionQueryEngine> {
        .build(),
    );

+    let information_extension = Arc::new(DistributedInformationExtension::new(meta_client.clone()));
    let catalog_manager = KvBackendCatalogManager::new(
-        Mode::Distributed,
-        Some(meta_client.clone()),
+        information_extension,
        cached_meta_backend.clone(),
        layered_cache_registry,
+        None,
    );
    let plugins: Plugins = Default::default();
    let state = Arc::new(QueryEngineState::new(
--- a/src/cmd/src/datanode.rs
+++ b/src/cmd/src/datanode.rs
@@ -272,9 +272,10 @@ impl StartCommand {
        info!("Datanode start command: {:#?}", self);
        info!("Datanode options: {:#?}", opts);

+        let plugin_opts = opts.plugins;
        let opts = opts.component;
        let mut plugins = Plugins::new();
-        plugins::setup_datanode_plugins(&mut plugins, &opts)
+        plugins::setup_datanode_plugins(&mut plugins, &plugin_opts, &opts)
            .await
            .context(StartDatanodeSnafu)?;

--- a/src/cmd/src/error.rs
+++ b/src/cmd/src/error.rs
@@ -24,6 +24,12 @@ use snafu::{Location, Snafu};
 #[snafu(visibility(pub))]
 #[stack_trace_debug]
 pub enum Error {
+    #[snafu(display("Failed to install ring crypto provider: {}", msg))]
+    InitTlsProvider {
+        #[snafu(implicit)]
+        location: Location,
+        msg: String,
+    },
    #[snafu(display("Failed to create default catalog and schema"))]
    InitMetadata {
        #[snafu(implicit)]
@@ -369,9 +375,10 @@ impl ErrorExt for Error {
            }
            Error::SubstraitEncodeLogicalPlan { source, .. } => source.status_code(),

-            Error::SerdeJson { .. } | Error::FileIo { .. } | Error::SpawnThread { .. } => {
-                StatusCode::Unexpected
-            }
+            Error::SerdeJson { .. }
+            | Error::FileIo { .. }
+            | Error::SpawnThread { .. }
+            | Error::InitTlsProvider { .. } => StatusCode::Unexpected,

            Error::Other { source, .. } => source.status_code(),

--- a/src/cmd/src/flownode.rs
+++ b/src/cmd/src/flownode.rs
@@ -41,7 +41,7 @@ use crate::error::{
    MissingConfigSnafu, Result, ShutdownFlownodeSnafu, StartFlownodeSnafu,
 };
 use crate::options::{GlobalOptions, GreptimeOptions};
-use crate::{log_versions, App};
+use crate::{log_versions, App, DistributedInformationExtension};

 pub const APP_NAME: &str = "greptime-flownode";

@@ -269,11 +269,13 @@ impl StartCommand {
            .build(),
        );

+        let information_extension =
+            Arc::new(DistributedInformationExtension::new(meta_client.clone()));
        let catalog_manager = KvBackendCatalogManager::new(
-            opts.mode,
-            Some(meta_client.clone()),
+            information_extension,
            cached_meta_backend.clone(),
            layered_cache_registry.clone(),
+            None,
        );

        let table_metadata_manager =
--- a/src/cmd/src/frontend.rs
+++ b/src/cmd/src/frontend.rs
@@ -36,8 +36,8 @@ use frontend::instance::builder::FrontendBuilder;
 use frontend::instance::{FrontendInstance, Instance as FeInstance};
 use frontend::server::Services;
 use meta_client::{MetaClientOptions, MetaClientType};
+use query::stats::StatementStatistics;
 use servers::tls::{TlsMode, TlsOption};
-use servers::Mode;
 use snafu::{OptionExt, ResultExt};
 use tracing_appender::non_blocking::WorkerGuard;

@@ -46,7 +46,7 @@ use crate::error::{
    Result, StartFrontendSnafu,
 };
 use crate::options::{GlobalOptions, GreptimeOptions};
-use crate::{log_versions, App};
+use crate::{log_versions, App, DistributedInformationExtension};

 type FrontendOptions = GreptimeOptions<frontend::frontend::FrontendOptions>;

@@ -266,9 +266,10 @@ impl StartCommand {
        info!("Frontend start command: {:#?}", self);
        info!("Frontend options: {:#?}", opts);

+        let plugin_opts = opts.plugins;
        let opts = opts.component;
        let mut plugins = Plugins::new();
-        plugins::setup_frontend_plugins(&mut plugins, &opts)
+        plugins::setup_frontend_plugins(&mut plugins, &plugin_opts, &opts)
            .await
            .context(StartFrontendSnafu)?;

@@ -315,11 +316,13 @@ impl StartCommand {
            .build(),
        );

+        let information_extension =
+            Arc::new(DistributedInformationExtension::new(meta_client.clone()));
        let catalog_manager = KvBackendCatalogManager::new(
-            Mode::Distributed,
-            Some(meta_client.clone()),
+            information_extension,
            cached_meta_backend.clone(),
            layered_cache_registry.clone(),
+            None,
        );

        let executor = HandlerGroupExecutor::new(vec![
@@ -340,6 +343,8 @@ impl StartCommand {
        // Some queries are expected to take long time.
        let channel_config = ChannelConfig {
            timeout: None,
+            tcp_nodelay: opts.datanode.client.tcp_nodelay,
+            connect_timeout: Some(opts.datanode.client.connect_timeout),
            ..Default::default()
        };
        let client = NodeClients::new(channel_config);
@@ -351,6 +356,7 @@ impl StartCommand {
            catalog_manager,
            Arc::new(client),
            meta_client,
+            StatementStatistics::new(opts.logging.slow_query.clone()),
        )
        .with_plugin(plugins.clone())
        .with_local_cache_invalidator(layered_cache_registry)
@@ -469,7 +475,7 @@ mod tests {
        };

        let mut plugins = Plugins::new();
-        plugins::setup_frontend_plugins(&mut plugins, &fe_opts)
+        plugins::setup_frontend_plugins(&mut plugins, &[], &fe_opts)
            .await
            .unwrap();

--- a/src/cmd/src/lib.rs
+++ b/src/cmd/src/lib.rs
@@ -15,7 +15,17 @@
 #![feature(assert_matches, let_chains)]

 use async_trait::async_trait;
+use catalog::information_schema::InformationExtension;
+use client::api::v1::meta::ProcedureStatus;
+use common_error::ext::BoxedError;
+use common_meta::cluster::{ClusterInfo, NodeInfo};
+use common_meta::datanode::RegionStat;
+use common_meta::ddl::{ExecutorContext, ProcedureExecutor};
+use common_meta::rpc::procedure;
+use common_procedure::{ProcedureInfo, ProcedureState};
 use common_telemetry::{error, info};
+use meta_client::MetaClientRef;
+use snafu::ResultExt;

 use crate::error::Result;

@@ -74,6 +84,7 @@ pub trait App: Send {
 }

 /// Log the versions of the application, and the arguments passed to the cli.
+///
 /// `version` should be the same as the output of cli "--version";
 /// and the `short_version` is the short version of the codes, often consist of git branch and commit.
 pub fn log_versions(version: &str, short_version: &str, app: &str) {
@@ -94,3 +105,69 @@ fn log_env_flags() {
        info!("argument: {}", argument);
    }
 }
+
+pub struct DistributedInformationExtension {
+    meta_client: MetaClientRef,
+}
+
+impl DistributedInformationExtension {
+    pub fn new(meta_client: MetaClientRef) -> Self {
+        Self { meta_client }
+    }
+}
+
+#[async_trait::async_trait]
+impl InformationExtension for DistributedInformationExtension {
+    type Error = catalog::error::Error;
+
+    async fn nodes(&self) -> std::result::Result<Vec<NodeInfo>, Self::Error> {
+        self.meta_client
+            .list_nodes(None)
+            .await
+            .map_err(BoxedError::new)
+            .context(catalog::error::ListNodesSnafu)
+    }
+
+    async fn procedures(&self) -> std::result::Result<Vec<(String, ProcedureInfo)>, Self::Error> {
+        let procedures = self
+            .meta_client
+            .list_procedures(&ExecutorContext::default())
+            .await
+            .map_err(BoxedError::new)
+            .context(catalog::error::ListProceduresSnafu)?
+            .procedures;
+        let mut result = Vec::with_capacity(procedures.len());
+        for procedure in procedures {
+            let pid = match procedure.id {
+                Some(pid) => pid,
+                None => return catalog::error::ProcedureIdNotFoundSnafu {}.fail(),
+            };
+            let pid = procedure::pb_pid_to_pid(&pid)
+                .map_err(BoxedError::new)
+                .context(catalog::error::ConvertProtoDataSnafu)?;
+            let status = ProcedureStatus::try_from(procedure.status)
+                .map(|v| v.as_str_name())
+                .unwrap_or("Unknown")
+                .to_string();
+            let procedure_info = ProcedureInfo {
+                id: pid,
+                type_name: procedure.type_name,
+                start_time_ms: procedure.start_time_ms,
+                end_time_ms: procedure.end_time_ms,
+                state: ProcedureState::Running,
+                lock_keys: procedure.lock_keys,
+            };
+            result.push((status, procedure_info));
+        }
+
+        Ok(result)
+    }
+
+    async fn region_stats(&self) -> std::result::Result<Vec<RegionStat>, Self::Error> {
+        self.meta_client
+            .list_region_stats()
+            .await
+            .map_err(BoxedError::new)
+            .context(catalog::error::ListRegionStatsSnafu)
+    }
+}
--- a/src/cmd/src/metasrv.rs
+++ b/src/cmd/src/metasrv.rs
@@ -48,6 +48,10 @@ impl Instance {
            _guard: guard,
        }
    }
+
+    pub fn get_inner(&self) -> &MetasrvInstance {
+        &self.instance
+    }
 }

 #[async_trait]
@@ -86,6 +90,14 @@ impl Command {
    pub fn load_options(&self, global_options: &GlobalOptions) -> Result<MetasrvOptions> {
        self.subcmd.load_options(global_options)
    }
+
+    pub fn config_file(&self) -> &Option<String> {
+        self.subcmd.config_file()
+    }
+
+    pub fn env_prefix(&self) -> &String {
+        self.subcmd.env_prefix()
+    }
 }

 #[derive(Parser)]
@@ -105,6 +117,18 @@ impl SubCommand {
            SubCommand::Start(cmd) => cmd.load_options(global_options),
        }
    }
+
+    fn config_file(&self) -> &Option<String> {
+        match self {
+            SubCommand::Start(cmd) => &cmd.config_file,
+        }
+    }
+
+    fn env_prefix(&self) -> &String {
+        match self {
+            SubCommand::Start(cmd) => &cmd.env_prefix,
+        }
+    }
 }

 #[derive(Debug, Default, Parser)]
@@ -249,9 +273,10 @@ impl StartCommand {
        info!("Metasrv start command: {:#?}", self);
        info!("Metasrv options: {:#?}", opts);

+        let plugin_opts = opts.plugins;
        let opts = opts.component;
        let mut plugins = Plugins::new();
-        plugins::setup_metasrv_plugins(&mut plugins, &opts)
+        plugins::setup_metasrv_plugins(&mut plugins, &plugin_opts, &opts)
            .await
            .context(StartMetaServerSnafu)?;

--- a/src/cmd/src/options.rs
+++ b/src/cmd/src/options.rs
@@ -15,6 +15,7 @@
 use clap::Parser;
 use common_config::Configurable;
 use common_runtime::global::RuntimeOptions;
+use plugins::PluginOptions;
 use serde::{Deserialize, Serialize};

 #[derive(Parser, Default, Debug, Clone)]
@@ -40,6 +41,8 @@ pub struct GlobalOptions {
 pub struct GreptimeOptions<T> {
    /// The runtime options.
    pub runtime: RuntimeOptions,
+    /// The plugin options.
+    pub plugins: Vec<PluginOptions>,

    /// The options of each component (like Datanode or Standalone) of GreptimeDB.
    #[serde(flatten)]
--- a/src/cmd/src/standalone.rs
+++ b/src/cmd/src/standalone.rs
@@ -12,19 +12,24 @@
 // See the License for the specific language governing permissions and
 // limitations under the License.

+use std::net::SocketAddr;
 use std::sync::Arc;
 use std::{fs, path};

 use async_trait::async_trait;
 use cache::{build_fundamental_cache_registry, with_default_composite_cache_registry};
+use catalog::information_schema::InformationExtension;
 use catalog::kvbackend::KvBackendCatalogManager;
 use clap::Parser;
+use client::api::v1::meta::RegionRole;
 use common_base::Plugins;
 use common_catalog::consts::{MIN_USER_FLOW_ID, MIN_USER_TABLE_ID};
 use common_config::{metadata_store_dir, Configurable, KvBackendConfig};
 use common_error::ext::BoxedError;
 use common_meta::cache::LayeredCacheRegistryBuilder;
 use common_meta::cache_invalidator::CacheInvalidatorRef;
+use common_meta::cluster::{NodeInfo, NodeStatus};
+use common_meta::datanode::RegionStat;
 use common_meta::ddl::flow_meta::{FlowMetadataAllocator, FlowMetadataAllocatorRef};
 use common_meta::ddl::table_meta::{TableMetadataAllocator, TableMetadataAllocatorRef};
 use common_meta::ddl::{DdlContext, NoopRegionFailureDetectorControl, ProcedureExecutorRef};
@@ -33,10 +38,11 @@ use common_meta::key::flow::{FlowMetadataManager, FlowMetadataManagerRef};
 use common_meta::key::{TableMetadataManager, TableMetadataManagerRef};
 use common_meta::kv_backend::KvBackendRef;
 use common_meta::node_manager::NodeManagerRef;
+use common_meta::peer::Peer;
 use common_meta::region_keeper::MemoryRegionKeeper;
 use common_meta::sequence::SequenceBuilder;
 use common_meta::wal_options_allocator::{WalOptionsAllocator, WalOptionsAllocatorRef};
-use common_procedure::ProcedureManagerRef;
+use common_procedure::{ProcedureInfo, ProcedureManagerRef};
 use common_telemetry::info;
 use common_telemetry::logging::{LoggingOptions, TracingOptions};
 use common_time::timezone::set_default_timezone;
@@ -44,6 +50,7 @@ use common_version::{short_version, version};
 use common_wal::config::DatanodeWalConfig;
 use datanode::config::{DatanodeOptions, ProcedureConfig, RegionEngineConfig, StorageConfig};
 use datanode::datanode::{Datanode, DatanodeBuilder};
+use datanode::region_server::RegionServer;
 use file_engine::config::EngineConfig as FileEngineConfig;
 use flow::{FlowWorkerManager, FlownodeBuilder, FrontendInvoker};
 use frontend::frontend::FrontendOptions;
@@ -55,6 +62,7 @@ use frontend::service_config::{
 };
 use meta_srv::metasrv::{FLOW_ID_SEQ, TABLE_ID_SEQ};
 use mito2::config::MitoConfig;
+use query::stats::StatementStatistics;
 use serde::{Deserialize, Serialize};
 use servers::export_metrics::ExportMetricsOption;
 use servers::grpc::GrpcOptions;
@@ -243,6 +251,13 @@ pub struct Instance {
    _guard: Vec<WorkerGuard>,
 }

+impl Instance {
+    /// Find the socket addr of a server by its `name`.
+    pub async fn server_addr(&self, name: &str) -> Option<SocketAddr> {
+        self.frontend.server_handlers().addr(name).await
+    }
+}
+
 #[async_trait]
 impl App for Instance {
    fn name(&self) -> &str {
@@ -333,7 +348,8 @@ pub struct StartCommand {
 }

 impl StartCommand {
-    fn load_options(
+    /// Load the GreptimeDB options from various sources (command line, config file or env).
+    pub fn load_options(
        &self,
        global_options: &GlobalOptions,
    ) -> Result<GreptimeOptions<StandaloneOptions>> {
@@ -423,7 +439,8 @@ impl StartCommand {
    #[allow(unreachable_code)]
    #[allow(unused_variables)]
    #[allow(clippy::diverging_sub_expression)]
-    async fn build(&self, opts: GreptimeOptions<StandaloneOptions>) -> Result<Instance> {
+    /// Build GreptimeDB instance with the loaded options.
+    pub async fn build(&self, opts: GreptimeOptions<StandaloneOptions>) -> Result<Instance> {
        common_runtime::init_global_runtimes(&opts.runtime);

        let guard = common_telemetry::init_global_logging(
@@ -438,15 +455,16 @@ impl StartCommand {
        info!("Standalone options: {opts:#?}");

        let mut plugins = Plugins::new();
+        let plugin_opts = opts.plugins;
        let opts = opts.component;
        let fe_opts = opts.frontend_options();
        let dn_opts = opts.datanode_options();

-        plugins::setup_frontend_plugins(&mut plugins, &fe_opts)
+        plugins::setup_frontend_plugins(&mut plugins, &plugin_opts, &fe_opts)
            .await
            .context(StartFrontendSnafu)?;

-        plugins::setup_datanode_plugins(&mut plugins, &dn_opts)
+        plugins::setup_datanode_plugins(&mut plugins, &plugin_opts, &dn_opts)
            .await
            .context(StartDatanodeSnafu)?;

@@ -477,22 +495,26 @@ impl StartCommand {
            .build(),
        );

-        let catalog_manager = KvBackendCatalogManager::new(
-            dn_opts.mode,
-            None,
-            kv_backend.clone(),
-            layered_cache_registry.clone(),
-        );
-
-        let table_metadata_manager =
-            Self::create_table_metadata_manager(kv_backend.clone()).await?;
-
        let datanode = DatanodeBuilder::new(dn_opts, plugins.clone())
            .with_kv_backend(kv_backend.clone())
            .build()
            .await
            .context(StartDatanodeSnafu)?;

+        let information_extension = Arc::new(StandaloneInformationExtension::new(
+            datanode.region_server(),
+            procedure_manager.clone(),
+        ));
+        let catalog_manager = KvBackendCatalogManager::new(
+            information_extension,
+            kv_backend.clone(),
+            layered_cache_registry.clone(),
+            Some(procedure_manager.clone()),
+        );
+
+        let table_metadata_manager =
+            Self::create_table_metadata_manager(kv_backend.clone()).await?;
+
        let flow_metadata_manager = Arc::new(FlowMetadataManager::new(kv_backend.clone()));
        let flow_builder = FlownodeBuilder::new(
            Default::default(),
@@ -556,6 +578,7 @@ impl StartCommand {
            catalog_manager.clone(),
            node_manager.clone(),
            ddl_task_executor.clone(),
+            StatementStatistics::new(opts.logging.slow_query.clone()),
        )
        .with_plugin(plugins.clone())
        .try_build()
@@ -641,6 +664,93 @@ impl StartCommand {
    }
 }

+pub struct StandaloneInformationExtension {
+    region_server: RegionServer,
+    procedure_manager: ProcedureManagerRef,
+    start_time_ms: u64,
+}
+
+impl StandaloneInformationExtension {
+    pub fn new(region_server: RegionServer, procedure_manager: ProcedureManagerRef) -> Self {
+        Self {
+            region_server,
+            procedure_manager,
+            start_time_ms: common_time::util::current_time_millis() as u64,
+        }
+    }
+}
+
+#[async_trait::async_trait]
+impl InformationExtension for StandaloneInformationExtension {
+    type Error = catalog::error::Error;
+
+    async fn nodes(&self) -> std::result::Result<Vec<NodeInfo>, Self::Error> {
+        let build_info = common_version::build_info();
+        let node_info = NodeInfo {
+            // For the standalone:
+            // - id always 0
+            // - empty string for peer_addr
+            peer: Peer {
+                id: 0,
+                addr: "".to_string(),
+            },
+            last_activity_ts: -1,
+            status: NodeStatus::Standalone,
+            version: build_info.version.to_string(),
+            git_commit: build_info.commit_short.to_string(),
+            // Use `self.start_time_ms` instead.
+            // It's not precise but enough.
+            start_time_ms: self.start_time_ms,
+        };
+        Ok(vec![node_info])
+    }
+
+    async fn procedures(&self) -> std::result::Result<Vec<(String, ProcedureInfo)>, Self::Error> {
+        self.procedure_manager
+            .list_procedures()
+            .await
+            .map_err(BoxedError::new)
+            .map(|procedures| {
+                procedures
+                    .into_iter()
+                    .map(|procedure| {
+                        let status = procedure.state.as_str_name().to_string();
+                        (status, procedure)
+                    })
+                    .collect::<Vec<_>>()
+            })
+            .context(catalog::error::ListProceduresSnafu)
+    }
+
+    async fn region_stats(&self) -> std::result::Result<Vec<RegionStat>, Self::Error> {
+        let stats = self
+            .region_server
+            .reportable_regions()
+            .into_iter()
+            .map(|stat| {
+                let region_stat = self
+                    .region_server
+                    .region_statistic(stat.region_id)
+                    .unwrap_or_default();
+                RegionStat {
+                    id: stat.region_id,
+                    rcus: 0,
+                    wcus: 0,
+                    approximate_bytes: region_stat.estimated_disk_size(),
+                    engine: stat.engine,
+                    role: RegionRole::from(stat.role).into(),
+                    num_rows: region_stat.num_rows,
+                    memtable_size: region_stat.memtable_size,
+                    manifest_size: region_stat.manifest_size,
+                    sst_size: region_stat.sst_size,
+                    index_size: region_stat.index_size,
+                }
+            })
+            .collect::<Vec<_>>();
+        Ok(stats)
+    }
+}
+
 #[cfg(test)]
 mod tests {
    use std::default::Default;
@@ -665,7 +775,7 @@ mod tests {
        };

        let mut plugins = Plugins::new();
-        plugins::setup_frontend_plugins(&mut plugins, &fe_opts)
+        plugins::setup_frontend_plugins(&mut plugins, &[], &fe_opts)
            .await
            .unwrap();

--- a/src/cmd/tests/load_config_test.rs
+++ b/src/cmd/tests/load_config_test.rs
@@ -16,13 +16,11 @@ use std::time::Duration;

 use cmd::options::GreptimeOptions;
 use cmd::standalone::StandaloneOptions;
-use common_base::readable_size::ReadableSize;
 use common_config::Configurable;
 use common_grpc::channel_manager::{
    DEFAULT_MAX_GRPC_RECV_MESSAGE_SIZE, DEFAULT_MAX_GRPC_SEND_MESSAGE_SIZE,
 };
-use common_runtime::global::RuntimeOptions;
-use common_telemetry::logging::{LoggingOptions, DEFAULT_OTLP_ENDPOINT};
+use common_telemetry::logging::{LoggingOptions, SlowQueryOptions, DEFAULT_OTLP_ENDPOINT};
 use common_wal::config::raft_engine::RaftEngineConfig;
 use common_wal::config::DatanodeWalConfig;
 use datanode::config::{DatanodeOptions, RegionEngineConfig, StorageConfig};
@@ -45,10 +43,6 @@ fn test_load_datanode_example_config() {
            .unwrap();

    let expected = GreptimeOptions::<DatanodeOptions> {
-        runtime: RuntimeOptions {
-            global_rt_size: 8,
-            compact_rt_size: 4,
-        },
        component: DatanodeOptions {
            node_id: Some(42),
            meta_client: Some(MetaClientOptions {
@@ -65,6 +59,7 @@ fn test_load_datanode_example_config() {
            wal: DatanodeWalConfig::RaftEngine(RaftEngineConfig {
                dir: Some("/tmp/greptimedb/wal".to_string()),
                sync_period: Some(Duration::from_secs(10)),
+                recovery_parallelism: 2,
                ..Default::default()
            }),
            storage: StorageConfig {
@@ -73,16 +68,8 @@ fn test_load_datanode_example_config() {
            },
            region_engine: vec![
                RegionEngineConfig::Mito(MitoConfig {
-                    num_workers: 8,
                    auto_flush_interval: Duration::from_secs(3600),
                    scan_parallelism: 0,
-                    global_write_buffer_size: ReadableSize::gb(1),
-                    global_write_buffer_reject_size: ReadableSize::gb(2),
-                    sst_meta_cache_size: ReadableSize::mb(128),
-                    vector_cache_size: ReadableSize::mb(512),
-                    page_cache_size: ReadableSize::mb(512),
-                    selector_result_cache_size: ReadableSize::mb(512),
-                    max_background_jobs: 4,
                    experimental_write_cache_ttl: Some(Duration::from_secs(60 * 60 * 8)),
                    ..Default::default()
                }),
@@ -107,9 +94,10 @@ fn test_load_datanode_example_config() {
            rpc_max_send_message_size: Some(DEFAULT_MAX_GRPC_SEND_MESSAGE_SIZE),
            ..Default::default()
        },
+        ..Default::default()
    };

-    assert_eq!(options, expected);
+    similar_asserts::assert_eq!(options, expected);
 }

 #[test]
@@ -119,10 +107,6 @@ fn test_load_frontend_example_config() {
        GreptimeOptions::<FrontendOptions>::load_layered_options(example_config.to_str(), "")
            .unwrap();
    let expected = GreptimeOptions::<FrontendOptions> {
-        runtime: RuntimeOptions {
-            global_rt_size: 8,
-            compact_rt_size: 4,
-        },
        component: FrontendOptions {
            default_timezone: Some("UTC".to_string()),
            meta_client: Some(MetaClientOptions {
@@ -155,8 +139,9 @@ fn test_load_frontend_example_config() {
            },
            ..Default::default()
        },
+        ..Default::default()
    };
-    assert_eq!(options, expected);
+    similar_asserts::assert_eq!(options, expected);
 }

 #[test]
@@ -166,10 +151,6 @@ fn test_load_metasrv_example_config() {
        GreptimeOptions::<MetasrvOptions>::load_layered_options(example_config.to_str(), "")
            .unwrap();
    let expected = GreptimeOptions::<MetasrvOptions> {
-        runtime: RuntimeOptions {
-            global_rt_size: 8,
-            compact_rt_size: 4,
-        },
        component: MetasrvOptions {
            selector: SelectorType::default(),
            data_home: "/tmp/metasrv/".to_string(),
@@ -178,8 +159,20 @@ fn test_load_metasrv_example_config() {
                level: Some("info".to_string()),
                otlp_endpoint: Some(DEFAULT_OTLP_ENDPOINT.to_string()),
                tracing_sample_ratio: Some(Default::default()),
+                slow_query: SlowQueryOptions {
+                    enable: false,
+                    threshold: Some(Duration::from_secs(10)),
+                    sample_ratio: Some(1.0),
+                },
                ..Default::default()
            },
+            datanode: meta_srv::metasrv::DatanodeOptions {
+                client: meta_srv::metasrv::DatanodeClientOptions {
+                    timeout: Duration::from_secs(10),
+                    connect_timeout: Duration::from_secs(10),
+                    tcp_nodelay: true,
+                },
+            },
            export_metrics: ExportMetricsOption {
                self_import: Some(Default::default()),
                remote_write: Some(Default::default()),
@@ -187,8 +180,9 @@ fn test_load_metasrv_example_config() {
            },
            ..Default::default()
        },
+        ..Default::default()
    };
-    assert_eq!(options, expected);
+    similar_asserts::assert_eq!(options, expected);
 }

 #[test]
@@ -198,30 +192,19 @@ fn test_load_standalone_example_config() {
        GreptimeOptions::<StandaloneOptions>::load_layered_options(example_config.to_str(), "")
            .unwrap();
    let expected = GreptimeOptions::<StandaloneOptions> {
-        runtime: RuntimeOptions {
-            global_rt_size: 8,
-            compact_rt_size: 4,
-        },
        component: StandaloneOptions {
            default_timezone: Some("UTC".to_string()),
            wal: DatanodeWalConfig::RaftEngine(RaftEngineConfig {
                dir: Some("/tmp/greptimedb/wal".to_string()),
                sync_period: Some(Duration::from_secs(10)),
+                recovery_parallelism: 2,
                ..Default::default()
            }),
            region_engine: vec![
                RegionEngineConfig::Mito(MitoConfig {
-                    num_workers: 8,
                    auto_flush_interval: Duration::from_secs(3600),
-                    scan_parallelism: 0,
-                    global_write_buffer_size: ReadableSize::gb(1),
-                    global_write_buffer_reject_size: ReadableSize::gb(2),
-                    sst_meta_cache_size: ReadableSize::mb(128),
-                    vector_cache_size: ReadableSize::mb(512),
-                    page_cache_size: ReadableSize::mb(512),
-                    selector_result_cache_size: ReadableSize::mb(512),
-                    max_background_jobs: 4,
                    experimental_write_cache_ttl: Some(Duration::from_secs(60 * 60 * 8)),
+                    scan_parallelism: 0,
                    ..Default::default()
                }),
                RegionEngineConfig::File(EngineConfig {}),
@@ -243,6 +226,7 @@ fn test_load_standalone_example_config() {
            },
            ..Default::default()
        },
+        ..Default::default()
    };
-    assert_eq!(options, expected);
+    similar_asserts::assert_eq!(options, expected);
 }
--- a/src/common/base/Cargo.toml
+++ b/src/common/base/Cargo.toml
@@ -8,11 +8,13 @@ license.workspace = true
 workspace = true

 [dependencies]
-anymap = "1.0.0-beta.2"
+anymap2 = "0.13"
+async-trait.workspace = true
 bitvec = "1.0"
 bytes.workspace = true
 common-error.workspace = true
 common-macro.workspace = true
+futures.workspace = true
 paste = "1.0"
 serde = { version = "1.0", features = ["derive"] }
 snafu.workspace = true
--- a/src/common/base/src/buffer.rs
+++ b/src/common/base/src/buffer.rs
@@ -1,242 +0,0 @@
-// Copyright 2023 Greptime Team
-//
-// Licensed under the Apache License, Version 2.0 (the "License");
-// you may not use this file except in compliance with the License.
-// You may obtain a copy of the License at
-//
-//     http://www.apache.org/licenses/LICENSE-2.0
-//
-// Unless required by applicable law or agreed to in writing, software
-// distributed under the License is distributed on an "AS IS" BASIS,
-// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
-// See the License for the specific language governing permissions and
-// limitations under the License.
-
-use std::any::Any;
-use std::io::{Read, Write};
-
-use bytes::{Buf, BufMut, BytesMut};
-use common_error::ext::ErrorExt;
-use common_macro::stack_trace_debug;
-use paste::paste;
-use snafu::{ensure, Location, ResultExt, Snafu};
-
-#[derive(Snafu)]
-#[snafu(visibility(pub))]
-#[stack_trace_debug]
-pub enum Error {
-    #[snafu(display(
-        "Destination buffer overflow, src_len: {}, dst_len: {}",
-        src_len,
-        dst_len
-    ))]
-    Overflow {
-        src_len: usize,
-        dst_len: usize,
-        #[snafu(implicit)]
-        location: Location,
-    },
-
-    #[snafu(display("Buffer underflow"))]
-    Underflow {
-        #[snafu(implicit)]
-        location: Location,
-    },
-
-    #[snafu(display("IO operation reach EOF"))]
-    Eof {
-        #[snafu(source)]
-        error: std::io::Error,
-        #[snafu(implicit)]
-        location: Location,
-    },
-}
-
-pub type Result<T> = std::result::Result<T, Error>;
-
-impl ErrorExt for Error {
-    fn as_any(&self) -> &dyn Any {
-        self
-    }
-}
-
-macro_rules! impl_read_le {
-    ( $($num_ty: ty), *) => {
-        $(
-            paste!{
-                // TODO(hl): default implementation requires allocating a
-                // temp buffer. maybe use more efficient impls in concrete buffers.
-                // see https://github.com/GrepTimeTeam/greptimedb/pull/97#discussion_r930798941
-                fn [<read_ $num_ty _le>](&mut self) -> Result<$num_ty> {
-                    let mut buf = [0u8; std::mem::size_of::<$num_ty>()];
-                    self.read_to_slice(&mut buf)?;
-                    Ok($num_ty::from_le_bytes(buf))
-                }
-
-                fn [<peek_ $num_ty _le>](&mut self) -> Result<$num_ty> {
-                    let mut buf = [0u8; std::mem::size_of::<$num_ty>()];
-                    self.peek_to_slice(&mut buf)?;
-                    Ok($num_ty::from_le_bytes(buf))
-                }
-            }
-        )*
-    }
-}
-
-macro_rules! impl_write_le {
-    ( $($num_ty: ty), *) => {
-        $(
-            paste!{
-                fn [<write_ $num_ty _le>](&mut self, n: $num_ty) -> Result<()> {
-                    self.write_from_slice(&n.to_le_bytes())?;
-                    Ok(())
-                }
-            }
-        )*
-    }
-}
-
-pub trait Buffer {
-    /// Returns remaining data size for read.
-    fn remaining_size(&self) -> usize;
-
-    /// Returns true if buffer has no data for read.
-    fn is_empty(&self) -> bool {
-        self.remaining_size() == 0
-    }
-
-    /// Peeks data into dst. This method should not change internal cursor,
-    /// invoke `advance_by` if needed.
-    /// # Panics
-    /// This method **may** panic if buffer does not have enough data to be copied to dst.
-    fn peek_to_slice(&self, dst: &mut [u8]) -> Result<()>;
-
-    /// Reads data into dst. This method will change internal cursor.
-    /// # Panics
-    /// This method **may** panic if buffer does not have enough data to be copied to dst.
-    fn read_to_slice(&mut self, dst: &mut [u8]) -> Result<()> {
-        self.peek_to_slice(dst)?;
-        self.advance_by(dst.len());
-        Ok(())
-    }
-
-    /// Advances internal cursor for next read.
-    /// # Panics
-    /// This method **may** panic if the offset after advancing exceeds the length of underlying buffer.
-    fn advance_by(&mut self, by: usize);
-
-    impl_read_le![u8, i8, u16, i16, u32, i32, u64, i64, f32, f64];
-}
-
-macro_rules! impl_buffer_for_bytes {
-    ( $($buf_ty:ty), *) => {
-        $(
-        impl Buffer for $buf_ty {
-            fn remaining_size(&self) -> usize{
-                self.len()
-            }
-
-            fn peek_to_slice(&self, dst: &mut [u8]) -> Result<()> {
-                let dst_len = dst.len();
-                ensure!(self.remaining() >= dst.len(), OverflowSnafu {
-                        src_len: self.remaining_size(),
-                        dst_len,
-                    }
-                );
-                dst.copy_from_slice(&self[0..dst_len]);
-                Ok(())
-            }
-
-            #[inline]
-            fn advance_by(&mut self, by: usize) {
-                self.advance(by);
-            }
-        }
-        )*
-    };
-}
-
-impl_buffer_for_bytes![bytes::Bytes, bytes::BytesMut];
-
-impl Buffer for &[u8] {
-    fn remaining_size(&self) -> usize {
-        self.len()
-    }
-
-    fn peek_to_slice(&self, dst: &mut [u8]) -> Result<()> {
-        let dst_len = dst.len();
-        ensure!(
-            self.len() >= dst.len(),
-            OverflowSnafu {
-                src_len: self.remaining_size(),
-                dst_len,
-            }
-        );
-        dst.copy_from_slice(&self[0..dst_len]);
-        Ok(())
-    }
-
-    fn read_to_slice(&mut self, dst: &mut [u8]) -> Result<()> {
-        ensure!(
-            self.len() >= dst.len(),
-            OverflowSnafu {
-                src_len: self.remaining_size(),
-                dst_len: dst.len(),
-            }
-        );
-        self.read_exact(dst).context(EofSnafu)
-    }
-
-    fn advance_by(&mut self, by: usize) {
-        *self = &self[by..];
-    }
-}
-
-/// Mutable buffer.
-pub trait BufferMut {
-    fn as_slice(&self) -> &[u8];
-
-    fn write_from_slice(&mut self, src: &[u8]) -> Result<()>;
-
-    impl_write_le![i8, u8, i16, u16, i32, u32, i64, u64, f32, f64];
-}
-
-impl BufferMut for BytesMut {
-    fn as_slice(&self) -> &[u8] {
-        self
-    }
-
-    fn write_from_slice(&mut self, src: &[u8]) -> Result<()> {
-        self.put_slice(src);
-        Ok(())
-    }
-}
-
-impl BufferMut for &mut [u8] {
-    fn as_slice(&self) -> &[u8] {
-        self
-    }
-
-    fn write_from_slice(&mut self, src: &[u8]) -> Result<()> {
-        // see std::io::Write::write_all
-        // https://doc.rust-lang.org/src/std/io/impls.rs.html#363
-        self.write_all(src).map_err(|_| {
-            OverflowSnafu {
-                src_len: src.len(),
-                dst_len: self.as_slice().len(),
-            }
-            .build()
-        })
-    }
-}
-
-impl BufferMut for Vec<u8> {
-    fn as_slice(&self) -> &[u8] {
-        self
-    }
-
-    fn write_from_slice(&mut self, src: &[u8]) -> Result<()> {
-        self.extend_from_slice(src);
-        Ok(())
-    }
-}
--- a/src/common/base/src/bytes.rs
+++ b/src/common/base/src/bytes.rs
@@ -44,6 +44,12 @@ impl From<Vec<u8>> for Bytes {
    }
 }

+impl From<Bytes> for Vec<u8> {
+    fn from(bytes: Bytes) -> Vec<u8> {
+        bytes.0.into()
+    }
+}
+
 impl Deref for Bytes {
    type Target = [u8];

--- a/src/common/base/src/lib.rs
+++ b/src/common/base/src/lib.rs
@@ -13,9 +13,9 @@
 // limitations under the License.

 pub mod bit_vec;
-pub mod buffer;
 pub mod bytes;
 pub mod plugins;
+pub mod range_read;
 #[allow(clippy::all)]
 pub mod readable_size;
 pub mod secrets;
--- a/src/common/base/src/plugins.rs
+++ b/src/common/base/src/plugins.rs
@@ -12,20 +12,21 @@
 // See the License for the specific language governing permissions and
 // limitations under the License.

-use std::any::Any;
 use std::sync::{Arc, RwLock, RwLockReadGuard, RwLockWriteGuard};

-/// [`Plugins`] is a wrapper of [AnyMap](https://github.com/chris-morgan/anymap) and provides a thread-safe way to store and retrieve plugins.
+use anymap2::SendSyncAnyMap;
+
+/// [`Plugins`] is a wrapper of [anymap2](https://github.com/azriel91/anymap2) and provides a thread-safe way to store and retrieve plugins.
 /// Make it Cloneable and we can treat it like an Arc struct.
 #[derive(Default, Clone)]
 pub struct Plugins {
-    inner: Arc<RwLock<anymap::Map<dyn Any + Send + Sync>>>,
+    inner: Arc<RwLock<SendSyncAnyMap>>,
 }

 impl Plugins {
    pub fn new() -> Self {
        Self {
-            inner: Arc::new(RwLock::new(anymap::Map::new())),
+            inner: Arc::new(RwLock::new(SendSyncAnyMap::new())),
        }
    }

@@ -37,6 +38,18 @@ impl Plugins {
        self.read().get::<T>().cloned()
    }

+    pub fn get_or_insert<T, F>(&self, f: F) -> T
+    where
+        T: 'static + Send + Sync + Clone,
+        F: FnOnce() -> T,
+    {
+        let mut binding = self.write();
+        if !binding.contains::<T>() {
+            binding.insert(f());
+        }
+        binding.get::<T>().cloned().unwrap()
+    }
+
    pub fn map_mut<T: 'static + Send + Sync, F, R>(&self, mapper: F) -> R
    where
        F: FnOnce(Option<&mut T>) -> R,
@@ -61,11 +74,11 @@ impl Plugins {
        self.read().is_empty()
    }

-    fn read(&self) -> RwLockReadGuard<anymap::Map<dyn Any + Send + Sync>> {
+    fn read(&self) -> RwLockReadGuard<SendSyncAnyMap> {
        self.inner.read().unwrap()
    }

-    fn write(&self) -> RwLockWriteGuard<anymap::Map<dyn Any + Send + Sync>> {
+    fn write(&self) -> RwLockWriteGuard<SendSyncAnyMap> {
        self.inner.write().unwrap()
    }
 }
--- a/src/common/base/src/range_read.rs
+++ b/src/common/base/src/range_read.rs
@@ -0,0 +1,105 @@
+// Copyright 2023 Greptime Team
+//
+// Licensed under the Apache License, Version 2.0 (the "License");
+// you may not use this file except in compliance with the License.
+// You may obtain a copy of the License at
+//
+//     http://www.apache.org/licenses/LICENSE-2.0
+//
+// Unless required by applicable law or agreed to in writing, software
+// distributed under the License is distributed on an "AS IS" BASIS,
+// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+// See the License for the specific language governing permissions and
+// limitations under the License.
+
+use std::io;
+use std::ops::Range;
+
+use async_trait::async_trait;
+use bytes::{BufMut, Bytes};
+use futures::{AsyncReadExt, AsyncSeekExt};
+
+/// `Metadata` contains the metadata of a source.
+pub struct Metadata {
+    /// The length of the source in bytes.
+    pub content_length: u64,
+}
+
+/// `RangeReader` reads a range of bytes from a source.
+#[async_trait]
+pub trait RangeReader: Send + Unpin {
+    /// Returns the metadata of the source.
+    async fn metadata(&mut self) -> io::Result<Metadata>;
+
+    /// Reads the bytes in the given range.
+    async fn read(&mut self, range: Range<u64>) -> io::Result<Bytes>;
+
+    /// Reads the bytes in the given range into the buffer.
+    ///
+    /// Handles the buffer based on its capacity:
+    /// - If the buffer is insufficient to hold the bytes, it will either:
+    ///   - Allocate additional space (e.g., for `Vec<u8>`)
+    ///   - Panic (e.g., for `&mut [u8]`)
+    async fn read_into(
+        &mut self,
+        range: Range<u64>,
+        buf: &mut (impl BufMut + Send),
+    ) -> io::Result<()> {
+        let bytes = self.read(range).await?;
+        buf.put_slice(&bytes);
+        Ok(())
+    }
+
+    /// Reads the bytes in the given ranges.
+    async fn read_vec(&mut self, ranges: &[Range<u64>]) -> io::Result<Vec<Bytes>> {
+        let mut result = Vec::with_capacity(ranges.len());
+        for range in ranges {
+            result.push(self.read(range.clone()).await?);
+        }
+        Ok(result)
+    }
+}
+
+#[async_trait]
+impl<R: RangeReader + Send + Unpin> RangeReader for &mut R {
+    async fn metadata(&mut self) -> io::Result<Metadata> {
+        (*self).metadata().await
+    }
+    async fn read(&mut self, range: Range<u64>) -> io::Result<Bytes> {
+        (*self).read(range).await
+    }
+    async fn read_into(
+        &mut self,
+        range: Range<u64>,
+        buf: &mut (impl BufMut + Send),
+    ) -> io::Result<()> {
+        (*self).read_into(range, buf).await
+    }
+    async fn read_vec(&mut self, ranges: &[Range<u64>]) -> io::Result<Vec<Bytes>> {
+        (*self).read_vec(ranges).await
+    }
+}
+
+/// `RangeReaderAdapter` bridges `RangeReader` and `AsyncRead + AsyncSeek`.
+pub struct RangeReaderAdapter<R>(pub R);
+
+/// Implements `RangeReader` for a type that implements `AsyncRead + AsyncSeek`.
+///
+/// TODO(zhongzc): It's a temporary solution for porting the codebase from `AsyncRead + AsyncSeek` to `RangeReader`.
+/// Until the codebase is fully ported to `RangeReader`, remove this implementation.
+#[async_trait]
+impl<R: futures::AsyncRead + futures::AsyncSeek + Send + Unpin> RangeReader
+    for RangeReaderAdapter<R>
+{
+    async fn metadata(&mut self) -> io::Result<Metadata> {
+        let content_length = self.0.seek(io::SeekFrom::End(0)).await?;
+        Ok(Metadata { content_length })
+    }
+
+    async fn read(&mut self, range: Range<u64>) -> io::Result<Bytes> {
+        let mut buf = vec![0; (range.end - range.start) as usize];
+        self.0.seek(io::SeekFrom::Start(range.start)).await?;
+        self.0.read_exact(&mut buf).await?;
+        Ok(Bytes::from(buf))
+    }
+}
--- a/src/common/base/src/secrets.rs
+++ b/src/common/base/src/secrets.rs
@@ -46,8 +46,9 @@ impl From<String> for SecretString {
    }
 }

-/// Wrapper type for values that contains secrets, which attempts to limit
-/// accidental exposure and ensure secrets are wiped from memory when dropped.
+/// Wrapper type for values that contains secrets.
+///
+/// It attempts to limit accidental exposure and ensure secrets are wiped from memory when dropped.
 /// (e.g. passwords, cryptographic keys, access tokens or other credentials)
 ///
 /// Access to the secret inner value occurs through the [`ExposeSecret`]
--- a/src/common/base/tests/buffer_tests.rs
+++ b/src/common/base/tests/buffer_tests.rs
@@ -1,182 +0,0 @@
-// Copyright 2023 Greptime Team
-//
-// Licensed under the Apache License, Version 2.0 (the "License");
-// you may not use this file except in compliance with the License.
-// You may obtain a copy of the License at
-//
-//     http://www.apache.org/licenses/LICENSE-2.0
-//
-// Unless required by applicable law or agreed to in writing, software
-// distributed under the License is distributed on an "AS IS" BASIS,
-// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
-// See the License for the specific language governing permissions and
-// limitations under the License.
-
-#![feature(assert_matches)]
-
-#[cfg(test)]
-mod tests {
-    use std::assert_matches::assert_matches;
-
-    use bytes::{Buf, Bytes, BytesMut};
-    use common_base::buffer::Error::Overflow;
-    use common_base::buffer::{Buffer, BufferMut};
-    use paste::paste;
-
-    #[test]
-    pub fn test_buffer_read_write() {
-        let mut buf = BytesMut::with_capacity(16);
-        buf.write_u64_le(1234u64).unwrap();
-        let result = buf.peek_u64_le().unwrap();
-        assert_eq!(1234u64, result);
-        buf.advance_by(8);
-
-        buf.write_from_slice("hello, world".as_bytes()).unwrap();
-        let mut content = vec![0u8; 5];
-        buf.peek_to_slice(&mut content).unwrap();
-        let read = String::from_utf8_lossy(&content);
-        assert_eq!("hello", read);
-        buf.advance_by(5);
-        // after read, buffer should still have 7 bytes to read.
-        assert_eq!(7, buf.remaining());
-
-        let mut content = vec![0u8; 6];
-        buf.read_to_slice(&mut content).unwrap();
-        let read = String::from_utf8_lossy(&content);
-        assert_eq!(", worl", read);
-        // after read, buffer should still have 1 byte to read.
-        assert_eq!(1, buf.remaining());
-    }
-
-    #[test]
-    pub fn test_buffer_read() {
-        let mut bytes = Bytes::from_static("hello".as_bytes());
-        assert_eq!(5, bytes.remaining_size());
-        assert_eq!(b'h', bytes.peek_u8_le().unwrap());
-        bytes.advance_by(1);
-        assert_eq!(4, bytes.remaining_size());
-    }
-
-    macro_rules! test_primitive_read_write {
-        ( $($num_ty: ty), *) => {
-            $(
-                paste!{
-                    #[test]
-                    fn [<test_read_write_ $num_ty>]() {
-                        assert_eq!($num_ty::MAX,(&mut $num_ty::MAX.to_le_bytes() as &[u8]).[<read_ $num_ty _le>]().unwrap());
-                        assert_eq!($num_ty::MIN,(&mut $num_ty::MIN.to_le_bytes() as &[u8]).[<read_ $num_ty _le>]().unwrap());
-                    }
-                }
-            )*
-        }
-    }
-
-    test_primitive_read_write![u8, u16, u32, u64, i8, i16, i32, i64, f32, f64];
-
-    #[test]
-    pub fn test_read_write_from_slice_buffer() {
-        let mut buf = "hello".as_bytes();
-        assert_eq!(104, buf.peek_u8_le().unwrap());
-        buf.advance_by(1);
-        assert_eq!(101, buf.peek_u8_le().unwrap());
-        buf.advance_by(1);
-        assert_eq!(108, buf.peek_u8_le().unwrap());
-        buf.advance_by(1);
-        assert_eq!(108, buf.peek_u8_le().unwrap());
-        buf.advance_by(1);
-        assert_eq!(111, buf.peek_u8_le().unwrap());
-        buf.advance_by(1);
-        assert_matches!(buf.peek_u8_le(), Err(Overflow { .. }));
-    }
-
-    #[test]
-    pub fn test_read_u8_from_slice_buffer() {
-        let mut buf = "hello".as_bytes();
-        assert_eq!(104, buf.read_u8_le().unwrap());
-        assert_eq!(101, buf.read_u8_le().unwrap());
-        assert_eq!(108, buf.read_u8_le().unwrap());
-        assert_eq!(108, buf.read_u8_le().unwrap());
-        assert_eq!(111, buf.read_u8_le().unwrap());
-        assert_matches!(buf.read_u8_le(), Err(Overflow { .. }));
-    }
-
-    #[test]
-    pub fn test_read_write_numbers() {
-        let mut buf: Vec<u8> = vec![];
-        buf.write_u64_le(1234).unwrap();
-        assert_eq!(1234, (&buf[..]).read_u64_le().unwrap());
-
-        buf.write_u32_le(4242).unwrap();
-        let mut p = &buf[..];
-        assert_eq!(1234, p.read_u64_le().unwrap());
-        assert_eq!(4242, p.read_u32_le().unwrap());
-    }
-
-    macro_rules! test_primitive_vec_read_write {
-        ( $($num_ty: ty), *) => {
-            $(
-                paste!{
-                    #[test]
-                    fn [<test_read_write_ $num_ty _from_vec_buffer>]() {
-                        let mut buf = vec![];
-                        let _ = buf.[<write_ $num_ty _le>]($num_ty::MAX).unwrap();
-                        assert_eq!($num_ty::MAX, buf.as_slice().[<read_ $num_ty _le>]().unwrap());
-                    }
-                }
-            )*
-        }
-    }
-
-    test_primitive_vec_read_write![u8, u16, u32, u64, i8, i16, i32, i64, f32, f64];
-
-    #[test]
-    pub fn test_peek_write_from_vec_buffer() {
-        let mut buf: Vec<u8> = vec![];
-        buf.write_from_slice("hello".as_bytes()).unwrap();
-        let mut slice = buf.as_slice();
-        assert_eq!(104, slice.peek_u8_le().unwrap());
-        slice.advance_by(1);
-        assert_eq!(101, slice.peek_u8_le().unwrap());
-        slice.advance_by(1);
-        assert_eq!(108, slice.peek_u8_le().unwrap());
-        slice.advance_by(1);
-        assert_eq!(108, slice.peek_u8_le().unwrap());
-        slice.advance_by(1);
-        assert_eq!(111, slice.peek_u8_le().unwrap());
-        slice.advance_by(1);
-        assert_matches!(slice.read_u8_le(), Err(Overflow { .. }));
-    }
-
-    macro_rules! test_primitive_bytes_read_write {
-        ( $($num_ty: ty), *) => {
-            $(
-                paste!{
-                    #[test]
-                    fn [<test_read_write_ $num_ty _from_bytes>]() {
-                        let mut bytes = bytes::Bytes::from($num_ty::MAX.to_le_bytes().to_vec());
-                        assert_eq!($num_ty::MAX, bytes.[<read_ $num_ty _le>]().unwrap());
-
-                        let mut bytes = bytes::Bytes::from($num_ty::MIN.to_le_bytes().to_vec());
-                        assert_eq!($num_ty::MIN, bytes.[<read_ $num_ty _le>]().unwrap());
-                    }
-                }
-            )*
-        }
-    }
-
-    test_primitive_bytes_read_write![u8, u16, u32, u64, i8, i16, i32, i64, f32, f64];
-
-    #[test]
-    pub fn test_write_overflow() {
-        let mut buf = [0u8; 4];
-        assert_matches!(
-            (&mut buf[..]).write_from_slice("hell".as_bytes()),
-            Ok { .. }
-        );
-
-        assert_matches!(
-            (&mut buf[..]).write_from_slice("hello".as_bytes()),
-            Err(common_base::buffer::Error::Overflow { .. })
-        );
-    }
-}
--- a/src/common/catalog/src/consts.rs
+++ b/src/common/catalog/src/consts.rs
@@ -98,14 +98,20 @@ pub const INFORMATION_SCHEMA_CLUSTER_INFO_TABLE_ID: u32 = 31;
 pub const INFORMATION_SCHEMA_VIEW_TABLE_ID: u32 = 32;
 /// id for information_schema.FLOWS
 pub const INFORMATION_SCHEMA_FLOW_TABLE_ID: u32 = 33;
-/// ----- End of information_schema tables -----
+/// id for information_schema.procedure_info
+pub const INFORMATION_SCHEMA_PROCEDURE_INFO_TABLE_ID: u32 = 34;
+/// id for information_schema.region_statistics
+pub const INFORMATION_SCHEMA_REGION_STATISTICS_TABLE_ID: u32 = 35;
+
+// ----- End of information_schema tables -----

 /// ----- Begin of pg_catalog tables -----
 pub const PG_CATALOG_PG_CLASS_TABLE_ID: u32 = 256;
 pub const PG_CATALOG_PG_TYPE_TABLE_ID: u32 = 257;
 pub const PG_CATALOG_PG_NAMESPACE_TABLE_ID: u32 = 258;

-/// ----- End of pg_catalog tables -----
+// ----- End of pg_catalog tables -----
+
 pub const MITO_ENGINE: &str = "mito";
 pub const MITO2_ENGINE: &str = "mito2";
 pub const METRIC_ENGINE: &str = "metric";
--- a/src/common/error/src/status_code.rs
+++ b/src/common/error/src/status_code.rs
@@ -38,6 +38,8 @@ pub enum StatusCode {
    Cancelled = 1005,
    /// Illegal state, can be exposed to users.
    IllegalState = 1006,
+    /// Caused by some error originated from external system.
+    External = 1007,
    // ====== End of common status code ================

    // ====== Begin of SQL related status code =========
@@ -162,7 +164,8 @@ impl StatusCode {
            | StatusCode::InvalidAuthHeader
            | StatusCode::AccessDenied
            | StatusCode::PermissionDenied
-            | StatusCode::RequestOutdated => false,
+            | StatusCode::RequestOutdated
+            | StatusCode::External => false,
        }
    }

@@ -177,7 +180,9 @@ impl StatusCode {
            | StatusCode::IllegalState
            | StatusCode::EngineExecuteQuery
            | StatusCode::StorageUnavailable
-            | StatusCode::RuntimeResourcesExhausted => true,
+            | StatusCode::RuntimeResourcesExhausted
+            | StatusCode::External => true,
+
            StatusCode::Success
            | StatusCode::Unsupported
            | StatusCode::InvalidArguments
@@ -256,7 +261,7 @@ macro_rules! define_into_tonic_status {
 pub fn status_to_tonic_code(status_code: StatusCode) -> Code {
    match status_code {
        StatusCode::Success => Code::Ok,
-        StatusCode::Unknown => Code::Unknown,
+        StatusCode::Unknown | StatusCode::External => Code::Unknown,
        StatusCode::Unsupported => Code::Unimplemented,
        StatusCode::Unexpected
        | StatusCode::IllegalState
--- a/src/common/function/Cargo.toml
+++ b/src/common/function/Cargo.toml
@@ -9,7 +9,7 @@ workspace = true

 [features]
 default = ["geo"]
-geo = ["geohash", "h3o"]
+geo = ["geohash", "h3o", "s2"]

 [dependencies]
 api.workspace = true
@@ -27,12 +27,15 @@ common-time.workspace = true
 common-version.workspace = true
 datafusion.workspace = true
 datatypes.workspace = true
+derive_more = { version = "1", default-features = false, features = ["display"] }
 geohash = { version = "0.13", optional = true }
 h3o = { version = "0.6", optional = true }
+jsonb.workspace = true
 num = "0.4"
 num-traits = "0.2"
 once_cell.workspace = true
 paste = "1.0"
+s2 = { version = "0.0.12", optional = true }
 serde.workspace = true
 serde_json.workspace = true
 session.workspace = true
--- a/src/common/function/src/function_registry.rs
+++ b/src/common/function/src/function_registry.rs
@@ -22,6 +22,7 @@ use crate::function::{AsyncFunctionRef, FunctionRef};
 use crate::scalars::aggregate::{AggregateFunctionMetaRef, AggregateFunctions};
 use crate::scalars::date::DateFunction;
 use crate::scalars::expression::ExpressionFunction;
+use crate::scalars::json::JsonFunction;
 use crate::scalars::matches::MatchesFunction;
 use crate::scalars::math::MathFunction;
 use crate::scalars::numpy::NumpyFunction;
@@ -116,6 +117,9 @@ pub static FUNCTION_REGISTRY: Lazy<Arc<FunctionRegistry>> = Lazy::new(|| {
    SystemFunction::register(&function_registry);
    TableFunction::register(&function_registry);

+    // Json related functions
+    JsonFunction::register(&function_registry);
+
    // Geo functions
    #[cfg(feature = "geo")]
    crate::scalars::geo::GeoFunctions::register(&function_registry);
--- a/src/common/function/src/scalars.rs
+++ b/src/common/function/src/scalars.rs
@@ -17,9 +17,11 @@ pub(crate) mod date;
 pub mod expression;
 #[cfg(feature = "geo")]
 pub mod geo;
+pub mod json;
 pub mod matches;
 pub mod math;
 pub mod numpy;
+
 #[cfg(test)]
 pub(crate) mod test;
 pub(crate) mod timestamp;
--- a/src/common/function/src/scalars/aggregate.rs
+++ b/src/common/function/src/scalars/aggregate.rs
@@ -16,7 +16,6 @@ mod argmax;
 mod argmin;
 mod diff;
 mod mean;
-mod percentile;
 mod polyval;
 mod scipy_stats_norm_cdf;
 mod scipy_stats_norm_pdf;
@@ -28,7 +27,6 @@ pub use argmin::ArgminAccumulatorCreator;
 use common_query::logical_plan::AggregateFunctionCreatorRef;
 pub use diff::DiffAccumulatorCreator;
 pub use mean::MeanAccumulatorCreator;
-pub use percentile::PercentileAccumulatorCreator;
 pub use polyval::PolyvalAccumulatorCreator;
 pub use scipy_stats_norm_cdf::ScipyStatsNormCdfAccumulatorCreator;
 pub use scipy_stats_norm_pdf::ScipyStatsNormPdfAccumulatorCreator;
@@ -91,8 +89,14 @@ impl AggregateFunctions {
        register_aggr_func!("polyval", 2, PolyvalAccumulatorCreator);
        register_aggr_func!("argmax", 1, ArgmaxAccumulatorCreator);
        register_aggr_func!("argmin", 1, ArgminAccumulatorCreator);
-        register_aggr_func!("percentile", 2, PercentileAccumulatorCreator);
        register_aggr_func!("scipystatsnormcdf", 2, ScipyStatsNormCdfAccumulatorCreator);
        register_aggr_func!("scipystatsnormpdf", 2, ScipyStatsNormPdfAccumulatorCreator);
+
+        #[cfg(feature = "geo")]
+        register_aggr_func!(
+            "json_encode_path",
+            3,
+            super::geo::encoding::JsonPathEncodeFunctionCreator
+        );
    }
 }
--- a/src/common/function/src/scalars/aggregate/argmax.rs
+++ b/src/common/function/src/scalars/aggregate/argmax.rs
@@ -16,7 +16,10 @@ use std::cmp::Ordering;
 use std::sync::Arc;

 use common_macro::{as_aggr_func_creator, AggrFuncTypeStore};
-use common_query::error::{BadAccumulatorImplSnafu, CreateAccumulatorSnafu, Result};
+use common_query::error::{
+    BadAccumulatorImplSnafu, CreateAccumulatorSnafu, InvalidInputStateSnafu, Result,
+};
+use common_query::logical_plan::accumulator::AggrFuncTypeStore;
 use common_query::logical_plan::{Accumulator, AggregateFunctionCreator};
 use common_query::prelude::*;
 use datatypes::prelude::*;
--- a/src/common/function/src/scalars/aggregate/argmin.rs
+++ b/src/common/function/src/scalars/aggregate/argmin.rs
@@ -16,7 +16,10 @@ use std::cmp::Ordering;
 use std::sync::Arc;

 use common_macro::{as_aggr_func_creator, AggrFuncTypeStore};
-use common_query::error::{BadAccumulatorImplSnafu, CreateAccumulatorSnafu, Result};
+use common_query::error::{
+    BadAccumulatorImplSnafu, CreateAccumulatorSnafu, InvalidInputStateSnafu, Result,
+};
+use common_query::logical_plan::accumulator::AggrFuncTypeStore;
 use common_query::logical_plan::{Accumulator, AggregateFunctionCreator};
 use common_query::prelude::*;
 use datatypes::prelude::*;
--- a/src/common/function/src/scalars/aggregate/diff.rs
+++ b/src/common/function/src/scalars/aggregate/diff.rs
@@ -17,8 +17,10 @@ use std::sync::Arc;

 use common_macro::{as_aggr_func_creator, AggrFuncTypeStore};
 use common_query::error::{
-    CreateAccumulatorSnafu, DowncastVectorSnafu, FromScalarValueSnafu, Result,
+    CreateAccumulatorSnafu, DowncastVectorSnafu, FromScalarValueSnafu, InvalidInputStateSnafu,
+    Result,
 };
+use common_query::logical_plan::accumulator::AggrFuncTypeStore;
 use common_query::logical_plan::{Accumulator, AggregateFunctionCreator};
 use common_query::prelude::*;
 use datatypes::prelude::*;
--- a/src/common/function/src/scalars/aggregate/mean.rs
+++ b/src/common/function/src/scalars/aggregate/mean.rs
@@ -17,8 +17,10 @@ use std::sync::Arc;

 use common_macro::{as_aggr_func_creator, AggrFuncTypeStore};
 use common_query::error::{
-    BadAccumulatorImplSnafu, CreateAccumulatorSnafu, DowncastVectorSnafu, Result,
+    BadAccumulatorImplSnafu, CreateAccumulatorSnafu, DowncastVectorSnafu, InvalidInputStateSnafu,
+    Result,
 };
+use common_query::logical_plan::accumulator::AggrFuncTypeStore;
 use common_query::logical_plan::{Accumulator, AggregateFunctionCreator};
 use common_query::prelude::*;
 use datatypes::prelude::*;
--- a/src/common/function/src/scalars/aggregate/percentile.rs
+++ b/src/common/function/src/scalars/aggregate/percentile.rs
@@ -1,436 +0,0 @@
-// Copyright 2023 Greptime Team
-//
-// Licensed under the Apache License, Version 2.0 (the "License");
-// you may not use this file except in compliance with the License.
-// You may obtain a copy of the License at
-//
-//     http://www.apache.org/licenses/LICENSE-2.0
-//
-// Unless required by applicable law or agreed to in writing, software
-// distributed under the License is distributed on an "AS IS" BASIS,
-// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
-// See the License for the specific language governing permissions and
-// limitations under the License.
-
-use std::cmp::Reverse;
-use std::collections::BinaryHeap;
-use std::sync::Arc;
-
-use common_macro::{as_aggr_func_creator, AggrFuncTypeStore};
-use common_query::error::{
-    self, BadAccumulatorImplSnafu, CreateAccumulatorSnafu, DowncastVectorSnafu,
-    FromScalarValueSnafu, InvalidInputColSnafu, Result,
-};
-use common_query::logical_plan::{Accumulator, AggregateFunctionCreator};
-use common_query::prelude::*;
-use datatypes::prelude::*;
-use datatypes::types::OrdPrimitive;
-use datatypes::value::{ListValue, OrderedFloat};
-use datatypes::vectors::{ConstantVector, Float64Vector, Helper, ListVector};
-use datatypes::with_match_primitive_type_id;
-use num::NumCast;
-use snafu::{ensure, OptionExt, ResultExt};
-
-// https://numpy.org/doc/stable/reference/generated/numpy.percentile.html?highlight=percentile#numpy.percentile
-// if the p is 50,then the Percentile become median
-// we use two heap great and not_greater
-// the not_greater push the value that smaller than P-value
-// the greater push the value that bigger than P-value
-// just like the percentile in numpy:
-// Given a vector V of length N, the q-th percentile of V is the value q/100 of the way from the minimum to the maximum in a sorted copy of V.
-// The values and distances of the two nearest neighbors as well as the method parameter will determine the percentile
-// if the normalized ranking does not match the location of q exactly.
-// This function is the same as the median if q=50, the same as the minimum if q=0 and the same as the maximum if q=100.
-// This optional method parameter specifies the method to use when the desired quantile lies between two data points i < j.
-// If g is the fractional part of the index surrounded by i and alpha and beta are correction constants modifying i and j.
-//              i+g = (q-alpha)/(n-alpha-beta+1)
-// Below, 'q' is the quantile value, 'n' is the sample size and alpha and beta are constants. The following formula gives an interpolation "i + g" of where the quantile would be in the sorted sample.
-// With 'i' being the floor and 'g' the fractional part of the result.
-// the default method is linear where
-// alpha = 1
-// beta = 1
-#[derive(Debug, Default)]
-pub struct Percentile<T>
-where
-    T: WrapperType,
-{
-    greater: BinaryHeap<Reverse<OrdPrimitive<T>>>,
-    not_greater: BinaryHeap<OrdPrimitive<T>>,
-    n: u64,
-    p: Option<f64>,
-}
-
-impl<T> Percentile<T>
-where
-    T: WrapperType,
-{
-    fn push(&mut self, value: T) {
-        let value = OrdPrimitive::<T>(value);
-
-        self.n += 1;
-        if self.not_greater.is_empty() {
-            self.not_greater.push(value);
-            return;
-        }
-        // to keep the not_greater length == floor+1
-        // so to ensure the peek of the not_greater is array[floor]
-        // and the peek of the greater is array[floor+1]
-        let p = self.p.unwrap_or(0.0_f64);
-        let floor = (((self.n - 1) as f64) * p / (100_f64)).floor();
-        if value <= *self.not_greater.peek().unwrap() {
-            self.not_greater.push(value);
-            if self.not_greater.len() > (floor + 1.0) as usize {
-                self.greater.push(Reverse(self.not_greater.pop().unwrap()));
-            }
-        } else {
-            self.greater.push(Reverse(value));
-            if self.not_greater.len() < (floor + 1.0) as usize {
-                self.not_greater.push(self.greater.pop().unwrap().0);
-            }
-        }
-    }
-}
-
-impl<T> Accumulator for Percentile<T>
-where
-    T: WrapperType,
-{
-    fn state(&self) -> Result<Vec<Value>> {
-        let nums = self
-            .greater
-            .iter()
-            .map(|x| &x.0)
-            .chain(self.not_greater.iter())
-            .map(|&n| n.into())
-            .collect::<Vec<Value>>();
-        Ok(vec![
-            Value::List(ListValue::new(nums, T::LogicalType::build_data_type())),
-            self.p.into(),
-        ])
-    }
-
-    fn update_batch(&mut self, values: &[VectorRef]) -> Result<()> {
-        if values.is_empty() {
-            return Ok(());
-        }
-        ensure!(values.len() == 2, InvalidInputStateSnafu);
-        ensure!(values[0].len() == values[1].len(), InvalidInputStateSnafu);
-
-        if values[0].len() == 0 {
-            return Ok(());
-        }
-
-        // This is a unary accumulator, so only one column is provided.
-        let column = &values[0];
-        let mut len = 1;
-        let column: &<T as Scalar>::VectorType = if column.is_const() {
-            len = column.len();
-            let column: &ConstantVector = unsafe { Helper::static_cast(column) };
-            unsafe { Helper::static_cast(column.inner()) }
-        } else {
-            unsafe { Helper::static_cast(column) }
-        };
-
-        let x = &values[1];
-        let x = Helper::check_get_scalar::<f64>(x).context(error::InvalidInputTypeSnafu {
-            err_msg: "expecting \"POLYVAL\" function's second argument to be float64",
-        })?;
-        // `get(0)` is safe because we have checked `values[1].len() == values[0].len() != 0`
-        let first = x.get(0);
-        ensure!(!first.is_null(), InvalidInputColSnafu);
-
-        for i in 1..x.len() {
-            ensure!(first == x.get(i), InvalidInputColSnafu);
-        }
-
-        let first = match first {
-            Value::Float64(OrderedFloat(v)) => v,
-            // unreachable because we have checked `first` is not null and is i64 above
-            _ => unreachable!(),
-        };
-        if let Some(p) = self.p {
-            ensure!(p == first, InvalidInputColSnafu);
-        } else {
-            self.p = Some(first);
-        };
-
-        (0..len).for_each(|_| {
-            for v in column.iter_data().flatten() {
-                self.push(v);
-            }
-        });
-        Ok(())
-    }
-
-    fn merge_batch(&mut self, states: &[VectorRef]) -> Result<()> {
-        if states.is_empty() {
-            return Ok(());
-        }
-
-        ensure!(
-            states.len() == 2,
-            BadAccumulatorImplSnafu {
-                err_msg: "expect 2 states in `merge_batch`"
-            }
-        );
-
-        let p = &states[1];
-        let p = p
-            .as_any()
-            .downcast_ref::<Float64Vector>()
-            .with_context(|| DowncastVectorSnafu {
-                err_msg: format!(
-                    "expect float64vector, got vector type {}",
-                    p.vector_type_name()
-                ),
-            })?;
-        let p = p.get(0);
-        if p.is_null() {
-            return Ok(());
-        }
-        let p = match p {
-            Value::Float64(OrderedFloat(p)) => p,
-            _ => unreachable!(),
-        };
-        self.p = Some(p);
-
-        let values = &states[0];
-        let values = values
-            .as_any()
-            .downcast_ref::<ListVector>()
-            .with_context(|| DowncastVectorSnafu {
-                err_msg: format!(
-                    "expect ListVector, got vector type {}",
-                    values.vector_type_name()
-                ),
-            })?;
-        for value in values.values_iter() {
-            if let Some(value) = value.context(FromScalarValueSnafu)? {
-                let column: &<T as Scalar>::VectorType = unsafe { Helper::static_cast(&value) };
-                for v in column.iter_data().flatten() {
-                    self.push(v);
-                }
-            }
-        }
-        Ok(())
-    }
-
-    fn evaluate(&self) -> Result<Value> {
-        if self.not_greater.is_empty() {
-            assert!(
-                self.greater.is_empty(),
-                "not expected in two-heap percentile algorithm, there must be a bug when implementing it"
-            );
-        }
-        let not_greater = self.not_greater.peek();
-        if not_greater.is_none() {
-            return Ok(Value::Null);
-        }
-        let not_greater = (*self.not_greater.peek().unwrap()).as_primitive();
-        let percentile = if self.greater.is_empty() {
-            NumCast::from(not_greater).unwrap()
-        } else {
-            let greater = self.greater.peek().unwrap();
-            let p = if let Some(p) = self.p {
-                p
-            } else {
-                return Ok(Value::Null);
-            };
-            let fract = (((self.n - 1) as f64) * p / 100_f64).fract();
-            let not_greater_v: f64 = NumCast::from(not_greater).unwrap();
-            let greater_v: f64 = NumCast::from(greater.0.as_primitive()).unwrap();
-            not_greater_v * (1.0 - fract) + greater_v * fract
-        };
-        Ok(Value::from(percentile))
-    }
-}
-
-#[as_aggr_func_creator]
-#[derive(Debug, Default, AggrFuncTypeStore)]
-pub struct PercentileAccumulatorCreator {}
-
-impl AggregateFunctionCreator for PercentileAccumulatorCreator {
-    fn creator(&self) -> AccumulatorCreatorFunction {
-        let creator: AccumulatorCreatorFunction = Arc::new(move |types: &[ConcreteDataType]| {
-            let input_type = &types[0];
-            with_match_primitive_type_id!(
-                input_type.logical_type_id(),
-                |$S| {
-                    Ok(Box::new(Percentile::<<$S as LogicalPrimitiveType>::Wrapper>::default()))
-                },
-                {
-                    let err_msg = format!(
-                        "\"PERCENTILE\" aggregate function not support data type {:?}",
-                        input_type.logical_type_id(),
-                    );
-                    CreateAccumulatorSnafu { err_msg }.fail()?
-                }
-            )
-        });
-        creator
-    }
-
-    fn output_type(&self) -> Result<ConcreteDataType> {
-        let input_types = self.input_types()?;
-        ensure!(input_types.len() == 2, InvalidInputStateSnafu);
-        // unwrap is safe because we have checked input_types len must equals 1
-        Ok(ConcreteDataType::float64_datatype())
-    }
-
-    fn state_types(&self) -> Result<Vec<ConcreteDataType>> {
-        let input_types = self.input_types()?;
-        ensure!(input_types.len() == 2, InvalidInputStateSnafu);
-        Ok(vec![
-            ConcreteDataType::list_datatype(input_types.into_iter().next().unwrap()),
-            ConcreteDataType::float64_datatype(),
-        ])
-    }
-}
-
-#[cfg(test)]
-mod test {
-    use datatypes::vectors::{Float64Vector, Int32Vector};
-
-    use super::*;
-    #[test]
-    fn test_update_batch() {
-        // test update empty batch, expect not updating anything
-        let mut percentile = Percentile::<i32>::default();
-        percentile.update_batch(&[]).unwrap();
-        assert!(percentile.not_greater.is_empty());
-        assert!(percentile.greater.is_empty());
-        assert_eq!(Value::Null, percentile.evaluate().unwrap());
-
-        // test update one not-null value
-        let mut percentile = Percentile::<i32>::default();
-        let v: Vec<VectorRef> = vec![
-            Arc::new(Int32Vector::from(vec![Some(42)])),
-            Arc::new(Float64Vector::from(vec![Some(100.0_f64)])),
-        ];
-        percentile.update_batch(&v).unwrap();
-        assert_eq!(Value::from(42.0_f64), percentile.evaluate().unwrap());
-
-        // test update one null value
-        let mut percentile = Percentile::<i32>::default();
-        let v: Vec<VectorRef> = vec![
-            Arc::new(Int32Vector::from(vec![Option::<i32>::None])),
-            Arc::new(Float64Vector::from(vec![Some(100.0_f64)])),
-        ];
-        percentile.update_batch(&v).unwrap();
-        assert_eq!(Value::Null, percentile.evaluate().unwrap());
-
-        // test update no null-value batch
-        let mut percentile = Percentile::<i32>::default();
-        let v: Vec<VectorRef> = vec![
-            Arc::new(Int32Vector::from(vec![Some(-1i32), Some(1), Some(2)])),
-            Arc::new(Float64Vector::from(vec![
-                Some(100.0_f64),
-                Some(100.0_f64),
-                Some(100.0_f64),
-            ])),
-        ];
-        percentile.update_batch(&v).unwrap();
-        assert_eq!(Value::from(2_f64), percentile.evaluate().unwrap());
-
-        // test update null-value batch
-        let mut percentile = Percentile::<i32>::default();
-        let v: Vec<VectorRef> = vec![
-            Arc::new(Int32Vector::from(vec![Some(-2i32), None, Some(3), Some(4)])),
-            Arc::new(Float64Vector::from(vec![
-                Some(100.0_f64),
-                Some(100.0_f64),
-                Some(100.0_f64),
-                Some(100.0_f64),
-            ])),
-        ];
-        percentile.update_batch(&v).unwrap();
-        assert_eq!(Value::from(4_f64), percentile.evaluate().unwrap());
-
-        // test update with constant vector
-        let mut percentile = Percentile::<i32>::default();
-        let v: Vec<VectorRef> = vec![
-            Arc::new(ConstantVector::new(
-                Arc::new(Int32Vector::from_vec(vec![4])),
-                2,
-            )),
-            Arc::new(Float64Vector::from(vec![Some(100.0_f64), Some(100.0_f64)])),
-        ];
-        percentile.update_batch(&v).unwrap();
-        assert_eq!(Value::from(4_f64), percentile.evaluate().unwrap());
-
-        // test left border
-        let mut percentile = Percentile::<i32>::default();
-        let v: Vec<VectorRef> = vec![
-            Arc::new(Int32Vector::from(vec![Some(-1i32), Some(1), Some(2)])),
-            Arc::new(Float64Vector::from(vec![
-                Some(0.0_f64),
-                Some(0.0_f64),
-                Some(0.0_f64),
-            ])),
-        ];
-        percentile.update_batch(&v).unwrap();
-        assert_eq!(Value::from(-1.0_f64), percentile.evaluate().unwrap());
-
-        // test medium
-        let mut percentile = Percentile::<i32>::default();
-        let v: Vec<VectorRef> = vec![
-            Arc::new(Int32Vector::from(vec![Some(-1i32), Some(1), Some(2)])),
-            Arc::new(Float64Vector::from(vec![
-                Some(50.0_f64),
-                Some(50.0_f64),
-                Some(50.0_f64),
-            ])),
-        ];
-        percentile.update_batch(&v).unwrap();
-        assert_eq!(Value::from(1.0_f64), percentile.evaluate().unwrap());
-
-        // test right border
-        let mut percentile = Percentile::<i32>::default();
-        let v: Vec<VectorRef> = vec![
-            Arc::new(Int32Vector::from(vec![Some(-1i32), Some(1), Some(2)])),
-            Arc::new(Float64Vector::from(vec![
-                Some(100.0_f64),
-                Some(100.0_f64),
-                Some(100.0_f64),
-            ])),
-        ];
-        percentile.update_batch(&v).unwrap();
-        assert_eq!(Value::from(2.0_f64), percentile.evaluate().unwrap());
-
-        // the following is the result of numpy.percentile
-        // numpy.percentile
-        // a = np.array([[10,7,4]])
-        // np.percentile(a,40)
-        // >> 6.400000000000
-        let mut percentile = Percentile::<i32>::default();
-        let v: Vec<VectorRef> = vec![
-            Arc::new(Int32Vector::from(vec![Some(10i32), Some(7), Some(4)])),
-            Arc::new(Float64Vector::from(vec![
-                Some(40.0_f64),
-                Some(40.0_f64),
-                Some(40.0_f64),
-            ])),
-        ];
-        percentile.update_batch(&v).unwrap();
-        assert_eq!(Value::from(6.400000000_f64), percentile.evaluate().unwrap());
-
-        // the following is the result of numpy.percentile
-        // a = np.array([[10,7,4]])
-        // np.percentile(a,95)
-        // >> 9.7000000000000011
-        let mut percentile = Percentile::<i32>::default();
-        let v: Vec<VectorRef> = vec![
-            Arc::new(Int32Vector::from(vec![Some(10i32), Some(7), Some(4)])),
-            Arc::new(Float64Vector::from(vec![
-                Some(95.0_f64),
-                Some(95.0_f64),
-                Some(95.0_f64),
-            ])),
-        ];
-        percentile.update_batch(&v).unwrap();
-        assert_eq!(
-            Value::from(9.700_000_000_000_001_f64),
-            percentile.evaluate().unwrap()
-        );
-    }
-}
--- a/src/common/function/src/scalars/aggregate/polyval.rs
+++ b/src/common/function/src/scalars/aggregate/polyval.rs
@@ -18,8 +18,9 @@ use std::sync::Arc;
 use common_macro::{as_aggr_func_creator, AggrFuncTypeStore};
 use common_query::error::{
    self, BadAccumulatorImplSnafu, CreateAccumulatorSnafu, DowncastVectorSnafu,
-    FromScalarValueSnafu, InvalidInputColSnafu, Result,
+    FromScalarValueSnafu, InvalidInputColSnafu, InvalidInputStateSnafu, Result,
 };
+use common_query::logical_plan::accumulator::AggrFuncTypeStore;
 use common_query::logical_plan::{Accumulator, AggregateFunctionCreator};
 use common_query::prelude::*;
 use datatypes::prelude::*;
--- a/src/common/function/src/scalars/aggregate/scipy_stats_norm_cdf.rs
+++ b/src/common/function/src/scalars/aggregate/scipy_stats_norm_cdf.rs
@@ -17,8 +17,10 @@ use std::sync::Arc;
 use common_macro::{as_aggr_func_creator, AggrFuncTypeStore};
 use common_query::error::{
    self, BadAccumulatorImplSnafu, CreateAccumulatorSnafu, DowncastVectorSnafu,
-    FromScalarValueSnafu, GenerateFunctionSnafu, InvalidInputColSnafu, Result,
+    FromScalarValueSnafu, GenerateFunctionSnafu, InvalidInputColSnafu, InvalidInputStateSnafu,
+    Result,
 };
+use common_query::logical_plan::accumulator::AggrFuncTypeStore;
 use common_query::logical_plan::{Accumulator, AggregateFunctionCreator};
 use common_query::prelude::*;
 use datatypes::prelude::*;
--- a/src/common/function/src/scalars/aggregate/scipy_stats_norm_pdf.rs
+++ b/src/common/function/src/scalars/aggregate/scipy_stats_norm_pdf.rs
@@ -17,8 +17,10 @@ use std::sync::Arc;
 use common_macro::{as_aggr_func_creator, AggrFuncTypeStore};
 use common_query::error::{
    self, BadAccumulatorImplSnafu, CreateAccumulatorSnafu, DowncastVectorSnafu,
-    FromScalarValueSnafu, GenerateFunctionSnafu, InvalidInputColSnafu, Result,
+    FromScalarValueSnafu, GenerateFunctionSnafu, InvalidInputColSnafu, InvalidInputStateSnafu,
+    Result,
 };
+use common_query::logical_plan::accumulator::AggrFuncTypeStore;
 use common_query::logical_plan::{Accumulator, AggregateFunctionCreator};
 use common_query::prelude::*;
 use datatypes::prelude::*;
--- a/src/common/function/src/scalars/date/date_add.rs
+++ b/src/common/function/src/scalars/date/date_add.rs
@@ -14,18 +14,19 @@

 use std::fmt;

-use common_query::error::{InvalidFuncArgsSnafu, Result, UnsupportedInputDataTypeSnafu};
+use common_query::error::{ArrowComputeSnafu, IntoVectorSnafu, InvalidFuncArgsSnafu, Result};
 use common_query::prelude::Signature;
-use datatypes::data_type::DataType;
+use datatypes::arrow::compute::kernels::numeric;
 use datatypes::prelude::ConcreteDataType;
-use datatypes::value::ValueRef;
-use datatypes::vectors::VectorRef;
-use snafu::ensure;
+use datatypes::vectors::{Helper, VectorRef};
+use snafu::{ensure, ResultExt};

 use crate::function::{Function, FunctionContext};
 use crate::helper;

-/// A function adds an interval value to Timestamp, Date or DateTime, and return the result.
+/// A function adds an interval value to Timestamp, Date, and return the result.
+/// The implementation of datetime type is based on Date64 which is incorrect so this function
+/// doesn't support the datetime type.
 #[derive(Clone, Debug, Default)]
 pub struct DateAddFunction;

@@ -44,7 +45,6 @@ impl Function for DateAddFunction {
        helper::one_of_sigs2(
            vec![
                ConcreteDataType::date_datatype(),
-                ConcreteDataType::datetime_datatype(),
                ConcreteDataType::timestamp_second_datatype(),
                ConcreteDataType::timestamp_millisecond_datatype(),
                ConcreteDataType::timestamp_microsecond_datatype(),
@@ -69,64 +69,14 @@ impl Function for DateAddFunction {
            }
        );

-        let left = &columns[0];
-        let right = &columns[1];
+        let left = columns[0].to_arrow_array();
+        let right = columns[1].to_arrow_array();

-        let size = left.len();
-        let left_datatype = columns[0].data_type();
-        match left_datatype {
-            ConcreteDataType::Timestamp(_) => {
-                let mut result = left_datatype.create_mutable_vector(size);
-                for i in 0..size {
-                    let ts = left.get(i).as_timestamp();
-                    let interval = right.get(i).as_interval();
-
-                    let new_ts = match (ts, interval) {
-                        (Some(ts), Some(interval)) => ts.add_interval(interval),
-                        _ => ts,
-                    };
-
-                    result.push_value_ref(ValueRef::from(new_ts));
-                }
-
-                Ok(result.to_vector())
-            }
-            ConcreteDataType::Date(_) => {
-                let mut result = left_datatype.create_mutable_vector(size);
-                for i in 0..size {
-                    let date = left.get(i).as_date();
-                    let interval = right.get(i).as_interval();
-                    let new_date = match (date, interval) {
-                        (Some(date), Some(interval)) => date.add_interval(interval),
-                        _ => date,
-                    };
-
-                    result.push_value_ref(ValueRef::from(new_date));
-                }
-
-                Ok(result.to_vector())
-            }
-            ConcreteDataType::DateTime(_) => {
-                let mut result = left_datatype.create_mutable_vector(size);
-                for i in 0..size {
-                    let datetime = left.get(i).as_datetime();
-                    let interval = right.get(i).as_interval();
-                    let new_datetime = match (datetime, interval) {
-                        (Some(datetime), Some(interval)) => datetime.add_interval(interval),
-                        _ => datetime,
-                    };
-
-                    result.push_value_ref(ValueRef::from(new_datetime));
-                }
-
-                Ok(result.to_vector())
-            }
-            _ => UnsupportedInputDataTypeSnafu {
-                function: NAME,
-                datatypes: columns.iter().map(|c| c.data_type()).collect::<Vec<_>>(),
-            }
-            .fail(),
-        }
+        let result = numeric::add(&left, &right).context(ArrowComputeSnafu)?;
+        let arrow_type = result.data_type().clone();
+        Helper::try_into_vector(result).context(IntoVectorSnafu {
+            data_type: arrow_type,
+        })
    }
 }

@@ -144,8 +94,7 @@ mod tests {
    use datatypes::prelude::ConcreteDataType;
    use datatypes::value::Value;
    use datatypes::vectors::{
-        DateTimeVector, DateVector, IntervalDayTimeVector, IntervalYearMonthVector,
-        TimestampSecondVector,
+        DateVector, IntervalDayTimeVector, IntervalYearMonthVector, TimestampSecondVector,
    };

    use super::{DateAddFunction, *};
@@ -168,16 +117,15 @@ mod tests {
            ConcreteDataType::date_datatype(),
            f.return_type(&[ConcreteDataType::date_datatype()]).unwrap()
        );
-        assert_eq!(
-            ConcreteDataType::datetime_datatype(),
-            f.return_type(&[ConcreteDataType::datetime_datatype()])
-                .unwrap()
-        );
-        assert!(matches!(f.signature(),
+        assert!(
+            matches!(f.signature(),
                         Signature {
                             type_signature: TypeSignature::OneOf(sigs),
                             volatility: Volatility::Immutable
-                         } if  sigs.len() == 18));
+                         } if  sigs.len() == 15),
+            "{:?}",
+            f.signature()
+        );
    }

    #[test]
@@ -243,36 +191,4 @@ mod tests {
            }
        }
    }
-
-    #[test]
-    fn test_datetime_date_add() {
-        let f = DateAddFunction;
-
-        let dates = vec![Some(123), None, Some(42), None];
-        // Intervals in months
-        let intervals = vec![1, 2, 3, 1];
-        let results = [Some(2678400123), None, Some(7776000042), None];
-
-        let date_vector = DateTimeVector::from(dates.clone());
-        let interval_vector = IntervalYearMonthVector::from_vec(intervals);
-        let args: Vec<VectorRef> = vec![Arc::new(date_vector), Arc::new(interval_vector)];
-        let vector = f.eval(FunctionContext::default(), &args).unwrap();
-
-        assert_eq!(4, vector.len());
-        for (i, _t) in dates.iter().enumerate() {
-            let v = vector.get(i);
-            let result = results.get(i).unwrap();
-
-            if result.is_none() {
-                assert_eq!(Value::Null, v);
-                continue;
-            }
-            match v {
-                Value::DateTime(date) => {
-                    assert_eq!(date.val(), result.unwrap());
-                }
-                _ => unreachable!(),
-            }
-        }
-    }
 }
--- a/src/common/function/src/scalars/date/date_sub.rs
+++ b/src/common/function/src/scalars/date/date_sub.rs
@@ -14,18 +14,19 @@

 use std::fmt;

-use common_query::error::{InvalidFuncArgsSnafu, Result, UnsupportedInputDataTypeSnafu};
+use common_query::error::{ArrowComputeSnafu, IntoVectorSnafu, InvalidFuncArgsSnafu, Result};
 use common_query::prelude::Signature;
-use datatypes::data_type::DataType;
+use datatypes::arrow::compute::kernels::numeric;
 use datatypes::prelude::ConcreteDataType;
-use datatypes::value::ValueRef;
-use datatypes::vectors::VectorRef;
-use snafu::ensure;
+use datatypes::vectors::{Helper, VectorRef};
+use snafu::{ensure, ResultExt};

 use crate::function::{Function, FunctionContext};
 use crate::helper;

-/// A function subtracts an interval value to Timestamp, Date or DateTime, and return the result.
+/// A function subtracts an interval value to Timestamp, Date, and return the result.
+/// The implementation of datetime type is based on Date64 which is incorrect so this function
+/// doesn't support the datetime type.
 #[derive(Clone, Debug, Default)]
 pub struct DateSubFunction;

@@ -44,7 +45,6 @@ impl Function for DateSubFunction {
        helper::one_of_sigs2(
            vec![
                ConcreteDataType::date_datatype(),
-                ConcreteDataType::datetime_datatype(),
                ConcreteDataType::timestamp_second_datatype(),
                ConcreteDataType::timestamp_millisecond_datatype(),
                ConcreteDataType::timestamp_microsecond_datatype(),
@@ -69,65 +69,14 @@ impl Function for DateSubFunction {
            }
        );

-        let left = &columns[0];
-        let right = &columns[1];
+        let left = columns[0].to_arrow_array();
+        let right = columns[1].to_arrow_array();

-        let size = left.len();
-        let left_datatype = columns[0].data_type();
-
-        match left_datatype {
-            ConcreteDataType::Timestamp(_) => {
-                let mut result = left_datatype.create_mutable_vector(size);
-                for i in 0..size {
-                    let ts = left.get(i).as_timestamp();
-                    let interval = right.get(i).as_interval();
-
-                    let new_ts = match (ts, interval) {
-                        (Some(ts), Some(interval)) => ts.sub_interval(interval),
-                        _ => ts,
-                    };
-
-                    result.push_value_ref(ValueRef::from(new_ts));
-                }
-
-                Ok(result.to_vector())
-            }
-            ConcreteDataType::Date(_) => {
-                let mut result = left_datatype.create_mutable_vector(size);
-                for i in 0..size {
-                    let date = left.get(i).as_date();
-                    let interval = right.get(i).as_interval();
-                    let new_date = match (date, interval) {
-                        (Some(date), Some(interval)) => date.sub_interval(interval),
-                        _ => date,
-                    };
-
-                    result.push_value_ref(ValueRef::from(new_date));
-                }
-
-                Ok(result.to_vector())
-            }
-            ConcreteDataType::DateTime(_) => {
-                let mut result = left_datatype.create_mutable_vector(size);
-                for i in 0..size {
-                    let datetime = left.get(i).as_datetime();
-                    let interval = right.get(i).as_interval();
-                    let new_datetime = match (datetime, interval) {
-                        (Some(datetime), Some(interval)) => datetime.sub_interval(interval),
-                        _ => datetime,
-                    };
-
-                    result.push_value_ref(ValueRef::from(new_datetime));
-                }
-
-                Ok(result.to_vector())
-            }
-            _ => UnsupportedInputDataTypeSnafu {
-                function: NAME,
-                datatypes: columns.iter().map(|c| c.data_type()).collect::<Vec<_>>(),
-            }
-            .fail(),
-        }
+        let result = numeric::sub(&left, &right).context(ArrowComputeSnafu)?;
+        let arrow_type = result.data_type().clone();
+        Helper::try_into_vector(result).context(IntoVectorSnafu {
+            data_type: arrow_type,
+        })
    }
 }

@@ -145,8 +94,7 @@ mod tests {
    use datatypes::prelude::ConcreteDataType;
    use datatypes::value::Value;
    use datatypes::vectors::{
-        DateTimeVector, DateVector, IntervalDayTimeVector, IntervalYearMonthVector,
-        TimestampSecondVector,
+        DateVector, IntervalDayTimeVector, IntervalYearMonthVector, TimestampSecondVector,
    };

    use super::{DateSubFunction, *};
@@ -174,11 +122,15 @@ mod tests {
            f.return_type(&[ConcreteDataType::datetime_datatype()])
                .unwrap()
        );
-        assert!(matches!(f.signature(),
+        assert!(
+            matches!(f.signature(),
                         Signature {
                             type_signature: TypeSignature::OneOf(sigs),
                             volatility: Volatility::Immutable
-                         } if  sigs.len() == 18));
+                         } if  sigs.len() == 15),
+            "{:?}",
+            f.signature()
+        );
    }

    #[test]
@@ -250,42 +202,4 @@ mod tests {
            }
        }
    }
-
-    #[test]
-    fn test_datetime_date_sub() {
-        let f = DateSubFunction;
-        let millis_per_month = 3600 * 24 * 30 * 1000;
-
-        let dates = vec![
-            Some(123 * millis_per_month),
-            None,
-            Some(42 * millis_per_month),
-            None,
-        ];
-        // Intervals in months
-        let intervals = vec![1, 2, 3, 1];
-        let results = [Some(316137600000), None, Some(100915200000), None];
-
-        let date_vector = DateTimeVector::from(dates.clone());
-        let interval_vector = IntervalYearMonthVector::from_vec(intervals);
-        let args: Vec<VectorRef> = vec![Arc::new(date_vector), Arc::new(interval_vector)];
-        let vector = f.eval(FunctionContext::default(), &args).unwrap();
-
-        assert_eq!(4, vector.len());
-        for (i, _t) in dates.iter().enumerate() {
-            let v = vector.get(i);
-            let result = results.get(i).unwrap();
-
-            if result.is_none() {
-                assert_eq!(Value::Null, v);
-                continue;
-            }
-            match v {
-                Value::DateTime(date) => {
-                    assert_eq!(date.val(), result.unwrap());
-                }
-                _ => unreachable!(),
-            }
-        }
-    }
 }
--- a/src/common/function/src/scalars/geo.rs
+++ b/src/common/function/src/scalars/geo.rs
@@ -13,11 +13,11 @@
 // limitations under the License.

 use std::sync::Arc;
+pub(crate) mod encoding;
 mod geohash;
 mod h3;
-
-use geohash::GeohashFunction;
-use h3::H3Function;
+mod helpers;
+mod s2;

 use crate::function_registry::FunctionRegistry;

@@ -25,7 +25,40 @@ pub(crate) struct GeoFunctions;

 impl GeoFunctions {
    pub fn register(registry: &FunctionRegistry) {
-        registry.register(Arc::new(GeohashFunction));
-        registry.register(Arc::new(H3Function));
+        // geohash
+        registry.register(Arc::new(geohash::GeohashFunction));
+        registry.register(Arc::new(geohash::GeohashNeighboursFunction));
+
+        // h3 index
+        registry.register(Arc::new(h3::H3LatLngToCell));
+        registry.register(Arc::new(h3::H3LatLngToCellString));
+
+        // h3 index inspection
+        registry.register(Arc::new(h3::H3CellBase));
+        registry.register(Arc::new(h3::H3CellIsPentagon));
+        registry.register(Arc::new(h3::H3StringToCell));
+        registry.register(Arc::new(h3::H3CellToString));
+        registry.register(Arc::new(h3::H3CellCenterLatLng));
+        registry.register(Arc::new(h3::H3CellResolution));
+
+        // h3 hierarchical grid
+        registry.register(Arc::new(h3::H3CellCenterChild));
+        registry.register(Arc::new(h3::H3CellParent));
+        registry.register(Arc::new(h3::H3CellToChildren));
+        registry.register(Arc::new(h3::H3CellToChildrenSize));
+        registry.register(Arc::new(h3::H3CellToChildPos));
+        registry.register(Arc::new(h3::H3ChildPosToCell));
+
+        // h3 grid traversal
+        registry.register(Arc::new(h3::H3GridDisk));
+        registry.register(Arc::new(h3::H3GridDiskDistances));
+        registry.register(Arc::new(h3::H3GridDistance));
+        registry.register(Arc::new(h3::H3GridPathCells));
+
+        // s2
+        registry.register(Arc::new(s2::S2LatLngToCell));
+        registry.register(Arc::new(s2::S2CellLevel));
+        registry.register(Arc::new(s2::S2CellToToken));
+        registry.register(Arc::new(s2::S2CellParent));
    }
 }
--- a/src/common/function/src/scalars/geo/encoding.rs
+++ b/src/common/function/src/scalars/geo/encoding.rs
@@ -0,0 +1,223 @@
+// Copyright 2023 Greptime Team
+//
+// Licensed under the Apache License, Version 2.0 (the "License");
+// you may not use this file except in compliance with the License.
+// You may obtain a copy of the License at
+//
+//     http://www.apache.org/licenses/LICENSE-2.0
+//
+// Unless required by applicable law or agreed to in writing, software
+// distributed under the License is distributed on an "AS IS" BASIS,
+// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+// See the License for the specific language governing permissions and
+// limitations under the License.
+
+use std::sync::Arc;
+
+use common_error::ext::{BoxedError, PlainError};
+use common_error::status_code::StatusCode;
+use common_macro::{as_aggr_func_creator, AggrFuncTypeStore};
+use common_query::error::{self, InvalidInputStateSnafu, Result};
+use common_query::logical_plan::accumulator::AggrFuncTypeStore;
+use common_query::logical_plan::{Accumulator, AggregateFunctionCreator};
+use common_query::prelude::AccumulatorCreatorFunction;
+use common_time::Timestamp;
+use datatypes::prelude::ConcreteDataType;
+use datatypes::value::{ListValue, Value};
+use datatypes::vectors::VectorRef;
+use snafu::{ensure, ResultExt};
+
+use super::helpers::{ensure_columns_len, ensure_columns_n};
+
+/// Accumulator of lat, lng, timestamp tuples
+#[derive(Debug)]
+pub struct JsonPathAccumulator {
+    timestamp_type: ConcreteDataType,
+    lat: Vec<Option<f64>>,
+    lng: Vec<Option<f64>>,
+    timestamp: Vec<Option<Timestamp>>,
+}
+
+impl JsonPathAccumulator {
+    fn new(timestamp_type: ConcreteDataType) -> Self {
+        Self {
+            lat: Vec::default(),
+            lng: Vec::default(),
+            timestamp: Vec::default(),
+            timestamp_type,
+        }
+    }
+}
+
+impl Accumulator for JsonPathAccumulator {
+    fn state(&self) -> Result<Vec<Value>> {
+        Ok(vec![
+            Value::List(ListValue::new(
+                self.lat.iter().map(|i| Value::from(*i)).collect(),
+                ConcreteDataType::float64_datatype(),
+            )),
+            Value::List(ListValue::new(
+                self.lng.iter().map(|i| Value::from(*i)).collect(),
+                ConcreteDataType::float64_datatype(),
+            )),
+            Value::List(ListValue::new(
+                self.timestamp.iter().map(|i| Value::from(*i)).collect(),
+                self.timestamp_type.clone(),
+            )),
+        ])
+    }
+
+    fn update_batch(&mut self, columns: &[VectorRef]) -> Result<()> {
+        // update batch as in datafusion just provides the accumulator original
+        //  input.
+        //
+        // columns is vec of [`lat`, `lng`, `timestamp`]
+        // where
+        // - `lat` is a vector of `Value::Float64` or similar type. Each item in
+        //  the vector is a row in given dataset.
+        // - so on so forth for `lng` and `timestamp`
+        ensure_columns_n!(columns, 3);
+
+        let lat = &columns[0];
+        let lng = &columns[1];
+        let ts = &columns[2];
+
+        let size = lat.len();
+
+        for idx in 0..size {
+            self.lat.push(lat.get(idx).as_f64_lossy());
+            self.lng.push(lng.get(idx).as_f64_lossy());
+            self.timestamp.push(ts.get(idx).as_timestamp());
+        }
+
+        Ok(())
+    }
+
+    fn merge_batch(&mut self, states: &[VectorRef]) -> Result<()> {
+        // merge batch as in datafusion gives state accumulated from the data
+        //  returned from child accumulators' state() call
+        // In our particular implementation, the data structure is like
+        //
+        // states is vec of [`lat`, `lng`, `timestamp`]
+        // where
+        // - `lat` is a vector of `Value::List`. Each item in the list is all
+        //  coordinates from a child accumulator.
+        // - so on so forth for `lng` and `timestamp`
+
+        ensure_columns_n!(states, 3);
+
+        let lat_lists = &states[0];
+        let lng_lists = &states[1];
+        let ts_lists = &states[2];
+
+        let len = lat_lists.len();
+
+        for idx in 0..len {
+            if let Some(lat_list) = lat_lists
+                .get(idx)
+                .as_list()
+                .map_err(BoxedError::new)
+                .context(error::ExecuteSnafu)?
+            {
+                for v in lat_list.items() {
+                    self.lat.push(v.as_f64_lossy());
+                }
+            }
+
+            if let Some(lng_list) = lng_lists
+                .get(idx)
+                .as_list()
+                .map_err(BoxedError::new)
+                .context(error::ExecuteSnafu)?
+            {
+                for v in lng_list.items() {
+                    self.lng.push(v.as_f64_lossy());
+                }
+            }
+
+            if let Some(ts_list) = ts_lists
+                .get(idx)
+                .as_list()
+                .map_err(BoxedError::new)
+                .context(error::ExecuteSnafu)?
+            {
+                for v in ts_list.items() {
+                    self.timestamp.push(v.as_timestamp());
+                }
+            }
+        }
+
+        Ok(())
+    }
+
+    fn evaluate(&self) -> Result<Value> {
+        let mut work_vec: Vec<(&Option<f64>, &Option<f64>, &Option<Timestamp>)> = self
+            .lat
+            .iter()
+            .zip(self.lng.iter())
+            .zip(self.timestamp.iter())
+            .map(|((a, b), c)| (a, b, c))
+            .collect();
+
+        // sort by timestamp, we treat null timestamp as 0
+        work_vec.sort_unstable_by_key(|tuple| tuple.2.unwrap_or_else(|| Timestamp::new_second(0)));
+
+        let result = serde_json::to_string(
+            &work_vec
+                .into_iter()
+                // note that we transform to lng,lat for geojson compatibility
+                .map(|(lat, lng, _)| vec![lng, lat])
+                .collect::<Vec<Vec<&Option<f64>>>>(),
+        )
+        .map_err(|e| {
+            BoxedError::new(PlainError::new(
+                format!("Serialization failure: {}", e),
+                StatusCode::EngineExecuteQuery,
+            ))
+        })
+        .context(error::ExecuteSnafu)?;
+
+        Ok(Value::String(result.into()))
+    }
+}
+
+/// This function accept rows of lat, lng and timestamp, sort with timestamp and
+/// encoding them into a geojson-like path.
+///
+/// Example:
+///
+/// ```sql
+/// SELECT json_encode_path(lat, lon, timestamp) FROM table [group by ...];
+/// ```
+///
+#[as_aggr_func_creator]
+#[derive(Debug, Default, AggrFuncTypeStore)]
+pub struct JsonPathEncodeFunctionCreator {}
+
+impl AggregateFunctionCreator for JsonPathEncodeFunctionCreator {
+    fn creator(&self) -> AccumulatorCreatorFunction {
+        let creator: AccumulatorCreatorFunction = Arc::new(move |types: &[ConcreteDataType]| {
+            let ts_type = types[2].clone();
+            Ok(Box::new(JsonPathAccumulator::new(ts_type)))
+        });
+
+        creator
+    }
+
+    fn output_type(&self) -> Result<ConcreteDataType> {
+        Ok(ConcreteDataType::string_datatype())
+    }
+
+    fn state_types(&self) -> Result<Vec<ConcreteDataType>> {
+        let input_types = self.input_types()?;
+        ensure!(input_types.len() == 3, InvalidInputStateSnafu);
+
+        let timestamp_type = input_types[2].clone();
+
+        Ok(vec![
+            ConcreteDataType::list_datatype(ConcreteDataType::float64_datatype()),
+            ConcreteDataType::list_datatype(ConcreteDataType::float64_datatype()),
+            ConcreteDataType::list_datatype(timestamp_type),
+        ])
+    }
+}
--- a/src/common/function/src/scalars/geo/geohash.rs
+++ b/src/common/function/src/scalars/geo/geohash.rs
@@ -20,23 +20,69 @@ use common_query::error::{self, InvalidFuncArgsSnafu, Result};
 use common_query::prelude::{Signature, TypeSignature};
 use datafusion::logical_expr::Volatility;
 use datatypes::prelude::ConcreteDataType;
-use datatypes::scalars::ScalarVectorBuilder;
-use datatypes::value::Value;
-use datatypes::vectors::{MutableVector, StringVectorBuilder, VectorRef};
+use datatypes::scalars::{Scalar, ScalarVectorBuilder};
+use datatypes::value::{ListValue, Value};
+use datatypes::vectors::{ListVectorBuilder, MutableVector, StringVectorBuilder, VectorRef};
 use geohash::Coord;
 use snafu::{ensure, ResultExt};

 use crate::function::{Function, FunctionContext};

+macro_rules! ensure_resolution_usize {
+    ($v: ident) => {
+        if !($v > 0 && $v <= 12) {
+            Err(BoxedError::new(PlainError::new(
+                format!("Invalid geohash resolution {}, expect value: [1, 12]", $v),
+                StatusCode::EngineExecuteQuery,
+            )))
+            .context(error::ExecuteSnafu)
+        } else {
+            Ok($v as usize)
+        }
+    };
+}
+
+fn try_into_resolution(v: Value) -> Result<usize> {
+    match v {
+        Value::Int8(v) => {
+            ensure_resolution_usize!(v)
+        }
+        Value::Int16(v) => {
+            ensure_resolution_usize!(v)
+        }
+        Value::Int32(v) => {
+            ensure_resolution_usize!(v)
+        }
+        Value::Int64(v) => {
+            ensure_resolution_usize!(v)
+        }
+        Value::UInt8(v) => {
+            ensure_resolution_usize!(v)
+        }
+        Value::UInt16(v) => {
+            ensure_resolution_usize!(v)
+        }
+        Value::UInt32(v) => {
+            ensure_resolution_usize!(v)
+        }
+        Value::UInt64(v) => {
+            ensure_resolution_usize!(v)
+        }
+        _ => unreachable!(),
+    }
+}
+
 /// Function that return geohash string for a given geospatial coordinate.
 #[derive(Clone, Debug, Default)]
 pub struct GeohashFunction;

-const NAME: &str = "geohash";
+impl GeohashFunction {
+    const NAME: &'static str = "geohash";
+}

 impl Function for GeohashFunction {
    fn name(&self) -> &str {
-        NAME
+        Self::NAME
    }

    fn return_type(&self, _input_types: &[ConcreteDataType]) -> Result<ConcreteDataType> {
@@ -93,17 +139,7 @@ impl Function for GeohashFunction {
        for i in 0..size {
            let lat = lat_vec.get(i).as_f64_lossy();
            let lon = lon_vec.get(i).as_f64_lossy();
-            let r = match resolution_vec.get(i) {
-                Value::Int8(v) => v as usize,
-                Value::Int16(v) => v as usize,
-                Value::Int32(v) => v as usize,
-                Value::Int64(v) => v as usize,
-                Value::UInt8(v) => v as usize,
-                Value::UInt16(v) => v as usize,
-                Value::UInt32(v) => v as usize,
-                Value::UInt64(v) => v as usize,
-                _ => unreachable!(),
-            };
+            let r = try_into_resolution(resolution_vec.get(i))?;

            let result = match (lat, lon) {
                (Some(lat), Some(lon)) => {
@@ -130,6 +166,134 @@ impl Function for GeohashFunction {

 impl fmt::Display for GeohashFunction {
    fn fmt(&self, f: &mut fmt::Formatter) -> fmt::Result {
-        write!(f, "{}", NAME)
+        write!(f, "{}", Self::NAME)
+    }
+}
+
+/// Function that return geohash string for a given geospatial coordinate.
+#[derive(Clone, Debug, Default)]
+pub struct GeohashNeighboursFunction;
+
+impl GeohashNeighboursFunction {
+    const NAME: &'static str = "geohash_neighbours";
+}
+
+impl Function for GeohashNeighboursFunction {
+    fn name(&self) -> &str {
+        GeohashNeighboursFunction::NAME
+    }
+
+    fn return_type(&self, _input_types: &[ConcreteDataType]) -> Result<ConcreteDataType> {
+        Ok(ConcreteDataType::list_datatype(
+            ConcreteDataType::string_datatype(),
+        ))
+    }
+
+    fn signature(&self) -> Signature {
+        let mut signatures = Vec::new();
+        for coord_type in &[
+            ConcreteDataType::float32_datatype(),
+            ConcreteDataType::float64_datatype(),
+        ] {
+            for resolution_type in &[
+                ConcreteDataType::int8_datatype(),
+                ConcreteDataType::int16_datatype(),
+                ConcreteDataType::int32_datatype(),
+                ConcreteDataType::int64_datatype(),
+                ConcreteDataType::uint8_datatype(),
+                ConcreteDataType::uint16_datatype(),
+                ConcreteDataType::uint32_datatype(),
+                ConcreteDataType::uint64_datatype(),
+            ] {
+                signatures.push(TypeSignature::Exact(vec![
+                    // latitude
+                    coord_type.clone(),
+                    // longitude
+                    coord_type.clone(),
+                    // resolution
+                    resolution_type.clone(),
+                ]));
+            }
+        }
+        Signature::one_of(signatures, Volatility::Stable)
+    }
+
+    fn eval(&self, _func_ctx: FunctionContext, columns: &[VectorRef]) -> Result<VectorRef> {
+        ensure!(
+            columns.len() == 3,
+            InvalidFuncArgsSnafu {
+                err_msg: format!(
+                    "The length of the args is not correct, expect 3, provided : {}",
+                    columns.len()
+                ),
+            }
+        );
+
+        let lat_vec = &columns[0];
+        let lon_vec = &columns[1];
+        let resolution_vec = &columns[2];
+
+        let size = lat_vec.len();
+        let mut results =
+            ListVectorBuilder::with_type_capacity(ConcreteDataType::string_datatype(), size);
+
+        for i in 0..size {
+            let lat = lat_vec.get(i).as_f64_lossy();
+            let lon = lon_vec.get(i).as_f64_lossy();
+            let r = try_into_resolution(resolution_vec.get(i))?;
+
+            let result = match (lat, lon) {
+                (Some(lat), Some(lon)) => {
+                    let coord = Coord { x: lon, y: lat };
+                    let encoded = geohash::encode(coord, r)
+                        .map_err(|e| {
+                            BoxedError::new(PlainError::new(
+                                format!("Geohash error: {}", e),
+                                StatusCode::EngineExecuteQuery,
+                            ))
+                        })
+                        .context(error::ExecuteSnafu)?;
+                    let neighbours = geohash::neighbors(&encoded)
+                        .map_err(|e| {
+                            BoxedError::new(PlainError::new(
+                                format!("Geohash error: {}", e),
+                                StatusCode::EngineExecuteQuery,
+                            ))
+                        })
+                        .context(error::ExecuteSnafu)?;
+                    Some(ListValue::new(
+                        vec![
+                            neighbours.n,
+                            neighbours.nw,
+                            neighbours.w,
+                            neighbours.sw,
+                            neighbours.s,
+                            neighbours.se,
+                            neighbours.e,
+                            neighbours.ne,
+                        ]
+                        .into_iter()
+                        .map(Value::from)
+                        .collect(),
+                        ConcreteDataType::string_datatype(),
+                    ))
+                }
+                _ => None,
+            };
+
+            if let Some(list_value) = result {
+                results.push(Some(list_value.as_scalar_ref()));
+            } else {
+                results.push(None);
+            }
+        }
+
+        Ok(results.to_vector())
+    }
+}
+
+impl fmt::Display for GeohashNeighboursFunction {
+    fn fmt(&self, f: &mut fmt::Formatter) -> fmt::Result {
+        write!(f, "{}", GeohashNeighboursFunction::NAME)
    }
 }
--- a/src/common/function/src/scalars/geo/h3.rs
+++ b/src/common/function/src/scalars/geo/h3.rs
--- a/Show More
+++ b/Show More