poc/create-alter-for-metrics:

### Commit Message Enhance Prometheus Bulk Write Handling - **`server.rs`**: Introduced `start_background_task` in `PromBulkState` to handle asynchronous batch processing and SST file writing. Added a new `tx` field to manage task communication. - **`access_layer.rs`**: Added `file_id` method to `ParquetWriter` for file identification. - **`batch_builder.rs`**: Modified `MetricsBatchBuilder` to utilize session catalog and schema, and updated batch processing logic to handle column metadata. - **`prom_store.rs`**: Updated `remote_write` to use `decode_remote_write_request_to_batch` for batch processing and send data to the background task. - **`prom_row_builder.rs`**: Made `TableBuilder` and `TablesBuilder` fields public for external access. - **`proto.rs`**: Exposed `table_data` in `PromWriteRequest` for batch processing. Signed-off-by: Lei, HUANG <mrsatangel@gmail.com>
2025-12-22 22:20:02 +00:00 · 2025-06-30 08:27:26 +00:00 · 2025-06-29 14:07:57 +00:00 · 2025-06-28 09:58:33 +00:00 · 2025-06-27 08:50:25 +00:00 · 2025-06-26 13:00:54 +00:00
452 changed files with 28134 additions and 12989 deletions
--- a/.github/actions/setup-greptimedb-cluster/action.yml
+++ b/.github/actions/setup-greptimedb-cluster/action.yml
@@ -10,13 +10,13 @@ inputs:
  meta-replicas:
    default: 2
    description: "Number of Metasrv replicas"
-  image-registry: 
+  image-registry:
    default: "docker.io"
    description: "Image registry"
-  image-repository: 
+  image-repository:
    default: "greptime/greptimedb"
    description: "Image repository"
-  image-tag: 
+  image-tag:
    default: "latest"
    description: 'Image tag'
  etcd-endpoints:
@@ -32,12 +32,12 @@ runs:
  steps:
  - name: Install GreptimeDB operator
    uses: nick-fields/retry@v3
-    with: 
+    with:
      timeout_minutes: 3
      max_attempts: 3
      shell: bash
      command: |
-        helm repo add greptime https://greptimeteam.github.io/helm-charts/ 
+        helm repo add greptime https://greptimeteam.github.io/helm-charts/
        helm repo update
        helm upgrade \
          --install \
@@ -48,10 +48,10 @@ runs:
          --wait-for-jobs
  - name: Install GreptimeDB cluster
    shell: bash
-    run: | 
+    run: |
      helm upgrade \
        --install my-greptimedb \
-        --set meta.etcdEndpoints=${{ inputs.etcd-endpoints }} \
+        --set meta.backendStorage.etcd.endpoints=${{ inputs.etcd-endpoints }} \
        --set meta.enableRegionFailover=${{ inputs.enable-region-failover }} \
        --set image.registry=${{ inputs.image-registry }} \
        --set image.repository=${{ inputs.image-repository }}  \
@@ -72,7 +72,7 @@ runs:
  - name: Wait for GreptimeDB
    shell: bash
    run: |
-      while true; do 
+      while true; do
        PHASE=$(kubectl -n my-greptimedb get gtc my-greptimedb -o jsonpath='{.status.clusterPhase}')
        if [ "$PHASE" == "Running" ]; then
          echo "Cluster is ready"
@@ -86,10 +86,10 @@ runs:
  - name: Print GreptimeDB info
    if: always()
    shell: bash
-    run: | 
+    run: |
      kubectl get all --show-labels -n my-greptimedb
  - name: Describe Nodes
    if: always()
    shell: bash
-    run: | 
+    run: |
      kubectl describe nodes
--- a/.github/labeler.yaml
+++ b/.github/labeler.yaml
@@ -0,0 +1,15 @@
+ci:
+  - changed-files:
+      - any-glob-to-any-file: .github/**
+
+docker:
+  - changed-files:
+      - any-glob-to-any-file: docker/**
+
+documentation:
+  - changed-files:
+      - any-glob-to-any-file: docs/**
+
+dashboard:
+  - changed-files:
+      - any-glob-to-any-file: grafana/**
--- a/.github/scripts/deploy-greptimedb.sh
+++ b/.github/scripts/deploy-greptimedb.sh
@@ -68,7 +68,7 @@ function deploy_greptimedb_cluster() {

  helm install "$cluster_name" greptime/greptimedb-cluster \
    --set image.tag="$GREPTIMEDB_IMAGE_TAG" \
-    --set meta.etcdEndpoints="etcd.$install_namespace:2379" \
+    --set meta.backendStorage.etcd.endpoints="etcd.$install_namespace:2379" \
    -n "$install_namespace"

  # Wait for greptimedb cluster to be ready.
@@ -103,7 +103,7 @@ function deploy_greptimedb_cluster_with_s3_storage() {

  helm install "$cluster_name" greptime/greptimedb-cluster -n "$install_namespace" \
    --set image.tag="$GREPTIMEDB_IMAGE_TAG" \
-    --set meta.etcdEndpoints="etcd.$install_namespace:2379" \
+    --set meta.backendStorage.etcd.endpoints="etcd.$install_namespace:2379" \
    --set storage.s3.bucket="$AWS_CI_TEST_BUCKET" \
    --set storage.s3.region="$AWS_REGION" \
    --set storage.s3.root="$DATA_ROOT" \
--- a/.github/workflows/pr-labeling.yaml
+++ b/.github/workflows/pr-labeling.yaml
@@ -0,0 +1,42 @@
+name: 'PR Labeling'
+
+on:
+  pull_request_target:
+    types:
+      - opened
+      - synchronize
+      - reopened
+
+permissions:
+  contents: read
+  pull-requests: write
+  issues: write
+
+jobs:
+  labeler:
+    runs-on: ubuntu-latest
+    steps:
+      - name: Checkout sources
+        uses: actions/checkout@v4
+
+      - uses: actions/labeler@v5
+        with:
+          configuration-path: ".github/labeler.yaml"
+          repo-token: "${{ secrets.GITHUB_TOKEN }}"
+
+  size-label:
+    runs-on: ubuntu-latest
+    steps:
+      - uses: pascalgn/size-label-action@v0.5.5
+        env:
+          GITHUB_TOKEN: "${{ secrets.GITHUB_TOKEN }}"
+        with:
+          sizes: >
+            {
+              "0": "XS",
+              "100": "S",
+              "300": "M",
+              "1000": "L",
+              "1500": "XL",
+              "2000": "XXL"
+            }
--- a/Cargo.lock
+++ b/Cargo.lock
@@ -1621,8 +1621,10 @@ dependencies = [
 "cache",
 "catalog",
 "chrono",
+ "common-base",
 "common-catalog",
 "common-error",
+ "common-frontend",
 "common-macro",
 "common-meta",
 "common-procedure",
@@ -1668,9 +1670,9 @@ dependencies = [

 [[package]]
 name = "cc"
-version = "1.1.24"
+version = "1.2.27"
 source = "registry+https://github.com/rust-lang/crates.io-index"
-checksum = "812acba72f0a070b003d3697490d2b55b837230ae7c6c6497f05cc2ddbb8d938"
+checksum = "d487aa071b5f64da6f19a3e848e3578944b726ee5a4854b82172f02aa876bfdc"
 dependencies = [
 "jobserver",
 "libc",
@@ -1948,6 +1950,7 @@ checksum = "1462739cb27611015575c0c11df5df7601141071f07518d56fcc1be504cbec97"
 name = "cli"
 version = "0.15.0"
 dependencies = [
+ "async-stream",
 "async-trait",
 "auth",
 "base64 0.22.1",
@@ -1980,9 +1983,10 @@ dependencies = [
 "meta-srv",
 "nu-ansi-term",
 "object-store",
+ "operator",
 "query",
 "rand 0.9.0",
- "reqwest",
+ "reqwest 0.12.9",
 "serde",
 "serde_json",
 "servers",
@@ -2114,13 +2118,14 @@ dependencies = [
 "mito2",
 "moka",
 "nu-ansi-term",
+ "object-store",
 "plugins",
 "prometheus",
 "prost 0.13.5",
 "query",
 "rand 0.9.0",
 "regex",
- "reqwest",
+ "reqwest 0.12.9",
 "rexpect",
 "serde",
 "serde_json",
@@ -2216,6 +2221,7 @@ dependencies = [
 "humantime-serde",
 "meta-client",
 "num_cpus",
+ "object-store",
 "serde",
 "serde_json",
 "serde_with",
@@ -2293,8 +2299,14 @@ version = "0.15.0"
 dependencies = [
 "async-trait",
 "common-error",
+ "common-grpc",
 "common-macro",
+ "common-meta",
+ "greptime-proto",
+ "meta-client",
 "snafu 0.8.5",
+ "tokio",
+ "tonic 0.12.3",
 ]

 [[package]]
@@ -2322,6 +2334,7 @@ dependencies = [
 "datafusion",
 "datafusion-common",
 "datafusion-expr",
+ "datafusion-functions-aggregate-common",
 "datatypes",
 "derive_more",
 "geo",
@@ -2360,7 +2373,7 @@ dependencies = [
 "common-test-util",
 "common-version",
 "hyper 0.14.30",
- "reqwest",
+ "reqwest 0.12.9",
 "serde",
 "tempfile",
 "tokio",
@@ -2661,7 +2674,6 @@ dependencies = [
 name = "common-telemetry"
 version = "0.15.0"
 dependencies = [
- "atty",
 "backtrace",
 "common-error",
 "console-subscriber",
@@ -3062,9 +3074,9 @@ dependencies = [

 [[package]]
 name = "crossbeam-channel"
-version = "0.5.13"
+version = "0.5.15"
 source = "registry+https://github.com/rust-lang/crates.io-index"
-checksum = "33480d6946193aa8033910124896ca395333cae7e2d1113d1fef6c3272217df2"
+checksum = "82b8f8f868b36967f9606790d1903570de9ceaf870a7bf9fbbd3016d636a2cb2"
 dependencies = [
 "crossbeam-utils",
 ]
@@ -3753,7 +3765,7 @@ dependencies = [
 "prometheus",
 "prost 0.13.5",
 "query",
- "reqwest",
+ "reqwest 0.12.9",
 "serde",
 "serde_json",
 "servers",
@@ -4701,6 +4713,7 @@ dependencies = [
 "common-config",
 "common-datasource",
 "common-error",
+ "common-frontend",
 "common-function",
 "common-grpc",
 "common-macro",
@@ -4725,6 +4738,7 @@ dependencies = [
 "log-store",
 "meta-client",
 "num_cpus",
+ "object-store",
 "opentelemetry-proto 0.27.0",
 "operator",
 "otel-arrow-rust",
@@ -4747,6 +4761,7 @@ dependencies = [
 "substrait 0.15.0",
 "table",
 "tokio",
+ "tokio-util",
 "toml 0.8.19",
 "tonic 0.12.3",
 "tower 0.5.2",
@@ -5133,7 +5148,7 @@ dependencies = [
 [[package]]
 name = "greptime-proto"
 version = "0.1.0"
-source = "git+https://github.com/GreptimeTeam/greptime-proto.git?rev=454c52634c3bac27de10bf0d85d5533eed1cf03f#454c52634c3bac27de10bf0d85d5533eed1cf03f"
+source = "git+https://github.com/GreptimeTeam/greptime-proto.git?rev=464226cf8a4a22696503536a123d0b9e318582f4#464226cf8a4a22696503536a123d0b9e318582f4"
 dependencies = [
 "prost 0.13.5",
 "serde",
@@ -7220,6 +7235,7 @@ dependencies = [
 name = "metric-engine"
 version = "0.15.0"
 dependencies = [
+ "ahash 0.8.11",
 "api",
 "aquamarine",
 "async-stream",
@@ -7240,6 +7256,7 @@ dependencies = [
 "humantime-serde",
 "itertools 0.14.0",
 "lazy_static",
+ "mito-codec",
 "mito2",
 "mur3",
 "object-store",
@@ -7305,6 +7322,29 @@ dependencies = [
 "windows-sys 0.52.0",
 ]

+[[package]]
+name = "mito-codec"
+version = "0.15.0"
+dependencies = [
+ "api",
+ "bytes",
+ "common-base",
+ "common-decimal",
+ "common-error",
+ "common-macro",
+ "common-recordbatch",
+ "common-telemetry",
+ "common-time",
+ "datafusion-common",
+ "datafusion-expr",
+ "datatypes",
+ "memcomparable",
+ "paste",
+ "serde",
+ "snafu 0.8.5",
+ "store-api",
+]
+
 [[package]]
 name = "mito2"
 version = "0.15.0"
@@ -7347,6 +7387,7 @@ dependencies = [
 "lazy_static",
 "log-store",
 "memcomparable",
+ "mito-codec",
 "moka",
 "object-store",
 "parquet",
@@ -8060,14 +8101,21 @@ version = "0.15.0"
 dependencies = [
 "anyhow",
 "bytes",
+ "common-base",
+ "common-error",
+ "common-macro",
 "common-telemetry",
 "common-test-util",
 "futures",
+ "humantime-serde",
 "lazy_static",
 "md5",
 "moka",
 "opendal",
 "prometheus",
+ "reqwest 0.12.9",
+ "serde",
+ "snafu 0.8.5",
 "tokio",
 "uuid",
 ]
@@ -8202,7 +8250,7 @@ dependencies = [
 "prometheus",
 "quick-xml 0.36.2",
 "reqsign",
- "reqwest",
+ "reqwest 0.12.9",
 "serde",
 "serde_json",
 "sha2",
@@ -8274,6 +8322,19 @@ dependencies = [
 "tracing",
 ]

+[[package]]
+name = "opentelemetry-http"
+version = "0.10.0"
+source = "registry+https://github.com/rust-lang/crates.io-index"
+checksum = "7f51189ce8be654f9b5f7e70e49967ed894e84a06fc35c6c042e64ac1fc5399e"
+dependencies = [
+ "async-trait",
+ "bytes",
+ "http 0.2.12",
+ "opentelemetry 0.21.0",
+ "reqwest 0.11.27",
+]
+
 [[package]]
 name = "opentelemetry-otlp"
 version = "0.14.0"
@@ -8284,10 +8345,12 @@ dependencies = [
 "futures-core",
 "http 0.2.12",
 "opentelemetry 0.21.0",
+ "opentelemetry-http",
 "opentelemetry-proto 0.4.0",
 "opentelemetry-semantic-conventions",
 "opentelemetry_sdk 0.21.2",
 "prost 0.11.9",
+ "reqwest 0.11.27",
 "thiserror 1.0.64",
 "tokio",
 "tonic 0.9.2",
@@ -8386,6 +8449,7 @@ dependencies = [
 "common-catalog",
 "common-datasource",
 "common-error",
+ "common-frontend",
 "common-function",
 "common-grpc",
 "common-grpc-expr",
@@ -8885,9 +8949,8 @@ dependencies = [

 [[package]]
 name = "pgwire"
-version = "0.30.1"
-source = "registry+https://github.com/rust-lang/crates.io-index"
-checksum = "ec79ee18e6cafde8698885646780b967ecc905120798b8359dd0da64f9688e89"
+version = "0.30.2"
+source = "git+https://github.com/sunng87/pgwire?rev=127573d997228cfb70c7699881c568eae8131270#127573d997228cfb70c7699881c568eae8131270"
 dependencies = [
 "async-trait",
 "bytes",
@@ -10274,7 +10337,7 @@ dependencies = [
 "percent-encoding",
 "quick-xml 0.35.0",
 "rand 0.8.5",
- "reqwest",
+ "reqwest 0.12.9",
 "rsa",
 "rust-ini 0.21.1",
 "serde",
@@ -10283,6 +10346,42 @@ dependencies = [
 "sha2",
 ]

+[[package]]
+name = "reqwest"
+version = "0.11.27"
+source = "registry+https://github.com/rust-lang/crates.io-index"
+checksum = "dd67538700a17451e7cba03ac727fb961abb7607553461627b97de0b89cf4a62"
+dependencies = [
+ "base64 0.21.7",
+ "bytes",
+ "encoding_rs",
+ "futures-core",
+ "futures-util",
+ "h2 0.3.26",
+ "http 0.2.12",
+ "http-body 0.4.6",
+ "hyper 0.14.30",
+ "ipnet",
+ "js-sys",
+ "log",
+ "mime",
+ "once_cell",
+ "percent-encoding",
+ "pin-project-lite",
+ "serde",
+ "serde_json",
+ "serde_urlencoded",
+ "sync_wrapper 0.1.2",
+ "system-configuration",
+ "tokio",
+ "tower-service",
+ "url",
+ "wasm-bindgen",
+ "wasm-bindgen-futures",
+ "web-sys",
+ "winreg",
+]
+
 [[package]]
 name = "reqwest"
 version = "0.12.9"
@@ -10353,15 +10452,14 @@ dependencies = [

 [[package]]
 name = "ring"
-version = "0.17.8"
+version = "0.17.14"
 source = "registry+https://github.com/rust-lang/crates.io-index"
-checksum = "c17fa4cb658e3583423e915b9f3acc01cceaee1860e33d59ebae66adc3a2dc0d"
+checksum = "a4689e6c2294d81e88dc6261c768b63bc4fcdb852be6d1352498b114f61383b7"
 dependencies = [
 "cc",
 "cfg-if",
 "getrandom 0.2.15",
 "libc",
- "spin",
 "untrusted",
 "windows-sys 0.52.0",
 ]
@@ -11134,7 +11232,9 @@ dependencies = [
 "common-base",
 "common-catalog",
 "common-config",
+ "common-datasource",
 "common-error",
+ "common-frontend",
 "common-grpc",
 "common-macro",
 "common-mem-prof",
@@ -11175,16 +11275,23 @@ dependencies = [
 "local-ip-address",
 "log-query",
 "loki-proto",
+ "metric-engine",
 "mime_guess",
+ "mito-codec",
+ "mito2",
 "mysql_async",
 "notify",
 "object-pool",
+ "object-store",
 "once_cell",
 "openmetrics-parser",
 "opensrv-mysql",
 "opentelemetry-proto 0.27.0",
+ "operator",
 "otel-arrow-rust",
 "parking_lot 0.12.3",
+ "parquet",
+ "partition",
 "permutation",
 "pgwire",
 "pin-project",
@@ -11198,7 +11305,7 @@ dependencies = [
 "quoted-string",
 "rand 0.9.0",
 "regex",
- "reqwest",
+ "reqwest 0.12.9",
 "rust-embed",
 "rustls",
 "rustls-pemfile",
@@ -11642,7 +11749,7 @@ dependencies = [
 "local-ip-address",
 "mysql",
 "num_cpus",
- "reqwest",
+ "reqwest 0.12.9",
 "serde",
 "serde_json",
 "sha2",
@@ -12292,6 +12399,27 @@ dependencies = [
 "nom",
 ]

+[[package]]
+name = "system-configuration"
+version = "0.5.1"
+source = "registry+https://github.com/rust-lang/crates.io-index"
+checksum = "ba3a3adc5c275d719af8cb4272ea1c4a6d668a777f37e115f6d11ddbc1c8e0e7"
+dependencies = [
+ "bitflags 1.3.2",
+ "core-foundation",
+ "system-configuration-sys",
+]
+
+[[package]]
+name = "system-configuration-sys"
+version = "0.5.0"
+source = "registry+https://github.com/rust-lang/crates.io-index"
+checksum = "a75fb188eb626b924683e3b95e3a48e63551fcfb51949de2f06a9d91dbee93c9"
+dependencies = [
+ "core-foundation-sys",
+ "libc",
+]
+
 [[package]]
 name = "table"
 version = "0.15.0"
@@ -12582,7 +12710,7 @@ dependencies = [
 "paste",
 "rand 0.9.0",
 "rand_chacha 0.9.0",
- "reqwest",
+ "reqwest 0.12.9",
 "schemars",
 "serde",
 "serde_json",
@@ -13738,12 +13866,13 @@ checksum = "06abde3611657adf66d383f00b093d7faecc7fa57071cce2578660c9f1010821"

 [[package]]
 name = "uuid"
-version = "1.10.0"
+version = "1.17.0"
 source = "registry+https://github.com/rust-lang/crates.io-index"
-checksum = "81dfa00651efa65069b0b6b651f4aaa31ba9e3c3ce0137aaad053604ee7e0314"
+checksum = "3cf4199d1e5d15ddd86a694e4d0dffa9c323ce759fea589f00fef9d81cc1931d"
 dependencies = [
- "getrandom 0.2.15",
- "rand 0.8.5",
+ "getrandom 0.3.2",
+ "js-sys",
+ "rand 0.9.0",
 "serde",
 "wasm-bindgen",
 ]
@@ -14465,6 +14594,16 @@ dependencies = [
 "memchr",
 ]

+[[package]]
+name = "winreg"
+version = "0.50.0"
+source = "registry+https://github.com/rust-lang/crates.io-index"
+checksum = "524e57b2c537c0f9b1e69f1965311ec12182b4122e45035b1508cd24d2adadb1"
+dependencies = [
+ "cfg-if",
+ "windows-sys 0.48.0",
+]
+
 [[package]]
 name = "wit-bindgen-rt"
 version = "0.39.0"
--- a/Cargo.toml
+++ b/Cargo.toml
@@ -49,6 +49,7 @@ members = [
    "src/meta-client",
    "src/meta-srv",
    "src/metric-engine",
+    "src/mito-codec",
    "src/mito2",
    "src/object-store",
    "src/operator",
@@ -120,6 +121,7 @@ datafusion = { git = "https://github.com/waynexia/arrow-datafusion.git", rev = "
 datafusion-common = { git = "https://github.com/waynexia/arrow-datafusion.git", rev = "12c0381babd52c681043957e9d6ee083a03f7646" }
 datafusion-expr = { git = "https://github.com/waynexia/arrow-datafusion.git", rev = "12c0381babd52c681043957e9d6ee083a03f7646" }
 datafusion-functions = { git = "https://github.com/waynexia/arrow-datafusion.git", rev = "12c0381babd52c681043957e9d6ee083a03f7646" }
+datafusion-functions-aggregate-common = { git = "https://github.com/waynexia/arrow-datafusion.git", rev = "12c0381babd52c681043957e9d6ee083a03f7646" }
 datafusion-optimizer = { git = "https://github.com/waynexia/arrow-datafusion.git", rev = "12c0381babd52c681043957e9d6ee083a03f7646" }
 datafusion-physical-expr = { git = "https://github.com/waynexia/arrow-datafusion.git", rev = "12c0381babd52c681043957e9d6ee083a03f7646" }
 datafusion-physical-plan = { git = "https://github.com/waynexia/arrow-datafusion.git", rev = "12c0381babd52c681043957e9d6ee083a03f7646" }
@@ -133,7 +135,7 @@ etcd-client = "0.14"
 fst = "0.4.7"
 futures = "0.3"
 futures-util = "0.3"
-greptime-proto = { git = "https://github.com/GreptimeTeam/greptime-proto.git", rev = "454c52634c3bac27de10bf0d85d5533eed1cf03f" }
+greptime-proto = { git = "https://github.com/GreptimeTeam/greptime-proto.git", rev = "464226cf8a4a22696503536a123d0b9e318582f4" }
 hex = "0.4"
 http = "1"
 humantime = "2.1"
@@ -274,6 +276,7 @@ log-store = { path = "src/log-store" }
 meta-client = { path = "src/meta-client" }
 meta-srv = { path = "src/meta-srv" }
 metric-engine = { path = "src/metric-engine" }
+mito-codec = { path = "src/mito-codec" }
 mito2 = { path = "src/mito2" }
 object-store = { path = "src/object-store" }
 operator = { path = "src/operator" }
--- a/README.md
+++ b/README.md
@@ -189,7 +189,8 @@ We invite you to engage and contribute!
 - [Official Website](https://greptime.com/)
 - [Blog](https://greptime.com/blogs/)
 - [LinkedIn](https://www.linkedin.com/company/greptime/)
- [Twitter](https://twitter.com/greptime)
+- [X (Twitter)](https://X.com/greptime)
+- [YouTube](https://www.youtube.com/@greptime)

 ## License

--- a/config/config.md
+++ b/config/config.md
@@ -123,6 +123,7 @@
 | `storage.http_client.connect_timeout` | String | `30s` | The timeout for only the connect phase of a http client. |
 | `storage.http_client.timeout` | String | `30s` | The total request timeout, applied from when the request starts connecting until the response body has finished.<br/>Also considered a total deadline. |
 | `storage.http_client.pool_idle_timeout` | String | `90s` | The timeout for idle sockets being kept-alive. |
+| `storage.http_client.skip_ssl_validation` | Bool | `false` | To skip the ssl verification<br/>**Security Notice**: Setting `skip_ssl_validation = true` disables certificate verification, making connections vulnerable to man-in-the-middle attacks. Only use this in development or trusted private networks. |
 | `[[region_engine]]` | -- | -- | The region engine options. You can configure multiple region engines. |
 | `region_engine.mito` | -- | -- | The Mito engine options. |
 | `region_engine.mito.num_workers` | Integer | `8` | Number of region workers. |
@@ -184,10 +185,11 @@
 | `logging.dir` | String | `./greptimedb_data/logs` | The directory to store the log files. If set to empty, logs will not be written to files. |
 | `logging.level` | String | Unset | The log level. Can be `info`/`debug`/`warn`/`error`. |
 | `logging.enable_otlp_tracing` | Bool | `false` | Enable OTLP tracing. |
-| `logging.otlp_endpoint` | String | `http://localhost:4317` | The OTLP tracing endpoint. |
+| `logging.otlp_endpoint` | String | `http://localhost:4318` | The OTLP tracing endpoint. |
 | `logging.append_stdout` | Bool | `true` | Whether to append logs to stdout. |
 | `logging.log_format` | String | `text` | The log format. Can be `text`/`json`. |
 | `logging.max_log_files` | Integer | `720` | The maximum amount of log files. |
+| `logging.otlp_export_protocol` | String | `http` | The OTLP tracing export protocol. Can be `grpc`/`http`. |
 | `logging.tracing_sample_ratio` | -- | -- | The percentage of tracing will be sampled and exported.<br/>Valid range `[0, 1]`, 1 means all traces are sampled, 0 means all traces are not sampled, the default value is 1.<br/>ratio > 1 are treated as 1. Fractions < 0 are treated as 0 |
 | `logging.tracing_sample_ratio.default_ratio` | Float | `1.0` | -- |
 | `slow_query` | -- | -- | The slow query log options. |
@@ -232,7 +234,7 @@
 | `grpc.bind_addr` | String | `127.0.0.1:4001` | The address to bind the gRPC server. |
 | `grpc.server_addr` | String | `127.0.0.1:4001` | The address advertised to the metasrv, and used for connections from outside the host.<br/>If left empty or unset, the server will automatically use the IP address of the first network interface<br/>on the host, with the same port number as the one specified in `grpc.bind_addr`. |
 | `grpc.runtime_size` | Integer | `8` | The number of server worker threads. |
-| `grpc.flight_compression` | String | `arrow_ipc` | Compression mode for frontend side Arrow IPC service. Available options:<br/>- `none`: disable all compression<br/>- `transport`: only enable gRPC transport compression (zstd)<br/>- `arrow_ipc`: only enable Arrow IPC compression (lz4)<br/>- `all`: enable all compression. |
+| `grpc.flight_compression` | String | `arrow_ipc` | Compression mode for frontend side Arrow IPC service. Available options:<br/>- `none`: disable all compression<br/>- `transport`: only enable gRPC transport compression (zstd)<br/>- `arrow_ipc`: only enable Arrow IPC compression (lz4)<br/>- `all`: enable all compression.<br/>Default to `none` |
 | `grpc.tls` | -- | -- | gRPC server TLS options, see `mysql.tls` section. |
 | `grpc.tls.mode` | String | `disable` | TLS mode. |
 | `grpc.tls.cert_path` | String | Unset | Certificate file path. |
@@ -287,10 +289,11 @@
 | `logging.dir` | String | `./greptimedb_data/logs` | The directory to store the log files. If set to empty, logs will not be written to files. |
 | `logging.level` | String | Unset | The log level. Can be `info`/`debug`/`warn`/`error`. |
 | `logging.enable_otlp_tracing` | Bool | `false` | Enable OTLP tracing. |
-| `logging.otlp_endpoint` | String | `http://localhost:4317` | The OTLP tracing endpoint. |
+| `logging.otlp_endpoint` | String | `http://localhost:4318` | The OTLP tracing endpoint. |
 | `logging.append_stdout` | Bool | `true` | Whether to append logs to stdout. |
 | `logging.log_format` | String | `text` | The log format. Can be `text`/`json`. |
 | `logging.max_log_files` | Integer | `720` | The maximum amount of log files. |
+| `logging.otlp_export_protocol` | String | `http` | The OTLP tracing export protocol. Can be `grpc`/`http`. |
 | `logging.tracing_sample_ratio` | -- | -- | The percentage of tracing will be sampled and exported.<br/>Valid range `[0, 1]`, 1 means all traces are sampled, 0 means all traces are not sampled, the default value is 1.<br/>ratio > 1 are treated as 1. Fractions < 0 are treated as 0 |
 | `logging.tracing_sample_ratio.default_ratio` | Float | `1.0` | -- |
 | `slow_query` | -- | -- | The slow query log options. |
@@ -322,6 +325,7 @@
 | `selector` | String | `round_robin` | Datanode selector type.<br/>- `round_robin` (default value)<br/>- `lease_based`<br/>- `load_based`<br/>For details, please see "https://docs.greptime.com/developer-guide/metasrv/selector". |
 | `use_memory_store` | Bool | `false` | Store data in memory. |
 | `enable_region_failover` | Bool | `false` | Whether to enable region failover.<br/>This feature is only available on GreptimeDB running on cluster mode and<br/>- Using Remote WAL<br/>- Using shared storage (e.g., s3). |
+| `region_failure_detector_initialization_delay` | String | `10m` | Delay before initializing region failure detectors.<br/>This delay helps prevent premature initialization of region failure detectors in cases where<br/>cluster maintenance mode is enabled right after metasrv starts, especially when the cluster<br/>is not deployed via the recommended GreptimeDB Operator. Without this delay, early detector registration<br/>may trigger unnecessary region failovers during datanode startup. |
 | `allow_region_failover_on_local_wal` | Bool | `false` | Whether to allow region failover on local WAL.<br/>**This option is not recommended to be set to true, because it may lead to data loss during failover.** |
 | `node_max_idle_time` | String | `24hours` | Max allowed idle time before removing node info from metasrv memory. |
 | `enable_telemetry` | Bool | `true` | Whether to enable greptimedb telemetry. Enabled by default. |
@@ -369,10 +373,11 @@
 | `logging.dir` | String | `./greptimedb_data/logs` | The directory to store the log files. If set to empty, logs will not be written to files. |
 | `logging.level` | String | Unset | The log level. Can be `info`/`debug`/`warn`/`error`. |
 | `logging.enable_otlp_tracing` | Bool | `false` | Enable OTLP tracing. |
-| `logging.otlp_endpoint` | String | `http://localhost:4317` | The OTLP tracing endpoint. |
+| `logging.otlp_endpoint` | String | `http://localhost:4318` | The OTLP tracing endpoint. |
 | `logging.append_stdout` | Bool | `true` | Whether to append logs to stdout. |
 | `logging.log_format` | String | `text` | The log format. Can be `text`/`json`. |
 | `logging.max_log_files` | Integer | `720` | The maximum amount of log files. |
+| `logging.otlp_export_protocol` | String | `http` | The OTLP tracing export protocol. Can be `grpc`/`http`. |
 | `logging.tracing_sample_ratio` | -- | -- | The percentage of tracing will be sampled and exported.<br/>Valid range `[0, 1]`, 1 means all traces are sampled, 0 means all traces are not sampled, the default value is 1.<br/>ratio > 1 are treated as 1. Fractions < 0 are treated as 0 |
 | `logging.tracing_sample_ratio.default_ratio` | Float | `1.0` | -- |
 | `export_metrics` | -- | -- | The metasrv can export its metrics and send to Prometheus compatible service (e.g. `greptimedb` itself) from remote-write API.<br/>This is only used for `greptimedb` to export its own metrics internally. It's different from prometheus scrape. |
@@ -405,7 +410,7 @@
 | `grpc.runtime_size` | Integer | `8` | The number of server worker threads. |
 | `grpc.max_recv_message_size` | String | `512MB` | The maximum receive message size for gRPC server. |
 | `grpc.max_send_message_size` | String | `512MB` | The maximum send message size for gRPC server. |
-| `grpc.flight_compression` | String | `arrow_ipc` | Compression mode for datanode side Arrow IPC service. Available options:<br/>- `none`: disable all compression<br/>- `transport`: only enable gRPC transport compression (zstd)<br/>- `arrow_ipc`: only enable Arrow IPC compression (lz4)<br/>- `all`: enable all compression. |
+| `grpc.flight_compression` | String | `arrow_ipc` | Compression mode for datanode side Arrow IPC service. Available options:<br/>- `none`: disable all compression<br/>- `transport`: only enable gRPC transport compression (zstd)<br/>- `arrow_ipc`: only enable Arrow IPC compression (lz4)<br/>- `all`: enable all compression.<br/>Default to `none` |
 | `grpc.tls` | -- | -- | gRPC server TLS options, see `mysql.tls` section. |
 | `grpc.tls.mode` | String | `disable` | TLS mode. |
 | `grpc.tls.cert_path` | String | Unset | Certificate file path. |
@@ -471,6 +476,7 @@
 | `storage.http_client.connect_timeout` | String | `30s` | The timeout for only the connect phase of a http client. |
 | `storage.http_client.timeout` | String | `30s` | The total request timeout, applied from when the request starts connecting until the response body has finished.<br/>Also considered a total deadline. |
 | `storage.http_client.pool_idle_timeout` | String | `90s` | The timeout for idle sockets being kept-alive. |
+| `storage.http_client.skip_ssl_validation` | Bool | `false` | To skip the ssl verification<br/>**Security Notice**: Setting `skip_ssl_validation = true` disables certificate verification, making connections vulnerable to man-in-the-middle attacks. Only use this in development or trusted private networks. |
 | `[[region_engine]]` | -- | -- | The region engine options. You can configure multiple region engines. |
 | `region_engine.mito` | -- | -- | The Mito engine options. |
 | `region_engine.mito.num_workers` | Integer | `8` | Number of region workers. |
@@ -532,10 +538,11 @@
 | `logging.dir` | String | `./greptimedb_data/logs` | The directory to store the log files. If set to empty, logs will not be written to files. |
 | `logging.level` | String | Unset | The log level. Can be `info`/`debug`/`warn`/`error`. |
 | `logging.enable_otlp_tracing` | Bool | `false` | Enable OTLP tracing. |
-| `logging.otlp_endpoint` | String | `http://localhost:4317` | The OTLP tracing endpoint. |
+| `logging.otlp_endpoint` | String | `http://localhost:4318` | The OTLP tracing endpoint. |
 | `logging.append_stdout` | Bool | `true` | Whether to append logs to stdout. |
 | `logging.log_format` | String | `text` | The log format. Can be `text`/`json`. |
 | `logging.max_log_files` | Integer | `720` | The maximum amount of log files. |
+| `logging.otlp_export_protocol` | String | `http` | The OTLP tracing export protocol. Can be `grpc`/`http`. |
 | `logging.tracing_sample_ratio` | -- | -- | The percentage of tracing will be sampled and exported.<br/>Valid range `[0, 1]`, 1 means all traces are sampled, 0 means all traces are not sampled, the default value is 1.<br/>ratio > 1 are treated as 1. Fractions < 0 are treated as 0 |
 | `logging.tracing_sample_ratio.default_ratio` | Float | `1.0` | -- |
 | `export_metrics` | -- | -- | The datanode can export its metrics and send to Prometheus compatible service (e.g. `greptimedb` itself) from remote-write API.<br/>This is only used for `greptimedb` to export its own metrics internally. It's different from prometheus scrape. |
@@ -582,10 +589,11 @@
 | `logging.dir` | String | `./greptimedb_data/logs` | The directory to store the log files. If set to empty, logs will not be written to files. |
 | `logging.level` | String | Unset | The log level. Can be `info`/`debug`/`warn`/`error`. |
 | `logging.enable_otlp_tracing` | Bool | `false` | Enable OTLP tracing. |
-| `logging.otlp_endpoint` | String | `http://localhost:4317` | The OTLP tracing endpoint. |
+| `logging.otlp_endpoint` | String | `http://localhost:4318` | The OTLP tracing endpoint. |
 | `logging.append_stdout` | Bool | `true` | Whether to append logs to stdout. |
 | `logging.log_format` | String | `text` | The log format. Can be `text`/`json`. |
 | `logging.max_log_files` | Integer | `720` | The maximum amount of log files. |
+| `logging.otlp_export_protocol` | String | `http` | The OTLP tracing export protocol. Can be `grpc`/`http`. |
 | `logging.tracing_sample_ratio` | -- | -- | The percentage of tracing will be sampled and exported.<br/>Valid range `[0, 1]`, 1 means all traces are sampled, 0 means all traces are not sampled, the default value is 1.<br/>ratio > 1 are treated as 1. Fractions < 0 are treated as 0 |
 | `logging.tracing_sample_ratio.default_ratio` | Float | `1.0` | -- |
 | `tracing` | -- | -- | The tracing options. Only effect when compiled with `tokio-console` feature. |
--- a/config/datanode.example.toml
+++ b/config/datanode.example.toml
@@ -367,6 +367,10 @@ timeout = "30s"
 ## The timeout for idle sockets being kept-alive.
 pool_idle_timeout = "90s"

+## To skip the ssl verification
+## **Security Notice**: Setting `skip_ssl_validation = true` disables certificate verification, making connections vulnerable to man-in-the-middle attacks. Only use this in development or trusted private networks.
+skip_ssl_validation = false
+
 # Custom storage options
 # [[storage.providers]]
 # name = "S3"
@@ -625,7 +629,7 @@ level = "info"
 enable_otlp_tracing = false

 ## The OTLP tracing endpoint.
-otlp_endpoint = "http://localhost:4317"
+otlp_endpoint = "http://localhost:4318"

 ## Whether to append logs to stdout.
 append_stdout = true
@@ -636,6 +640,9 @@ log_format = "text"
 ## The maximum amount of log files.
 max_log_files = 720

+## The OTLP tracing export protocol. Can be `grpc`/`http`.
+otlp_export_protocol = "http"
+
 ## The percentage of tracing will be sampled and exported.
 ## Valid range `[0, 1]`, 1 means all traces are sampled, 0 means all traces are not sampled, the default value is 1.
 ## ratio > 1 are treated as 1. Fractions < 0 are treated as 0
--- a/config/flownode.example.toml
+++ b/config/flownode.example.toml
@@ -83,7 +83,7 @@ level = "info"
 enable_otlp_tracing = false

 ## The OTLP tracing endpoint.
-otlp_endpoint = "http://localhost:4317"
+otlp_endpoint = "http://localhost:4318"

 ## Whether to append logs to stdout.
 append_stdout = true
@@ -94,6 +94,9 @@ log_format = "text"
 ## The maximum amount of log files.
 max_log_files = 720

+## The OTLP tracing export protocol. Can be `grpc`/`http`.
+otlp_export_protocol = "http"
+
 ## The percentage of tracing will be sampled and exported.
 ## Valid range `[0, 1]`, 1 means all traces are sampled, 0 means all traces are not sampled, the default value is 1.
 ## ratio > 1 are treated as 1. Fractions < 0 are treated as 0
--- a/config/frontend.example.toml
+++ b/config/frontend.example.toml
@@ -218,7 +218,7 @@ level = "info"
 enable_otlp_tracing = false

 ## The OTLP tracing endpoint.
-otlp_endpoint = "http://localhost:4317"
+otlp_endpoint = "http://localhost:4318"

 ## Whether to append logs to stdout.
 append_stdout = true
@@ -229,6 +229,9 @@ log_format = "text"
 ## The maximum amount of log files.
 max_log_files = 720

+## The OTLP tracing export protocol. Can be `grpc`/`http`.
+otlp_export_protocol = "http"
+
 ## The percentage of tracing will be sampled and exported.
 ## Valid range `[0, 1]`, 1 means all traces are sampled, 0 means all traces are not sampled, the default value is 1.
 ## ratio > 1 are treated as 1. Fractions < 0 are treated as 0
--- a/config/metasrv.example.toml
+++ b/config/metasrv.example.toml
@@ -43,6 +43,13 @@ use_memory_store = false
 ## - Using shared storage (e.g., s3).
 enable_region_failover = false

+## Delay before initializing region failure detectors.
+## This delay helps prevent premature initialization of region failure detectors in cases where
+## cluster maintenance mode is enabled right after metasrv starts, especially when the cluster
+## is not deployed via the recommended GreptimeDB Operator. Without this delay, early detector registration
+## may trigger unnecessary region failovers during datanode startup.
+region_failure_detector_initialization_delay = '10m'
+
 ## Whether to allow region failover on local WAL.
 ## **This option is not recommended to be set to true, because it may lead to data loss during failover.**
 allow_region_failover_on_local_wal = false
@@ -220,7 +227,7 @@ level = "info"
 enable_otlp_tracing = false

 ## The OTLP tracing endpoint.
-otlp_endpoint = "http://localhost:4317"
+otlp_endpoint = "http://localhost:4318"

 ## Whether to append logs to stdout.
 append_stdout = true
@@ -231,6 +238,9 @@ log_format = "text"
 ## The maximum amount of log files.
 max_log_files = 720

+## The OTLP tracing export protocol. Can be `grpc`/`http`.
+otlp_export_protocol = "http"
+
 ## The percentage of tracing will be sampled and exported.
 ## Valid range `[0, 1]`, 1 means all traces are sampled, 0 means all traces are not sampled, the default value is 1.
 ## ratio > 1 are treated as 1. Fractions < 0 are treated as 0
--- a/config/standalone.example.toml
+++ b/config/standalone.example.toml
@@ -458,6 +458,10 @@ timeout = "30s"
 ## The timeout for idle sockets being kept-alive.
 pool_idle_timeout = "90s"

+## To skip the ssl verification
+## **Security Notice**: Setting `skip_ssl_validation = true` disables certificate verification, making connections vulnerable to man-in-the-middle attacks. Only use this in development or trusted private networks.
+skip_ssl_validation = false
+
 # Custom storage options
 # [[storage.providers]]
 # name = "S3"
@@ -716,7 +720,7 @@ level = "info"
 enable_otlp_tracing = false

 ## The OTLP tracing endpoint.
-otlp_endpoint = "http://localhost:4317"
+otlp_endpoint = "http://localhost:4318"

 ## Whether to append logs to stdout.
 append_stdout = true
@@ -727,6 +731,9 @@ log_format = "text"
 ## The maximum amount of log files.
 max_log_files = 720

+## The OTLP tracing export protocol. Can be `grpc`/`http`.
+otlp_export_protocol = "http"
+
 ## The percentage of tracing will be sampled and exported.
 ## Valid range `[0, 1]`, 1 means all traces are sampled, 0 means all traces are not sampled, the default value is 1.
 ## ratio > 1 are treated as 1. Fractions < 0 are treated as 0
--- a/grafana/README.md
+++ b/grafana/README.md
@@ -9,7 +9,7 @@ We highly recommend using the self-monitoring feature provided by [GreptimeDB Op
 - **Metrics Dashboards**

  - `dashboards/metrics/cluster/dashboard.json`: The Grafana dashboard for the GreptimeDB cluster. Read the [dashboard.md](./dashboards/metrics/cluster/dashboard.md) for more details.
-  
+
  - `dashboards/metrics/standalone/dashboard.json`: The Grafana dashboard for the standalone GreptimeDB instance. **It's generated from the `cluster/dashboard.json` by removing the instance filter through the `make dashboards` command**. Read the [dashboard.md](./dashboards/metrics/standalone/dashboard.md) for more details.

 - **Logs Dashboard**
@@ -83,7 +83,7 @@ If you use the [Helm Chart](https://github.com/GreptimeTeam/helm-charts) to depl
 - `monitoring.enabled=true`: Deploys a standalone GreptimeDB instance dedicated to monitoring the cluster;
 - `grafana.enabled=true`: Deploys Grafana and automatically imports the monitoring dashboard;

-The standalone GreptimeDB instance will collect metrics from your cluster, and the dashboard will be available in the Grafana UI. For detailed deployment instructions, please refer to our [Kubernetes deployment guide](https://docs.greptime.com/nightly/user-guide/deployments/deploy-on-kubernetes/getting-started).
+The standalone GreptimeDB instance will collect metrics from your cluster, and the dashboard will be available in the Grafana UI. For detailed deployment instructions, please refer to our [Kubernetes deployment guide](https://docs.greptime.com/user-guide/deployments-administration/deploy-on-kubernetes/getting-started).

 ### Self-host Prometheus and import dashboards manually

--- a/grafana/dashboards/metrics/cluster/dashboard.json
+++ b/grafana/dashboards/metrics/cluster/dashboard.json
--- a/grafana/dashboards/metrics/cluster/dashboard.md
+++ b/grafana/dashboards/metrics/cluster/dashboard.md
@@ -70,6 +70,7 @@
 | Inflight Flush | `greptime_mito_inflight_flush_count` | `timeseries` | Ongoing flush task count | `prometheus` | `none` | `[{{instance}}]-[{{pod}}]` |
 | Compaction Input/Output Bytes | `sum by(instance, pod) (greptime_mito_compaction_input_bytes)`<br/>`sum by(instance, pod) (greptime_mito_compaction_output_bytes)` | `timeseries` | Compaction oinput output bytes | `prometheus` | `bytes` | `[{{instance}}]-[{{pod}}]-input` |
 | Region Worker Handle Bulk Insert Requests | `histogram_quantile(0.95, sum by(le,instance, stage, pod) (rate(greptime_region_worker_handle_write_bucket[$__rate_interval])))`<br/>`sum by(instance, stage, pod) (rate(greptime_region_worker_handle_write_sum[$__rate_interval]))/sum by(instance, stage, pod) (rate(greptime_region_worker_handle_write_count[$__rate_interval]))` | `timeseries` | Per-stage elapsed time for region worker to handle bulk insert region requests. | `prometheus` | `s` | `[{{instance}}]-[{{pod}}]-[{{stage}}]-P95` |
+| Active Series and Field Builders Count | `sum by(instance, pod) (greptime_mito_memtable_active_series_count)`<br/>`sum by(instance, pod) (greptime_mito_memtable_field_builder_count)` | `timeseries` | Compaction oinput output bytes | `prometheus` | `none` | `[{{instance}}]-[{{pod}}]-series` |
 | Region Worker Convert Requests | `histogram_quantile(0.95, sum by(le, instance, stage, pod) (rate(greptime_datanode_convert_region_request_bucket[$__rate_interval])))`<br/>`sum by(le,instance, stage, pod) (rate(greptime_datanode_convert_region_request_sum[$__rate_interval]))/sum by(le,instance, stage, pod) (rate(greptime_datanode_convert_region_request_count[$__rate_interval]))` | `timeseries` | Per-stage elapsed time for region worker to decode requests. | `prometheus` | `s` | `[{{instance}}]-[{{pod}}]-[{{stage}}]-P95` |
 # OpenDAL
 | Title | Query | Type | Description | Datasource | Unit | Legend Format |
--- a/grafana/dashboards/metrics/cluster/dashboard.yaml
+++ b/grafana/dashboards/metrics/cluster/dashboard.yaml
@@ -612,6 +612,21 @@ groups:
                type: prometheus
                uid: ${metrics}
              legendFormat: '[{{instance}}]-[{{pod}}]-[{{stage}}]-AVG'
+        - title: Active Series and Field Builders Count
+          type: timeseries
+          description: Compaction oinput output bytes
+          unit: none
+          queries:
+            - expr: sum by(instance, pod) (greptime_mito_memtable_active_series_count)
+              datasource:
+                type: prometheus
+                uid: ${metrics}
+              legendFormat: '[{{instance}}]-[{{pod}}]-series'
+            - expr: sum by(instance, pod) (greptime_mito_memtable_field_builder_count)
+              datasource:
+                type: prometheus
+                uid: ${metrics}
+              legendFormat: '[{{instance}}]-[{{pod}}]-field_builders'
        - title: Region Worker Convert Requests
          type: timeseries
          description: Per-stage elapsed time for region worker to decode requests.
--- a/grafana/dashboards/metrics/standalone/dashboard.json
+++ b/grafana/dashboards/metrics/standalone/dashboard.json
--- a/grafana/dashboards/metrics/standalone/dashboard.md
+++ b/grafana/dashboards/metrics/standalone/dashboard.md
@@ -70,6 +70,7 @@
 | Inflight Flush | `greptime_mito_inflight_flush_count` | `timeseries` | Ongoing flush task count | `prometheus` | `none` | `[{{instance}}]-[{{pod}}]` |
 | Compaction Input/Output Bytes | `sum by(instance, pod) (greptime_mito_compaction_input_bytes)`<br/>`sum by(instance, pod) (greptime_mito_compaction_output_bytes)` | `timeseries` | Compaction oinput output bytes | `prometheus` | `bytes` | `[{{instance}}]-[{{pod}}]-input` |
 | Region Worker Handle Bulk Insert Requests | `histogram_quantile(0.95, sum by(le,instance, stage, pod) (rate(greptime_region_worker_handle_write_bucket[$__rate_interval])))`<br/>`sum by(instance, stage, pod) (rate(greptime_region_worker_handle_write_sum[$__rate_interval]))/sum by(instance, stage, pod) (rate(greptime_region_worker_handle_write_count[$__rate_interval]))` | `timeseries` | Per-stage elapsed time for region worker to handle bulk insert region requests. | `prometheus` | `s` | `[{{instance}}]-[{{pod}}]-[{{stage}}]-P95` |
+| Active Series and Field Builders Count | `sum by(instance, pod) (greptime_mito_memtable_active_series_count)`<br/>`sum by(instance, pod) (greptime_mito_memtable_field_builder_count)` | `timeseries` | Compaction oinput output bytes | `prometheus` | `none` | `[{{instance}}]-[{{pod}}]-series` |
 | Region Worker Convert Requests | `histogram_quantile(0.95, sum by(le, instance, stage, pod) (rate(greptime_datanode_convert_region_request_bucket[$__rate_interval])))`<br/>`sum by(le,instance, stage, pod) (rate(greptime_datanode_convert_region_request_sum[$__rate_interval]))/sum by(le,instance, stage, pod) (rate(greptime_datanode_convert_region_request_count[$__rate_interval]))` | `timeseries` | Per-stage elapsed time for region worker to decode requests. | `prometheus` | `s` | `[{{instance}}]-[{{pod}}]-[{{stage}}]-P95` |
 # OpenDAL
 | Title | Query | Type | Description | Datasource | Unit | Legend Format |
--- a/grafana/dashboards/metrics/standalone/dashboard.yaml
+++ b/grafana/dashboards/metrics/standalone/dashboard.yaml
@@ -612,6 +612,21 @@ groups:
                type: prometheus
                uid: ${metrics}
              legendFormat: '[{{instance}}]-[{{pod}}]-[{{stage}}]-AVG'
+        - title: Active Series and Field Builders Count
+          type: timeseries
+          description: Compaction oinput output bytes
+          unit: none
+          queries:
+            - expr: sum by(instance, pod) (greptime_mito_memtable_active_series_count)
+              datasource:
+                type: prometheus
+                uid: ${metrics}
+              legendFormat: '[{{instance}}]-[{{pod}}]-series'
+            - expr: sum by(instance, pod) (greptime_mito_memtable_field_builder_count)
+              datasource:
+                type: prometheus
+                uid: ${metrics}
+              legendFormat: '[{{instance}}]-[{{pod}}]-field_builders'
        - title: Region Worker Convert Requests
          type: timeseries
          description: Per-stage elapsed time for region worker to decode requests.
--- a/licenserc.toml
+++ b/licenserc.toml
@@ -31,6 +31,7 @@ excludes = [
    "src/operator/src/expr_helper/trigger.rs",
    "src/sql/src/statements/create/trigger.rs",
    "src/sql/src/statements/show/trigger.rs",
+    "src/sql/src/statements/drop/trigger.rs",
    "src/sql/src/parsers/create_parser/trigger.rs",
    "src/sql/src/parsers/show_parser/trigger.rs",
 ]
--- a/src/api/src/region.rs
+++ b/src/api/src/region.rs
@@ -22,6 +22,7 @@ use greptime_proto::v1::region::RegionResponse as RegionResponseV1;
 pub struct RegionResponse {
    pub affected_rows: AffectedRows,
    pub extensions: HashMap<String, Vec<u8>>,
+    pub metadata: Vec<u8>,
 }

 impl RegionResponse {
@@ -29,6 +30,7 @@ impl RegionResponse {
        Self {
            affected_rows: region_response.affected_rows as _,
            extensions: region_response.extensions,
+            metadata: region_response.metadata,
        }
    }

@@ -37,6 +39,16 @@ impl RegionResponse {
        Self {
            affected_rows,
            extensions: Default::default(),
+            metadata: Vec::new(),
+        }
+    }
+
+    /// Creates one response with metadata.
+    pub fn from_metadata(metadata: Vec<u8>) -> Self {
+        Self {
+            affected_rows: 0,
+            extensions: Default::default(),
+            metadata,
        }
    }
 }
--- a/src/catalog/Cargo.toml
+++ b/src/catalog/Cargo.toml
@@ -17,8 +17,10 @@ arrow-schema.workspace = true
 async-stream.workspace = true
 async-trait.workspace = true
 bytes.workspace = true
+common-base.workspace = true
 common-catalog.workspace = true
 common-error.workspace = true
+common-frontend.workspace = true
 common-macro.workspace = true
 common-meta.workspace = true
 common-procedure.workspace = true
--- a/src/catalog/src/error.rs
+++ b/src/catalog/src/error.rs
@@ -277,6 +277,26 @@ pub enum Error {
        #[snafu(implicit)]
        location: Location,
    },
+
+    #[snafu(display("Failed to invoke frontend services"))]
+    InvokeFrontend {
+        source: common_frontend::error::Error,
+        #[snafu(implicit)]
+        location: Location,
+    },
+
+    #[snafu(display("Meta client is not provided"))]
+    MetaClientMissing {
+        #[snafu(implicit)]
+        location: Location,
+    },
+
+    #[snafu(display("Failed to find frontend node: {}", addr))]
+    FrontendNotFound {
+        addr: String,
+        #[snafu(implicit)]
+        location: Location,
+    },
 }

 impl Error {
@@ -345,6 +365,10 @@ impl ErrorExt for Error {
            Error::GetViewCache { source, .. } | Error::GetTableCache { source, .. } => {
                source.status_code()
            }
+            Error::InvokeFrontend { source, .. } => source.status_code(),
+            Error::FrontendNotFound { .. } | Error::MetaClientMissing { .. } => {
+                StatusCode::Unexpected
+            }
        }
    }

--- a/src/catalog/src/kvbackend/manager.rs
+++ b/src/catalog/src/kvbackend/manager.rs
@@ -22,7 +22,9 @@ use common_catalog::consts::{
    PG_CATALOG_NAME,
 };
 use common_error::ext::BoxedError;
-use common_meta::cache::{LayeredCacheRegistryRef, ViewInfoCacheRef};
+use common_meta::cache::{
+    LayeredCacheRegistryRef, TableRoute, TableRouteCacheRef, ViewInfoCacheRef,
+};
 use common_meta::key::catalog_name::CatalogNameKey;
 use common_meta::key::flow::FlowMetadataManager;
 use common_meta::key::schema_name::SchemaNameKey;
@@ -51,6 +53,7 @@ use crate::error::{
 };
 use crate::information_schema::{InformationExtensionRef, InformationSchemaProvider};
 use crate::kvbackend::TableCacheRef;
+use crate::process_manager::ProcessManagerRef;
 use crate::system_schema::pg_catalog::PGCatalogProvider;
 use crate::system_schema::SystemSchemaProvider;
 use crate::CatalogManager;
@@ -84,6 +87,7 @@ impl KvBackendCatalogManager {
        backend: KvBackendRef,
        cache_registry: LayeredCacheRegistryRef,
        procedure_manager: Option<ProcedureManagerRef>,
+        process_manager: Option<ProcessManagerRef>,
    ) -> Arc<Self> {
        Arc::new_cyclic(|me| Self {
            information_extension,
@@ -102,12 +106,14 @@ impl KvBackendCatalogManager {
                    DEFAULT_CATALOG_NAME.to_string(),
                    me.clone(),
                    Arc::new(FlowMetadataManager::new(backend.clone())),
+                    process_manager.clone(),
                )),
                pg_catalog_provider: Arc::new(PGCatalogProvider::new(
                    DEFAULT_CATALOG_NAME.to_string(),
                    me.clone(),
                )),
                backend,
+                process_manager,
            },
            cache_registry,
            procedure_manager,
@@ -262,16 +268,68 @@ impl CatalogManager for KvBackendCatalogManager {
        let table_cache: TableCacheRef = self.cache_registry.get().context(CacheNotFoundSnafu {
            name: "table_cache",
        })?;
-        if let Some(table) = table_cache
+        let table_route_cache: TableRouteCacheRef =
+            self.cache_registry.get().context(CacheNotFoundSnafu {
+                name: "table_route_cache",
+            })?;
+        let table = table_cache
            .get_by_ref(&TableName {
                catalog_name: catalog_name.to_string(),
                schema_name: schema_name.to_string(),
                table_name: table_name.to_string(),
            })
            .await
-            .context(GetTableCacheSnafu)?
+            .context(GetTableCacheSnafu)?;
+
+        // Override logical table's partition key indices with physical table's.
+        if let Some(table) = &table
+            && let Some(table_route_value) = table_route_cache
+                .get(table.table_info().table_id())
+                .await
+                .context(TableMetadataManagerSnafu)?
+            && let TableRoute::Logical(logical_route) = &*table_route_value
+            && let Some(physical_table_info_value) = self
+                .table_metadata_manager
+                .table_info_manager()
+                .get(logical_route.physical_table_id())
+                .await
+                .context(TableMetadataManagerSnafu)?
        {
-            return Ok(Some(table));
+            let mut new_table_info = (*table.table_info()).clone();
+            // Gather all column names from the logical table
+            let logical_column_names: std::collections::HashSet<_> = new_table_info
+                .meta
+                .schema
+                .column_schemas()
+                .iter()
+                .map(|col| &col.name)
+                .collect();
+
+            // Only preserve partition key indices where the corresponding columns exist in logical table
+            new_table_info.meta.partition_key_indices = physical_table_info_value
+                .table_info
+                .meta
+                .partition_key_indices
+                .iter()
+                .filter(|&&index| {
+                    if let Some(physical_column) = physical_table_info_value
+                        .table_info
+                        .meta
+                        .schema
+                        .column_schemas
+                        .get(index)
+                    {
+                        logical_column_names.contains(&physical_column.name)
+                    } else {
+                        false
+                    }
+                })
+                .cloned()
+                .collect();
+
+            let new_table = DistTable::table(Arc::new(new_table_info));
+
+            return Ok(Some(new_table));
        }

        if channel == Channel::Postgres {
@@ -284,7 +342,7 @@ impl CatalogManager for KvBackendCatalogManager {
            }
        }

-        return Ok(None);
+        Ok(table)
    }

    async fn tables_by_ids(
@@ -419,6 +477,7 @@ struct SystemCatalog {
    information_schema_provider: Arc<InformationSchemaProvider>,
    pg_catalog_provider: Arc<PGCatalogProvider>,
    backend: KvBackendRef,
+    process_manager: Option<ProcessManagerRef>,
 }

 impl SystemCatalog {
@@ -486,6 +545,7 @@ impl SystemCatalog {
                        catalog.to_string(),
                        self.catalog_manager.clone(),
                        Arc::new(FlowMetadataManager::new(self.backend.clone())),
+                        self.process_manager.clone(),
                    ))
                });
            information_schema_provider.table(table_name)
--- a/src/catalog/src/lib.rs
+++ b/src/catalog/src/lib.rs
@@ -14,6 +14,7 @@

 #![feature(assert_matches)]
 #![feature(try_blocks)]
+#![feature(let_chains)]

 use std::any::Any;
 use std::fmt::{Debug, Formatter};
@@ -40,6 +41,7 @@ pub mod information_schema {
    pub use crate::system_schema::information_schema::*;
 }

+pub mod process_manager;
 pub mod table_source;

 #[async_trait::async_trait]
--- a/src/catalog/src/memory/manager.rs
+++ b/src/catalog/src/memory/manager.rs
@@ -356,6 +356,7 @@ impl MemoryCatalogManager {
            catalog,
            Arc::downgrade(self) as Weak<dyn CatalogManager>,
            Arc::new(FlowMetadataManager::new(Arc::new(MemoryKvBackend::new()))),
+            None, // we don't need ProcessManager on regions server.
        );
        let information_schema = information_schema_provider.tables().clone();

--- a/src/catalog/src/metrics.rs
+++ b/src/catalog/src/metrics.rs
@@ -34,4 +34,20 @@ lazy_static! {
        register_histogram!("greptime_catalog_kv_get", "catalog kv get").unwrap();
    pub static ref METRIC_CATALOG_KV_BATCH_GET: Histogram =
        register_histogram!("greptime_catalog_kv_batch_get", "catalog kv batch get").unwrap();
+
+    /// Count of running process in each catalog.
+    pub static ref PROCESS_LIST_COUNT: IntGaugeVec = register_int_gauge_vec!(
+        "greptime_process_list_count",
+        "Running process count per catalog",
+        &["catalog"]
+    )
+    .unwrap();
+
+    /// Count of killed process in each catalog.
+    pub static ref PROCESS_KILL_COUNT: IntCounterVec = register_int_counter_vec!(
+        "greptime_process_kill_count",
+        "Completed kill process requests count",
+        &["catalog"]
+    )
+    .unwrap();
 }
--- a/src/catalog/src/process_manager.rs
+++ b/src/catalog/src/process_manager.rs
@@ -0,0 +1,494 @@
+// Copyright 2023 Greptime Team
+//
+// Licensed under the Apache License, Version 2.0 (the "License");
+// you may not use this file except in compliance with the License.
+// You may obtain a copy of the License at
+//
+//     http://www.apache.org/licenses/LICENSE-2.0
+//
+// Unless required by applicable law or agreed to in writing, software
+// distributed under the License is distributed on an "AS IS" BASIS,
+// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+// See the License for the specific language governing permissions and
+// limitations under the License.
+
+use std::collections::hash_map::Entry;
+use std::collections::HashMap;
+use std::fmt::{Debug, Formatter};
+use std::sync::atomic::{AtomicU32, Ordering};
+use std::sync::{Arc, RwLock};
+
+use api::v1::frontend::{KillProcessRequest, ListProcessRequest, ProcessInfo};
+use common_base::cancellation::CancellationHandle;
+use common_frontend::selector::{FrontendSelector, MetaClientSelector};
+use common_telemetry::{debug, info};
+use common_time::util::current_time_millis;
+use meta_client::MetaClientRef;
+use snafu::{ensure, OptionExt, ResultExt};
+
+use crate::error;
+use crate::metrics::{PROCESS_KILL_COUNT, PROCESS_LIST_COUNT};
+
+pub type ProcessId = u32;
+pub type ProcessManagerRef = Arc<ProcessManager>;
+
+/// Query process manager.
+pub struct ProcessManager {
+    /// Local frontend server address,
+    server_addr: String,
+    /// Next process id for local queries.
+    next_id: AtomicU32,
+    /// Running process per catalog.
+    catalogs: RwLock<HashMap<String, HashMap<ProcessId, CancellableProcess>>>,
+    /// Frontend selector to locate frontend nodes.
+    frontend_selector: Option<MetaClientSelector>,
+}
+
+impl ProcessManager {
+    /// Create a [ProcessManager] instance with server address and kv client.
+    pub fn new(server_addr: String, meta_client: Option<MetaClientRef>) -> Self {
+        let frontend_selector = meta_client.map(MetaClientSelector::new);
+        Self {
+            server_addr,
+            next_id: Default::default(),
+            catalogs: Default::default(),
+            frontend_selector,
+        }
+    }
+}
+
+impl ProcessManager {
+    /// Registers a submitted query. Use the provided id if present.
+    #[must_use]
+    pub fn register_query(
+        self: &Arc<Self>,
+        catalog: String,
+        schemas: Vec<String>,
+        query: String,
+        client: String,
+        query_id: Option<ProcessId>,
+    ) -> Ticket {
+        let id = query_id.unwrap_or_else(|| self.next_id.fetch_add(1, Ordering::Relaxed));
+        let process = ProcessInfo {
+            id,
+            catalog: catalog.clone(),
+            schemas,
+            query,
+            start_timestamp: current_time_millis(),
+            client,
+            frontend: self.server_addr.clone(),
+        };
+        let cancellation_handle = Arc::new(CancellationHandle::default());
+        let cancellable_process = CancellableProcess::new(cancellation_handle.clone(), process);
+
+        self.catalogs
+            .write()
+            .unwrap()
+            .entry(catalog.clone())
+            .or_default()
+            .insert(id, cancellable_process);
+
+        Ticket {
+            catalog,
+            manager: self.clone(),
+            id,
+            cancellation_handle,
+        }
+    }
+
+    /// Generates the next process id.
+    pub fn next_id(&self) -> u32 {
+        self.next_id.fetch_add(1, Ordering::Relaxed)
+    }
+
+    /// De-register a query from process list.
+    pub fn deregister_query(&self, catalog: String, id: ProcessId) {
+        if let Entry::Occupied(mut o) = self.catalogs.write().unwrap().entry(catalog) {
+            let process = o.get_mut().remove(&id);
+            debug!("Deregister process: {:?}", process);
+            if o.get().is_empty() {
+                o.remove();
+            }
+        }
+    }
+
+    /// List local running processes in given catalog.
+    pub fn local_processes(&self, catalog: Option<&str>) -> error::Result<Vec<ProcessInfo>> {
+        let catalogs = self.catalogs.read().unwrap();
+        let result = if let Some(catalog) = catalog {
+            if let Some(catalogs) = catalogs.get(catalog) {
+                catalogs.values().map(|p| p.process.clone()).collect()
+            } else {
+                vec![]
+            }
+        } else {
+            catalogs
+                .values()
+                .flat_map(|v| v.values().map(|p| p.process.clone()))
+                .collect()
+        };
+        Ok(result)
+    }
+
+    pub async fn list_all_processes(
+        &self,
+        catalog: Option<&str>,
+    ) -> error::Result<Vec<ProcessInfo>> {
+        let mut processes = vec![];
+        if let Some(remote_frontend_selector) = self.frontend_selector.as_ref() {
+            let frontends = remote_frontend_selector
+                .select(|node| node.peer.addr != self.server_addr)
+                .await
+                .context(error::InvokeFrontendSnafu)?;
+            for mut f in frontends {
+                processes.extend(
+                    f.list_process(ListProcessRequest {
+                        catalog: catalog.unwrap_or_default().to_string(),
+                    })
+                    .await
+                    .context(error::InvokeFrontendSnafu)?
+                    .processes,
+                );
+            }
+        }
+        processes.extend(self.local_processes(catalog)?);
+        Ok(processes)
+    }
+
+    /// Kills query with provided catalog and id.
+    pub async fn kill_process(
+        &self,
+        server_addr: String,
+        catalog: String,
+        id: ProcessId,
+    ) -> error::Result<bool> {
+        if server_addr == self.server_addr {
+            self.kill_local_process(catalog, id).await
+        } else {
+            let mut nodes = self
+                .frontend_selector
+                .as_ref()
+                .context(error::MetaClientMissingSnafu)?
+                .select(|node| node.peer.addr == server_addr)
+                .await
+                .context(error::InvokeFrontendSnafu)?;
+            ensure!(
+                !nodes.is_empty(),
+                error::FrontendNotFoundSnafu { addr: server_addr }
+            );
+
+            let request = KillProcessRequest {
+                server_addr,
+                catalog,
+                process_id: id,
+            };
+            nodes[0]
+                .kill_process(request)
+                .await
+                .context(error::InvokeFrontendSnafu)?;
+            Ok(true)
+        }
+    }
+
+    /// Kills local query with provided catalog and id.
+    pub async fn kill_local_process(&self, catalog: String, id: ProcessId) -> error::Result<bool> {
+        if let Some(catalogs) = self.catalogs.write().unwrap().get_mut(&catalog) {
+            if let Some(process) = catalogs.remove(&id) {
+                process.handle.cancel();
+                info!(
+                    "Killed process, catalog: {}, id: {:?}",
+                    process.process.catalog, process.process.id
+                );
+                PROCESS_KILL_COUNT.with_label_values(&[&catalog]).inc();
+                Ok(true)
+            } else {
+                debug!("Failed to kill process, id not found: {}", id);
+                Ok(false)
+            }
+        } else {
+            debug!("Failed to kill process, catalog not found: {}", catalog);
+            Ok(false)
+        }
+    }
+}
+
+pub struct Ticket {
+    pub(crate) catalog: String,
+    pub(crate) manager: ProcessManagerRef,
+    pub(crate) id: ProcessId,
+    pub cancellation_handle: Arc<CancellationHandle>,
+}
+
+impl Drop for Ticket {
+    fn drop(&mut self) {
+        self.manager
+            .deregister_query(std::mem::take(&mut self.catalog), self.id);
+    }
+}
+
+struct CancellableProcess {
+    handle: Arc<CancellationHandle>,
+    process: ProcessInfo,
+}
+
+impl Drop for CancellableProcess {
+    fn drop(&mut self) {
+        PROCESS_LIST_COUNT
+            .with_label_values(&[&self.process.catalog])
+            .dec();
+    }
+}
+
+impl CancellableProcess {
+    fn new(handle: Arc<CancellationHandle>, process: ProcessInfo) -> Self {
+        PROCESS_LIST_COUNT
+            .with_label_values(&[&process.catalog])
+            .inc();
+        Self { handle, process }
+    }
+}
+
+impl Debug for CancellableProcess {
+    fn fmt(&self, f: &mut Formatter<'_>) -> std::fmt::Result {
+        f.debug_struct("CancellableProcess")
+            .field("cancelled", &self.handle.is_cancelled())
+            .field("process", &self.process)
+            .finish()
+    }
+}
+
+#[cfg(test)]
+mod tests {
+    use std::sync::Arc;
+
+    use crate::process_manager::ProcessManager;
+
+    #[tokio::test]
+    async fn test_register_query() {
+        let process_manager = Arc::new(ProcessManager::new("127.0.0.1:8000".to_string(), None));
+        let ticket = process_manager.clone().register_query(
+            "public".to_string(),
+            vec!["test".to_string()],
+            "SELECT * FROM table".to_string(),
+            "".to_string(),
+            None,
+        );
+
+        let running_processes = process_manager.local_processes(None).unwrap();
+        assert_eq!(running_processes.len(), 1);
+        assert_eq!(&running_processes[0].frontend, "127.0.0.1:8000");
+        assert_eq!(running_processes[0].id, ticket.id);
+        assert_eq!(&running_processes[0].query, "SELECT * FROM table");
+
+        drop(ticket);
+        assert_eq!(process_manager.local_processes(None).unwrap().len(), 0);
+    }
+
+    #[tokio::test]
+    async fn test_register_query_with_custom_id() {
+        let process_manager = Arc::new(ProcessManager::new("127.0.0.1:8000".to_string(), None));
+        let custom_id = 12345;
+
+        let ticket = process_manager.clone().register_query(
+            "public".to_string(),
+            vec!["test".to_string()],
+            "SELECT * FROM table".to_string(),
+            "client1".to_string(),
+            Some(custom_id),
+        );
+
+        assert_eq!(ticket.id, custom_id);
+
+        let running_processes = process_manager.local_processes(None).unwrap();
+        assert_eq!(running_processes.len(), 1);
+        assert_eq!(running_processes[0].id, custom_id);
+        assert_eq!(&running_processes[0].client, "client1");
+    }
+
+    #[tokio::test]
+    async fn test_multiple_queries_same_catalog() {
+        let process_manager = Arc::new(ProcessManager::new("127.0.0.1:8000".to_string(), None));
+
+        let ticket1 = process_manager.clone().register_query(
+            "public".to_string(),
+            vec!["schema1".to_string()],
+            "SELECT * FROM table1".to_string(),
+            "client1".to_string(),
+            None,
+        );
+
+        let ticket2 = process_manager.clone().register_query(
+            "public".to_string(),
+            vec!["schema2".to_string()],
+            "SELECT * FROM table2".to_string(),
+            "client2".to_string(),
+            None,
+        );
+
+        let running_processes = process_manager.local_processes(Some("public")).unwrap();
+        assert_eq!(running_processes.len(), 2);
+
+        // Verify both processes are present
+        let ids: Vec<u32> = running_processes.iter().map(|p| p.id).collect();
+        assert!(ids.contains(&ticket1.id));
+        assert!(ids.contains(&ticket2.id));
+    }
+
+    #[tokio::test]
+    async fn test_multiple_catalogs() {
+        let process_manager = Arc::new(ProcessManager::new("127.0.0.1:8000".to_string(), None));
+
+        let _ticket1 = process_manager.clone().register_query(
+            "catalog1".to_string(),
+            vec!["schema1".to_string()],
+            "SELECT * FROM table1".to_string(),
+            "client1".to_string(),
+            None,
+        );
+
+        let _ticket2 = process_manager.clone().register_query(
+            "catalog2".to_string(),
+            vec!["schema2".to_string()],
+            "SELECT * FROM table2".to_string(),
+            "client2".to_string(),
+            None,
+        );
+
+        // Test listing processes for specific catalog
+        let catalog1_processes = process_manager.local_processes(Some("catalog1")).unwrap();
+        assert_eq!(catalog1_processes.len(), 1);
+        assert_eq!(&catalog1_processes[0].catalog, "catalog1");
+
+        let catalog2_processes = process_manager.local_processes(Some("catalog2")).unwrap();
+        assert_eq!(catalog2_processes.len(), 1);
+        assert_eq!(&catalog2_processes[0].catalog, "catalog2");
+
+        // Test listing all processes
+        let all_processes = process_manager.local_processes(None).unwrap();
+        assert_eq!(all_processes.len(), 2);
+    }
+
+    #[tokio::test]
+    async fn test_deregister_query() {
+        let process_manager = Arc::new(ProcessManager::new("127.0.0.1:8000".to_string(), None));
+
+        let ticket = process_manager.clone().register_query(
+            "public".to_string(),
+            vec!["test".to_string()],
+            "SELECT * FROM table".to_string(),
+            "client1".to_string(),
+            None,
+        );
+        assert_eq!(process_manager.local_processes(None).unwrap().len(), 1);
+        process_manager.deregister_query("public".to_string(), ticket.id);
+        assert_eq!(process_manager.local_processes(None).unwrap().len(), 0);
+    }
+
+    #[tokio::test]
+    async fn test_cancellation_handle() {
+        let process_manager = Arc::new(ProcessManager::new("127.0.0.1:8000".to_string(), None));
+
+        let ticket = process_manager.clone().register_query(
+            "public".to_string(),
+            vec!["test".to_string()],
+            "SELECT * FROM table".to_string(),
+            "client1".to_string(),
+            None,
+        );
+
+        assert!(!ticket.cancellation_handle.is_cancelled());
+        ticket.cancellation_handle.cancel();
+        assert!(ticket.cancellation_handle.is_cancelled());
+    }
+
+    #[tokio::test]
+    async fn test_kill_local_process() {
+        let process_manager = Arc::new(ProcessManager::new("127.0.0.1:8000".to_string(), None));
+
+        let ticket = process_manager.clone().register_query(
+            "public".to_string(),
+            vec!["test".to_string()],
+            "SELECT * FROM table".to_string(),
+            "client1".to_string(),
+            None,
+        );
+        assert!(!ticket.cancellation_handle.is_cancelled());
+        let killed = process_manager
+            .kill_process(
+                "127.0.0.1:8000".to_string(),
+                "public".to_string(),
+                ticket.id,
+            )
+            .await
+            .unwrap();
+
+        assert!(killed);
+        assert_eq!(process_manager.local_processes(None).unwrap().len(), 0);
+    }
+
+    #[tokio::test]
+    async fn test_kill_nonexistent_process() {
+        let process_manager = Arc::new(ProcessManager::new("127.0.0.1:8000".to_string(), None));
+        let killed = process_manager
+            .kill_process("127.0.0.1:8000".to_string(), "public".to_string(), 999)
+            .await
+            .unwrap();
+        assert!(!killed);
+    }
+
+    #[tokio::test]
+    async fn test_kill_process_nonexistent_catalog() {
+        let process_manager = Arc::new(ProcessManager::new("127.0.0.1:8000".to_string(), None));
+        let killed = process_manager
+            .kill_process("127.0.0.1:8000".to_string(), "nonexistent".to_string(), 1)
+            .await
+            .unwrap();
+        assert!(!killed);
+    }
+
+    #[tokio::test]
+    async fn test_process_info_fields() {
+        let process_manager = Arc::new(ProcessManager::new("127.0.0.1:8000".to_string(), None));
+
+        let _ticket = process_manager.clone().register_query(
+            "test_catalog".to_string(),
+            vec!["schema1".to_string(), "schema2".to_string()],
+            "SELECT COUNT(*) FROM users WHERE age > 18".to_string(),
+            "test_client".to_string(),
+            Some(42),
+        );
+
+        let processes = process_manager.local_processes(None).unwrap();
+        assert_eq!(processes.len(), 1);
+
+        let process = &processes[0];
+        assert_eq!(process.id, 42);
+        assert_eq!(&process.catalog, "test_catalog");
+        assert_eq!(process.schemas, vec!["schema1", "schema2"]);
+        assert_eq!(&process.query, "SELECT COUNT(*) FROM users WHERE age > 18");
+        assert_eq!(&process.client, "test_client");
+        assert_eq!(&process.frontend, "127.0.0.1:8000");
+        assert!(process.start_timestamp > 0);
+    }
+
+    #[tokio::test]
+    async fn test_ticket_drop_deregisters_process() {
+        let process_manager = Arc::new(ProcessManager::new("127.0.0.1:8000".to_string(), None));
+
+        {
+            let _ticket = process_manager.clone().register_query(
+                "public".to_string(),
+                vec!["test".to_string()],
+                "SELECT * FROM table".to_string(),
+                "client1".to_string(),
+                None,
+            );
+
+            // Process should be registered
+            assert_eq!(process_manager.local_processes(None).unwrap().len(), 1);
+        } // ticket goes out of scope here
+
+        // Process should be automatically deregistered
+        assert_eq!(process_manager.local_processes(None).unwrap().len(), 0);
+    }
+}
--- a/src/catalog/src/system_schema/information_schema.rs
+++ b/src/catalog/src/system_schema/information_schema.rs
@@ -19,6 +19,7 @@ mod information_memory_table;
 pub mod key_column_usage;
 mod partitions;
 mod procedure_info;
+pub mod process_list;
 pub mod region_peers;
 mod region_statistics;
 mod runtime_metrics;
@@ -42,6 +43,7 @@ use common_recordbatch::SendableRecordBatchStream;
 use datatypes::schema::SchemaRef;
 use lazy_static::lazy_static;
 use paste::paste;
+use process_list::InformationSchemaProcessList;
 use store_api::storage::{ScanRequest, TableId};
 use table::metadata::TableType;
 use table::TableRef;
@@ -50,6 +52,7 @@ use views::InformationSchemaViews;

 use self::columns::InformationSchemaColumns;
 use crate::error::{Error, Result};
+use crate::process_manager::ProcessManagerRef;
 use crate::system_schema::information_schema::cluster_info::InformationSchemaClusterInfo;
 use crate::system_schema::information_schema::flows::InformationSchemaFlows;
 use crate::system_schema::information_schema::information_memory_table::get_schema_columns;
@@ -113,6 +116,7 @@ macro_rules! setup_memory_table {
 pub struct InformationSchemaProvider {
    catalog_name: String,
    catalog_manager: Weak<dyn CatalogManager>,
+    process_manager: Option<ProcessManagerRef>,
    flow_metadata_manager: Arc<FlowMetadataManager>,
    tables: HashMap<String, TableRef>,
 }
@@ -207,6 +211,10 @@ impl SystemSchemaProviderInner for InformationSchemaProvider {
                    self.catalog_manager.clone(),
                ),
            ) as _),
+            PROCESS_LIST => self
+                .process_manager
+                .as_ref()
+                .map(|p| Arc::new(InformationSchemaProcessList::new(p.clone())) as _),
            _ => None,
        }
    }
@@ -217,11 +225,13 @@ impl InformationSchemaProvider {
        catalog_name: String,
        catalog_manager: Weak<dyn CatalogManager>,
        flow_metadata_manager: Arc<FlowMetadataManager>,
+        process_manager: Option<ProcessManagerRef>,
    ) -> Self {
        let mut provider = Self {
            catalog_name,
            catalog_manager,
            flow_metadata_manager,
+            process_manager,
            tables: HashMap::new(),
        };

@@ -277,6 +287,9 @@ impl InformationSchemaProvider {
            self.build_table(TABLE_CONSTRAINTS).unwrap(),
        );
        tables.insert(FLOWS.to_string(), self.build_table(FLOWS).unwrap());
+        if let Some(process_list) = self.build_table(PROCESS_LIST) {
+            tables.insert(PROCESS_LIST.to_string(), process_list);
+        }
        // Add memory tables
        for name in MEMORY_TABLES.iter() {
            tables.insert((*name).to_string(), self.build_table(name).expect(name));
--- a/src/catalog/src/system_schema/information_schema/process_list.rs
+++ b/src/catalog/src/system_schema/information_schema/process_list.rs
@@ -0,0 +1,189 @@
+// Copyright 2023 Greptime Team
+//
+// Licensed under the Apache License, Version 2.0 (the "License");
+// you may not use this file except in compliance with the License.
+// You may obtain a copy of the License at
+//
+//     http://www.apache.org/licenses/LICENSE-2.0
+//
+// Unless required by applicable law or agreed to in writing, software
+// distributed under the License is distributed on an "AS IS" BASIS,
+// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+// See the License for the specific language governing permissions and
+// limitations under the License.
+
+use std::sync::Arc;
+
+use common_catalog::consts::INFORMATION_SCHEMA_PROCESS_LIST_TABLE_ID;
+use common_error::ext::BoxedError;
+use common_frontend::DisplayProcessId;
+use common_recordbatch::adapter::RecordBatchStreamAdapter;
+use common_recordbatch::{RecordBatch, SendableRecordBatchStream};
+use common_time::util::current_time_millis;
+use common_time::{Duration, Timestamp};
+use datafusion::physical_plan::stream::RecordBatchStreamAdapter as DfRecordBatchStreamAdapter;
+use datatypes::prelude::ConcreteDataType as CDT;
+use datatypes::scalars::ScalarVectorBuilder;
+use datatypes::schema::{ColumnSchema, Schema, SchemaRef};
+use datatypes::value::Value;
+use datatypes::vectors::{
+    DurationMillisecondVectorBuilder, StringVectorBuilder, TimestampMillisecondVectorBuilder,
+    VectorRef,
+};
+use snafu::ResultExt;
+use store_api::storage::{ScanRequest, TableId};
+
+use crate::error::{self, InternalSnafu};
+use crate::information_schema::Predicates;
+use crate::process_manager::ProcessManagerRef;
+use crate::system_schema::information_schema::InformationTable;
+
+/// Column names of `information_schema.process_list`
+pub const ID: &str = "id";
+pub const CATALOG: &str = "catalog";
+pub const SCHEMAS: &str = "schemas";
+pub const QUERY: &str = "query";
+pub const CLIENT: &str = "client";
+pub const FRONTEND: &str = "frontend";
+pub const START_TIMESTAMP: &str = "start_timestamp";
+pub const ELAPSED_TIME: &str = "elapsed_time";
+
+/// `information_schema.process_list` table implementation that tracks running
+/// queries in current cluster.
+pub struct InformationSchemaProcessList {
+    schema: SchemaRef,
+    process_manager: ProcessManagerRef,
+}
+
+impl InformationSchemaProcessList {
+    pub fn new(process_manager: ProcessManagerRef) -> Self {
+        Self {
+            schema: Self::schema(),
+            process_manager,
+        }
+    }
+
+    fn schema() -> SchemaRef {
+        Arc::new(Schema::new(vec![
+            ColumnSchema::new(ID, CDT::string_datatype(), false),
+            ColumnSchema::new(CATALOG, CDT::string_datatype(), false),
+            ColumnSchema::new(SCHEMAS, CDT::string_datatype(), false),
+            ColumnSchema::new(QUERY, CDT::string_datatype(), false),
+            ColumnSchema::new(CLIENT, CDT::string_datatype(), false),
+            ColumnSchema::new(FRONTEND, CDT::string_datatype(), false),
+            ColumnSchema::new(
+                START_TIMESTAMP,
+                CDT::timestamp_millisecond_datatype(),
+                false,
+            ),
+            ColumnSchema::new(ELAPSED_TIME, CDT::duration_millisecond_datatype(), false),
+        ]))
+    }
+}
+
+impl InformationTable for InformationSchemaProcessList {
+    fn table_id(&self) -> TableId {
+        INFORMATION_SCHEMA_PROCESS_LIST_TABLE_ID
+    }
+
+    fn table_name(&self) -> &'static str {
+        "process_list"
+    }
+
+    fn schema(&self) -> SchemaRef {
+        self.schema.clone()
+    }
+
+    fn to_stream(&self, request: ScanRequest) -> error::Result<SendableRecordBatchStream> {
+        let process_manager = self.process_manager.clone();
+        let stream = Box::pin(DfRecordBatchStreamAdapter::new(
+            self.schema.arrow_schema().clone(),
+            futures::stream::once(async move {
+                make_process_list(process_manager, request)
+                    .await
+                    .map(RecordBatch::into_df_record_batch)
+                    .map_err(|e| datafusion::error::DataFusionError::External(Box::new(e)))
+            }),
+        ));
+
+        Ok(Box::pin(
+            RecordBatchStreamAdapter::try_new(stream)
+                .map_err(BoxedError::new)
+                .context(InternalSnafu)?,
+        ))
+    }
+}
+
+/// Build running process list.
+async fn make_process_list(
+    process_manager: ProcessManagerRef,
+    request: ScanRequest,
+) -> error::Result<RecordBatch> {
+    let predicates = Predicates::from_scan_request(&Some(request));
+    let current_time = current_time_millis();
+    // todo(hl): find a way to extract user catalog to filter queries from other users.
+    let queries = process_manager.list_all_processes(None).await?;
+
+    let mut id_builder = StringVectorBuilder::with_capacity(queries.len());
+    let mut catalog_builder = StringVectorBuilder::with_capacity(queries.len());
+    let mut schemas_builder = StringVectorBuilder::with_capacity(queries.len());
+    let mut query_builder = StringVectorBuilder::with_capacity(queries.len());
+    let mut client_builder = StringVectorBuilder::with_capacity(queries.len());
+    let mut frontend_builder = StringVectorBuilder::with_capacity(queries.len());
+    let mut start_time_builder = TimestampMillisecondVectorBuilder::with_capacity(queries.len());
+    let mut elapsed_time_builder = DurationMillisecondVectorBuilder::with_capacity(queries.len());
+
+    for process in queries {
+        let display_id = DisplayProcessId {
+            server_addr: process.frontend.to_string(),
+            id: process.id,
+        }
+        .to_string();
+        let schemas = process.schemas.join(",");
+        let id = Value::from(display_id);
+        let catalog = Value::from(process.catalog);
+        let schemas = Value::from(schemas);
+        let query = Value::from(process.query);
+        let client = Value::from(process.client);
+        let frontend = Value::from(process.frontend);
+        let start_timestamp = Value::from(Timestamp::new_millisecond(process.start_timestamp));
+        let elapsed_time = Value::from(Duration::new_millisecond(
+            current_time - process.start_timestamp,
+        ));
+        let row = [
+            (ID, &id),
+            (CATALOG, &catalog),
+            (SCHEMAS, &schemas),
+            (QUERY, &query),
+            (CLIENT, &client),
+            (FRONTEND, &frontend),
+            (START_TIMESTAMP, &start_timestamp),
+            (ELAPSED_TIME, &elapsed_time),
+        ];
+        if predicates.eval(&row) {
+            id_builder.push(id.as_string().as_deref());
+            catalog_builder.push(catalog.as_string().as_deref());
+            schemas_builder.push(schemas.as_string().as_deref());
+            query_builder.push(query.as_string().as_deref());
+            client_builder.push(client.as_string().as_deref());
+            frontend_builder.push(frontend.as_string().as_deref());
+            start_time_builder.push(start_timestamp.as_timestamp().map(|t| t.value().into()));
+            elapsed_time_builder.push(elapsed_time.as_duration().map(|d| d.value().into()));
+        }
+    }
+
+    RecordBatch::new(
+        InformationSchemaProcessList::schema(),
+        vec![
+            Arc::new(id_builder.finish()) as VectorRef,
+            Arc::new(catalog_builder.finish()) as VectorRef,
+            Arc::new(schemas_builder.finish()) as VectorRef,
+            Arc::new(query_builder.finish()) as VectorRef,
+            Arc::new(client_builder.finish()) as VectorRef,
+            Arc::new(frontend_builder.finish()) as VectorRef,
+            Arc::new(start_time_builder.finish()) as VectorRef,
+            Arc::new(elapsed_time_builder.finish()) as VectorRef,
+        ],
+    )
+    .context(error::CreateRecordBatchSnafu)
+}
--- a/src/catalog/src/system_schema/information_schema/table_names.rs
+++ b/src/catalog/src/system_schema/information_schema/table_names.rs
@@ -47,3 +47,4 @@ pub const VIEWS: &str = "views";
 pub const FLOWS: &str = "flows";
 pub const PROCEDURE_INFO: &str = "procedure_info";
 pub const REGION_STATISTICS: &str = "region_statistics";
+pub const PROCESS_LIST: &str = "process_list";
--- a/src/catalog/src/table_source.rs
+++ b/src/catalog/src/table_source.rs
@@ -328,6 +328,7 @@ mod tests {
            backend.clone(),
            layered_cache_registry,
            None,
+            None,
        );
        let table_metadata_manager = TableMetadataManager::new(backend);
        let mut view_info = common_meta::key::test_utils::new_test_table_info(1024, vec![]);
--- a/src/cli/Cargo.toml
+++ b/src/cli/Cargo.toml
@@ -16,6 +16,7 @@ mysql_kvbackend = ["common-meta/mysql_kvbackend", "meta-srv/mysql_kvbackend"]
 workspace = true

 [dependencies]
+async-stream.workspace = true
 async-trait.workspace = true
 auth.workspace = true
 base64.workspace = true
@@ -50,6 +51,7 @@ meta-client.workspace = true
 meta-srv.workspace = true
 nu-ansi-term = "0.46"
 object-store.workspace = true
+operator.workspace = true
 query.workspace = true
 rand.workspace = true
 reqwest.workspace = true
@@ -65,6 +67,7 @@ tokio.workspace = true
 tracing-appender.workspace = true

 [dev-dependencies]
+common-meta = { workspace = true, features = ["testing"] }
 common-version.workspace = true
 serde.workspace = true
 tempfile.workspace = true
--- a/src/cli/src/data.rs
+++ b/src/cli/src/data.rs
@@ -0,0 +1,39 @@
+// Copyright 2023 Greptime Team
+//
+// Licensed under the Apache License, Version 2.0 (the "License");
+// you may not use this file except in compliance with the License.
+// You may obtain a copy of the License at
+//
+//     http://www.apache.org/licenses/LICENSE-2.0
+//
+// Unless required by applicable law or agreed to in writing, software
+// distributed under the License is distributed on an "AS IS" BASIS,
+// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+// See the License for the specific language governing permissions and
+// limitations under the License.
+
+mod export;
+mod import;
+
+use clap::Subcommand;
+use common_error::ext::BoxedError;
+
+use crate::data::export::ExportCommand;
+use crate::data::import::ImportCommand;
+use crate::Tool;
+
+/// Command for data operations including exporting data from and importing data into GreptimeDB.
+#[derive(Subcommand)]
+pub enum DataCommand {
+    Export(ExportCommand),
+    Import(ImportCommand),
+}
+
+impl DataCommand {
+    pub async fn build(&self) -> std::result::Result<Box<dyn Tool>, BoxedError> {
+        match self {
+            DataCommand::Export(cmd) => cmd.build().await,
+            DataCommand::Import(cmd) => cmd.build().await,
+        }
+    }
+}
--- a/src/cli/src/data/export.rs
+++ b/src/cli/src/data/export.rs
--- a/src/cli/src/data/import.rs
+++ b/src/cli/src/data/import.rs
--- a/src/cli/src/error.rs
+++ b/src/cli/src/error.rs
@@ -17,8 +17,10 @@ use std::any::Any;
 use common_error::ext::{BoxedError, ErrorExt};
 use common_error::status_code::StatusCode;
 use common_macro::stack_trace_debug;
+use common_meta::peer::Peer;
 use object_store::Error as ObjectStoreError;
 use snafu::{Location, Snafu};
+use store_api::storage::TableId;

 #[derive(Snafu)]
 #[snafu(visibility(pub))]
@@ -30,6 +32,7 @@ pub enum Error {
        location: Location,
        msg: String,
    },
+
    #[snafu(display("Failed to create default catalog and schema"))]
    InitMetadata {
        #[snafu(implicit)]
@@ -72,6 +75,20 @@ pub enum Error {
        source: common_meta::error::Error,
    },

+    #[snafu(display("Failed to get table metadata"))]
+    TableMetadata {
+        #[snafu(implicit)]
+        location: Location,
+        source: common_meta::error::Error,
+    },
+
+    #[snafu(display("Unexpected error: {}", msg))]
+    Unexpected {
+        msg: String,
+        #[snafu(implicit)]
+        location: Location,
+    },
+
    #[snafu(display("Missing config, msg: {}", msg))]
    MissingConfig {
        msg: String,
@@ -221,6 +238,13 @@ pub enum Error {
        location: Location,
    },

+    #[snafu(display("Table not found: {table_id}"))]
+    TableNotFound {
+        table_id: TableId,
+        #[snafu(implicit)]
+        location: Location,
+    },
+
    #[snafu(display("OpenDAL operator failed"))]
    OpenDal {
        #[snafu(implicit)]
@@ -228,22 +252,25 @@ pub enum Error {
        #[snafu(source)]
        error: ObjectStoreError,
    },
+
    #[snafu(display("S3 config need be set"))]
    S3ConfigNotSet {
        #[snafu(implicit)]
        location: Location,
    },
+
    #[snafu(display("Output directory not set"))]
    OutputDirNotSet {
        #[snafu(implicit)]
        location: Location,
    },
-    #[snafu(display("KV backend not set: {}", backend))]
-    KvBackendNotSet {
-        backend: String,
+
+    #[snafu(display("Empty store addresses"))]
+    EmptyStoreAddrs {
        #[snafu(implicit)]
        location: Location,
    },
+
    #[snafu(display("Unsupported memory backend"))]
    UnsupportedMemoryBackend {
        #[snafu(implicit)]
@@ -256,6 +283,36 @@ pub enum Error {
        #[snafu(implicit)]
        location: Location,
    },
+
+    #[snafu(display("Invalid arguments: {}", msg))]
+    InvalidArguments {
+        msg: String,
+        #[snafu(implicit)]
+        location: Location,
+    },
+
+    #[snafu(display("Failed to init backend"))]
+    InitBackend {
+        #[snafu(implicit)]
+        location: Location,
+        #[snafu(source)]
+        error: ObjectStoreError,
+    },
+
+    #[snafu(display("Covert column schemas to defs failed"))]
+    CovertColumnSchemasToDefs {
+        #[snafu(implicit)]
+        location: Location,
+        source: operator::error::Error,
+    },
+
+    #[snafu(display("Failed to send request to datanode: {}", peer))]
+    SendRequestToDatanode {
+        peer: Peer,
+        #[snafu(implicit)]
+        location: Location,
+        source: common_meta::error::Error,
+    },
 }

 pub type Result<T> = std::result::Result<T, Error>;
@@ -263,9 +320,9 @@ pub type Result<T> = std::result::Result<T, Error>;
 impl ErrorExt for Error {
    fn status_code(&self) -> StatusCode {
        match self {
-            Error::InitMetadata { source, .. } | Error::InitDdlManager { source, .. } => {
-                source.status_code()
-            }
+            Error::InitMetadata { source, .. }
+            | Error::InitDdlManager { source, .. }
+            | Error::TableMetadata { source, .. } => source.status_code(),

            Error::MissingConfig { .. }
            | Error::LoadLayeredConfig { .. }
@@ -276,8 +333,12 @@ impl ErrorExt for Error {
            | Error::EmptyResult { .. }
            | Error::InvalidFilePath { .. }
            | Error::UnsupportedMemoryBackend { .. }
+            | Error::InvalidArguments { .. }
            | Error::ParseProxyOpts { .. } => StatusCode::InvalidArguments,

+            Error::CovertColumnSchemasToDefs { source, .. } => source.status_code(),
+            Error::SendRequestToDatanode { source, .. } => source.status_code(),
+
            Error::StartProcedureManager { source, .. }
            | Error::StopProcedureManager { source, .. } => source.status_code(),
            Error::StartWalOptionsAllocator { source, .. } => source.status_code(),
@@ -285,6 +346,7 @@ impl ErrorExt for Error {
            Error::ParseSql { source, .. } | Error::PlanStatement { source, .. } => {
                source.status_code()
            }
+            Error::Unexpected { .. } => StatusCode::Unexpected,

            Error::SerdeJson { .. }
            | Error::FileIo { .. }
@@ -293,15 +355,16 @@ impl ErrorExt for Error {
            | Error::BuildClient { .. } => StatusCode::Unexpected,

            Error::Other { source, .. } => source.status_code(),
-            Error::OpenDal { .. } => StatusCode::Internal,
+            Error::OpenDal { .. } | Error::InitBackend { .. } => StatusCode::Internal,
            Error::S3ConfigNotSet { .. }
            | Error::OutputDirNotSet { .. }
-            | Error::KvBackendNotSet { .. } => StatusCode::InvalidArguments,
+            | Error::EmptyStoreAddrs { .. } => StatusCode::InvalidArguments,

            Error::BuildRuntime { source, .. } => source.status_code(),

            Error::CacheRequired { .. } | Error::BuildCacheRegistry { .. } => StatusCode::Internal,
            Error::MetaClientInit { source, .. } => source.status_code(),
+            Error::TableNotFound { .. } => StatusCode::TableNotFound,
            Error::SchemaNotFound { .. } => StatusCode::DatabaseNotFound,
        }
    }
--- a/src/cli/src/lib.rs
+++ b/src/cli/src/lib.rs
@@ -13,22 +13,20 @@
 // limitations under the License.

 mod bench;
+mod data;
 mod database;
 pub mod error;
-mod export;
-mod import;
-mod meta_snapshot;
+mod metadata;

 use async_trait::async_trait;
-use clap::{Parser, Subcommand};
+use clap::Parser;
 use common_error::ext::BoxedError;
 pub use database::DatabaseClient;
 use error::Result;

 pub use crate::bench::BenchTableMetadataCommand;
-pub use crate::export::ExportCommand;
-pub use crate::import::ImportCommand;
-pub use crate::meta_snapshot::{MetaCommand, MetaInfoCommand, MetaRestoreCommand, MetaSaveCommand};
+pub use crate::data::DataCommand;
+pub use crate::metadata::MetadataCommand;

 #[async_trait]
 pub trait Tool: Send + Sync {
@@ -51,19 +49,3 @@ impl AttachCommand {
        unimplemented!("Wait for https://github.com/GreptimeTeam/greptimedb/issues/2373")
    }
 }
-
-/// Subcommand for data operations like export and import.
-#[derive(Subcommand)]
-pub enum DataCommand {
-    Export(ExportCommand),
-    Import(ImportCommand),
-}
-
-impl DataCommand {
-    pub async fn build(&self) -> std::result::Result<Box<dyn Tool>, BoxedError> {
-        match self {
-            DataCommand::Export(cmd) => cmd.build().await,
-            DataCommand::Import(cmd) => cmd.build().await,
-        }
-    }
-}
--- a/src/cli/src/metadata.rs
+++ b/src/cli/src/metadata.rs
@@ -0,0 +1,52 @@
+// Copyright 2023 Greptime Team
+//
+// Licensed under the Apache License, Version 2.0 (the "License");
+// you may not use this file except in compliance with the License.
+// You may obtain a copy of the License at
+//
+//     http://www.apache.org/licenses/LICENSE-2.0
+//
+// Unless required by applicable law or agreed to in writing, software
+// distributed under the License is distributed on an "AS IS" BASIS,
+// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+// See the License for the specific language governing permissions and
+// limitations under the License.
+
+mod common;
+mod control;
+mod repair;
+mod snapshot;
+mod utils;
+
+use clap::Subcommand;
+use common_error::ext::BoxedError;
+
+use crate::metadata::control::{DelCommand, GetCommand};
+use crate::metadata::repair::RepairLogicalTablesCommand;
+use crate::metadata::snapshot::SnapshotCommand;
+use crate::Tool;
+
+/// Command for managing metadata operations,
+/// including saving and restoring metadata snapshots,
+/// controlling metadata operations, and diagnosing and repairing metadata.
+#[derive(Subcommand)]
+pub enum MetadataCommand {
+    #[clap(subcommand)]
+    Snapshot(SnapshotCommand),
+    #[clap(subcommand)]
+    Get(GetCommand),
+    #[clap(subcommand)]
+    Del(DelCommand),
+    RepairLogicalTables(RepairLogicalTablesCommand),
+}
+
+impl MetadataCommand {
+    pub async fn build(&self) -> Result<Box<dyn Tool>, BoxedError> {
+        match self {
+            MetadataCommand::Snapshot(cmd) => cmd.build().await,
+            MetadataCommand::RepairLogicalTables(cmd) => cmd.build().await,
+            MetadataCommand::Get(cmd) => cmd.build().await,
+            MetadataCommand::Del(cmd) => cmd.build().await,
+        }
+    }
+}
--- a/src/cli/src/metadata/common.rs
+++ b/src/cli/src/metadata/common.rs
@@ -0,0 +1,116 @@
+// Copyright 2023 Greptime Team
+//
+// Licensed under the Apache License, Version 2.0 (the "License");
+// you may not use this file except in compliance with the License.
+// You may obtain a copy of the License at
+//
+//     http://www.apache.org/licenses/LICENSE-2.0
+//
+// Unless required by applicable law or agreed to in writing, software
+// distributed under the License is distributed on an "AS IS" BASIS,
+// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+// See the License for the specific language governing permissions and
+// limitations under the License.
+
+use std::sync::Arc;
+
+use clap::Parser;
+use common_error::ext::BoxedError;
+use common_meta::kv_backend::chroot::ChrootKvBackend;
+use common_meta::kv_backend::etcd::EtcdStore;
+use common_meta::kv_backend::KvBackendRef;
+use meta_srv::bootstrap::create_etcd_client;
+use meta_srv::metasrv::BackendImpl;
+
+use crate::error::{EmptyStoreAddrsSnafu, UnsupportedMemoryBackendSnafu};
+
+#[derive(Debug, Default, Parser)]
+pub(crate) struct StoreConfig {
+    /// The endpoint of store. one of etcd, postgres or mysql.
+    ///
+    /// For postgres store, the format is:
+    /// "password=password dbname=postgres user=postgres host=localhost port=5432"
+    ///
+    /// For etcd store, the format is:
+    /// "127.0.0.1:2379"
+    ///
+    /// For mysql store, the format is:
+    /// "mysql://user:password@ip:port/dbname"
+    #[clap(long, alias = "store-addr", value_delimiter = ',', num_args = 1..)]
+    store_addrs: Vec<String>,
+
+    /// The maximum number of operations in a transaction. Only used when using [etcd-store].
+    #[clap(long, default_value = "128")]
+    max_txn_ops: usize,
+
+    /// The metadata store backend.
+    #[clap(long, value_enum, default_value = "etcd-store")]
+    backend: BackendImpl,
+
+    /// The key prefix of the metadata store.
+    #[clap(long, default_value = "")]
+    store_key_prefix: String,
+
+    /// The table name in RDS to store metadata. Only used when using [postgres-store] or [mysql-store].
+    #[cfg(any(feature = "pg_kvbackend", feature = "mysql_kvbackend"))]
+    #[clap(long, default_value = common_meta::kv_backend::DEFAULT_META_TABLE_NAME)]
+    meta_table_name: String,
+}
+
+impl StoreConfig {
+    /// Builds a [`KvBackendRef`] from the store configuration.
+    pub async fn build(&self) -> Result<KvBackendRef, BoxedError> {
+        let max_txn_ops = self.max_txn_ops;
+        let store_addrs = &self.store_addrs;
+        if store_addrs.is_empty() {
+            EmptyStoreAddrsSnafu.fail().map_err(BoxedError::new)
+        } else {
+            let kvbackend = match self.backend {
+                BackendImpl::EtcdStore => {
+                    let etcd_client = create_etcd_client(store_addrs)
+                        .await
+                        .map_err(BoxedError::new)?;
+                    Ok(EtcdStore::with_etcd_client(etcd_client, max_txn_ops))
+                }
+                #[cfg(feature = "pg_kvbackend")]
+                BackendImpl::PostgresStore => {
+                    let table_name = &self.meta_table_name;
+                    let pool = meta_srv::bootstrap::create_postgres_pool(store_addrs)
+                        .await
+                        .map_err(BoxedError::new)?;
+                    Ok(common_meta::kv_backend::rds::PgStore::with_pg_pool(
+                        pool,
+                        table_name,
+                        max_txn_ops,
+                    )
+                    .await
+                    .map_err(BoxedError::new)?)
+                }
+                #[cfg(feature = "mysql_kvbackend")]
+                BackendImpl::MysqlStore => {
+                    let table_name = &self.meta_table_name;
+                    let pool = meta_srv::bootstrap::create_mysql_pool(store_addrs)
+                        .await
+                        .map_err(BoxedError::new)?;
+                    Ok(common_meta::kv_backend::rds::MySqlStore::with_mysql_pool(
+                        pool,
+                        table_name,
+                        max_txn_ops,
+                    )
+                    .await
+                    .map_err(BoxedError::new)?)
+                }
+                BackendImpl::MemoryStore => UnsupportedMemoryBackendSnafu
+                    .fail()
+                    .map_err(BoxedError::new),
+            };
+            if self.store_key_prefix.is_empty() {
+                kvbackend
+            } else {
+                let chroot_kvbackend =
+                    ChrootKvBackend::new(self.store_key_prefix.as_bytes().to_vec(), kvbackend?);
+                Ok(Arc::new(chroot_kvbackend))
+            }
+        }
+    }
+}
--- a/src/cli/src/metadata/control.rs
+++ b/src/cli/src/metadata/control.rs
@@ -0,0 +1,22 @@
+// Copyright 2023 Greptime Team
+//
+// Licensed under the Apache License, Version 2.0 (the "License");
+// you may not use this file except in compliance with the License.
+// You may obtain a copy of the License at
+//
+//     http://www.apache.org/licenses/LICENSE-2.0
+//
+// Unless required by applicable law or agreed to in writing, software
+// distributed under the License is distributed on an "AS IS" BASIS,
+// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+// See the License for the specific language governing permissions and
+// limitations under the License.
+
+mod del;
+mod get;
+#[cfg(test)]
+mod test_utils;
+mod utils;
+
+pub(crate) use del::DelCommand;
+pub(crate) use get::GetCommand;
--- a/src/cli/src/metadata/control/del.rs
+++ b/src/cli/src/metadata/control/del.rs
@@ -0,0 +1,42 @@
+// Copyright 2023 Greptime Team
+//
+// Licensed under the Apache License, Version 2.0 (the "License");
+// you may not use this file except in compliance with the License.
+// You may obtain a copy of the License at
+//
+//     http://www.apache.org/licenses/LICENSE-2.0
+//
+// Unless required by applicable law or agreed to in writing, software
+// distributed under the License is distributed on an "AS IS" BASIS,
+// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+// See the License for the specific language governing permissions and
+// limitations under the License.
+
+mod key;
+mod table;
+
+use clap::Subcommand;
+use common_error::ext::BoxedError;
+
+use crate::metadata::control::del::key::DelKeyCommand;
+use crate::metadata::control::del::table::DelTableCommand;
+use crate::Tool;
+
+/// The prefix of the tombstone keys.
+pub(crate) const CLI_TOMBSTONE_PREFIX: &str = "__cli_tombstone/";
+
+/// Subcommand for deleting metadata from the metadata store.
+#[derive(Subcommand)]
+pub enum DelCommand {
+    Key(DelKeyCommand),
+    Table(DelTableCommand),
+}
+
+impl DelCommand {
+    pub async fn build(&self) -> Result<Box<dyn Tool>, BoxedError> {
+        match self {
+            DelCommand::Key(cmd) => cmd.build().await,
+            DelCommand::Table(cmd) => cmd.build().await,
+        }
+    }
+}
--- a/src/cli/src/metadata/control/del/key.rs
+++ b/src/cli/src/metadata/control/del/key.rs
@@ -0,0 +1,132 @@
+// Copyright 2023 Greptime Team
+//
+// Licensed under the Apache License, Version 2.0 (the "License");
+// you may not use this file except in compliance with the License.
+// You may obtain a copy of the License at
+//
+//     http://www.apache.org/licenses/LICENSE-2.0
+//
+// Unless required by applicable law or agreed to in writing, software
+// distributed under the License is distributed on an "AS IS" BASIS,
+// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+// See the License for the specific language governing permissions and
+// limitations under the License.
+
+use async_trait::async_trait;
+use clap::Parser;
+use common_error::ext::BoxedError;
+use common_meta::key::tombstone::TombstoneManager;
+use common_meta::kv_backend::KvBackendRef;
+use common_meta::rpc::store::RangeRequest;
+
+use crate::metadata::common::StoreConfig;
+use crate::metadata::control::del::CLI_TOMBSTONE_PREFIX;
+use crate::Tool;
+
+/// Delete key-value pairs logically from the metadata store.
+#[derive(Debug, Default, Parser)]
+pub struct DelKeyCommand {
+    /// The key to delete from the metadata store.
+    key: String,
+
+    /// Delete key-value pairs with the given prefix.
+    #[clap(long)]
+    prefix: bool,
+
+    #[clap(flatten)]
+    store: StoreConfig,
+}
+
+impl DelKeyCommand {
+    pub async fn build(&self) -> Result<Box<dyn Tool>, BoxedError> {
+        let kv_backend = self.store.build().await?;
+        Ok(Box::new(DelKeyTool {
+            key: self.key.to_string(),
+            prefix: self.prefix,
+            key_deleter: KeyDeleter::new(kv_backend),
+        }))
+    }
+}
+
+struct KeyDeleter {
+    kv_backend: KvBackendRef,
+    tombstone_manager: TombstoneManager,
+}
+
+impl KeyDeleter {
+    fn new(kv_backend: KvBackendRef) -> Self {
+        Self {
+            kv_backend: kv_backend.clone(),
+            tombstone_manager: TombstoneManager::new_with_prefix(kv_backend, CLI_TOMBSTONE_PREFIX),
+        }
+    }
+
+    async fn delete(&self, key: &str, prefix: bool) -> Result<usize, BoxedError> {
+        let mut req = RangeRequest::default().with_keys_only();
+        if prefix {
+            req = req.with_prefix(key.as_bytes());
+        } else {
+            req = req.with_key(key.as_bytes());
+        }
+        let resp = self.kv_backend.range(req).await.map_err(BoxedError::new)?;
+        let keys = resp.kvs.iter().map(|kv| kv.key.clone()).collect::<Vec<_>>();
+        self.tombstone_manager
+            .create(keys)
+            .await
+            .map_err(BoxedError::new)
+    }
+}
+
+struct DelKeyTool {
+    key: String,
+    prefix: bool,
+    key_deleter: KeyDeleter,
+}
+
+#[async_trait]
+impl Tool for DelKeyTool {
+    async fn do_work(&self) -> Result<(), BoxedError> {
+        let deleted = self.key_deleter.delete(&self.key, self.prefix).await?;
+        // Print the number of deleted keys.
+        println!("{}", deleted);
+        Ok(())
+    }
+}
+
+#[cfg(test)]
+mod tests {
+    use std::sync::Arc;
+
+    use common_meta::kv_backend::chroot::ChrootKvBackend;
+    use common_meta::kv_backend::memory::MemoryKvBackend;
+    use common_meta::kv_backend::{KvBackend, KvBackendRef};
+    use common_meta::rpc::store::RangeRequest;
+
+    use crate::metadata::control::del::key::KeyDeleter;
+    use crate::metadata::control::del::CLI_TOMBSTONE_PREFIX;
+    use crate::metadata::control::test_utils::put_key;
+
+    #[tokio::test]
+    async fn test_delete_keys() {
+        let kv_backend = Arc::new(MemoryKvBackend::new()) as KvBackendRef;
+        let key_deleter = KeyDeleter::new(kv_backend.clone());
+        put_key(&kv_backend, "foo", "bar").await;
+        put_key(&kv_backend, "foo/bar", "baz").await;
+        put_key(&kv_backend, "foo/baz", "qux").await;
+        let deleted = key_deleter.delete("foo", true).await.unwrap();
+        assert_eq!(deleted, 3);
+        let deleted = key_deleter.delete("foo/bar", false).await.unwrap();
+        assert_eq!(deleted, 0);
+
+        let chroot = ChrootKvBackend::new(CLI_TOMBSTONE_PREFIX.as_bytes().to_vec(), kv_backend);
+        let req = RangeRequest::default().with_prefix(b"foo");
+        let resp = chroot.range(req).await.unwrap();
+        assert_eq!(resp.kvs.len(), 3);
+        assert_eq!(resp.kvs[0].key, b"foo");
+        assert_eq!(resp.kvs[0].value, b"bar");
+        assert_eq!(resp.kvs[1].key, b"foo/bar");
+        assert_eq!(resp.kvs[1].value, b"baz");
+        assert_eq!(resp.kvs[2].key, b"foo/baz");
+        assert_eq!(resp.kvs[2].value, b"qux");
+    }
+}
--- a/src/cli/src/metadata/control/del/table.rs
+++ b/src/cli/src/metadata/control/del/table.rs
@@ -0,0 +1,235 @@
+// Copyright 2023 Greptime Team
+//
+// Licensed under the Apache License, Version 2.0 (the "License");
+// you may not use this file except in compliance with the License.
+// You may obtain a copy of the License at
+//
+//     http://www.apache.org/licenses/LICENSE-2.0
+//
+// Unless required by applicable law or agreed to in writing, software
+// distributed under the License is distributed on an "AS IS" BASIS,
+// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+// See the License for the specific language governing permissions and
+// limitations under the License.
+
+use async_trait::async_trait;
+use clap::Parser;
+use client::{DEFAULT_CATALOG_NAME, DEFAULT_SCHEMA_NAME};
+use common_catalog::format_full_table_name;
+use common_error::ext::BoxedError;
+use common_meta::ddl::utils::get_region_wal_options;
+use common_meta::key::table_name::TableNameManager;
+use common_meta::key::TableMetadataManager;
+use common_meta::kv_backend::KvBackendRef;
+use store_api::storage::TableId;
+
+use crate::error::{InvalidArgumentsSnafu, TableNotFoundSnafu};
+use crate::metadata::common::StoreConfig;
+use crate::metadata::control::del::CLI_TOMBSTONE_PREFIX;
+use crate::metadata::control::utils::get_table_id_by_name;
+use crate::Tool;
+
+/// Delete table metadata logically from the metadata store.
+#[derive(Debug, Default, Parser)]
+pub struct DelTableCommand {
+    /// The table id to delete from the metadata store.
+    #[clap(long)]
+    table_id: Option<u32>,
+
+    /// The table name to delete from the metadata store.
+    #[clap(long)]
+    table_name: Option<String>,
+
+    /// The schema name of the table.
+    #[clap(long, default_value = DEFAULT_SCHEMA_NAME)]
+    schema_name: String,
+
+    /// The catalog name of the table.
+    #[clap(long, default_value = DEFAULT_CATALOG_NAME)]
+    catalog_name: String,
+
+    #[clap(flatten)]
+    store: StoreConfig,
+}
+
+impl DelTableCommand {
+    fn validate(&self) -> Result<(), BoxedError> {
+        if matches!(
+            (&self.table_id, &self.table_name),
+            (Some(_), Some(_)) | (None, None)
+        ) {
+            return Err(BoxedError::new(
+                InvalidArgumentsSnafu {
+                    msg: "You must specify either --table-id or --table-name.",
+                }
+                .build(),
+            ));
+        }
+        Ok(())
+    }
+}
+
+impl DelTableCommand {
+    pub async fn build(&self) -> Result<Box<dyn Tool>, BoxedError> {
+        self.validate()?;
+        let kv_backend = self.store.build().await?;
+        Ok(Box::new(DelTableTool {
+            table_id: self.table_id,
+            table_name: self.table_name.clone(),
+            schema_name: self.schema_name.clone(),
+            catalog_name: self.catalog_name.clone(),
+            table_name_manager: TableNameManager::new(kv_backend.clone()),
+            table_metadata_deleter: TableMetadataDeleter::new(kv_backend),
+        }))
+    }
+}
+
+struct DelTableTool {
+    table_id: Option<u32>,
+    table_name: Option<String>,
+    schema_name: String,
+    catalog_name: String,
+    table_name_manager: TableNameManager,
+    table_metadata_deleter: TableMetadataDeleter,
+}
+
+#[async_trait]
+impl Tool for DelTableTool {
+    async fn do_work(&self) -> Result<(), BoxedError> {
+        let table_id = if let Some(table_name) = &self.table_name {
+            let catalog_name = &self.catalog_name;
+            let schema_name = &self.schema_name;
+
+            let Some(table_id) = get_table_id_by_name(
+                &self.table_name_manager,
+                catalog_name,
+                schema_name,
+                table_name,
+            )
+            .await?
+            else {
+                println!(
+                    "Table({}) not found",
+                    format_full_table_name(catalog_name, schema_name, table_name)
+                );
+                return Ok(());
+            };
+            table_id
+        } else {
+            // Safety: we have validated that table_id or table_name is not None
+            self.table_id.unwrap()
+        };
+        self.table_metadata_deleter.delete(table_id).await?;
+        println!("Table({}) deleted", table_id);
+
+        Ok(())
+    }
+}
+
+struct TableMetadataDeleter {
+    table_metadata_manager: TableMetadataManager,
+}
+
+impl TableMetadataDeleter {
+    fn new(kv_backend: KvBackendRef) -> Self {
+        Self {
+            table_metadata_manager: TableMetadataManager::new_with_custom_tombstone_prefix(
+                kv_backend,
+                CLI_TOMBSTONE_PREFIX,
+            ),
+        }
+    }
+
+    async fn delete(&self, table_id: TableId) -> Result<(), BoxedError> {
+        let (table_info, table_route) = self
+            .table_metadata_manager
+            .get_full_table_info(table_id)
+            .await
+            .map_err(BoxedError::new)?;
+        let Some(table_info) = table_info else {
+            return Err(BoxedError::new(TableNotFoundSnafu { table_id }.build()));
+        };
+        let Some(table_route) = table_route else {
+            return Err(BoxedError::new(TableNotFoundSnafu { table_id }.build()));
+        };
+        let physical_table_id = self
+            .table_metadata_manager
+            .table_route_manager()
+            .get_physical_table_id(table_id)
+            .await
+            .map_err(BoxedError::new)?;
+
+        let table_name = table_info.table_name();
+        let region_wal_options = get_region_wal_options(
+            &self.table_metadata_manager,
+            &table_route,
+            physical_table_id,
+        )
+        .await
+        .map_err(BoxedError::new)?;
+
+        self.table_metadata_manager
+            .delete_table_metadata(table_id, &table_name, &table_route, &region_wal_options)
+            .await
+            .map_err(BoxedError::new)?;
+        Ok(())
+    }
+}
+
+#[cfg(test)]
+mod tests {
+    use std::collections::HashMap;
+    use std::sync::Arc;
+
+    use common_error::ext::ErrorExt;
+    use common_error::status_code::StatusCode;
+    use common_meta::key::table_route::TableRouteValue;
+    use common_meta::key::TableMetadataManager;
+    use common_meta::kv_backend::chroot::ChrootKvBackend;
+    use common_meta::kv_backend::memory::MemoryKvBackend;
+    use common_meta::kv_backend::{KvBackend, KvBackendRef};
+    use common_meta::rpc::store::RangeRequest;
+
+    use crate::metadata::control::del::table::TableMetadataDeleter;
+    use crate::metadata::control::del::CLI_TOMBSTONE_PREFIX;
+    use crate::metadata::control::test_utils::prepare_physical_table_metadata;
+
+    #[tokio::test]
+    async fn test_delete_table_not_found() {
+        let kv_backend = Arc::new(MemoryKvBackend::new()) as KvBackendRef;
+
+        let table_metadata_deleter = TableMetadataDeleter::new(kv_backend);
+        let table_id = 1;
+        let err = table_metadata_deleter.delete(table_id).await.unwrap_err();
+        assert_eq!(err.status_code(), StatusCode::TableNotFound);
+    }
+
+    #[tokio::test]
+    async fn test_delete_table_metadata() {
+        let kv_backend = Arc::new(MemoryKvBackend::new());
+        let table_metadata_manager = TableMetadataManager::new(kv_backend.clone());
+        let table_id = 1024;
+        let (table_info, table_route) = prepare_physical_table_metadata("my_table", table_id).await;
+        table_metadata_manager
+            .create_table_metadata(
+                table_info,
+                TableRouteValue::Physical(table_route),
+                HashMap::new(),
+            )
+            .await
+            .unwrap();
+
+        let total_keys = kv_backend.len();
+        assert!(total_keys > 0);
+
+        let table_metadata_deleter = TableMetadataDeleter::new(kv_backend.clone());
+        table_metadata_deleter.delete(table_id).await.unwrap();
+
+        // Check the tombstone keys are deleted
+        let chroot =
+            ChrootKvBackend::new(CLI_TOMBSTONE_PREFIX.as_bytes().to_vec(), kv_backend.clone());
+        let req = RangeRequest::default().with_range(vec![0], vec![0]);
+        let resp = chroot.range(req).await.unwrap();
+        assert_eq!(resp.kvs.len(), total_keys);
+    }
+}
--- a/src/cli/src/metadata/control/get.rs
+++ b/src/cli/src/metadata/control/get.rs
@@ -0,0 +1,247 @@
+// Copyright 2023 Greptime Team
+//
+// Licensed under the Apache License, Version 2.0 (the "License");
+// you may not use this file except in compliance with the License.
+// You may obtain a copy of the License at
+//
+//     http://www.apache.org/licenses/LICENSE-2.0
+//
+// Unless required by applicable law or agreed to in writing, software
+// distributed under the License is distributed on an "AS IS" BASIS,
+// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+// See the License for the specific language governing permissions and
+// limitations under the License.
+
+use std::cmp::min;
+
+use async_trait::async_trait;
+use clap::{Parser, Subcommand};
+use client::{DEFAULT_CATALOG_NAME, DEFAULT_SCHEMA_NAME};
+use common_catalog::format_full_table_name;
+use common_error::ext::BoxedError;
+use common_meta::key::table_info::TableInfoKey;
+use common_meta::key::table_route::TableRouteKey;
+use common_meta::key::TableMetadataManager;
+use common_meta::kv_backend::KvBackendRef;
+use common_meta::range_stream::{PaginationStream, DEFAULT_PAGE_SIZE};
+use common_meta::rpc::store::RangeRequest;
+use futures::TryStreamExt;
+
+use crate::error::InvalidArgumentsSnafu;
+use crate::metadata::common::StoreConfig;
+use crate::metadata::control::utils::{decode_key_value, get_table_id_by_name, json_fromatter};
+use crate::Tool;
+
+/// Getting metadata from metadata store.
+#[derive(Subcommand)]
+pub enum GetCommand {
+    Key(GetKeyCommand),
+    Table(GetTableCommand),
+}
+
+impl GetCommand {
+    pub async fn build(&self) -> Result<Box<dyn Tool>, BoxedError> {
+        match self {
+            GetCommand::Key(cmd) => cmd.build().await,
+            GetCommand::Table(cmd) => cmd.build().await,
+        }
+    }
+}
+
+/// Get key-value pairs from the metadata store.
+#[derive(Debug, Default, Parser)]
+pub struct GetKeyCommand {
+    /// The key to get from the metadata store.
+    #[clap(default_value = "")]
+    key: String,
+
+    /// Whether to perform a prefix query. If true, returns all key-value pairs where the key starts with the given prefix.
+    #[clap(long, default_value = "false")]
+    prefix: bool,
+
+    /// The maximum number of key-value pairs to return. If 0, returns all key-value pairs.
+    #[clap(long, default_value = "0")]
+    limit: u64,
+
+    #[clap(flatten)]
+    store: StoreConfig,
+}
+
+impl GetKeyCommand {
+    pub async fn build(&self) -> Result<Box<dyn Tool>, BoxedError> {
+        let kvbackend = self.store.build().await?;
+        Ok(Box::new(GetKeyTool {
+            kvbackend,
+            key: self.key.clone(),
+            prefix: self.prefix,
+            limit: self.limit,
+        }))
+    }
+}
+
+struct GetKeyTool {
+    kvbackend: KvBackendRef,
+    key: String,
+    prefix: bool,
+    limit: u64,
+}
+
+#[async_trait]
+impl Tool for GetKeyTool {
+    async fn do_work(&self) -> Result<(), BoxedError> {
+        let mut req = RangeRequest::default();
+        if self.prefix {
+            req = req.with_prefix(self.key.as_bytes());
+        } else {
+            req = req.with_key(self.key.as_bytes());
+        }
+        let page_size = if self.limit > 0 {
+            min(self.limit as usize, DEFAULT_PAGE_SIZE)
+        } else {
+            DEFAULT_PAGE_SIZE
+        };
+        let pagination_stream =
+            PaginationStream::new(self.kvbackend.clone(), req, page_size, decode_key_value);
+        let mut stream = Box::pin(pagination_stream.into_stream());
+        let mut counter = 0;
+
+        while let Some((key, value)) = stream.try_next().await.map_err(BoxedError::new)? {
+            print!("{}\n{}\n", key, value);
+            counter += 1;
+            if self.limit > 0 && counter >= self.limit {
+                break;
+            }
+        }
+
+        Ok(())
+    }
+}
+
+/// Get table metadata from the metadata store via table id.
+#[derive(Debug, Default, Parser)]
+pub struct GetTableCommand {
+    /// Get table metadata by table id.
+    #[clap(long)]
+    table_id: Option<u32>,
+
+    /// Get table metadata by table name.
+    #[clap(long)]
+    table_name: Option<String>,
+
+    /// The schema name of the table.
+    #[clap(long, default_value = DEFAULT_SCHEMA_NAME)]
+    schema_name: String,
+
+    /// The catalog name of the table.
+    #[clap(long, default_value = DEFAULT_CATALOG_NAME)]
+    catalog_name: String,
+
+    /// Pretty print the output.
+    #[clap(long, default_value = "false")]
+    pretty: bool,
+
+    #[clap(flatten)]
+    store: StoreConfig,
+}
+
+impl GetTableCommand {
+    pub fn validate(&self) -> Result<(), BoxedError> {
+        if matches!(
+            (&self.table_id, &self.table_name),
+            (Some(_), Some(_)) | (None, None)
+        ) {
+            return Err(BoxedError::new(
+                InvalidArgumentsSnafu {
+                    msg: "You must specify either --table-id or --table-name.",
+                }
+                .build(),
+            ));
+        }
+        Ok(())
+    }
+}
+
+struct GetTableTool {
+    kvbackend: KvBackendRef,
+    table_id: Option<u32>,
+    table_name: Option<String>,
+    schema_name: String,
+    catalog_name: String,
+    pretty: bool,
+}
+
+#[async_trait]
+impl Tool for GetTableTool {
+    async fn do_work(&self) -> Result<(), BoxedError> {
+        let table_metadata_manager = TableMetadataManager::new(self.kvbackend.clone());
+        let table_name_manager = table_metadata_manager.table_name_manager();
+        let table_info_manager = table_metadata_manager.table_info_manager();
+        let table_route_manager = table_metadata_manager.table_route_manager();
+
+        let table_id = if let Some(table_name) = &self.table_name {
+            let catalog_name = &self.catalog_name;
+            let schema_name = &self.schema_name;
+
+            let Some(table_id) =
+                get_table_id_by_name(table_name_manager, catalog_name, schema_name, table_name)
+                    .await?
+            else {
+                println!(
+                    "Table({}) not found",
+                    format_full_table_name(catalog_name, schema_name, table_name)
+                );
+                return Ok(());
+            };
+            table_id
+        } else {
+            // Safety: we have validated that table_id or table_name is not None
+            self.table_id.unwrap()
+        };
+
+        let table_info = table_info_manager
+            .get(table_id)
+            .await
+            .map_err(BoxedError::new)?;
+        if let Some(table_info) = table_info {
+            println!(
+                "{}\n{}",
+                TableInfoKey::new(table_id),
+                json_fromatter(self.pretty, &*table_info)
+            );
+        } else {
+            println!("Table info not found");
+        }
+
+        let table_route = table_route_manager
+            .table_route_storage()
+            .get(table_id)
+            .await
+            .map_err(BoxedError::new)?;
+        if let Some(table_route) = table_route {
+            println!(
+                "{}\n{}",
+                TableRouteKey::new(table_id),
+                json_fromatter(self.pretty, &table_route)
+            );
+        } else {
+            println!("Table route not found");
+        }
+
+        Ok(())
+    }
+}
+
+impl GetTableCommand {
+    pub async fn build(&self) -> Result<Box<dyn Tool>, BoxedError> {
+        self.validate()?;
+        let kvbackend = self.store.build().await?;
+        Ok(Box::new(GetTableTool {
+            kvbackend,
+            table_id: self.table_id,
+            table_name: self.table_name.clone(),
+            schema_name: self.schema_name.clone(),
+            catalog_name: self.catalog_name.clone(),
+            pretty: self.pretty,
+        }))
+    }
+}
--- a/src/cli/src/metadata/control/test_utils.rs
+++ b/src/cli/src/metadata/control/test_utils.rs
@@ -0,0 +1,51 @@
+// Copyright 2023 Greptime Team
+//
+// Licensed under the Apache License, Version 2.0 (the "License");
+// you may not use this file except in compliance with the License.
+// You may obtain a copy of the License at
+//
+//     http://www.apache.org/licenses/LICENSE-2.0
+//
+// Unless required by applicable law or agreed to in writing, software
+// distributed under the License is distributed on an "AS IS" BASIS,
+// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+// See the License for the specific language governing permissions and
+// limitations under the License.
+
+use common_meta::ddl::test_util::test_create_physical_table_task;
+use common_meta::key::table_route::PhysicalTableRouteValue;
+use common_meta::kv_backend::KvBackendRef;
+use common_meta::peer::Peer;
+use common_meta::rpc::router::{Region, RegionRoute};
+use common_meta::rpc::store::PutRequest;
+use store_api::storage::{RegionId, TableId};
+use table::metadata::RawTableInfo;
+
+/// Puts a key-value pair into the kv backend.
+pub async fn put_key(kv_backend: &KvBackendRef, key: &str, value: &str) {
+    let put_req = PutRequest::new()
+        .with_key(key.as_bytes())
+        .with_value(value.as_bytes());
+    kv_backend.put(put_req).await.unwrap();
+}
+
+/// Prepares the physical table metadata for testing.
+///
+/// Returns the table info and the table route.
+pub async fn prepare_physical_table_metadata(
+    table_name: &str,
+    table_id: TableId,
+) -> (RawTableInfo, PhysicalTableRouteValue) {
+    let mut create_physical_table_task = test_create_physical_table_task(table_name);
+    let table_route = PhysicalTableRouteValue::new(vec![RegionRoute {
+        region: Region {
+            id: RegionId::new(table_id, 1),
+            ..Default::default()
+        },
+        leader_peer: Some(Peer::empty(1)),
+        ..Default::default()
+    }]);
+    create_physical_table_task.set_table_id(table_id);
+
+    (create_physical_table_task.table_info, table_route)
+}
--- a/src/cli/src/metadata/control/utils.rs
+++ b/src/cli/src/metadata/control/utils.rs
@@ -0,0 +1,57 @@
+// Copyright 2023 Greptime Team
+//
+// Licensed under the Apache License, Version 2.0 (the "License");
+// you may not use this file except in compliance with the License.
+// You may obtain a copy of the License at
+//
+//     http://www.apache.org/licenses/LICENSE-2.0
+//
+// Unless required by applicable law or agreed to in writing, software
+// distributed under the License is distributed on an "AS IS" BASIS,
+// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+// See the License for the specific language governing permissions and
+// limitations under the License.
+
+use common_error::ext::BoxedError;
+use common_meta::error::Result as CommonMetaResult;
+use common_meta::key::table_name::{TableNameKey, TableNameManager};
+use common_meta::rpc::KeyValue;
+use serde::Serialize;
+use store_api::storage::TableId;
+
+/// Decodes a key-value pair into a string.
+pub fn decode_key_value(kv: KeyValue) -> CommonMetaResult<(String, String)> {
+    let key = String::from_utf8_lossy(&kv.key).to_string();
+    let value = String::from_utf8_lossy(&kv.value).to_string();
+    Ok((key, value))
+}
+
+/// Formats a value as a JSON string.
+pub fn json_fromatter<T>(pretty: bool, value: &T) -> String
+where
+    T: Serialize,
+{
+    if pretty {
+        serde_json::to_string_pretty(value).unwrap()
+    } else {
+        serde_json::to_string(value).unwrap()
+    }
+}
+
+/// Gets the table id by table name.
+pub async fn get_table_id_by_name(
+    table_name_manager: &TableNameManager,
+    catalog_name: &str,
+    schema_name: &str,
+    table_name: &str,
+) -> Result<Option<TableId>, BoxedError> {
+    let table_name_key = TableNameKey::new(catalog_name, schema_name, table_name);
+    let Some(table_name_value) = table_name_manager
+        .get(table_name_key)
+        .await
+        .map_err(BoxedError::new)?
+    else {
+        return Ok(None);
+    };
+    Ok(Some(table_name_value.table_id()))
+}
--- a/src/cli/src/metadata/repair.rs
+++ b/src/cli/src/metadata/repair.rs
@@ -0,0 +1,369 @@
+// Copyright 2023 Greptime Team
+//
+// Licensed under the Apache License, Version 2.0 (the "License");
+// you may not use this file except in compliance with the License.
+// You may obtain a copy of the License at
+//
+//     http://www.apache.org/licenses/LICENSE-2.0
+//
+// Unless required by applicable law or agreed to in writing, software
+// distributed under the License is distributed on an "AS IS" BASIS,
+// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+// See the License for the specific language governing permissions and
+// limitations under the License.
+
+mod alter_table;
+mod create_table;
+
+use std::sync::Arc;
+use std::time::Duration;
+
+use async_trait::async_trait;
+use clap::Parser;
+use client::api::v1::CreateTableExpr;
+use client::client_manager::NodeClients;
+use client::{DEFAULT_CATALOG_NAME, DEFAULT_SCHEMA_NAME};
+use common_error::ext::{BoxedError, ErrorExt};
+use common_error::status_code::StatusCode;
+use common_grpc::channel_manager::ChannelConfig;
+use common_meta::error::Error as CommonMetaError;
+use common_meta::key::TableMetadataManager;
+use common_meta::kv_backend::KvBackendRef;
+use common_meta::node_manager::NodeManagerRef;
+use common_meta::peer::Peer;
+use common_meta::rpc::router::{find_leaders, RegionRoute};
+use common_telemetry::{error, info, warn};
+use futures::TryStreamExt;
+use snafu::{ensure, ResultExt};
+use store_api::storage::TableId;
+
+use crate::error::{
+    InvalidArgumentsSnafu, Result, SendRequestToDatanodeSnafu, TableMetadataSnafu, UnexpectedSnafu,
+};
+use crate::metadata::common::StoreConfig;
+use crate::metadata::utils::{FullTableMetadata, IteratorInput, TableMetadataIterator};
+use crate::Tool;
+
+/// Repair metadata of logical tables.
+#[derive(Debug, Default, Parser)]
+pub struct RepairLogicalTablesCommand {
+    /// The names of the tables to repair.
+    #[clap(long, value_delimiter = ',', alias = "table-name")]
+    table_names: Vec<String>,
+
+    /// The id of the table to repair.
+    #[clap(long, value_delimiter = ',', alias = "table-id")]
+    table_ids: Vec<TableId>,
+
+    /// The schema of the tables to repair.
+    #[clap(long, default_value = DEFAULT_SCHEMA_NAME)]
+    schema_name: String,
+
+    /// The catalog of the tables to repair.
+    #[clap(long, default_value = DEFAULT_CATALOG_NAME)]
+    catalog_name: String,
+
+    /// Whether to fail fast if any repair operation fails.
+    #[clap(long)]
+    fail_fast: bool,
+
+    #[clap(flatten)]
+    store: StoreConfig,
+
+    /// The timeout for the client to operate the datanode.
+    #[clap(long, default_value_t = 30)]
+    client_timeout_secs: u64,
+
+    /// The timeout for the client to connect to the datanode.
+    #[clap(long, default_value_t = 3)]
+    client_connect_timeout_secs: u64,
+}
+
+impl RepairLogicalTablesCommand {
+    fn validate(&self) -> Result<()> {
+        ensure!(
+            !self.table_names.is_empty() || !self.table_ids.is_empty(),
+            InvalidArgumentsSnafu {
+                msg: "You must specify --table-names or --table-ids.",
+            }
+        );
+        Ok(())
+    }
+}
+
+impl RepairLogicalTablesCommand {
+    pub async fn build(&self) -> std::result::Result<Box<dyn Tool>, BoxedError> {
+        self.validate().map_err(BoxedError::new)?;
+        let kv_backend = self.store.build().await?;
+        let node_client_channel_config = ChannelConfig::new()
+            .timeout(Duration::from_secs(self.client_timeout_secs))
+            .connect_timeout(Duration::from_secs(self.client_connect_timeout_secs));
+        let node_manager = Arc::new(NodeClients::new(node_client_channel_config));
+
+        Ok(Box::new(RepairTool {
+            table_names: self.table_names.clone(),
+            table_ids: self.table_ids.clone(),
+            schema_name: self.schema_name.clone(),
+            catalog_name: self.catalog_name.clone(),
+            fail_fast: self.fail_fast,
+            kv_backend,
+            node_manager,
+        }))
+    }
+}
+
+struct RepairTool {
+    table_names: Vec<String>,
+    table_ids: Vec<TableId>,
+    schema_name: String,
+    catalog_name: String,
+    fail_fast: bool,
+    kv_backend: KvBackendRef,
+    node_manager: NodeManagerRef,
+}
+
+#[async_trait]
+impl Tool for RepairTool {
+    async fn do_work(&self) -> std::result::Result<(), BoxedError> {
+        self.repair_tables().await.map_err(BoxedError::new)
+    }
+}
+
+impl RepairTool {
+    fn generate_iterator_input(&self) -> Result<IteratorInput> {
+        if !self.table_names.is_empty() {
+            let table_names = &self.table_names;
+            let catalog = &self.catalog_name;
+            let schema_name = &self.schema_name;
+
+            let table_names = table_names
+                .iter()
+                .map(|table_name| {
+                    (
+                        catalog.to_string(),
+                        schema_name.to_string(),
+                        table_name.to_string(),
+                    )
+                })
+                .collect::<Vec<_>>();
+            return Ok(IteratorInput::new_table_names(table_names));
+        } else if !self.table_ids.is_empty() {
+            return Ok(IteratorInput::new_table_ids(self.table_ids.clone()));
+        };
+
+        InvalidArgumentsSnafu {
+            msg: "You must specify --table-names or --table-id.",
+        }
+        .fail()
+    }
+
+    async fn repair_tables(&self) -> Result<()> {
+        let input = self.generate_iterator_input()?;
+        let mut table_metadata_iterator =
+            Box::pin(TableMetadataIterator::new(self.kv_backend.clone(), input).into_stream());
+        let table_metadata_manager = TableMetadataManager::new(self.kv_backend.clone());
+
+        let mut skipped_table = 0;
+        let mut success_table = 0;
+        while let Some(full_table_metadata) = table_metadata_iterator.try_next().await? {
+            let full_table_name = full_table_metadata.full_table_name();
+            if !full_table_metadata.is_metric_engine() {
+                warn!(
+                    "Skipping repair for non-metric engine table: {}",
+                    full_table_name
+                );
+                skipped_table += 1;
+                continue;
+            }
+
+            if full_table_metadata.is_physical_table() {
+                warn!("Skipping repair for physical table: {}", full_table_name);
+                skipped_table += 1;
+                continue;
+            }
+
+            let (physical_table_id, physical_table_route) = table_metadata_manager
+                .table_route_manager()
+                .get_physical_table_route(full_table_metadata.table_id)
+                .await
+                .context(TableMetadataSnafu)?;
+
+            if let Err(err) = self
+                .repair_table(
+                    &full_table_metadata,
+                    physical_table_id,
+                    &physical_table_route.region_routes,
+                )
+                .await
+            {
+                error!(
+                    err;
+                    "Failed to repair table: {}, skipped table: {}",
+                    full_table_name,
+                    skipped_table,
+                );
+
+                if self.fail_fast {
+                    return Err(err);
+                }
+            } else {
+                success_table += 1;
+            }
+        }
+
+        info!(
+            "Repair logical tables result: {} tables repaired, {} tables skipped",
+            success_table, skipped_table
+        );
+
+        Ok(())
+    }
+
+    async fn alter_table_on_datanodes(
+        &self,
+        full_table_metadata: &FullTableMetadata,
+        physical_region_routes: &[RegionRoute],
+    ) -> Result<Vec<(Peer, CommonMetaError)>> {
+        let logical_table_id = full_table_metadata.table_id;
+        let alter_table_expr = alter_table::generate_alter_table_expr_for_all_columns(
+            &full_table_metadata.table_info,
+        )?;
+        let node_manager = self.node_manager.clone();
+
+        let mut failed_peers = Vec::new();
+        info!(
+            "Sending alter table requests to all datanodes for table: {}, number of regions:{}.",
+            full_table_metadata.full_table_name(),
+            physical_region_routes.len()
+        );
+        let leaders = find_leaders(physical_region_routes);
+        for peer in &leaders {
+            let alter_table_request = alter_table::make_alter_region_request_for_peer(
+                logical_table_id,
+                &alter_table_expr,
+                full_table_metadata.table_info.ident.version,
+                peer,
+                physical_region_routes,
+            )?;
+            let datanode = node_manager.datanode(peer).await;
+            if let Err(err) = datanode.handle(alter_table_request).await {
+                failed_peers.push((peer.clone(), err));
+            }
+        }
+
+        Ok(failed_peers)
+    }
+
+    async fn create_table_on_datanode(
+        &self,
+        create_table_expr: &CreateTableExpr,
+        logical_table_id: TableId,
+        physical_table_id: TableId,
+        peer: &Peer,
+        physical_region_routes: &[RegionRoute],
+    ) -> Result<()> {
+        let node_manager = self.node_manager.clone();
+        let datanode = node_manager.datanode(peer).await;
+        let create_table_request = create_table::make_create_region_request_for_peer(
+            logical_table_id,
+            physical_table_id,
+            create_table_expr,
+            peer,
+            physical_region_routes,
+        )?;
+
+        datanode
+            .handle(create_table_request)
+            .await
+            .with_context(|_| SendRequestToDatanodeSnafu { peer: peer.clone() })?;
+
+        Ok(())
+    }
+
+    async fn repair_table(
+        &self,
+        full_table_metadata: &FullTableMetadata,
+        physical_table_id: TableId,
+        physical_region_routes: &[RegionRoute],
+    ) -> Result<()> {
+        let full_table_name = full_table_metadata.full_table_name();
+        // First we sends alter table requests to all datanodes with all columns.
+        let failed_peers = self
+            .alter_table_on_datanodes(full_table_metadata, physical_region_routes)
+            .await?;
+
+        if failed_peers.is_empty() {
+            info!(
+                "All alter table requests sent successfully for table: {}",
+                full_table_name
+            );
+            return Ok(());
+        }
+        warn!(
+            "Sending alter table requests to datanodes for table: {} failed for the datanodes: {:?}",
+            full_table_name,
+            failed_peers.iter().map(|(peer, _)| peer.id).collect::<Vec<_>>()
+        );
+
+        let create_table_expr =
+            create_table::generate_create_table_expr(&full_table_metadata.table_info)?;
+
+        let mut errors = Vec::new();
+        for (peer, err) in failed_peers {
+            if err.status_code() != StatusCode::RegionNotFound {
+                error!(
+                    err;
+                    "Sending alter table requests to datanode: {} for table: {} failed",
+                    peer.id,
+                    full_table_name,
+                );
+                continue;
+            }
+            info!(
+                "Region not found for table: {}, datanode: {}, trying to create the logical table on that datanode",
+                full_table_name,
+                peer.id
+            );
+
+            // If the alter table request fails for any datanode, we attempt to create the table on that datanode
+            // as a fallback mechanism to ensure table consistency across the cluster.
+            if let Err(err) = self
+                .create_table_on_datanode(
+                    &create_table_expr,
+                    full_table_metadata.table_id,
+                    physical_table_id,
+                    &peer,
+                    physical_region_routes,
+                )
+                .await
+            {
+                error!(
+                    err;
+                    "Failed to create table on datanode: {} for table: {}",
+                    peer.id, full_table_name
+                );
+                errors.push(err);
+                if self.fail_fast {
+                    break;
+                }
+            } else {
+                info!(
+                    "Created table on datanode: {} for table: {}",
+                    peer.id, full_table_name
+                );
+            }
+        }
+
+        if !errors.is_empty() {
+            return UnexpectedSnafu {
+                msg: format!(
+                    "Failed to create table on datanodes for table: {}",
+                    full_table_name,
+                ),
+            }
+            .fail();
+        }
+
+        Ok(())
+    }
+}
--- a/src/cli/src/metadata/repair/alter_table.rs
+++ b/src/cli/src/metadata/repair/alter_table.rs
@@ -0,0 +1,85 @@
+// Copyright 2023 Greptime Team
+//
+// Licensed under the Apache License, Version 2.0 (the "License");
+// you may not use this file except in compliance with the License.
+// You may obtain a copy of the License at
+//
+//     http://www.apache.org/licenses/LICENSE-2.0
+//
+// Unless required by applicable law or agreed to in writing, software
+// distributed under the License is distributed on an "AS IS" BASIS,
+// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+// See the License for the specific language governing permissions and
+// limitations under the License.
+
+use client::api::v1::alter_table_expr::Kind;
+use client::api::v1::region::{region_request, AlterRequests, RegionRequest, RegionRequestHeader};
+use client::api::v1::{AddColumn, AddColumns, AlterTableExpr};
+use common_meta::ddl::alter_logical_tables::make_alter_region_request;
+use common_meta::peer::Peer;
+use common_meta::rpc::router::{find_leader_regions, RegionRoute};
+use operator::expr_helper::column_schemas_to_defs;
+use snafu::ResultExt;
+use store_api::storage::{RegionId, TableId};
+use table::metadata::RawTableInfo;
+
+use crate::error::{CovertColumnSchemasToDefsSnafu, Result};
+
+/// Generates alter table expression for all columns.
+pub fn generate_alter_table_expr_for_all_columns(
+    table_info: &RawTableInfo,
+) -> Result<AlterTableExpr> {
+    let schema = &table_info.meta.schema;
+
+    let mut alter_table_expr = AlterTableExpr {
+        catalog_name: table_info.catalog_name.to_string(),
+        schema_name: table_info.schema_name.to_string(),
+        table_name: table_info.name.to_string(),
+        ..Default::default()
+    };
+
+    let primary_keys = table_info
+        .meta
+        .primary_key_indices
+        .iter()
+        .map(|i| schema.column_schemas[*i].name.clone())
+        .collect::<Vec<_>>();
+
+    let add_columns = column_schemas_to_defs(schema.column_schemas.clone(), &primary_keys)
+        .context(CovertColumnSchemasToDefsSnafu)?;
+
+    alter_table_expr.kind = Some(Kind::AddColumns(AddColumns {
+        add_columns: add_columns
+            .into_iter()
+            .map(|col| AddColumn {
+                column_def: Some(col),
+                location: None,
+                add_if_not_exists: true,
+            })
+            .collect(),
+    }));
+
+    Ok(alter_table_expr)
+}
+
+/// Makes an alter region request for a peer.
+pub fn make_alter_region_request_for_peer(
+    logical_table_id: TableId,
+    alter_table_expr: &AlterTableExpr,
+    schema_version: u64,
+    peer: &Peer,
+    region_routes: &[RegionRoute],
+) -> Result<RegionRequest> {
+    let regions_on_this_peer = find_leader_regions(region_routes, peer);
+    let mut requests = Vec::with_capacity(regions_on_this_peer.len());
+    for region_number in &regions_on_this_peer {
+        let region_id = RegionId::new(logical_table_id, *region_number);
+        let request = make_alter_region_request(region_id, alter_table_expr, schema_version);
+        requests.push(request);
+    }
+
+    Ok(RegionRequest {
+        header: Some(RegionRequestHeader::default()),
+        body: Some(region_request::Body::Alters(AlterRequests { requests })),
+    })
+}
--- a/src/cli/src/metadata/repair/create_table.rs
+++ b/src/cli/src/metadata/repair/create_table.rs
@@ -0,0 +1,89 @@
+// Copyright 2023 Greptime Team
+//
+// Licensed under the Apache License, Version 2.0 (the "License");
+// you may not use this file except in compliance with the License.
+// You may obtain a copy of the License at
+//
+//     http://www.apache.org/licenses/LICENSE-2.0
+//
+// Unless required by applicable law or agreed to in writing, software
+// distributed under the License is distributed on an "AS IS" BASIS,
+// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+// See the License for the specific language governing permissions and
+// limitations under the License.
+
+use std::collections::HashMap;
+
+use client::api::v1::region::{region_request, CreateRequests, RegionRequest, RegionRequestHeader};
+use client::api::v1::CreateTableExpr;
+use common_meta::ddl::create_logical_tables::create_region_request_builder;
+use common_meta::ddl::utils::region_storage_path;
+use common_meta::peer::Peer;
+use common_meta::rpc::router::{find_leader_regions, RegionRoute};
+use operator::expr_helper::column_schemas_to_defs;
+use snafu::ResultExt;
+use store_api::storage::{RegionId, TableId};
+use table::metadata::RawTableInfo;
+
+use crate::error::{CovertColumnSchemasToDefsSnafu, Result};
+
+/// Generates a `CreateTableExpr` from a `RawTableInfo`.
+pub fn generate_create_table_expr(table_info: &RawTableInfo) -> Result<CreateTableExpr> {
+    let schema = &table_info.meta.schema;
+    let primary_keys = table_info
+        .meta
+        .primary_key_indices
+        .iter()
+        .map(|i| schema.column_schemas[*i].name.clone())
+        .collect::<Vec<_>>();
+
+    let timestamp_index = schema.timestamp_index.as_ref().unwrap();
+    let time_index = schema.column_schemas[*timestamp_index].name.clone();
+    let column_defs = column_schemas_to_defs(schema.column_schemas.clone(), &primary_keys)
+        .context(CovertColumnSchemasToDefsSnafu)?;
+    let table_options = HashMap::from(&table_info.meta.options);
+
+    Ok(CreateTableExpr {
+        catalog_name: table_info.catalog_name.to_string(),
+        schema_name: table_info.schema_name.to_string(),
+        table_name: table_info.name.to_string(),
+        desc: String::default(),
+        column_defs,
+        time_index,
+        primary_keys,
+        create_if_not_exists: true,
+        table_options,
+        table_id: None,
+        engine: table_info.meta.engine.to_string(),
+    })
+}
+
+/// Makes a create region request for a peer.
+pub fn make_create_region_request_for_peer(
+    logical_table_id: TableId,
+    physical_table_id: TableId,
+    create_table_expr: &CreateTableExpr,
+    peer: &Peer,
+    region_routes: &[RegionRoute],
+) -> Result<RegionRequest> {
+    let regions_on_this_peer = find_leader_regions(region_routes, peer);
+    let mut requests = Vec::with_capacity(regions_on_this_peer.len());
+    let request_builder =
+        create_region_request_builder(create_table_expr, physical_table_id).unwrap();
+
+    let catalog = &create_table_expr.catalog_name;
+    let schema = &create_table_expr.schema_name;
+    let storage_path = region_storage_path(catalog, schema);
+
+    for region_number in &regions_on_this_peer {
+        let region_id = RegionId::new(logical_table_id, *region_number);
+        let region_request =
+            request_builder.build_one(region_id, storage_path.clone(), &HashMap::new());
+        requests.push(region_request);
+    }
+
+    Ok(RegionRequest {
+        header: Some(RegionRequestHeader::default()),
+        body: Some(region_request::Body::Creates(CreateRequests { requests })),
+    })
+}
--- a/src/cli/src/metadata/snapshot.rs
+++ b/src/cli/src/metadata/snapshot.rs
@@ -13,139 +13,37 @@
 // limitations under the License.

 use std::path::Path;
-use std::sync::Arc;

 use async_trait::async_trait;
 use clap::{Parser, Subcommand};
 use common_base::secrets::{ExposeSecret, SecretString};
 use common_error::ext::BoxedError;
-use common_meta::kv_backend::chroot::ChrootKvBackend;
-use common_meta::kv_backend::etcd::EtcdStore;
-use common_meta::kv_backend::KvBackendRef;
 use common_meta::snapshot::MetadataSnapshotManager;
-use meta_srv::bootstrap::create_etcd_client;
-use meta_srv::metasrv::BackendImpl;
 use object_store::services::{Fs, S3};
 use object_store::ObjectStore;
 use snafu::{OptionExt, ResultExt};

-use crate::error::{
-    InvalidFilePathSnafu, KvBackendNotSetSnafu, OpenDalSnafu, S3ConfigNotSetSnafu,
-    UnsupportedMemoryBackendSnafu,
-};
+use crate::error::{InvalidFilePathSnafu, OpenDalSnafu, S3ConfigNotSetSnafu};
+use crate::metadata::common::StoreConfig;
 use crate::Tool;

-/// Subcommand for metadata snapshot management.
+/// Subcommand for metadata snapshot operations, including saving snapshots, restoring from snapshots, and viewing snapshot information.
 #[derive(Subcommand)]
-pub enum MetaCommand {
-    #[clap(subcommand)]
-    Snapshot(MetaSnapshotCommand),
+pub enum SnapshotCommand {
+    /// Save a snapshot of the current metadata state to a specified location.
+    Save(SaveCommand),
+    /// Restore metadata from a snapshot.
+    Restore(RestoreCommand),
+    /// Explore metadata from a snapshot.
+    Info(InfoCommand),
 }

-impl MetaCommand {
+impl SnapshotCommand {
    pub async fn build(&self) -> Result<Box<dyn Tool>, BoxedError> {
        match self {
-            MetaCommand::Snapshot(cmd) => cmd.build().await,
-        }
-    }
-}
-
-/// Subcommand for metadata snapshot operations. such as save, restore and info.
-#[derive(Subcommand)]
-pub enum MetaSnapshotCommand {
-    /// Export metadata snapshot tool.
-    Save(MetaSaveCommand),
-    /// Restore metadata snapshot tool.
-    Restore(MetaRestoreCommand),
-    /// Explore metadata from metadata snapshot.
-    Info(MetaInfoCommand),
-}
-
-impl MetaSnapshotCommand {
-    pub async fn build(&self) -> Result<Box<dyn Tool>, BoxedError> {
-        match self {
-            MetaSnapshotCommand::Save(cmd) => cmd.build().await,
-            MetaSnapshotCommand::Restore(cmd) => cmd.build().await,
-            MetaSnapshotCommand::Info(cmd) => cmd.build().await,
-        }
-    }
-}
-
-#[derive(Debug, Default, Parser)]
-struct MetaConnection {
-    /// The endpoint of store. one of etcd, pg or mysql.
-    #[clap(long, alias = "store-addr", value_delimiter = ',', num_args = 1..)]
-    store_addrs: Vec<String>,
-    /// The database backend.
-    #[clap(long, value_enum)]
-    backend: Option<BackendImpl>,
-    #[clap(long, default_value = "")]
-    store_key_prefix: String,
-    #[cfg(any(feature = "pg_kvbackend", feature = "mysql_kvbackend"))]
-    #[clap(long,default_value = common_meta::kv_backend::DEFAULT_META_TABLE_NAME)]
-    meta_table_name: String,
-    #[clap(long, default_value = "128")]
-    max_txn_ops: usize,
-}
-
-impl MetaConnection {
-    pub async fn build(&self) -> Result<KvBackendRef, BoxedError> {
-        let max_txn_ops = self.max_txn_ops;
-        let store_addrs = &self.store_addrs;
-        if store_addrs.is_empty() {
-            KvBackendNotSetSnafu { backend: "all" }
-                .fail()
-                .map_err(BoxedError::new)
-        } else {
-            let kvbackend = match self.backend {
-                Some(BackendImpl::EtcdStore) => {
-                    let etcd_client = create_etcd_client(store_addrs)
-                        .await
-                        .map_err(BoxedError::new)?;
-                    Ok(EtcdStore::with_etcd_client(etcd_client, max_txn_ops))
-                }
-                #[cfg(feature = "pg_kvbackend")]
-                Some(BackendImpl::PostgresStore) => {
-                    let table_name = &self.meta_table_name;
-                    let pool = meta_srv::bootstrap::create_postgres_pool(store_addrs)
-                        .await
-                        .map_err(BoxedError::new)?;
-                    Ok(common_meta::kv_backend::rds::PgStore::with_pg_pool(
-                        pool,
-                        table_name,
-                        max_txn_ops,
-                    )
-                    .await
-                    .map_err(BoxedError::new)?)
-                }
-                #[cfg(feature = "mysql_kvbackend")]
-                Some(BackendImpl::MysqlStore) => {
-                    let table_name = &self.meta_table_name;
-                    let pool = meta_srv::bootstrap::create_mysql_pool(store_addrs)
-                        .await
-                        .map_err(BoxedError::new)?;
-                    Ok(common_meta::kv_backend::rds::MySqlStore::with_mysql_pool(
-                        pool,
-                        table_name,
-                        max_txn_ops,
-                    )
-                    .await
-                    .map_err(BoxedError::new)?)
-                }
-                Some(BackendImpl::MemoryStore) => UnsupportedMemoryBackendSnafu
-                    .fail()
-                    .map_err(BoxedError::new),
-                _ => KvBackendNotSetSnafu { backend: "all" }
-                    .fail()
-                    .map_err(BoxedError::new),
-            };
-            if self.store_key_prefix.is_empty() {
-                kvbackend
-            } else {
-                let chroot_kvbackend =
-                    ChrootKvBackend::new(self.store_key_prefix.as_bytes().to_vec(), kvbackend?);
-                Ok(Arc::new(chroot_kvbackend))
-            }
+            SnapshotCommand::Save(cmd) => cmd.build().await,
+            SnapshotCommand::Restore(cmd) => cmd.build().await,
+            SnapshotCommand::Info(cmd) => cmd.build().await,
        }
    }
 }
@@ -214,10 +112,10 @@ impl S3Config {
 /// It will dump the metadata snapshot to local file or s3 bucket.
 /// The snapshot file will be in binary format.
 #[derive(Debug, Default, Parser)]
-pub struct MetaSaveCommand {
-    /// The connection to the metadata store.
+pub struct SaveCommand {
+    /// The store configuration.
    #[clap(flatten)]
-    connection: MetaConnection,
+    store: StoreConfig,
    /// The s3 config.
    #[clap(flatten)]
    s3_config: S3Config,
@@ -240,9 +138,9 @@ fn create_local_file_object_store(root: &str) -> Result<ObjectStore, BoxedError>
    Ok(object_store)
 }

-impl MetaSaveCommand {
+impl SaveCommand {
    pub async fn build(&self) -> Result<Box<dyn Tool>, BoxedError> {
-        let kvbackend = self.connection.build().await?;
+        let kvbackend = self.store.build().await?;
        let output_dir = &self.output_dir;
        let object_store = self.s3_config.build(output_dir).map_err(BoxedError::new)?;
        if let Some(store) = object_store {
@@ -262,7 +160,7 @@ impl MetaSaveCommand {
    }
 }

-pub struct MetaSnapshotTool {
+struct MetaSnapshotTool {
    inner: MetadataSnapshotManager,
    target_file: String,
 }
@@ -278,14 +176,16 @@ impl Tool for MetaSnapshotTool {
    }
 }

-/// Restore metadata snapshot tool.
-/// This tool is used to restore metadata snapshot from etcd, pg or mysql.
-/// It will restore the metadata snapshot from local file or s3 bucket.
+/// Restore metadata from a snapshot file.
+///
+/// This command restores the metadata state from a previously saved snapshot.
+/// The snapshot can be loaded from either a local file system or an S3 bucket,
+/// depending on the provided configuration.
 #[derive(Debug, Default, Parser)]
-pub struct MetaRestoreCommand {
-    /// The connection to the metadata store.
+pub struct RestoreCommand {
+    /// The store configuration.
    #[clap(flatten)]
-    connection: MetaConnection,
+    store: StoreConfig,
    /// The s3 config.
    #[clap(flatten)]
    s3_config: S3Config,
@@ -299,9 +199,9 @@ pub struct MetaRestoreCommand {
    force: bool,
 }

-impl MetaRestoreCommand {
+impl RestoreCommand {
    pub async fn build(&self) -> Result<Box<dyn Tool>, BoxedError> {
-        let kvbackend = self.connection.build().await?;
+        let kvbackend = self.store.build().await?;
        let input_dir = &self.input_dir;
        let object_store = self.s3_config.build(input_dir).map_err(BoxedError::new)?;
        if let Some(store) = object_store {
@@ -323,7 +223,7 @@ impl MetaRestoreCommand {
    }
 }

-pub struct MetaRestoreTool {
+struct MetaRestoreTool {
    inner: MetadataSnapshotManager,
    source_file: String,
    force: bool,
@@ -372,9 +272,12 @@ impl Tool for MetaRestoreTool {
    }
 }

-/// Explore metadata from metadata snapshot.
+/// Explore metadata from a snapshot file.
+///
+/// This command allows filtering the metadata by a specific key and limiting the number of results.
+/// It prints the filtered metadata to the console.
 #[derive(Debug, Default, Parser)]
-pub struct MetaInfoCommand {
+pub struct InfoCommand {
    /// The s3 config.
    #[clap(flatten)]
    s3_config: S3Config,
@@ -389,7 +292,7 @@ pub struct MetaInfoCommand {
    limit: Option<usize>,
 }

-pub struct MetaInfoTool {
+struct MetaInfoTool {
    inner: ObjectStore,
    source_file: String,
    inspect_key: String,
@@ -398,6 +301,7 @@ pub struct MetaInfoTool {

 #[async_trait]
 impl Tool for MetaInfoTool {
+    #[allow(clippy::print_stdout)]
    async fn do_work(&self) -> std::result::Result<(), BoxedError> {
        let result = MetadataSnapshotManager::info(
            &self.inner,
@@ -414,7 +318,7 @@ impl Tool for MetaInfoTool {
    }
 }

-impl MetaInfoCommand {
+impl InfoCommand {
    fn decide_object_store_root_for_local_store(
        file_path: &str,
    ) -> Result<(&str, &str), BoxedError> {
--- a/src/cli/src/metadata/utils.rs
+++ b/src/cli/src/metadata/utils.rs
@@ -0,0 +1,178 @@
+// Copyright 2023 Greptime Team
+//
+// Licensed under the Apache License, Version 2.0 (the "License");
+// you may not use this file except in compliance with the License.
+// You may obtain a copy of the License at
+//
+//     http://www.apache.org/licenses/LICENSE-2.0
+//
+// Unless required by applicable law or agreed to in writing, software
+// distributed under the License is distributed on an "AS IS" BASIS,
+// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+// See the License for the specific language governing permissions and
+// limitations under the License.
+
+use std::collections::VecDeque;
+
+use async_stream::try_stream;
+use common_catalog::consts::METRIC_ENGINE;
+use common_catalog::format_full_table_name;
+use common_meta::key::table_name::TableNameKey;
+use common_meta::key::table_route::TableRouteValue;
+use common_meta::key::TableMetadataManager;
+use common_meta::kv_backend::KvBackendRef;
+use futures::Stream;
+use snafu::{OptionExt, ResultExt};
+use store_api::storage::TableId;
+use table::metadata::RawTableInfo;
+
+use crate::error::{Result, TableMetadataSnafu, UnexpectedSnafu};
+
+/// The input for the iterator.
+pub enum IteratorInput {
+    TableIds(VecDeque<TableId>),
+    TableNames(VecDeque<(String, String, String)>),
+}
+
+impl IteratorInput {
+    /// Creates a new iterator input from a list of table ids.
+    pub fn new_table_ids(table_ids: Vec<TableId>) -> Self {
+        Self::TableIds(table_ids.into())
+    }
+
+    /// Creates a new iterator input from a list of table names.
+    pub fn new_table_names(table_names: Vec<(String, String, String)>) -> Self {
+        Self::TableNames(table_names.into())
+    }
+}
+
+/// An iterator for retrieving table metadata from the metadata store.
+///
+/// This struct provides functionality to iterate over table metadata based on
+/// either [`TableId`] and their associated regions or fully qualified table names.
+pub struct TableMetadataIterator {
+    input: IteratorInput,
+    table_metadata_manager: TableMetadataManager,
+}
+
+/// The full table metadata.
+pub struct FullTableMetadata {
+    pub table_id: TableId,
+    pub table_info: RawTableInfo,
+    pub table_route: TableRouteValue,
+}
+
+impl FullTableMetadata {
+    /// Returns true if it's [TableRouteValue::Physical].
+    pub fn is_physical_table(&self) -> bool {
+        self.table_route.is_physical()
+    }
+
+    /// Returns true if it's a metric engine table.
+    pub fn is_metric_engine(&self) -> bool {
+        self.table_info.meta.engine == METRIC_ENGINE
+    }
+
+    /// Returns the full table name.
+    pub fn full_table_name(&self) -> String {
+        format_full_table_name(
+            &self.table_info.catalog_name,
+            &self.table_info.schema_name,
+            &self.table_info.name,
+        )
+    }
+}
+
+impl TableMetadataIterator {
+    pub fn new(kvbackend: KvBackendRef, input: IteratorInput) -> Self {
+        let table_metadata_manager = TableMetadataManager::new(kvbackend);
+        Self {
+            input,
+            table_metadata_manager,
+        }
+    }
+
+    /// Returns the next table metadata.
+    ///
+    /// This method handles two types of inputs:
+    /// - TableIds: Returns metadata for a specific [`TableId`].
+    /// - TableNames: Returns metadata for a table identified by its full name (catalog.schema.table).
+    ///
+    /// Returns `None` when there are no more tables to process.
+    pub async fn next(&mut self) -> Result<Option<FullTableMetadata>> {
+        match &mut self.input {
+            IteratorInput::TableIds(table_ids) => {
+                if let Some(table_id) = table_ids.pop_front() {
+                    let full_table_metadata = self.get_table_metadata(table_id).await?;
+                    return Ok(Some(full_table_metadata));
+                }
+            }
+
+            IteratorInput::TableNames(table_names) => {
+                if let Some(full_table_name) = table_names.pop_front() {
+                    let table_id = self.get_table_id_by_name(full_table_name).await?;
+                    let full_table_metadata = self.get_table_metadata(table_id).await?;
+                    return Ok(Some(full_table_metadata));
+                }
+            }
+        }
+
+        Ok(None)
+    }
+
+    /// Converts the iterator into a stream of table metadata.
+    pub fn into_stream(mut self) -> impl Stream<Item = Result<FullTableMetadata>> {
+        try_stream!({
+            while let Some(full_table_metadata) = self.next().await? {
+                yield full_table_metadata;
+            }
+        })
+    }
+
+    async fn get_table_id_by_name(
+        &mut self,
+        (catalog_name, schema_name, table_name): (String, String, String),
+    ) -> Result<TableId> {
+        let key = TableNameKey::new(&catalog_name, &schema_name, &table_name);
+        let table_id = self
+            .table_metadata_manager
+            .table_name_manager()
+            .get(key)
+            .await
+            .context(TableMetadataSnafu)?
+            .with_context(|| UnexpectedSnafu {
+                msg: format!(
+                    "Table not found: {}",
+                    format_full_table_name(&catalog_name, &schema_name, &table_name)
+                ),
+            })?
+            .table_id();
+        Ok(table_id)
+    }
+
+    async fn get_table_metadata(&mut self, table_id: TableId) -> Result<FullTableMetadata> {
+        let (table_info, table_route) = self
+            .table_metadata_manager
+            .get_full_table_info(table_id)
+            .await
+            .context(TableMetadataSnafu)?;
+
+        let table_info = table_info
+            .with_context(|| UnexpectedSnafu {
+                msg: format!("Table info not found for table id: {table_id}"),
+            })?
+            .into_inner()
+            .table_info;
+        let table_route = table_route
+            .with_context(|| UnexpectedSnafu {
+                msg: format!("Table route not found for table id: {table_id}"),
+            })?
+            .into_inner();
+
+        Ok(FullTableMetadata {
+            table_id,
+            table_info,
+            table_route,
+        })
+    }
+}
--- a/src/client/src/flow.rs
+++ b/src/client/src/flow.rs
@@ -12,7 +12,7 @@
 // See the License for the specific language governing permissions and
 // limitations under the License.

-use api::v1::flow::{FlowRequest, FlowResponse};
+use api::v1::flow::{DirtyWindowRequest, DirtyWindowRequests, FlowRequest, FlowResponse};
 use api::v1::region::InsertRequests;
 use common_error::ext::BoxedError;
 use common_meta::node_manager::Flownode;
@@ -44,6 +44,16 @@ impl Flownode for FlowRequester {
            .map_err(BoxedError::new)
            .context(common_meta::error::ExternalSnafu)
    }
+
+    async fn handle_mark_window_dirty(
+        &self,
+        req: DirtyWindowRequest,
+    ) -> common_meta::error::Result<FlowResponse> {
+        self.handle_mark_window_dirty(req)
+            .await
+            .map_err(BoxedError::new)
+            .context(common_meta::error::ExternalSnafu)
+    }
 }

 impl FlowRequester {
@@ -91,4 +101,20 @@ impl FlowRequester {
            .into_inner();
        Ok(response)
    }
+
+    async fn handle_mark_window_dirty(&self, req: DirtyWindowRequest) -> Result<FlowResponse> {
+        let (addr, mut client) = self.client.raw_flow_client()?;
+        let response = client
+            .handle_mark_dirty_time_window(DirtyWindowRequests {
+                requests: vec![req],
+            })
+            .await
+            .or_else(|e| {
+                let code = e.code();
+                let err: crate::error::Error = e.into();
+                Err(BoxedError::new(err)).context(FlowServerSnafu { addr, code })
+            })?
+            .into_inner();
+        Ok(response)
+    }
 }
--- a/src/cmd/Cargo.toml
+++ b/src/cmd/Cargo.toml
@@ -67,6 +67,7 @@ metric-engine.workspace = true
 mito2.workspace = true
 moka.workspace = true
 nu-ansi-term = "0.46"
+object-store.workspace = true
 plugins.workspace = true
 prometheus.workspace = true
 prost.workspace = true
--- a/src/cmd/src/datanode.rs
+++ b/src/cmd/src/datanode.rs
@@ -280,7 +280,7 @@ mod tests {

    use common_config::ENV_VAR_SEP;
    use common_test_util::temp_dir::create_named_temp_file;
-    use datanode::config::{FileConfig, GcsConfig, ObjectStoreConfig, S3Config};
+    use object_store::config::{FileConfig, GcsConfig, ObjectStoreConfig, S3Config};
    use servers::heartbeat_options::HeartbeatOptions;

    use super::*;
--- a/src/cmd/src/datanode/builder.rs
+++ b/src/cmd/src/datanode/builder.rs
@@ -93,6 +93,7 @@ impl InstanceBuilder {
            MetaClientType::Datanode { member_id },
            meta_client_options,
            Some(&plugins),
+            None,
        )
        .await
        .context(MetaClientInitSnafu)?;
--- a/src/cmd/src/flownode.rs
+++ b/src/cmd/src/flownode.rs
@@ -55,14 +55,32 @@ type FlownodeOptions = GreptimeOptions<flow::FlownodeOptions>;
 pub struct Instance {
    flownode: FlownodeInstance,

+    // The components of flownode, which make it easier to expand based
+    // on the components.
+    #[cfg(feature = "enterprise")]
+    components: Components,
+
    // Keep the logging guard to prevent the worker from being dropped.
    _guard: Vec<WorkerGuard>,
 }

+#[cfg(feature = "enterprise")]
+pub struct Components {
+    pub catalog_manager: catalog::CatalogManagerRef,
+    pub fe_client: Arc<FrontendClient>,
+    pub kv_backend: common_meta::kv_backend::KvBackendRef,
+}
+
 impl Instance {
-    pub fn new(flownode: FlownodeInstance, guard: Vec<WorkerGuard>) -> Self {
+    pub fn new(
+        flownode: FlownodeInstance,
+        #[cfg(feature = "enterprise")] components: Components,
+        guard: Vec<WorkerGuard>,
+    ) -> Self {
        Self {
            flownode,
+            #[cfg(feature = "enterprise")]
+            components,
            _guard: guard,
        }
    }
@@ -75,6 +93,11 @@ impl Instance {
    pub fn flownode_mut(&mut self) -> &mut FlownodeInstance {
        &mut self.flownode
    }
+
+    #[cfg(feature = "enterprise")]
+    pub fn components(&self) -> &Components {
+        &self.components
+    }
 }

 #[async_trait::async_trait]
@@ -283,6 +306,7 @@ impl StartCommand {
            MetaClientType::Flownode { member_id },
            meta_config,
            None,
+            None,
        )
        .await
        .context(MetaClientInitSnafu)?;
@@ -323,6 +347,7 @@ impl StartCommand {
            cached_meta_backend.clone(),
            layered_cache_registry.clone(),
            None,
+            None,
        );

        let table_metadata_manager =
@@ -348,19 +373,20 @@ impl StartCommand {
        let flow_auth_header = get_flow_auth_options(&opts).context(StartFlownodeSnafu)?;
        let frontend_client =
            FrontendClient::from_meta_client(meta_client.clone(), flow_auth_header);
+        let frontend_client = Arc::new(frontend_client);
        let flownode_builder = FlownodeBuilder::new(
            opts.clone(),
            plugins,
            table_metadata_manager,
            catalog_manager.clone(),
            flow_metadata_manager,
-            Arc::new(frontend_client),
+            frontend_client.clone(),
        )
        .with_heartbeat_task(heartbeat_task);

        let mut flownode = flownode_builder.build().await.context(StartFlownodeSnafu)?;
        let services = FlownodeServiceBuilder::new(&opts)
-            .with_grpc_server(flownode.flownode_server().clone())
+            .with_default_grpc_server(flownode.flownode_server())
            .enable_http_service()
            .build()
            .context(StartFlownodeSnafu)?;
@@ -392,6 +418,16 @@ impl StartCommand {
            .set_frontend_invoker(invoker)
            .await;

-        Ok(Instance::new(flownode, guard))
+        #[cfg(feature = "enterprise")]
+        let components = Components {
+            catalog_manager: catalog_manager.clone(),
+            fe_client: frontend_client,
+            kv_backend: cached_meta_backend,
+        };
+
+        #[cfg(not(feature = "enterprise"))]
+        return Ok(Instance::new(flownode, guard));
+        #[cfg(feature = "enterprise")]
+        Ok(Instance::new(flownode, components, guard))
    }
 }
--- a/src/cmd/src/frontend.rs
+++ b/src/cmd/src/frontend.rs
@@ -20,6 +20,7 @@ use async_trait::async_trait;
 use cache::{build_fundamental_cache_registry, with_default_composite_cache_registry};
 use catalog::information_extension::DistributedInformationExtension;
 use catalog::kvbackend::{CachedKvBackendBuilder, KvBackendCatalogManager, MetaKvBackend};
+use catalog::process_manager::ProcessManager;
 use clap::Parser;
 use client::client_manager::NodeClients;
 use common_base::Plugins;
@@ -38,6 +39,7 @@ use frontend::heartbeat::HeartbeatTask;
 use frontend::instance::builder::FrontendBuilder;
 use frontend::server::Services;
 use meta_client::{MetaClientOptions, MetaClientType};
+use servers::addrs;
 use servers::export_metrics::ExportMetricsTask;
 use servers::tls::{TlsMode, TlsOption};
 use snafu::{OptionExt, ResultExt};
@@ -311,6 +313,7 @@ impl StartCommand {
            MetaClientType::Frontend,
            meta_client_options,
            Some(&plugins),
+            None,
        )
        .await
        .context(error::MetaClientInitSnafu)?;
@@ -342,11 +345,17 @@ impl StartCommand {

        let information_extension =
            Arc::new(DistributedInformationExtension::new(meta_client.clone()));
+
+        let process_manager = Arc::new(ProcessManager::new(
+            addrs::resolve_addr(&opts.grpc.bind_addr, Some(&opts.grpc.server_addr)),
+            Some(meta_client.clone()),
+        ));
        let catalog_manager = KvBackendCatalogManager::new(
            information_extension,
            cached_meta_backend.clone(),
            layered_cache_registry.clone(),
            None,
+            Some(process_manager.clone()),
        );

        let executor = HandlerGroupExecutor::new(vec![
@@ -383,6 +392,7 @@ impl StartCommand {
            catalog_manager,
            Arc::new(client),
            meta_client,
+            process_manager,
        )
        .with_plugin(plugins.clone())
        .with_local_cache_invalidator(layered_cache_registry)
--- a/src/cmd/src/standalone.rs
+++ b/src/cmd/src/standalone.rs
@@ -21,6 +21,7 @@ use async_trait::async_trait;
 use cache::{build_fundamental_cache_registry, with_default_composite_cache_registry};
 use catalog::information_schema::InformationExtension;
 use catalog::kvbackend::KvBackendCatalogManager;
+use catalog::process_manager::ProcessManager;
 use clap::Parser;
 use client::api::v1::meta::RegionRole;
 use common_base::readable_size::ReadableSize;
@@ -29,20 +30,16 @@ use common_catalog::consts::{MIN_USER_FLOW_ID, MIN_USER_TABLE_ID};
 use common_config::{metadata_store_dir, Configurable, KvBackendConfig};
 use common_error::ext::BoxedError;
 use common_meta::cache::LayeredCacheRegistryBuilder;
-use common_meta::cache_invalidator::CacheInvalidatorRef;
 use common_meta::cluster::{NodeInfo, NodeStatus};
 use common_meta::datanode::RegionStat;
-use common_meta::ddl::flow_meta::{FlowMetadataAllocator, FlowMetadataAllocatorRef};
-use common_meta::ddl::table_meta::{TableMetadataAllocator, TableMetadataAllocatorRef};
+use common_meta::ddl::flow_meta::FlowMetadataAllocator;
+use common_meta::ddl::table_meta::TableMetadataAllocator;
 use common_meta::ddl::{DdlContext, NoopRegionFailureDetectorControl, ProcedureExecutorRef};
 use common_meta::ddl_manager::DdlManager;
-#[cfg(feature = "enterprise")]
-use common_meta::ddl_manager::TriggerDdlManagerRef;
 use common_meta::key::flow::flow_state::FlowStat;
-use common_meta::key::flow::{FlowMetadataManager, FlowMetadataManagerRef};
+use common_meta::key::flow::FlowMetadataManager;
 use common_meta::key::{TableMetadataManager, TableMetadataManagerRef};
 use common_meta::kv_backend::KvBackendRef;
-use common_meta::node_manager::NodeManagerRef;
 use common_meta::peer::Peer;
 use common_meta::region_keeper::MemoryRegionKeeper;
 use common_meta::region_registry::LeaderRegionRegistry;
@@ -260,15 +257,34 @@ pub struct Instance {
    flownode: FlownodeInstance,
    procedure_manager: ProcedureManagerRef,
    wal_options_allocator: WalOptionsAllocatorRef,
+
+    // The components of standalone, which make it easier to expand based
+    // on the components.
+    #[cfg(feature = "enterprise")]
+    components: Components,
+
    // Keep the logging guard to prevent the worker from being dropped.
    _guard: Vec<WorkerGuard>,
 }

+#[cfg(feature = "enterprise")]
+pub struct Components {
+    pub plugins: Plugins,
+    pub kv_backend: KvBackendRef,
+    pub frontend_client: Arc<FrontendClient>,
+    pub catalog_manager: catalog::CatalogManagerRef,
+}
+
 impl Instance {
    /// Find the socket addr of a server by its `name`.
    pub fn server_addr(&self, name: &str) -> Option<SocketAddr> {
        self.frontend.server_handlers().addr(name)
    }
+
+    #[cfg(feature = "enterprise")]
+    pub fn components(&self) -> &Components {
+        &self.components
+    }
 }

 #[async_trait]
@@ -526,11 +542,14 @@ impl StartCommand {
            datanode.region_server(),
            procedure_manager.clone(),
        ));
+
+        let process_manager = Arc::new(ProcessManager::new(opts.grpc.server_addr.clone(), None));
        let catalog_manager = KvBackendCatalogManager::new(
            information_extension.clone(),
            kv_backend.clone(),
            layered_cache_registry.clone(),
            Some(procedure_manager.clone()),
+            Some(process_manager.clone()),
        );

        let table_metadata_manager =
@@ -546,13 +565,14 @@ impl StartCommand {
        // actually make a connection
        let (frontend_client, frontend_instance_handler) =
            FrontendClient::from_empty_grpc_handler();
+        let frontend_client = Arc::new(frontend_client);
        let flow_builder = FlownodeBuilder::new(
            flownode_options,
            plugins.clone(),
            table_metadata_manager.clone(),
            catalog_manager.clone(),
            flow_metadata_manager.clone(),
-            Arc::new(frontend_client.clone()),
+            frontend_client.clone(),
        );
        let flownode = flow_builder
            .build()
@@ -590,28 +610,36 @@ impl StartCommand {
            .await
            .context(error::BuildWalOptionsAllocatorSnafu)?;
        let wal_options_allocator = Arc::new(wal_options_allocator);
-        let table_meta_allocator = Arc::new(TableMetadataAllocator::new(
+        let table_metadata_allocator = Arc::new(TableMetadataAllocator::new(
            table_id_sequence,
            wal_options_allocator.clone(),
        ));
-        let flow_meta_allocator = Arc::new(FlowMetadataAllocator::with_noop_peer_allocator(
+        let flow_metadata_allocator = Arc::new(FlowMetadataAllocator::with_noop_peer_allocator(
            flow_id_sequence,
        ));

+        let ddl_context = DdlContext {
+            node_manager: node_manager.clone(),
+            cache_invalidator: layered_cache_registry.clone(),
+            memory_region_keeper: Arc::new(MemoryRegionKeeper::default()),
+            leader_region_registry: Arc::new(LeaderRegionRegistry::default()),
+            table_metadata_manager: table_metadata_manager.clone(),
+            table_metadata_allocator: table_metadata_allocator.clone(),
+            flow_metadata_manager: flow_metadata_manager.clone(),
+            flow_metadata_allocator: flow_metadata_allocator.clone(),
+            region_failure_detector_controller: Arc::new(NoopRegionFailureDetectorControl),
+        };
+        let procedure_manager_c = procedure_manager.clone();
+
+        let ddl_manager = DdlManager::try_new(ddl_context, procedure_manager_c, true)
+            .context(error::InitDdlManagerSnafu)?;
        #[cfg(feature = "enterprise")]
-        let trigger_ddl_manager: Option<TriggerDdlManagerRef> = plugins.get();
-        let ddl_task_executor = Self::create_ddl_task_executor(
-            procedure_manager.clone(),
-            node_manager.clone(),
-            layered_cache_registry.clone(),
-            table_metadata_manager,
-            table_meta_allocator,
-            flow_metadata_manager,
-            flow_meta_allocator,
-            #[cfg(feature = "enterprise")]
-            trigger_ddl_manager,
-        )
-        .await?;
+        let ddl_manager = {
+            let trigger_ddl_manager: Option<common_meta::ddl_manager::TriggerDdlManagerRef> =
+                plugins.get();
+            ddl_manager.with_trigger_ddl_manager(trigger_ddl_manager)
+        };
+        let ddl_task_executor: ProcedureExecutorRef = Arc::new(ddl_manager);

        let fe_instance = FrontendBuilder::new(
            fe_opts.clone(),
@@ -620,6 +648,7 @@ impl StartCommand {
            catalog_manager.clone(),
            node_manager.clone(),
            ddl_task_executor.clone(),
+            process_manager,
        )
        .with_plugin(plugins.clone())
        .try_build()
@@ -647,13 +676,13 @@ impl StartCommand {
            node_manager,
        )
        .await
-        .context(error::StartFlownodeSnafu)?;
+        .context(StartFlownodeSnafu)?;
        flow_streaming_engine.set_frontend_invoker(invoker).await;

        let export_metrics_task = ExportMetricsTask::try_new(&opts.export_metrics, Some(&plugins))
            .context(error::ServersSnafu)?;

-        let servers = Services::new(opts, fe_instance.clone(), plugins)
+        let servers = Services::new(opts, fe_instance.clone(), plugins.clone())
            .build()
            .context(error::StartFrontendSnafu)?;

@@ -664,51 +693,26 @@ impl StartCommand {
            export_metrics_task,
        };

+        #[cfg(feature = "enterprise")]
+        let components = Components {
+            plugins,
+            kv_backend,
+            frontend_client,
+            catalog_manager,
+        };
+
        Ok(Instance {
            datanode,
            frontend,
            flownode,
            procedure_manager,
            wal_options_allocator,
+            #[cfg(feature = "enterprise")]
+            components,
            _guard: guard,
        })
    }

-    #[allow(clippy::too_many_arguments)]
-    pub async fn create_ddl_task_executor(
-        procedure_manager: ProcedureManagerRef,
-        node_manager: NodeManagerRef,
-        cache_invalidator: CacheInvalidatorRef,
-        table_metadata_manager: TableMetadataManagerRef,
-        table_metadata_allocator: TableMetadataAllocatorRef,
-        flow_metadata_manager: FlowMetadataManagerRef,
-        flow_metadata_allocator: FlowMetadataAllocatorRef,
-        #[cfg(feature = "enterprise")] trigger_ddl_manager: Option<TriggerDdlManagerRef>,
-    ) -> Result<ProcedureExecutorRef> {
-        let procedure_executor: ProcedureExecutorRef = Arc::new(
-            DdlManager::try_new(
-                DdlContext {
-                    node_manager,
-                    cache_invalidator,
-                    memory_region_keeper: Arc::new(MemoryRegionKeeper::default()),
-                    leader_region_registry: Arc::new(LeaderRegionRegistry::default()),
-                    table_metadata_manager,
-                    table_metadata_allocator,
-                    flow_metadata_manager,
-                    flow_metadata_allocator,
-                    region_failure_detector_controller: Arc::new(NoopRegionFailureDetectorControl),
-                },
-                procedure_manager,
-                true,
-                #[cfg(feature = "enterprise")]
-                trigger_ddl_manager,
-            )
-            .context(error::InitDdlManagerSnafu)?,
-        );
-
-        Ok(procedure_executor)
-    }
-
    pub async fn create_table_metadata_manager(
        kv_backend: KvBackendRef,
    ) -> Result<TableMetadataManagerRef> {
@@ -844,7 +848,7 @@ mod tests {
    use common_config::ENV_VAR_SEP;
    use common_test_util::temp_dir::create_named_temp_file;
    use common_wal::config::DatanodeWalConfig;
-    use datanode::config::{FileConfig, GcsConfig};
+    use object_store::config::{FileConfig, GcsConfig};

    use super::*;
    use crate::options::GlobalOptions;
@@ -963,15 +967,15 @@ mod tests {

        assert!(matches!(
            &dn_opts.storage.store,
-            datanode::config::ObjectStoreConfig::File(FileConfig { .. })
+            object_store::config::ObjectStoreConfig::File(FileConfig { .. })
        ));
        assert_eq!(dn_opts.storage.providers.len(), 2);
        assert!(matches!(
            dn_opts.storage.providers[0],
-            datanode::config::ObjectStoreConfig::Gcs(GcsConfig { .. })
+            object_store::config::ObjectStoreConfig::Gcs(GcsConfig { .. })
        ));
        match &dn_opts.storage.providers[1] {
-            datanode::config::ObjectStoreConfig::S3(s3_config) => {
+            object_store::config::ObjectStoreConfig::S3(s3_config) => {
                assert_eq!(
                    "SecretBox<alloc::string::String>([REDACTED])".to_string(),
                    format!("{:?}", s3_config.access_key_id)
--- a/src/cmd/tests/load_config_test.rs
+++ b/src/cmd/tests/load_config_test.rs
@@ -12,14 +12,13 @@
 // See the License for the specific language governing permissions and
 // limitations under the License.

-use std::path::Path;
 use std::time::Duration;

 use cmd::options::GreptimeOptions;
 use cmd::standalone::StandaloneOptions;
 use common_config::{Configurable, DEFAULT_DATA_HOME};
 use common_options::datanode::{ClientOptions, DatanodeClientOptions};
-use common_telemetry::logging::{LoggingOptions, DEFAULT_LOGGING_DIR, DEFAULT_OTLP_ENDPOINT};
+use common_telemetry::logging::{LoggingOptions, DEFAULT_LOGGING_DIR, DEFAULT_OTLP_HTTP_ENDPOINT};
 use common_wal::config::raft_engine::RaftEngineConfig;
 use common_wal::config::DatanodeWalConfig;
 use datanode::config::{DatanodeOptions, RegionEngineConfig, StorageConfig};
@@ -58,12 +57,7 @@ fn test_load_datanode_example_config() {
                metadata_cache_tti: Duration::from_secs(300),
            }),
            wal: DatanodeWalConfig::RaftEngine(RaftEngineConfig {
-                dir: Some(
-                    Path::new(DEFAULT_DATA_HOME)
-                        .join(WAL_DIR)
-                        .to_string_lossy()
-                        .to_string(),
-                ),
+                dir: Some(format!("{}/{}", DEFAULT_DATA_HOME, WAL_DIR)),
                sync_period: Some(Duration::from_secs(10)),
                recovery_parallelism: 2,
                ..Default::default()
@@ -86,11 +80,8 @@ fn test_load_datanode_example_config() {
            ],
            logging: LoggingOptions {
                level: Some("info".to_string()),
-                dir: Path::new(DEFAULT_DATA_HOME)
-                    .join(DEFAULT_LOGGING_DIR)
-                    .to_string_lossy()
-                    .to_string(),
-                otlp_endpoint: Some(DEFAULT_OTLP_ENDPOINT.to_string()),
+                dir: format!("{}/{}", DEFAULT_DATA_HOME, DEFAULT_LOGGING_DIR),
+                otlp_endpoint: Some(DEFAULT_OTLP_HTTP_ENDPOINT.to_string()),
                tracing_sample_ratio: Some(Default::default()),
                ..Default::default()
            },
@@ -132,11 +123,8 @@ fn test_load_frontend_example_config() {
            }),
            logging: LoggingOptions {
                level: Some("info".to_string()),
-                dir: Path::new(DEFAULT_DATA_HOME)
-                    .join(DEFAULT_LOGGING_DIR)
-                    .to_string_lossy()
-                    .to_string(),
-                otlp_endpoint: Some(DEFAULT_OTLP_ENDPOINT.to_string()),
+                dir: format!("{}/{}", DEFAULT_DATA_HOME, DEFAULT_LOGGING_DIR),
+                otlp_endpoint: Some(DEFAULT_OTLP_HTTP_ENDPOINT.to_string()),
                tracing_sample_ratio: Some(Default::default()),
                ..Default::default()
            },
@@ -182,12 +170,9 @@ fn test_load_metasrv_example_config() {
                ..Default::default()
            },
            logging: LoggingOptions {
-                dir: Path::new(DEFAULT_DATA_HOME)
-                    .join(DEFAULT_LOGGING_DIR)
-                    .to_string_lossy()
-                    .to_string(),
+                dir: format!("{}/{}", DEFAULT_DATA_HOME, DEFAULT_LOGGING_DIR),
                level: Some("info".to_string()),
-                otlp_endpoint: Some(DEFAULT_OTLP_ENDPOINT.to_string()),
+                otlp_endpoint: Some(DEFAULT_OTLP_HTTP_ENDPOINT.to_string()),
                tracing_sample_ratio: Some(Default::default()),
                ..Default::default()
            },
@@ -220,12 +205,7 @@ fn test_load_standalone_example_config() {
        component: StandaloneOptions {
            default_timezone: Some("UTC".to_string()),
            wal: DatanodeWalConfig::RaftEngine(RaftEngineConfig {
-                dir: Some(
-                    Path::new(DEFAULT_DATA_HOME)
-                        .join(WAL_DIR)
-                        .to_string_lossy()
-                        .to_string(),
-                ),
+                dir: Some(format!("{}/{}", DEFAULT_DATA_HOME, WAL_DIR)),
                sync_period: Some(Duration::from_secs(10)),
                recovery_parallelism: 2,
                ..Default::default()
@@ -248,11 +228,8 @@ fn test_load_standalone_example_config() {
            },
            logging: LoggingOptions {
                level: Some("info".to_string()),
-                dir: Path::new(DEFAULT_DATA_HOME)
-                    .join(DEFAULT_LOGGING_DIR)
-                    .to_string_lossy()
-                    .to_string(),
-                otlp_endpoint: Some(DEFAULT_OTLP_ENDPOINT.to_string()),
+                dir: format!("{}/{}", DEFAULT_DATA_HOME, DEFAULT_LOGGING_DIR),
+                otlp_endpoint: Some(DEFAULT_OTLP_HTTP_ENDPOINT.to_string()),
                tracing_sample_ratio: Some(Default::default()),
                ..Default::default()
            },
--- a/src/common/base/src/cancellation.rs
+++ b/src/common/base/src/cancellation.rs
@@ -0,0 +1,240 @@
+// Copyright 2023 Greptime Team
+//
+// Licensed under the Apache License, Version 2.0 (the "License");
+// you may not use this file except in compliance with the License.
+// You may obtain a copy of the License at
+//
+//     http://www.apache.org/licenses/LICENSE-2.0
+//
+// Unless required by applicable law or agreed to in writing, software
+// distributed under the License is distributed on an "AS IS" BASIS,
+// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+// See the License for the specific language governing permissions and
+// limitations under the License.
+
+//! [CancellationHandle] is used to compose with manual implementation of [futures::future::Future]
+//! or [futures::stream::Stream] to facilitate cancellation.
+//! See example in [frontend::stream_wrapper::CancellableStreamWrapper] and [CancellableFuture].
+
+use std::fmt::{Debug, Display, Formatter};
+use std::future::Future;
+use std::pin::Pin;
+use std::sync::atomic::{AtomicBool, Ordering};
+use std::sync::Arc;
+use std::task::{Context, Poll};
+
+use futures::task::AtomicWaker;
+use pin_project::pin_project;
+
+#[derive(Default)]
+pub struct CancellationHandle {
+    waker: AtomicWaker,
+    cancelled: AtomicBool,
+}
+
+impl Debug for CancellationHandle {
+    fn fmt(&self, f: &mut Formatter<'_>) -> std::fmt::Result {
+        f.debug_struct("CancellationHandle")
+            .field("cancelled", &self.is_cancelled())
+            .finish()
+    }
+}
+
+impl CancellationHandle {
+    pub fn waker(&self) -> &AtomicWaker {
+        &self.waker
+    }
+
+    /// Cancels a future or stream.
+    pub fn cancel(&self) {
+        if self
+            .cancelled
+            .compare_exchange(false, true, Ordering::Acquire, Ordering::Relaxed)
+            .is_ok()
+        {
+            self.waker.wake();
+        }
+    }
+
+    /// Is this handle cancelled.
+    pub fn is_cancelled(&self) -> bool {
+        self.cancelled.load(Ordering::Relaxed)
+    }
+}
+
+#[pin_project]
+#[derive(Debug, Clone)]
+pub struct CancellableFuture<T> {
+    #[pin]
+    fut: T,
+    handle: Arc<CancellationHandle>,
+}
+
+impl<T> CancellableFuture<T> {
+    pub fn new(fut: T, handle: Arc<CancellationHandle>) -> Self {
+        Self { fut, handle }
+    }
+}
+
+impl<T> Future for CancellableFuture<T>
+where
+    T: Future,
+{
+    type Output = Result<T::Output, Cancelled>;
+
+    fn poll(mut self: Pin<&mut Self>, cx: &mut Context<'_>) -> Poll<Self::Output> {
+        let this = self.as_mut().project();
+        // Check if the task has been aborted
+        if this.handle.is_cancelled() {
+            return Poll::Ready(Err(Cancelled));
+        }
+
+        if let Poll::Ready(x) = this.fut.poll(cx) {
+            return Poll::Ready(Ok(x));
+        }
+
+        this.handle.waker().register(cx.waker());
+        if this.handle.is_cancelled() {
+            return Poll::Ready(Err(Cancelled));
+        }
+        Poll::Pending
+    }
+}
+
+#[derive(Copy, Clone, Debug)]
+pub struct Cancelled;
+
+impl Display for Cancelled {
+    fn fmt(&self, f: &mut Formatter<'_>) -> std::fmt::Result {
+        write!(f, "Future has been cancelled")
+    }
+}
+
+#[cfg(test)]
+mod tests {
+    use std::sync::Arc;
+    use std::time::Duration;
+
+    use tokio::time::{sleep, timeout};
+
+    use crate::cancellation::{CancellableFuture, CancellationHandle, Cancelled};
+
+    #[tokio::test]
+    async fn test_cancellable_future_completes_normally() {
+        let handle = Arc::new(CancellationHandle::default());
+        let future = async { 42 };
+        let cancellable = CancellableFuture::new(future, handle);
+
+        let result = cancellable.await;
+        assert!(result.is_ok());
+        assert_eq!(result.unwrap(), 42);
+    }
+
+    #[tokio::test]
+    async fn test_cancellable_future_cancelled_before_start() {
+        let handle = Arc::new(CancellationHandle::default());
+        handle.cancel();
+
+        let future = async { 42 };
+        let cancellable = CancellableFuture::new(future, handle);
+
+        let result = cancellable.await;
+        assert!(result.is_err());
+        assert!(matches!(result.unwrap_err(), Cancelled));
+    }
+
+    #[tokio::test]
+    async fn test_cancellable_future_cancelled_during_execution() {
+        let handle = Arc::new(CancellationHandle::default());
+        let handle_clone = handle.clone();
+
+        // Create a future that sleeps for a long time
+        let future = async {
+            sleep(Duration::from_secs(10)).await;
+            42
+        };
+        let cancellable = CancellableFuture::new(future, handle);
+
+        // Cancel the future after a short delay
+        tokio::spawn(async move {
+            sleep(Duration::from_millis(50)).await;
+            handle_clone.cancel();
+        });
+
+        let result = cancellable.await;
+        assert!(result.is_err());
+        assert!(matches!(result.unwrap_err(), Cancelled));
+    }
+
+    #[tokio::test]
+    async fn test_cancellable_future_completes_before_cancellation() {
+        let handle = Arc::new(CancellationHandle::default());
+        let handle_clone = handle.clone();
+
+        // Create a future that completes quickly
+        let future = async {
+            sleep(Duration::from_millis(10)).await;
+            42
+        };
+        let cancellable = CancellableFuture::new(future, handle);
+
+        // Try to cancel after the future should have completed
+        tokio::spawn(async move {
+            sleep(Duration::from_millis(100)).await;
+            handle_clone.cancel();
+        });
+
+        let result = cancellable.await;
+        assert!(result.is_ok());
+        assert_eq!(result.unwrap(), 42);
+    }
+
+    #[tokio::test]
+    async fn test_cancellation_handle_is_cancelled() {
+        let handle = CancellationHandle::default();
+        assert!(!handle.is_cancelled());
+
+        handle.cancel();
+        assert!(handle.is_cancelled());
+    }
+
+    #[tokio::test]
+    async fn test_multiple_cancellable_futures_with_same_handle() {
+        let handle = Arc::new(CancellationHandle::default());
+
+        let future1 = CancellableFuture::new(async { 1 }, handle.clone());
+        let future2 = CancellableFuture::new(async { 2 }, handle.clone());
+
+        // Cancel before starting
+        handle.cancel();
+
+        let (result1, result2) = tokio::join!(future1, future2);
+
+        assert!(result1.is_err());
+        assert!(result2.is_err());
+        assert!(matches!(result1.unwrap_err(), Cancelled));
+        assert!(matches!(result2.unwrap_err(), Cancelled));
+    }
+
+    #[tokio::test]
+    async fn test_cancellable_future_with_timeout() {
+        let handle = Arc::new(CancellationHandle::default());
+        let future = async {
+            sleep(Duration::from_secs(1)).await;
+            42
+        };
+        let cancellable = CancellableFuture::new(future, handle.clone());
+
+        // Use timeout to ensure the test doesn't hang
+        let result = timeout(Duration::from_millis(100), cancellable).await;
+
+        // Should timeout because the future takes 1 second but we timeout after 100ms
+        assert!(result.is_err());
+    }
+
+    #[tokio::test]
+    async fn test_cancelled_display() {
+        let cancelled = Cancelled;
+        assert_eq!(format!("{}", cancelled), "Future has been cancelled");
+    }
+}
--- a/src/common/base/src/lib.rs
+++ b/src/common/base/src/lib.rs
@@ -14,6 +14,7 @@

 pub mod bit_vec;
 pub mod bytes;
+pub mod cancellation;
 pub mod plugins;
 pub mod range_read;
 #[allow(clippy::all)]
--- a/src/common/catalog/src/consts.rs
+++ b/src/common/catalog/src/consts.rs
@@ -102,6 +102,8 @@ pub const INFORMATION_SCHEMA_FLOW_TABLE_ID: u32 = 33;
 pub const INFORMATION_SCHEMA_PROCEDURE_INFO_TABLE_ID: u32 = 34;
 /// id for information_schema.region_statistics
 pub const INFORMATION_SCHEMA_REGION_STATISTICS_TABLE_ID: u32 = 35;
+/// id for information_schema.process_list
+pub const INFORMATION_SCHEMA_PROCESS_LIST_TABLE_ID: u32 = 36;

 // ----- End of information_schema tables -----

--- a/src/common/config/Cargo.toml
+++ b/src/common/config/Cargo.toml
@@ -14,6 +14,7 @@ common-macro.workspace = true
 config.workspace = true
 humantime-serde.workspace = true
 num_cpus.workspace = true
+object-store.workspace = true
 serde.workspace = true
 serde_json.workspace = true
 serde_with.workspace = true
--- a/src/common/config/src/config.rs
+++ b/src/common/config/src/config.rs
@@ -106,7 +106,7 @@ mod tests {
    use common_telemetry::logging::LoggingOptions;
    use common_test_util::temp_dir::create_named_temp_file;
    use common_wal::config::DatanodeWalConfig;
-    use datanode::config::{ObjectStoreConfig, StorageConfig};
+    use datanode::config::StorageConfig;
    use meta_client::MetaClientOptions;
    use serde::{Deserialize, Serialize};

@@ -212,7 +212,7 @@ mod tests {

                // Check the configs from environment variables.
                match &opts.storage.store {
-                    ObjectStoreConfig::S3(s3_config) => {
+                    object_store::config::ObjectStoreConfig::S3(s3_config) => {
                        assert_eq!(s3_config.bucket, "mybucket".to_string());
                    }
                    _ => panic!("unexpected store type"),
--- a/src/common/datasource/src/lib.rs
+++ b/src/common/datasource/src/lib.rs
@@ -21,6 +21,7 @@ pub mod error;
 pub mod file_format;
 pub mod lister;
 pub mod object_store;
+pub mod parquet_writer;
 pub mod share_buffer;
 #[cfg(test)]
 pub mod test_util;
--- a/src/common/datasource/src/parquet_writer.rs
+++ b/src/common/datasource/src/parquet_writer.rs
@@ -0,0 +1,52 @@
+// Copyright 2023 Greptime Team
+//
+// Licensed under the Apache License, Version 2.0 (the "License");
+// you may not use this file except in compliance with the License.
+// You may obtain a copy of the License at
+//
+//     http://www.apache.org/licenses/LICENSE-2.0
+//
+// Unless required by applicable law or agreed to in writing, software
+// distributed under the License is distributed on an "AS IS" BASIS,
+// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+// See the License for the specific language governing permissions and
+// limitations under the License.
+
+use bytes::Bytes;
+use futures::future::BoxFuture;
+use object_store::Writer;
+use parquet::arrow::async_writer::AsyncFileWriter;
+use parquet::errors::ParquetError;
+
+/// Bridges opendal [Writer] with parquet [AsyncFileWriter].
+pub struct AsyncWriter {
+    inner: Writer,
+}
+
+impl AsyncWriter {
+    /// Create a [`AsyncWriter`] by given [`Writer`].
+    pub fn new(writer: Writer) -> Self {
+        Self { inner: writer }
+    }
+}
+
+impl AsyncFileWriter for AsyncWriter {
+    fn write(&mut self, bs: Bytes) -> BoxFuture<'_, parquet::errors::Result<()>> {
+        Box::pin(async move {
+            self.inner
+                .write(bs)
+                .await
+                .map_err(|err| ParquetError::External(Box::new(err)))
+        })
+    }
+
+    fn complete(&mut self) -> BoxFuture<'_, parquet::errors::Result<()>> {
+        Box::pin(async move {
+            self.inner
+                .close()
+                .await
+                .map(|_| ())
+                .map_err(|err| ParquetError::External(Box::new(err)))
+        })
+    }
+}
--- a/src/common/frontend/Cargo.toml
+++ b/src/common/frontend/Cargo.toml
@@ -7,5 +7,13 @@ license.workspace = true
 [dependencies]
 async-trait.workspace = true
 common-error.workspace = true
+common-grpc.workspace = true
 common-macro.workspace = true
+common-meta.workspace = true
+greptime-proto.workspace = true
+meta-client.workspace = true
 snafu.workspace = true
+tonic.workspace = true
+
+[dev-dependencies]
+tokio.workspace = true
--- a/src/common/frontend/src/error.rs
+++ b/src/common/frontend/src/error.rs
@@ -27,6 +27,35 @@ pub enum Error {
        location: Location,
        source: BoxedError,
    },
+
+    #[snafu(display("Failed to list nodes from metasrv"))]
+    Meta {
+        source: Box<meta_client::error::Error>,
+        #[snafu(implicit)]
+        location: Location,
+    },
+
+    #[snafu(display("Failed to parse process id: {}", s))]
+    ParseProcessId {
+        s: String,
+        #[snafu(implicit)]
+        location: Location,
+    },
+
+    #[snafu(display("Failed to invoke frontend service"))]
+    InvokeFrontend {
+        #[snafu(source)]
+        error: tonic::Status,
+        #[snafu(implicit)]
+        location: Location,
+    },
+
+    #[snafu(display("Failed to invoke list process service"))]
+    CreateChannel {
+        source: common_grpc::error::Error,
+        #[snafu(implicit)]
+        location: Location,
+    },
 }

 pub type Result<T> = std::result::Result<T, Error>;
@@ -36,6 +65,10 @@ impl ErrorExt for Error {
        use Error::*;
        match self {
            External { source, .. } => source.status_code(),
+            Meta { source, .. } => source.status_code(),
+            ParseProcessId { .. } => StatusCode::InvalidArguments,
+            InvokeFrontend { .. } => StatusCode::Unexpected,
+            CreateChannel { source, .. } => source.status_code(),
        }
    }

--- a/src/common/frontend/src/lib.rs
+++ b/src/common/frontend/src/lib.rs
@@ -12,4 +12,41 @@
 // See the License for the specific language governing permissions and
 // limitations under the License.

+use std::fmt::{Display, Formatter};
+use std::str::FromStr;
+
+use snafu::OptionExt;
+
 pub mod error;
+pub mod selector;
+
+#[derive(Debug, Clone, Eq, PartialEq)]
+pub struct DisplayProcessId {
+    pub server_addr: String,
+    pub id: u32,
+}
+
+impl Display for DisplayProcessId {
+    fn fmt(&self, f: &mut Formatter<'_>) -> std::fmt::Result {
+        write!(f, "{}/{}", self.server_addr, self.id)
+    }
+}
+
+impl TryFrom<&str> for DisplayProcessId {
+    type Error = error::Error;
+
+    fn try_from(value: &str) -> Result<Self, Self::Error> {
+        let mut split = value.split('/');
+        let server_addr = split
+            .next()
+            .context(error::ParseProcessIdSnafu { s: value })?
+            .to_string();
+        let id = split
+            .next()
+            .context(error::ParseProcessIdSnafu { s: value })?;
+        let id = u32::from_str(id)
+            .ok()
+            .context(error::ParseProcessIdSnafu { s: value })?;
+        Ok(DisplayProcessId { server_addr, id })
+    }
+}
--- a/src/common/frontend/src/selector.rs
+++ b/src/common/frontend/src/selector.rs
@@ -0,0 +1,112 @@
+// Copyright 2023 Greptime Team
+//
+// Licensed under the Apache License, Version 2.0 (the "License");
+// you may not use this file except in compliance with the License.
+// You may obtain a copy of the License at
+//
+//     http://www.apache.org/licenses/LICENSE-2.0
+//
+// Unless required by applicable law or agreed to in writing, software
+// distributed under the License is distributed on an "AS IS" BASIS,
+// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+// See the License for the specific language governing permissions and
+// limitations under the License.
+
+use std::time::Duration;
+
+use common_grpc::channel_manager::{ChannelConfig, ChannelManager};
+use common_meta::cluster::{ClusterInfo, NodeInfo, Role};
+use greptime_proto::v1::frontend::{
+    frontend_client, KillProcessRequest, KillProcessResponse, ListProcessRequest,
+    ListProcessResponse,
+};
+use meta_client::MetaClientRef;
+use snafu::ResultExt;
+use tonic::Response;
+
+use crate::error;
+use crate::error::{MetaSnafu, Result};
+
+pub type FrontendClientPtr = Box<dyn FrontendClient>;
+
+#[async_trait::async_trait]
+pub trait FrontendClient: Send {
+    async fn list_process(&mut self, req: ListProcessRequest) -> Result<ListProcessResponse>;
+
+    async fn kill_process(&mut self, req: KillProcessRequest) -> Result<KillProcessResponse>;
+}
+
+#[async_trait::async_trait]
+impl FrontendClient for frontend_client::FrontendClient<tonic::transport::channel::Channel> {
+    async fn list_process(&mut self, req: ListProcessRequest) -> Result<ListProcessResponse> {
+        frontend_client::FrontendClient::<tonic::transport::channel::Channel>::list_process(
+            self, req,
+        )
+        .await
+        .context(error::InvokeFrontendSnafu)
+        .map(Response::into_inner)
+    }
+
+    async fn kill_process(&mut self, req: KillProcessRequest) -> Result<KillProcessResponse> {
+        frontend_client::FrontendClient::<tonic::transport::channel::Channel>::kill_process(
+            self, req,
+        )
+        .await
+        .context(error::InvokeFrontendSnafu)
+        .map(Response::into_inner)
+    }
+}
+
+#[async_trait::async_trait]
+pub trait FrontendSelector {
+    async fn select<F>(&self, predicate: F) -> Result<Vec<FrontendClientPtr>>
+    where
+        F: Fn(&NodeInfo) -> bool + Send;
+}
+
+#[derive(Debug, Clone)]
+pub struct MetaClientSelector {
+    meta_client: MetaClientRef,
+    channel_manager: ChannelManager,
+}
+
+#[async_trait::async_trait]
+impl FrontendSelector for MetaClientSelector {
+    async fn select<F>(&self, predicate: F) -> Result<Vec<FrontendClientPtr>>
+    where
+        F: Fn(&NodeInfo) -> bool + Send,
+    {
+        let nodes = self
+            .meta_client
+            .list_nodes(Some(Role::Frontend))
+            .await
+            .map_err(Box::new)
+            .context(MetaSnafu)?;
+
+        nodes
+            .into_iter()
+            .filter(predicate)
+            .map(|node| {
+                let channel = self
+                    .channel_manager
+                    .get(node.peer.addr)
+                    .context(error::CreateChannelSnafu)?;
+                let client = frontend_client::FrontendClient::new(channel);
+                Ok(Box::new(client) as FrontendClientPtr)
+            })
+            .collect::<Result<Vec<_>>>()
+    }
+}
+
+impl MetaClientSelector {
+    pub fn new(meta_client: MetaClientRef) -> Self {
+        let cfg = ChannelConfig::new()
+            .connect_timeout(Duration::from_secs(30))
+            .timeout(Duration::from_secs(30));
+        let channel_manager = ChannelManager::with_config(cfg);
+        Self {
+            meta_client,
+            channel_manager,
+        }
+    }
+}
--- a/src/common/function/Cargo.toml
+++ b/src/common/function/Cargo.toml
@@ -33,6 +33,7 @@ common-version.workspace = true
 datafusion.workspace = true
 datafusion-common.workspace = true
 datafusion-expr.workspace = true
+datafusion-functions-aggregate-common.workspace = true
 datatypes.workspace = true
 derive_more = { version = "1", default-features = false, features = ["display"] }
 geo = { version = "0.29", optional = true }
--- a/src/common/function/src/aggrs.rs
+++ b/src/common/function/src/aggrs.rs
@@ -13,6 +13,7 @@
 // limitations under the License.

 pub mod approximate;
+pub mod count_hash;
 #[cfg(feature = "geo")]
 pub mod geo;
 pub mod vector;
--- a/src/common/function/src/aggrs/approximate.rs
+++ b/src/common/function/src/aggrs/approximate.rs
@@ -14,8 +14,8 @@

 use crate::function_registry::FunctionRegistry;

-pub(crate) mod hll;
-mod uddsketch;
+pub mod hll;
+pub mod uddsketch;

 pub(crate) struct ApproximateFunction;

--- a/src/common/function/src/aggrs/count_hash.rs
+++ b/src/common/function/src/aggrs/count_hash.rs
@@ -0,0 +1,647 @@
+// Copyright 2023 Greptime Team
+//
+// Licensed under the Apache License, Version 2.0 (the "License");
+// you may not use this file except in compliance with the License.
+// You may obtain a copy of the License at
+//
+//     http://www.apache.org/licenses/LICENSE-2.0
+//
+// Unless required by applicable law or agreed to in writing, software
+// distributed under the License is distributed on an "AS IS" BASIS,
+// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+// See the License for the specific language governing permissions and
+// limitations under the License.
+
+//! `CountHash` / `count_hash` is a hash-based approximate distinct count function.
+//!
+//! It is a variant of `CountDistinct` that uses a hash function to approximate the
+//! distinct count.
+//! It is designed to be more efficient than `CountDistinct` for large datasets,
+//! but it is not as accurate, as the hash value may be collision.
+
+use std::collections::HashSet;
+use std::fmt::Debug;
+use std::sync::Arc;
+
+use ahash::RandomState;
+use datafusion_common::cast::as_list_array;
+use datafusion_common::error::Result;
+use datafusion_common::hash_utils::create_hashes;
+use datafusion_common::utils::SingleRowListArrayBuilder;
+use datafusion_common::{internal_err, not_impl_err, ScalarValue};
+use datafusion_expr::function::{AccumulatorArgs, StateFieldsArgs};
+use datafusion_expr::utils::{format_state_name, AggregateOrderSensitivity};
+use datafusion_expr::{
+    Accumulator, AggregateUDF, AggregateUDFImpl, EmitTo, GroupsAccumulator, ReversedUDAF,
+    SetMonotonicity, Signature, TypeSignature, Volatility,
+};
+use datafusion_functions_aggregate_common::aggregate::groups_accumulator::nulls::filtered_null_mask;
+use datatypes::arrow;
+use datatypes::arrow::array::{
+    Array, ArrayRef, AsArray, BooleanArray, Int64Array, ListArray, UInt64Array,
+};
+use datatypes::arrow::buffer::{OffsetBuffer, ScalarBuffer};
+use datatypes::arrow::datatypes::{DataType, Field};
+
+use crate::function_registry::FunctionRegistry;
+
+type HashValueType = u64;
+
+// read from /dev/urandom 4047821dc6144e4b2abddf23ad4171126a52eeecd26eff2191cf673b965a7875
+const RANDOM_SEED_0: u64 = 0x4047821dc6144e4b;
+const RANDOM_SEED_1: u64 = 0x2abddf23ad417112;
+const RANDOM_SEED_2: u64 = 0x6a52eeecd26eff21;
+const RANDOM_SEED_3: u64 = 0x91cf673b965a7875;
+
+impl CountHash {
+    pub fn register(registry: &FunctionRegistry) {
+        registry.register_aggr(CountHash::udf_impl());
+    }
+
+    pub fn udf_impl() -> AggregateUDF {
+        AggregateUDF::new_from_impl(CountHash {
+            signature: Signature::one_of(
+                vec![TypeSignature::VariadicAny, TypeSignature::Nullary],
+                Volatility::Immutable,
+            ),
+        })
+    }
+}
+
+#[derive(Debug, Clone)]
+pub struct CountHash {
+    signature: Signature,
+}
+
+impl AggregateUDFImpl for CountHash {
+    fn as_any(&self) -> &dyn std::any::Any {
+        self
+    }
+
+    fn name(&self) -> &str {
+        "count_hash"
+    }
+
+    fn signature(&self) -> &Signature {
+        &self.signature
+    }
+
+    fn return_type(&self, _arg_types: &[DataType]) -> Result<DataType> {
+        Ok(DataType::Int64)
+    }
+
+    fn is_nullable(&self) -> bool {
+        false
+    }
+
+    fn state_fields(&self, args: StateFieldsArgs) -> Result<Vec<Field>> {
+        Ok(vec![Field::new_list(
+            format_state_name(args.name, "count_hash"),
+            Field::new_list_field(DataType::UInt64, true),
+            // For count_hash accumulator, null list item stands for an
+            // empty value set (i.e., all NULL value so far for that group).
+            true,
+        )])
+    }
+
+    fn accumulator(&self, acc_args: AccumulatorArgs) -> Result<Box<dyn Accumulator>> {
+        if acc_args.exprs.len() > 1 {
+            return not_impl_err!("count_hash with multiple arguments");
+        }
+
+        Ok(Box::new(CountHashAccumulator {
+            values: HashSet::default(),
+            random_state: RandomState::with_seeds(
+                RANDOM_SEED_0,
+                RANDOM_SEED_1,
+                RANDOM_SEED_2,
+                RANDOM_SEED_3,
+            ),
+            batch_hashes: vec![],
+        }))
+    }
+
+    fn aliases(&self) -> &[String] {
+        &[]
+    }
+
+    fn groups_accumulator_supported(&self, _args: AccumulatorArgs) -> bool {
+        true
+    }
+
+    fn create_groups_accumulator(
+        &self,
+        args: AccumulatorArgs,
+    ) -> Result<Box<dyn GroupsAccumulator>> {
+        if args.exprs.len() > 1 {
+            return not_impl_err!("count_hash with multiple arguments");
+        }
+
+        Ok(Box::new(CountHashGroupAccumulator::new()))
+    }
+
+    fn reverse_expr(&self) -> ReversedUDAF {
+        ReversedUDAF::Identical
+    }
+
+    fn order_sensitivity(&self) -> AggregateOrderSensitivity {
+        AggregateOrderSensitivity::Insensitive
+    }
+
+    fn default_value(&self, _data_type: &DataType) -> Result<ScalarValue> {
+        Ok(ScalarValue::Int64(Some(0)))
+    }
+
+    fn set_monotonicity(&self, _data_type: &DataType) -> SetMonotonicity {
+        SetMonotonicity::Increasing
+    }
+}
+
+/// GroupsAccumulator for `count_hash` aggregate function
+#[derive(Debug)]
+pub struct CountHashGroupAccumulator {
+    /// One HashSet per group to track distinct values
+    distinct_sets: Vec<HashSet<HashValueType, RandomState>>,
+    random_state: RandomState,
+    batch_hashes: Vec<HashValueType>,
+}
+
+impl Default for CountHashGroupAccumulator {
+    fn default() -> Self {
+        Self::new()
+    }
+}
+
+impl CountHashGroupAccumulator {
+    pub fn new() -> Self {
+        Self {
+            distinct_sets: vec![],
+            random_state: RandomState::with_seeds(
+                RANDOM_SEED_0,
+                RANDOM_SEED_1,
+                RANDOM_SEED_2,
+                RANDOM_SEED_3,
+            ),
+            batch_hashes: vec![],
+        }
+    }
+
+    fn ensure_sets(&mut self, total_num_groups: usize) {
+        if self.distinct_sets.len() < total_num_groups {
+            self.distinct_sets
+                .resize_with(total_num_groups, HashSet::default);
+        }
+    }
+}
+
+impl GroupsAccumulator for CountHashGroupAccumulator {
+    fn update_batch(
+        &mut self,
+        values: &[ArrayRef],
+        group_indices: &[usize],
+        opt_filter: Option<&BooleanArray>,
+        total_num_groups: usize,
+    ) -> Result<()> {
+        assert_eq!(values.len(), 1, "count_hash expects a single argument");
+        self.ensure_sets(total_num_groups);
+
+        let array = &values[0];
+        self.batch_hashes.clear();
+        self.batch_hashes.resize(array.len(), 0);
+        let hashes = create_hashes(
+            &[ArrayRef::clone(array)],
+            &self.random_state,
+            &mut self.batch_hashes,
+        )?;
+
+        // Use a pattern similar to accumulate_indices to process rows
+        // that are not null and pass the filter
+        let nulls = array.logical_nulls();
+
+        match (nulls.as_ref(), opt_filter) {
+            (None, None) => {
+                // No nulls, no filter - process all rows
+                for (row_idx, &group_idx) in group_indices.iter().enumerate() {
+                    self.distinct_sets[group_idx].insert(hashes[row_idx]);
+                }
+            }
+            (Some(nulls), None) => {
+                // Has nulls, no filter
+                for (row_idx, (&group_idx, is_valid)) in
+                    group_indices.iter().zip(nulls.iter()).enumerate()
+                {
+                    if is_valid {
+                        self.distinct_sets[group_idx].insert(hashes[row_idx]);
+                    }
+                }
+            }
+            (None, Some(filter)) => {
+                // No nulls, has filter
+                for (row_idx, (&group_idx, filter_value)) in
+                    group_indices.iter().zip(filter.iter()).enumerate()
+                {
+                    if let Some(true) = filter_value {
+                        self.distinct_sets[group_idx].insert(hashes[row_idx]);
+                    }
+                }
+            }
+            (Some(nulls), Some(filter)) => {
+                // Has nulls and filter
+                let iter = filter
+                    .iter()
+                    .zip(group_indices.iter())
+                    .zip(nulls.iter())
+                    .enumerate();
+
+                for (row_idx, ((filter_value, &group_idx), is_valid)) in iter {
+                    if is_valid && filter_value == Some(true) {
+                        self.distinct_sets[group_idx].insert(hashes[row_idx]);
+                    }
+                }
+            }
+        }
+
+        Ok(())
+    }
+
+    fn evaluate(&mut self, emit_to: EmitTo) -> Result<ArrayRef> {
+        let distinct_sets: Vec<HashSet<u64, RandomState>> =
+            emit_to.take_needed(&mut self.distinct_sets);
+
+        let counts = distinct_sets
+            .iter()
+            .map(|set| set.len() as i64)
+            .collect::<Vec<_>>();
+        Ok(Arc::new(Int64Array::from(counts)))
+    }
+
+    fn merge_batch(
+        &mut self,
+        values: &[ArrayRef],
+        group_indices: &[usize],
+        _opt_filter: Option<&BooleanArray>,
+        total_num_groups: usize,
+    ) -> Result<()> {
+        assert_eq!(
+            values.len(),
+            1,
+            "count_hash merge expects a single state array"
+        );
+        self.ensure_sets(total_num_groups);
+
+        let list_array = as_list_array(&values[0])?;
+
+        // For each group in the incoming batch
+        for (i, &group_idx) in group_indices.iter().enumerate() {
+            if i < list_array.len() {
+                let inner_array = list_array.value(i);
+                let inner_array = inner_array.as_any().downcast_ref::<UInt64Array>().unwrap();
+                // Add each value to our set for this group
+                for j in 0..inner_array.len() {
+                    if !inner_array.is_null(j) {
+                        self.distinct_sets[group_idx].insert(inner_array.value(j));
+                    }
+                }
+            }
+        }
+
+        Ok(())
+    }
+
+    fn state(&mut self, emit_to: EmitTo) -> Result<Vec<ArrayRef>> {
+        let distinct_sets: Vec<HashSet<u64, RandomState>> =
+            emit_to.take_needed(&mut self.distinct_sets);
+
+        let mut offsets = Vec::with_capacity(distinct_sets.len() + 1);
+        offsets.push(0);
+        let mut curr_len = 0i32;
+
+        let mut value_iter = distinct_sets
+            .into_iter()
+            .flat_map(|set| {
+                // build offset
+                curr_len += set.len() as i32;
+                offsets.push(curr_len);
+                // convert into iter
+                set.into_iter()
+            })
+            .peekable();
+        let data_array: ArrayRef = if value_iter.peek().is_none() {
+            arrow::array::new_empty_array(&DataType::UInt64) as _
+        } else {
+            Arc::new(UInt64Array::from_iter_values(value_iter))
+        };
+        let offset_buffer = OffsetBuffer::new(ScalarBuffer::from(offsets));
+
+        let list_array = ListArray::new(
+            Arc::new(Field::new_list_field(DataType::UInt64, true)),
+            offset_buffer,
+            data_array,
+            None,
+        );
+
+        Ok(vec![Arc::new(list_array) as _])
+    }
+
+    fn convert_to_state(
+        &self,
+        values: &[ArrayRef],
+        opt_filter: Option<&BooleanArray>,
+    ) -> Result<Vec<ArrayRef>> {
+        // For a single hash value per row, create a list array with that value
+        assert_eq!(values.len(), 1, "count_hash expects a single argument");
+        let values = ArrayRef::clone(&values[0]);
+
+        let offsets = OffsetBuffer::new(ScalarBuffer::from_iter(0..values.len() as i32 + 1));
+        let nulls = filtered_null_mask(opt_filter, &values);
+        let list_array = ListArray::new(
+            Arc::new(Field::new_list_field(DataType::UInt64, true)),
+            offsets,
+            values,
+            nulls,
+        );
+
+        Ok(vec![Arc::new(list_array)])
+    }
+
+    fn supports_convert_to_state(&self) -> bool {
+        true
+    }
+
+    fn size(&self) -> usize {
+        // Base size of the struct
+        let mut size = size_of::<Self>();
+
+        // Size of the vector holding the HashSets
+        size += size_of::<Vec<HashSet<HashValueType, RandomState>>>()
+            + self.distinct_sets.capacity() * size_of::<HashSet<HashValueType, RandomState>>();
+
+        // Estimate HashSet contents size more efficiently
+        // Instead of iterating through all values which is expensive, use an approximation
+        for set in &self.distinct_sets {
+            // Base size of the HashSet
+            size += set.capacity() * size_of::<HashValueType>();
+        }
+
+        size
+    }
+}
+
+#[derive(Debug)]
+struct CountHashAccumulator {
+    values: HashSet<HashValueType, RandomState>,
+    random_state: RandomState,
+    batch_hashes: Vec<HashValueType>,
+}
+
+impl CountHashAccumulator {
+    // calculating the size for fixed length values, taking first batch size *
+    // number of batches.
+    fn fixed_size(&self) -> usize {
+        size_of_val(self) + (size_of::<HashValueType>() * self.values.capacity())
+    }
+}
+
+impl Accumulator for CountHashAccumulator {
+    /// Returns the distinct values seen so far as (one element) ListArray.
+    fn state(&mut self) -> Result<Vec<ScalarValue>> {
+        let values = self.values.iter().cloned().collect::<Vec<_>>();
+        let arr = Arc::new(UInt64Array::from(values)) as _;
+        let list_scalar = SingleRowListArrayBuilder::new(arr).build_list_scalar();
+        Ok(vec![list_scalar])
+    }
+
+    fn update_batch(&mut self, values: &[ArrayRef]) -> Result<()> {
+        if values.is_empty() {
+            return Ok(());
+        }
+
+        let arr = &values[0];
+        if arr.data_type() == &DataType::Null {
+            return Ok(());
+        }
+
+        self.batch_hashes.clear();
+        self.batch_hashes.resize(arr.len(), 0);
+        let hashes = create_hashes(
+            &[ArrayRef::clone(arr)],
+            &self.random_state,
+            &mut self.batch_hashes,
+        )?;
+        for hash in hashes.as_slice() {
+            self.values.insert(*hash);
+        }
+        Ok(())
+    }
+
+    /// Merges multiple sets of distinct values into the current set.
+    ///
+    /// The input to this function is a `ListArray` with **multiple** rows,
+    /// where each row contains the values from a partial aggregate's phase (e.g.
+    /// the result of calling `Self::state` on multiple accumulators).
+    fn merge_batch(&mut self, states: &[ArrayRef]) -> Result<()> {
+        if states.is_empty() {
+            return Ok(());
+        }
+        assert_eq!(states.len(), 1, "array_agg states must be singleton!");
+        let array = &states[0];
+        let list_array = array.as_list::<i32>();
+        for inner_array in list_array.iter() {
+            let Some(inner_array) = inner_array else {
+                return internal_err!(
+                    "Intermediate results of count_hash should always be non null"
+                );
+            };
+            let hash_array = inner_array.as_any().downcast_ref::<UInt64Array>().unwrap();
+            for i in 0..hash_array.len() {
+                self.values.insert(hash_array.value(i));
+            }
+        }
+        Ok(())
+    }
+
+    fn evaluate(&mut self) -> Result<ScalarValue> {
+        Ok(ScalarValue::Int64(Some(self.values.len() as i64)))
+    }
+
+    fn size(&self) -> usize {
+        self.fixed_size()
+    }
+}
+
+#[cfg(test)]
+mod tests {
+    use datatypes::arrow::array::{Array, BooleanArray, Int32Array, Int64Array};
+
+    use super::*;
+
+    fn create_test_accumulator() -> CountHashAccumulator {
+        CountHashAccumulator {
+            values: HashSet::default(),
+            random_state: RandomState::with_seeds(
+                RANDOM_SEED_0,
+                RANDOM_SEED_1,
+                RANDOM_SEED_2,
+                RANDOM_SEED_3,
+            ),
+            batch_hashes: vec![],
+        }
+    }
+
+    #[test]
+    fn test_count_hash_accumulator() -> Result<()> {
+        let mut acc = create_test_accumulator();
+
+        // Test with some data
+        let array = Arc::new(Int32Array::from(vec![
+            Some(1),
+            Some(2),
+            Some(3),
+            Some(1),
+            Some(2),
+            None,
+        ])) as ArrayRef;
+        acc.update_batch(&[array])?;
+        let result = acc.evaluate()?;
+        assert_eq!(result, ScalarValue::Int64(Some(4)));
+
+        // Test with empty data
+        let mut acc = create_test_accumulator();
+        let array = Arc::new(Int32Array::from(vec![] as Vec<Option<i32>>)) as ArrayRef;
+        acc.update_batch(&[array])?;
+        let result = acc.evaluate()?;
+        assert_eq!(result, ScalarValue::Int64(Some(0)));
+
+        // Test with only nulls
+        let mut acc = create_test_accumulator();
+        let array = Arc::new(Int32Array::from(vec![None, None, None])) as ArrayRef;
+        acc.update_batch(&[array])?;
+        let result = acc.evaluate()?;
+        assert_eq!(result, ScalarValue::Int64(Some(1)));
+
+        Ok(())
+    }
+
+    #[test]
+    fn test_count_hash_accumulator_merge() -> Result<()> {
+        // Accumulator 1
+        let mut acc1 = create_test_accumulator();
+        let array1 = Arc::new(Int32Array::from(vec![Some(1), Some(2), Some(3)])) as ArrayRef;
+        acc1.update_batch(&[array1])?;
+        let state1 = acc1.state()?;
+
+        // Accumulator 2
+        let mut acc2 = create_test_accumulator();
+        let array2 = Arc::new(Int32Array::from(vec![Some(3), Some(4), Some(5)])) as ArrayRef;
+        acc2.update_batch(&[array2])?;
+        let state2 = acc2.state()?;
+
+        // Merge state1 and state2 into a new accumulator
+        let mut acc_merged = create_test_accumulator();
+        let state_array1 = state1[0].to_array()?;
+        let state_array2 = state2[0].to_array()?;
+
+        acc_merged.merge_batch(&[state_array1])?;
+        acc_merged.merge_batch(&[state_array2])?;
+
+        let result = acc_merged.evaluate()?;
+        // Distinct values are {1, 2, 3, 4, 5}, so count is 5
+        assert_eq!(result, ScalarValue::Int64(Some(5)));
+
+        Ok(())
+    }
+
+    fn create_test_group_accumulator() -> CountHashGroupAccumulator {
+        CountHashGroupAccumulator::new()
+    }
+
+    #[test]
+    fn test_count_hash_group_accumulator() -> Result<()> {
+        let mut acc = create_test_group_accumulator();
+        let values = Arc::new(Int32Array::from(vec![1, 2, 1, 3, 2, 4, 5])) as ArrayRef;
+        let group_indices = vec![0, 1, 0, 0, 1, 2, 0];
+        let total_num_groups = 3;
+
+        acc.update_batch(&[values], &group_indices, None, total_num_groups)?;
+
+        let result_array = acc.evaluate(EmitTo::All)?;
+        let result = result_array.as_any().downcast_ref::<Int64Array>().unwrap();
+
+        // Group 0: {1, 3, 5} -> 3
+        // Group 1: {2} -> 1
+        // Group 2: {4} -> 1
+        assert_eq!(result.value(0), 3);
+        assert_eq!(result.value(1), 1);
+        assert_eq!(result.value(2), 1);
+
+        Ok(())
+    }
+
+    #[test]
+    fn test_count_hash_group_accumulator_with_filter() -> Result<()> {
+        let mut acc = create_test_group_accumulator();
+        let values = Arc::new(Int32Array::from(vec![1, 2, 3, 4, 5, 6])) as ArrayRef;
+        let group_indices = vec![0, 0, 1, 1, 2, 2];
+        let filter = BooleanArray::from(vec![true, false, true, true, false, true]);
+        let total_num_groups = 3;
+
+        acc.update_batch(&[values], &group_indices, Some(&filter), total_num_groups)?;
+
+        let result_array = acc.evaluate(EmitTo::All)?;
+        let result = result_array.as_any().downcast_ref::<Int64Array>().unwrap();
+
+        // Group 0: {1} (2 is filtered out) -> 1
+        // Group 1: {3, 4} -> 2
+        // Group 2: {6} (5 is filtered out) -> 1
+        assert_eq!(result.value(0), 1);
+        assert_eq!(result.value(1), 2);
+        assert_eq!(result.value(2), 1);
+
+        Ok(())
+    }
+
+    #[test]
+    fn test_count_hash_group_accumulator_merge() -> Result<()> {
+        // Accumulator 1
+        let mut acc1 = create_test_group_accumulator();
+        let values1 = Arc::new(Int32Array::from(vec![1, 2, 3, 4])) as ArrayRef;
+        let group_indices1 = vec![0, 0, 1, 1];
+        acc1.update_batch(&[values1], &group_indices1, None, 2)?;
+        // acc1 state: group 0 -> {1, 2}, group 1 -> {3, 4}
+        let state1 = acc1.state(EmitTo::All)?;
+
+        // Accumulator 2
+        let mut acc2 = create_test_group_accumulator();
+        let values2 = Arc::new(Int32Array::from(vec![5, 6, 1, 3])) as ArrayRef;
+        // Merge into different group indices
+        let group_indices2 = vec![2, 2, 0, 1];
+        acc2.update_batch(&[values2], &group_indices2, None, 3)?;
+        // acc2 state: group 0 -> {1}, group 1 -> {3}, group 2 -> {5, 6}
+
+        // Merge state from acc1 into acc2
+        // We will merge acc1's group 0 into acc2's group 0
+        // and acc1's group 1 into acc2's group 2
+        let merge_group_indices = vec![0, 2];
+        acc2.merge_batch(&state1, &merge_group_indices, None, 3)?;
+
+        let result_array = acc2.evaluate(EmitTo::All)?;
+        let result = result_array.as_any().downcast_ref::<Int64Array>().unwrap();
+
+        // Final state of acc2:
+        // Group 0: {1} U {1, 2} -> {1, 2}, count = 2
+        // Group 1: {3}, count = 1
+        // Group 2: {5, 6} U {3, 4} -> {3, 4, 5, 6}, count = 4
+        assert_eq!(result.value(0), 2);
+        assert_eq!(result.value(1), 1);
+        assert_eq!(result.value(2), 4);
+
+        Ok(())
+    }
+
+    #[test]
+    fn test_size() {
+        let acc = create_test_group_accumulator();
+        // Just test it doesn't crash and returns a value.
+        assert!(acc.size() > 0);
+    }
+}
--- a/src/common/function/src/function_registry.rs
+++ b/src/common/function/src/function_registry.rs
@@ -21,6 +21,7 @@ use once_cell::sync::Lazy;

 use crate::admin::AdminFunction;
 use crate::aggrs::approximate::ApproximateFunction;
+use crate::aggrs::count_hash::CountHash;
 use crate::aggrs::vector::VectorFunction as VectorAggrFunction;
 use crate::function::{AsyncFunctionRef, Function, FunctionRef};
 use crate::function_factory::ScalarFunctionFactory;
@@ -144,6 +145,9 @@ pub static FUNCTION_REGISTRY: Lazy<Arc<FunctionRegistry>> = Lazy::new(|| {
    // Approximate functions
    ApproximateFunction::register(&function_registry);

+    // CountHash function
+    CountHash::register(&function_registry);
+
    Arc::new(function_registry)
 });

--- a/src/common/function/src/system.rs
+++ b/src/common/function/src/system.rs
@@ -23,7 +23,8 @@ use std::sync::Arc;

 use build::BuildFunction;
 use database::{
-    CurrentSchemaFunction, DatabaseFunction, ReadPreferenceFunction, SessionUserFunction,
+    ConnectionIdFunction, CurrentSchemaFunction, DatabaseFunction, PgBackendPidFunction,
+    ReadPreferenceFunction, SessionUserFunction,
 };
 use pg_catalog::PGCatalogFunction;
 use procedure_state::ProcedureStateFunction;
@@ -42,6 +43,8 @@ impl SystemFunction {
        registry.register_scalar(DatabaseFunction);
        registry.register_scalar(SessionUserFunction);
        registry.register_scalar(ReadPreferenceFunction);
+        registry.register_scalar(PgBackendPidFunction);
+        registry.register_scalar(ConnectionIdFunction);
        registry.register_scalar(TimezoneFunction);
        registry.register_async(Arc::new(ProcedureStateFunction));
        PGCatalogFunction::register(registry);
--- a/src/common/function/src/system/database.rs
+++ b/src/common/function/src/system/database.rs
@@ -18,7 +18,8 @@ use std::sync::Arc;
 use common_query::error::Result;
 use common_query::prelude::{Signature, Volatility};
 use datatypes::prelude::{ConcreteDataType, ScalarVector};
-use datatypes::vectors::{StringVector, VectorRef};
+use datatypes::vectors::{StringVector, UInt32Vector, VectorRef};
+use derive_more::Display;

 use crate::function::{Function, FunctionContext};

@@ -32,10 +33,20 @@ pub struct SessionUserFunction;

 pub struct ReadPreferenceFunction;

+#[derive(Display)]
+#[display("{}", self.name())]
+pub struct PgBackendPidFunction;
+
+#[derive(Display)]
+#[display("{}", self.name())]
+pub struct ConnectionIdFunction;
+
 const DATABASE_FUNCTION_NAME: &str = "database";
 const CURRENT_SCHEMA_FUNCTION_NAME: &str = "current_schema";
 const SESSION_USER_FUNCTION_NAME: &str = "session_user";
 const READ_PREFERENCE_FUNCTION_NAME: &str = "read_preference";
+const PG_BACKEND_PID: &str = "pg_backend_pid";
+const CONNECTION_ID: &str = "connection_id";

 impl Function for DatabaseFunction {
    fn name(&self) -> &str {
@@ -117,6 +128,46 @@ impl Function for ReadPreferenceFunction {
    }
 }

+impl Function for PgBackendPidFunction {
+    fn name(&self) -> &str {
+        PG_BACKEND_PID
+    }
+
+    fn return_type(&self, _input_types: &[ConcreteDataType]) -> Result<ConcreteDataType> {
+        Ok(ConcreteDataType::uint64_datatype())
+    }
+
+    fn signature(&self) -> Signature {
+        Signature::nullary(Volatility::Immutable)
+    }
+
+    fn eval(&self, func_ctx: &FunctionContext, _columns: &[VectorRef]) -> Result<VectorRef> {
+        let pid = func_ctx.query_ctx.process_id();
+
+        Ok(Arc::new(UInt32Vector::from_slice([pid])) as _)
+    }
+}
+
+impl Function for ConnectionIdFunction {
+    fn name(&self) -> &str {
+        CONNECTION_ID
+    }
+
+    fn return_type(&self, _input_types: &[ConcreteDataType]) -> Result<ConcreteDataType> {
+        Ok(ConcreteDataType::uint64_datatype())
+    }
+
+    fn signature(&self) -> Signature {
+        Signature::nullary(Volatility::Immutable)
+    }
+
+    fn eval(&self, func_ctx: &FunctionContext, _columns: &[VectorRef]) -> Result<VectorRef> {
+        let pid = func_ctx.query_ctx.process_id();
+
+        Ok(Arc::new(UInt32Vector::from_slice([pid])) as _)
+    }
+}
+
 impl fmt::Display for DatabaseFunction {
    fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result {
        write!(f, "DATABASE")
--- a/src/common/grpc-expr/src/alter.rs
+++ b/src/common/grpc-expr/src/alter.rs
@@ -180,6 +180,22 @@ pub fn alter_expr_to_request(table_id: TableId, expr: AlterTableExpr) -> Result<
            },
            None => return MissingAlterIndexOptionSnafu.fail(),
        },
+        Kind::DropDefaults(o) => {
+            let names = o
+                .drop_defaults
+                .into_iter()
+                .map(|col| {
+                    ensure!(
+                        !col.column_name.is_empty(),
+                        MissingFieldSnafu {
+                            field: "column_name"
+                        }
+                    );
+                    Ok(col.column_name)
+                })
+                .collect::<Result<Vec<_>>>()?;
+            AlterKind::DropDefaults { names }
+        }
    };

    let request = AlterTableRequest {
--- a/src/common/grpc/src/channel_manager.rs
+++ b/src/common/grpc/src/channel_manager.rs
@@ -201,8 +201,8 @@ impl ChannelManager {
            "http"
        };

-        let mut endpoint =
-            Endpoint::new(format!("{http_prefix}://{addr}")).context(CreateChannelSnafu)?;
+        let mut endpoint = Endpoint::new(format!("{http_prefix}://{addr}"))
+            .context(CreateChannelSnafu { addr })?;

        if let Some(dur) = self.config().timeout {
            endpoint = endpoint.timeout(dur);
@@ -237,7 +237,7 @@ impl ChannelManager {
        if let Some(tls_config) = &self.inner.client_tls_config {
            endpoint = endpoint
                .tls_config(tls_config.clone())
-                .context(CreateChannelSnafu)?;
+                .context(CreateChannelSnafu { addr })?;
        }

        endpoint = endpoint
--- a/src/common/grpc/src/error.rs
+++ b/src/common/grpc/src/error.rs
@@ -52,8 +52,9 @@ pub enum Error {
        location: Location,
    },

-    #[snafu(display("Failed to create gRPC channel"))]
+    #[snafu(display("Failed to create gRPC channel from '{addr}'"))]
    CreateChannel {
+        addr: String,
        #[snafu(source)]
        error: tonic::transport::Error,
        #[snafu(implicit)]
--- a/src/common/meta/Cargo.toml
+++ b/src/common/meta/Cargo.toml
@@ -17,7 +17,7 @@ workspace = true
 anymap2 = "0.13.0"
 api.workspace = true
 async-recursion = "1.0"
-async-stream = "0.3"
+async-stream.workspace = true
 async-trait.workspace = true
 backon = { workspace = true, optional = true }
 base64.workspace = true
--- a/src/common/meta/src/ddl/alter_logical_tables.rs
+++ b/src/common/meta/src/ddl/alter_logical_tables.rs
@@ -25,6 +25,7 @@ use common_procedure::error::{FromJsonSnafu, Result as ProcedureResult, ToJsonSn
 use common_procedure::{Context, LockKey, Procedure, Status};
 use common_telemetry::{error, info, warn};
 use futures_util::future;
+pub use region_request::make_alter_region_request;
 use serde::{Deserialize, Serialize};
 use snafu::{ensure, ResultExt};
 use store_api::metadata::ColumnMetadata;
--- a/src/common/meta/src/ddl/alter_logical_tables/region_request.rs
+++ b/src/common/meta/src/ddl/alter_logical_tables/region_request.rs
@@ -12,20 +12,18 @@
 // See the License for the specific language governing permissions and
 // limitations under the License.

-use api::v1;
 use api::v1::alter_table_expr::Kind;
 use api::v1::region::{
    alter_request, region_request, AddColumn, AddColumns, AlterRequest, AlterRequests,
    RegionColumnDef, RegionRequest, RegionRequestHeader,
 };
+use api::v1::{self, AlterTableExpr};
 use common_telemetry::tracing_context::TracingContext;
 use store_api::storage::RegionId;

 use crate::ddl::alter_logical_tables::AlterLogicalTablesProcedure;
 use crate::error::Result;
-use crate::key::table_info::TableInfoValue;
 use crate::peer::Peer;
-use crate::rpc::ddl::AlterTableTask;
 use crate::rpc::router::{find_leader_regions, RegionRoute};

 impl AlterLogicalTablesProcedure {
@@ -62,34 +60,37 @@ impl AlterLogicalTablesProcedure {
        {
            for region_number in &regions_on_this_peer {
                let region_id = RegionId::new(table.table_info.ident.table_id, *region_number);
-                let request = self.make_alter_region_request(region_id, task, table)?;
+                let request = make_alter_region_request(
+                    region_id,
+                    &task.alter_table,
+                    table.table_info.ident.version,
+                );
                requests.push(request);
            }
        }

        Ok(AlterRequests { requests })
    }
+}

-    fn make_alter_region_request(
-        &self,
-        region_id: RegionId,
-        task: &AlterTableTask,
-        table: &TableInfoValue,
-    ) -> Result<AlterRequest> {
-        let region_id = region_id.as_u64();
-        let schema_version = table.table_info.ident.version;
-        let kind = match &task.alter_table.kind {
-            Some(Kind::AddColumns(add_columns)) => Some(alter_request::Kind::AddColumns(
-                to_region_add_columns(add_columns),
-            )),
-            _ => unreachable!(), // Safety: we have checked the kind in check_input_tasks
-        };
+/// Makes an alter region request.
+pub fn make_alter_region_request(
+    region_id: RegionId,
+    alter_table_expr: &AlterTableExpr,
+    schema_version: u64,
+) -> AlterRequest {
+    let region_id = region_id.as_u64();
+    let kind = match &alter_table_expr.kind {
+        Some(Kind::AddColumns(add_columns)) => Some(alter_request::Kind::AddColumns(
+            to_region_add_columns(add_columns),
+        )),
+        _ => unreachable!(), // Safety: we have checked the kind in check_input_tasks
+    };

-        Ok(AlterRequest {
-            region_id,
-            schema_version,
-            kind,
-        })
+    AlterRequest {
+        region_id,
+        schema_version,
+        kind,
    }
 }

--- a/src/common/meta/src/ddl/alter_table/region_request.rs
+++ b/src/common/meta/src/ddl/alter_table/region_request.rs
@@ -135,6 +135,7 @@ fn create_proto_alter_kind(
        Kind::UnsetTableOptions(v) => Ok(Some(alter_request::Kind::UnsetTableOptions(v.clone()))),
        Kind::SetIndex(v) => Ok(Some(alter_request::Kind::SetIndex(v.clone()))),
        Kind::UnsetIndex(v) => Ok(Some(alter_request::Kind::UnsetIndex(v.clone()))),
+        Kind::DropDefaults(v) => Ok(Some(alter_request::Kind::DropDefaults(v.clone()))),
    }
 }

--- a/src/common/meta/src/ddl/alter_table/update_metadata.rs
+++ b/src/common/meta/src/ddl/alter_table/update_metadata.rs
@@ -61,7 +61,8 @@ impl AlterTableProcedure {
            | AlterKind::SetTableOptions { .. }
            | AlterKind::UnsetTableOptions { .. }
            | AlterKind::SetIndex { .. }
-            | AlterKind::UnsetIndex { .. } => {}
+            | AlterKind::UnsetIndex { .. }
+            | AlterKind::DropDefaults { .. } => {}
        }

        Ok(new_info)
--- a/src/common/meta/src/ddl/create_logical_tables.rs
+++ b/src/common/meta/src/ddl/create_logical_tables.rs
@@ -25,6 +25,7 @@ use common_procedure::error::{FromJsonSnafu, Result as ProcedureResult, ToJsonSn
 use common_procedure::{Context as ProcedureContext, LockKey, Procedure, Status};
 use common_telemetry::{debug, error, warn};
 use futures::future;
+pub use region_request::create_region_request_builder;
 use serde::{Deserialize, Serialize};
 use snafu::{ensure, ResultExt};
 use store_api::metadata::ColumnMetadata;
--- a/src/common/meta/src/ddl/create_logical_tables/region_request.rs
+++ b/src/common/meta/src/ddl/create_logical_tables/region_request.rs
@@ -15,16 +15,16 @@
 use std::collections::HashMap;

 use api::v1::region::{region_request, CreateRequests, RegionRequest, RegionRequestHeader};
+use api::v1::CreateTableExpr;
 use common_telemetry::debug;
 use common_telemetry::tracing_context::TracingContext;
-use store_api::storage::RegionId;
+use store_api::storage::{RegionId, TableId};

 use crate::ddl::create_logical_tables::CreateLogicalTablesProcedure;
 use crate::ddl::create_table_template::{build_template, CreateRequestBuilder};
 use crate::ddl::utils::region_storage_path;
 use crate::error::Result;
 use crate::peer::Peer;
-use crate::rpc::ddl::CreateTableTask;
 use crate::rpc::router::{find_leader_regions, RegionRoute};

 impl CreateLogicalTablesProcedure {
@@ -45,13 +45,15 @@ impl CreateLogicalTablesProcedure {
            let catalog = &create_table_expr.catalog_name;
            let schema = &create_table_expr.schema_name;
            let logical_table_id = task.table_info.ident.table_id;
+            let physical_table_id = self.data.physical_table_id;
            let storage_path = region_storage_path(catalog, schema);
-            let request_builder = self.create_region_request_builder(task)?;
+            let request_builder =
+                create_region_request_builder(&task.create_table, physical_table_id)?;

            for region_number in &regions_on_this_peer {
                let region_id = RegionId::new(logical_table_id, *region_number);
                let one_region_request =
-                    request_builder.build_one(region_id, storage_path.clone(), &HashMap::new())?;
+                    request_builder.build_one(region_id, storage_path.clone(), &HashMap::new());
                requests.push(one_region_request);
            }
        }
@@ -69,16 +71,13 @@ impl CreateLogicalTablesProcedure {
            body: Some(region_request::Body::Creates(CreateRequests { requests })),
        }))
    }
-
-    fn create_region_request_builder(
-        &self,
-        task: &CreateTableTask,
-    ) -> Result<CreateRequestBuilder> {
-        let create_expr = &task.create_table;
-        let template = build_template(create_expr)?;
-        Ok(CreateRequestBuilder::new(
-            template,
-            Some(self.data.physical_table_id),
-        ))
-    }
+}
+
+/// Creates a region request builder.
+pub fn create_region_request_builder(
+    create_table_expr: &CreateTableExpr,
+    physical_table_id: TableId,
+) -> Result<CreateRequestBuilder> {
+    let template = build_template(create_table_expr)?;
+    Ok(CreateRequestBuilder::new(template, Some(physical_table_id)))
 }
--- a/src/common/meta/src/ddl/create_table.rs
+++ b/src/common/meta/src/ddl/create_table.rs
@@ -218,11 +218,8 @@ impl CreateTableProcedure {
            let mut requests = Vec::with_capacity(regions.len());
            for region_number in regions {
                let region_id = RegionId::new(self.table_id(), region_number);
-                let create_region_request = request_builder.build_one(
-                    region_id,
-                    storage_path.clone(),
-                    region_wal_options,
-                )?;
+                let create_region_request =
+                    request_builder.build_one(region_id, storage_path.clone(), region_wal_options);
                requests.push(PbRegionRequest::Create(create_region_request));
            }

--- a/src/common/meta/src/ddl/create_table_template.rs
+++ b/src/common/meta/src/ddl/create_table_template.rs
@@ -105,12 +105,12 @@ impl CreateRequestBuilder {
        &self.template
    }

-    pub(crate) fn build_one(
+    pub fn build_one(
        &self,
        region_id: RegionId,
        storage_path: String,
        region_wal_options: &HashMap<RegionNumber, String>,
-    ) -> Result<CreateRequest> {
+    ) -> CreateRequest {
        let mut request = self.template.clone();

        request.region_id = region_id.as_u64();
@@ -130,6 +130,6 @@ impl CreateRequestBuilder {
            );
        }

-        Ok(request)
+        request
    }
 }
--- a/src/common/meta/src/ddl/drop_database/executor.rs
+++ b/src/common/meta/src/ddl/drop_database/executor.rs
@@ -13,7 +13,6 @@
 // limitations under the License.

 use std::any::Any;
-use std::collections::HashMap;

 use common_procedure::Status;
 use common_telemetry::info;
@@ -25,7 +24,7 @@ use table::table_name::TableName;
 use crate::ddl::drop_database::cursor::DropDatabaseCursor;
 use crate::ddl::drop_database::{DropDatabaseContext, DropTableTarget, State};
 use crate::ddl::drop_table::executor::DropTableExecutor;
-use crate::ddl::utils::extract_region_wal_options;
+use crate::ddl::utils::get_region_wal_options;
 use crate::ddl::DdlContext;
 use crate::error::{self, Result};
 use crate::key::table_route::TableRouteValue;
@@ -109,17 +108,12 @@ impl State for DropDatabaseExecutor {
        );

        // Deletes topic-region mapping if dropping physical table
-        let region_wal_options =
-            if let TableRouteValue::Physical(table_route_value) = &table_route_value {
-                let datanode_table_values = ddl_ctx
-                    .table_metadata_manager
-                    .datanode_table_manager()
-                    .regions(self.physical_table_id, table_route_value)
-                    .await?;
-                extract_region_wal_options(&datanode_table_values)?
-            } else {
-                HashMap::new()
-            };
+        let region_wal_options = get_region_wal_options(
+            &ddl_ctx.table_metadata_manager,
+            &table_route_value,
+            self.physical_table_id,
+        )
+        .await?;

        executor
            .on_destroy_metadata(ddl_ctx, &table_route_value, &region_wal_options)
--- a/src/common/meta/src/ddl/test_util/datanode_handler.rs
+++ b/src/common/meta/src/ddl/test_util/datanode_handler.rs
@@ -32,6 +32,7 @@ impl MockDatanodeHandler for () {
        Ok(RegionResponse {
            affected_rows: 0,
            extensions: Default::default(),
+            metadata: Vec::new(),
        })
    }

--- a/src/common/meta/src/ddl/utils.rs
+++ b/src/common/meta/src/ddl/utils.rs
@@ -42,7 +42,8 @@ use crate::error::{
 };
 use crate::key::datanode_table::DatanodeTableValue;
 use crate::key::table_name::TableNameKey;
-use crate::key::TableMetadataManagerRef;
+use crate::key::table_route::TableRouteValue;
+use crate::key::{TableMetadataManager, TableMetadataManagerRef};
 use crate::peer::Peer;
 use crate::rpc::ddl::CreateTableTask;
 use crate::rpc::router::{find_follower_regions, find_followers, RegionRoute};
@@ -187,6 +188,25 @@ pub fn parse_region_wal_options(
    Ok(region_wal_options)
 }

+/// Gets the wal options for a table.
+pub async fn get_region_wal_options(
+    table_metadata_manager: &TableMetadataManager,
+    table_route_value: &TableRouteValue,
+    physical_table_id: TableId,
+) -> Result<HashMap<RegionNumber, WalOptions>> {
+    let region_wal_options =
+        if let TableRouteValue::Physical(table_route_value) = &table_route_value {
+            let datanode_table_values = table_metadata_manager
+                .datanode_table_manager()
+                .regions(physical_table_id, table_route_value)
+                .await?;
+            extract_region_wal_options(&datanode_table_values)?
+        } else {
+            HashMap::new()
+        };
+    Ok(region_wal_options)
+}
+
 /// Extracts region wal options from [DatanodeTableValue]s.
 pub fn extract_region_wal_options(
    datanode_table_values: &Vec<DatanodeTableValue>,
--- a/src/common/meta/src/ddl_manager.rs
+++ b/src/common/meta/src/ddl_manager.rs
@@ -50,7 +50,11 @@ use crate::key::{DeserializedValueWithBytes, TableMetadataManagerRef};
 #[cfg(feature = "enterprise")]
 use crate::rpc::ddl::trigger::CreateTriggerTask;
 #[cfg(feature = "enterprise")]
+use crate::rpc::ddl::trigger::DropTriggerTask;
+#[cfg(feature = "enterprise")]
 use crate::rpc::ddl::DdlTask::CreateTrigger;
+#[cfg(feature = "enterprise")]
+use crate::rpc::ddl::DdlTask::DropTrigger;
 use crate::rpc::ddl::DdlTask::{
    AlterDatabase, AlterLogicalTables, AlterTable, CreateDatabase, CreateFlow, CreateLogicalTables,
    CreateTable, CreateView, DropDatabase, DropFlow, DropLogicalTables, DropTable, DropView,
@@ -91,6 +95,14 @@ pub trait TriggerDdlManager: Send + Sync {
        query_context: QueryContext,
    ) -> Result<SubmitDdlTaskResponse>;

+    async fn drop_trigger(
+        &self,
+        drop_trigger_task: DropTriggerTask,
+        procedure_manager: ProcedureManagerRef,
+        ddl_context: DdlContext,
+        query_context: QueryContext,
+    ) -> Result<SubmitDdlTaskResponse>;
+
    fn as_any(&self) -> &dyn std::any::Any;
 }

@@ -125,13 +137,12 @@ impl DdlManager {
        ddl_context: DdlContext,
        procedure_manager: ProcedureManagerRef,
        register_loaders: bool,
-        #[cfg(feature = "enterprise")] trigger_ddl_manager: Option<TriggerDdlManagerRef>,
    ) -> Result<Self> {
        let manager = Self {
            ddl_context,
            procedure_manager,
            #[cfg(feature = "enterprise")]
-            trigger_ddl_manager,
+            trigger_ddl_manager: None,
        };
        if register_loaders {
            manager.register_loaders()?;
@@ -139,6 +150,15 @@ impl DdlManager {
        Ok(manager)
    }

+    #[cfg(feature = "enterprise")]
+    pub fn with_trigger_ddl_manager(
+        mut self,
+        trigger_ddl_manager: Option<TriggerDdlManagerRef>,
+    ) -> Self {
+        self.trigger_ddl_manager = trigger_ddl_manager;
+        self
+    }
+
    /// Returns the [TableMetadataManagerRef].
    pub fn table_metadata_manager(&self) -> &TableMetadataManagerRef {
        &self.ddl_context.table_metadata_manager
@@ -640,6 +660,28 @@ async fn handle_drop_flow_task(
    })
 }

+#[cfg(feature = "enterprise")]
+async fn handle_drop_trigger_task(
+    ddl_manager: &DdlManager,
+    drop_trigger_task: DropTriggerTask,
+    query_context: QueryContext,
+) -> Result<SubmitDdlTaskResponse> {
+    let Some(m) = ddl_manager.trigger_ddl_manager.as_ref() else {
+        return UnsupportedSnafu {
+            operation: "drop trigger",
+        }
+        .fail();
+    };
+
+    m.drop_trigger(
+        drop_trigger_task,
+        ddl_manager.procedure_manager.clone(),
+        ddl_manager.ddl_context.clone(),
+        query_context,
+    )
+    .await
+}
+
 async fn handle_drop_view_task(
    ddl_manager: &DdlManager,
    drop_view_task: DropViewTask,
@@ -827,6 +869,11 @@ impl ProcedureExecutor for DdlManager {
                    handle_create_flow_task(self, create_flow_task, request.query_context.into())
                        .await
                }
+                DropFlow(drop_flow_task) => handle_drop_flow_task(self, drop_flow_task).await,
+                CreateView(create_view_task) => {
+                    handle_create_view_task(self, create_view_task).await
+                }
+                DropView(drop_view_task) => handle_drop_view_task(self, drop_view_task).await,
                #[cfg(feature = "enterprise")]
                CreateTrigger(create_trigger_task) => {
                    handle_create_trigger_task(
@@ -836,11 +883,11 @@ impl ProcedureExecutor for DdlManager {
                    )
                    .await
                }
-                DropFlow(drop_flow_task) => handle_drop_flow_task(self, drop_flow_task).await,
-                CreateView(create_view_task) => {
-                    handle_create_view_task(self, create_view_task).await
+                #[cfg(feature = "enterprise")]
+                DropTrigger(drop_trigger_task) => {
+                    handle_drop_trigger_task(self, drop_trigger_task, request.query_context.into())
+                        .await
                }
-                DropView(drop_view_task) => handle_drop_view_task(self, drop_view_task).await,
            }
        }
        .trace(span)
@@ -964,8 +1011,6 @@ mod tests {
            },
            procedure_manager.clone(),
            true,
-            #[cfg(feature = "enterprise")]
-            None,
        );

        let expected_loaders = vec![
--- a/src/common/meta/src/error.rs
+++ b/src/common/meta/src/error.rs
@@ -1001,7 +1001,7 @@ impl ErrorExt for Error {
            }
            #[cfg(any(feature = "pg_kvbackend", feature = "mysql_kvbackend"))]
            RdsTransactionRetryFailed { .. } => StatusCode::Internal,
-            Error::DatanodeTableInfoNotFound { .. } => StatusCode::Internal,
+            DatanodeTableInfoNotFound { .. } => StatusCode::Internal,
        }
    }

--- a/src/common/meta/src/key.rs
+++ b/src/common/meta/src/key.rs
@@ -109,7 +109,7 @@ pub mod table_name;
 pub mod table_route;
 #[cfg(any(test, feature = "testing"))]
 pub mod test_utils;
-mod tombstone;
+pub mod tombstone;
 pub mod topic_name;
 pub mod topic_region;
 pub mod txn_helper;
@@ -535,6 +535,29 @@ impl TableMetadataManager {
        }
    }

+    /// Creates a new `TableMetadataManager` with a custom tombstone prefix.
+    pub fn new_with_custom_tombstone_prefix(
+        kv_backend: KvBackendRef,
+        tombstone_prefix: &str,
+    ) -> Self {
+        Self {
+            table_name_manager: TableNameManager::new(kv_backend.clone()),
+            table_info_manager: TableInfoManager::new(kv_backend.clone()),
+            view_info_manager: ViewInfoManager::new(kv_backend.clone()),
+            datanode_table_manager: DatanodeTableManager::new(kv_backend.clone()),
+            catalog_manager: CatalogManager::new(kv_backend.clone()),
+            schema_manager: SchemaManager::new(kv_backend.clone()),
+            table_route_manager: TableRouteManager::new(kv_backend.clone()),
+            tombstone_manager: TombstoneManager::new_with_prefix(
+                kv_backend.clone(),
+                tombstone_prefix,
+            ),
+            topic_name_manager: TopicNameManager::new(kv_backend.clone()),
+            topic_region_manager: TopicRegionManager::new(kv_backend.clone()),
+            kv_backend,
+        }
+    }
+
    pub async fn init(&self) -> Result<()> {
        let catalog_name = CatalogNameKey::new(DEFAULT_CATALOG_NAME);

@@ -925,7 +948,7 @@ impl TableMetadataManager {
    ) -> Result<()> {
        let keys =
            self.table_metadata_keys(table_id, table_name, table_route_value, region_wal_options)?;
-        self.tombstone_manager.create(keys).await
+        self.tombstone_manager.create(keys).await.map(|_| ())
    }

    /// Deletes metadata tombstone for table **permanently**.
@@ -939,7 +962,10 @@ impl TableMetadataManager {
    ) -> Result<()> {
        let table_metadata_keys =
            self.table_metadata_keys(table_id, table_name, table_route_value, region_wal_options)?;
-        self.tombstone_manager.delete(table_metadata_keys).await
+        self.tombstone_manager
+            .delete(table_metadata_keys)
+            .await
+            .map(|_| ())
    }

    /// Restores metadata for table.
@@ -953,7 +979,7 @@ impl TableMetadataManager {
    ) -> Result<()> {
        let keys =
            self.table_metadata_keys(table_id, table_name, table_route_value, region_wal_options)?;
-        self.tombstone_manager.restore(keys).await
+        self.tombstone_manager.restore(keys).await.map(|_| ())
    }

    /// Deletes metadata for table **permanently**.
--- a/src/common/meta/src/key/table_route.rs
+++ b/src/common/meta/src/key/table_route.rs
@@ -48,6 +48,11 @@ impl TableRouteKey {
    pub fn new(table_id: TableId) -> Self {
        Self { table_id }
    }
+
+    /// Returns the range prefix of the table route key.
+    pub fn range_prefix() -> Vec<u8> {
+        format!("{}/", TABLE_ROUTE_PREFIX).into_bytes()
+    }
 }

 #[derive(Debug, PartialEq, Serialize, Deserialize, Clone)]
--- a/src/common/meta/src/key/tombstone.rs
+++ b/src/common/meta/src/key/tombstone.rs
@@ -25,20 +25,32 @@ use crate::rpc::store::BatchGetRequest;
 /// [TombstoneManager] provides the ability to:
 /// - logically delete values
 /// - restore the deleted values
-pub(crate) struct TombstoneManager {
+pub struct TombstoneManager {
    kv_backend: KvBackendRef,
+    tombstone_prefix: String,
 }

 const TOMBSTONE_PREFIX: &str = "__tombstone/";

-fn to_tombstone(key: &[u8]) -> Vec<u8> {
-    [TOMBSTONE_PREFIX.as_bytes(), key].concat()
-}
-
 impl TombstoneManager {
    /// Returns [TombstoneManager].
    pub fn new(kv_backend: KvBackendRef) -> Self {
-        Self { kv_backend }
+        Self {
+            kv_backend,
+            tombstone_prefix: TOMBSTONE_PREFIX.to_string(),
+        }
+    }
+
+    /// Returns [TombstoneManager] with a custom tombstone prefix.
+    pub fn new_with_prefix(kv_backend: KvBackendRef, prefix: &str) -> Self {
+        Self {
+            kv_backend,
+            tombstone_prefix: prefix.to_string(),
+        }
+    }
+
+    pub fn to_tombstone(&self, key: &[u8]) -> Vec<u8> {
+        [self.tombstone_prefix.as_bytes(), key].concat()
    }

    /// Moves value to `dest_key`.
@@ -67,7 +79,7 @@ impl TombstoneManager {
        (txn, TxnOpGetResponseSet::filter(src_key))
    }

-    async fn move_values_inner(&self, keys: &[Vec<u8>], dest_keys: &[Vec<u8>]) -> Result<()> {
+    async fn move_values_inner(&self, keys: &[Vec<u8>], dest_keys: &[Vec<u8>]) -> Result<usize> {
        ensure!(
            keys.len() == dest_keys.len(),
            error::UnexpectedSnafu {
@@ -102,7 +114,7 @@ impl TombstoneManager {
                .unzip();
            let mut resp = self.kv_backend.txn(Txn::merge_all(txns)).await?;
            if resp.succeeded {
-                return Ok(());
+                return Ok(keys.len());
            }
            let mut set = TxnOpGetResponseSet::from(&mut resp.responses);
            // Updates results.
@@ -125,7 +137,9 @@ impl TombstoneManager {
    }

    /// Moves values to `dest_key`.
-    async fn move_values(&self, keys: Vec<Vec<u8>>, dest_keys: Vec<Vec<u8>>) -> Result<()> {
+    ///
+    /// Returns the number of keys that were moved.
+    async fn move_values(&self, keys: Vec<Vec<u8>>, dest_keys: Vec<Vec<u8>>) -> Result<usize> {
        let chunk_size = self.kv_backend.max_txn_ops() / 2;
        if keys.len() > chunk_size {
            let keys_chunks = keys.chunks(chunk_size).collect::<Vec<_>>();
@@ -134,7 +148,7 @@ impl TombstoneManager {
                self.move_values_inner(keys, dest_keys).await?;
            }

-            Ok(())
+            Ok(keys.len())
        } else {
            self.move_values_inner(&keys, &dest_keys).await
        }
@@ -145,11 +159,13 @@ impl TombstoneManager {
    /// Preforms to:
    /// - deletes origin values.
    /// - stores tombstone values.
-    pub(crate) async fn create(&self, keys: Vec<Vec<u8>>) -> Result<()> {
+    ///
+    /// Returns the number of keys that were moved.
+    pub async fn create(&self, keys: Vec<Vec<u8>>) -> Result<usize> {
        let (keys, dest_keys): (Vec<_>, Vec<_>) = keys
            .into_iter()
            .map(|key| {
-                let tombstone_key = to_tombstone(&key);
+                let tombstone_key = self.to_tombstone(&key);
                (key, tombstone_key)
            })
            .unzip();
@@ -162,11 +178,13 @@ impl TombstoneManager {
    /// Preforms to:
    /// - restore origin value.
    /// - deletes tombstone values.
-    pub(crate) async fn restore(&self, keys: Vec<Vec<u8>>) -> Result<()> {
+    ///
+    /// Returns the number of keys that were restored.
+    pub async fn restore(&self, keys: Vec<Vec<u8>>) -> Result<usize> {
        let (keys, dest_keys): (Vec<_>, Vec<_>) = keys
            .into_iter()
            .map(|key| {
-                let tombstone_key = to_tombstone(&key);
+                let tombstone_key = self.to_tombstone(&key);
                (tombstone_key, key)
            })
            .unzip();
@@ -175,16 +193,18 @@ impl TombstoneManager {
    }

    /// Deletes tombstones values for the specified `keys`.
-    pub(crate) async fn delete(&self, keys: Vec<Vec<u8>>) -> Result<()> {
+    ///
+    /// Returns the number of keys that were deleted.
+    pub async fn delete(&self, keys: Vec<Vec<u8>>) -> Result<usize> {
        let operations = keys
            .iter()
-            .map(|key| TxnOp::Delete(to_tombstone(key)))
+            .map(|key| TxnOp::Delete(self.to_tombstone(key)))
            .collect::<Vec<_>>();

        let txn = Txn::new().and_then(operations);
        // Always success.
        let _ = self.kv_backend.txn(txn).await?;
-        Ok(())
+        Ok(keys.len())
    }
 }

@@ -194,7 +214,6 @@ mod tests {
    use std::collections::HashMap;
    use std::sync::Arc;

-    use super::to_tombstone;
    use crate::error::Error;
    use crate::key::tombstone::TombstoneManager;
    use crate::kv_backend::memory::MemoryKvBackend;
@@ -246,7 +265,7 @@ mod tests {
        assert!(!kv_backend.exists(b"foo").await.unwrap());
        assert_eq!(
            kv_backend
-                .get(&to_tombstone(b"bar"))
+                .get(&tombstone_manager.to_tombstone(b"bar"))
                .await
                .unwrap()
                .unwrap()
@@ -255,7 +274,7 @@ mod tests {
        );
        assert_eq!(
            kv_backend
-                .get(&to_tombstone(b"foo"))
+                .get(&tombstone_manager.to_tombstone(b"foo"))
                .await
                .unwrap()
                .unwrap()
@@ -287,7 +306,7 @@ mod tests {
            kv_backend.clone(),
            &[MoveValue {
                key: b"bar".to_vec(),
-                dest_key: to_tombstone(b"bar"),
+                dest_key: tombstone_manager.to_tombstone(b"bar"),
                value: b"baz".to_vec(),
            }],
        )
@@ -364,7 +383,7 @@ mod tests {
            .iter()
            .map(|(key, value)| MoveValue {
                key: key.clone(),
-                dest_key: to_tombstone(key),
+                dest_key: tombstone_manager.to_tombstone(key),
                value: value.clone(),
            })
            .collect::<Vec<_>>();
@@ -409,7 +428,7 @@ mod tests {
            .iter()
            .map(|(key, value)| MoveValue {
                key: key.clone(),
-                dest_key: to_tombstone(key),
+                dest_key: tombstone_manager.to_tombstone(key),
                value: value.clone(),
            })
            .collect::<Vec<_>>();
@@ -462,7 +481,7 @@ mod tests {
            .iter()
            .map(|(key, value)| MoveValue {
                key: key.clone(),
-                dest_key: to_tombstone(key),
+                dest_key: tombstone_manager.to_tombstone(key),
                value: value.clone(),
            })
            .collect::<Vec<_>>();
@@ -502,7 +521,7 @@ mod tests {
            .iter()
            .map(|(key, value)| MoveValue {
                key: key.clone(),
-                dest_key: to_tombstone(key),
+                dest_key: tombstone_manager.to_tombstone(key),
                value: value.clone(),
            })
            .collect::<Vec<_>>();
@@ -537,7 +556,7 @@ mod tests {
            .iter()
            .map(|(key, value)| MoveValue {
                key: key.clone(),
-                dest_key: to_tombstone(key),
+                dest_key: tombstone_manager.to_tombstone(key),
                value: value.clone(),
            })
            .collect::<Vec<_>>();
--- a/src/common/meta/src/key/view_info.rs
+++ b/src/common/meta/src/key/view_info.rs
@@ -70,11 +70,12 @@ impl MetadataKey<'_, ViewInfoKey> for ViewInfoKey {
            }
            .build()
        })?;
-        let captures = VIEW_INFO_KEY_PATTERN
-            .captures(key)
-            .context(InvalidViewInfoSnafu {
-                err_msg: format!("Invalid ViewInfoKey '{key}'"),
-            })?;
+        let captures =
+            VIEW_INFO_KEY_PATTERN
+                .captures(key)
+                .with_context(|| InvalidViewInfoSnafu {
+                    err_msg: format!("Invalid ViewInfoKey '{key}'"),
+                })?;
        // Safety: pass the regex check above
        let view_id = captures[1].parse::<TableId>().unwrap();
        Ok(ViewInfoKey { view_id })
--- a/Show More
+++ b/Show More