chore: rebase to main

chore: per review
docs: more comment on accums internal
2026-01-04 20:32:56 +00:00 · 2024-04-25 14:20:44 +08:00 · 2024-04-25 14:06:44 +08:00 · 2024-04-25 14:06:44 +08:00 · 2024-04-25 14:06:44 +08:00 · 2024-04-25 14:06:44 +08:00
603 changed files with 28804 additions and 9770 deletions
--- a/.github/CODEOWNERS
+++ b/.github/CODEOWNERS
@@ -0,0 +1,27 @@
+# GreptimeDB CODEOWNERS
+
+# These owners will be the default owners for everything in the repo.
+
+* @GreptimeTeam/db-approver
+
+## [Module] Databse Engine
+/src/index @zhongzc
+/src/mito2 @evenyag @v0y4g3r @waynexia
+/src/query @evenyag
+
+## [Module] Distributed
+/src/common/meta @MichaelScofield
+/src/common/procedure @MichaelScofield
+/src/meta-client @MichaelScofield
+/src/meta-srv @MichaelScofield
+
+## [Module] Write Ahead Log
+/src/log-store @v0y4g3r
+/src/store-api @v0y4g3r
+
+## [Module] Metrics Engine
+/src/metric-engine @waynexia
+/src/promql @waynexia
+
+## [Module] Flow
+/src/flow @zhongzc @waynexia
--- a/.github/ISSUE_TEMPLATE/bug_report.yml
+++ b/.github/ISSUE_TEMPLATE/bug_report.yml
@@ -39,7 +39,7 @@ body:
        - Query Engine
        - Table Engine
        - Write Protocols
-        - MetaSrv
+        - Metasrv
        - Frontend
        - Datanode
        - Other
--- a/.github/actions/build-dev-builder-images/action.yml
+++ b/.github/actions/build-dev-builder-images/action.yml
@@ -22,15 +22,15 @@ inputs:
  build-dev-builder-ubuntu:
    description: Build dev-builder-ubuntu image
    required: false
-    default: 'true'
+    default: "true"
  build-dev-builder-centos:
    description: Build dev-builder-centos image
    required: false
-    default: 'true'
+    default: "true"
  build-dev-builder-android:
    description: Build dev-builder-android image
    required: false
-    default: 'true'
+    default: "true"
 runs:
  using: composite
  steps:
@@ -47,7 +47,7 @@ runs:
      run: |
        make dev-builder \
          BASE_IMAGE=ubuntu \
-          BUILDX_MULTI_PLATFORM_BUILD=true \
+          BUILDX_MULTI_PLATFORM_BUILD=all \
          IMAGE_REGISTRY=${{ inputs.dockerhub-image-registry }} \
          IMAGE_NAMESPACE=${{ inputs.dockerhub-image-namespace }} \
          IMAGE_TAG=${{ inputs.version }}
@@ -58,7 +58,7 @@ runs:
      run: |
        make dev-builder \
          BASE_IMAGE=centos \
-          BUILDX_MULTI_PLATFORM_BUILD=true \
+          BUILDX_MULTI_PLATFORM_BUILD=amd64 \
          IMAGE_REGISTRY=${{ inputs.dockerhub-image-registry }} \
          IMAGE_NAMESPACE=${{ inputs.dockerhub-image-namespace }} \
          IMAGE_TAG=${{ inputs.version }}
@@ -72,5 +72,5 @@ runs:
          IMAGE_REGISTRY=${{ inputs.dockerhub-image-registry }} \
          IMAGE_NAMESPACE=${{ inputs.dockerhub-image-namespace }} \
          IMAGE_TAG=${{ inputs.version }} && \
-        
+
        docker push ${{ inputs.dockerhub-image-registry }}/${{ inputs.dockerhub-image-namespace }}/dev-builder-android:${{ inputs.version }}
--- a/.github/actions/build-linux-artifacts/action.yml
+++ b/.github/actions/build-linux-artifacts/action.yml
@@ -16,7 +16,7 @@ inputs:
  dev-mode:
    description: Enable dev mode, only build standard greptime
    required: false
-    default: 'false'
+    default: "false"
  working-dir:
    description: Working directory to build the artifacts
    required: false
@@ -68,7 +68,7 @@ runs:

    - name: Build greptime on centos base image
      uses: ./.github/actions/build-greptime-binary
-      if: ${{ inputs.arch == 'amd64' && inputs.dev-mode == 'false' }} # Only build centos7 base image for amd64.
+      if: ${{ inputs.arch == 'amd64' && inputs.dev-mode == 'false' }} # Builds greptime for centos if the host machine is amd64.
      with:
        base-image: centos
        features: servers/dashboard
@@ -79,7 +79,7 @@ runs:

    - name: Build greptime on android base image
      uses: ./.github/actions/build-greptime-binary
-      if: ${{ inputs.arch == 'amd64' && inputs.dev-mode == 'false' }} # Only build android base image on amd64.
+      if: ${{ inputs.arch == 'amd64' && inputs.dev-mode == 'false' }} # Builds arm64 greptime binary for android if the host machine amd64.
      with:
        base-image: android
        artifacts-dir: greptime-android-arm64-${{ inputs.version }}
--- a/.github/workflows/apidoc.yml
+++ b/.github/workflows/apidoc.yml
@@ -13,7 +13,7 @@ on:
 name: Build API docs

 env:
-  RUST_TOOLCHAIN: nightly-2023-12-19
+  RUST_TOOLCHAIN: nightly-2024-04-18

 jobs:
  apidoc:
--- a/.github/workflows/develop.yml
+++ b/.github/workflows/develop.yml
@@ -30,15 +30,20 @@ concurrency:
  cancel-in-progress: true

 env:
-  RUST_TOOLCHAIN: nightly-2023-12-19
+  RUST_TOOLCHAIN: nightly-2024-04-18

 jobs:
-  typos:
-    name: Spell Check with Typos
+  check-typos-and-docs:
+    name: Check typos and docs
    runs-on: ubuntu-20.04
    steps:
      - uses: actions/checkout@v4
      - uses: crate-ci/typos@v1.13.10
+      - name: Check the config docs
+        run: |
+          make config-docs && \
+          git diff --name-only --exit-code ./config/config.md  \
+          || (echo "'config/config.md' is not up-to-date, please run 'make config-docs'." && exit 1)

  check:
    name: Check
@@ -93,6 +98,8 @@ jobs:
    steps:
      - uses: actions/checkout@v4
      - uses: arduino/setup-protoc@v3
+        with:
+          repo-token: ${{ secrets.GITHUB_TOKEN }}
      - uses: dtolnay/rust-toolchain@master
        with:
          toolchain: ${{ env.RUST_TOOLCHAIN }}
@@ -123,10 +130,12 @@ jobs:
    runs-on: ubuntu-latest
    strategy:
      matrix:
-        target: [ "fuzz_create_table", "fuzz_alter_table" ]
+        target: [ "fuzz_create_table", "fuzz_alter_table", "fuzz_create_database" ]
    steps:
      - uses: actions/checkout@v4
      - uses: arduino/setup-protoc@v3
+        with:
+          repo-token: ${{ secrets.GITHUB_TOKEN }}
      - uses: dtolnay/rust-toolchain@master
        with:
          toolchain: ${{ env.RUST_TOOLCHAIN }}
@@ -138,8 +147,9 @@ jobs:
      - name: Set Rust Fuzz
        shell: bash
        run: |
-          sudo apt update && sudo apt install -y libfuzzer-14-dev
-          cargo install cargo-fuzz
+          sudo apt-get install -y libfuzzer-14-dev
+          rustup install nightly
+          cargo +nightly install cargo-fuzz
      - name: Download pre-built binaries
        uses: actions/download-artifact@v4
        with:
@@ -175,13 +185,13 @@ jobs:
      - name: Unzip binaries
        run: tar -xvf ./bins.tar.gz
      - name: Run sqlness
-        run: RUST_BACKTRACE=1 ./bins/sqlness-runner -c ./tests/cases --bins-dir ./bins
+        run: RUST_BACKTRACE=1 ./bins/sqlness-runner -c ./tests/cases --bins-dir ./bins --preserve-state
      - name: Upload sqlness logs
        if: always()
        uses: actions/upload-artifact@v4
        with:
          name: sqlness-logs
-          path: /tmp/greptime-*.log
+          path: /tmp/sqlness-*
          retention-days: 3

  sqlness-kafka-wal:
@@ -205,13 +215,13 @@ jobs:
        working-directory: tests-integration/fixtures/kafka
        run: docker compose -f docker-compose-standalone.yml up -d --wait
      - name: Run sqlness
-        run: RUST_BACKTRACE=1 ./bins/sqlness-runner -w kafka -k 127.0.0.1:9092 -c ./tests/cases --bins-dir ./bins
+        run: RUST_BACKTRACE=1 ./bins/sqlness-runner -w kafka -k 127.0.0.1:9092 -c ./tests/cases --bins-dir ./bins --preserve-state
      - name: Upload sqlness logs
        if: always()
        uses: actions/upload-artifact@v4
        with:
          name: sqlness-logs-with-kafka-wal
-          path: /tmp/greptime-*.log
+          path: /tmp/sqlness-*
          retention-days: 3

  fmt:
@@ -305,10 +315,10 @@ jobs:
          CARGO_BUILD_RUSTFLAGS: "-C link-arg=-fuse-ld=lld"
          RUST_BACKTRACE: 1
          CARGO_INCREMENTAL: 0
-          GT_S3_BUCKET: ${{ secrets.S3_BUCKET }}
-          GT_S3_ACCESS_KEY_ID: ${{ secrets.S3_ACCESS_KEY_ID }}
-          GT_S3_ACCESS_KEY: ${{ secrets.S3_ACCESS_KEY }}
-          GT_S3_REGION: ${{ secrets.S3_REGION }}
+          GT_S3_BUCKET: ${{ vars.AWS_CI_TEST_BUCKET }}
+          GT_S3_ACCESS_KEY_ID: ${{ secrets.AWS_CI_TEST_ACCESS_KEY_ID }}
+          GT_S3_ACCESS_KEY: ${{ secrets.AWS_CI_TEST_SECRET_ACCESS_KEY }}
+          GT_S3_REGION: ${{ vars.AWS_CI_TEST_BUCKET_REGION }}
          GT_ETCD_ENDPOINTS: http://127.0.0.1:2379
          GT_KAFKA_ENDPOINTS: 127.0.0.1:9092
          UNITTEST_LOG_DIR: "__unittest_logs"
@@ -321,20 +331,20 @@ jobs:
          fail_ci_if_error: false
          verbose: true

-  compat:
-    name: Compatibility Test
-    needs: build
-    runs-on: ubuntu-20.04
-    timeout-minutes: 60
-    steps:
-      - uses: actions/checkout@v4
-      - name: Download pre-built binaries
-        uses: actions/download-artifact@v4
-        with:
-          name: bins
-          path: .
-      - name: Unzip binaries
-        run: |
-          mkdir -p ./bins/current
-          tar -xvf ./bins.tar.gz --strip-components=1 -C ./bins/current
-      - run: ./tests/compat/test-compat.sh 0.6.0
+  # compat:
+  #   name: Compatibility Test
+  #   needs: build
+  #   runs-on: ubuntu-20.04
+  #   timeout-minutes: 60
+  #   steps:
+  #     - uses: actions/checkout@v4
+  #     - name: Download pre-built binaries
+  #       uses: actions/download-artifact@v4
+  #       with:
+  #         name: bins
+  #         path: .
+  #     - name: Unzip binaries
+  #       run: |
+  #         mkdir -p ./bins/current
+  #         tar -xvf ./bins.tar.gz --strip-components=1 -C ./bins/current
+  #     - run: ./tests/compat/test-compat.sh 0.6.0
--- a/.github/workflows/license.yaml
+++ b/.github/workflows/license.yaml
@@ -13,4 +13,4 @@ jobs:
    steps:
    - uses: actions/checkout@v4
    - name: Check License Header
-      uses: korandoru/hawkeye@v4
+      uses: korandoru/hawkeye@v5
--- a/.github/workflows/nightly-ci.yml
+++ b/.github/workflows/nightly-ci.yml
@@ -12,7 +12,7 @@ concurrency:
  cancel-in-progress: true

 env:
-  RUST_TOOLCHAIN: nightly-2023-12-19
+  RUST_TOOLCHAIN: nightly-2024-04-18

 jobs:
  sqlness:
@@ -85,10 +85,10 @@ jobs:
        env:
          RUST_BACKTRACE: 1
          CARGO_INCREMENTAL: 0
-          GT_S3_BUCKET: ${{ secrets.S3_BUCKET }}
-          GT_S3_ACCESS_KEY_ID: ${{ secrets.S3_ACCESS_KEY_ID }}
-          GT_S3_ACCESS_KEY: ${{ secrets.S3_ACCESS_KEY }}
-          GT_S3_REGION: ${{ secrets.S3_REGION }}
+          GT_S3_BUCKET: ${{ vars.AWS_CI_TEST_BUCKET }}
+          GT_S3_ACCESS_KEY_ID: ${{ secrets.AWS_CI_TEST_ACCESS_KEY_ID }}
+          GT_S3_ACCESS_KEY: ${{ secrets.AWS_CI_TEST_SECRET_ACCESS_KEY }}
+          GT_S3_REGION: ${{ vars.AWS_CI_TEST_BUCKET_REGION }}
          UNITTEST_LOG_DIR: "__unittest_logs"
      - name: Notify slack if failed
        if: failure()
--- a/.github/workflows/release.yml
+++ b/.github/workflows/release.yml
@@ -82,7 +82,7 @@ on:
 # Use env variables to control all the release process.
 env:
  # The arguments of building greptime.
-  RUST_TOOLCHAIN: nightly-2023-12-19
+  RUST_TOOLCHAIN: nightly-2024-04-18
  CARGO_PROFILE: nightly

  # Controls whether to run tests, include unit-test, integration-test and sqlness.
--- a/CONTRIBUTING.md
+++ b/CONTRIBUTING.md
@@ -50,7 +50,7 @@ GreptimeDB uses the [Apache 2.0 license](https://github.com/GreptimeTeam/greptim

 - To ensure that community is free and confident in its ability to use your contributions, please sign the Contributor License Agreement (CLA) which will be incorporated in the pull request process.
 - Make sure all files have proper license header (running `docker run --rm -v $(pwd):/github/workspace ghcr.io/korandoru/hawkeye-native:v3 format` from the project root).
- Make sure all your codes are formatted and follow the [coding style](https://pingcap.github.io/style-guide/rust/).
+- Make sure all your codes are formatted and follow the [coding style](https://pingcap.github.io/style-guide/rust/) and [style guide](http://github.com/greptimeTeam/docs/style-guide.md).
 - Make sure all unit tests are passed (using `cargo test --workspace` or [nextest](https://nexte.st/index.html) `cargo nextest run`).
 - Make sure all clippy warnings are fixed (you can check it locally by running `cargo clippy --workspace --all-targets -- -D warnings`).

--- a/Cargo.lock
+++ b/Cargo.lock
--- a/Cargo.toml
+++ b/Cargo.toml
@@ -62,7 +62,7 @@ members = [
 resolver = "2"

 [workspace.package]
-version = "0.7.1"
+version = "0.7.2"
 edition = "2021"
 license = "Apache-2.0"

@@ -70,16 +70,24 @@ license = "Apache-2.0"
 clippy.print_stdout = "warn"
 clippy.print_stderr = "warn"
 clippy.implicit_clone = "warn"
+clippy.readonly_write_lock = "allow"
 rust.unknown_lints = "deny"
+# Remove this after https://github.com/PyO3/pyo3/issues/4094
+rust.non_local_definitions = "allow"

 [workspace.dependencies]
+# We turn off default-features for some dependencies here so the workspaces which inherit them can
+# selectively turn them on if needed, since we can override default-features = true (from false)
+# for the inherited dependency but cannot do the reverse (override from true to false).
+#
+# See for more detaiils: https://github.com/rust-lang/cargo/issues/11329
 ahash = { version = "0.8", features = ["compile-time-rng"] }
 aquamarine = "0.3"
-arrow = { version = "47.0" }
-arrow-array = "47.0"
-arrow-flight = "47.0"
-arrow-ipc = { version = "47.0", features = ["lz4"] }
-arrow-schema = { version = "47.0", features = ["serde"] }
+arrow = { version = "51.0.0", features = ["prettyprint"] }
+arrow-array = { version = "51.0.0", default-features = false, features = ["chrono-tz"] }
+arrow-flight = "51.0"
+arrow-ipc = { version = "51.0.0", default-features = false, features = ["lz4"] }
+arrow-schema = { version = "51.0", features = ["serde"] }
 async-stream = "0.3"
 async-trait = "0.1"
 axum = { version = "0.6", features = ["headers"] }
@@ -90,20 +98,25 @@ bytemuck = "1.12"
 bytes = { version = "1.5", features = ["serde"] }
 chrono = { version = "0.4", features = ["serde"] }
 clap = { version = "4.4", features = ["derive"] }
+crossbeam-utils = "0.8"
 dashmap = "5.4"
-datafusion = { git = "https://github.com/apache/arrow-datafusion.git", rev = "26e43acac3a96cec8dd4c8365f22dfb1a84306e9" }
-datafusion-common = { git = "https://github.com/apache/arrow-datafusion.git", rev = "26e43acac3a96cec8dd4c8365f22dfb1a84306e9" }
-datafusion-expr = { git = "https://github.com/apache/arrow-datafusion.git", rev = "26e43acac3a96cec8dd4c8365f22dfb1a84306e9" }
-datafusion-optimizer = { git = "https://github.com/apache/arrow-datafusion.git", rev = "26e43acac3a96cec8dd4c8365f22dfb1a84306e9" }
-datafusion-physical-expr = { git = "https://github.com/apache/arrow-datafusion.git", rev = "26e43acac3a96cec8dd4c8365f22dfb1a84306e9" }
-datafusion-sql = { git = "https://github.com/apache/arrow-datafusion.git", rev = "26e43acac3a96cec8dd4c8365f22dfb1a84306e9" }
-datafusion-substrait = { git = "https://github.com/apache/arrow-datafusion.git", rev = "26e43acac3a96cec8dd4c8365f22dfb1a84306e9" }
+datafusion = { git = "https://github.com/apache/arrow-datafusion.git", rev = "34eda15b73a9e278af8844b30ed2f1c21c10359c" }
+datafusion-common = { git = "https://github.com/apache/arrow-datafusion.git", rev = "34eda15b73a9e278af8844b30ed2f1c21c10359c" }
+datafusion-expr = { git = "https://github.com/apache/arrow-datafusion.git", rev = "34eda15b73a9e278af8844b30ed2f1c21c10359c" }
+datafusion-functions = { git = "https://github.com/apache/arrow-datafusion.git", rev = "34eda15b73a9e278af8844b30ed2f1c21c10359c" }
+datafusion-optimizer = { git = "https://github.com/apache/arrow-datafusion.git", rev = "34eda15b73a9e278af8844b30ed2f1c21c10359c" }
+datafusion-physical-expr = { git = "https://github.com/apache/arrow-datafusion.git", rev = "34eda15b73a9e278af8844b30ed2f1c21c10359c" }
+datafusion-sql = { git = "https://github.com/apache/arrow-datafusion.git", rev = "34eda15b73a9e278af8844b30ed2f1c21c10359c" }
+datafusion-substrait = { git = "https://github.com/apache/arrow-datafusion.git", rev = "34eda15b73a9e278af8844b30ed2f1c21c10359c" }
 derive_builder = "0.12"
-etcd-client = "0.12"
+dotenv = "0.15"
+# TODO(LFC): Wait for https://github.com/etcdv3/etcd-client/pull/76
+etcd-client = { git = "https://github.com/MichaelScofield/etcd-client.git", rev = "4c371e9b3ea8e0a8ee2f9cbd7ded26e54a45df3b" }
 fst = "0.4.7"
 futures = "0.3"
 futures-util = "0.3"
-greptime-proto = { git = "https://github.com/GreptimeTeam/greptime-proto.git", rev = "06f6297ff3cab578a1589741b504342fbad70453" }
+greptime-proto = { git = "https://github.com/GreptimeTeam/greptime-proto.git", rev = "73ac0207ab71dfea48f30259ffdb611501b5ecb8" }
+humantime = "2.1"
 humantime-serde = "1.1"
 itertools = "0.10"
 lazy_static = "1.4"
@@ -113,12 +126,12 @@ moka = "0.12"
 notify = "6.1"
 num_cpus = "1.16"
 once_cell = "1.18"
-opentelemetry-proto = { git = "https://github.com/waynexia/opentelemetry-rust.git", rev = "33841b38dda79b15f2024952be5f32533325ca02", features = [
+opentelemetry-proto = { version = "0.5", features = [
    "gen-tonic",
    "metrics",
    "trace",
 ] }
-parquet = "47.0"
+parquet = { version = "51.0.0", default-features = false, features = ["arrow", "async", "object_store"] }
 paste = "1.0"
 pin-project = "1.0"
 prometheus = { version = "0.13.3", features = ["process"] }
@@ -131,6 +144,7 @@ reqwest = { version = "0.11", default-features = false, features = [
    "json",
    "rustls-tls-native-roots",
    "stream",
+    "multipart",
 ] }
 rskafka = "0.5"
 rust_decimal = "1.33"
@@ -141,18 +155,19 @@ serde_with = "3"
 smallvec = { version = "1", features = ["serde"] }
 snafu = "0.7"
 sysinfo = "0.30"
-# on branch v0.38.x
-sqlparser = { git = "https://github.com/GreptimeTeam/sqlparser-rs.git", rev = "6a93567ae38d42be5c8d08b13c8ff4dde26502ef", features = [
+# on branch v0.44.x
+sqlparser = { git = "https://github.com/GreptimeTeam/sqlparser-rs.git", rev = "c919990bf62ad38d2b0c0a3bc90b26ad919d51b0", features = [
    "visitor",
 ] }
 strum = { version = "0.25", features = ["derive"] }
 tempfile = "3"
-tokio = { version = "1.28", features = ["full"] }
+tokio = { version = "1.36", features = ["full"] }
 tokio-stream = { version = "0.1" }
 tokio-util = { version = "0.7", features = ["io-util", "compat"] }
 toml = "0.8.8"
-tonic = { version = "0.10", features = ["tls"] }
-uuid = { version = "1", features = ["serde", "v4", "fast-rng"] }
+tonic = { version = "0.11", features = ["tls"] }
+uuid = { version = "1.7", features = ["serde", "v4", "fast-rng"] }
+zstd = "0.13"

 ## workspaces members
 api = { path = "src/api" }
--- a/18
+++ b/18
@@ -54,8 +54,10 @@ ifneq ($(strip $(RELEASE)),)
 	CARGO_BUILD_OPTS += --release
 endif

-ifeq ($(BUILDX_MULTI_PLATFORM_BUILD), true)
+ifeq ($(BUILDX_MULTI_PLATFORM_BUILD), all)
 	BUILDX_MULTI_PLATFORM_BUILD_OPTS := --platform linux/amd64,linux/arm64 --push
+else ifeq ($(BUILDX_MULTI_PLATFORM_BUILD), amd64)
+	BUILDX_MULTI_PLATFORM_BUILD_OPTS := --platform linux/amd64 --push
 else
 	BUILDX_MULTI_PLATFORM_BUILD_OPTS := -o type=docker
 endif
@@ -169,6 +171,10 @@ check: ## Cargo check all the targets.
 clippy: ## Check clippy rules.
 	cargo clippy --workspace --all-targets --all-features -- -D warnings

+.PHONY: fix-clippy
+fix-clippy: ## Fix clippy violations.
+	cargo clippy --workspace --all-targets --all-features --fix
+
 .PHONY: fmt-check
 fmt-check: ## Check code format.
 	cargo fmt --all -- --check
@@ -188,6 +194,16 @@ run-it-in-container: start-etcd ## Run integration tests in dev-builder.
 	-w /greptimedb ${IMAGE_REGISTRY}/${IMAGE_NAMESPACE}/dev-builder-${BASE_IMAGE}:latest \
 	make test sqlness-test BUILD_JOBS=${BUILD_JOBS}

+##@ Docs
+config-docs: ## Generate configuration documentation from toml files.
+	docker run --rm \
+    -v ${PWD}:/greptimedb \
+    -w /greptimedb/config \
+    toml2docs/toml2docs:latest \
+    -p '##' \
+    -t ./config-docs-template.md \
+    -o ./config.md
+
 ##@ General

 # The help target prints out all targets with their descriptions organized
--- a/README.md
+++ b/README.md
@@ -143,7 +143,7 @@ cargo run -- standalone start
 - [GreptimeDB C++ Ingester](https://github.com/GreptimeTeam/greptimedb-ingester-cpp)
 - [GreptimeDB Erlang Ingester](https://github.com/GreptimeTeam/greptimedb-ingester-erl)
 - [GreptimeDB Rust Ingester](https://github.com/GreptimeTeam/greptimedb-ingester-rust)
- [GreptimeDB JavaScript Ingester](https://github.com/GreptimeTeam/greptime-ingester-js)
+- [GreptimeDB JavaScript Ingester](https://github.com/GreptimeTeam/greptimedb-ingester-js)

 ### Grafana Dashboard

--- a/benchmarks/Cargo.toml
+++ b/benchmarks/Cargo.toml
@@ -8,11 +8,31 @@ license.workspace = true
 workspace = true

 [dependencies]
+api.workspace = true
 arrow.workspace = true
 chrono.workspace = true
 clap.workspace = true
 client.workspace = true
+common-base.workspace = true
+common-telemetry.workspace = true
+common-wal.workspace = true
+dotenv.workspace = true
+futures.workspace = true
 futures-util.workspace = true
+humantime.workspace = true
+humantime-serde.workspace = true
 indicatif = "0.17.1"
+itertools.workspace = true
+lazy_static.workspace = true
+log-store.workspace = true
+mito2.workspace = true
+num_cpus.workspace = true
 parquet.workspace = true
+prometheus.workspace = true
+rand.workspace = true
+rskafka.workspace = true
+serde.workspace = true
+store-api.workspace = true
 tokio.workspace = true
+toml.workspace = true
+uuid.workspace = true
--- a/benchmarks/README.md
+++ b/benchmarks/README.md
@@ -0,0 +1,11 @@
+Benchmarkers for GreptimeDB
+--------------------------------
+
+## Wal Benchmarker
+The wal benchmarker serves to evaluate the performance of GreptimeDB's Write-Ahead Log (WAL) component. It meticulously assesses the read/write performance of the WAL under diverse workloads generated by the benchmarker. 
+
+
+### How to use
+To compile the benchmarker, navigate to the `greptimedb/benchmarks` directory and execute `cargo build --release`. Subsequently, you'll find the compiled target located at `greptimedb/target/release/wal_bench`.
+
+The `./wal_bench -h` command reveals numerous arguments that the target accepts. Among these, a notable one is the `cfg-file` argument. By utilizing a configuration file in the TOML format, you can bypass the need to repeatedly specify cumbersome arguments.
--- a/benchmarks/config/wal_bench.example.toml
+++ b/benchmarks/config/wal_bench.example.toml
@@ -0,0 +1,21 @@
+# Refers to the documents of `Args` in benchmarks/src/wal.rs`.
+wal_provider = "kafka"
+bootstrap_brokers = ["localhost:9092"]
+num_workers = 10
+num_topics = 32
+num_regions = 1000
+num_scrapes = 1000
+num_rows = 5
+col_types = "ifs"
+max_batch_size = "512KB"
+linger = "1ms"
+backoff_init = "10ms"
+backoff_max = "1ms"
+backoff_base = 2
+backoff_deadline = "3s"
+compression = "zstd"
+rng_seed = 42
+skip_read = false
+skip_write = false
+random_topics = true
+report_metrics = false
--- a/benchmarks/src/bin/nyc-taxi.rs
+++ b/benchmarks/src/bin/nyc-taxi.rs
@@ -215,37 +215,7 @@ fn build_values(column: &ArrayRef) -> (Values, ColumnDataType) {
                ColumnDataType::String,
            )
        }
-        DataType::Null
-        | DataType::Boolean
-        | DataType::Int8
-        | DataType::Int16
-        | DataType::Int32
-        | DataType::UInt8
-        | DataType::UInt16
-        | DataType::UInt32
-        | DataType::UInt64
-        | DataType::Float16
-        | DataType::Float32
-        | DataType::Date32
-        | DataType::Date64
-        | DataType::Time32(_)
-        | DataType::Time64(_)
-        | DataType::Duration(_)
-        | DataType::Interval(_)
-        | DataType::Binary
-        | DataType::FixedSizeBinary(_)
-        | DataType::LargeBinary
-        | DataType::LargeUtf8
-        | DataType::List(_)
-        | DataType::FixedSizeList(_, _)
-        | DataType::LargeList(_)
-        | DataType::Struct(_)
-        | DataType::Union(_, _)
-        | DataType::Dictionary(_, _)
-        | DataType::Decimal128(_, _)
-        | DataType::Decimal256(_, _)
-        | DataType::RunEndEncoded(_, _)
-        | DataType::Map(_, _) => todo!(),
+        _ => unimplemented!(),
    }
 }

@@ -444,7 +414,7 @@ fn create_table_expr(table_name: &str) -> CreateTableExpr {
 fn query_set(table_name: &str) -> HashMap<String, String> {
    HashMap::from([
        (
-            "count_all".to_string(), 
+            "count_all".to_string(),
            format!("SELECT COUNT(*) FROM {table_name};"),
        ),
        (
--- a/benchmarks/src/bin/wal_bench.rs
+++ b/benchmarks/src/bin/wal_bench.rs
@@ -0,0 +1,326 @@
+// Copyright 2023 Greptime Team
+//
+// Licensed under the Apache License, Version 2.0 (the "License");
+// you may not use this file except in compliance with the License.
+// You may obtain a copy of the License at
+//
+//     http://www.apache.org/licenses/LICENSE-2.0
+//
+// Unless required by applicable law or agreed to in writing, software
+// distributed under the License is distributed on an "AS IS" BASIS,
+// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+// See the License for the specific language governing permissions and
+// limitations under the License.
+
+#![feature(int_roundings)]
+
+use std::fs;
+use std::sync::Arc;
+use std::time::Instant;
+
+use api::v1::{ColumnDataType, ColumnSchema, SemanticType};
+use benchmarks::metrics;
+use benchmarks::wal_bench::{Args, Config, Region, WalProvider};
+use clap::Parser;
+use common_telemetry::info;
+use common_wal::config::kafka::common::BackoffConfig;
+use common_wal::config::kafka::DatanodeKafkaConfig as KafkaConfig;
+use common_wal::config::raft_engine::RaftEngineConfig;
+use common_wal::options::{KafkaWalOptions, WalOptions};
+use itertools::Itertools;
+use log_store::kafka::log_store::KafkaLogStore;
+use log_store::raft_engine::log_store::RaftEngineLogStore;
+use mito2::wal::Wal;
+use prometheus::{Encoder, TextEncoder};
+use rand::distributions::{Alphanumeric, DistString};
+use rand::rngs::SmallRng;
+use rand::SeedableRng;
+use rskafka::client::partition::Compression;
+use rskafka::client::ClientBuilder;
+use store_api::logstore::LogStore;
+use store_api::storage::RegionId;
+
+async fn run_benchmarker<S: LogStore>(cfg: &Config, topics: &[String], wal: Arc<Wal<S>>) {
+    let chunk_size = cfg.num_regions.div_ceil(cfg.num_workers);
+    let region_chunks = (0..cfg.num_regions)
+        .map(|id| {
+            build_region(
+                id as u64,
+                topics,
+                &mut SmallRng::seed_from_u64(cfg.rng_seed),
+                cfg,
+            )
+        })
+        .chunks(chunk_size as usize)
+        .into_iter()
+        .map(|chunk| Arc::new(chunk.collect::<Vec<_>>()))
+        .collect::<Vec<_>>();
+
+    let mut write_elapsed = 0;
+    let mut read_elapsed = 0;
+
+    if !cfg.skip_write {
+        info!("Benchmarking write ...");
+
+        let num_scrapes = cfg.num_scrapes;
+        let timer = Instant::now();
+        futures::future::join_all((0..cfg.num_workers).map(|i| {
+            let wal = wal.clone();
+            let regions = region_chunks[i as usize].clone();
+            tokio::spawn(async move {
+                for _ in 0..num_scrapes {
+                    let mut wal_writer = wal.writer();
+                    regions
+                        .iter()
+                        .for_each(|region| region.add_wal_entry(&mut wal_writer));
+                    wal_writer.write_to_wal().await.unwrap();
+                }
+            })
+        }))
+        .await;
+        write_elapsed += timer.elapsed().as_millis();
+    }
+
+    if !cfg.skip_read {
+        info!("Benchmarking read ...");
+
+        let timer = Instant::now();
+        futures::future::join_all((0..cfg.num_workers).map(|i| {
+            let wal = wal.clone();
+            let regions = region_chunks[i as usize].clone();
+            tokio::spawn(async move {
+                for region in regions.iter() {
+                    region.replay(&wal).await;
+                }
+            })
+        }))
+        .await;
+        read_elapsed = timer.elapsed().as_millis();
+    }
+
+    dump_report(cfg, write_elapsed, read_elapsed);
+}
+
+fn build_region(id: u64, topics: &[String], rng: &mut SmallRng, cfg: &Config) -> Region {
+    let wal_options = match cfg.wal_provider {
+        WalProvider::Kafka => {
+            assert!(!topics.is_empty());
+            WalOptions::Kafka(KafkaWalOptions {
+                topic: topics.get(id as usize % topics.len()).cloned().unwrap(),
+            })
+        }
+        WalProvider::RaftEngine => WalOptions::RaftEngine,
+    };
+    Region::new(
+        RegionId::from_u64(id),
+        build_schema(&parse_col_types(&cfg.col_types), rng),
+        wal_options,
+        cfg.num_rows,
+        cfg.rng_seed,
+    )
+}
+
+fn build_schema(col_types: &[ColumnDataType], mut rng: &mut SmallRng) -> Vec<ColumnSchema> {
+    col_types
+        .iter()
+        .map(|col_type| ColumnSchema {
+            column_name: Alphanumeric.sample_string(&mut rng, 5),
+            datatype: *col_type as i32,
+            semantic_type: SemanticType::Field as i32,
+            datatype_extension: None,
+        })
+        .chain(vec![ColumnSchema {
+            column_name: "ts".to_string(),
+            datatype: ColumnDataType::TimestampMillisecond as i32,
+            semantic_type: SemanticType::Tag as i32,
+            datatype_extension: None,
+        }])
+        .collect()
+}
+
+fn dump_report(cfg: &Config, write_elapsed: u128, read_elapsed: u128) {
+    let cost_report = format!(
+        "write costs: {} ms, read costs: {} ms",
+        write_elapsed, read_elapsed,
+    );
+
+    let total_written_bytes = metrics::METRIC_WAL_WRITE_BYTES_TOTAL.get() as u128;
+    let write_throughput = if write_elapsed > 0 {
+        (total_written_bytes * 1000).div_floor(write_elapsed)
+    } else {
+        0
+    };
+    let total_read_bytes = metrics::METRIC_WAL_READ_BYTES_TOTAL.get() as u128;
+    let read_throughput = if read_elapsed > 0 {
+        (total_read_bytes * 1000).div_floor(read_elapsed)
+    } else {
+        0
+    };
+
+    let throughput_report = format!(
+        "total written bytes: {} bytes, total read bytes: {} bytes, write throuput: {} bytes/s ({} mb/s), read throughput: {} bytes/s ({} mb/s)",
+        total_written_bytes,
+        total_read_bytes,
+        write_throughput,
+        write_throughput.div_floor(1 << 20),
+        read_throughput,
+        read_throughput.div_floor(1 << 20),
+    );
+
+    let metrics_report = if cfg.report_metrics {
+        let mut buffer = Vec::new();
+        let encoder = TextEncoder::new();
+        let metrics = prometheus::gather();
+        encoder.encode(&metrics, &mut buffer).unwrap();
+        String::from_utf8(buffer).unwrap()
+    } else {
+        String::new()
+    };
+
+    info!(
+        r#"
+Benchmark config: 
+{cfg:?}
+
+Benchmark report:
+{cost_report}
+{throughput_report}
+{metrics_report}"#
+    );
+}
+
+async fn create_topics(cfg: &Config) -> Vec<String> {
+    // Creates topics.
+    let client = ClientBuilder::new(cfg.bootstrap_brokers.clone())
+        .build()
+        .await
+        .unwrap();
+    let ctrl_client = client.controller_client().unwrap();
+    let (topics, tasks): (Vec<_>, Vec<_>) = (0..cfg.num_topics)
+        .map(|i| {
+            let topic = if cfg.random_topics {
+                format!(
+                    "greptime_wal_bench_topic_{}_{}",
+                    uuid::Uuid::new_v4().as_u128(),
+                    i
+                )
+            } else {
+                format!("greptime_wal_bench_topic_{}", i)
+            };
+            let task = ctrl_client.create_topic(
+                topic.clone(),
+                1,
+                cfg.bootstrap_brokers.len() as i16,
+                2000,
+            );
+            (topic, task)
+        })
+        .unzip();
+    // Must ignore errors since we allow topics being created more than once.
+    let _ = futures::future::try_join_all(tasks).await;
+
+    topics
+}
+
+fn parse_compression(comp: &str) -> Compression {
+    match comp {
+        "no" => Compression::NoCompression,
+        "gzip" => Compression::Gzip,
+        "lz4" => Compression::Lz4,
+        "snappy" => Compression::Snappy,
+        "zstd" => Compression::Zstd,
+        other => unreachable!("Unrecognized compression {other}"),
+    }
+}
+
+fn parse_col_types(col_types: &str) -> Vec<ColumnDataType> {
+    let parts = col_types.split('x').collect::<Vec<_>>();
+    assert!(parts.len() <= 2);
+
+    let pattern = parts[0];
+    let repeat = parts
+        .get(1)
+        .map(|r| r.parse::<usize>().unwrap())
+        .unwrap_or(1);
+
+    pattern
+        .chars()
+        .map(|c| match c {
+            'i' | 'I' => ColumnDataType::Int64,
+            'f' | 'F' => ColumnDataType::Float64,
+            's' | 'S' => ColumnDataType::String,
+            other => unreachable!("Cannot parse {other} as a column data type"),
+        })
+        .cycle()
+        .take(pattern.len() * repeat)
+        .collect()
+}
+
+fn main() {
+    // Sets the global logging to INFO and suppress loggings from rskafka other than ERROR and upper ones.
+    std::env::set_var("UNITTEST_LOG_LEVEL", "info,rskafka=error");
+    common_telemetry::init_default_ut_logging();
+
+    let args = Args::parse();
+    let cfg = if !args.cfg_file.is_empty() {
+        toml::from_str(&fs::read_to_string(&args.cfg_file).unwrap()).unwrap()
+    } else {
+        Config::from(args)
+    };
+
+    // Validates arguments.
+    if cfg.num_regions < cfg.num_workers {
+        panic!("num_regions must be greater than or equal to num_workers");
+    }
+    if cfg
+        .num_workers
+        .min(cfg.num_topics)
+        .min(cfg.num_regions)
+        .min(cfg.num_scrapes)
+        .min(cfg.max_batch_size.as_bytes() as u32)
+        .min(cfg.bootstrap_brokers.len() as u32)
+        == 0
+    {
+        panic!("Invalid arguments");
+    }
+
+    tokio::runtime::Builder::new_multi_thread()
+        .enable_all()
+        .build()
+        .unwrap()
+        .block_on(async {
+            match cfg.wal_provider {
+                WalProvider::Kafka => {
+                    let topics = create_topics(&cfg).await;
+                    let kafka_cfg = KafkaConfig {
+                        broker_endpoints: cfg.bootstrap_brokers.clone(),
+                        max_batch_size: cfg.max_batch_size,
+                        linger: cfg.linger,
+                        backoff: BackoffConfig {
+                            init: cfg.backoff_init,
+                            max: cfg.backoff_max,
+                            base: cfg.backoff_base,
+                            deadline: Some(cfg.backoff_deadline),
+                        },
+                        compression: parse_compression(&cfg.compression),
+                        ..Default::default()
+                    };
+                    let store = Arc::new(KafkaLogStore::try_new(&kafka_cfg).await.unwrap());
+                    let wal = Arc::new(Wal::new(store));
+                    run_benchmarker(&cfg, &topics, wal).await;
+                }
+                WalProvider::RaftEngine => {
+                    // The benchmarker assumes the raft engine directory exists.
+                    let store = RaftEngineLogStore::try_new(
+                        "/tmp/greptimedb/raft-engine-wal".to_string(),
+                        RaftEngineConfig::default(),
+                    )
+                    .await
+                    .map(Arc::new)
+                    .unwrap();
+                    let wal = Arc::new(Wal::new(store));
+                    run_benchmarker(&cfg, &[], wal).await;
+                }
+            }
+        });
+}
--- a/benchmarks/src/lib.rs
+++ b/benchmarks/src/lib.rs
@@ -0,0 +1,16 @@
+// Copyright 2023 Greptime Team
+//
+// Licensed under the Apache License, Version 2.0 (the "License");
+// you may not use this file except in compliance with the License.
+// You may obtain a copy of the License at
+//
+//     http://www.apache.org/licenses/LICENSE-2.0
+//
+// Unless required by applicable law or agreed to in writing, software
+// distributed under the License is distributed on an "AS IS" BASIS,
+// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+// See the License for the specific language governing permissions and
+// limitations under the License.
+
+pub mod metrics;
+pub mod wal_bench;
--- a/benchmarks/src/metrics.rs
+++ b/benchmarks/src/metrics.rs
@@ -0,0 +1,39 @@
+// Copyright 2023 Greptime Team
+//
+// Licensed under the Apache License, Version 2.0 (the "License");
+// you may not use this file except in compliance with the License.
+// You may obtain a copy of the License at
+//
+//     http://www.apache.org/licenses/LICENSE-2.0
+//
+// Unless required by applicable law or agreed to in writing, software
+// distributed under the License is distributed on an "AS IS" BASIS,
+// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+// See the License for the specific language governing permissions and
+// limitations under the License.
+
+use lazy_static::lazy_static;
+use prometheus::*;
+
+/// Logstore label.
+pub const LOGSTORE_LABEL: &str = "logstore";
+/// Operation type label.
+pub const OPTYPE_LABEL: &str = "optype";
+
+lazy_static! {
+    /// Counters of bytes of each operation on a logstore.
+    pub static ref METRIC_WAL_OP_BYTES_TOTAL: IntCounterVec = register_int_counter_vec!(
+        "greptime_bench_wal_op_bytes_total",
+        "wal operation bytes total",
+        &[OPTYPE_LABEL],
+    )
+    .unwrap();
+    /// Counter of bytes of the append_batch operation.
+    pub static ref METRIC_WAL_WRITE_BYTES_TOTAL: IntCounter = METRIC_WAL_OP_BYTES_TOTAL.with_label_values(
+        &["write"],
+    );
+    /// Counter of bytes of the read operation.
+    pub static ref METRIC_WAL_READ_BYTES_TOTAL: IntCounter = METRIC_WAL_OP_BYTES_TOTAL.with_label_values(
+        &["read"],
+    );
+}
--- a/benchmarks/src/wal_bench.rs
+++ b/benchmarks/src/wal_bench.rs
@@ -0,0 +1,361 @@
+// Copyright 2023 Greptime Team
+//
+// Licensed under the Apache License, Version 2.0 (the "License");
+// you may not use this file except in compliance with the License.
+// You may obtain a copy of the License at
+//
+//     http://www.apache.org/licenses/LICENSE-2.0
+//
+// Unless required by applicable law or agreed to in writing, software
+// distributed under the License is distributed on an "AS IS" BASIS,
+// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+// See the License for the specific language governing permissions and
+// limitations under the License.
+
+use std::mem::size_of;
+use std::sync::atomic::{AtomicI64, AtomicU64, Ordering};
+use std::sync::{Arc, Mutex};
+use std::time::Duration;
+
+use api::v1::value::ValueData;
+use api::v1::{ColumnDataType, ColumnSchema, Mutation, OpType, Row, Rows, Value, WalEntry};
+use clap::{Parser, ValueEnum};
+use common_base::readable_size::ReadableSize;
+use common_wal::options::WalOptions;
+use futures::StreamExt;
+use mito2::wal::{Wal, WalWriter};
+use rand::distributions::{Alphanumeric, DistString, Uniform};
+use rand::rngs::SmallRng;
+use rand::{Rng, SeedableRng};
+use serde::{Deserialize, Serialize};
+use store_api::logstore::LogStore;
+use store_api::storage::RegionId;
+
+use crate::metrics;
+
+/// The wal provider.
+#[derive(Clone, ValueEnum, Default, Debug, PartialEq, Serialize, Deserialize)]
+#[serde(rename_all = "snake_case")]
+pub enum WalProvider {
+    #[default]
+    RaftEngine,
+    Kafka,
+}
+
+#[derive(Parser)]
+pub struct Args {
+    /// The provided configuration file.
+    /// The example configuration file can be found at `greptimedb/benchmarks/config/wal_bench.example.toml`.
+    #[clap(long, short = 'c')]
+    pub cfg_file: String,
+
+    /// The wal provider.
+    #[clap(long, value_enum, default_value_t = WalProvider::default())]
+    pub wal_provider: WalProvider,
+
+    /// The advertised addresses of the kafka brokers.
+    /// If there're multiple bootstrap brokers, their addresses should be separated by comma, for e.g. "localhost:9092,localhost:9093".
+    #[clap(long, short = 'b', default_value = "localhost:9092")]
+    pub bootstrap_brokers: String,
+
+    /// The number of workers each running in a dedicated thread.
+    #[clap(long, default_value_t = num_cpus::get() as u32)]
+    pub num_workers: u32,
+
+    /// The number of kafka topics to be created.
+    #[clap(long, default_value_t = 32)]
+    pub num_topics: u32,
+
+    /// The number of regions.
+    #[clap(long, default_value_t = 1000)]
+    pub num_regions: u32,
+
+    /// The number of times each region is scraped.
+    #[clap(long, default_value_t = 1000)]
+    pub num_scrapes: u32,
+
+    /// The number of rows in each wal entry.
+    /// Each time a region is scraped, a wal entry containing will be produced.
+    #[clap(long, default_value_t = 5)]
+    pub num_rows: u32,
+
+    /// The column types of the schema for each region.
+    /// Currently, three column types are supported:
+    /// - i = ColumnDataType::Int64
+    /// - f = ColumnDataType::Float64
+    /// - s = ColumnDataType::String  
+    /// For e.g., "ifs" will be parsed as three columns: i64, f64, and string.
+    ///
+    /// Additionally, a "x" sign can be provided to repeat the column types for a given number of times.
+    /// For e.g., "iix2" will be parsed as 4 columns: i64, i64, i64, and i64.
+    /// This feature is useful if you want to specify many columns.
+    #[clap(long, default_value = "ifs")]
+    pub col_types: String,
+
+    /// The maximum size of a batch of kafka records.
+    /// The default value is 1mb.
+    #[clap(long, default_value = "512KB")]
+    pub max_batch_size: ReadableSize,
+
+    /// The minimum latency the kafka client issues a batch of kafka records.
+    /// However, a batch of kafka records would be immediately issued if a record cannot be fit into the batch.
+    #[clap(long, default_value = "1ms")]
+    pub linger: String,
+
+    /// The initial backoff delay of the kafka consumer.
+    #[clap(long, default_value = "10ms")]
+    pub backoff_init: String,
+
+    /// The maximum backoff delay of the kafka consumer.
+    #[clap(long, default_value = "1s")]
+    pub backoff_max: String,
+
+    /// The exponential backoff rate of the kafka consumer. The next back off = base * the current backoff.
+    #[clap(long, default_value_t = 2)]
+    pub backoff_base: u32,
+
+    /// The deadline of backoff. The backoff ends if the total backoff delay reaches the deadline.
+    #[clap(long, default_value = "3s")]
+    pub backoff_deadline: String,
+
+    /// The client-side compression algorithm for kafka records.
+    #[clap(long, default_value = "zstd")]
+    pub compression: String,
+
+    /// The seed of random number generators.
+    #[clap(long, default_value_t = 42)]
+    pub rng_seed: u64,
+
+    /// Skips the read phase, aka. region replay, if set to true.
+    #[clap(long, default_value_t = false)]
+    pub skip_read: bool,
+
+    /// Skips the write phase if set to true.
+    #[clap(long, default_value_t = false)]
+    pub skip_write: bool,
+
+    /// Randomly generates topic names if set to true.
+    /// Useful when you want to run the benchmarker without worrying about the topics created before.
+    #[clap(long, default_value_t = false)]
+    pub random_topics: bool,
+
+    /// Logs out the gathered prometheus metrics when the benchmarker ends.
+    #[clap(long, default_value_t = false)]
+    pub report_metrics: bool,
+}
+
+/// Benchmarker config.
+#[derive(Debug, Clone, Serialize, Deserialize)]
+pub struct Config {
+    pub wal_provider: WalProvider,
+    pub bootstrap_brokers: Vec<String>,
+    pub num_workers: u32,
+    pub num_topics: u32,
+    pub num_regions: u32,
+    pub num_scrapes: u32,
+    pub num_rows: u32,
+    pub col_types: String,
+    pub max_batch_size: ReadableSize,
+    #[serde(with = "humantime_serde")]
+    pub linger: Duration,
+    #[serde(with = "humantime_serde")]
+    pub backoff_init: Duration,
+    #[serde(with = "humantime_serde")]
+    pub backoff_max: Duration,
+    pub backoff_base: u32,
+    #[serde(with = "humantime_serde")]
+    pub backoff_deadline: Duration,
+    pub compression: String,
+    pub rng_seed: u64,
+    pub skip_read: bool,
+    pub skip_write: bool,
+    pub random_topics: bool,
+    pub report_metrics: bool,
+}
+
+impl From<Args> for Config {
+    fn from(args: Args) -> Self {
+        let cfg = Self {
+            wal_provider: args.wal_provider,
+            bootstrap_brokers: args
+                .bootstrap_brokers
+                .split(',')
+                .map(ToString::to_string)
+                .collect::<Vec<_>>(),
+            num_workers: args.num_workers.min(num_cpus::get() as u32),
+            num_topics: args.num_topics,
+            num_regions: args.num_regions,
+            num_scrapes: args.num_scrapes,
+            num_rows: args.num_rows,
+            col_types: args.col_types,
+            max_batch_size: args.max_batch_size,
+            linger: humantime::parse_duration(&args.linger).unwrap(),
+            backoff_init: humantime::parse_duration(&args.backoff_init).unwrap(),
+            backoff_max: humantime::parse_duration(&args.backoff_max).unwrap(),
+            backoff_base: args.backoff_base,
+            backoff_deadline: humantime::parse_duration(&args.backoff_deadline).unwrap(),
+            compression: args.compression,
+            rng_seed: args.rng_seed,
+            skip_read: args.skip_read,
+            skip_write: args.skip_write,
+            random_topics: args.random_topics,
+            report_metrics: args.report_metrics,
+        };
+
+        cfg
+    }
+}
+
+/// The region used for wal benchmarker.
+pub struct Region {
+    id: RegionId,
+    schema: Vec<ColumnSchema>,
+    wal_options: WalOptions,
+    next_sequence: AtomicU64,
+    next_entry_id: AtomicU64,
+    next_timestamp: AtomicI64,
+    rng: Mutex<Option<SmallRng>>,
+    num_rows: u32,
+}
+
+impl Region {
+    /// Creates a new region.
+    pub fn new(
+        id: RegionId,
+        schema: Vec<ColumnSchema>,
+        wal_options: WalOptions,
+        num_rows: u32,
+        rng_seed: u64,
+    ) -> Self {
+        Self {
+            id,
+            schema,
+            wal_options,
+            next_sequence: AtomicU64::new(1),
+            next_entry_id: AtomicU64::new(1),
+            next_timestamp: AtomicI64::new(1655276557000),
+            rng: Mutex::new(Some(SmallRng::seed_from_u64(rng_seed))),
+            num_rows,
+        }
+    }
+
+    /// Scrapes the region and adds the generated entry to wal.
+    pub fn add_wal_entry<S: LogStore>(&self, wal_writer: &mut WalWriter<S>) {
+        let mutation = Mutation {
+            op_type: OpType::Put as i32,
+            sequence: self
+                .next_sequence
+                .fetch_add(self.num_rows as u64, Ordering::Relaxed),
+            rows: Some(self.build_rows()),
+        };
+        let entry = WalEntry {
+            mutations: vec![mutation],
+        };
+        metrics::METRIC_WAL_WRITE_BYTES_TOTAL.inc_by(Self::entry_estimated_size(&entry) as u64);
+
+        wal_writer
+            .add_entry(
+                self.id,
+                self.next_entry_id.fetch_add(1, Ordering::Relaxed),
+                &entry,
+                &self.wal_options,
+            )
+            .unwrap();
+    }
+
+    /// Replays the region.
+    pub async fn replay<S: LogStore>(&self, wal: &Arc<Wal<S>>) {
+        let mut wal_stream = wal.scan(self.id, 0, &self.wal_options).unwrap();
+        while let Some(res) = wal_stream.next().await {
+            let (_, entry) = res.unwrap();
+            metrics::METRIC_WAL_READ_BYTES_TOTAL.inc_by(Self::entry_estimated_size(&entry) as u64);
+        }
+    }
+
+    /// Computes the estimated size in bytes of the entry.
+    pub fn entry_estimated_size(entry: &WalEntry) -> usize {
+        let wrapper_size = size_of::<WalEntry>()
+            + entry.mutations.capacity() * size_of::<Mutation>()
+            + size_of::<Rows>();
+
+        let rows = entry.mutations[0].rows.as_ref().unwrap();
+
+        let schema_size = rows.schema.capacity() * size_of::<ColumnSchema>()
+            + rows
+                .schema
+                .iter()
+                .map(|s| s.column_name.capacity())
+                .sum::<usize>();
+        let values_size = (rows.rows.capacity() * size_of::<Row>())
+            + rows
+                .rows
+                .iter()
+                .map(|r| r.values.capacity() * size_of::<Value>())
+                .sum::<usize>();
+
+        wrapper_size + schema_size + values_size
+    }
+
+    fn build_rows(&self) -> Rows {
+        let cols = self
+            .schema
+            .iter()
+            .map(|col_schema| {
+                let col_data_type = ColumnDataType::try_from(col_schema.datatype).unwrap();
+                self.build_col(&col_data_type, self.num_rows)
+            })
+            .collect::<Vec<_>>();
+
+        let rows = (0..self.num_rows)
+            .map(|i| {
+                let values = cols.iter().map(|col| col[i as usize].clone()).collect();
+                Row { values }
+            })
+            .collect();
+
+        Rows {
+            schema: self.schema.clone(),
+            rows,
+        }
+    }
+
+    fn build_col(&self, col_data_type: &ColumnDataType, num_rows: u32) -> Vec<Value> {
+        let mut rng_guard = self.rng.lock().unwrap();
+        let rng = rng_guard.as_mut().unwrap();
+        match col_data_type {
+            ColumnDataType::TimestampMillisecond => (0..num_rows)
+                .map(|_| {
+                    let ts = self.next_timestamp.fetch_add(1000, Ordering::Relaxed);
+                    Value {
+                        value_data: Some(ValueData::TimestampMillisecondValue(ts)),
+                    }
+                })
+                .collect(),
+            ColumnDataType::Int64 => (0..num_rows)
+                .map(|_| {
+                    let v = rng.sample(Uniform::new(0, 10_000));
+                    Value {
+                        value_data: Some(ValueData::I64Value(v)),
+                    }
+                })
+                .collect(),
+            ColumnDataType::Float64 => (0..num_rows)
+                .map(|_| {
+                    let v = rng.sample(Uniform::new(0.0, 5000.0));
+                    Value {
+                        value_data: Some(ValueData::F64Value(v)),
+                    }
+                })
+                .collect(),
+            ColumnDataType::String => (0..num_rows)
+                .map(|_| {
+                    let v = Alphanumeric.sample_string(rng, 10);
+                    Value {
+                        value_data: Some(ValueData::StringValue(v)),
+                    }
+                })
+                .collect(),
+            _ => unreachable!(),
+        }
+    }
+}
--- a/cliff.toml
+++ b/cliff.toml
@@ -0,0 +1,127 @@
+# https://git-cliff.org/docs/configuration
+
+[remote.github]
+owner = "GreptimeTeam"
+repo = "greptimedb"
+
+[changelog]
+header = ""
+footer = ""
+# template for the changelog body
+# https://keats.github.io/tera/docs/#introduction
+body = """
+# {{ version }}
+
+Release date: {{ timestamp | date(format="%B %d, %Y") }}
+
+{%- set breakings = commits | filter(attribute="breaking", value=true) -%}
+{%- if breakings | length > 0 %}
+
+## Breaking changes
+    {% for commit in breakings %}
+      * {{ commit.github.pr_title }}\
+        {% if commit.github.username %} by \
+          {% set author = commit.github.username -%}
+          [@{{ author }}](https://github.com/{{ author }})
+        {%- endif -%}
+        {% if commit.github.pr_number %} in \
+          {% set number = commit.github.pr_number -%}
+          [#{{ number }}]({{ self::remote_url() }}/pull/{{ number }})
+        {%- endif %}
+    {%- endfor %}
+{%- endif -%}
+
+{%- set grouped_commits = commits | filter(attribute="breaking", value=false) | group_by(attribute="group") -%}
+{% for group, commits in grouped_commits %}
+
+    ### {{ group | striptags | trim | upper_first }}
+    {% for commit in commits %}
+        * {{ commit.github.pr_title }}\
+            {% if commit.github.username %} by \
+              {% set author = commit.github.username -%}
+              [@{{ author }}](https://github.com/{{ author }})
+            {%- endif -%}
+            {% if commit.github.pr_number %} in \
+              {% set number = commit.github.pr_number -%}
+              [#{{ number }}]({{ self::remote_url() }}/pull/{{ number }})
+            {%- endif %}
+    {%- endfor -%}
+{% endfor %}
+
+{%- if github.contributors | filter(attribute="is_first_time", value=true) | length != 0 %}
+  {% raw %}\n{% endraw -%}
+  ## New Contributors
+{% endif -%}
+{% for contributor in github.contributors | filter(attribute="is_first_time", value=true) %}
+  * [@{{ contributor.username }}](https://github.com/{{ contributor.username }}) made their first contribution
+    {%- if contributor.pr_number %} in \
+      [#{{ contributor.pr_number }}]({{ self::remote_url() }}/pull/{{ contributor.pr_number }}) \
+    {%- endif %}
+{%- endfor -%}
+
+{% if github.contributors | length != 0 %}
+  {% raw %}\n{% endraw -%}
+## All Contributors
+
+We would like to thank the following contributors from the GreptimeDB community:
+
+{%- set contributors = github.contributors | sort(attribute="username") | map(attribute="username") -%}
+{%- set bots = ['dependabot[bot]'] %}
+
+{% for contributor in contributors %}
+{%- if bots is containing(contributor) -%}{% continue %}{%- endif -%}
+{%- if loop.first -%}
+  [@{{ contributor }}](https://github.com/{{ contributor }})
+{%- else -%}
+  , [@{{ contributor }}](https://github.com/{{ contributor }})
+{%- endif -%}
+{%- endfor %}
+{%- endif %}
+{% raw %}\n{% endraw %}
+
+{%- macro remote_url() -%}
+  https://github.com/{{ remote.github.owner }}/{{ remote.github.repo }}
+{%- endmacro -%}
+"""
+trim = true
+
+[git]
+# parse the commits based on https://www.conventionalcommits.org
+conventional_commits = true
+# filter out the commits that are not conventional
+filter_unconventional = true
+# process each line of a commit as an individual commit
+split_commits = false
+# regex for parsing and grouping commits
+commit_parsers = [
+  { message = "^feat", group = "<!-- 0 -->🚀 Features" },
+  { message = "^fix", group = "<!-- 1 -->🐛 Bug Fixes" },
+  { message = "^doc", group = "<!-- 3 -->📚 Documentation" },
+  { message = "^perf", group = "<!-- 4 -->⚡ Performance" },
+  { message = "^refactor", group = "<!-- 2 -->🚜 Refactor" },
+  { message = "^style", group = "<!-- 5 -->🎨 Styling" },
+  { message = "^test", group = "<!-- 6 -->🧪 Testing" },
+  { message = "^chore\\(release\\): prepare for", skip = true },
+  { message = "^chore\\(deps.*\\)", skip = true },
+  { message = "^chore\\(pr\\)", skip = true },
+  { message = "^chore\\(pull\\)", skip = true },
+  { message = "^chore|^ci", group = "<!-- 7 -->⚙️ Miscellaneous Tasks" },
+  { body = ".*security", group = "<!-- 8 -->🛡️ Security" },
+  { message = "^revert", group = "<!-- 9 -->◀️ Revert" },
+]
+# protect breaking changes from being skipped due to matching a skipping commit_parser
+protect_breaking_commits = false
+# filter out the commits that are not matched by commit parsers
+filter_commits = false
+# regex for matching git tags
+# tag_pattern = "v[0-9].*"
+# regex for skipping tags
+# skip_tags = ""
+# regex for ignoring tags
+ignore_tags = ".*-nightly-.*"
+# sort the tags topologically
+topo_order = false
+# sort the commits inside sections by oldest/newest order
+sort_commits = "oldest"
+# limit the number of commits included in the changelog.
+# limit_commits = 42
--- a/config/config-docs-template.md
+++ b/config/config-docs-template.md
@@ -0,0 +1,19 @@
+# Configurations
+
+## Standalone Mode
+
+{{ toml2docs "./standalone.example.toml" }}
+
+## Cluster Mode
+
+### Frontend
+
+{{ toml2docs "./frontend.example.toml" }}
+
+### Metasrv
+
+{{ toml2docs "./metasrv.example.toml" }}
+
+### Datanode
+
+{{ toml2docs "./datanode.example.toml" }}
--- a/config/config.md
+++ b/config/config.md
@@ -0,0 +1,376 @@
+# Configurations
+
+## Standalone Mode
+
+| Key | Type | Default | Descriptions |
+| --- | -----| ------- | ----------- |
+| `mode` | String | `standalone` | The running mode of the datanode. It can be `standalone` or `distributed`. |
+| `enable_telemetry` | Bool | `true` | Enable telemetry to collect anonymous usage data. |
+| `default_timezone` | String | `None` | The default timezone of the server. |
+| `http` | -- | -- | The HTTP server options. |
+| `http.addr` | String | `127.0.0.1:4000` | The address to bind the HTTP server. |
+| `http.timeout` | String | `30s` | HTTP request timeout. |
+| `http.body_limit` | String | `64MB` | HTTP request body limit.<br/>Support the following units are supported: `B`, `KB`, `KiB`, `MB`, `MiB`, `GB`, `GiB`, `TB`, `TiB`, `PB`, `PiB`. |
+| `grpc` | -- | -- | The gRPC server options. |
+| `grpc.addr` | String | `127.0.0.1:4001` | The address to bind the gRPC server. |
+| `grpc.runtime_size` | Integer | `8` | The number of server worker threads. |
+| `mysql` | -- | -- | MySQL server options. |
+| `mysql.enable` | Bool | `true` | Whether to enable. |
+| `mysql.addr` | String | `127.0.0.1:4002` | The addr to bind the MySQL server. |
+| `mysql.runtime_size` | Integer | `2` | The number of server worker threads. |
+| `mysql.tls` | -- | -- | -- |
+| `mysql.tls.mode` | String | `disable` | TLS mode, refer to https://www.postgresql.org/docs/current/libpq-ssl.html<br/>- `disable` (default value)<br/>- `prefer`<br/>- `require`<br/>- `verify-ca`<br/>- `verify-full` |
+| `mysql.tls.cert_path` | String | `None` | Certificate file path. |
+| `mysql.tls.key_path` | String | `None` | Private key file path. |
+| `mysql.tls.watch` | Bool | `false` | Watch for Certificate and key file change and auto reload |
+| `postgres` | -- | -- | PostgresSQL server options. |
+| `postgres.enable` | Bool | `true` | Whether to enable |
+| `postgres.addr` | String | `127.0.0.1:4003` | The addr to bind the PostgresSQL server. |
+| `postgres.runtime_size` | Integer | `2` | The number of server worker threads. |
+| `postgres.tls` | -- | -- | PostgresSQL server TLS options, see `mysql_options.tls` section. |
+| `postgres.tls.mode` | String | `disable` | TLS mode. |
+| `postgres.tls.cert_path` | String | `None` | Certificate file path. |
+| `postgres.tls.key_path` | String | `None` | Private key file path. |
+| `postgres.tls.watch` | Bool | `false` | Watch for Certificate and key file change and auto reload |
+| `opentsdb` | -- | -- | OpenTSDB protocol options. |
+| `opentsdb.enable` | Bool | `true` | Whether to enable |
+| `opentsdb.addr` | String | `127.0.0.1:4242` | OpenTSDB telnet API server address. |
+| `opentsdb.runtime_size` | Integer | `2` | The number of server worker threads. |
+| `influxdb` | -- | -- | InfluxDB protocol options. |
+| `influxdb.enable` | Bool | `true` | Whether to enable InfluxDB protocol in HTTP API. |
+| `prom_store` | -- | -- | Prometheus remote storage options |
+| `prom_store.enable` | Bool | `true` | Whether to enable Prometheus remote write and read in HTTP API. |
+| `prom_store.with_metric_engine` | Bool | `true` | Whether to store the data from Prometheus remote write in metric engine. |
+| `wal` | -- | -- | The WAL options. |
+| `wal.provider` | String | `raft_engine` | The provider of the WAL.<br/>- `raft_engine`: the wal is stored in the local file system by raft-engine.<br/>- `kafka`: it's remote wal that data is stored in Kafka. |
+| `wal.dir` | String | `None` | The directory to store the WAL files.<br/>**It's only used when the provider is `raft_engine`**. |
+| `wal.file_size` | String | `256MB` | The size of the WAL segment file.<br/>**It's only used when the provider is `raft_engine`**. |
+| `wal.purge_threshold` | String | `4GB` | The threshold of the WAL size to trigger a flush.<br/>**It's only used when the provider is `raft_engine`**. |
+| `wal.purge_interval` | String | `10m` | The interval to trigger a flush.<br/>**It's only used when the provider is `raft_engine`**. |
+| `wal.read_batch_size` | Integer | `128` | The read batch size.<br/>**It's only used when the provider is `raft_engine`**. |
+| `wal.sync_write` | Bool | `false` | Whether to use sync write.<br/>**It's only used when the provider is `raft_engine`**. |
+| `wal.enable_log_recycle` | Bool | `true` | Whether to reuse logically truncated log files.<br/>**It's only used when the provider is `raft_engine`**. |
+| `wal.prefill_log_files` | Bool | `false` | Whether to pre-create log files on start up.<br/>**It's only used when the provider is `raft_engine`**. |
+| `wal.sync_period` | String | `10s` | Duration for fsyncing log files.<br/>**It's only used when the provider is `raft_engine`**. |
+| `wal.broker_endpoints` | Array | -- | The Kafka broker endpoints.<br/>**It's only used when the provider is `kafka`**. |
+| `wal.max_batch_size` | String | `1MB` | The max size of a single producer batch.<br/>Warning: Kafka has a default limit of 1MB per message in a topic.<br/>**It's only used when the provider is `kafka`**. |
+| `wal.linger` | String | `200ms` | The linger duration of a kafka batch producer.<br/>**It's only used when the provider is `kafka`**. |
+| `wal.consumer_wait_timeout` | String | `100ms` | The consumer wait timeout.<br/>**It's only used when the provider is `kafka`**. |
+| `wal.backoff_init` | String | `500ms` | The initial backoff delay.<br/>**It's only used when the provider is `kafka`**. |
+| `wal.backoff_max` | String | `10s` | The maximum backoff delay.<br/>**It's only used when the provider is `kafka`**. |
+| `wal.backoff_base` | Integer | `2` | The exponential backoff rate, i.e. next backoff = base * current backoff.<br/>**It's only used when the provider is `kafka`**. |
+| `wal.backoff_deadline` | String | `5mins` | The deadline of retries.<br/>**It's only used when the provider is `kafka`**. |
+| `metadata_store` | -- | -- | Metadata storage options. |
+| `metadata_store.file_size` | String | `256MB` | Kv file size in bytes. |
+| `metadata_store.purge_threshold` | String | `4GB` | Kv purge threshold. |
+| `procedure` | -- | -- | Procedure storage options. |
+| `procedure.max_retry_times` | Integer | `3` | Procedure max retry time. |
+| `procedure.retry_delay` | String | `500ms` | Initial retry delay of procedures, increases exponentially |
+| `storage` | -- | -- | The data storage options. |
+| `storage.data_home` | String | `/tmp/greptimedb/` | The working home directory. |
+| `storage.type` | String | `File` | The storage type used to store the data.<br/>- `File`: the data is stored in the local file system.<br/>- `S3`: the data is stored in the S3 object storage.<br/>- `Gcs`: the data is stored in the Google Cloud Storage.<br/>- `Azblob`: the data is stored in the Azure Blob Storage.<br/>- `Oss`: the data is stored in the Aliyun OSS. |
+| `storage.cache_path` | String | `None` | Cache configuration for object storage such as 'S3' etc.<br/>The local file cache directory. |
+| `storage.cache_capacity` | String | `None` | The local file cache capacity in bytes. |
+| `storage.bucket` | String | `None` | The S3 bucket name.<br/>**It's only used when the storage type is `S3`, `Oss` and `Gcs`**. |
+| `storage.root` | String | `None` | The S3 data will be stored in the specified prefix, for example, `s3://${bucket}/${root}`.<br/>**It's only used when the storage type is `S3`, `Oss` and `Azblob`**. |
+| `storage.access_key_id` | String | `None` | The access key id of the aws account.<br/>It's **highly recommended** to use AWS IAM roles instead of hardcoding the access key id and secret key.<br/>**It's only used when the storage type is `S3` and `Oss`**. |
+| `storage.secret_access_key` | String | `None` | The secret access key of the aws account.<br/>It's **highly recommended** to use AWS IAM roles instead of hardcoding the access key id and secret key.<br/>**It's only used when the storage type is `S3`**. |
+| `storage.access_key_secret` | String | `None` | The secret access key of the aliyun account.<br/>**It's only used when the storage type is `Oss`**. |
+| `storage.account_name` | String | `None` | The account key of the azure account.<br/>**It's only used when the storage type is `Azblob`**. |
+| `storage.account_key` | String | `None` | The account key of the azure account.<br/>**It's only used when the storage type is `Azblob`**. |
+| `storage.scope` | String | `None` | The scope of the google cloud storage.<br/>**It's only used when the storage type is `Gcs`**. |
+| `storage.credential_path` | String | `None` | The credential path of the google cloud storage.<br/>**It's only used when the storage type is `Gcs`**. |
+| `storage.container` | String | `None` | The container of the azure account.<br/>**It's only used when the storage type is `Azblob`**. |
+| `storage.sas_token` | String | `None` | The sas token of the azure account.<br/>**It's only used when the storage type is `Azblob`**. |
+| `storage.endpoint` | String | `None` | The endpoint of the S3 service.<br/>**It's only used when the storage type is `S3`, `Oss`, `Gcs` and `Azblob`**. |
+| `storage.region` | String | `None` | The region of the S3 service.<br/>**It's only used when the storage type is `S3`, `Oss`, `Gcs` and `Azblob`**. |
+| `[[region_engine]]` | -- | -- | The region engine options. You can configure multiple region engines. |
+| `region_engine.mito` | -- | -- | The Mito engine options. |
+| `region_engine.mito.num_workers` | Integer | `8` | Number of region workers. |
+| `region_engine.mito.worker_channel_size` | Integer | `128` | Request channel size of each worker. |
+| `region_engine.mito.worker_request_batch_size` | Integer | `64` | Max batch size for a worker to handle requests. |
+| `region_engine.mito.manifest_checkpoint_distance` | Integer | `10` | Number of meta action updated to trigger a new checkpoint for the manifest. |
+| `region_engine.mito.compress_manifest` | Bool | `false` | Whether to compress manifest and checkpoint file by gzip (default false). |
+| `region_engine.mito.max_background_jobs` | Integer | `4` | Max number of running background jobs |
+| `region_engine.mito.auto_flush_interval` | String | `1h` | Interval to auto flush a region if it has not flushed yet. |
+| `region_engine.mito.global_write_buffer_size` | String | `1GB` | Global write buffer size for all regions. If not set, it's default to 1/8 of OS memory with a max limitation of 1GB. |
+| `region_engine.mito.global_write_buffer_reject_size` | String | `2GB` | Global write buffer size threshold to reject write requests. If not set, it's default to 2 times of `global_write_buffer_size` |
+| `region_engine.mito.sst_meta_cache_size` | String | `128MB` | Cache size for SST metadata. Setting it to 0 to disable the cache.<br/>If not set, it's default to 1/32 of OS memory with a max limitation of 128MB. |
+| `region_engine.mito.vector_cache_size` | String | `512MB` | Cache size for vectors and arrow arrays. Setting it to 0 to disable the cache.<br/>If not set, it's default to 1/16 of OS memory with a max limitation of 512MB. |
+| `region_engine.mito.page_cache_size` | String | `512MB` | Cache size for pages of SST row groups. Setting it to 0 to disable the cache.<br/>If not set, it's default to 1/16 of OS memory with a max limitation of 512MB. |
+| `region_engine.mito.sst_write_buffer_size` | String | `8MB` | Buffer size for SST writing. |
+| `region_engine.mito.scan_parallelism` | Integer | `0` | Parallelism to scan a region (default: 1/4 of cpu cores).<br/>- `0`: using the default value (1/4 of cpu cores).<br/>- `1`: scan in current thread.<br/>- `n`: scan in parallelism n. |
+| `region_engine.mito.parallel_scan_channel_size` | Integer | `32` | Capacity of the channel to send data from parallel scan tasks to the main task. |
+| `region_engine.mito.allow_stale_entries` | Bool | `false` | Whether to allow stale WAL entries read during replay. |
+| `region_engine.mito.inverted_index` | -- | -- | The options for inverted index in Mito engine. |
+| `region_engine.mito.inverted_index.create_on_flush` | String | `auto` | Whether to create the index on flush.<br/>- `auto`: automatically<br/>- `disable`: never |
+| `region_engine.mito.inverted_index.create_on_compaction` | String | `auto` | Whether to create the index on compaction.<br/>- `auto`: automatically<br/>- `disable`: never |
+| `region_engine.mito.inverted_index.apply_on_query` | String | `auto` | Whether to apply the index on query<br/>- `auto`: automatically<br/>- `disable`: never |
+| `region_engine.mito.inverted_index.mem_threshold_on_create` | String | `64M` | Memory threshold for performing an external sort during index creation.<br/>Setting to empty will disable external sorting, forcing all sorting operations to happen in memory. |
+| `region_engine.mito.inverted_index.intermediate_path` | String | `""` | File system path to store intermediate files for external sorting (default `{data_home}/index_intermediate`). |
+| `region_engine.mito.memtable` | -- | -- | -- |
+| `region_engine.mito.memtable.type` | String | `time_series` | Memtable type.<br/>- `time_series`: time-series memtable<br/>- `partition_tree`: partition tree memtable (experimental) |
+| `region_engine.mito.memtable.index_max_keys_per_shard` | Integer | `8192` | The max number of keys in one shard.<br/>Only available for `partition_tree` memtable. |
+| `region_engine.mito.memtable.data_freeze_threshold` | Integer | `32768` | The max rows of data inside the actively writing buffer in one shard.<br/>Only available for `partition_tree` memtable. |
+| `region_engine.mito.memtable.fork_dictionary_bytes` | String | `1GiB` | Max dictionary bytes.<br/>Only available for `partition_tree` memtable. |
+| `logging` | -- | -- | The logging options. |
+| `logging.dir` | String | `/tmp/greptimedb/logs` | The directory to store the log files. |
+| `logging.level` | String | `None` | The log level. Can be `info`/`debug`/`warn`/`error`. |
+| `logging.enable_otlp_tracing` | Bool | `false` | Enable OTLP tracing. |
+| `logging.otlp_endpoint` | String | `None` | The OTLP tracing endpoint. |
+| `logging.append_stdout` | Bool | `true` | Whether to append logs to stdout. |
+| `logging.tracing_sample_ratio` | -- | -- | The percentage of tracing will be sampled and exported.<br/>Valid range `[0, 1]`, 1 means all traces are sampled, 0 means all traces are not sampled, the default value is 1.<br/>ratio > 1 are treated as 1. Fractions < 0 are treated as 0 |
+| `logging.tracing_sample_ratio.default_ratio` | Float | `1.0` | -- |
+| `export_metrics` | -- | -- | The datanode can export its metrics and send to Prometheus compatible service (e.g. send to `greptimedb` itself) from remote-write API.<br/>This is only used for `greptimedb` to export its own metrics internally. It's different from prometheus scrape. |
+| `export_metrics.enable` | Bool | `false` | whether enable export metrics. |
+| `export_metrics.write_interval` | String | `30s` | The interval of export metrics. |
+| `export_metrics.self_import` | -- | -- | For `standalone` mode, `self_import` is recommend to collect metrics generated by itself |
+| `export_metrics.self_import.db` | String | `None` | -- |
+| `export_metrics.remote_write` | -- | -- | -- |
+| `export_metrics.remote_write.url` | String | `""` | The url the metrics send to. The url example can be: `http://127.0.0.1:4000/v1/prometheus/write?db=information_schema`. |
+| `export_metrics.remote_write.headers` | InlineTable | -- | HTTP headers of Prometheus remote-write carry. |
+
+
+## Cluster Mode
+
+### Frontend
+
+| Key | Type | Default | Descriptions |
+| --- | -----| ------- | ----------- |
+| `mode` | String | `standalone` | The running mode of the datanode. It can be `standalone` or `distributed`. |
+| `default_timezone` | String | `None` | The default timezone of the server. |
+| `heartbeat` | -- | -- | The heartbeat options. |
+| `heartbeat.interval` | String | `18s` | Interval for sending heartbeat messages to the metasrv. |
+| `heartbeat.retry_interval` | String | `3s` | Interval for retrying to send heartbeat messages to the metasrv. |
+| `http` | -- | -- | The HTTP server options. |
+| `http.addr` | String | `127.0.0.1:4000` | The address to bind the HTTP server. |
+| `http.timeout` | String | `30s` | HTTP request timeout. |
+| `http.body_limit` | String | `64MB` | HTTP request body limit.<br/>Support the following units are supported: `B`, `KB`, `KiB`, `MB`, `MiB`, `GB`, `GiB`, `TB`, `TiB`, `PB`, `PiB`. |
+| `grpc` | -- | -- | The gRPC server options. |
+| `grpc.addr` | String | `127.0.0.1:4001` | The address to bind the gRPC server. |
+| `grpc.runtime_size` | Integer | `8` | The number of server worker threads. |
+| `mysql` | -- | -- | MySQL server options. |
+| `mysql.enable` | Bool | `true` | Whether to enable. |
+| `mysql.addr` | String | `127.0.0.1:4002` | The addr to bind the MySQL server. |
+| `mysql.runtime_size` | Integer | `2` | The number of server worker threads. |
+| `mysql.tls` | -- | -- | -- |
+| `mysql.tls.mode` | String | `disable` | TLS mode, refer to https://www.postgresql.org/docs/current/libpq-ssl.html<br/>- `disable` (default value)<br/>- `prefer`<br/>- `require`<br/>- `verify-ca`<br/>- `verify-full` |
+| `mysql.tls.cert_path` | String | `None` | Certificate file path. |
+| `mysql.tls.key_path` | String | `None` | Private key file path. |
+| `mysql.tls.watch` | Bool | `false` | Watch for Certificate and key file change and auto reload |
+| `postgres` | -- | -- | PostgresSQL server options. |
+| `postgres.enable` | Bool | `true` | Whether to enable |
+| `postgres.addr` | String | `127.0.0.1:4003` | The addr to bind the PostgresSQL server. |
+| `postgres.runtime_size` | Integer | `2` | The number of server worker threads. |
+| `postgres.tls` | -- | -- | PostgresSQL server TLS options, see `mysql_options.tls` section. |
+| `postgres.tls.mode` | String | `disable` | TLS mode. |
+| `postgres.tls.cert_path` | String | `None` | Certificate file path. |
+| `postgres.tls.key_path` | String | `None` | Private key file path. |
+| `postgres.tls.watch` | Bool | `false` | Watch for Certificate and key file change and auto reload |
+| `opentsdb` | -- | -- | OpenTSDB protocol options. |
+| `opentsdb.enable` | Bool | `true` | Whether to enable |
+| `opentsdb.addr` | String | `127.0.0.1:4242` | OpenTSDB telnet API server address. |
+| `opentsdb.runtime_size` | Integer | `2` | The number of server worker threads. |
+| `influxdb` | -- | -- | InfluxDB protocol options. |
+| `influxdb.enable` | Bool | `true` | Whether to enable InfluxDB protocol in HTTP API. |
+| `prom_store` | -- | -- | Prometheus remote storage options |
+| `prom_store.enable` | Bool | `true` | Whether to enable Prometheus remote write and read in HTTP API. |
+| `prom_store.with_metric_engine` | Bool | `true` | Whether to store the data from Prometheus remote write in metric engine. |
+| `meta_client` | -- | -- | The metasrv client options. |
+| `meta_client.metasrv_addrs` | Array | -- | The addresses of the metasrv. |
+| `meta_client.timeout` | String | `3s` | Operation timeout. |
+| `meta_client.heartbeat_timeout` | String | `500ms` | Heartbeat timeout. |
+| `meta_client.ddl_timeout` | String | `10s` | DDL timeout. |
+| `meta_client.connect_timeout` | String | `1s` | Connect server timeout. |
+| `meta_client.tcp_nodelay` | Bool | `true` | `TCP_NODELAY` option for accepted connections. |
+| `meta_client.metadata_cache_max_capacity` | Integer | `100000` | The configuration about the cache of the metadata. |
+| `meta_client.metadata_cache_ttl` | String | `10m` | TTL of the metadata cache. |
+| `meta_client.metadata_cache_tti` | String | `5m` | -- |
+| `datanode` | -- | -- | Datanode options. |
+| `datanode.client` | -- | -- | Datanode client options. |
+| `datanode.client.timeout` | String | `10s` | -- |
+| `datanode.client.connect_timeout` | String | `10s` | -- |
+| `datanode.client.tcp_nodelay` | Bool | `true` | -- |
+| `logging` | -- | -- | The logging options. |
+| `logging.dir` | String | `/tmp/greptimedb/logs` | The directory to store the log files. |
+| `logging.level` | String | `None` | The log level. Can be `info`/`debug`/`warn`/`error`. |
+| `logging.enable_otlp_tracing` | Bool | `false` | Enable OTLP tracing. |
+| `logging.otlp_endpoint` | String | `None` | The OTLP tracing endpoint. |
+| `logging.append_stdout` | Bool | `true` | Whether to append logs to stdout. |
+| `logging.tracing_sample_ratio` | -- | -- | The percentage of tracing will be sampled and exported.<br/>Valid range `[0, 1]`, 1 means all traces are sampled, 0 means all traces are not sampled, the default value is 1.<br/>ratio > 1 are treated as 1. Fractions < 0 are treated as 0 |
+| `logging.tracing_sample_ratio.default_ratio` | Float | `1.0` | -- |
+| `export_metrics` | -- | -- | The datanode can export its metrics and send to Prometheus compatible service (e.g. send to `greptimedb` itself) from remote-write API.<br/>This is only used for `greptimedb` to export its own metrics internally. It's different from prometheus scrape. |
+| `export_metrics.enable` | Bool | `false` | whether enable export metrics. |
+| `export_metrics.write_interval` | String | `30s` | The interval of export metrics. |
+| `export_metrics.self_import` | -- | -- | For `standalone` mode, `self_import` is recommend to collect metrics generated by itself |
+| `export_metrics.self_import.db` | String | `None` | -- |
+| `export_metrics.remote_write` | -- | -- | -- |
+| `export_metrics.remote_write.url` | String | `""` | The url the metrics send to. The url example can be: `http://127.0.0.1:4000/v1/prometheus/write?db=information_schema`. |
+| `export_metrics.remote_write.headers` | InlineTable | -- | HTTP headers of Prometheus remote-write carry. |
+
+
+### Metasrv
+
+| Key | Type | Default | Descriptions |
+| --- | -----| ------- | ----------- |
+| `data_home` | String | `/tmp/metasrv/` | The working home directory. |
+| `bind_addr` | String | `127.0.0.1:3002` | The bind address of metasrv. |
+| `server_addr` | String | `127.0.0.1:3002` | The communication server address for frontend and datanode to connect to metasrv,  "127.0.0.1:3002" by default for localhost. |
+| `store_addr` | String | `127.0.0.1:2379` | Etcd server address. |
+| `selector` | String | `lease_based` | Datanode selector type.<br/>- `lease_based` (default value).<br/>- `load_based`<br/>For details, please see "https://docs.greptime.com/developer-guide/metasrv/selector". |
+| `use_memory_store` | Bool | `false` | Store data in memory. |
+| `enable_telemetry` | Bool | `true` | Whether to enable greptimedb telemetry. |
+| `store_key_prefix` | String | `""` | If it's not empty, the metasrv will store all data with this key prefix. |
+| `procedure` | -- | -- | Procedure storage options. |
+| `procedure.max_retry_times` | Integer | `12` | Procedure max retry time. |
+| `procedure.retry_delay` | String | `500ms` | Initial retry delay of procedures, increases exponentially |
+| `procedure.max_metadata_value_size` | String | `1500KiB` | Auto split large value<br/>GreptimeDB procedure uses etcd as the default metadata storage backend.<br/>The etcd the maximum size of any request is 1.5 MiB<br/>1500KiB = 1536KiB (1.5MiB) - 36KiB (reserved size of key)<br/>Comments out the `max_metadata_value_size`, for don't split large value (no limit). |
+| `failure_detector` | -- | -- | -- |
+| `failure_detector.threshold` | Float | `8.0` | -- |
+| `failure_detector.min_std_deviation` | String | `100ms` | -- |
+| `failure_detector.acceptable_heartbeat_pause` | String | `3000ms` | -- |
+| `failure_detector.first_heartbeat_estimate` | String | `1000ms` | -- |
+| `datanode` | -- | -- | Datanode options. |
+| `datanode.client` | -- | -- | Datanode client options. |
+| `datanode.client.timeout` | String | `10s` | -- |
+| `datanode.client.connect_timeout` | String | `10s` | -- |
+| `datanode.client.tcp_nodelay` | Bool | `true` | -- |
+| `wal` | -- | -- | -- |
+| `wal.provider` | String | `raft_engine` | -- |
+| `wal.broker_endpoints` | Array | -- | The broker endpoints of the Kafka cluster. |
+| `wal.num_topics` | Integer | `64` | Number of topics to be created upon start. |
+| `wal.selector_type` | String | `round_robin` | Topic selector type.<br/>Available selector types:<br/>- `round_robin` (default) |
+| `wal.topic_name_prefix` | String | `greptimedb_wal_topic` | A Kafka topic is constructed by concatenating `topic_name_prefix` and `topic_id`. |
+| `wal.replication_factor` | Integer | `1` | Expected number of replicas of each partition. |
+| `wal.create_topic_timeout` | String | `30s` | Above which a topic creation operation will be cancelled. |
+| `wal.backoff_init` | String | `500ms` | The initial backoff for kafka clients. |
+| `wal.backoff_max` | String | `10s` | The maximum backoff for kafka clients. |
+| `wal.backoff_base` | Integer | `2` | Exponential backoff rate, i.e. next backoff = base * current backoff. |
+| `wal.backoff_deadline` | String | `5mins` | Stop reconnecting if the total wait time reaches the deadline. If this config is missing, the reconnecting won't terminate. |
+| `logging` | -- | -- | The logging options. |
+| `logging.dir` | String | `/tmp/greptimedb/logs` | The directory to store the log files. |
+| `logging.level` | String | `None` | The log level. Can be `info`/`debug`/`warn`/`error`. |
+| `logging.enable_otlp_tracing` | Bool | `false` | Enable OTLP tracing. |
+| `logging.otlp_endpoint` | String | `None` | The OTLP tracing endpoint. |
+| `logging.append_stdout` | Bool | `true` | Whether to append logs to stdout. |
+| `logging.tracing_sample_ratio` | -- | -- | The percentage of tracing will be sampled and exported.<br/>Valid range `[0, 1]`, 1 means all traces are sampled, 0 means all traces are not sampled, the default value is 1.<br/>ratio > 1 are treated as 1. Fractions < 0 are treated as 0 |
+| `logging.tracing_sample_ratio.default_ratio` | Float | `1.0` | -- |
+| `export_metrics` | -- | -- | The datanode can export its metrics and send to Prometheus compatible service (e.g. send to `greptimedb` itself) from remote-write API.<br/>This is only used for `greptimedb` to export its own metrics internally. It's different from prometheus scrape. |
+| `export_metrics.enable` | Bool | `false` | whether enable export metrics. |
+| `export_metrics.write_interval` | String | `30s` | The interval of export metrics. |
+| `export_metrics.self_import` | -- | -- | For `standalone` mode, `self_import` is recommend to collect metrics generated by itself |
+| `export_metrics.self_import.db` | String | `None` | -- |
+| `export_metrics.remote_write` | -- | -- | -- |
+| `export_metrics.remote_write.url` | String | `""` | The url the metrics send to. The url example can be: `http://127.0.0.1:4000/v1/prometheus/write?db=information_schema`. |
+| `export_metrics.remote_write.headers` | InlineTable | -- | HTTP headers of Prometheus remote-write carry. |
+
+
+### Datanode
+
+| Key | Type | Default | Descriptions |
+| --- | -----| ------- | ----------- |
+| `mode` | String | `standalone` | The running mode of the datanode. It can be `standalone` or `distributed`. |
+| `node_id` | Integer | `None` | The datanode identifier and should be unique in the cluster. |
+| `require_lease_before_startup` | Bool | `false` | Start services after regions have obtained leases.<br/>It will block the datanode start if it can't receive leases in the heartbeat from metasrv. |
+| `init_regions_in_background` | Bool | `false` | Initialize all regions in the background during the startup.<br/>By default, it provides services after all regions have been initialized. |
+| `rpc_addr` | String | `127.0.0.1:3001` | The gRPC address of the datanode. |
+| `rpc_hostname` | String | `None` | The hostname of the datanode. |
+| `rpc_runtime_size` | Integer | `8` | The number of gRPC server worker threads. |
+| `rpc_max_recv_message_size` | String | `512MB` | The maximum receive message size for gRPC server. |
+| `rpc_max_send_message_size` | String | `512MB` | The maximum send message size for gRPC server. |
+| `enable_telemetry` | Bool | `true` | Enable telemetry to collect anonymous usage data. |
+| `heartbeat` | -- | -- | The heartbeat options. |
+| `heartbeat.interval` | String | `3s` | Interval for sending heartbeat messages to the metasrv. |
+| `heartbeat.retry_interval` | String | `3s` | Interval for retrying to send heartbeat messages to the metasrv. |
+| `meta_client` | -- | -- | The metasrv client options. |
+| `meta_client.metasrv_addrs` | Array | -- | The addresses of the metasrv. |
+| `meta_client.timeout` | String | `3s` | Operation timeout. |
+| `meta_client.heartbeat_timeout` | String | `500ms` | Heartbeat timeout. |
+| `meta_client.ddl_timeout` | String | `10s` | DDL timeout. |
+| `meta_client.connect_timeout` | String | `1s` | Connect server timeout. |
+| `meta_client.tcp_nodelay` | Bool | `true` | `TCP_NODELAY` option for accepted connections. |
+| `meta_client.metadata_cache_max_capacity` | Integer | `100000` | The configuration about the cache of the metadata. |
+| `meta_client.metadata_cache_ttl` | String | `10m` | TTL of the metadata cache. |
+| `meta_client.metadata_cache_tti` | String | `5m` | -- |
+| `wal` | -- | -- | The WAL options. |
+| `wal.provider` | String | `raft_engine` | The provider of the WAL.<br/>- `raft_engine`: the wal is stored in the local file system by raft-engine.<br/>- `kafka`: it's remote wal that data is stored in Kafka. |
+| `wal.dir` | String | `None` | The directory to store the WAL files.<br/>**It's only used when the provider is `raft_engine`**. |
+| `wal.file_size` | String | `256MB` | The size of the WAL segment file.<br/>**It's only used when the provider is `raft_engine`**. |
+| `wal.purge_threshold` | String | `4GB` | The threshold of the WAL size to trigger a flush.<br/>**It's only used when the provider is `raft_engine`**. |
+| `wal.purge_interval` | String | `10m` | The interval to trigger a flush.<br/>**It's only used when the provider is `raft_engine`**. |
+| `wal.read_batch_size` | Integer | `128` | The read batch size.<br/>**It's only used when the provider is `raft_engine`**. |
+| `wal.sync_write` | Bool | `false` | Whether to use sync write.<br/>**It's only used when the provider is `raft_engine`**. |
+| `wal.enable_log_recycle` | Bool | `true` | Whether to reuse logically truncated log files.<br/>**It's only used when the provider is `raft_engine`**. |
+| `wal.prefill_log_files` | Bool | `false` | Whether to pre-create log files on start up.<br/>**It's only used when the provider is `raft_engine`**. |
+| `wal.sync_period` | String | `10s` | Duration for fsyncing log files.<br/>**It's only used when the provider is `raft_engine`**. |
+| `wal.broker_endpoints` | Array | -- | The Kafka broker endpoints.<br/>**It's only used when the provider is `kafka`**. |
+| `wal.max_batch_size` | String | `1MB` | The max size of a single producer batch.<br/>Warning: Kafka has a default limit of 1MB per message in a topic.<br/>**It's only used when the provider is `kafka`**. |
+| `wal.linger` | String | `200ms` | The linger duration of a kafka batch producer.<br/>**It's only used when the provider is `kafka`**. |
+| `wal.consumer_wait_timeout` | String | `100ms` | The consumer wait timeout.<br/>**It's only used when the provider is `kafka`**. |
+| `wal.backoff_init` | String | `500ms` | The initial backoff delay.<br/>**It's only used when the provider is `kafka`**. |
+| `wal.backoff_max` | String | `10s` | The maximum backoff delay.<br/>**It's only used when the provider is `kafka`**. |
+| `wal.backoff_base` | Integer | `2` | The exponential backoff rate, i.e. next backoff = base * current backoff.<br/>**It's only used when the provider is `kafka`**. |
+| `wal.backoff_deadline` | String | `5mins` | The deadline of retries.<br/>**It's only used when the provider is `kafka`**. |
+| `storage` | -- | -- | The data storage options. |
+| `storage.data_home` | String | `/tmp/greptimedb/` | The working home directory. |
+| `storage.type` | String | `File` | The storage type used to store the data.<br/>- `File`: the data is stored in the local file system.<br/>- `S3`: the data is stored in the S3 object storage.<br/>- `Gcs`: the data is stored in the Google Cloud Storage.<br/>- `Azblob`: the data is stored in the Azure Blob Storage.<br/>- `Oss`: the data is stored in the Aliyun OSS. |
+| `storage.cache_path` | String | `None` | Cache configuration for object storage such as 'S3' etc.<br/>The local file cache directory. |
+| `storage.cache_capacity` | String | `None` | The local file cache capacity in bytes. |
+| `storage.bucket` | String | `None` | The S3 bucket name.<br/>**It's only used when the storage type is `S3`, `Oss` and `Gcs`**. |
+| `storage.root` | String | `None` | The S3 data will be stored in the specified prefix, for example, `s3://${bucket}/${root}`.<br/>**It's only used when the storage type is `S3`, `Oss` and `Azblob`**. |
+| `storage.access_key_id` | String | `None` | The access key id of the aws account.<br/>It's **highly recommended** to use AWS IAM roles instead of hardcoding the access key id and secret key.<br/>**It's only used when the storage type is `S3` and `Oss`**. |
+| `storage.secret_access_key` | String | `None` | The secret access key of the aws account.<br/>It's **highly recommended** to use AWS IAM roles instead of hardcoding the access key id and secret key.<br/>**It's only used when the storage type is `S3`**. |
+| `storage.access_key_secret` | String | `None` | The secret access key of the aliyun account.<br/>**It's only used when the storage type is `Oss`**. |
+| `storage.account_name` | String | `None` | The account key of the azure account.<br/>**It's only used when the storage type is `Azblob`**. |
+| `storage.account_key` | String | `None` | The account key of the azure account.<br/>**It's only used when the storage type is `Azblob`**. |
+| `storage.scope` | String | `None` | The scope of the google cloud storage.<br/>**It's only used when the storage type is `Gcs`**. |
+| `storage.credential_path` | String | `None` | The credential path of the google cloud storage.<br/>**It's only used when the storage type is `Gcs`**. |
+| `storage.container` | String | `None` | The container of the azure account.<br/>**It's only used when the storage type is `Azblob`**. |
+| `storage.sas_token` | String | `None` | The sas token of the azure account.<br/>**It's only used when the storage type is `Azblob`**. |
+| `storage.endpoint` | String | `None` | The endpoint of the S3 service.<br/>**It's only used when the storage type is `S3`, `Oss`, `Gcs` and `Azblob`**. |
+| `storage.region` | String | `None` | The region of the S3 service.<br/>**It's only used when the storage type is `S3`, `Oss`, `Gcs` and `Azblob`**. |
+| `[[region_engine]]` | -- | -- | The region engine options. You can configure multiple region engines. |
+| `region_engine.mito` | -- | -- | The Mito engine options. |
+| `region_engine.mito.num_workers` | Integer | `8` | Number of region workers. |
+| `region_engine.mito.worker_channel_size` | Integer | `128` | Request channel size of each worker. |
+| `region_engine.mito.worker_request_batch_size` | Integer | `64` | Max batch size for a worker to handle requests. |
+| `region_engine.mito.manifest_checkpoint_distance` | Integer | `10` | Number of meta action updated to trigger a new checkpoint for the manifest. |
+| `region_engine.mito.compress_manifest` | Bool | `false` | Whether to compress manifest and checkpoint file by gzip (default false). |
+| `region_engine.mito.max_background_jobs` | Integer | `4` | Max number of running background jobs |
+| `region_engine.mito.auto_flush_interval` | String | `1h` | Interval to auto flush a region if it has not flushed yet. |
+| `region_engine.mito.global_write_buffer_size` | String | `1GB` | Global write buffer size for all regions. If not set, it's default to 1/8 of OS memory with a max limitation of 1GB. |
+| `region_engine.mito.global_write_buffer_reject_size` | String | `2GB` | Global write buffer size threshold to reject write requests. If not set, it's default to 2 times of `global_write_buffer_size` |
+| `region_engine.mito.sst_meta_cache_size` | String | `128MB` | Cache size for SST metadata. Setting it to 0 to disable the cache.<br/>If not set, it's default to 1/32 of OS memory with a max limitation of 128MB. |
+| `region_engine.mito.vector_cache_size` | String | `512MB` | Cache size for vectors and arrow arrays. Setting it to 0 to disable the cache.<br/>If not set, it's default to 1/16 of OS memory with a max limitation of 512MB. |
+| `region_engine.mito.page_cache_size` | String | `512MB` | Cache size for pages of SST row groups. Setting it to 0 to disable the cache.<br/>If not set, it's default to 1/16 of OS memory with a max limitation of 512MB. |
+| `region_engine.mito.sst_write_buffer_size` | String | `8MB` | Buffer size for SST writing. |
+| `region_engine.mito.scan_parallelism` | Integer | `0` | Parallelism to scan a region (default: 1/4 of cpu cores).<br/>- `0`: using the default value (1/4 of cpu cores).<br/>- `1`: scan in current thread.<br/>- `n`: scan in parallelism n. |
+| `region_engine.mito.parallel_scan_channel_size` | Integer | `32` | Capacity of the channel to send data from parallel scan tasks to the main task. |
+| `region_engine.mito.allow_stale_entries` | Bool | `false` | Whether to allow stale WAL entries read during replay. |
+| `region_engine.mito.inverted_index` | -- | -- | The options for inverted index in Mito engine. |
+| `region_engine.mito.inverted_index.create_on_flush` | String | `auto` | Whether to create the index on flush.<br/>- `auto`: automatically<br/>- `disable`: never |
+| `region_engine.mito.inverted_index.create_on_compaction` | String | `auto` | Whether to create the index on compaction.<br/>- `auto`: automatically<br/>- `disable`: never |
+| `region_engine.mito.inverted_index.apply_on_query` | String | `auto` | Whether to apply the index on query<br/>- `auto`: automatically<br/>- `disable`: never |
+| `region_engine.mito.inverted_index.mem_threshold_on_create` | String | `64M` | Memory threshold for performing an external sort during index creation.<br/>Setting to empty will disable external sorting, forcing all sorting operations to happen in memory. |
+| `region_engine.mito.inverted_index.intermediate_path` | String | `""` | File system path to store intermediate files for external sorting (default `{data_home}/index_intermediate`). |
+| `region_engine.mito.memtable` | -- | -- | -- |
+| `region_engine.mito.memtable.type` | String | `time_series` | Memtable type.<br/>- `time_series`: time-series memtable<br/>- `partition_tree`: partition tree memtable (experimental) |
+| `region_engine.mito.memtable.index_max_keys_per_shard` | Integer | `8192` | The max number of keys in one shard.<br/>Only available for `partition_tree` memtable. |
+| `region_engine.mito.memtable.data_freeze_threshold` | Integer | `32768` | The max rows of data inside the actively writing buffer in one shard.<br/>Only available for `partition_tree` memtable. |
+| `region_engine.mito.memtable.fork_dictionary_bytes` | String | `1GiB` | Max dictionary bytes.<br/>Only available for `partition_tree` memtable. |
+| `logging` | -- | -- | The logging options. |
+| `logging.dir` | String | `/tmp/greptimedb/logs` | The directory to store the log files. |
+| `logging.level` | String | `None` | The log level. Can be `info`/`debug`/`warn`/`error`. |
+| `logging.enable_otlp_tracing` | Bool | `false` | Enable OTLP tracing. |
+| `logging.otlp_endpoint` | String | `None` | The OTLP tracing endpoint. |
+| `logging.append_stdout` | Bool | `true` | Whether to append logs to stdout. |
+| `logging.tracing_sample_ratio` | -- | -- | The percentage of tracing will be sampled and exported.<br/>Valid range `[0, 1]`, 1 means all traces are sampled, 0 means all traces are not sampled, the default value is 1.<br/>ratio > 1 are treated as 1. Fractions < 0 are treated as 0 |
+| `logging.tracing_sample_ratio.default_ratio` | Float | `1.0` | -- |
+| `export_metrics` | -- | -- | The datanode can export its metrics and send to Prometheus compatible service (e.g. send to `greptimedb` itself) from remote-write API.<br/>This is only used for `greptimedb` to export its own metrics internally. It's different from prometheus scrape. |
+| `export_metrics.enable` | Bool | `false` | whether enable export metrics. |
+| `export_metrics.write_interval` | String | `30s` | The interval of export metrics. |
+| `export_metrics.self_import` | -- | -- | For `standalone` mode, `self_import` is recommend to collect metrics generated by itself |
+| `export_metrics.self_import.db` | String | `None` | -- |
+| `export_metrics.remote_write` | -- | -- | -- |
+| `export_metrics.remote_write.url` | String | `""` | The url the metrics send to. The url example can be: `http://127.0.0.1:4000/v1/prometheus/write?db=information_schema`. |
+| `export_metrics.remote_write.headers` | InlineTable | -- | HTTP headers of Prometheus remote-write carry. |
--- a/config/datanode.example.toml
+++ b/config/datanode.example.toml
@@ -1,171 +1,430 @@
-# Node running mode, see `standalone.example.toml`.
-mode = "distributed"
-# The datanode identifier, should be unique.
+## The running mode of the datanode. It can be `standalone` or `distributed`.
+mode = "standalone"
+
+## The datanode identifier and should be unique in the cluster.
+## +toml2docs:none-default
 node_id = 42
-# gRPC server address, "127.0.0.1:3001" by default.
-rpc_addr = "127.0.0.1:3001"
-# Hostname of this node.
-rpc_hostname = "127.0.0.1"
-# The number of gRPC server worker threads, 8 by default.
-rpc_runtime_size = 8
-# Start services after regions have obtained leases.
-# It will block the datanode start if it can't receive leases in the heartbeat from metasrv.
+
+## Start services after regions have obtained leases.
+## It will block the datanode start if it can't receive leases in the heartbeat from metasrv.
 require_lease_before_startup = false

-# Initialize all regions in the background during the startup.
-# By default, it provides services after all regions have been initialized.
+## Initialize all regions in the background during the startup.
+## By default, it provides services after all regions have been initialized.
 init_regions_in_background = false

+## The gRPC address of the datanode.
+rpc_addr = "127.0.0.1:3001"
+
+## The hostname of the datanode.
+## +toml2docs:none-default
+rpc_hostname = "127.0.0.1"
+
+## The number of gRPC server worker threads.
+rpc_runtime_size = 8
+
+## The maximum receive message size for gRPC server.
+rpc_max_recv_message_size = "512MB"
+
+## The maximum send message size for gRPC server.
+rpc_max_send_message_size = "512MB"
+
+## Enable telemetry to collect anonymous usage data.
+enable_telemetry = true
+
+## The heartbeat options.
 [heartbeat]
-# Interval for sending heartbeat messages to the Metasrv, 3 seconds by default.
+## Interval for sending heartbeat messages to the metasrv.
 interval = "3s"

-# Metasrv client options.
+## Interval for retrying to send heartbeat messages to the metasrv.
+retry_interval = "3s"
+
+## The metasrv client options.
 [meta_client]
-# Metasrv address list.
+## The addresses of the metasrv.
 metasrv_addrs = ["127.0.0.1:3002"]
-# Heartbeat timeout, 500 milliseconds by default.
-heartbeat_timeout = "500ms"
-# Operation timeout, 3 seconds by default.
+
+## Operation timeout.
 timeout = "3s"
-# Connect server timeout, 1 second by default.
+
+## Heartbeat timeout.
+heartbeat_timeout = "500ms"
+
+## DDL timeout.
+ddl_timeout = "10s"
+
+## Connect server timeout.
 connect_timeout = "1s"
-# `TCP_NODELAY` option for accepted connections, true by default.
+
+## `TCP_NODELAY` option for accepted connections.
 tcp_nodelay = true

-# WAL options.
+## The configuration about the cache of the metadata.
+metadata_cache_max_capacity = 100000
+
+## TTL of the metadata cache.
+metadata_cache_ttl = "10m"
+
+# TTI of the metadata cache.
+metadata_cache_tti = "5m"
+
+## The WAL options.
 [wal]
+## The provider of the WAL.
+## - `raft_engine`: the wal is stored in the local file system by raft-engine.
+## - `kafka`: it's remote wal that data is stored in Kafka.
 provider = "raft_engine"

-# Raft-engine wal options, see `standalone.example.toml`.
-# dir = "/tmp/greptimedb/wal"
+## The directory to store the WAL files.
+## **It's only used when the provider is `raft_engine`**.
+## +toml2docs:none-default
+dir = "/tmp/greptimedb/wal"
+
+## The size of the WAL segment file.
+## **It's only used when the provider is `raft_engine`**.
 file_size = "256MB"
+
+## The threshold of the WAL size to trigger a flush.
+## **It's only used when the provider is `raft_engine`**.
 purge_threshold = "4GB"
+
+## The interval to trigger a flush.
+## **It's only used when the provider is `raft_engine`**.
 purge_interval = "10m"
+
+## The read batch size.
+## **It's only used when the provider is `raft_engine`**.
 read_batch_size = 128
+
+## Whether to use sync write.
+## **It's only used when the provider is `raft_engine`**.
 sync_write = false

-# Kafka wal options, see `standalone.example.toml`.
-# broker_endpoints = ["127.0.0.1:9092"]
-# Warning: Kafka has a default limit of 1MB per message in a topic.
-# max_batch_size = "1MB"
-# linger = "200ms"
-# consumer_wait_timeout = "100ms"
-# backoff_init = "500ms"
-# backoff_max = "10s"
-# backoff_base = 2
-# backoff_deadline = "5mins"
+## Whether to reuse logically truncated log files.
+## **It's only used when the provider is `raft_engine`**.
+enable_log_recycle = true

-# Storage options, see `standalone.example.toml`.
+## Whether to pre-create log files on start up.
+## **It's only used when the provider is `raft_engine`**.
+prefill_log_files = false
+
+## Duration for fsyncing log files.
+## **It's only used when the provider is `raft_engine`**.
+sync_period = "10s"
+
+## The Kafka broker endpoints.
+## **It's only used when the provider is `kafka`**.
+broker_endpoints = ["127.0.0.1:9092"]
+
+## The max size of a single producer batch.
+## Warning: Kafka has a default limit of 1MB per message in a topic.
+## **It's only used when the provider is `kafka`**.
+max_batch_size = "1MB"
+
+## The linger duration of a kafka batch producer.
+## **It's only used when the provider is `kafka`**.
+linger = "200ms"
+
+## The consumer wait timeout.
+## **It's only used when the provider is `kafka`**.
+consumer_wait_timeout = "100ms"
+
+## The initial backoff delay.
+## **It's only used when the provider is `kafka`**.
+backoff_init = "500ms"
+
+## The maximum backoff delay.
+## **It's only used when the provider is `kafka`**.
+backoff_max = "10s"
+
+## The exponential backoff rate, i.e. next backoff = base * current backoff.
+## **It's only used when the provider is `kafka`**.
+backoff_base = 2
+
+## The deadline of retries.
+## **It's only used when the provider is `kafka`**.
+backoff_deadline = "5mins"
+
+# Example of using S3 as the storage.
+# [storage]
+# type = "S3"
+# bucket = "greptimedb"
+# root = "data"
+# access_key_id = "test"
+# secret_access_key = "123456"
+# endpoint = "https://s3.amazonaws.com"
+# region = "us-west-2"
+
+# Example of using Oss as the storage.
+# [storage]
+# type = "Oss"
+# bucket = "greptimedb"
+# root = "data"
+# access_key_id = "test"
+# access_key_secret = "123456"
+# endpoint = "https://oss-cn-hangzhou.aliyuncs.com"
+
+# Example of using Azblob as the storage.
+# [storage]
+# type = "Azblob"
+# container = "greptimedb"
+# root = "data"
+# account_name = "test"
+# account_key = "123456"
+# endpoint = "https://greptimedb.blob.core.windows.net"
+# sas_token = ""
+
+# Example of using Gcs as the storage.
+# [storage]
+# type = "Gcs"
+# bucket = "greptimedb"
+# root = "data"
+# scope = "test"
+# credential_path = "123456"
+# endpoint = "https://storage.googleapis.com"
+
+## The data storage options.
 [storage]
-# The working home directory.
+## The working home directory.
 data_home = "/tmp/greptimedb/"
-# Storage type.
-type = "File"
-# TTL for all tables. Disabled by default.
-# global_ttl = "7d"

-# Cache configuration for object storage such as 'S3' etc.
-# The local file cache directory
-# cache_path = "/path/local_cache"
-# The local file cache capacity in bytes.
-# cache_capacity = "256MB"
+## The storage type used to store the data.
+## - `File`: the data is stored in the local file system.
+## - `S3`: the data is stored in the S3 object storage.
+## - `Gcs`: the data is stored in the Google Cloud Storage.
+## - `Azblob`: the data is stored in the Azure Blob Storage.
+## - `Oss`: the data is stored in the Aliyun OSS.
+type = "File"
+
+## Cache configuration for object storage such as 'S3' etc.
+## The local file cache directory.
+## +toml2docs:none-default
+cache_path = "/path/local_cache"
+
+## The local file cache capacity in bytes.
+## +toml2docs:none-default
+cache_capacity = "256MB"
+
+## The S3 bucket name.
+## **It's only used when the storage type is `S3`, `Oss` and `Gcs`**.
+## +toml2docs:none-default
+bucket = "greptimedb"
+
+## The S3 data will be stored in the specified prefix, for example, `s3://${bucket}/${root}`.
+## **It's only used when the storage type is `S3`, `Oss` and `Azblob`**.
+## +toml2docs:none-default
+root = "greptimedb"
+
+## The access key id of the aws account.
+## It's **highly recommended** to use AWS IAM roles instead of hardcoding the access key id and secret key.
+## **It's only used when the storage type is `S3` and `Oss`**.
+## +toml2docs:none-default
+access_key_id = "test"
+
+## The secret access key of the aws account.
+## It's **highly recommended** to use AWS IAM roles instead of hardcoding the access key id and secret key.
+## **It's only used when the storage type is `S3`**.
+## +toml2docs:none-default
+secret_access_key = "test"
+
+## The secret access key of the aliyun account.
+## **It's only used when the storage type is `Oss`**.
+## +toml2docs:none-default
+access_key_secret = "test"
+
+## The account key of the azure account.
+## **It's only used when the storage type is `Azblob`**.
+## +toml2docs:none-default
+account_name = "test"
+
+## The account key of the azure account.
+## **It's only used when the storage type is `Azblob`**.
+## +toml2docs:none-default
+account_key = "test"
+
+## The scope of the google cloud storage.
+## **It's only used when the storage type is `Gcs`**.
+## +toml2docs:none-default
+scope = "test"
+
+## The credential path of the google cloud storage.
+## **It's only used when the storage type is `Gcs`**.
+## +toml2docs:none-default
+credential_path = "test"
+
+## The container of the azure account.
+## **It's only used when the storage type is `Azblob`**.
+## +toml2docs:none-default
+container = "greptimedb"
+
+## The sas token of the azure account.
+## **It's only used when the storage type is `Azblob`**.
+## +toml2docs:none-default
+sas_token = ""
+
+## The endpoint of the S3 service.
+## **It's only used when the storage type is `S3`, `Oss`, `Gcs` and `Azblob`**.
+## +toml2docs:none-default
+endpoint = "https://s3.amazonaws.com"
+
+## The region of the S3 service.
+## **It's only used when the storage type is `S3`, `Oss`, `Gcs` and `Azblob`**.
+## +toml2docs:none-default
+region = "us-west-2"

 # Custom storage options
-#[[storage.providers]]
-#type = "S3"
-#[[storage.providers]]
-#type = "Gcs"
+# [[storage.providers]]
+# type = "S3"
+# [[storage.providers]]
+# type = "Gcs"

-# Mito engine options
+## The region engine options. You can configure multiple region engines.
 [[region_engine]]
+
+## The Mito engine options.
 [region_engine.mito]
-# Number of region workers
+
+## Number of region workers.
 num_workers = 8
-# Request channel size of each worker
+
+## Request channel size of each worker.
 worker_channel_size = 128
-# Max batch size for a worker to handle requests
+
+## Max batch size for a worker to handle requests.
 worker_request_batch_size = 64
-# Number of meta action updated to trigger a new checkpoint for the manifest
+
+## Number of meta action updated to trigger a new checkpoint for the manifest.
 manifest_checkpoint_distance = 10
-# Whether to compress manifest and checkpoint file by gzip (default false).
+
+## Whether to compress manifest and checkpoint file by gzip (default false).
 compress_manifest = false
-# Max number of running background jobs
+
+## Max number of running background jobs
 max_background_jobs = 4
-# Interval to auto flush a region if it has not flushed yet.
+
+## Interval to auto flush a region if it has not flushed yet.
 auto_flush_interval = "1h"
-# Global write buffer size for all regions. If not set, it's default to 1/8 of OS memory with a max limitation of 1GB.
+
+## Global write buffer size for all regions. If not set, it's default to 1/8 of OS memory with a max limitation of 1GB.
 global_write_buffer_size = "1GB"
-# Global write buffer size threshold to reject write requests. If not set, it's default to 2 times of `global_write_buffer_size`
+
+## Global write buffer size threshold to reject write requests. If not set, it's default to 2 times of `global_write_buffer_size`
 global_write_buffer_reject_size = "2GB"
-# Cache size for SST metadata. Setting it to 0 to disable the cache.
-# If not set, it's default to 1/32 of OS memory with a max limitation of 128MB.
+
+## Cache size for SST metadata. Setting it to 0 to disable the cache.
+## If not set, it's default to 1/32 of OS memory with a max limitation of 128MB.
 sst_meta_cache_size = "128MB"
-# Cache size for vectors and arrow arrays. Setting it to 0 to disable the cache.
-# If not set, it's default to 1/16 of OS memory with a max limitation of 512MB.
+
+## Cache size for vectors and arrow arrays. Setting it to 0 to disable the cache.
+## If not set, it's default to 1/16 of OS memory with a max limitation of 512MB.
 vector_cache_size = "512MB"
-# Cache size for pages of SST row groups. Setting it to 0 to disable the cache.
-# If not set, it's default to 1/16 of OS memory with a max limitation of 512MB.
+
+## Cache size for pages of SST row groups. Setting it to 0 to disable the cache.
+## If not set, it's default to 1/16 of OS memory with a max limitation of 512MB.
 page_cache_size = "512MB"
-# Buffer size for SST writing.
+
+## Buffer size for SST writing.
 sst_write_buffer_size = "8MB"
-# Parallelism to scan a region (default: 1/4 of cpu cores).
-# - 0: using the default value (1/4 of cpu cores).
-# - 1: scan in current thread.
-# - n: scan in parallelism n.
+
+## Parallelism to scan a region (default: 1/4 of cpu cores).
+## - `0`: using the default value (1/4 of cpu cores).
+## - `1`: scan in current thread.
+## - `n`: scan in parallelism n.
 scan_parallelism = 0
-# Capacity of the channel to send data from parallel scan tasks to the main task (default 32).
+
+## Capacity of the channel to send data from parallel scan tasks to the main task.
 parallel_scan_channel_size = 32
-# Whether to allow stale WAL entries read during replay.
+
+## Whether to allow stale WAL entries read during replay.
 allow_stale_entries = false

+## The options for inverted index in Mito engine.
 [region_engine.mito.inverted_index]
-# Whether to create the index on flush.
-# - "auto": automatically
-# - "disable": never
+
+## Whether to create the index on flush.
+## - `auto`: automatically
+## - `disable`: never
 create_on_flush = "auto"
-# Whether to create the index on compaction.
-# - "auto": automatically
-# - "disable": never
+
+## Whether to create the index on compaction.
+## - `auto`: automatically
+## - `disable`: never
 create_on_compaction = "auto"
-# Whether to apply the index on query
-# - "auto": automatically
-# - "disable": never
+
+## Whether to apply the index on query
+## - `auto`: automatically
+## - `disable`: never
 apply_on_query = "auto"
-# Memory threshold for performing an external sort during index creation.
-# Setting to empty will disable external sorting, forcing all sorting operations to happen in memory.
+
+## Memory threshold for performing an external sort during index creation.
+## Setting to empty will disable external sorting, forcing all sorting operations to happen in memory.
 mem_threshold_on_create = "64M"
-# File system path to store intermediate files for external sorting (default `{data_home}/index_intermediate`).
+
+## File system path to store intermediate files for external sorting (default `{data_home}/index_intermediate`).
 intermediate_path = ""

 [region_engine.mito.memtable]
-# Memtable type.
-# - "partition_tree": partition tree memtable
-# - "time_series": time-series memtable (deprecated)
-type = "partition_tree"
-# The max number of keys in one shard.
+## Memtable type.
+## - `time_series`: time-series memtable
+## - `partition_tree`: partition tree memtable (experimental)
+type = "time_series"
+
+## The max number of keys in one shard.
+## Only available for `partition_tree` memtable.
 index_max_keys_per_shard = 8192
-# The max rows of data inside the actively writing buffer in one shard.
+
+## The max rows of data inside the actively writing buffer in one shard.
+## Only available for `partition_tree` memtable.
 data_freeze_threshold = 32768
-# Max dictionary bytes.
+
+## Max dictionary bytes.
+## Only available for `partition_tree` memtable.
 fork_dictionary_bytes = "1GiB"

-# Log options, see `standalone.example.toml`
-# [logging]
-# dir = "/tmp/greptimedb/logs"
-# level = "info"
+## The logging options.
+[logging]
+## The directory to store the log files.
+dir = "/tmp/greptimedb/logs"

-# Datanode export the metrics generated by itself
-# encoded to Prometheus remote-write format
-# and send to Prometheus remote-write compatible receiver (e.g. send to `greptimedb` itself)
-# This is only used for `greptimedb` to export its own metrics internally. It's different from prometheus scrape.
-# [export_metrics]
-# whether enable export metrics, default is false
-# enable = false
-# The interval of export metrics
-# write_interval = "30s"
-# [export_metrics.remote_write]
-# The url the metrics send to. The url is empty by default, url example: `http://127.0.0.1:4000/v1/prometheus/write?db=information_schema`
-# url = ""
-# HTTP headers of Prometheus remote-write carry
-# headers = {}
+## The log level. Can be `info`/`debug`/`warn`/`error`.
+## +toml2docs:none-default
+level = "info"
+
+## Enable OTLP tracing.
+enable_otlp_tracing = false
+
+## The OTLP tracing endpoint.
+## +toml2docs:none-default
+otlp_endpoint = ""
+
+## Whether to append logs to stdout.
+append_stdout = true
+
+## The percentage of tracing will be sampled and exported.
+## Valid range `[0, 1]`, 1 means all traces are sampled, 0 means all traces are not sampled, the default value is 1.
+## ratio > 1 are treated as 1. Fractions < 0 are treated as 0
+[logging.tracing_sample_ratio]
+default_ratio = 1.0
+
+## The datanode can export its metrics and send to Prometheus compatible service (e.g. send to `greptimedb` itself) from remote-write API.
+## This is only used for `greptimedb` to export its own metrics internally. It's different from prometheus scrape.
+[export_metrics]
+
+## whether enable export metrics.
+enable = false
+
+## The interval of export metrics.
+write_interval = "30s"
+
+## For `standalone` mode, `self_import` is recommend to collect metrics generated by itself
+[export_metrics.self_import]
+## +toml2docs:none-default
+db = "information_schema"
+
+[export_metrics.remote_write]
+## The url the metrics send to. The url example can be: `http://127.0.0.1:4000/v1/prometheus/write?db=information_schema`.
+url = ""
+
+## HTTP headers of Prometheus remote-write carry.
+headers = { }
--- a/config/frontend.example.toml
+++ b/config/frontend.example.toml
@@ -1,106 +1,192 @@
-# Node running mode, see `standalone.example.toml`.
-mode = "distributed"
-# The default timezone of the server
-# default_timezone = "UTC"
+## The running mode of the datanode. It can be `standalone` or `distributed`.
+mode = "standalone"

+## The default timezone of the server.
+## +toml2docs:none-default
+default_timezone = "UTC"
+
+## The heartbeat options.
 [heartbeat]
-# Interval for sending heartbeat task to the Metasrv, 5 seconds by default.
-interval = "5s"
-# Interval for retry sending heartbeat task, 5 seconds by default.
-retry_interval = "5s"
+## Interval for sending heartbeat messages to the metasrv.
+interval = "18s"

-# HTTP server options, see `standalone.example.toml`.
+## Interval for retrying to send heartbeat messages to the metasrv.
+retry_interval = "3s"
+
+## The HTTP server options.
 [http]
+## The address to bind the HTTP server.
 addr = "127.0.0.1:4000"
+## HTTP request timeout.
 timeout = "30s"
+## HTTP request body limit.
+## Support the following units are supported: `B`, `KB`, `KiB`, `MB`, `MiB`, `GB`, `GiB`, `TB`, `TiB`, `PB`, `PiB`.
 body_limit = "64MB"

-# gRPC server options, see `standalone.example.toml`.
+## The gRPC server options.
 [grpc]
+## The address to bind the gRPC server.
 addr = "127.0.0.1:4001"
+## The number of server worker threads.
 runtime_size = 8

-# MySQL server options, see `standalone.example.toml`.
+## MySQL server options.
 [mysql]
+## Whether to enable.
 enable = true
+## The addr to bind the MySQL server.
 addr = "127.0.0.1:4002"
+## The number of server worker threads.
 runtime_size = 2

-# MySQL server TLS options, see `standalone.example.toml`.
+# MySQL server TLS options.
 [mysql.tls]
+
+## TLS mode, refer to https://www.postgresql.org/docs/current/libpq-ssl.html
+## - `disable` (default value)
+## - `prefer`
+## - `require`
+## - `verify-ca`
+## - `verify-full`
 mode = "disable"
+
+## Certificate file path.
+## +toml2docs:none-default
 cert_path = ""
+
+## Private key file path.
+## +toml2docs:none-default
 key_path = ""
+
+## Watch for Certificate and key file change and auto reload
 watch = false

-# PostgresSQL server options, see `standalone.example.toml`.
+## PostgresSQL server options.
 [postgres]
+## Whether to enable
 enable = true
+## The addr to bind the PostgresSQL server.
 addr = "127.0.0.1:4003"
+## The number of server worker threads.
 runtime_size = 2

-# PostgresSQL server TLS options, see `standalone.example.toml`.
+## PostgresSQL server TLS options, see `mysql_options.tls` section.
 [postgres.tls]
+## TLS mode.
 mode = "disable"
+
+## Certificate file path.
+## +toml2docs:none-default
 cert_path = ""
+
+## Private key file path.
+## +toml2docs:none-default
 key_path = ""
+
+## Watch for Certificate and key file change and auto reload
 watch = false

-# OpenTSDB protocol options, see `standalone.example.toml`.
+## OpenTSDB protocol options.
 [opentsdb]
+## Whether to enable
 enable = true
+## OpenTSDB telnet API server address.
 addr = "127.0.0.1:4242"
+## The number of server worker threads.
 runtime_size = 2

-# InfluxDB protocol options, see `standalone.example.toml`.
+## InfluxDB protocol options.
 [influxdb]
+## Whether to enable InfluxDB protocol in HTTP API.
 enable = true

-# Prometheus remote storage options, see `standalone.example.toml`.
+## Prometheus remote storage options
 [prom_store]
+## Whether to enable Prometheus remote write and read in HTTP API.
 enable = true
-# Whether to store the data from Prometheus remote write in metric engine.
-# true by default
+## Whether to store the data from Prometheus remote write in metric engine.
 with_metric_engine = true

-# Metasrv client options, see `datanode.example.toml`.
+## The metasrv client options.
 [meta_client]
+## The addresses of the metasrv.
 metasrv_addrs = ["127.0.0.1:3002"]
+
+## Operation timeout.
 timeout = "3s"
-# DDL timeouts options.
+
+## Heartbeat timeout.
+heartbeat_timeout = "500ms"
+
+## DDL timeout.
 ddl_timeout = "10s"
+
+## Connect server timeout.
 connect_timeout = "1s"
+
+## `TCP_NODELAY` option for accepted connections.
 tcp_nodelay = true
-# The configuration about the cache of the Metadata.
-# default: 100000
+
+## The configuration about the cache of the metadata.
 metadata_cache_max_capacity = 100000
-# default: 10m
+
+## TTL of the metadata cache.
 metadata_cache_ttl = "10m"
-# default: 5m
+
+# TTI of the metadata cache.
 metadata_cache_tti = "5m"

-# Log options, see `standalone.example.toml`
-# [logging]
-# dir = "/tmp/greptimedb/logs"
-# level = "info"
-
-# Datanode options.
+## Datanode options.
 [datanode]
-# Datanode client options.
+## Datanode client options.
 [datanode.client]
 timeout = "10s"
 connect_timeout = "10s"
 tcp_nodelay = true

-# Frontend export the metrics generated by itself
-# encoded to Prometheus remote-write format
-# and send to Prometheus remote-write compatible receiver (e.g. send to `greptimedb` itself)
-# This is only used for `greptimedb` to export its own metrics internally. It's different from prometheus scrape.
-# [export_metrics]
-# whether enable export metrics, default is false
-# enable = false
-# The interval of export metrics
-# write_interval = "30s"
-# for `frontend`, `self_import` is recommend to collect metrics generated by itself
-# [export_metrics.self_import]
-# db = "information_schema"
+## The logging options.
+[logging]
+## The directory to store the log files.
+dir = "/tmp/greptimedb/logs"
+
+## The log level. Can be `info`/`debug`/`warn`/`error`.
+## +toml2docs:none-default
+level = "info"
+
+## Enable OTLP tracing.
+enable_otlp_tracing = false
+
+## The OTLP tracing endpoint.
+## +toml2docs:none-default
+otlp_endpoint = ""
+
+## Whether to append logs to stdout.
+append_stdout = true
+
+## The percentage of tracing will be sampled and exported.
+## Valid range `[0, 1]`, 1 means all traces are sampled, 0 means all traces are not sampled, the default value is 1.
+## ratio > 1 are treated as 1. Fractions < 0 are treated as 0
+[logging.tracing_sample_ratio]
+default_ratio = 1.0
+
+## The datanode can export its metrics and send to Prometheus compatible service (e.g. send to `greptimedb` itself) from remote-write API.
+## This is only used for `greptimedb` to export its own metrics internally. It's different from prometheus scrape.
+[export_metrics]
+
+## whether enable export metrics.
+enable = false
+
+## The interval of export metrics.
+write_interval = "30s"
+
+## For `standalone` mode, `self_import` is recommend to collect metrics generated by itself
+[export_metrics.self_import]
+## +toml2docs:none-default
+db = "information_schema"
+
+[export_metrics.remote_write]
+## The url the metrics send to. The url example can be: `http://127.0.0.1:4000/v1/prometheus/write?db=information_schema`.
+url = ""
+
+## HTTP headers of Prometheus remote-write carry.
+headers = { }
--- a/config/metasrv.example.toml
+++ b/config/metasrv.example.toml
@@ -1,35 +1,46 @@
-# The working home directory.
+## The working home directory.
 data_home = "/tmp/metasrv/"
-# The bind address of metasrv, "127.0.0.1:3002" by default.
+
+## The bind address of metasrv.
 bind_addr = "127.0.0.1:3002"
-# The communication server address for frontend and datanode to connect to metasrv,  "127.0.0.1:3002" by default for localhost.
+
+## The communication server address for frontend and datanode to connect to metasrv,  "127.0.0.1:3002" by default for localhost.
 server_addr = "127.0.0.1:3002"
-# Etcd server address, "127.0.0.1:2379" by default.
+
+## Etcd server address.
 store_addr = "127.0.0.1:2379"
-# Datanode selector type.
-# - "lease_based" (default value).
-# - "load_based"
-# For details, please see "https://docs.greptime.com/developer-guide/metasrv/selector".
+
+## Datanode selector type.
+## - `lease_based` (default value).
+## - `load_based`
+## For details, please see "https://docs.greptime.com/developer-guide/metasrv/selector".
 selector = "lease_based"
-# Store data in memory, false by default.
+
+## Store data in memory.
 use_memory_store = false
-# Whether to enable greptimedb telemetry, true by default.
+
+## Whether to enable greptimedb telemetry.
 enable_telemetry = true
-# If it's not empty, the metasrv will store all data with this key prefix.
+
+## If it's not empty, the metasrv will store all data with this key prefix.
 store_key_prefix = ""

-# Log options, see `standalone.example.toml`
-# [logging]
-# dir = "/tmp/greptimedb/logs"
-# level = "info"
-
-# Procedure storage options.
+## Procedure storage options.
 [procedure]
-# Procedure max retry time.
+
+## Procedure max retry time.
 max_retry_times = 12
-# Initial retry delay of procedures, increases exponentially
+
+## Initial retry delay of procedures, increases exponentially
 retry_delay = "500ms"

+## Auto split large value
+## GreptimeDB procedure uses etcd as the default metadata storage backend.
+## The etcd the maximum size of any request is 1.5 MiB
+## 1500KiB = 1536KiB (1.5MiB) - 36KiB (reserved size of key)
+## Comments out the `max_metadata_value_size`, for don't split large value (no limit).
+max_metadata_value_size = "1500KiB"
+
 # Failure detectors options.
 [failure_detector]
 threshold = 8.0
@@ -37,57 +48,96 @@ min_std_deviation = "100ms"
 acceptable_heartbeat_pause = "3000ms"
 first_heartbeat_estimate = "1000ms"

-# # Datanode options.
-# [datanode]
-# # Datanode client options.
-# [datanode.client_options]
-# timeout = "10s"
-# connect_timeout = "10s"
-# tcp_nodelay = true
+## Datanode options.
+[datanode]
+## Datanode client options.
+[datanode.client]
+timeout = "10s"
+connect_timeout = "10s"
+tcp_nodelay = true

 [wal]
 # Available wal providers:
-# - "raft_engine" (default)
-# - "kafka"
+# - `raft_engine` (default): there're none raft-engine wal config since metasrv only involves in remote wal currently.
+# - `kafka`: metasrv **have to be** configured with kafka wal config when using kafka wal provider in datanode.
 provider = "raft_engine"

-# There're none raft-engine wal config since meta srv only involves in remote wal currently.
-
 # Kafka wal config.
-# The broker endpoints of the Kafka cluster. ["127.0.0.1:9092"] by default.
-# broker_endpoints = ["127.0.0.1:9092"]
-# Number of topics to be created upon start.
-# num_topics = 64
-# Topic selector type.
-# Available selector types: 
-# - "round_robin" (default)
-# selector_type = "round_robin"
-# A Kafka topic is constructed by concatenating `topic_name_prefix` and `topic_id`.
-# topic_name_prefix = "greptimedb_wal_topic"
-# Expected number of replicas of each partition.
-# replication_factor = 1
-# Above which a topic creation operation will be cancelled.
-# create_topic_timeout = "30s"
-# The initial backoff for kafka clients.
-# backoff_init = "500ms"
-# The maximum backoff for kafka clients.
-# backoff_max = "10s"
-# Exponential backoff rate, i.e. next backoff = base * current backoff.
-# backoff_base = 2
-# Stop reconnecting if the total wait time reaches the deadline. If this config is missing, the reconnecting won't terminate.
-# backoff_deadline = "5mins"

-# Metasrv export the metrics generated by itself
-# encoded to Prometheus remote-write format
-# and send to Prometheus remote-write compatible receiver (e.g. send to `greptimedb` itself)
-# This is only used for `greptimedb` to export its own metrics internally. It's different from prometheus scrape.
-# [export_metrics]
-# whether enable export metrics, default is false
-# enable = false
-# The interval of export metrics
-# write_interval = "30s"
-# [export_metrics.remote_write]
-# The url the metrics send to. The url is empty by default, url example: `http://127.0.0.1:4000/v1/prometheus/write?db=information_schema`
-# url = ""
-# HTTP headers of Prometheus remote-write carry
-# headers = {}
+## The broker endpoints of the Kafka cluster.
+broker_endpoints = ["127.0.0.1:9092"]
+
+## Number of topics to be created upon start.
+num_topics = 64
+
+## Topic selector type.
+## Available selector types:
+## - `round_robin` (default)
+selector_type = "round_robin"
+
+## A Kafka topic is constructed by concatenating `topic_name_prefix` and `topic_id`.
+topic_name_prefix = "greptimedb_wal_topic"
+
+## Expected number of replicas of each partition.
+replication_factor = 1
+
+## Above which a topic creation operation will be cancelled.
+create_topic_timeout = "30s"
+## The initial backoff for kafka clients.
+backoff_init = "500ms"
+
+## The maximum backoff for kafka clients.
+backoff_max = "10s"
+
+## Exponential backoff rate, i.e. next backoff = base * current backoff.
+backoff_base = 2
+
+## Stop reconnecting if the total wait time reaches the deadline. If this config is missing, the reconnecting won't terminate.
+backoff_deadline = "5mins"
+
+## The logging options.
+[logging]
+## The directory to store the log files.
+dir = "/tmp/greptimedb/logs"
+
+## The log level. Can be `info`/`debug`/`warn`/`error`.
+## +toml2docs:none-default
+level = "info"
+
+## Enable OTLP tracing.
+enable_otlp_tracing = false
+
+## The OTLP tracing endpoint.
+## +toml2docs:none-default
+otlp_endpoint = ""
+
+## Whether to append logs to stdout.
+append_stdout = true
+
+## The percentage of tracing will be sampled and exported.
+## Valid range `[0, 1]`, 1 means all traces are sampled, 0 means all traces are not sampled, the default value is 1.
+## ratio > 1 are treated as 1. Fractions < 0 are treated as 0
+[logging.tracing_sample_ratio]
+default_ratio = 1.0
+
+## The datanode can export its metrics and send to Prometheus compatible service (e.g. send to `greptimedb` itself) from remote-write API.
+## This is only used for `greptimedb` to export its own metrics internally. It's different from prometheus scrape.
+[export_metrics]
+
+## whether enable export metrics.
+enable = false
+
+## The interval of export metrics.
+write_interval = "30s"
+
+## For `standalone` mode, `self_import` is recommend to collect metrics generated by itself
+[export_metrics.self_import]
+## +toml2docs:none-default
+db = "information_schema"
+
+[export_metrics.remote_write]
+## The url the metrics send to. The url example can be: `http://127.0.0.1:4000/v1/prometheus/write?db=information_schema`.
+url = ""
+
+## HTTP headers of Prometheus remote-write carry.
+headers = { }
--- a/config/standalone.example.toml
+++ b/config/standalone.example.toml
@@ -1,286 +1,477 @@
-# Node running mode, "standalone" or "distributed".
+## The running mode of the datanode. It can be `standalone` or `distributed`.
 mode = "standalone"
-# Whether to enable greptimedb telemetry, true by default.
-enable_telemetry = true
-# The default timezone of the server
-# default_timezone = "UTC"

-# HTTP server options.
+## Enable telemetry to collect anonymous usage data.
+enable_telemetry = true
+
+## The default timezone of the server.
+## +toml2docs:none-default
+default_timezone = "UTC"
+
+## The HTTP server options.
 [http]
-# Server address, "127.0.0.1:4000" by default.
+## The address to bind the HTTP server.
 addr = "127.0.0.1:4000"
-# HTTP request timeout, 30s by default.
+## HTTP request timeout.
 timeout = "30s"
-# HTTP request body limit, 64Mb by default.
-# the following units are supported: B, KB, KiB, MB, MiB, GB, GiB, TB, TiB, PB, PiB
+## HTTP request body limit.
+## Support the following units are supported: `B`, `KB`, `KiB`, `MB`, `MiB`, `GB`, `GiB`, `TB`, `TiB`, `PB`, `PiB`.
 body_limit = "64MB"

-# gRPC server options.
+## The gRPC server options.
 [grpc]
-# Server address, "127.0.0.1:4001" by default.
+## The address to bind the gRPC server.
 addr = "127.0.0.1:4001"
-# The number of server worker threads, 8 by default.
+## The number of server worker threads.
 runtime_size = 8

-# MySQL server options.
+## MySQL server options.
 [mysql]
-# Whether to enable
+## Whether to enable.
 enable = true
-# Server address, "127.0.0.1:4002" by default.
+## The addr to bind the MySQL server.
 addr = "127.0.0.1:4002"
-# The number of server worker threads, 2 by default.
+## The number of server worker threads.
 runtime_size = 2

 # MySQL server TLS options.
 [mysql.tls]
-# TLS mode, refer to https://www.postgresql.org/docs/current/libpq-ssl.html
-# - "disable" (default value)
-# - "prefer"
-# - "require"
-# - "verify-ca"
-# - "verify-full"
+
+## TLS mode, refer to https://www.postgresql.org/docs/current/libpq-ssl.html
+## - `disable` (default value)
+## - `prefer`
+## - `require`
+## - `verify-ca`
+## - `verify-full`
 mode = "disable"
-# Certificate file path.
+
+## Certificate file path.
+## +toml2docs:none-default
 cert_path = ""
-# Private key file path.
+
+## Private key file path.
+## +toml2docs:none-default
 key_path = ""
-# Watch for Certificate and key file change and auto reload
+
+## Watch for Certificate and key file change and auto reload
 watch = false

-# PostgresSQL server options.
+## PostgresSQL server options.
 [postgres]
-# Whether to enable
+## Whether to enable
 enable = true
-# Server address, "127.0.0.1:4003" by default.
+## The addr to bind the PostgresSQL server.
 addr = "127.0.0.1:4003"
-# The number of server worker threads, 2 by default.
+## The number of server worker threads.
 runtime_size = 2

-# PostgresSQL server TLS options, see `[mysql_options.tls]` section.
+## PostgresSQL server TLS options, see `mysql_options.tls` section.
 [postgres.tls]
-# TLS mode.
+## TLS mode.
 mode = "disable"
-# certificate file path.
+
+## Certificate file path.
+## +toml2docs:none-default
 cert_path = ""
-# private key file path.
+
+## Private key file path.
+## +toml2docs:none-default
 key_path = ""
-# Watch for Certificate and key file change and auto reload
+
+## Watch for Certificate and key file change and auto reload
 watch = false

-# OpenTSDB protocol options.
+## OpenTSDB protocol options.
 [opentsdb]
-# Whether to enable
+## Whether to enable
 enable = true
-# OpenTSDB telnet API server address, "127.0.0.1:4242" by default.
+## OpenTSDB telnet API server address.
 addr = "127.0.0.1:4242"
-# The number of server worker threads, 2 by default.
+## The number of server worker threads.
 runtime_size = 2

-# InfluxDB protocol options.
+## InfluxDB protocol options.
 [influxdb]
-# Whether to enable InfluxDB protocol in HTTP API, true by default.
+## Whether to enable InfluxDB protocol in HTTP API.
 enable = true

-# Prometheus remote storage options
+## Prometheus remote storage options
 [prom_store]
-# Whether to enable Prometheus remote write and read in HTTP API, true by default.
+## Whether to enable Prometheus remote write and read in HTTP API.
 enable = true
-# Whether to store the data from Prometheus remote write in metric engine.
-# true by default
+## Whether to store the data from Prometheus remote write in metric engine.
 with_metric_engine = true

+## The WAL options.
 [wal]
-# Available wal providers:
-# - "raft_engine" (default)
-# - "kafka"
+## The provider of the WAL.
+## - `raft_engine`: the wal is stored in the local file system by raft-engine.
+## - `kafka`: it's remote wal that data is stored in Kafka.
 provider = "raft_engine"

-# Raft-engine wal options.
-# WAL data directory
-# dir = "/tmp/greptimedb/wal"
-# WAL file size in bytes.
+## The directory to store the WAL files.
+## **It's only used when the provider is `raft_engine`**.
+## +toml2docs:none-default
+dir = "/tmp/greptimedb/wal"
+
+## The size of the WAL segment file.
+## **It's only used when the provider is `raft_engine`**.
 file_size = "256MB"
-# WAL purge threshold.
+
+## The threshold of the WAL size to trigger a flush.
+## **It's only used when the provider is `raft_engine`**.
 purge_threshold = "4GB"
-# WAL purge interval in seconds.
+
+## The interval to trigger a flush.
+## **It's only used when the provider is `raft_engine`**.
 purge_interval = "10m"
-# WAL read batch size.
+
+## The read batch size.
+## **It's only used when the provider is `raft_engine`**.
 read_batch_size = 128
-# Whether to sync log file after every write.
+
+## Whether to use sync write.
+## **It's only used when the provider is `raft_engine`**.
 sync_write = false
-# Whether to reuse logically truncated log files.
+
+## Whether to reuse logically truncated log files.
+## **It's only used when the provider is `raft_engine`**.
 enable_log_recycle = true
-# Whether to pre-create log files on start up
+
+## Whether to pre-create log files on start up.
+## **It's only used when the provider is `raft_engine`**.
 prefill_log_files = false
-# Duration for fsyncing log files.
-sync_period = "1000ms"

-# Kafka wal options.
-# The broker endpoints of the Kafka cluster. ["127.0.0.1:9092"] by default.
-# broker_endpoints = ["127.0.0.1:9092"]
+## Duration for fsyncing log files.
+## **It's only used when the provider is `raft_engine`**.
+sync_period = "10s"

-# Number of topics to be created upon start.
-# num_topics = 64
-# Topic selector type.
-# Available selector types:
-# - "round_robin" (default)
-# selector_type = "round_robin"
-# The prefix of topic name.
-# topic_name_prefix = "greptimedb_wal_topic"
-# The number of replicas of each partition.
-# Warning: the replication factor must be positive and must not be greater than the number of broker endpoints.
-# replication_factor = 1
+## The Kafka broker endpoints.
+## **It's only used when the provider is `kafka`**.
+broker_endpoints = ["127.0.0.1:9092"]

-# The max size of a single producer batch.
-# Warning: Kafka has a default limit of 1MB per message in a topic.
-# max_batch_size = "1MB"
-# The linger duration.
-# linger = "200ms"
-# The consumer wait timeout.
-# consumer_wait_timeout = "100ms"
-# Create topic timeout.
-# create_topic_timeout = "30s"
+## The max size of a single producer batch.
+## Warning: Kafka has a default limit of 1MB per message in a topic.
+## **It's only used when the provider is `kafka`**.
+max_batch_size = "1MB"

-# The initial backoff delay.
-# backoff_init = "500ms"
-# The maximum backoff delay.
-# backoff_max = "10s"
-# Exponential backoff rate, i.e. next backoff = base * current backoff.
-# backoff_base = 2
-# The deadline of retries.
-# backoff_deadline = "5mins"
+## The linger duration of a kafka batch producer.
+## **It's only used when the provider is `kafka`**.
+linger = "200ms"

-# Metadata storage options.
+## The consumer wait timeout.
+## **It's only used when the provider is `kafka`**.
+consumer_wait_timeout = "100ms"
+
+## The initial backoff delay.
+## **It's only used when the provider is `kafka`**.
+backoff_init = "500ms"
+
+## The maximum backoff delay.
+## **It's only used when the provider is `kafka`**.
+backoff_max = "10s"
+
+## The exponential backoff rate, i.e. next backoff = base * current backoff.
+## **It's only used when the provider is `kafka`**.
+backoff_base = 2
+
+## The deadline of retries.
+## **It's only used when the provider is `kafka`**.
+backoff_deadline = "5mins"
+
+## Metadata storage options.
 [metadata_store]
-# Kv file size in bytes.
+## Kv file size in bytes.
 file_size = "256MB"
-# Kv purge threshold.
+## Kv purge threshold.
 purge_threshold = "4GB"

-# Procedure storage options.
+## Procedure storage options.
 [procedure]
-# Procedure max retry time.
+## Procedure max retry time.
 max_retry_times = 3
-# Initial retry delay of procedures, increases exponentially
+## Initial retry delay of procedures, increases exponentially
 retry_delay = "500ms"

-# Storage options.
+# Example of using S3 as the storage.
+# [storage]
+# type = "S3"
+# bucket = "greptimedb"
+# root = "data"
+# access_key_id = "test"
+# secret_access_key = "123456"
+# endpoint = "https://s3.amazonaws.com"
+# region = "us-west-2"
+
+# Example of using Oss as the storage.
+# [storage]
+# type = "Oss"
+# bucket = "greptimedb"
+# root = "data"
+# access_key_id = "test"
+# access_key_secret = "123456"
+# endpoint = "https://oss-cn-hangzhou.aliyuncs.com"
+
+# Example of using Azblob as the storage.
+# [storage]
+# type = "Azblob"
+# container = "greptimedb"
+# root = "data"
+# account_name = "test"
+# account_key = "123456"
+# endpoint = "https://greptimedb.blob.core.windows.net"
+# sas_token = ""
+
+# Example of using Gcs as the storage.
+# [storage]
+# type = "Gcs"
+# bucket = "greptimedb"
+# root = "data"
+# scope = "test"
+# credential_path = "123456"
+# endpoint = "https://storage.googleapis.com"
+
+## The data storage options.
 [storage]
-# The working home directory.
+## The working home directory.
 data_home = "/tmp/greptimedb/"
-# Storage type.
+
+## The storage type used to store the data.
+## - `File`: the data is stored in the local file system.
+## - `S3`: the data is stored in the S3 object storage.
+## - `Gcs`: the data is stored in the Google Cloud Storage.
+## - `Azblob`: the data is stored in the Azure Blob Storage.
+## - `Oss`: the data is stored in the Aliyun OSS.
 type = "File"
-# TTL for all tables. Disabled by default.
-# global_ttl = "7d"
-# Cache configuration for object storage such as 'S3' etc.
-# cache_path = "/path/local_cache"
-# The local file cache capacity in bytes.
-# cache_capacity = "256MB"
+
+## Cache configuration for object storage such as 'S3' etc.
+## The local file cache directory.
+## +toml2docs:none-default
+cache_path = "/path/local_cache"
+
+## The local file cache capacity in bytes.
+## +toml2docs:none-default
+cache_capacity = "256MB"
+
+## The S3 bucket name.
+## **It's only used when the storage type is `S3`, `Oss` and `Gcs`**.
+## +toml2docs:none-default
+bucket = "greptimedb"
+
+## The S3 data will be stored in the specified prefix, for example, `s3://${bucket}/${root}`.
+## **It's only used when the storage type is `S3`, `Oss` and `Azblob`**.
+## +toml2docs:none-default
+root = "greptimedb"
+
+## The access key id of the aws account.
+## It's **highly recommended** to use AWS IAM roles instead of hardcoding the access key id and secret key.
+## **It's only used when the storage type is `S3` and `Oss`**.
+## +toml2docs:none-default
+access_key_id = "test"
+
+## The secret access key of the aws account.
+## It's **highly recommended** to use AWS IAM roles instead of hardcoding the access key id and secret key.
+## **It's only used when the storage type is `S3`**.
+## +toml2docs:none-default
+secret_access_key = "test"
+
+## The secret access key of the aliyun account.
+## **It's only used when the storage type is `Oss`**.
+## +toml2docs:none-default
+access_key_secret = "test"
+
+## The account key of the azure account.
+## **It's only used when the storage type is `Azblob`**.
+## +toml2docs:none-default
+account_name = "test"
+
+## The account key of the azure account.
+## **It's only used when the storage type is `Azblob`**.
+## +toml2docs:none-default
+account_key = "test"
+
+## The scope of the google cloud storage.
+## **It's only used when the storage type is `Gcs`**.
+## +toml2docs:none-default
+scope = "test"
+
+## The credential path of the google cloud storage.
+## **It's only used when the storage type is `Gcs`**.
+## +toml2docs:none-default
+credential_path = "test"
+
+## The container of the azure account.
+## **It's only used when the storage type is `Azblob`**.
+## +toml2docs:none-default
+container = "greptimedb"
+
+## The sas token of the azure account.
+## **It's only used when the storage type is `Azblob`**.
+## +toml2docs:none-default
+sas_token = ""
+
+## The endpoint of the S3 service.
+## **It's only used when the storage type is `S3`, `Oss`, `Gcs` and `Azblob`**.
+## +toml2docs:none-default
+endpoint = "https://s3.amazonaws.com"
+
+## The region of the S3 service.
+## **It's only used when the storage type is `S3`, `Oss`, `Gcs` and `Azblob`**.
+## +toml2docs:none-default
+region = "us-west-2"

 # Custom storage options
-#[[storage.providers]]
-#type = "S3"
-#[[storage.providers]]
-#type = "Gcs"
+# [[storage.providers]]
+# type = "S3"
+# [[storage.providers]]
+# type = "Gcs"

-# Mito engine options
+## The region engine options. You can configure multiple region engines.
 [[region_engine]]
+
+## The Mito engine options.
 [region_engine.mito]
-# Number of region workers
+
+## Number of region workers.
 num_workers = 8
-# Request channel size of each worker
+
+## Request channel size of each worker.
 worker_channel_size = 128
-# Max batch size for a worker to handle requests
+
+## Max batch size for a worker to handle requests.
 worker_request_batch_size = 64
-# Number of meta action updated to trigger a new checkpoint for the manifest
+
+## Number of meta action updated to trigger a new checkpoint for the manifest.
 manifest_checkpoint_distance = 10
-# Whether to compress manifest and checkpoint file by gzip (default false).
+
+## Whether to compress manifest and checkpoint file by gzip (default false).
 compress_manifest = false
-# Max number of running background jobs
+
+## Max number of running background jobs
 max_background_jobs = 4
-# Interval to auto flush a region if it has not flushed yet.
+
+## Interval to auto flush a region if it has not flushed yet.
 auto_flush_interval = "1h"
-# Global write buffer size for all regions. If not set, it's default to 1/8 of OS memory with a max limitation of 1GB.
+
+## Global write buffer size for all regions. If not set, it's default to 1/8 of OS memory with a max limitation of 1GB.
 global_write_buffer_size = "1GB"
-# Global write buffer size threshold to reject write requests. If not set, it's default to 2 times of `global_write_buffer_size`
+
+## Global write buffer size threshold to reject write requests. If not set, it's default to 2 times of `global_write_buffer_size`
 global_write_buffer_reject_size = "2GB"
-# Cache size for SST metadata. Setting it to 0 to disable the cache.
-# If not set, it's default to 1/32 of OS memory with a max limitation of 128MB.
+
+## Cache size for SST metadata. Setting it to 0 to disable the cache.
+## If not set, it's default to 1/32 of OS memory with a max limitation of 128MB.
 sst_meta_cache_size = "128MB"
-# Cache size for vectors and arrow arrays. Setting it to 0 to disable the cache.
-# If not set, it's default to 1/16 of OS memory with a max limitation of 512MB.
+
+## Cache size for vectors and arrow arrays. Setting it to 0 to disable the cache.
+## If not set, it's default to 1/16 of OS memory with a max limitation of 512MB.
 vector_cache_size = "512MB"
-# Cache size for pages of SST row groups. Setting it to 0 to disable the cache.
-# If not set, it's default to 1/16 of OS memory with a max limitation of 512MB.
+
+## Cache size for pages of SST row groups. Setting it to 0 to disable the cache.
+## If not set, it's default to 1/16 of OS memory with a max limitation of 512MB.
 page_cache_size = "512MB"
-# Buffer size for SST writing.
+
+## Buffer size for SST writing.
 sst_write_buffer_size = "8MB"
-# Parallelism to scan a region (default: 1/4 of cpu cores).
-# - 0: using the default value (1/4 of cpu cores).
-# - 1: scan in current thread.
-# - n: scan in parallelism n.
+
+## Parallelism to scan a region (default: 1/4 of cpu cores).
+## - `0`: using the default value (1/4 of cpu cores).
+## - `1`: scan in current thread.
+## - `n`: scan in parallelism n.
 scan_parallelism = 0
-# Capacity of the channel to send data from parallel scan tasks to the main task (default 32).
+
+## Capacity of the channel to send data from parallel scan tasks to the main task.
 parallel_scan_channel_size = 32
-# Whether to allow stale WAL entries read during replay.
+
+## Whether to allow stale WAL entries read during replay.
 allow_stale_entries = false

+## The options for inverted index in Mito engine.
 [region_engine.mito.inverted_index]
-# Whether to create the index on flush.
-# - "auto": automatically
-# - "disable": never
+
+## Whether to create the index on flush.
+## - `auto`: automatically
+## - `disable`: never
 create_on_flush = "auto"
-# Whether to create the index on compaction.
-# - "auto": automatically
-# - "disable": never
+
+## Whether to create the index on compaction.
+## - `auto`: automatically
+## - `disable`: never
 create_on_compaction = "auto"
-# Whether to apply the index on query
-# - "auto": automatically
-# - "disable": never
+
+## Whether to apply the index on query
+## - `auto`: automatically
+## - `disable`: never
 apply_on_query = "auto"
-# Memory threshold for performing an external sort during index creation.
-# Setting to empty will disable external sorting, forcing all sorting operations to happen in memory.
+
+## Memory threshold for performing an external sort during index creation.
+## Setting to empty will disable external sorting, forcing all sorting operations to happen in memory.
 mem_threshold_on_create = "64M"
-# File system path to store intermediate files for external sorting (default `{data_home}/index_intermediate`).
+
+## File system path to store intermediate files for external sorting (default `{data_home}/index_intermediate`).
 intermediate_path = ""

 [region_engine.mito.memtable]
-# Memtable type.
-# - "partition_tree": partition tree memtable
-# - "time_series": time-series memtable (deprecated)
-type = "partition_tree"
-# The max number of keys in one shard.
+## Memtable type.
+## - `time_series`: time-series memtable
+## - `partition_tree`: partition tree memtable (experimental)
+type = "time_series"
+
+## The max number of keys in one shard.
+## Only available for `partition_tree` memtable.
 index_max_keys_per_shard = 8192
-# The max rows of data inside the actively writing buffer in one shard.
+
+## The max rows of data inside the actively writing buffer in one shard.
+## Only available for `partition_tree` memtable.
 data_freeze_threshold = 32768
-# Max dictionary bytes.
+
+## Max dictionary bytes.
+## Only available for `partition_tree` memtable.
 fork_dictionary_bytes = "1GiB"

-# Log options
-# [logging]
-# Specify logs directory.
-# dir = "/tmp/greptimedb/logs"
-# Specify the log level [info | debug | error | warn]
-# level = "info"
-# whether enable tracing, default is false
-# enable_otlp_tracing = false
-# tracing exporter endpoint with format `ip:port`, we use grpc oltp as exporter, default endpoint is `localhost:4317`
-# otlp_endpoint = "localhost:4317"
-# Whether to append logs to stdout. Defaults to true.
-# append_stdout = true
-# The percentage of tracing will be sampled and exported. Valid range `[0, 1]`, 1 means all traces are sampled, 0 means all traces are not sampled, the default value is 1. ratio > 1 are treated as 1. Fractions < 0 are treated as 0
-# [logging.tracing_sample_ratio]
-# default_ratio = 0.0
+## The logging options.
+[logging]
+## The directory to store the log files.
+dir = "/tmp/greptimedb/logs"

-# Standalone export the metrics generated by itself
-# encoded to Prometheus remote-write format
-# and send to Prometheus remote-write compatible receiver (e.g. send to `greptimedb` itself)
-# This is only used for `greptimedb` to export its own metrics internally. It's different from prometheus scrape.
-# [export_metrics]
-# whether enable export metrics, default is false
-# enable = false
-# The interval of export metrics
-# write_interval = "30s"
-# for `standalone`, `self_import` is recommend to collect metrics generated by itself
-# [export_metrics.self_import]
-# db = "information_schema"
+## The log level. Can be `info`/`debug`/`warn`/`error`.
+## +toml2docs:none-default
+level = "info"
+
+## Enable OTLP tracing.
+enable_otlp_tracing = false
+
+## The OTLP tracing endpoint.
+## +toml2docs:none-default
+otlp_endpoint = ""
+
+## Whether to append logs to stdout.
+append_stdout = true
+
+## The percentage of tracing will be sampled and exported.
+## Valid range `[0, 1]`, 1 means all traces are sampled, 0 means all traces are not sampled, the default value is 1.
+## ratio > 1 are treated as 1. Fractions < 0 are treated as 0
+[logging.tracing_sample_ratio]
+default_ratio = 1.0
+
+## The datanode can export its metrics and send to Prometheus compatible service (e.g. send to `greptimedb` itself) from remote-write API.
+## This is only used for `greptimedb` to export its own metrics internally. It's different from prometheus scrape.
+[export_metrics]
+
+## whether enable export metrics.
+enable = false
+
+## The interval of export metrics.
+write_interval = "30s"
+
+## For `standalone` mode, `self_import` is recommend to collect metrics generated by itself
+[export_metrics.self_import]
+## +toml2docs:none-default
+db = "information_schema"
+
+[export_metrics.remote_write]
+## The url the metrics send to. The url example can be: `http://127.0.0.1:4000/v1/prometheus/write?db=information_schema`.
+url = ""
+
+## HTTP headers of Prometheus remote-write carry.
+headers = { }
--- a/docs/how-to/how-to-write-fuzz-tests.md
+++ b/docs/how-to/how-to-write-fuzz-tests.md
@@ -0,0 +1,136 @@
+# How to write fuzz tests
+
+This document introduces how to write fuzz tests in GreptimeDB.
+
+## What is a fuzz test
+Fuzz test is tool that leverage deterministic random generation to assist in finding bugs. The goal of fuzz tests is to identify inputs generated by the fuzzer that cause system panics, crashes, or unexpected behaviors to occur. And we are using the [cargo-fuzz](https://github.com/rust-fuzz/cargo-fuzz) to run our fuzz test targets. 
+
+## Why we need them
+- Find bugs by leveraging random generation
+- Integrate with other tests (e.g., e2e)
+
+## Resources
+All fuzz test-related resources are located in the `/tests-fuzz` directory.
+There are two types of resources: (1) fundamental components and (2) test targets.
+
+### Fundamental components 
+They are located in the `/tests-fuzz/src` directory. The fundamental components define how to generate SQLs (including dialects for different protocols) and validate execution results (e.g., column attribute validation), etc.
+
+### Test targets
+They are located in the `/tests-fuzz/targets` directory, with each file representing an independent fuzz test case. The target utilizes fundamental components to generate SQLs, sends the generated SQLs via specified protocol, and validates the results of SQL execution.
+
+Figure 1 illustrates the fundamental components of the fuzz test provide the ability to generate random SQLs. It utilizes a Random Number Generator (Rng) to generate the Intermediate Representation (IR), then employs a DialectTranslator to produce specified dialects for different protocols. Finally, the fuzz tests send the generated SQL via the specified protocol and verify that the execution results meet expectations.
+```
+                            Rng                                 
+                             |                                  
+                             |                                  
+                             v                                  
+                       ExprGenerator                            
+                             |                                  
+                             |                                  
+                             v                                  
+               Intermediate representation (IR)                 
+                             |                                  
+                             |                                  
+      +----------------------+----------------------+           
+      |                      |                      |           
+      v                      v                      v           
+MySQLTranslator    PostgreSQLTranslator   OtherDialectTranslator
+      |                      |                      |           
+      |                      |                      |           
+      v                      v                      v           
+SQL(MySQL Dialect)         .....                  .....         
+      |
+      |
+      v
+  Fuzz Test
+
+```
+(Figure1: Overview of fuzz tests)
+
+For more details about fuzz targets and fundamental components, please refer to this [tracking issue](https://github.com/GreptimeTeam/greptimedb/issues/3174).
+
+## How to add a fuzz test target
+
+1. Create an empty rust source file under the `/tests-fuzz/targets/<fuzz-target>.rs` directory.
+
+2. Register the fuzz test target in the `/tests-fuzz/Cargo.toml` file.
+
+```toml
+[[bin]]
+name = "<fuzz-target>"
+path = "targets/<fuzz-target>.rs"
+test = false
+bench = false
+doc = false
+```
+
+3. Define the `FuzzInput` in the `/tests-fuzz/targets/<fuzz-target>.rs`.
+
+```rust
+#![no_main]
+use libfuzzer_sys::arbitrary::{Arbitrary, Unstructured};
+
+#[derive(Clone, Debug)]
+struct FuzzInput {
+    seed: u64,
+}
+
+impl Arbitrary<'_> for FuzzInput {
+    fn arbitrary(u: &mut Unstructured<'_>) -> arbitrary::Result<Self> {
+        let seed = u.int_in_range(u64::MIN..=u64::MAX)?;
+        Ok(FuzzInput { seed })
+    }
+}
+```
+
+4. Write your first fuzz test target in the `/tests-fuzz/targets/<fuzz-target>.rs`.
+
+```rust
+use libfuzzer_sys::fuzz_target;
+use rand::{Rng, SeedableRng};
+use rand_chacha::ChaChaRng;
+use snafu::ResultExt;
+use sqlx::{MySql, Pool};
+use tests_fuzz::fake::{
+    merge_two_word_map_fn, random_capitalize_map, uppercase_and_keyword_backtick_map,
+    MappedGenerator, WordGenerator,
+};
+use tests_fuzz::generator::create_expr::CreateTableExprGeneratorBuilder;
+use tests_fuzz::generator::Generator;
+use tests_fuzz::ir::CreateTableExpr;
+use tests_fuzz::translator::mysql::create_expr::CreateTableExprTranslator;
+use tests_fuzz::translator::DslTranslator;
+use tests_fuzz::utils::{init_greptime_connections, Connections};
+
+fuzz_target!(|input: FuzzInput| {
+    common_telemetry::init_default_ut_logging();
+    common_runtime::block_on_write(async {
+        let Connections { mysql } = init_greptime_connections().await;
+            let mut rng = ChaChaRng::seed_from_u64(input.seed);
+            let columns = rng.gen_range(2..30);
+            let create_table_generator = CreateTableExprGeneratorBuilder::default()
+                .name_generator(Box::new(MappedGenerator::new(
+                    WordGenerator,
+                    merge_two_word_map_fn(random_capitalize_map, uppercase_and_keyword_backtick_map),
+                )))
+                .columns(columns)
+                .engine("mito")
+                .if_not_exists(if_not_exists)
+                .build()
+                .unwrap();
+            let ir = create_table_generator.generate(&mut rng);
+            let translator = CreateTableExprTranslator;
+            let sql = translator.translate(&expr).unwrap();
+            mysql.execute(&sql).await
+    })
+});
+```
+
+5. Run your fuzz test target
+
+```bash
+    cargo fuzz run <fuzz-target> --fuzz-dir tests-fuzz
+```
+
+For more details, please refer to this [document](/tests-fuzz/README.md).
--- a/docs/rfcs/2023-07-06-table-engine-refactor.md
+++ b/docs/rfcs/2023-07-06-table-engine-refactor.md
@@ -27,8 +27,8 @@ subgraph Frontend["Frontend"]
    end
 end

-MyTable --> MetaSrv
-MetaSrv --> ETCD
+MyTable --> Metasrv
+Metasrv --> ETCD

 MyTable-->TableEngine0
 MyTable-->TableEngine1
@@ -95,8 +95,8 @@ subgraph Frontend["Frontend"]
    end
 end

-MyTable --> MetaSrv
-MetaSrv --> ETCD
+MyTable --> Metasrv
+Metasrv --> ETCD

 MyTable-->RegionEngine
 MyTable-->RegionEngine1
--- a/docs/rfcs/2024-01-17-dataflow-framework.md
+++ b/docs/rfcs/2024-01-17-dataflow-framework.md
@@ -36,7 +36,7 @@ Hence, we choose the third option, and use a simple logical plan that's anagonis
 ## Deploy mode and protocol
 - Greptime Flow is an independent streaming compute component. It can be used either within a standalone node or as a dedicated node at the same level as frontend in distributed mode.
 - It accepts insert request Rows, which is used between frontend and datanode.
- New flow job is submitted in the format of modified SQL query like snowflake do, like: `CREATE TASK avg_over_5m WINDOW_SIZE = "5m" AS SELECT avg(value) FROM table WHERE time > now() - 5m GROUP BY time(1m)`. Flow job then got stored in MetaSrv.
+- New flow job is submitted in the format of modified SQL query like snowflake do, like: `CREATE TASK avg_over_5m WINDOW_SIZE = "5m" AS SELECT avg(value) FROM table WHERE time > now() - 5m GROUP BY time(1m)`. Flow job then got stored in Metasrv.
 - It also persists results in the format of Rows to frontend.
 - The query plan uses Substrait as codec format. It's the same with GreptimeDB's query engine.
 - Greptime Flow needs a WAL for recovering. It's possible to reuse datanode's.
--- a/docs/schema-structs.md
+++ b/docs/schema-structs.md
@@ -73,7 +73,7 @@ CREATE TABLE cpu (
    usage_system DOUBLE,
    datacenter STRING,
    TIME INDEX (ts),
-    PRIMARY KEY(datacenter, host)) ENGINE=mito WITH(regions=1);
+    PRIMARY KEY(datacenter, host)) ENGINE=mito;
 ```

 Then the table's `TableMeta` may look like this:
@@ -249,7 +249,7 @@ CREATE TABLE cpu (
    usage_system DOUBLE,
    datacenter STRING,
    TIME INDEX (ts),
-    PRIMARY KEY(datacenter, host)) ENGINE=mito WITH(regions=1);
+    PRIMARY KEY(datacenter, host)) ENGINE=mito;

 select ts, usage_system from cpu;
 ```
--- a/docs/style-guide.md
+++ b/docs/style-guide.md
@@ -0,0 +1,46 @@
+# GreptimeDB Style Guide
+
+This style guide is intended to help contributors to GreptimeDB write code that is consistent with the rest of the codebase. It is a living document and will be updated as the codebase evolves.
+
+It's mainly an complement to the [Rust Style Guide](https://pingcap.github.io/style-guide/rust/).
+
+## Table of Contents
+
+- Formatting
+- Modules
+- Comments
+
+## Formatting
+
+- Place all `mod` declaration before any `use`.
+- Use `unimplemented!()` instead of `todo!()` for things that aren't likely to be implemented.
+- Add an empty line before and after declaration blocks.
+- Place comment before attributes (`#[]`) and derive (`#[derive]`).
+
+## Modules
+
+- Use the file with same name instead of `mod.rs` to define a module. E.g.:
+
+```
+.
+├── cache
+│  ├── cache_size.rs
+│  └── write_cache.rs
+└── cache.rs
+```
+
+## Comments
+
+- Add comments for public functions and structs.
+- Prefer document comment (`///`) over normal comment (`//`) for structs, fields, functions etc.
+- Add link (`[]`) to struct, method, or any other reference. And make sure that link works.
+
+## Error handling
+
+- Define a custom error type for the module if needed.
+- Prefer `with_context()` over `context()` when allocation is needed to construct an error.
+- Use `error!()` or `warn!()` macros in the `common_telemetry` crate to log errors. E.g.:
+
+```rust
+error!(e; "Failed to do something");
+```
--- a/rust-toolchain.toml
+++ b/rust-toolchain.toml
@@ -1,2 +1,2 @@
 [toolchain]
-channel = "nightly-2023-12-19"
+channel = "nightly-2024-04-18"
--- a/src/api/src/lib.rs
+++ b/src/api/src/lib.rs
@@ -21,6 +21,7 @@ pub mod prom_store {
    }
 }

+pub mod region;
 pub mod v1;

 pub use greptime_proto;
--- a/src/api/src/region.rs
+++ b/src/api/src/region.rs
@@ -0,0 +1,42 @@
+// Copyright 2023 Greptime Team
+//
+// Licensed under the Apache License, Version 2.0 (the "License");
+// you may not use this file except in compliance with the License.
+// You may obtain a copy of the License at
+//
+//     http://www.apache.org/licenses/LICENSE-2.0
+//
+// Unless required by applicable law or agreed to in writing, software
+// distributed under the License is distributed on an "AS IS" BASIS,
+// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+// See the License for the specific language governing permissions and
+// limitations under the License.
+
+use std::collections::HashMap;
+
+use common_base::AffectedRows;
+use greptime_proto::v1::region::RegionResponse as RegionResponseV1;
+
+/// This result struct is derived from [RegionResponseV1]
+#[derive(Debug)]
+pub struct RegionResponse {
+    pub affected_rows: AffectedRows,
+    pub extension: HashMap<String, Vec<u8>>,
+}
+
+impl RegionResponse {
+    pub fn from_region_response(region_response: RegionResponseV1) -> Self {
+        Self {
+            affected_rows: region_response.affected_rows as _,
+            extension: region_response.extension,
+        }
+    }
+
+    /// Creates one response without extension
+    pub fn new(affected_rows: AffectedRows) -> Self {
+        Self {
+            affected_rows,
+            extension: Default::default(),
+        }
+    }
+}
--- a/src/auth/src/tests.rs
+++ b/src/auth/src/tests.rs
@@ -45,9 +45,9 @@ impl Default for MockUserProvider {

 impl MockUserProvider {
    pub fn set_authorization_info(&mut self, info: DatabaseAuthInfo) {
-        self.catalog = info.catalog.to_owned();
-        self.schema = info.schema.to_owned();
-        self.username = info.username.to_owned();
+        info.catalog.clone_into(&mut self.catalog);
+        info.schema.clone_into(&mut self.schema);
+        info.username.clone_into(&mut self.username);
    }
 }

--- a/src/catalog/Cargo.toml
+++ b/src/catalog/Cargo.toml
@@ -40,6 +40,7 @@ prometheus.workspace = true
 serde_json.workspace = true
 session.workspace = true
 snafu.workspace = true
+sql.workspace = true
 store-api.workspace = true
 table.workspace = true
 tokio.workspace = true
--- a/src/catalog/src/error.rs
+++ b/src/catalog/src/error.rs
@@ -216,7 +216,7 @@ pub enum Error {
    },

    #[snafu(display("Failed to perform metasrv operation"))]
-    MetaSrv {
+    Metasrv {
        location: Location,
        source: meta_client::error::Error,
    },
@@ -304,7 +304,7 @@ impl ErrorExt for Error {
            | Error::CreateTable { source, .. }
            | Error::TableSchemaMismatch { source, .. } => source.status_code(),

-            Error::MetaSrv { source, .. } => source.status_code(),
+            Error::Metasrv { source, .. } => source.status_code(),
            Error::SystemCatalogTableScan { source, .. } => source.status_code(),
            Error::SystemCatalogTableScanExec { source, .. } => source.status_code(),
            Error::InvalidTableInfoInCatalog { source, .. } => source.status_code(),
--- a/src/catalog/src/information_schema.rs
+++ b/src/catalog/src/information_schema.rs
@@ -20,6 +20,7 @@ mod predicate;
 mod region_peers;
 mod runtime_metrics;
 pub mod schemata;
+mod table_constraints;
 mod table_names;
 pub mod tables;

@@ -41,8 +42,7 @@ use table::error::{SchemaConversionSnafu, TablesRecordBatchSnafu};
 use table::metadata::{
    FilterPushDownType, TableInfoBuilder, TableInfoRef, TableMetaBuilder, TableType,
 };
-use table::thin_table::{ThinTable, ThinTableAdapter};
-use table::TableRef;
+use table::{Table, TableRef};
 pub use table_names::*;

 use self::columns::InformationSchemaColumns;
@@ -53,6 +53,7 @@ use crate::information_schema::partitions::InformationSchemaPartitions;
 use crate::information_schema::region_peers::InformationSchemaRegionPeers;
 use crate::information_schema::runtime_metrics::InformationSchemaMetrics;
 use crate::information_schema::schemata::InformationSchemaSchemata;
+use crate::information_schema::table_constraints::InformationSchemaTableConstraints;
 use crate::information_schema::tables::InformationSchemaTables;
 use crate::CatalogManager;

@@ -174,6 +175,10 @@ impl InformationSchemaProvider {
            KEY_COLUMN_USAGE.to_string(),
            self.build_table(KEY_COLUMN_USAGE).unwrap(),
        );
+        tables.insert(
+            TABLE_CONSTRAINTS.to_string(),
+            self.build_table(TABLE_CONSTRAINTS).unwrap(),
+        );

        // Add memory tables
        for name in MEMORY_TABLES.iter() {
@@ -187,10 +192,9 @@ impl InformationSchemaProvider {
        self.information_table(name).map(|table| {
            let table_info = Self::table_info(self.catalog_name.clone(), &table);
            let filter_pushdown = FilterPushDownType::Inexact;
-            let thin_table = ThinTable::new(table_info, filter_pushdown);
-
            let data_source = Arc::new(InformationTableDataSource::new(table));
-            Arc::new(ThinTableAdapter::new(thin_table, data_source)) as _
+            let table = Table::new(table_info, filter_pushdown, data_source);
+            Arc::new(table)
        })
    }

@@ -243,6 +247,10 @@ impl InformationSchemaProvider {
                self.catalog_name.clone(),
                self.catalog_manager.clone(),
            )) as _),
+            TABLE_CONSTRAINTS => Some(Arc::new(InformationSchemaTableConstraints::new(
+                self.catalog_name.clone(),
+                self.catalog_manager.clone(),
+            )) as _),
            _ => None,
        }
    }
--- a/src/catalog/src/information_schema/columns.rs
+++ b/src/catalog/src/information_schema/columns.rs
@@ -26,13 +26,16 @@ use common_recordbatch::{RecordBatch, SendableRecordBatchStream};
 use datafusion::physical_plan::stream::RecordBatchStreamAdapter as DfRecordBatchStreamAdapter;
 use datafusion::physical_plan::streaming::PartitionStream as DfPartitionStream;
 use datafusion::physical_plan::SendableRecordBatchStream as DfSendableRecordBatchStream;
-use datatypes::prelude::{ConcreteDataType, DataType};
+use datatypes::prelude::{ConcreteDataType, DataType, MutableVector};
 use datatypes::scalars::ScalarVectorBuilder;
 use datatypes::schema::{ColumnSchema, Schema, SchemaRef};
 use datatypes::value::Value;
-use datatypes::vectors::{StringVectorBuilder, VectorRef};
+use datatypes::vectors::{
+    ConstantVector, Int64Vector, Int64VectorBuilder, StringVector, StringVectorBuilder, VectorRef,
+};
 use futures::TryStreamExt;
 use snafu::{OptionExt, ResultExt};
+use sql::statements;
 use store_api::storage::{ScanRequest, TableId};

 use super::{InformationTable, COLUMNS};
@@ -52,14 +55,38 @@ pub const TABLE_CATALOG: &str = "table_catalog";
 pub const TABLE_SCHEMA: &str = "table_schema";
 pub const TABLE_NAME: &str = "table_name";
 pub const COLUMN_NAME: &str = "column_name";
+const ORDINAL_POSITION: &str = "ordinal_position";
+const CHARACTER_MAXIMUM_LENGTH: &str = "character_maximum_length";
+const CHARACTER_OCTET_LENGTH: &str = "character_octet_length";
+const NUMERIC_PRECISION: &str = "numeric_precision";
+const NUMERIC_SCALE: &str = "numeric_scale";
+const DATETIME_PRECISION: &str = "datetime_precision";
+const CHARACTER_SET_NAME: &str = "character_set_name";
+pub const COLLATION_NAME: &str = "collation_name";
+pub const COLUMN_KEY: &str = "column_key";
+pub const EXTRA: &str = "extra";
+pub const PRIVILEGES: &str = "privileges";
+const GENERATION_EXPRESSION: &str = "generation_expression";
+// Extension field to keep greptime data type name
+pub const GREPTIME_DATA_TYPE: &str = "greptime_data_type";
 pub const DATA_TYPE: &str = "data_type";
 pub const SEMANTIC_TYPE: &str = "semantic_type";
 pub const COLUMN_DEFAULT: &str = "column_default";
 pub const IS_NULLABLE: &str = "is_nullable";
 const COLUMN_TYPE: &str = "column_type";
 pub const COLUMN_COMMENT: &str = "column_comment";
+const SRS_ID: &str = "srs_id";
 const INIT_CAPACITY: usize = 42;

+// The maximum length of string type
+const MAX_STRING_LENGTH: i64 = 2147483647;
+const UTF8_CHARSET_NAME: &str = "utf8";
+const UTF8_COLLATE_NAME: &str = "utf8_bin";
+const PRI_COLUMN_KEY: &str = "PRI";
+const TIME_INDEX_COLUMN_KEY: &str = "TIME INDEX";
+const DEFAULT_PRIVILEGES: &str = "select,insert";
+const EMPTY_STR: &str = "";
+
 impl InformationSchemaColumns {
    pub(super) fn new(catalog_name: String, catalog_manager: Weak<dyn CatalogManager>) -> Self {
        Self {
@@ -75,12 +102,46 @@ impl InformationSchemaColumns {
            ColumnSchema::new(TABLE_SCHEMA, ConcreteDataType::string_datatype(), false),
            ColumnSchema::new(TABLE_NAME, ConcreteDataType::string_datatype(), false),
            ColumnSchema::new(COLUMN_NAME, ConcreteDataType::string_datatype(), false),
+            ColumnSchema::new(ORDINAL_POSITION, ConcreteDataType::int64_datatype(), false),
+            ColumnSchema::new(
+                CHARACTER_MAXIMUM_LENGTH,
+                ConcreteDataType::int64_datatype(),
+                true,
+            ),
+            ColumnSchema::new(
+                CHARACTER_OCTET_LENGTH,
+                ConcreteDataType::int64_datatype(),
+                true,
+            ),
+            ColumnSchema::new(NUMERIC_PRECISION, ConcreteDataType::int64_datatype(), true),
+            ColumnSchema::new(NUMERIC_SCALE, ConcreteDataType::int64_datatype(), true),
+            ColumnSchema::new(DATETIME_PRECISION, ConcreteDataType::int64_datatype(), true),
+            ColumnSchema::new(
+                CHARACTER_SET_NAME,
+                ConcreteDataType::string_datatype(),
+                true,
+            ),
+            ColumnSchema::new(COLLATION_NAME, ConcreteDataType::string_datatype(), true),
+            ColumnSchema::new(COLUMN_KEY, ConcreteDataType::string_datatype(), false),
+            ColumnSchema::new(EXTRA, ConcreteDataType::string_datatype(), false),
+            ColumnSchema::new(PRIVILEGES, ConcreteDataType::string_datatype(), false),
+            ColumnSchema::new(
+                GENERATION_EXPRESSION,
+                ConcreteDataType::string_datatype(),
+                false,
+            ),
+            ColumnSchema::new(
+                GREPTIME_DATA_TYPE,
+                ConcreteDataType::string_datatype(),
+                false,
+            ),
            ColumnSchema::new(DATA_TYPE, ConcreteDataType::string_datatype(), false),
            ColumnSchema::new(SEMANTIC_TYPE, ConcreteDataType::string_datatype(), false),
            ColumnSchema::new(COLUMN_DEFAULT, ConcreteDataType::string_datatype(), true),
            ColumnSchema::new(IS_NULLABLE, ConcreteDataType::string_datatype(), false),
            ColumnSchema::new(COLUMN_TYPE, ConcreteDataType::string_datatype(), false),
            ColumnSchema::new(COLUMN_COMMENT, ConcreteDataType::string_datatype(), true),
+            ColumnSchema::new(SRS_ID, ConcreteDataType::int64_datatype(), true),
        ]))
    }

@@ -136,9 +197,18 @@ struct InformationSchemaColumnsBuilder {
    schema_names: StringVectorBuilder,
    table_names: StringVectorBuilder,
    column_names: StringVectorBuilder,
+    ordinal_positions: Int64VectorBuilder,
+    character_maximum_lengths: Int64VectorBuilder,
+    character_octet_lengths: Int64VectorBuilder,
+    numeric_precisions: Int64VectorBuilder,
+    numeric_scales: Int64VectorBuilder,
+    datetime_precisions: Int64VectorBuilder,
+    character_set_names: StringVectorBuilder,
+    collation_names: StringVectorBuilder,
+    column_keys: StringVectorBuilder,
+    greptime_data_types: StringVectorBuilder,
    data_types: StringVectorBuilder,
    semantic_types: StringVectorBuilder,
-
    column_defaults: StringVectorBuilder,
    is_nullables: StringVectorBuilder,
    column_types: StringVectorBuilder,
@@ -159,6 +229,16 @@ impl InformationSchemaColumnsBuilder {
            schema_names: StringVectorBuilder::with_capacity(INIT_CAPACITY),
            table_names: StringVectorBuilder::with_capacity(INIT_CAPACITY),
            column_names: StringVectorBuilder::with_capacity(INIT_CAPACITY),
+            ordinal_positions: Int64VectorBuilder::with_capacity(INIT_CAPACITY),
+            character_maximum_lengths: Int64VectorBuilder::with_capacity(INIT_CAPACITY),
+            character_octet_lengths: Int64VectorBuilder::with_capacity(INIT_CAPACITY),
+            numeric_precisions: Int64VectorBuilder::with_capacity(INIT_CAPACITY),
+            numeric_scales: Int64VectorBuilder::with_capacity(INIT_CAPACITY),
+            datetime_precisions: Int64VectorBuilder::with_capacity(INIT_CAPACITY),
+            character_set_names: StringVectorBuilder::with_capacity(INIT_CAPACITY),
+            collation_names: StringVectorBuilder::with_capacity(INIT_CAPACITY),
+            column_keys: StringVectorBuilder::with_capacity(INIT_CAPACITY),
+            greptime_data_types: StringVectorBuilder::with_capacity(INIT_CAPACITY),
            data_types: StringVectorBuilder::with_capacity(INIT_CAPACITY),
            semantic_types: StringVectorBuilder::with_capacity(INIT_CAPACITY),
            column_defaults: StringVectorBuilder::with_capacity(INIT_CAPACITY),
@@ -195,6 +275,7 @@ impl InformationSchemaColumnsBuilder {

                    self.add_column(
                        &predicates,
+                        idx,
                        &catalog_name,
                        &schema_name,
                        &table.table_info().name,
@@ -208,16 +289,27 @@ impl InformationSchemaColumnsBuilder {
        self.finish()
    }

+    #[allow(clippy::too_many_arguments)]
    fn add_column(
        &mut self,
        predicates: &Predicates,
+        index: usize,
        catalog_name: &str,
        schema_name: &str,
        table_name: &str,
        semantic_type: &str,
        column_schema: &ColumnSchema,
    ) {
-        let data_type = &column_schema.data_type.name();
+        // Use sql data type name
+        let data_type = statements::concrete_data_type_to_sql_data_type(&column_schema.data_type)
+            .map(|dt| dt.to_string().to_lowercase())
+            .unwrap_or_else(|_| column_schema.data_type.name());
+
+        let column_key = match semantic_type {
+            SEMANTIC_TYPE_PRIMARY_KEY => PRI_COLUMN_KEY,
+            SEMANTIC_TYPE_TIME_INDEX => TIME_INDEX_COLUMN_KEY,
+            _ => EMPTY_STR,
+        };

        let row = [
            (TABLE_CATALOG, &Value::from(catalog_name)),
@@ -226,6 +318,8 @@ impl InformationSchemaColumnsBuilder {
            (COLUMN_NAME, &Value::from(column_schema.name.as_str())),
            (DATA_TYPE, &Value::from(data_type.as_str())),
            (SEMANTIC_TYPE, &Value::from(semantic_type)),
+            (ORDINAL_POSITION, &Value::from((index + 1) as i64)),
+            (COLUMN_KEY, &Value::from(column_key)),
        ];

        if !predicates.eval(&row) {
@@ -236,7 +330,63 @@ impl InformationSchemaColumnsBuilder {
        self.schema_names.push(Some(schema_name));
        self.table_names.push(Some(table_name));
        self.column_names.push(Some(&column_schema.name));
-        self.data_types.push(Some(data_type));
+        // Starts from 1
+        self.ordinal_positions.push(Some((index + 1) as i64));
+
+        if column_schema.data_type.is_string() {
+            self.character_maximum_lengths.push(Some(MAX_STRING_LENGTH));
+            self.character_octet_lengths.push(Some(MAX_STRING_LENGTH));
+            self.numeric_precisions.push(None);
+            self.numeric_scales.push(None);
+            self.datetime_precisions.push(None);
+            self.character_set_names.push(Some(UTF8_CHARSET_NAME));
+            self.collation_names.push(Some(UTF8_COLLATE_NAME));
+        } else if column_schema.data_type.is_numeric() || column_schema.data_type.is_decimal() {
+            self.character_maximum_lengths.push(None);
+            self.character_octet_lengths.push(None);
+
+            self.numeric_precisions.push(
+                column_schema
+                    .data_type
+                    .numeric_precision()
+                    .map(|x| x as i64),
+            );
+            self.numeric_scales
+                .push(column_schema.data_type.numeric_scale().map(|x| x as i64));
+
+            self.datetime_precisions.push(None);
+            self.character_set_names.push(None);
+            self.collation_names.push(None);
+        } else {
+            self.character_maximum_lengths.push(None);
+            self.character_octet_lengths.push(None);
+            self.numeric_precisions.push(None);
+            self.numeric_scales.push(None);
+
+            match &column_schema.data_type {
+                ConcreteDataType::DateTime(datetime_type) => {
+                    self.datetime_precisions
+                        .push(Some(datetime_type.precision() as i64));
+                }
+                ConcreteDataType::Timestamp(ts_type) => {
+                    self.datetime_precisions
+                        .push(Some(ts_type.precision() as i64));
+                }
+                ConcreteDataType::Time(time_type) => {
+                    self.datetime_precisions
+                        .push(Some(time_type.precision() as i64));
+                }
+                _ => self.datetime_precisions.push(None),
+            }
+
+            self.character_set_names.push(None);
+            self.collation_names.push(None);
+        }
+
+        self.column_keys.push(Some(column_key));
+        self.greptime_data_types
+            .push(Some(&column_schema.data_type.name()));
+        self.data_types.push(Some(&data_type));
        self.semantic_types.push(Some(semantic_type));
        self.column_defaults.push(
            column_schema
@@ -249,23 +399,52 @@ impl InformationSchemaColumnsBuilder {
        } else {
            self.is_nullables.push(Some("No"));
        }
-        self.column_types.push(Some(data_type));
+        self.column_types.push(Some(&data_type));
        self.column_comments
            .push(column_schema.column_comment().map(|x| x.as_ref()));
    }

    fn finish(&mut self) -> Result<RecordBatch> {
+        let rows_num = self.collation_names.len();
+
+        let privileges = Arc::new(ConstantVector::new(
+            Arc::new(StringVector::from(vec![DEFAULT_PRIVILEGES])),
+            rows_num,
+        ));
+        let empty_string = Arc::new(ConstantVector::new(
+            Arc::new(StringVector::from(vec![EMPTY_STR])),
+            rows_num,
+        ));
+        let srs_ids = Arc::new(ConstantVector::new(
+            Arc::new(Int64Vector::from(vec![None])),
+            rows_num,
+        ));
+
        let columns: Vec<VectorRef> = vec![
            Arc::new(self.catalog_names.finish()),
            Arc::new(self.schema_names.finish()),
            Arc::new(self.table_names.finish()),
            Arc::new(self.column_names.finish()),
+            Arc::new(self.ordinal_positions.finish()),
+            Arc::new(self.character_maximum_lengths.finish()),
+            Arc::new(self.character_octet_lengths.finish()),
+            Arc::new(self.numeric_precisions.finish()),
+            Arc::new(self.numeric_scales.finish()),
+            Arc::new(self.datetime_precisions.finish()),
+            Arc::new(self.character_set_names.finish()),
+            Arc::new(self.collation_names.finish()),
+            Arc::new(self.column_keys.finish()),
+            empty_string.clone(),
+            privileges,
+            empty_string,
+            Arc::new(self.greptime_data_types.finish()),
            Arc::new(self.data_types.finish()),
            Arc::new(self.semantic_types.finish()),
            Arc::new(self.column_defaults.finish()),
            Arc::new(self.is_nullables.finish()),
            Arc::new(self.column_types.finish()),
            Arc::new(self.column_comments.finish()),
+            srs_ids,
        ];

        RecordBatch::new(self.schema.clone(), columns).context(CreateRecordBatchSnafu)
--- a/src/catalog/src/information_schema/key_column_usage.rs
+++ b/src/catalog/src/information_schema/key_column_usage.rs
@@ -49,6 +49,11 @@ pub const COLUMN_NAME: &str = "column_name";
 pub const ORDINAL_POSITION: &str = "ordinal_position";
 const INIT_CAPACITY: usize = 42;

+/// Primary key constraint name
+pub(crate) const PRI_CONSTRAINT_NAME: &str = "PRIMARY";
+/// Time index constraint name
+pub(crate) const TIME_INDEX_CONSTRAINT_NAME: &str = "TIME INDEX";
+
 /// The virtual table implementation for `information_schema.KEY_COLUMN_USAGE`.
 pub(super) struct InformationSchemaKeyColumnUsage {
    schema: SchemaRef,
@@ -232,7 +237,7 @@ impl InformationSchemaKeyColumnUsageBuilder {
                            self.add_key_column_usage(
                                &predicates,
                                &schema_name,
-                                "TIME INDEX",
+                                TIME_INDEX_CONSTRAINT_NAME,
                                &catalog_name,
                                &schema_name,
                                &table_name,
@@ -262,7 +267,7 @@ impl InformationSchemaKeyColumnUsageBuilder {
            self.add_key_column_usage(
                &predicates,
                &schema_name,
-                "PRIMARY",
+                PRI_CONSTRAINT_NAME,
                &catalog_name,
                &schema_name,
                &table_name,
--- a/src/catalog/src/information_schema/predicate.rs
+++ b/src/catalog/src/information_schema/predicate.rs
@@ -109,11 +109,7 @@ impl Predicate {
                };
            }
            Predicate::Not(p) => {
-                let Some(b) = p.eval(row) else {
-                    return None;
-                };
-
-                return Some(!b);
+                return Some(!p.eval(row)?);
            }
        }

@@ -125,13 +121,7 @@ impl Predicate {
    fn from_expr(expr: DfExpr) -> Option<Predicate> {
        match expr {
            // NOT expr
-            DfExpr::Not(expr) => {
-                let Some(p) = Self::from_expr(*expr) else {
-                    return None;
-                };
-
-                Some(Predicate::Not(Box::new(p)))
-            }
+            DfExpr::Not(expr) => Some(Predicate::Not(Box::new(Self::from_expr(*expr)?))),
            // expr LIKE pattern
            DfExpr::Like(Like {
                negated,
@@ -178,25 +168,15 @@ impl Predicate {
                }
                // left AND right
                (left, Operator::And, right) => {
-                    let Some(left) = Self::from_expr(left) else {
-                        return None;
-                    };
-
-                    let Some(right) = Self::from_expr(right) else {
-                        return None;
-                    };
+                    let left = Self::from_expr(left)?;
+                    let right = Self::from_expr(right)?;

                    Some(Predicate::And(Box::new(left), Box::new(right)))
                }
                // left OR right
                (left, Operator::Or, right) => {
-                    let Some(left) = Self::from_expr(left) else {
-                        return None;
-                    };
-
-                    let Some(right) = Self::from_expr(right) else {
-                        return None;
-                    };
+                    let left = Self::from_expr(left)?;
+                    let right = Self::from_expr(right)?;

                    Some(Predicate::Or(Box::new(left), Box::new(right)))
                }
--- a/src/catalog/src/information_schema/table_constraints.rs
+++ b/src/catalog/src/information_schema/table_constraints.rs
@@ -0,0 +1,286 @@
+// Copyright 2023 Greptime Team
+//
+// Licensed under the Apache License, Version 2.0 (the "License");
+// you may not use this file except in compliance with the License.
+// You may obtain a copy of the License at
+//
+//     http://www.apache.org/licenses/LICENSE-2.0
+//
+// Unless required by applicable law or agreed to in writing, software
+// distributed under the License is distributed on an "AS IS" BASIS,
+// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+// See the License for the specific language governing permissions and
+// limitations under the License.
+
+use std::sync::{Arc, Weak};
+
+use arrow_schema::SchemaRef as ArrowSchemaRef;
+use common_catalog::consts::INFORMATION_SCHEMA_TABLE_CONSTRAINTS_TABLE_ID;
+use common_error::ext::BoxedError;
+use common_query::physical_plan::TaskContext;
+use common_recordbatch::adapter::RecordBatchStreamAdapter;
+use common_recordbatch::{RecordBatch, SendableRecordBatchStream};
+use datafusion::physical_plan::stream::RecordBatchStreamAdapter as DfRecordBatchStreamAdapter;
+use datafusion::physical_plan::streaming::PartitionStream as DfPartitionStream;
+use datafusion::physical_plan::SendableRecordBatchStream as DfSendableRecordBatchStream;
+use datatypes::prelude::{ConcreteDataType, MutableVector};
+use datatypes::scalars::ScalarVectorBuilder;
+use datatypes::schema::{ColumnSchema, Schema, SchemaRef};
+use datatypes::value::Value;
+use datatypes::vectors::{ConstantVector, StringVector, StringVectorBuilder, VectorRef};
+use futures::TryStreamExt;
+use snafu::{OptionExt, ResultExt};
+use store_api::storage::{ScanRequest, TableId};
+
+use super::{InformationTable, TABLE_CONSTRAINTS};
+use crate::error::{
+    CreateRecordBatchSnafu, InternalSnafu, Result, UpgradeWeakCatalogManagerRefSnafu,
+};
+use crate::information_schema::key_column_usage::{
+    PRI_CONSTRAINT_NAME, TIME_INDEX_CONSTRAINT_NAME,
+};
+use crate::information_schema::Predicates;
+use crate::CatalogManager;
+
+/// The `TABLE_CONSTRAINTS` table describes which tables have constraints.
+pub(super) struct InformationSchemaTableConstraints {
+    schema: SchemaRef,
+    catalog_name: String,
+    catalog_manager: Weak<dyn CatalogManager>,
+}
+
+const CONSTRAINT_CATALOG: &str = "constraint_catalog";
+const CONSTRAINT_SCHEMA: &str = "constraint_schema";
+const CONSTRAINT_NAME: &str = "constraint_name";
+const TABLE_SCHEMA: &str = "table_schema";
+const TABLE_NAME: &str = "table_name";
+const CONSTRAINT_TYPE: &str = "constraint_type";
+const ENFORCED: &str = "enforced";
+
+const INIT_CAPACITY: usize = 42;
+
+const TIME_INDEX_CONSTRAINT_TYPE: &str = "TIME INDEX";
+const PRI_KEY_CONSTRAINT_TYPE: &str = "PRIMARY KEY";
+
+impl InformationSchemaTableConstraints {
+    pub(super) fn new(catalog_name: String, catalog_manager: Weak<dyn CatalogManager>) -> Self {
+        Self {
+            schema: Self::schema(),
+            catalog_name,
+            catalog_manager,
+        }
+    }
+
+    fn schema() -> SchemaRef {
+        Arc::new(Schema::new(vec![
+            ColumnSchema::new(
+                CONSTRAINT_CATALOG,
+                ConcreteDataType::string_datatype(),
+                false,
+            ),
+            ColumnSchema::new(
+                CONSTRAINT_SCHEMA,
+                ConcreteDataType::string_datatype(),
+                false,
+            ),
+            ColumnSchema::new(CONSTRAINT_NAME, ConcreteDataType::string_datatype(), false),
+            ColumnSchema::new(TABLE_SCHEMA, ConcreteDataType::string_datatype(), false),
+            ColumnSchema::new(TABLE_NAME, ConcreteDataType::string_datatype(), false),
+            ColumnSchema::new(CONSTRAINT_TYPE, ConcreteDataType::string_datatype(), false),
+            ColumnSchema::new(ENFORCED, ConcreteDataType::string_datatype(), false),
+        ]))
+    }
+
+    fn builder(&self) -> InformationSchemaTableConstraintsBuilder {
+        InformationSchemaTableConstraintsBuilder::new(
+            self.schema.clone(),
+            self.catalog_name.clone(),
+            self.catalog_manager.clone(),
+        )
+    }
+}
+
+impl InformationTable for InformationSchemaTableConstraints {
+    fn table_id(&self) -> TableId {
+        INFORMATION_SCHEMA_TABLE_CONSTRAINTS_TABLE_ID
+    }
+
+    fn table_name(&self) -> &'static str {
+        TABLE_CONSTRAINTS
+    }
+
+    fn schema(&self) -> SchemaRef {
+        self.schema.clone()
+    }
+
+    fn to_stream(&self, request: ScanRequest) -> Result<SendableRecordBatchStream> {
+        let schema = self.schema.arrow_schema().clone();
+        let mut builder = self.builder();
+        let stream = Box::pin(DfRecordBatchStreamAdapter::new(
+            schema,
+            futures::stream::once(async move {
+                builder
+                    .make_table_constraints(Some(request))
+                    .await
+                    .map(|x| x.into_df_record_batch())
+                    .map_err(Into::into)
+            }),
+        ));
+        Ok(Box::pin(
+            RecordBatchStreamAdapter::try_new(stream)
+                .map_err(BoxedError::new)
+                .context(InternalSnafu)?,
+        ))
+    }
+}
+
+struct InformationSchemaTableConstraintsBuilder {
+    schema: SchemaRef,
+    catalog_name: String,
+    catalog_manager: Weak<dyn CatalogManager>,
+
+    constraint_schemas: StringVectorBuilder,
+    constraint_names: StringVectorBuilder,
+    table_schemas: StringVectorBuilder,
+    table_names: StringVectorBuilder,
+    constraint_types: StringVectorBuilder,
+}
+
+impl InformationSchemaTableConstraintsBuilder {
+    fn new(
+        schema: SchemaRef,
+        catalog_name: String,
+        catalog_manager: Weak<dyn CatalogManager>,
+    ) -> Self {
+        Self {
+            schema,
+            catalog_name,
+            catalog_manager,
+            constraint_schemas: StringVectorBuilder::with_capacity(INIT_CAPACITY),
+            constraint_names: StringVectorBuilder::with_capacity(INIT_CAPACITY),
+            table_schemas: StringVectorBuilder::with_capacity(INIT_CAPACITY),
+            table_names: StringVectorBuilder::with_capacity(INIT_CAPACITY),
+            constraint_types: StringVectorBuilder::with_capacity(INIT_CAPACITY),
+        }
+    }
+
+    /// Construct the `information_schema.table_constraints` virtual table
+    async fn make_table_constraints(
+        &mut self,
+        request: Option<ScanRequest>,
+    ) -> Result<RecordBatch> {
+        let catalog_name = self.catalog_name.clone();
+        let catalog_manager = self
+            .catalog_manager
+            .upgrade()
+            .context(UpgradeWeakCatalogManagerRefSnafu)?;
+        let predicates = Predicates::from_scan_request(&request);
+
+        for schema_name in catalog_manager.schema_names(&catalog_name).await? {
+            let mut stream = catalog_manager.tables(&catalog_name, &schema_name).await;
+
+            while let Some(table) = stream.try_next().await? {
+                let keys = &table.table_info().meta.primary_key_indices;
+                let schema = table.schema();
+
+                if schema.timestamp_index().is_some() {
+                    self.add_table_constraint(
+                        &predicates,
+                        &schema_name,
+                        TIME_INDEX_CONSTRAINT_NAME,
+                        &schema_name,
+                        &table.table_info().name,
+                        TIME_INDEX_CONSTRAINT_TYPE,
+                    );
+                }
+
+                if !keys.is_empty() {
+                    self.add_table_constraint(
+                        &predicates,
+                        &schema_name,
+                        PRI_CONSTRAINT_NAME,
+                        &schema_name,
+                        &table.table_info().name,
+                        PRI_KEY_CONSTRAINT_TYPE,
+                    );
+                }
+            }
+        }
+
+        self.finish()
+    }
+
+    fn add_table_constraint(
+        &mut self,
+        predicates: &Predicates,
+        constraint_schema: &str,
+        constraint_name: &str,
+        table_schema: &str,
+        table_name: &str,
+        constraint_type: &str,
+    ) {
+        let row = [
+            (CONSTRAINT_SCHEMA, &Value::from(constraint_schema)),
+            (CONSTRAINT_NAME, &Value::from(constraint_name)),
+            (TABLE_SCHEMA, &Value::from(table_schema)),
+            (TABLE_NAME, &Value::from(table_name)),
+            (CONSTRAINT_TYPE, &Value::from(constraint_type)),
+        ];
+
+        if !predicates.eval(&row) {
+            return;
+        }
+
+        self.constraint_schemas.push(Some(constraint_schema));
+        self.constraint_names.push(Some(constraint_name));
+        self.table_schemas.push(Some(table_schema));
+        self.table_names.push(Some(table_name));
+        self.constraint_types.push(Some(constraint_type));
+    }
+
+    fn finish(&mut self) -> Result<RecordBatch> {
+        let rows_num = self.constraint_names.len();
+
+        let constraint_catalogs = Arc::new(ConstantVector::new(
+            Arc::new(StringVector::from(vec!["def"])),
+            rows_num,
+        ));
+        let enforceds = Arc::new(ConstantVector::new(
+            Arc::new(StringVector::from(vec!["YES"])),
+            rows_num,
+        ));
+
+        let columns: Vec<VectorRef> = vec![
+            constraint_catalogs,
+            Arc::new(self.constraint_schemas.finish()),
+            Arc::new(self.constraint_names.finish()),
+            Arc::new(self.table_schemas.finish()),
+            Arc::new(self.table_names.finish()),
+            Arc::new(self.constraint_types.finish()),
+            enforceds,
+        ];
+
+        RecordBatch::new(self.schema.clone(), columns).context(CreateRecordBatchSnafu)
+    }
+}
+
+impl DfPartitionStream for InformationSchemaTableConstraints {
+    fn schema(&self) -> &ArrowSchemaRef {
+        self.schema.arrow_schema()
+    }
+
+    fn execute(&self, _: Arc<TaskContext>) -> DfSendableRecordBatchStream {
+        let schema = self.schema.arrow_schema().clone();
+        let mut builder = self.builder();
+        Box::pin(DfRecordBatchStreamAdapter::new(
+            schema,
+            futures::stream::once(async move {
+                builder
+                    .make_table_constraints(None)
+                    .await
+                    .map(|x| x.into_df_record_batch())
+                    .map_err(Into::into)
+            }),
+        ))
+    }
+}
--- a/src/catalog/src/information_schema/table_names.rs
+++ b/src/catalog/src/information_schema/table_names.rs
@@ -41,3 +41,4 @@ pub const SESSION_STATUS: &str = "session_status";
 pub const RUNTIME_METRICS: &str = "runtime_metrics";
 pub const PARTITIONS: &str = "partitions";
 pub const REGION_PEERS: &str = "greptime_region_peers";
+pub const TABLE_CONSTRAINTS: &str = "table_constraints";
--- a/src/catalog/src/kvbackend/client.rs
+++ b/src/catalog/src/kvbackend/client.rs
@@ -17,7 +17,6 @@ use std::fmt::Debug;
 use std::sync::atomic::{AtomicUsize, Ordering};
 use std::sync::{Arc, Mutex};
 use std::time::Duration;
-use std::usize;

 use common_error::ext::BoxedError;
 use common_meta::cache_invalidator::KvCacheInvalidator;
@@ -364,6 +363,10 @@ impl KvBackend for MetaKvBackend {
        "MetaKvBackend"
    }

+    fn as_any(&self) -> &dyn Any {
+        self
+    }
+
    async fn range(&self, req: RangeRequest) -> Result<RangeResponse> {
        self.client
            .range(req)
@@ -372,27 +375,6 @@ impl KvBackend for MetaKvBackend {
            .context(ExternalSnafu)
    }

-    async fn get(&self, key: &[u8]) -> Result<Option<KeyValue>> {
-        let mut response = self
-            .client
-            .range(RangeRequest::new().with_key(key))
-            .await
-            .map_err(BoxedError::new)
-            .context(ExternalSnafu)?;
-        Ok(response.take_kvs().get_mut(0).map(|kv| KeyValue {
-            key: kv.take_key(),
-            value: kv.take_value(),
-        }))
-    }
-
-    async fn batch_put(&self, req: BatchPutRequest) -> Result<BatchPutResponse> {
-        self.client
-            .batch_put(req)
-            .await
-            .map_err(BoxedError::new)
-            .context(ExternalSnafu)
-    }
-
    async fn put(&self, req: PutRequest) -> Result<PutResponse> {
        self.client
            .put(req)
@@ -401,17 +383,9 @@ impl KvBackend for MetaKvBackend {
            .context(ExternalSnafu)
    }

-    async fn delete_range(&self, req: DeleteRangeRequest) -> Result<DeleteRangeResponse> {
+    async fn batch_put(&self, req: BatchPutRequest) -> Result<BatchPutResponse> {
        self.client
-            .delete_range(req)
-            .await
-            .map_err(BoxedError::new)
-            .context(ExternalSnafu)
-    }
-
-    async fn batch_delete(&self, req: BatchDeleteRequest) -> Result<BatchDeleteResponse> {
-        self.client
-            .batch_delete(req)
+            .batch_put(req)
            .await
            .map_err(BoxedError::new)
            .context(ExternalSnafu)
@@ -436,8 +410,33 @@ impl KvBackend for MetaKvBackend {
            .context(ExternalSnafu)
    }

-    fn as_any(&self) -> &dyn Any {
-        self
+    async fn delete_range(&self, req: DeleteRangeRequest) -> Result<DeleteRangeResponse> {
+        self.client
+            .delete_range(req)
+            .await
+            .map_err(BoxedError::new)
+            .context(ExternalSnafu)
+    }
+
+    async fn batch_delete(&self, req: BatchDeleteRequest) -> Result<BatchDeleteResponse> {
+        self.client
+            .batch_delete(req)
+            .await
+            .map_err(BoxedError::new)
+            .context(ExternalSnafu)
+    }
+
+    async fn get(&self, key: &[u8]) -> Result<Option<KeyValue>> {
+        let mut response = self
+            .client
+            .range(RangeRequest::new().with_key(key))
+            .await
+            .map_err(BoxedError::new)
+            .context(ExternalSnafu)?;
+        Ok(response.take_kvs().get_mut(0).map(|kv| KeyValue {
+            key: kv.take_key(),
+            value: kv.take_value(),
+        }))
    }
 }

@@ -506,32 +505,32 @@ mod tests {
        }

        async fn range(&self, _req: RangeRequest) -> Result<RangeResponse, Self::Error> {
-            todo!()
+            unimplemented!()
        }

        async fn batch_put(&self, _req: BatchPutRequest) -> Result<BatchPutResponse, Self::Error> {
-            todo!()
+            unimplemented!()
        }

        async fn compare_and_put(
            &self,
            _req: CompareAndPutRequest,
        ) -> Result<CompareAndPutResponse, Self::Error> {
-            todo!()
+            unimplemented!()
        }

        async fn delete_range(
            &self,
            _req: DeleteRangeRequest,
        ) -> Result<DeleteRangeResponse, Self::Error> {
-            todo!()
+            unimplemented!()
        }

        async fn batch_delete(
            &self,
            _req: BatchDeleteRequest,
        ) -> Result<BatchDeleteResponse, Self::Error> {
-            todo!()
+            unimplemented!()
        }
    }

--- a/src/catalog/src/table_source.rs
+++ b/src/catalog/src/table_source.rs
@@ -49,10 +49,7 @@ impl DfTableSourceProvider {
        }
    }

-    pub fn resolve_table_ref<'a>(
-        &'a self,
-        table_ref: TableReference<'a>,
-    ) -> Result<ResolvedTableReference<'a>> {
+    pub fn resolve_table_ref(&self, table_ref: TableReference) -> Result<ResolvedTableReference> {
        if self.disallow_cross_catalog_query {
            match &table_ref {
                TableReference::Bare { .. } => (),
@@ -76,7 +73,7 @@ impl DfTableSourceProvider {

    pub async fn resolve_table(
        &mut self,
-        table_ref: TableReference<'_>,
+        table_ref: TableReference,
    ) -> Result<Arc<dyn TableSource>> {
        let table_ref = self.resolve_table_ref(table_ref)?;

@@ -106,8 +103,6 @@ impl DfTableSourceProvider {

 #[cfg(test)]
 mod tests {
-    use std::borrow::Cow;
-
    use session::context::QueryContext;

    use super::*;
@@ -120,68 +115,37 @@ mod tests {
        let table_provider =
            DfTableSourceProvider::new(MemoryCatalogManager::with_default_setup(), true, query_ctx);

-        let table_ref = TableReference::Bare {
-            table: Cow::Borrowed("table_name"),
-        };
+        let table_ref = TableReference::bare("table_name");
        let result = table_provider.resolve_table_ref(table_ref);
        assert!(result.is_ok());

-        let table_ref = TableReference::Partial {
-            schema: Cow::Borrowed("public"),
-            table: Cow::Borrowed("table_name"),
-        };
+        let table_ref = TableReference::partial("public", "table_name");
        let result = table_provider.resolve_table_ref(table_ref);
        assert!(result.is_ok());

-        let table_ref = TableReference::Partial {
-            schema: Cow::Borrowed("wrong_schema"),
-            table: Cow::Borrowed("table_name"),
-        };
+        let table_ref = TableReference::partial("wrong_schema", "table_name");
        let result = table_provider.resolve_table_ref(table_ref);
        assert!(result.is_ok());

-        let table_ref = TableReference::Full {
-            catalog: Cow::Borrowed("greptime"),
-            schema: Cow::Borrowed("public"),
-            table: Cow::Borrowed("table_name"),
-        };
+        let table_ref = TableReference::full("greptime", "public", "table_name");
        let result = table_provider.resolve_table_ref(table_ref);
        assert!(result.is_ok());

-        let table_ref = TableReference::Full {
-            catalog: Cow::Borrowed("wrong_catalog"),
-            schema: Cow::Borrowed("public"),
-            table: Cow::Borrowed("table_name"),
-        };
+        let table_ref = TableReference::full("wrong_catalog", "public", "table_name");
        let result = table_provider.resolve_table_ref(table_ref);
        assert!(result.is_err());

-        let table_ref = TableReference::Partial {
-            schema: Cow::Borrowed("information_schema"),
-            table: Cow::Borrowed("columns"),
-        };
+        let table_ref = TableReference::partial("information_schema", "columns");
        let result = table_provider.resolve_table_ref(table_ref);
        assert!(result.is_ok());

-        let table_ref = TableReference::Full {
-            catalog: Cow::Borrowed("greptime"),
-            schema: Cow::Borrowed("information_schema"),
-            table: Cow::Borrowed("columns"),
-        };
+        let table_ref = TableReference::full("greptime", "information_schema", "columns");
        assert!(table_provider.resolve_table_ref(table_ref).is_ok());

-        let table_ref = TableReference::Full {
-            catalog: Cow::Borrowed("dummy"),
-            schema: Cow::Borrowed("information_schema"),
-            table: Cow::Borrowed("columns"),
-        };
+        let table_ref = TableReference::full("dummy", "information_schema", "columns");
        assert!(table_provider.resolve_table_ref(table_ref).is_err());

-        let table_ref = TableReference::Full {
-            catalog: Cow::Borrowed("greptime"),
-            schema: Cow::Borrowed("greptime_private"),
-            table: Cow::Borrowed("columns"),
-        };
+        let table_ref = TableReference::full("greptime", "greptime_private", "columns");
        assert!(table_provider.resolve_table_ref(table_ref).is_ok());
    }
 }
--- a/src/client/src/database.rs
+++ b/src/client/src/database.rs
@@ -37,6 +37,8 @@ use snafu::{ensure, ResultExt};
 use crate::error::{ConvertFlightDataSnafu, Error, IllegalFlightMessagesSnafu, ServerSnafu};
 use crate::{error, from_grpc_response, metrics, Client, Result, StreamInserter};

+pub const DEFAULT_LOOKBACK_STRING: &str = "5m";
+
 #[derive(Clone, Debug, Default)]
 pub struct Database {
    // The "catalog" and "schema" to be used in processing the requests at the server side.
@@ -215,6 +217,7 @@ impl Database {
                start: start.to_string(),
                end: end.to_string(),
                step: step.to_string(),
+                lookback: DEFAULT_LOOKBACK_STRING.to_string(),
            })),
        }))
        .await
--- a/src/client/src/region.rs
+++ b/src/client/src/region.rs
@@ -14,6 +14,7 @@

 use std::sync::Arc;

+use api::region::RegionResponse;
 use api::v1::region::{QueryRequest, RegionRequest};
 use api::v1::ResponseHeader;
 use arc_swap::ArcSwapOption;
@@ -23,7 +24,7 @@ use async_trait::async_trait;
 use common_error::ext::{BoxedError, ErrorExt};
 use common_error::status_code::StatusCode;
 use common_grpc::flight::{FlightDecoder, FlightMessage};
-use common_meta::datanode_manager::{Datanode, HandleResponse};
+use common_meta::datanode_manager::Datanode;
 use common_meta::error::{self as meta_error, Result as MetaResult};
 use common_recordbatch::error::ExternalSnafu;
 use common_recordbatch::{RecordBatchStreamWrapper, SendableRecordBatchStream};
@@ -46,7 +47,7 @@ pub struct RegionRequester {

 #[async_trait]
 impl Datanode for RegionRequester {
-    async fn handle(&self, request: RegionRequest) -> MetaResult<HandleResponse> {
+    async fn handle(&self, request: RegionRequest) -> MetaResult<RegionResponse> {
        self.handle_inner(request).await.map_err(|err| {
            if err.should_retry() {
                meta_error::Error::RetryLater {
@@ -165,7 +166,7 @@ impl RegionRequester {
        Ok(Box::pin(record_batch_stream))
    }

-    async fn handle_inner(&self, request: RegionRequest) -> Result<HandleResponse> {
+    async fn handle_inner(&self, request: RegionRequest) -> Result<RegionResponse> {
        let request_type = request
            .body
            .as_ref()
@@ -194,10 +195,10 @@ impl RegionRequester {

        check_response_header(&response.header)?;

-        Ok(HandleResponse::from_region_response(response))
+        Ok(RegionResponse::from_region_response(response))
    }

-    pub async fn handle(&self, request: RegionRequest) -> Result<HandleResponse> {
+    pub async fn handle(&self, request: RegionRequest) -> Result<RegionResponse> {
        self.handle_inner(request).await
    }
 }
--- a/src/cmd/Cargo.toml
+++ b/src/cmd/Cargo.toml
@@ -36,6 +36,7 @@ common-telemetry = { workspace = true, features = [
    "deadlock_detection",
 ] }
 common-time.workspace = true
+common-version.workspace = true
 common-wal.workspace = true
 config = "0.13"
 datanode.workspace = true
@@ -76,6 +77,7 @@ tikv-jemallocator = "0.5"
 common-test-util.workspace = true
 serde.workspace = true
 temp-env = "0.3"
+tempfile.workspace = true

 [target.'cfg(not(windows))'.dev-dependencies]
 rexpect = "0.5"
--- a/src/cmd/src/bin/greptime.rs
+++ b/src/cmd/src/bin/greptime.rs
@@ -22,6 +22,7 @@ use cmd::options::{CliOptions, Options};
 use cmd::{
    cli, datanode, frontend, greptimedb_cli, log_versions, metasrv, standalone, start_app, App,
 };
+use common_version::{short_version, version};

 #[derive(Parser)]
 enum SubCommand {
@@ -105,7 +106,8 @@ async fn main() -> Result<()> {

    common_telemetry::set_panic_hook();

-    let cli = greptimedb_cli();
+    let version = version!();
+    let cli = greptimedb_cli().version(version);

    let cli = SubCommand::augment_subcommands(cli);

@@ -129,7 +131,7 @@ async fn main() -> Result<()> {
        opts.node_id(),
    );

-    log_versions();
+    log_versions(version, short_version!());

    let app = subcmd.build(opts).await?;

--- a/src/cmd/src/cli.rs
+++ b/src/cmd/src/cli.rs
@@ -84,10 +84,10 @@ impl Command {
        let mut logging_opts = LoggingOptions::default();

        if let Some(dir) = &cli_options.log_dir {
-            logging_opts.dir = dir.clone();
+            logging_opts.dir.clone_from(dir);
        }

-        logging_opts.level = cli_options.log_level.clone();
+        logging_opts.level.clone_from(&cli_options.log_level);

        Ok(Options::Cli(Box::new(logging_opts)))
    }
--- a/src/cmd/src/cli/bench/metadata.rs
+++ b/src/cmd/src/cli/bench/metadata.rs
@@ -107,14 +107,11 @@ impl TableMetadataBencher {
                    .unwrap();
                let start = Instant::now();
                let table_info = table_info.unwrap();
+                let table_route = table_route.unwrap();
                let table_id = table_info.table_info.ident.table_id;
                let _ = self
                    .table_metadata_manager
-                    .delete_table_metadata(
-                        table_id,
-                        &table_info.table_name(),
-                        table_route.unwrap().region_routes().unwrap(),
-                    )
+                    .delete_table_metadata(table_id, &table_info.table_name(), &table_route)
                    .await;
                start.elapsed()
            },
@@ -140,7 +137,7 @@ impl TableMetadataBencher {
                let start = Instant::now();
                let _ = self
                    .table_metadata_manager
-                    .rename_table(table_info.unwrap(), new_table_name)
+                    .rename_table(&table_info.unwrap(), new_table_name)
                    .await;

                start.elapsed()
--- a/src/cmd/src/cli/export.rs
+++ b/src/cmd/src/cli/export.rs
@@ -226,7 +226,10 @@ impl Export {
    }

    async fn show_create_table(&self, catalog: &str, schema: &str, table: &str) -> Result<String> {
-        let sql = format!("show create table {}.{}.{}", catalog, schema, table);
+        let sql = format!(
+            r#"show create table "{}"."{}"."{}""#,
+            catalog, schema, table
+        );
        let mut client = self.client.clone();
        client.set_catalog(catalog);
        client.set_schema(schema);
@@ -273,7 +276,7 @@ impl Export {
                for (c, s, t) in table_list {
                    match self.show_create_table(&c, &s, &t).await {
                        Err(e) => {
-                            error!(e; "Failed to export table {}.{}.{}", c, s, t)
+                            error!(e; r#"Failed to export table "{}"."{}"."{}""#, c, s, t)
                        }
                        Ok(create_table) => {
                            file.write_all(create_table.as_bytes())
@@ -417,3 +420,82 @@ fn split_database(database: &str) -> Result<(String, Option<String>)> {
        Ok((catalog.to_string(), Some(schema.to_string())))
    }
 }
+
+#[cfg(test)]
+mod tests {
+    use clap::Parser;
+    use client::{Client, Database};
+    use common_catalog::consts::{DEFAULT_CATALOG_NAME, DEFAULT_SCHEMA_NAME};
+
+    use crate::error::Result;
+    use crate::options::{CliOptions, Options};
+    use crate::{cli, standalone, App};
+
+    #[tokio::test(flavor = "multi_thread")]
+    async fn test_export_create_table_with_quoted_names() -> Result<()> {
+        let output_dir = tempfile::tempdir().unwrap();
+
+        let standalone = standalone::Command::parse_from([
+            "standalone",
+            "start",
+            "--data-home",
+            &*output_dir.path().to_string_lossy(),
+        ]);
+        let Options::Standalone(standalone_opts) =
+            standalone.load_options(&CliOptions::default())?
+        else {
+            unreachable!()
+        };
+        let mut instance = standalone.build(*standalone_opts).await?;
+        instance.start().await?;
+
+        let client = Client::with_urls(["127.0.0.1:4001"]);
+        let database = Database::new(DEFAULT_CATALOG_NAME, DEFAULT_SCHEMA_NAME, client);
+        database
+            .sql(r#"CREATE DATABASE "cli.export.create_table";"#)
+            .await
+            .unwrap();
+        database
+            .sql(
+                r#"CREATE TABLE "cli.export.create_table"."a.b.c"(
+                        ts TIMESTAMP,
+                        TIME INDEX (ts)
+                    ) engine=mito;
+                "#,
+            )
+            .await
+            .unwrap();
+
+        let output_dir = tempfile::tempdir().unwrap();
+        let cli = cli::Command::parse_from([
+            "cli",
+            "export",
+            "--addr",
+            "127.0.0.1:4001",
+            "--output-dir",
+            &*output_dir.path().to_string_lossy(),
+            "--target",
+            "create-table",
+        ]);
+        let mut cli_app = cli.build().await?;
+        cli_app.start().await?;
+
+        instance.stop().await?;
+
+        let output_file = output_dir
+            .path()
+            .join("greptime-cli.export.create_table.sql");
+        let res = std::fs::read_to_string(output_file).unwrap();
+        let expect = r#"CREATE TABLE IF NOT EXISTS "a.b.c" (
+  "ts" TIMESTAMP(3) NOT NULL,
+  TIME INDEX ("ts")
+)
+
+ENGINE=mito
+;
+"#;
+        assert_eq!(res.trim(), expect.trim());
+
+        Ok(())
+    }
+}
--- a/src/cmd/src/cli/upgrade.rs
+++ b/src/cmd/src/cli/upgrade.rs
@@ -192,10 +192,10 @@ impl MigrateTableMetadata {
                let key = v1SchemaKey::parse(key_str)
                    .unwrap_or_else(|e| panic!("schema key is corrupted: {e}, key: {key_str}"));

-                Ok((key, ()))
+                Ok(key)
            }),
        );
-        while let Some((key, _)) = stream.try_next().await.context(error::IterStreamSnafu)? {
+        while let Some(key) = stream.try_next().await.context(error::IterStreamSnafu)? {
            let _ = self.migrate_schema_key(&key).await;
            keys.push(key.to_string().as_bytes().to_vec());
        }
@@ -244,10 +244,10 @@ impl MigrateTableMetadata {
                let key = v1CatalogKey::parse(key_str)
                    .unwrap_or_else(|e| panic!("catalog key is corrupted: {e}, key: {key_str}"));

-                Ok((key, ()))
+                Ok(key)
            }),
        );
-        while let Some((key, _)) = stream.try_next().await.context(error::IterStreamSnafu)? {
+        while let Some(key) = stream.try_next().await.context(error::IterStreamSnafu)? {
            let _ = self.migrate_catalog_key(&key).await;
            keys.push(key.to_string().as_bytes().to_vec());
        }
--- a/src/cmd/src/datanode.rs
+++ b/src/cmd/src/datanode.rs
@@ -139,19 +139,19 @@ impl StartCommand {
        )?;

        if let Some(dir) = &cli_options.log_dir {
-            opts.logging.dir = dir.clone();
+            opts.logging.dir.clone_from(dir);
        }

        if cli_options.log_level.is_some() {
-            opts.logging.level = cli_options.log_level.clone();
+            opts.logging.level.clone_from(&cli_options.log_level);
        }

        if let Some(addr) = &self.rpc_addr {
-            opts.rpc_addr = addr.clone();
+            opts.rpc_addr.clone_from(addr);
        }

        if self.rpc_hostname.is_some() {
-            opts.rpc_hostname = self.rpc_hostname.clone();
+            opts.rpc_hostname.clone_from(&self.rpc_hostname);
        }

        if let Some(node_id) = self.node_id {
@@ -161,7 +161,8 @@ impl StartCommand {
        if let Some(metasrv_addrs) = &self.metasrv_addr {
            opts.meta_client
                .get_or_insert_with(MetaClientOptions::default)
-                .metasrv_addrs = metasrv_addrs.clone();
+                .metasrv_addrs
+                .clone_from(metasrv_addrs);
            opts.mode = Mode::Distributed;
        }

@@ -173,7 +174,7 @@ impl StartCommand {
        }

        if let Some(data_home) = &self.data_home {
-            opts.storage.data_home = data_home.clone();
+            opts.storage.data_home.clone_from(data_home);
        }

        // `wal_dir` only affects raft-engine config.
@@ -191,7 +192,7 @@ impl StartCommand {
        }

        if let Some(http_addr) = &self.http_addr {
-            opts.http.addr = http_addr.clone();
+            opts.http.addr.clone_from(http_addr);
        }

        if let Some(http_timeout) = self.http_timeout {
--- a/src/cmd/src/frontend.rs
+++ b/src/cmd/src/frontend.rs
@@ -157,11 +157,11 @@ impl StartCommand {
        )?;

        if let Some(dir) = &cli_options.log_dir {
-            opts.logging.dir = dir.clone();
+            opts.logging.dir.clone_from(dir);
        }

        if cli_options.log_level.is_some() {
-            opts.logging.level = cli_options.log_level.clone();
+            opts.logging.level.clone_from(&cli_options.log_level);
        }

        let tls_opts = TlsOption::new(
@@ -171,7 +171,7 @@ impl StartCommand {
        );

        if let Some(addr) = &self.http_addr {
-            opts.http.addr = addr.clone()
+            opts.http.addr.clone_from(addr);
        }

        if let Some(http_timeout) = self.http_timeout {
@@ -183,24 +183,24 @@ impl StartCommand {
        }

        if let Some(addr) = &self.rpc_addr {
-            opts.grpc.addr = addr.clone()
+            opts.grpc.addr.clone_from(addr);
        }

        if let Some(addr) = &self.mysql_addr {
            opts.mysql.enable = true;
-            opts.mysql.addr = addr.clone();
+            opts.mysql.addr.clone_from(addr);
            opts.mysql.tls = tls_opts.clone();
        }

        if let Some(addr) = &self.postgres_addr {
            opts.postgres.enable = true;
-            opts.postgres.addr = addr.clone();
+            opts.postgres.addr.clone_from(addr);
            opts.postgres.tls = tls_opts;
        }

        if let Some(addr) = &self.opentsdb_addr {
            opts.opentsdb.enable = true;
-            opts.opentsdb.addr = addr.clone();
+            opts.opentsdb.addr.clone_from(addr);
        }

        if let Some(enable) = self.influxdb_enable {
@@ -210,11 +210,12 @@ impl StartCommand {
        if let Some(metasrv_addrs) = &self.metasrv_addr {
            opts.meta_client
                .get_or_insert_with(MetaClientOptions::default)
-                .metasrv_addrs = metasrv_addrs.clone();
+                .metasrv_addrs
+                .clone_from(metasrv_addrs);
            opts.mode = Mode::Distributed;
        }

-        opts.user_provider = self.user_provider.clone();
+        opts.user_provider.clone_from(&self.user_provider);

        Ok(Options::Frontend(Box::new(opts)))
    }
--- a/src/cmd/src/lib.rs
+++ b/src/cmd/src/lib.rs
@@ -64,26 +64,23 @@ pub async fn start_app(mut app: Box<dyn App>) -> error::Result<()> {
    Ok(())
 }

-pub fn log_versions() {
+/// Log the versions of the application, and the arguments passed to the cli.
+/// `version_string` should be the same as the output of cli "--version";
+/// and the `app_version` is the short version of the codes, often consist of git branch and commit.
+pub fn log_versions(version_string: &str, app_version: &str) {
    // Report app version as gauge.
    APP_VERSION
-        .with_label_values(&[short_version(), full_version()])
+        .with_label_values(&[env!("CARGO_PKG_VERSION"), app_version])
        .inc();

    // Log version and argument flags.
-    info!(
-        "short_version: {}, full_version: {}",
-        short_version(),
-        full_version()
-    );
+    info!("GreptimeDB version: {}", version_string);

    log_env_flags();
 }

 pub fn greptimedb_cli() -> clap::Command {
-    let cmd = clap::Command::new("greptimedb")
-        .version(print_version())
-        .subcommand_required(true);
+    let cmd = clap::Command::new("greptimedb").subcommand_required(true);

    #[cfg(feature = "tokio-console")]
    let cmd = cmd.arg(arg!(--"tokio-console-addr"[TOKIO_CONSOLE_ADDR]));
@@ -91,35 +88,6 @@ pub fn greptimedb_cli() -> clap::Command {
    cmd.args([arg!(--"log-dir"[LOG_DIR]), arg!(--"log-level"[LOG_LEVEL])])
 }

-fn print_version() -> &'static str {
-    concat!(
-        "\nbranch: ",
-        env!("GIT_BRANCH"),
-        "\ncommit: ",
-        env!("GIT_COMMIT"),
-        "\ndirty: ",
-        env!("GIT_DIRTY"),
-        "\nversion: ",
-        env!("CARGO_PKG_VERSION")
-    )
-}
-
-fn short_version() -> &'static str {
-    env!("CARGO_PKG_VERSION")
-}
-
-// {app_name}-{branch_name}-{commit_short}
-// The branch name (tag) of a release build should already contain the short
-// version so the full version doesn't concat the short version explicitly.
-fn full_version() -> &'static str {
-    concat!(
-        "greptimedb-",
-        env!("GIT_BRANCH"),
-        "-",
-        env!("GIT_COMMIT_SHORT")
-    )
-}
-
 fn log_env_flags() {
    info!("command line arguments");
    for argument in std::env::args() {
--- a/src/cmd/src/metasrv.rs
+++ b/src/cmd/src/metasrv.rs
@@ -17,8 +17,8 @@ use std::time::Duration;
 use async_trait::async_trait;
 use clap::Parser;
 use common_telemetry::logging;
-use meta_srv::bootstrap::MetaSrvInstance;
-use meta_srv::metasrv::MetaSrvOptions;
+use meta_srv::bootstrap::MetasrvInstance;
+use meta_srv::metasrv::MetasrvOptions;
 use snafu::ResultExt;

 use crate::error::{self, Result, StartMetaServerSnafu};
@@ -26,11 +26,11 @@ use crate::options::{CliOptions, Options};
 use crate::App;

 pub struct Instance {
-    instance: MetaSrvInstance,
+    instance: MetasrvInstance,
 }

 impl Instance {
-    fn new(instance: MetaSrvInstance) -> Self {
+    fn new(instance: MetasrvInstance) -> Self {
        Self { instance }
    }
 }
@@ -42,7 +42,7 @@ impl App for Instance {
    }

    async fn start(&mut self) -> Result<()> {
-        plugins::start_meta_srv_plugins(self.instance.plugins())
+        plugins::start_metasrv_plugins(self.instance.plugins())
            .await
            .context(StartMetaServerSnafu)?;

@@ -64,7 +64,7 @@ pub struct Command {
 }

 impl Command {
-    pub async fn build(self, opts: MetaSrvOptions) -> Result<Instance> {
+    pub async fn build(self, opts: MetasrvOptions) -> Result<Instance> {
        self.subcmd.build(opts).await
    }

@@ -79,7 +79,7 @@ enum SubCommand {
 }

 impl SubCommand {
-    async fn build(self, opts: MetaSrvOptions) -> Result<Instance> {
+    async fn build(self, opts: MetasrvOptions) -> Result<Instance> {
        match self {
            SubCommand::Start(cmd) => cmd.build(opts).await,
        }
@@ -127,30 +127,30 @@ struct StartCommand {

 impl StartCommand {
    fn load_options(&self, cli_options: &CliOptions) -> Result<Options> {
-        let mut opts: MetaSrvOptions = Options::load_layered_options(
+        let mut opts: MetasrvOptions = Options::load_layered_options(
            self.config_file.as_deref(),
            self.env_prefix.as_ref(),
-            MetaSrvOptions::env_list_keys(),
+            MetasrvOptions::env_list_keys(),
        )?;

        if let Some(dir) = &cli_options.log_dir {
-            opts.logging.dir = dir.clone();
+            opts.logging.dir.clone_from(dir);
        }

        if cli_options.log_level.is_some() {
-            opts.logging.level = cli_options.log_level.clone();
+            opts.logging.level.clone_from(&cli_options.log_level);
        }

        if let Some(addr) = &self.bind_addr {
-            opts.bind_addr = addr.clone();
+            opts.bind_addr.clone_from(addr);
        }

        if let Some(addr) = &self.server_addr {
-            opts.server_addr = addr.clone();
+            opts.server_addr.clone_from(addr);
        }

        if let Some(addr) = &self.store_addr {
-            opts.store_addr = addr.clone();
+            opts.store_addr.clone_from(addr);
        }

        if let Some(selector_type) = &self.selector {
@@ -168,7 +168,7 @@ impl StartCommand {
        }

        if let Some(http_addr) = &self.http_addr {
-            opts.http.addr = http_addr.clone();
+            opts.http.addr.clone_from(http_addr);
        }

        if let Some(http_timeout) = self.http_timeout {
@@ -176,11 +176,11 @@ impl StartCommand {
        }

        if let Some(data_home) = &self.data_home {
-            opts.data_home = data_home.clone();
+            opts.data_home.clone_from(data_home);
        }

        if !self.store_key_prefix.is_empty() {
-            opts.store_key_prefix = self.store_key_prefix.clone()
+            opts.store_key_prefix.clone_from(&self.store_key_prefix)
        }

        if let Some(max_txn_ops) = self.max_txn_ops {
@@ -193,20 +193,20 @@ impl StartCommand {
        Ok(Options::Metasrv(Box::new(opts)))
    }

-    async fn build(self, mut opts: MetaSrvOptions) -> Result<Instance> {
-        let plugins = plugins::setup_meta_srv_plugins(&mut opts)
+    async fn build(self, mut opts: MetasrvOptions) -> Result<Instance> {
+        let plugins = plugins::setup_metasrv_plugins(&mut opts)
            .await
            .context(StartMetaServerSnafu)?;

-        logging::info!("MetaSrv start command: {:#?}", self);
-        logging::info!("MetaSrv options: {:#?}", opts);
+        logging::info!("Metasrv start command: {:#?}", self);
+        logging::info!("Metasrv options: {:#?}", opts);

        let builder = meta_srv::bootstrap::metasrv_builder(&opts, plugins.clone(), None)
            .await
            .context(error::BuildMetaServerSnafu)?;
        let metasrv = builder.build().await.context(error::BuildMetaServerSnafu)?;

-        let instance = MetaSrvInstance::new(opts, plugins, metasrv)
+        let instance = MetasrvInstance::new(opts, plugins, metasrv)
            .await
            .context(error::BuildMetaServerSnafu)?;

@@ -218,6 +218,7 @@ impl StartCommand {
 mod tests {
    use std::io::Write;

+    use common_base::readable_size::ReadableSize;
    use common_test_util::temp_dir::create_named_temp_file;
    use meta_srv::selector::SelectorType;

@@ -297,6 +298,10 @@ mod tests {
                .first_heartbeat_estimate
                .as_millis()
        );
+        assert_eq!(
+            options.procedure.max_metadata_value_size,
+            Some(ReadableSize::kb(1500))
+        );
    }

    #[test]
--- a/src/cmd/src/options.rs
+++ b/src/cmd/src/options.rs
@@ -15,12 +15,12 @@
 use clap::ArgMatches;
 use common_config::KvBackendConfig;
 use common_telemetry::logging::{LoggingOptions, TracingOptions};
-use common_wal::config::MetaSrvWalConfig;
+use common_wal::config::MetasrvWalConfig;
 use config::{Config, Environment, File, FileFormat};
 use datanode::config::{DatanodeOptions, ProcedureConfig};
 use frontend::error::{Result as FeResult, TomlFormatSnafu};
 use frontend::frontend::{FrontendOptions, TomlSerializable};
-use meta_srv::metasrv::MetaSrvOptions;
+use meta_srv::metasrv::MetasrvOptions;
 use serde::{Deserialize, Serialize};
 use snafu::ResultExt;

@@ -38,7 +38,7 @@ pub struct MixOptions {
    pub frontend: FrontendOptions,
    pub datanode: DatanodeOptions,
    pub logging: LoggingOptions,
-    pub wal_meta: MetaSrvWalConfig,
+    pub wal_meta: MetasrvWalConfig,
 }

 impl From<MixOptions> for FrontendOptions {
@@ -56,7 +56,7 @@ impl TomlSerializable for MixOptions {
 pub enum Options {
    Datanode(Box<DatanodeOptions>),
    Frontend(Box<FrontendOptions>),
-    Metasrv(Box<MetaSrvOptions>),
+    Metasrv(Box<MetasrvOptions>),
    Standalone(Box<MixOptions>),
    Cli(Box<LoggingOptions>),
 }
--- a/src/cmd/src/standalone.rs
+++ b/src/cmd/src/standalone.rs
@@ -293,11 +293,11 @@ impl StartCommand {
        opts.mode = Mode::Standalone;

        if let Some(dir) = &cli_options.log_dir {
-            opts.logging.dir = dir.clone();
+            opts.logging.dir.clone_from(dir);
        }

        if cli_options.log_level.is_some() {
-            opts.logging.level = cli_options.log_level.clone();
+            opts.logging.level.clone_from(&cli_options.log_level);
        }

        let tls_opts = TlsOption::new(
@@ -307,11 +307,11 @@ impl StartCommand {
        );

        if let Some(addr) = &self.http_addr {
-            opts.http.addr = addr.clone()
+            opts.http.addr.clone_from(addr);
        }

        if let Some(data_home) = &self.data_home {
-            opts.storage.data_home = data_home.clone();
+            opts.storage.data_home.clone_from(data_home);
        }

        if let Some(addr) = &self.rpc_addr {
@@ -325,31 +325,31 @@ impl StartCommand {
                }
                .fail();
            }
-            opts.grpc.addr = addr.clone()
+            opts.grpc.addr.clone_from(addr)
        }

        if let Some(addr) = &self.mysql_addr {
            opts.mysql.enable = true;
-            opts.mysql.addr = addr.clone();
+            opts.mysql.addr.clone_from(addr);
            opts.mysql.tls = tls_opts.clone();
        }

        if let Some(addr) = &self.postgres_addr {
            opts.postgres.enable = true;
-            opts.postgres.addr = addr.clone();
+            opts.postgres.addr.clone_from(addr);
            opts.postgres.tls = tls_opts;
        }

        if let Some(addr) = &self.opentsdb_addr {
            opts.opentsdb.enable = true;
-            opts.opentsdb.addr = addr.clone();
+            opts.opentsdb.addr.clone_from(addr);
        }

        if self.influxdb_enable {
            opts.influxdb.enable = self.influxdb_enable;
        }

-        opts.user_provider = self.user_provider.clone();
+        opts.user_provider.clone_from(&self.user_provider);

        let metadata_store = opts.metadata_store.clone();
        let procedure = opts.procedure.clone();
--- a/src/common/catalog/src/consts.rs
+++ b/src/common/catalog/src/consts.rs
@@ -86,6 +86,8 @@ pub const INFORMATION_SCHEMA_RUNTIME_METRICS_TABLE_ID: u32 = 27;
 pub const INFORMATION_SCHEMA_PARTITIONS_TABLE_ID: u32 = 28;
 /// id for information_schema.REGION_PEERS
 pub const INFORMATION_SCHEMA_REGION_PEERS_TABLE_ID: u32 = 29;
+/// id for information_schema.columns
+pub const INFORMATION_SCHEMA_TABLE_CONSTRAINTS_TABLE_ID: u32 = 30;
 /// ----- End of information_schema tables -----

 pub const MITO_ENGINE: &str = "mito";
--- a/src/common/datasource/Cargo.toml
+++ b/src/common/datasource/Cargo.toml
@@ -30,7 +30,7 @@ derive_builder.workspace = true
 futures.workspace = true
 lazy_static.workspace = true
 object-store.workspace = true
-orc-rust = "0.2"
+orc-rust = { git = "https://github.com/MichaelScofield/orc-rs.git", rev = "17347f5f084ac937863317df882218055c4ea8c1" }
 parquet.workspace = true
 paste = "1.0"
 regex = "1.7"
--- a/src/common/datasource/src/buffered_writer.rs
+++ b/src/common/datasource/src/buffered_writer.rs
@@ -60,12 +60,6 @@ impl<
            .context(error::BufferedWriterClosedSnafu)?;
        let metadata = encoder.close().await?;

-        // Use `rows_written` to keep a track of if any rows have been written.
-        // If no row's been written, then we can simply close the underlying
-        // writer without flush so that no file will be actually created.
-        if self.rows_written != 0 {
-            self.bytes_written += self.try_flush(true).await?;
-        }
        // It's important to shut down! flushes all pending writes
        self.close_inner_writer().await?;
        Ok((metadata, self.bytes_written))
@@ -79,8 +73,15 @@ impl<
        Fut: Future<Output = Result<T>>,
    > LazyBufferedWriter<T, U, F>
 {
-    /// Closes the writer without flushing the buffer data.
+    /// Closes the writer and flushes the buffer data.
    pub async fn close_inner_writer(&mut self) -> Result<()> {
+        // Use `rows_written` to keep a track of if any rows have been written.
+        // If no row's been written, then we can simply close the underlying
+        // writer without flush so that no file will be actually created.
+        if self.rows_written != 0 {
+            self.bytes_written += self.try_flush(true).await?;
+        }
+
        if let Some(writer) = &mut self.writer {
            writer.shutdown().await.context(error::AsyncWriteSnafu)?;
        }
@@ -117,7 +118,7 @@ impl<
        Ok(())
    }

-    pub async fn try_flush(&mut self, all: bool) -> Result<u64> {
+    async fn try_flush(&mut self, all: bool) -> Result<u64> {
        let mut bytes_written: u64 = 0;

        // Once buffered data size reaches threshold, split the data in chunks (typically 4MB)
--- a/src/common/datasource/src/file_format.rs
+++ b/src/common/datasource/src/file_format.rs
@@ -213,10 +213,6 @@ pub async fn stream_to_file<T: DfRecordBatchEncoder, U: Fn(SharedBuffer) -> T>(
        writer.write(&batch).await?;
        rows += batch.num_rows();
    }
-
-    // Flushes all pending writes
-    let _ = writer.try_flush(true).await?;
    writer.close_inner_writer().await?;
-
    Ok(rows)
 }
--- a/src/common/datasource/src/file_format/csv.rs
+++ b/src/common/datasource/src/file_format/csv.rs
@@ -117,7 +117,7 @@ impl CsvConfig {
        let mut builder = csv::ReaderBuilder::new(self.file_schema.clone())
            .with_delimiter(self.delimiter)
            .with_batch_size(self.batch_size)
-            .has_header(self.has_header);
+            .with_header(self.has_header);

        if let Some(proj) = &self.file_projection {
            builder = builder.with_projection(proj.clone());
--- a/src/common/datasource/src/file_format/parquet.rs
+++ b/src/common/datasource/src/file_format/parquet.rs
@@ -215,10 +215,7 @@ impl BufferedWriter {

    /// Write a record batch to stream writer.
    pub async fn write(&mut self, arrow_batch: &RecordBatch) -> error::Result<()> {
-        self.inner.write(arrow_batch).await?;
-        self.inner.try_flush(false).await?;
-
-        Ok(())
+        self.inner.write(arrow_batch).await
    }

    /// Close parquet writer.
--- a/src/common/datasource/src/file_format/tests.rs
+++ b/src/common/datasource/src/file_format/tests.rs
@@ -19,6 +19,7 @@ use std::vec;

 use common_test_util::find_workspace_path;
 use datafusion::assert_batches_eq;
+use datafusion::config::TableParquetOptions;
 use datafusion::datasource::physical_plan::{FileOpener, FileScanConfig, FileStream, ParquetExec};
 use datafusion::execution::context::TaskContext;
 use datafusion::physical_plan::metrics::ExecutionPlanMetricsSet;
@@ -166,7 +167,7 @@ async fn test_parquet_exec() {
        .to_string();
    let base_config = scan_config(schema.clone(), None, path);

-    let exec = ParquetExec::new(base_config, None, None)
+    let exec = ParquetExec::new(base_config, None, None, TableParquetOptions::default())
        .with_parquet_file_reader_factory(Arc::new(DefaultParquetFileReaderFactory::new(store)));

    let ctx = SessionContext::new();
--- a/src/common/datasource/src/test_util.rs
+++ b/src/common/datasource/src/test_util.rs
@@ -16,6 +16,7 @@ use std::sync::Arc;

 use arrow_schema::{DataType, Field, Schema, SchemaRef};
 use common_test_util::temp_dir::{create_temp_dir, TempDir};
+use datafusion::common::Statistics;
 use datafusion::datasource::listing::PartitionedFile;
 use datafusion::datasource::object_store::ObjectStoreUrl;
 use datafusion::datasource::physical_plan::{FileScanConfig, FileStream};
@@ -72,17 +73,16 @@ pub fn test_basic_schema() -> SchemaRef {
 pub fn scan_config(file_schema: SchemaRef, limit: Option<usize>, filename: &str) -> FileScanConfig {
    // object_store only recognize the Unix style path, so make it happy.
    let filename = &filename.replace('\\', "/");
-
+    let statistics = Statistics::new_unknown(file_schema.as_ref());
    FileScanConfig {
        object_store_url: ObjectStoreUrl::parse("empty://").unwrap(), // won't be used
        file_schema,
        file_groups: vec![vec![PartitionedFile::new(filename.to_string(), 10)]],
-        statistics: Default::default(),
+        statistics,
        projection: None,
        limit,
        table_partition_cols: vec![],
        output_ordering: vec![],
-        infinite_source: false,
    }
 }

--- a/src/common/error/src/status_code.rs
+++ b/src/common/error/src/status_code.rs
@@ -59,6 +59,7 @@ pub enum StatusCode {
    RegionNotFound = 4005,
    RegionAlreadyExists = 4006,
    RegionReadonly = 4007,
+    /// Region is not in a proper state to handle specific request.
    RegionNotReady = 4008,
    // If mutually exclusive operations are reached at the same time,
    // only one can be executed, another one will get region busy.
--- a/src/common/function/src/scalars/aggregate/diff.rs
+++ b/src/common/function/src/scalars/aggregate/diff.rs
@@ -56,7 +56,7 @@ where
            .map(|&n| n.into())
            .collect::<Vec<Value>>();
        Ok(vec![Value::List(ListValue::new(
-            Some(Box::new(nums)),
+            nums,
            I::LogicalType::build_data_type(),
        ))])
    }
@@ -120,10 +120,7 @@ where
                O::from_native(native).into()
            })
            .collect::<Vec<Value>>();
-        let diff = Value::List(ListValue::new(
-            Some(Box::new(diff)),
-            O::LogicalType::build_data_type(),
-        ));
+        let diff = Value::List(ListValue::new(diff, O::LogicalType::build_data_type()));
        Ok(diff)
    }
 }
@@ -218,10 +215,7 @@ mod test {
        let values = vec![Value::from(2_i64), Value::from(1_i64)];
        diff.update_batch(&v).unwrap();
        assert_eq!(
-            Value::List(ListValue::new(
-                Some(Box::new(values)),
-                ConcreteDataType::int64_datatype()
-            )),
+            Value::List(ListValue::new(values, ConcreteDataType::int64_datatype())),
            diff.evaluate().unwrap()
        );

@@ -236,10 +230,7 @@ mod test {
        let values = vec![Value::from(5_i64), Value::from(1_i64)];
        diff.update_batch(&v).unwrap();
        assert_eq!(
-            Value::List(ListValue::new(
-                Some(Box::new(values)),
-                ConcreteDataType::int64_datatype()
-            )),
+            Value::List(ListValue::new(values, ConcreteDataType::int64_datatype())),
            diff.evaluate().unwrap()
        );

@@ -252,10 +243,7 @@ mod test {
        let values = vec![Value::from(0_i64), Value::from(0_i64), Value::from(0_i64)];
        diff.update_batch(&v).unwrap();
        assert_eq!(
-            Value::List(ListValue::new(
-                Some(Box::new(values)),
-                ConcreteDataType::int64_datatype()
-            )),
+            Value::List(ListValue::new(values, ConcreteDataType::int64_datatype())),
            diff.evaluate().unwrap()
        );
    }
--- a/src/common/function/src/scalars/aggregate/percentile.rs
+++ b/src/common/function/src/scalars/aggregate/percentile.rs
@@ -104,10 +104,7 @@ where
            .map(|&n| n.into())
            .collect::<Vec<Value>>();
        Ok(vec![
-            Value::List(ListValue::new(
-                Some(Box::new(nums)),
-                T::LogicalType::build_data_type(),
-            )),
+            Value::List(ListValue::new(nums, T::LogicalType::build_data_type())),
            self.p.into(),
        ])
    }
--- a/src/common/function/src/scalars/aggregate/polyval.rs
+++ b/src/common/function/src/scalars/aggregate/polyval.rs
@@ -72,10 +72,7 @@ where
            .map(|&n| n.into())
            .collect::<Vec<Value>>();
        Ok(vec![
-            Value::List(ListValue::new(
-                Some(Box::new(nums)),
-                T::LogicalType::build_data_type(),
-            )),
+            Value::List(ListValue::new(nums, T::LogicalType::build_data_type())),
            self.x.into(),
        ])
    }
--- a/src/common/function/src/scalars/aggregate/scipy_stats_norm_cdf.rs
+++ b/src/common/function/src/scalars/aggregate/scipy_stats_norm_cdf.rs
@@ -56,10 +56,7 @@ where
            .map(|&x| x.into())
            .collect::<Vec<Value>>();
        Ok(vec![
-            Value::List(ListValue::new(
-                Some(Box::new(nums)),
-                T::LogicalType::build_data_type(),
-            )),
+            Value::List(ListValue::new(nums, T::LogicalType::build_data_type())),
            self.x.into(),
        ])
    }
--- a/src/common/function/src/scalars/aggregate/scipy_stats_norm_pdf.rs
+++ b/src/common/function/src/scalars/aggregate/scipy_stats_norm_pdf.rs
@@ -56,10 +56,7 @@ where
            .map(|&x| x.into())
            .collect::<Vec<Value>>();
        Ok(vec![
-            Value::List(ListValue::new(
-                Some(Box::new(nums)),
-                T::LogicalType::build_data_type(),
-            )),
+            Value::List(ListValue::new(nums, T::LogicalType::build_data_type())),
            self.x.into(),
        ])
    }
--- a/src/common/function/src/scalars/math.rs
+++ b/src/common/function/src/scalars/math.rs
@@ -77,7 +77,7 @@ impl Function for RangeFunction {
    /// `range_fn` will never been used. As long as a legal signature is returned, the specific content of the signature does not matter.
    /// In fact, the arguments loaded by `range_fn` are very complicated, and it is difficult to use `Signature` to describe
    fn signature(&self) -> Signature {
-        Signature::any(0, Volatility::Immutable)
+        Signature::variadic_any(Volatility::Immutable)
    }

    fn eval(&self, _func_ctx: FunctionContext, _columns: &[VectorRef]) -> Result<VectorRef> {
--- a/src/common/grpc/src/lib.rs
+++ b/src/common/grpc/src/lib.rs
@@ -15,7 +15,7 @@
 pub mod channel_manager;
 pub mod error;
 pub mod flight;
+pub mod precision;
 pub mod select;
-pub mod writer;

 pub use error::Error;
--- a/src/common/grpc/src/precision.rs
+++ b/src/common/grpc/src/precision.rs
@@ -0,0 +1,141 @@
+// Copyright 2023 Greptime Team
+//
+// Licensed under the Apache License, Version 2.0 (the "License");
+// you may not use this file except in compliance with the License.
+// You may obtain a copy of the License at
+//
+//     http://www.apache.org/licenses/LICENSE-2.0
+//
+// Unless required by applicable law or agreed to in writing, software
+// distributed under the License is distributed on an "AS IS" BASIS,
+// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+// See the License for the specific language governing permissions and
+// limitations under the License.
+
+use std::fmt::Display;
+
+use common_time::timestamp::TimeUnit;
+
+use crate::Error;
+
+/// Precision represents the precision of a timestamp.
+/// It is used to convert timestamps between different precisions.
+#[derive(Debug, Clone, Copy, PartialEq, Eq)]
+pub enum Precision {
+    Nanosecond,
+    Microsecond,
+    Millisecond,
+    Second,
+    Minute,
+    Hour,
+}
+
+impl Precision {
+    pub fn to_nanos(&self, amount: i64) -> Option<i64> {
+        match self {
+            Precision::Nanosecond => Some(amount),
+            Precision::Microsecond => amount.checked_mul(1_000),
+            Precision::Millisecond => amount.checked_mul(1_000_000),
+            Precision::Second => amount.checked_mul(1_000_000_000),
+            Precision::Minute => amount
+                .checked_mul(60)
+                .and_then(|a| a.checked_mul(1_000_000_000)),
+            Precision::Hour => amount
+                .checked_mul(3600)
+                .and_then(|a| a.checked_mul(1_000_000_000)),
+        }
+    }
+
+    pub fn to_millis(&self, amount: i64) -> Option<i64> {
+        match self {
+            Precision::Nanosecond => amount.checked_div(1_000_000),
+            Precision::Microsecond => amount.checked_div(1_000),
+            Precision::Millisecond => Some(amount),
+            Precision::Second => amount.checked_mul(1_000),
+            Precision::Minute => amount.checked_mul(60_000),
+            Precision::Hour => amount.checked_mul(3_600_000),
+        }
+    }
+}
+
+impl Display for Precision {
+    fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result {
+        match self {
+            Precision::Nanosecond => write!(f, "Precision::Nanosecond"),
+            Precision::Microsecond => write!(f, "Precision::Microsecond"),
+            Precision::Millisecond => write!(f, "Precision::Millisecond"),
+            Precision::Second => write!(f, "Precision::Second"),
+            Precision::Minute => write!(f, "Precision::Minute"),
+            Precision::Hour => write!(f, "Precision::Hour"),
+        }
+    }
+}
+
+impl TryFrom<Precision> for TimeUnit {
+    type Error = Error;
+
+    fn try_from(precision: Precision) -> Result<Self, Self::Error> {
+        Ok(match precision {
+            Precision::Second => TimeUnit::Second,
+            Precision::Millisecond => TimeUnit::Millisecond,
+            Precision::Microsecond => TimeUnit::Microsecond,
+            Precision::Nanosecond => TimeUnit::Nanosecond,
+            _ => {
+                return Err(Error::NotSupported {
+                    feat: format!("convert {precision} into TimeUnit"),
+                })
+            }
+        })
+    }
+}
+
+#[cfg(test)]
+mod tests {
+    use crate::precision::Precision;
+
+    #[test]
+    fn test_to_nanos() {
+        assert_eq!(Precision::Nanosecond.to_nanos(1).unwrap(), 1);
+        assert_eq!(Precision::Microsecond.to_nanos(1).unwrap(), 1_000);
+        assert_eq!(Precision::Millisecond.to_nanos(1).unwrap(), 1_000_000);
+        assert_eq!(Precision::Second.to_nanos(1).unwrap(), 1_000_000_000);
+        assert_eq!(Precision::Minute.to_nanos(1).unwrap(), 60 * 1_000_000_000);
+        assert_eq!(
+            Precision::Hour.to_nanos(1).unwrap(),
+            60 * 60 * 1_000_000_000
+        );
+    }
+
+    #[test]
+    fn test_to_millis() {
+        assert_eq!(Precision::Nanosecond.to_millis(1_000_000).unwrap(), 1);
+        assert_eq!(Precision::Microsecond.to_millis(1_000).unwrap(), 1);
+        assert_eq!(Precision::Millisecond.to_millis(1).unwrap(), 1);
+        assert_eq!(Precision::Second.to_millis(1).unwrap(), 1_000);
+        assert_eq!(Precision::Minute.to_millis(1).unwrap(), 60 * 1_000);
+        assert_eq!(Precision::Hour.to_millis(1).unwrap(), 60 * 60 * 1_000);
+    }
+
+    #[test]
+    fn test_to_nanos_basic() {
+        assert_eq!(Precision::Second.to_nanos(1), Some(1_000_000_000));
+        assert_eq!(Precision::Minute.to_nanos(1), Some(60 * 1_000_000_000));
+    }
+
+    #[test]
+    fn test_to_millis_basic() {
+        assert_eq!(Precision::Second.to_millis(1), Some(1_000));
+        assert_eq!(Precision::Minute.to_millis(1), Some(60_000));
+    }
+
+    #[test]
+    fn test_to_nanos_overflow() {
+        assert_eq!(Precision::Hour.to_nanos(i64::MAX / 100), None);
+    }
+
+    #[test]
+    fn test_zero_input() {
+        assert_eq!(Precision::Second.to_nanos(0), Some(0));
+        assert_eq!(Precision::Minute.to_millis(0), Some(0));
+    }
+}
--- a/src/common/grpc/src/writer.rs
+++ b/src/common/grpc/src/writer.rs
@@ -1,441 +0,0 @@
-// Copyright 2023 Greptime Team
-//
-// Licensed under the Apache License, Version 2.0 (the "License");
-// you may not use this file except in compliance with the License.
-// You may obtain a copy of the License at
-//
-//     http://www.apache.org/licenses/LICENSE-2.0
-//
-// Unless required by applicable law or agreed to in writing, software
-// distributed under the License is distributed on an "AS IS" BASIS,
-// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
-// See the License for the specific language governing permissions and
-// limitations under the License.
-
-use std::collections::HashMap;
-use std::fmt::Display;
-
-use api::helper::values_with_capacity;
-use api::v1::{Column, ColumnDataType, ColumnDataTypeExtension, SemanticType};
-use common_base::BitVec;
-use common_time::timestamp::TimeUnit;
-use snafu::ensure;
-
-use crate::error::{Result, TypeMismatchSnafu};
-use crate::Error;
-
-type ColumnName = String;
-
-type RowCount = u32;
-
-// TODO(fys): will remove in the future.
-#[derive(Default)]
-pub struct LinesWriter {
-    column_name_index: HashMap<ColumnName, usize>,
-    null_masks: Vec<BitVec>,
-    batch: (Vec<Column>, RowCount),
-    lines: usize,
-}
-
-impl LinesWriter {
-    pub fn with_lines(lines: usize) -> Self {
-        Self {
-            lines,
-            ..Default::default()
-        }
-    }
-
-    pub fn write_ts(&mut self, column_name: &str, value: (i64, Precision)) -> Result<()> {
-        let (idx, column) = self.mut_column(
-            column_name,
-            ColumnDataType::TimestampMillisecond,
-            SemanticType::Timestamp,
-            None,
-        );
-        ensure!(
-            column.datatype == ColumnDataType::TimestampMillisecond as i32,
-            TypeMismatchSnafu {
-                column_name,
-                expected: "timestamp",
-                actual: format!("{:?}", column.datatype)
-            }
-        );
-        // It is safe to use unwrap here, because values has been initialized in mut_column()
-        let values = column.values.as_mut().unwrap();
-        values
-            .timestamp_millisecond_values
-            .push(to_ms_ts(value.1, value.0));
-        self.null_masks[idx].push(false);
-        Ok(())
-    }
-
-    pub fn write_tag(&mut self, column_name: &str, value: &str) -> Result<()> {
-        let (idx, column) =
-            self.mut_column(column_name, ColumnDataType::String, SemanticType::Tag, None);
-        ensure!(
-            column.datatype == ColumnDataType::String as i32,
-            TypeMismatchSnafu {
-                column_name,
-                expected: "string",
-                actual: format!("{:?}", column.datatype)
-            }
-        );
-        // It is safe to use unwrap here, because values has been initialized in mut_column()
-        let values = column.values.as_mut().unwrap();
-        values.string_values.push(value.to_string());
-        self.null_masks[idx].push(false);
-        Ok(())
-    }
-
-    pub fn write_u64(&mut self, column_name: &str, value: u64) -> Result<()> {
-        let (idx, column) = self.mut_column(
-            column_name,
-            ColumnDataType::Uint64,
-            SemanticType::Field,
-            None,
-        );
-        ensure!(
-            column.datatype == ColumnDataType::Uint64 as i32,
-            TypeMismatchSnafu {
-                column_name,
-                expected: "u64",
-                actual: format!("{:?}", column.datatype)
-            }
-        );
-        // It is safe to use unwrap here, because values has been initialized in mut_column()
-        let values = column.values.as_mut().unwrap();
-        values.u64_values.push(value);
-        self.null_masks[idx].push(false);
-        Ok(())
-    }
-
-    pub fn write_i64(&mut self, column_name: &str, value: i64) -> Result<()> {
-        let (idx, column) = self.mut_column(
-            column_name,
-            ColumnDataType::Int64,
-            SemanticType::Field,
-            None,
-        );
-        ensure!(
-            column.datatype == ColumnDataType::Int64 as i32,
-            TypeMismatchSnafu {
-                column_name,
-                expected: "i64",
-                actual: format!("{:?}", column.datatype)
-            }
-        );
-        // It is safe to use unwrap here, because values has been initialized in mut_column()
-        let values = column.values.as_mut().unwrap();
-        values.i64_values.push(value);
-        self.null_masks[idx].push(false);
-        Ok(())
-    }
-
-    pub fn write_f64(&mut self, column_name: &str, value: f64) -> Result<()> {
-        let (idx, column) = self.mut_column(
-            column_name,
-            ColumnDataType::Float64,
-            SemanticType::Field,
-            None,
-        );
-        ensure!(
-            column.datatype == ColumnDataType::Float64 as i32,
-            TypeMismatchSnafu {
-                column_name,
-                expected: "f64",
-                actual: format!("{:?}", column.datatype)
-            }
-        );
-        // It is safe to use unwrap here, because values has been initialized in mut_column()
-        let values = column.values.as_mut().unwrap();
-        values.f64_values.push(value);
-        self.null_masks[idx].push(false);
-        Ok(())
-    }
-
-    pub fn write_string(&mut self, column_name: &str, value: &str) -> Result<()> {
-        let (idx, column) = self.mut_column(
-            column_name,
-            ColumnDataType::String,
-            SemanticType::Field,
-            None,
-        );
-        ensure!(
-            column.datatype == ColumnDataType::String as i32,
-            TypeMismatchSnafu {
-                column_name,
-                expected: "string",
-                actual: format!("{:?}", column.datatype)
-            }
-        );
-        // It is safe to use unwrap here, because values has been initialized in mut_column()
-        let values = column.values.as_mut().unwrap();
-        values.string_values.push(value.to_string());
-        self.null_masks[idx].push(false);
-        Ok(())
-    }
-
-    pub fn write_bool(&mut self, column_name: &str, value: bool) -> Result<()> {
-        let (idx, column) = self.mut_column(
-            column_name,
-            ColumnDataType::Boolean,
-            SemanticType::Field,
-            None,
-        );
-        ensure!(
-            column.datatype == ColumnDataType::Boolean as i32,
-            TypeMismatchSnafu {
-                column_name,
-                expected: "boolean",
-                actual: format!("{:?}", column.datatype)
-            }
-        );
-        // It is safe to use unwrap here, because values has been initialized in mut_column()
-        let values = column.values.as_mut().unwrap();
-        values.bool_values.push(value);
-        self.null_masks[idx].push(false);
-        Ok(())
-    }
-
-    pub fn commit(&mut self) {
-        let batch = &mut self.batch;
-        batch.1 += 1;
-
-        for i in 0..batch.0.len() {
-            let null_mask = &mut self.null_masks[i];
-            if batch.1 as usize > null_mask.len() {
-                null_mask.push(true);
-            }
-        }
-    }
-
-    pub fn finish(mut self) -> (Vec<Column>, RowCount) {
-        let null_masks = self.null_masks;
-        for (i, null_mask) in null_masks.into_iter().enumerate() {
-            let columns = &mut self.batch.0;
-            columns[i].null_mask = null_mask.into_vec();
-        }
-        self.batch
-    }
-
-    fn mut_column(
-        &mut self,
-        column_name: &str,
-        datatype: ColumnDataType,
-        semantic_type: SemanticType,
-        datatype_extension: Option<ColumnDataTypeExtension>,
-    ) -> (usize, &mut Column) {
-        let column_names = &mut self.column_name_index;
-        let column_idx = match column_names.get(column_name) {
-            Some(i) => *i,
-            None => {
-                let new_idx = column_names.len();
-                let batch = &mut self.batch;
-                let to_insert = self.lines;
-                let mut null_mask = BitVec::with_capacity(to_insert);
-                null_mask.extend(BitVec::repeat(true, batch.1 as usize));
-                self.null_masks.push(null_mask);
-                batch.0.push(Column {
-                    column_name: column_name.to_string(),
-                    semantic_type: semantic_type.into(),
-                    values: Some(values_with_capacity(datatype, to_insert)),
-                    datatype: datatype as i32,
-                    null_mask: Vec::default(),
-                    datatype_extension,
-                });
-                let _ = column_names.insert(column_name.to_string(), new_idx);
-                new_idx
-            }
-        };
-        (column_idx, &mut self.batch.0[column_idx])
-    }
-}
-
-pub fn to_ms_ts(p: Precision, ts: i64) -> i64 {
-    match p {
-        Precision::Nanosecond => ts / 1_000_000,
-        Precision::Microsecond => ts / 1000,
-        Precision::Millisecond => ts,
-        Precision::Second => ts * 1000,
-        Precision::Minute => ts * 1000 * 60,
-        Precision::Hour => ts * 1000 * 60 * 60,
-    }
-}
-
-#[derive(Debug, Clone, Copy, PartialEq, Eq)]
-pub enum Precision {
-    Nanosecond,
-    Microsecond,
-    Millisecond,
-    Second,
-    Minute,
-    Hour,
-}
-
-impl Display for Precision {
-    fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result {
-        match self {
-            Precision::Nanosecond => write!(f, "Precision::Nanosecond"),
-            Precision::Microsecond => write!(f, "Precision::Microsecond"),
-            Precision::Millisecond => write!(f, "Precision::Millisecond"),
-            Precision::Second => write!(f, "Precision::Second"),
-            Precision::Minute => write!(f, "Precision::Minute"),
-            Precision::Hour => write!(f, "Precision::Hour"),
-        }
-    }
-}
-
-impl TryFrom<Precision> for TimeUnit {
-    type Error = Error;
-
-    fn try_from(precision: Precision) -> std::result::Result<Self, Self::Error> {
-        Ok(match precision {
-            Precision::Second => TimeUnit::Second,
-            Precision::Millisecond => TimeUnit::Millisecond,
-            Precision::Microsecond => TimeUnit::Microsecond,
-            Precision::Nanosecond => TimeUnit::Nanosecond,
-            _ => {
-                return Err(Error::NotSupported {
-                    feat: format!("convert {precision} into TimeUnit"),
-                })
-            }
-        })
-    }
-}
-
-#[cfg(test)]
-mod tests {
-    use api::v1::{ColumnDataType, SemanticType};
-    use common_base::BitVec;
-
-    use super::LinesWriter;
-    use crate::writer::{to_ms_ts, Precision};
-
-    #[test]
-    fn test_lines_writer() {
-        let mut writer = LinesWriter::with_lines(3);
-
-        writer.write_tag("host", "host1").unwrap();
-        writer.write_f64("cpu", 0.5).unwrap();
-        writer.write_f64("memory", 0.4).unwrap();
-        writer.write_string("name", "name1").unwrap();
-        writer
-            .write_ts("ts", (101011000, Precision::Millisecond))
-            .unwrap();
-        writer.commit();
-
-        writer.write_tag("host", "host2").unwrap();
-        writer
-            .write_ts("ts", (102011001, Precision::Millisecond))
-            .unwrap();
-        writer.write_bool("enable_reboot", true).unwrap();
-        writer.write_u64("year_of_service", 2).unwrap();
-        writer.write_i64("temperature", 4).unwrap();
-        writer.commit();
-
-        writer.write_tag("host", "host3").unwrap();
-        writer.write_f64("cpu", 0.4).unwrap();
-        writer.write_u64("cpu_core_num", 16).unwrap();
-        writer
-            .write_ts("ts", (103011002, Precision::Millisecond))
-            .unwrap();
-        writer.commit();
-
-        let insert_batch = writer.finish();
-        assert_eq!(3, insert_batch.1);
-
-        let columns = insert_batch.0;
-        assert_eq!(9, columns.len());
-
-        let column = &columns[0];
-        assert_eq!("host", columns[0].column_name);
-        assert_eq!(ColumnDataType::String as i32, column.datatype);
-        assert_eq!(SemanticType::Tag as i32, column.semantic_type);
-        assert_eq!(
-            vec!["host1", "host2", "host3"],
-            column.values.as_ref().unwrap().string_values
-        );
-        verify_null_mask(&column.null_mask, vec![false, false, false]);
-
-        let column = &columns[1];
-        assert_eq!("cpu", column.column_name);
-        assert_eq!(ColumnDataType::Float64 as i32, column.datatype);
-        assert_eq!(SemanticType::Field as i32, column.semantic_type);
-        assert_eq!(vec![0.5, 0.4], column.values.as_ref().unwrap().f64_values);
-        verify_null_mask(&column.null_mask, vec![false, true, false]);
-
-        let column = &columns[2];
-        assert_eq!("memory", column.column_name);
-        assert_eq!(ColumnDataType::Float64 as i32, column.datatype);
-        assert_eq!(SemanticType::Field as i32, column.semantic_type);
-        assert_eq!(vec![0.4], column.values.as_ref().unwrap().f64_values);
-        verify_null_mask(&column.null_mask, vec![false, true, true]);
-
-        let column = &columns[3];
-        assert_eq!("name", column.column_name);
-        assert_eq!(ColumnDataType::String as i32, column.datatype);
-        assert_eq!(SemanticType::Field as i32, column.semantic_type);
-        assert_eq!(vec!["name1"], column.values.as_ref().unwrap().string_values);
-        verify_null_mask(&column.null_mask, vec![false, true, true]);
-
-        let column = &columns[4];
-        assert_eq!("ts", column.column_name);
-        assert_eq!(ColumnDataType::TimestampMillisecond as i32, column.datatype);
-        assert_eq!(SemanticType::Timestamp as i32, column.semantic_type);
-        assert_eq!(
-            vec![101011000, 102011001, 103011002],
-            column.values.as_ref().unwrap().timestamp_millisecond_values
-        );
-        verify_null_mask(&column.null_mask, vec![false, false, false]);
-
-        let column = &columns[5];
-        assert_eq!("enable_reboot", column.column_name);
-        assert_eq!(ColumnDataType::Boolean as i32, column.datatype);
-        assert_eq!(SemanticType::Field as i32, column.semantic_type);
-        assert_eq!(vec![true], column.values.as_ref().unwrap().bool_values);
-        verify_null_mask(&column.null_mask, vec![true, false, true]);
-
-        let column = &columns[6];
-        assert_eq!("year_of_service", column.column_name);
-        assert_eq!(ColumnDataType::Uint64 as i32, column.datatype);
-        assert_eq!(SemanticType::Field as i32, column.semantic_type);
-        assert_eq!(vec![2], column.values.as_ref().unwrap().u64_values);
-        verify_null_mask(&column.null_mask, vec![true, false, true]);
-
-        let column = &columns[7];
-        assert_eq!("temperature", column.column_name);
-        assert_eq!(ColumnDataType::Int64 as i32, column.datatype);
-        assert_eq!(SemanticType::Field as i32, column.semantic_type);
-        assert_eq!(vec![4], column.values.as_ref().unwrap().i64_values);
-        verify_null_mask(&column.null_mask, vec![true, false, true]);
-
-        let column = &columns[8];
-        assert_eq!("cpu_core_num", column.column_name);
-        assert_eq!(ColumnDataType::Uint64 as i32, column.datatype);
-        assert_eq!(SemanticType::Field as i32, column.semantic_type);
-        assert_eq!(vec![16], column.values.as_ref().unwrap().u64_values);
-        verify_null_mask(&column.null_mask, vec![true, true, false]);
-    }
-
-    fn verify_null_mask(data: &[u8], expected: Vec<bool>) {
-        let bitvec = BitVec::from_slice(data);
-        for (idx, b) in expected.iter().enumerate() {
-            assert_eq!(b, bitvec.get(idx).unwrap())
-        }
-    }
-
-    #[test]
-    fn test_to_ms() {
-        assert_eq!(100, to_ms_ts(Precision::Nanosecond, 100110000));
-        assert_eq!(100110, to_ms_ts(Precision::Microsecond, 100110000));
-        assert_eq!(100110000, to_ms_ts(Precision::Millisecond, 100110000));
-        assert_eq!(
-            100110000 * 1000 * 60,
-            to_ms_ts(Precision::Minute, 100110000)
-        );
-        assert_eq!(
-            100110000 * 1000 * 60 * 60,
-            to_ms_ts(Precision::Hour, 100110000)
-        );
-    }
-}
--- a/src/common/macro/src/range_fn.rs
+++ b/src/common/macro/src/range_fn.rs
@@ -119,15 +119,17 @@ fn build_struct(
            }

            pub fn scalar_udf() -> ScalarUDF {
-                ScalarUDF {
-                    name: Self::name().to_string(),
-                    signature: Signature::new(
+                // TODO(LFC): Use the new Datafusion UDF impl.
+                #[allow(deprecated)]
+                ScalarUDF::new(
+                    Self::name(),
+                    &Signature::new(
                        TypeSignature::Exact(Self::input_type()),
                        Volatility::Immutable,
                    ),
-                    return_type: Arc::new(|_| Ok(Arc::new(Self::return_type()))),
-                    fun: Arc::new(Self::calc),
-                }
+                    &(Arc::new(|_: &_| Ok(Arc::new(Self::return_type()))) as _),
+                    &(Arc::new(Self::calc) as _),
+                )
            }

            fn input_type() -> Vec<DataType> {
--- a/src/common/meta/src/cache_invalidator.rs
+++ b/src/common/meta/src/cache_invalidator.rs
@@ -18,6 +18,7 @@ use tokio::sync::RwLock;

 use crate::error::Result;
 use crate::instruction::CacheIdent;
+use crate::key::schema_name::SchemaNameKey;
 use crate::key::table_info::TableInfoKey;
 use crate::key::table_name::TableNameKey;
 use crate::key::table_route::TableRouteKey;
@@ -107,6 +108,10 @@ where
                    let key: TableNameKey = (&table_name).into();
                    self.invalidate_key(&key.as_raw_key()).await
                }
+                CacheIdent::SchemaName(schema_name) => {
+                    let key: SchemaNameKey = (&schema_name).into();
+                    self.invalidate_key(&key.as_raw_key()).await;
+                }
            }
        }
        Ok(())
--- a/src/common/meta/src/cluster.rs
+++ b/src/common/meta/src/cluster.rs
@@ -0,0 +1,302 @@
+// Copyright 2023 Greptime Team
+//
+// Licensed under the Apache License, Version 2.0 (the "License");
+// you may not use this file except in compliance with the License.
+// You may obtain a copy of the License at
+//
+//     http://www.apache.org/licenses/LICENSE-2.0
+//
+// Unless required by applicable law or agreed to in writing, software
+// distributed under the License is distributed on an "AS IS" BASIS,
+// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+// See the License for the specific language governing permissions and
+// limitations under the License.
+
+use std::str::FromStr;
+
+use common_error::ext::ErrorExt;
+use lazy_static::lazy_static;
+use regex::Regex;
+use serde::{Deserialize, Serialize};
+use snafu::{ensure, OptionExt, ResultExt};
+
+use crate::error::{
+    DecodeJsonSnafu, EncodeJsonSnafu, Error, FromUtf8Snafu, InvalidNodeInfoKeySnafu,
+    InvalidRoleSnafu, ParseNumSnafu, Result,
+};
+use crate::peer::Peer;
+
+const CLUSTER_NODE_INFO_PREFIX: &str = "__meta_cluster_node_info";
+
+lazy_static! {
+    static ref CLUSTER_NODE_INFO_PREFIX_PATTERN: Regex = Regex::new(&format!(
+        "^{CLUSTER_NODE_INFO_PREFIX}-([0-9]+)-([0-9]+)-([0-9]+)$"
+    ))
+    .unwrap();
+}
+
+/// [ClusterInfo] provides information about the cluster.
+#[async_trait::async_trait]
+pub trait ClusterInfo {
+    type Error: ErrorExt;
+
+    /// List all nodes by role in the cluster. If `role` is `None`, list all nodes.
+    async fn list_nodes(
+        &self,
+        role: Option<Role>,
+    ) -> std::result::Result<Vec<NodeInfo>, Self::Error>;
+
+    // TODO(jeremy): Other info, like region status, etc.
+}
+
+/// The key of [NodeInfo] in the storage. The format is `__meta_cluster_node_info-{cluster_id}-{role}-{node_id}`.
+/// This key cannot be used to describe the `Metasrv` because the `Metasrv` does not have
+/// a `cluster_id`, it serves multiple clusters.
+#[derive(Debug, Clone, Eq, Hash, PartialEq, Serialize, Deserialize)]
+pub struct NodeInfoKey {
+    /// The cluster id.
+    pub cluster_id: u64,
+    /// The role of the node. It can be `[Role::Datanode]` or `[Role::Frontend]`.
+    pub role: Role,
+    /// The node id.
+    pub node_id: u64,
+}
+
+impl NodeInfoKey {
+    pub fn key_prefix_with_cluster_id(cluster_id: u64) -> String {
+        format!("{}-{}-", CLUSTER_NODE_INFO_PREFIX, cluster_id)
+    }
+
+    pub fn key_prefix_with_role(cluster_id: u64, role: Role) -> String {
+        format!(
+            "{}-{}-{}-",
+            CLUSTER_NODE_INFO_PREFIX,
+            cluster_id,
+            i32::from(role)
+        )
+    }
+}
+
+/// The information of a node in the cluster.
+#[derive(Debug, Serialize, Deserialize)]
+pub struct NodeInfo {
+    /// The peer information. [node_id, address]
+    pub peer: Peer,
+    /// Last activity time in milliseconds.
+    pub last_activity_ts: i64,
+    /// The status of the node. Different roles have different node status.
+    pub status: NodeStatus,
+}
+
+#[derive(Debug, Clone, Eq, Hash, PartialEq, Serialize, Deserialize)]
+pub enum Role {
+    Datanode,
+    Frontend,
+    Metasrv,
+}
+
+#[derive(Debug, Serialize, Deserialize)]
+pub enum NodeStatus {
+    Datanode(DatanodeStatus),
+    Frontend(FrontendStatus),
+    Metasrv(MetasrvStatus),
+}
+
+/// The status of a datanode.
+#[derive(Debug, Serialize, Deserialize)]
+pub struct DatanodeStatus {
+    /// The read capacity units during this period.
+    pub rcus: i64,
+    /// The write capacity units during this period.
+    pub wcus: i64,
+    /// How many leader regions on this node.
+    pub leader_regions: usize,
+    /// How many follower regions on this node.
+    pub follower_regions: usize,
+}
+
+/// The status of a frontend.
+#[derive(Debug, Serialize, Deserialize)]
+pub struct FrontendStatus {}
+
+/// The status of a metasrv.
+#[derive(Debug, Serialize, Deserialize)]
+pub struct MetasrvStatus {
+    pub is_leader: bool,
+}
+
+impl FromStr for NodeInfoKey {
+    type Err = Error;
+
+    fn from_str(key: &str) -> Result<Self> {
+        let caps = CLUSTER_NODE_INFO_PREFIX_PATTERN
+            .captures(key)
+            .context(InvalidNodeInfoKeySnafu { key })?;
+
+        ensure!(caps.len() == 4, InvalidNodeInfoKeySnafu { key });
+
+        let cluster_id = caps[1].to_string();
+        let role = caps[2].to_string();
+        let node_id = caps[3].to_string();
+        let cluster_id: u64 = cluster_id.parse().context(ParseNumSnafu {
+            err_msg: format!("invalid cluster_id: {cluster_id}"),
+        })?;
+        let role: i32 = role.parse().context(ParseNumSnafu {
+            err_msg: format!("invalid role {role}"),
+        })?;
+        let role = Role::try_from(role)?;
+        let node_id: u64 = node_id.parse().context(ParseNumSnafu {
+            err_msg: format!("invalid node_id: {node_id}"),
+        })?;
+
+        Ok(Self {
+            cluster_id,
+            role,
+            node_id,
+        })
+    }
+}
+
+impl TryFrom<Vec<u8>> for NodeInfoKey {
+    type Error = Error;
+
+    fn try_from(bytes: Vec<u8>) -> Result<Self> {
+        String::from_utf8(bytes)
+            .context(FromUtf8Snafu {
+                name: "NodeInfoKey",
+            })
+            .map(|x| x.parse())?
+    }
+}
+
+impl From<NodeInfoKey> for Vec<u8> {
+    fn from(key: NodeInfoKey) -> Self {
+        format!(
+            "{}-{}-{}-{}",
+            CLUSTER_NODE_INFO_PREFIX,
+            key.cluster_id,
+            i32::from(key.role),
+            key.node_id
+        )
+        .into_bytes()
+    }
+}
+
+impl FromStr for NodeInfo {
+    type Err = Error;
+
+    fn from_str(value: &str) -> Result<Self> {
+        serde_json::from_str(value).context(DecodeJsonSnafu)
+    }
+}
+
+impl TryFrom<Vec<u8>> for NodeInfo {
+    type Error = Error;
+
+    fn try_from(bytes: Vec<u8>) -> Result<Self> {
+        String::from_utf8(bytes)
+            .context(FromUtf8Snafu { name: "NodeInfo" })
+            .map(|x| x.parse())?
+    }
+}
+
+impl TryFrom<NodeInfo> for Vec<u8> {
+    type Error = Error;
+
+    fn try_from(info: NodeInfo) -> Result<Self> {
+        Ok(serde_json::to_string(&info)
+            .context(EncodeJsonSnafu)?
+            .into_bytes())
+    }
+}
+
+impl From<Role> for i32 {
+    fn from(role: Role) -> Self {
+        match role {
+            Role::Datanode => 0,
+            Role::Frontend => 1,
+            Role::Metasrv => 2,
+        }
+    }
+}
+
+impl TryFrom<i32> for Role {
+    type Error = Error;
+
+    fn try_from(role: i32) -> Result<Self> {
+        match role {
+            0 => Ok(Self::Datanode),
+            1 => Ok(Self::Frontend),
+            2 => Ok(Self::Metasrv),
+            _ => InvalidRoleSnafu { role }.fail(),
+        }
+    }
+}
+
+#[cfg(test)]
+mod tests {
+    use std::assert_matches::assert_matches;
+
+    use crate::cluster::Role::{Datanode, Frontend};
+    use crate::cluster::{DatanodeStatus, NodeInfo, NodeInfoKey, NodeStatus};
+    use crate::peer::Peer;
+
+    #[test]
+    fn test_node_info_key_round_trip() {
+        let key = NodeInfoKey {
+            cluster_id: 1,
+            role: Datanode,
+            node_id: 2,
+        };
+
+        let key_bytes: Vec<u8> = key.into();
+        let new_key: NodeInfoKey = key_bytes.try_into().unwrap();
+
+        assert_eq!(1, new_key.cluster_id);
+        assert_eq!(Datanode, new_key.role);
+        assert_eq!(2, new_key.node_id);
+    }
+
+    #[test]
+    fn test_node_info_round_trip() {
+        let node_info = NodeInfo {
+            peer: Peer {
+                id: 1,
+                addr: "127.0.0.1".to_string(),
+            },
+            last_activity_ts: 123,
+            status: NodeStatus::Datanode(DatanodeStatus {
+                rcus: 1,
+                wcus: 2,
+                leader_regions: 3,
+                follower_regions: 4,
+            }),
+        };
+
+        let node_info_bytes: Vec<u8> = node_info.try_into().unwrap();
+        let new_node_info: NodeInfo = node_info_bytes.try_into().unwrap();
+
+        assert_matches!(
+            new_node_info,
+            NodeInfo {
+                peer: Peer { id: 1, .. },
+                last_activity_ts: 123,
+                status: NodeStatus::Datanode(DatanodeStatus {
+                    rcus: 1,
+                    wcus: 2,
+                    leader_regions: 3,
+                    follower_regions: 4,
+                }),
+            }
+        );
+    }
+
+    #[test]
+    fn test_node_info_key_prefix() {
+        let prefix = NodeInfoKey::key_prefix_with_cluster_id(1);
+        assert_eq!(prefix, "__meta_cluster_node_info-1-");
+
+        let prefix = NodeInfoKey::key_prefix_with_role(2, Frontend);
+        assert_eq!(prefix, "__meta_cluster_node_info-2-1-");
+    }
+}
--- a/src/common/meta/src/datanode_manager.rs
+++ b/src/common/meta/src/datanode_manager.rs
@@ -12,10 +12,10 @@
 // See the License for the specific language governing permissions and
 // limitations under the License.

-use std::collections::HashMap;
 use std::sync::Arc;

-use api::v1::region::{QueryRequest, RegionRequest, RegionResponse};
+use api::region::RegionResponse;
+use api::v1::region::{QueryRequest, RegionRequest};
 pub use common_base::AffectedRows;
 use common_recordbatch::SendableRecordBatchStream;

@@ -26,7 +26,7 @@ use crate::peer::Peer;
 #[async_trait::async_trait]
 pub trait Datanode: Send + Sync {
    /// Handles DML, and DDL requests.
-    async fn handle(&self, request: RegionRequest) -> Result<HandleResponse>;
+    async fn handle(&self, request: RegionRequest) -> Result<RegionResponse>;

    /// Handles query requests
    async fn handle_query(&self, request: QueryRequest) -> Result<SendableRecordBatchStream>;
@@ -42,27 +42,3 @@ pub trait DatanodeManager: Send + Sync {
 }

 pub type DatanodeManagerRef = Arc<dyn DatanodeManager>;
-
-/// This result struct is derived from [RegionResponse]
-#[derive(Debug)]
-pub struct HandleResponse {
-    pub affected_rows: AffectedRows,
-    pub extension: HashMap<String, Vec<u8>>,
-}
-
-impl HandleResponse {
-    pub fn from_region_response(region_response: RegionResponse) -> Self {
-        Self {
-            affected_rows: region_response.affected_rows as _,
-            extension: region_response.extension,
-        }
-    }
-
-    /// Creates one response without extension
-    pub fn new(affected_rows: AffectedRows) -> Self {
-        Self {
-            affected_rows,
-            extension: Default::default(),
-        }
-    }
-}
--- a/src/common/meta/src/ddl.rs
+++ b/src/common/meta/src/ddl.rs
@@ -30,6 +30,7 @@ use crate::rpc::procedure::{MigrateRegionRequest, MigrateRegionResponse, Procedu

 pub mod alter_logical_tables;
 pub mod alter_table;
+pub mod create_database;
 pub mod create_logical_tables;
 pub mod create_table;
 mod create_table_template;
--- a/src/common/meta/src/ddl/alter_logical_tables.rs
+++ b/src/common/meta/src/ddl/alter_logical_tables.rs
@@ -23,7 +23,6 @@ use common_procedure::error::{FromJsonSnafu, Result as ProcedureResult, ToJsonSn
 use common_procedure::{Context, LockKey, Procedure, Status};
 use common_telemetry::{info, warn};
 use futures_util::future;
-use itertools::Itertools;
 use serde::{Deserialize, Serialize};
 use snafu::{ensure, ResultExt};
 use store_api::metadata::ColumnMetadata;
@@ -32,14 +31,14 @@ use strum::AsRefStr;
 use table::metadata::TableId;

 use crate::ddl::utils::add_peer_context_if_needed;
-use crate::ddl::{physical_table_metadata, DdlContext};
+use crate::ddl::DdlContext;
 use crate::error::{DecodeJsonSnafu, Error, MetadataCorruptionSnafu, Result};
 use crate::key::table_info::TableInfoValue;
 use crate::key::table_route::PhysicalTableRouteValue;
 use crate::key::DeserializedValueWithBytes;
 use crate::lock_key::{CatalogLock, SchemaLock, TableLock};
 use crate::rpc::ddl::AlterTableTask;
-use crate::rpc::router::{find_leader_regions, find_leaders};
+use crate::rpc::router::find_leaders;
 use crate::{cache_invalidator, metrics, ClusterId};

 pub struct AlterLogicalTablesProcedure {
@@ -118,23 +117,17 @@ impl AlterLogicalTablesProcedure {

        for peer in leaders {
            let requester = self.context.datanode_manager.datanode(&peer).await;
-            let region_numbers = find_leader_regions(&physical_table_route.region_routes, &peer);
+            let request = self.make_request(&peer, &physical_table_route.region_routes)?;

-            for region_number in region_numbers {
-                let request = self.make_request(region_number)?;
-                let peer = peer.clone();
-                let requester = requester.clone();
-
-                alter_region_tasks.push(async move {
-                    requester
-                        .handle(request)
-                        .await
-                        .map_err(add_peer_context_if_needed(peer))
-                });
-            }
+            alter_region_tasks.push(async move {
+                requester
+                    .handle(request)
+                    .await
+                    .map_err(add_peer_context_if_needed(peer))
+            });
        }

-        // Collects responses from all the alter region tasks.
+        // Collects responses from datanodes.
        let phy_raw_schemas = future::join_all(alter_region_tasks)
            .await
            .into_iter()
@@ -169,44 +162,8 @@ impl AlterLogicalTablesProcedure {
    }

    pub(crate) async fn on_update_metadata(&mut self) -> Result<Status> {
-        if !self.data.physical_columns.is_empty() {
-            let physical_table_info = self.data.physical_table_info.as_ref().unwrap();
-
-            // Generates new table info
-            let old_raw_table_info = physical_table_info.table_info.clone();
-            let new_raw_table_info = physical_table_metadata::build_new_physical_table_info(
-                old_raw_table_info,
-                &self.data.physical_columns,
-            );
-
-            // Updates physical table's metadata
-            self.context
-                .table_metadata_manager
-                .update_table_info(
-                    DeserializedValueWithBytes::from_inner(physical_table_info.clone()),
-                    new_raw_table_info,
-                )
-                .await?;
-        }
-
-        let table_info_values = self.build_update_metadata()?;
-        let manager = &self.context.table_metadata_manager;
-        let chunk_size = manager.batch_update_table_info_value_chunk_size();
-        if table_info_values.len() > chunk_size {
-            let chunks = table_info_values
-                .into_iter()
-                .chunks(chunk_size)
-                .into_iter()
-                .map(|check| check.collect::<Vec<_>>())
-                .collect::<Vec<_>>();
-            for chunk in chunks {
-                manager.batch_update_table_info_values(chunk).await?;
-            }
-        } else {
-            manager
-                .batch_update_table_info_values(table_info_values)
-                .await?;
-        }
+        self.update_physical_table_metadata().await?;
+        self.update_logical_tables_metadata().await?;

        self.data.state = AlterTablesState::InvalidateTableCache;
        Ok(Status::executing(true))
@@ -289,10 +246,10 @@ pub struct AlterTablesData {
    tasks: Vec<AlterTableTask>,
    /// Table info values before the alter operation.
    /// Corresponding one-to-one with the AlterTableTask in tasks.
-    table_info_values: Vec<TableInfoValue>,
+    table_info_values: Vec<DeserializedValueWithBytes<TableInfoValue>>,
    /// Physical table info
    physical_table_id: TableId,
-    physical_table_info: Option<TableInfoValue>,
+    physical_table_info: Option<DeserializedValueWithBytes<TableInfoValue>>,
    physical_table_route: Option<PhysicalTableRouteValue>,
    physical_columns: Vec<ColumnMetadata>,
 }
--- a/src/common/meta/src/ddl/alter_logical_tables/metadata.rs
+++ b/src/common/meta/src/ddl/alter_logical_tables/metadata.rs
@@ -24,6 +24,7 @@ use crate::error::{
 use crate::key::table_info::TableInfoValue;
 use crate::key::table_name::TableNameKey;
 use crate::key::table_route::TableRouteValue;
+use crate::key::DeserializedValueWithBytes;
 use crate::rpc::ddl::AlterTableTask;

 impl AlterLogicalTablesProcedure {
@@ -61,11 +62,9 @@ impl AlterLogicalTablesProcedure {
            .get_full_table_info(self.data.physical_table_id)
            .await?;

-        let physical_table_info = physical_table_info
-            .with_context(|| TableInfoNotFoundSnafu {
-                table: format!("table id - {}", self.data.physical_table_id),
-            })?
-            .into_inner();
+        let physical_table_info = physical_table_info.with_context(|| TableInfoNotFoundSnafu {
+            table: format!("table id - {}", self.data.physical_table_id),
+        })?;
        let physical_table_route = physical_table_route
            .context(TableRouteNotFoundSnafu {
                table_id: self.data.physical_table_id,
@@ -99,9 +98,9 @@ impl AlterLogicalTablesProcedure {
    async fn get_all_table_info_values(
        &self,
        table_ids: &[TableId],
-    ) -> Result<Vec<TableInfoValue>> {
+    ) -> Result<Vec<DeserializedValueWithBytes<TableInfoValue>>> {
        let table_info_manager = self.context.table_metadata_manager.table_info_manager();
-        let mut table_info_map = table_info_manager.batch_get(table_ids).await?;
+        let mut table_info_map = table_info_manager.batch_get_raw(table_ids).await?;
        let mut table_info_values = Vec::with_capacity(table_ids.len());
        for (table_id, task) in table_ids.iter().zip(self.data.tasks.iter()) {
            let table_info_value =
--- a/src/common/meta/src/ddl/alter_logical_tables/region_request.rs
+++ b/src/common/meta/src/ddl/alter_logical_tables/region_request.rs
@@ -19,16 +19,22 @@ use api::v1::region::{
    RegionColumnDef, RegionRequest, RegionRequestHeader,
 };
 use common_telemetry::tracing_context::TracingContext;
-use store_api::storage::{RegionId, RegionNumber};
+use store_api::storage::RegionId;

 use crate::ddl::alter_logical_tables::AlterLogicalTablesProcedure;
 use crate::error::Result;
 use crate::key::table_info::TableInfoValue;
+use crate::peer::Peer;
 use crate::rpc::ddl::AlterTableTask;
+use crate::rpc::router::{find_leader_regions, RegionRoute};

 impl AlterLogicalTablesProcedure {
-    pub(crate) fn make_request(&self, region_number: RegionNumber) -> Result<RegionRequest> {
-        let alter_requests = self.make_alter_region_requests(region_number)?;
+    pub(crate) fn make_request(
+        &self,
+        peer: &Peer,
+        region_routes: &[RegionRoute],
+    ) -> Result<RegionRequest> {
+        let alter_requests = self.make_alter_region_requests(peer, region_routes)?;
        let request = RegionRequest {
            header: Some(RegionRequestHeader {
                tracing_context: TracingContext::from_current_span().to_w3c(),
@@ -40,17 +46,25 @@ impl AlterLogicalTablesProcedure {
        Ok(request)
    }

-    fn make_alter_region_requests(&self, region_number: RegionNumber) -> Result<AlterRequests> {
-        let mut requests = Vec::with_capacity(self.data.tasks.len());
+    fn make_alter_region_requests(
+        &self,
+        peer: &Peer,
+        region_routes: &[RegionRoute],
+    ) -> Result<AlterRequests> {
+        let tasks = &self.data.tasks;
+        let regions_on_this_peer = find_leader_regions(region_routes, peer);
+        let mut requests = Vec::with_capacity(tasks.len() * regions_on_this_peer.len());
        for (task, table) in self
            .data
            .tasks
            .iter()
            .zip(self.data.table_info_values.iter())
        {
-            let region_id = RegionId::new(table.table_info.ident.table_id, region_number);
-            let request = self.make_alter_region_request(region_id, task, table)?;
-            requests.push(request);
+            for region_number in &regions_on_this_peer {
+                let region_id = RegionId::new(table.table_info.ident.table_id, *region_number);
+                let request = self.make_alter_region_request(region_id, task, table)?;
+                requests.push(request);
+            }
        }

        Ok(AlterRequests { requests })
--- a/src/common/meta/src/ddl/alter_logical_tables/update_metadata.rs
+++ b/src/common/meta/src/ddl/alter_logical_tables/update_metadata.rs
@@ -13,17 +13,71 @@
 // limitations under the License.

 use common_grpc_expr::alter_expr_to_request;
+use common_telemetry::warn;
+use itertools::Itertools;
 use snafu::ResultExt;
 use table::metadata::{RawTableInfo, TableInfo};

 use crate::ddl::alter_logical_tables::AlterLogicalTablesProcedure;
+use crate::ddl::physical_table_metadata;
 use crate::error;
 use crate::error::{ConvertAlterTableRequestSnafu, Result};
 use crate::key::table_info::TableInfoValue;
+use crate::key::DeserializedValueWithBytes;
 use crate::rpc::ddl::AlterTableTask;

 impl AlterLogicalTablesProcedure {
-    pub(crate) fn build_update_metadata(&self) -> Result<Vec<(TableInfoValue, RawTableInfo)>> {
+    pub(crate) async fn update_physical_table_metadata(&mut self) -> Result<()> {
+        if self.data.physical_columns.is_empty() {
+            warn!("No physical columns found, leaving the physical table's schema unchanged when altering logical tables");
+            return Ok(());
+        }
+
+        // Safety: must exist.
+        let physical_table_info = self.data.physical_table_info.as_ref().unwrap();
+
+        // Generates new table info
+        let old_raw_table_info = physical_table_info.table_info.clone();
+        let new_raw_table_info = physical_table_metadata::build_new_physical_table_info(
+            old_raw_table_info,
+            &self.data.physical_columns,
+        );
+
+        // Updates physical table's metadata
+        self.context
+            .table_metadata_manager
+            .update_table_info(physical_table_info, new_raw_table_info)
+            .await?;
+
+        Ok(())
+    }
+
+    pub(crate) async fn update_logical_tables_metadata(&mut self) -> Result<()> {
+        let table_info_values = self.build_update_metadata()?;
+        let manager = &self.context.table_metadata_manager;
+        let chunk_size = manager.batch_update_table_info_value_chunk_size();
+        if table_info_values.len() > chunk_size {
+            let chunks = table_info_values
+                .into_iter()
+                .chunks(chunk_size)
+                .into_iter()
+                .map(|check| check.collect::<Vec<_>>())
+                .collect::<Vec<_>>();
+            for chunk in chunks {
+                manager.batch_update_table_info_values(chunk).await?;
+            }
+        } else {
+            manager
+                .batch_update_table_info_values(table_info_values)
+                .await?;
+        }
+
+        Ok(())
+    }
+
+    pub(crate) fn build_update_metadata(
+        &self,
+    ) -> Result<Vec<(DeserializedValueWithBytes<TableInfoValue>, RawTableInfo)>> {
        let mut table_info_values_to_update = Vec::with_capacity(self.data.tasks.len());
        for (task, table) in self
            .data
@@ -40,8 +94,8 @@ impl AlterLogicalTablesProcedure {
    fn build_new_table_info(
        &self,
        task: &AlterTableTask,
-        table: &TableInfoValue,
-    ) -> Result<(TableInfoValue, RawTableInfo)> {
+        table: &DeserializedValueWithBytes<TableInfoValue>,
+    ) -> Result<(DeserializedValueWithBytes<TableInfoValue>, RawTableInfo)> {
        // Builds new_meta
        let table_info = TableInfo::try_from(table.table_info.clone())
            .context(error::ConvertRawTableInfoSnafu)?;
--- a/src/common/meta/src/ddl/alter_table.rs
+++ b/src/common/meta/src/ddl/alter_table.rs
@@ -12,52 +12,49 @@
 // See the License for the specific language governing permissions and
 // limitations under the License.

+mod check;
+mod metadata;
+mod region_request;
+mod update_metadata;
+
 use std::vec;

 use api::v1::alter_expr::Kind;
-use api::v1::region::{
-    alter_request, region_request, AddColumn, AddColumns, AlterRequest, DropColumn, DropColumns,
-    RegionColumnDef, RegionRequest, RegionRequestHeader,
-};
-use api::v1::{AlterExpr, RenameTable};
+use api::v1::RenameTable;
 use async_trait::async_trait;
 use common_error::ext::ErrorExt;
 use common_error::status_code::StatusCode;
-use common_grpc_expr::alter_expr_to_request;
 use common_procedure::error::{FromJsonSnafu, Result as ProcedureResult, ToJsonSnafu};
 use common_procedure::{
    Context as ProcedureContext, Error as ProcedureError, LockKey, Procedure, Status, StringKey,
 };
-use common_telemetry::tracing_context::TracingContext;
 use common_telemetry::{debug, info};
 use futures::future;
 use serde::{Deserialize, Serialize};
-use snafu::{ensure, OptionExt, ResultExt};
-use store_api::storage::{ColumnId, RegionId};
+use snafu::ResultExt;
+use store_api::storage::RegionId;
 use strum::AsRefStr;
-use table::metadata::{RawTableInfo, TableId, TableInfo};
-use table::requests::AlterKind;
+use table::metadata::{RawTableInfo, TableId};
 use table::table_reference::TableReference;

 use crate::cache_invalidator::Context;
 use crate::ddl::utils::add_peer_context_if_needed;
 use crate::ddl::DdlContext;
-use crate::error::{self, ConvertAlterTableRequestSnafu, Error, InvalidProtoMsgSnafu, Result};
+use crate::error::{Error, Result};
 use crate::instruction::CacheIdent;
 use crate::key::table_info::TableInfoValue;
-use crate::key::table_name::TableNameKey;
 use crate::key::DeserializedValueWithBytes;
 use crate::lock_key::{CatalogLock, SchemaLock, TableLock, TableNameLock};
 use crate::metrics;
 use crate::rpc::ddl::AlterTableTask;
 use crate::rpc::router::{find_leader_regions, find_leaders};
-use crate::table_name::TableName;

+/// The alter table procedure
 pub struct AlterTableProcedure {
+    // The runtime context.
    context: DdlContext,
+    // The serialized data.
    data: AlterTableData,
-    /// proto alter Kind for adding/dropping columns.
-    kind: Option<alter_request::Kind>,
 }

 impl AlterTableProcedure {
@@ -65,123 +62,36 @@ impl AlterTableProcedure {

    pub fn new(
        cluster_id: u64,
+        table_id: TableId,
        task: AlterTableTask,
-        table_info_value: DeserializedValueWithBytes<TableInfoValue>,
        context: DdlContext,
    ) -> Result<Self> {
-        let alter_kind = task
-            .alter_table
-            .kind
-            .as_ref()
-            .context(InvalidProtoMsgSnafu {
-                err_msg: "'kind' is absent",
-            })?;
-        let (kind, next_column_id) =
-            create_proto_alter_kind(&table_info_value.table_info, alter_kind)?;
-
-        debug!(
-            "New AlterTableProcedure, kind: {:?}, next_column_id: {:?}",
-            kind, next_column_id
-        );
-
+        task.validate()?;
        Ok(Self {
            context,
-            data: AlterTableData::new(task, table_info_value, cluster_id, next_column_id),
-            kind,
+            data: AlterTableData::new(task, table_id, cluster_id),
        })
    }

    pub fn from_json(json: &str, context: DdlContext) -> ProcedureResult<Self> {
        let data: AlterTableData = serde_json::from_str(json).context(FromJsonSnafu)?;
-        let alter_kind = data
-            .task
-            .alter_table
-            .kind
-            .as_ref()
-            .context(InvalidProtoMsgSnafu {
-                err_msg: "'kind' is absent",
-            })
-            .map_err(ProcedureError::external)?;
-        let (kind, next_column_id) =
-            create_proto_alter_kind(&data.table_info_value.table_info, alter_kind)
-                .map_err(ProcedureError::external)?;
-        assert_eq!(data.next_column_id, next_column_id);
-
-        Ok(AlterTableProcedure {
-            context,
-            data,
-            kind,
-        })
+        Ok(AlterTableProcedure { context, data })
    }

    // Checks whether the table exists.
-    async fn on_prepare(&mut self) -> Result<Status> {
-        let alter_expr = &self.alter_expr();
-        let catalog = &alter_expr.catalog_name;
-        let schema = &alter_expr.schema_name;
-
-        let alter_kind = self.alter_kind()?;
-        let manager = &self.context.table_metadata_manager;
-
-        if let Kind::RenameTable(RenameTable { new_table_name }) = alter_kind {
-            let new_table_name_key = TableNameKey::new(catalog, schema, new_table_name);
-
-            let exist = manager
-                .table_name_manager()
-                .exists(new_table_name_key)
-                .await?;
-
-            ensure!(
-                !exist,
-                error::TableAlreadyExistsSnafu {
-                    table_name: TableName::from(new_table_name_key).to_string(),
-                }
-            )
-        }
-
-        let table_name_key = TableNameKey::new(catalog, schema, &alter_expr.table_name);
-
-        let exist = manager.table_name_manager().exists(table_name_key).await?;
-
-        ensure!(
-            exist,
-            error::TableNotFoundSnafu {
-                table_name: TableName::from(table_name_key).to_string()
-            }
-        );
-
+    pub(crate) async fn on_prepare(&mut self) -> Result<Status> {
+        self.check_alter().await?;
+        self.fill_table_info().await?;
+        // Safety: Checked in `AlterTableProcedure::new`.
+        let alter_kind = self.data.task.alter_table.kind.as_ref().unwrap();
        if matches!(alter_kind, Kind::RenameTable { .. }) {
            self.data.state = AlterTableState::UpdateMetadata;
        } else {
            self.data.state = AlterTableState::SubmitAlterRegionRequests;
        };
-
        Ok(Status::executing(true))
    }

-    fn alter_expr(&self) -> &AlterExpr {
-        &self.data.task.alter_table
-    }
-
-    fn alter_kind(&self) -> Result<&Kind> {
-        self.alter_expr()
-            .kind
-            .as_ref()
-            .context(InvalidProtoMsgSnafu {
-                err_msg: "'kind' is absent",
-            })
-    }
-
-    pub fn create_alter_region_request(&self, region_id: RegionId) -> Result<AlterRequest> {
-        let table_info = self.data.table_info();
-
-        Ok(AlterRequest {
-            region_id: region_id.as_u64(),
-            schema_version: table_info.ident.version,
-            kind: self.kind.clone(),
-        })
-    }
-
    pub async fn submit_alter_region_requests(&mut self) -> Result<Status> {
        let table_id = self.data.table_id();
        let (_, physical_table_route) = self
@@ -200,14 +110,7 @@ impl AlterTableProcedure {

            for region in regions {
                let region_id = RegionId::new(table_id, region);
-                let request = self.create_alter_region_request(region_id)?;
-                let request = RegionRequest {
-                    header: Some(RegionRequestHeader {
-                        tracing_context: TracingContext::from_current_span().to_w3c(),
-                        ..Default::default()
-                    }),
-                    body: Some(region_request::Body::Alter(request)),
-                };
+                let request = self.make_alter_region_request(region_id)?;
                debug!("Submitting {request:?} to {datanode}");

                let datanode = datanode.clone();
@@ -238,91 +141,39 @@ impl AlterTableProcedure {
        Ok(Status::executing(true))
    }

-    /// Update table metadata for rename table operation.
-    async fn on_update_metadata_for_rename(&self, new_table_name: String) -> Result<()> {
-        let table_metadata_manager = &self.context.table_metadata_manager;
-
-        let current_table_info_value = self.data.table_info_value.clone();
-
-        table_metadata_manager
-            .rename_table(current_table_info_value, new_table_name)
-            .await?;
-
-        Ok(())
-    }
-
-    async fn on_update_metadata_for_alter(&self, new_table_info: RawTableInfo) -> Result<()> {
-        let table_metadata_manager = &self.context.table_metadata_manager;
-        let current_table_info_value = self.data.table_info_value.clone();
-
-        table_metadata_manager
-            .update_table_info(current_table_info_value, new_table_info)
-            .await?;
-
-        Ok(())
-    }
-
-    fn build_new_table_info(&self) -> Result<TableInfo> {
-        // Builds new_meta
-        let table_info = TableInfo::try_from(self.data.table_info().clone())
-            .context(error::ConvertRawTableInfoSnafu)?;
-
-        let table_ref = self.data.table_ref();
-
-        let request = alter_expr_to_request(self.data.table_id(), self.alter_expr().clone())
-            .context(ConvertAlterTableRequestSnafu)?;
-
-        let new_meta = table_info
-            .meta
-            .builder_with_alter_kind(table_ref.table, &request.alter_kind, false)
-            .context(error::TableSnafu)?
-            .build()
-            .with_context(|_| error::BuildTableMetaSnafu {
-                table_name: table_ref.table,
-            })?;
-
-        let mut new_info = table_info.clone();
-        new_info.meta = new_meta;
-        new_info.ident.version = table_info.ident.version + 1;
-        if let Some(column_id) = self.data.next_column_id {
-            new_info.meta.next_column_id = new_info.meta.next_column_id.max(column_id);
-        }
-
-        if let AlterKind::RenameTable { new_table_name } = &request.alter_kind {
-            new_info.name = new_table_name.to_string();
-        }
-
-        Ok(new_info)
-    }
-
    /// Update table metadata.
-    async fn on_update_metadata(&mut self) -> Result<Status> {
+    pub(crate) async fn on_update_metadata(&mut self) -> Result<Status> {
        let table_id = self.data.table_id();
        let table_ref = self.data.table_ref();
-        let new_info = self.build_new_table_info()?;
+        // Safety: checked before.
+        let table_info_value = self.data.table_info_value.as_ref().unwrap();
+        let new_info = self.build_new_table_info(&table_info_value.table_info)?;

        debug!(
-            "starting update table: {} metadata, new table info {:?}",
+            "Starting update table: {} metadata, new table info {:?}",
            table_ref.to_string(),
            new_info
        );

-        if let Kind::RenameTable(RenameTable { new_table_name }) = self.alter_kind()? {
-            self.on_update_metadata_for_rename(new_table_name.to_string())
+        // Safety: Checked in `AlterTableProcedure::new`.
+        let alter_kind = self.data.task.alter_table.kind.as_ref().unwrap();
+        if let Kind::RenameTable(RenameTable { new_table_name }) = alter_kind {
+            self.on_update_metadata_for_rename(new_table_name.to_string(), table_info_value)
                .await?;
        } else {
-            self.on_update_metadata_for_alter(new_info.into()).await?;
+            self.on_update_metadata_for_alter(new_info.into(), table_info_value)
+                .await?;
        }

        info!("Updated table metadata for table {table_ref}, table_id: {table_id}");
-
        self.data.state = AlterTableState::InvalidateTableCache;
        Ok(Status::executing(true))
    }

    /// Broadcasts the invalidating table cache instructions.
    async fn on_broadcast(&mut self) -> Result<Status> {
-        let alter_kind = self.alter_kind()?;
+        // Safety: Checked in `AlterTableProcedure::new`.
+        let alter_kind = self.data.task.alter_table.kind.as_ref().unwrap();
        let cache_invalidator = &self.context.cache_invalidator;
        let cache_keys = if matches!(alter_kind, Kind::RenameTable { .. }) {
            vec![CacheIdent::TableName(self.data.table_ref().into())]
@@ -348,7 +199,9 @@ impl AlterTableProcedure {
        lock_key.push(SchemaLock::read(table_ref.catalog, table_ref.schema).into());
        lock_key.push(TableLock::Write(table_id).into());

-        if let Ok(Kind::RenameTable(RenameTable { new_table_name })) = self.alter_kind() {
+        // Safety: Checked in `AlterTableProcedure::new`.
+        let alter_kind = self.data.task.alter_table.kind.as_ref().unwrap();
+        if let Kind::RenameTable(RenameTable { new_table_name }) = alter_kind {
            lock_key.push(
                TableNameLock::new(table_ref.catalog, table_ref.schema, new_table_name).into(),
            )
@@ -403,8 +256,9 @@ impl Procedure for AlterTableProcedure {

 #[derive(Debug, Serialize, Deserialize, AsRefStr)]
 enum AlterTableState {
-    /// Prepares to alter the table
+    /// Prepares to alter the table.
    Prepare,
+    /// Sends alter region requests to Datanode.
    SubmitAlterRegionRequests,
    /// Updates table metadata.
    UpdateMetadata,
@@ -412,30 +266,25 @@ enum AlterTableState {
    InvalidateTableCache,
 }

+// The serialized data of alter table.
 #[derive(Debug, Serialize, Deserialize)]
 pub struct AlterTableData {
    cluster_id: u64,
    state: AlterTableState,
    task: AlterTableTask,
+    table_id: TableId,
    /// Table info value before alteration.
-    table_info_value: DeserializedValueWithBytes<TableInfoValue>,
-    /// Next column id of the table if the task adds columns to the table.
-    next_column_id: Option<ColumnId>,
+    table_info_value: Option<DeserializedValueWithBytes<TableInfoValue>>,
 }

 impl AlterTableData {
-    pub fn new(
-        task: AlterTableTask,
-        table_info_value: DeserializedValueWithBytes<TableInfoValue>,
-        cluster_id: u64,
-        next_column_id: Option<ColumnId>,
-    ) -> Self {
+    pub fn new(task: AlterTableTask, table_id: TableId, cluster_id: u64) -> Self {
        Self {
            state: AlterTableState::Prepare,
            task,
-            table_info_value,
+            table_id,
            cluster_id,
-            next_column_id,
+            table_info_value: None,
        }
    }

@@ -444,76 +293,12 @@ impl AlterTableData {
    }

    fn table_id(&self) -> TableId {
-        self.table_info().ident.table_id
+        self.table_id
    }

-    fn table_info(&self) -> &RawTableInfo {
-        &self.table_info_value.table_info
-    }
-}
-
-/// Creates region proto alter kind from `table_info` and `alter_kind`.
-///
-/// Returns the kind and next column id if it adds new columns.
-///
-/// # Panics
-/// Panics if kind is rename.
-pub fn create_proto_alter_kind(
-    table_info: &RawTableInfo,
-    alter_kind: &Kind,
-) -> Result<(Option<alter_request::Kind>, Option<ColumnId>)> {
-    match alter_kind {
-        Kind::AddColumns(x) => {
-            let mut next_column_id = table_info.meta.next_column_id;
-
-            let add_columns = x
-                .add_columns
-                .iter()
-                .map(|add_column| {
-                    let column_def =
-                        add_column
-                            .column_def
-                            .as_ref()
-                            .context(InvalidProtoMsgSnafu {
-                                err_msg: "'column_def' is absent",
-                            })?;
-
-                    let column_id = next_column_id;
-                    next_column_id += 1;
-
-                    let column_def = RegionColumnDef {
-                        column_def: Some(column_def.clone()),
-                        column_id,
-                    };
-
-                    Ok(AddColumn {
-                        column_def: Some(column_def),
-                        location: add_column.location.clone(),
-                    })
-                })
-                .collect::<Result<Vec<_>>>()?;
-
-            Ok((
-                Some(alter_request::Kind::AddColumns(AddColumns { add_columns })),
-                Some(next_column_id),
-            ))
-        }
-        Kind::DropColumns(x) => {
-            let drop_columns = x
-                .drop_columns
-                .iter()
-                .map(|x| DropColumn {
-                    name: x.name.clone(),
-                })
-                .collect::<Vec<_>>();
-
-            Ok((
-                Some(alter_request::Kind::DropColumns(DropColumns {
-                    drop_columns,
-                })),
-                None,
-            ))
-        }
-        Kind::RenameTable(_) => Ok((None, None)),
+    fn table_info(&self) -> Option<&RawTableInfo> {
+        self.table_info_value
+            .as_ref()
+            .map(|value| &value.table_info)
    }
 }
--- a/src/common/meta/src/ddl/alter_table/check.rs
+++ b/src/common/meta/src/ddl/alter_table/check.rs
@@ -0,0 +1,62 @@
+// Copyright 2023 Greptime Team
+//
+// Licensed under the Apache License, Version 2.0 (the "License");
+// you may not use this file except in compliance with the License.
+// You may obtain a copy of the License at
+//
+//     http://www.apache.org/licenses/LICENSE-2.0
+//
+// Unless required by applicable law or agreed to in writing, software
+// distributed under the License is distributed on an "AS IS" BASIS,
+// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+// See the License for the specific language governing permissions and
+// limitations under the License.
+
+use api::v1::alter_expr::Kind;
+use api::v1::RenameTable;
+use common_catalog::format_full_table_name;
+use snafu::ensure;
+
+use crate::ddl::alter_table::AlterTableProcedure;
+use crate::error::{self, Result};
+use crate::key::table_name::TableNameKey;
+
+impl AlterTableProcedure {
+    /// Checks:
+    /// - The new table name doesn't exist (rename).
+    /// - Table exists.
+    pub(crate) async fn check_alter(&self) -> Result<()> {
+        let alter_expr = &self.data.task.alter_table;
+        let catalog = &alter_expr.catalog_name;
+        let schema = &alter_expr.schema_name;
+        let table_name = &alter_expr.table_name;
+        // Safety: Checked in `AlterTableProcedure::new`.
+        let alter_kind = self.data.task.alter_table.kind.as_ref().unwrap();
+
+        let manager = &self.context.table_metadata_manager;
+        if let Kind::RenameTable(RenameTable { new_table_name }) = alter_kind {
+            let new_table_name_key = TableNameKey::new(catalog, schema, new_table_name);
+            let exists = manager
+                .table_name_manager()
+                .exists(new_table_name_key)
+                .await?;
+            ensure!(
+                !exists,
+                error::TableAlreadyExistsSnafu {
+                    table_name: format_full_table_name(catalog, schema, new_table_name),
+                }
+            )
+        }
+
+        let table_name_key = TableNameKey::new(catalog, schema, table_name);
+        let exists = manager.table_name_manager().exists(table_name_key).await?;
+        ensure!(
+            exists,
+            error::TableNotFoundSnafu {
+                table_name: format_full_table_name(catalog, schema, &alter_expr.table_name),
+            }
+        );
+
+        Ok(())
+    }
+}
--- a/src/common/meta/src/ddl/alter_table/metadata.rs
+++ b/src/common/meta/src/ddl/alter_table/metadata.rs
@@ -0,0 +1,42 @@
+// Copyright 2023 Greptime Team
+//
+// Licensed under the Apache License, Version 2.0 (the "License");
+// you may not use this file except in compliance with the License.
+// You may obtain a copy of the License at
+//
+//     http://www.apache.org/licenses/LICENSE-2.0
+//
+// Unless required by applicable law or agreed to in writing, software
+// distributed under the License is distributed on an "AS IS" BASIS,
+// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+// See the License for the specific language governing permissions and
+// limitations under the License.
+
+use common_catalog::format_full_table_name;
+use snafu::OptionExt;
+
+use crate::ddl::alter_table::AlterTableProcedure;
+use crate::error::{self, Result};
+
+impl AlterTableProcedure {
+    /// Fetches the table info.
+    pub(crate) async fn fill_table_info(&mut self) -> Result<()> {
+        let table_id = self.data.table_id();
+        let alter_expr = &self.data.task.alter_table;
+        let catalog = &alter_expr.catalog_name;
+        let schema = &alter_expr.schema_name;
+        let table_name = &alter_expr.table_name;
+
+        let table_info_value = self
+            .context
+            .table_metadata_manager
+            .table_info_manager()
+            .get(table_id)
+            .await?
+            .with_context(|| error::TableNotFoundSnafu {
+                table_name: format_full_table_name(catalog, schema, table_name),
+            })?;
+        self.data.table_info_value = Some(table_info_value);
+        Ok(())
+    }
+}
--- a/src/common/meta/src/ddl/alter_table/region_request.rs
+++ b/src/common/meta/src/ddl/alter_table/region_request.rs
@@ -0,0 +1,258 @@
+// Copyright 2023 Greptime Team
+//
+// Licensed under the Apache License, Version 2.0 (the "License");
+// you may not use this file except in compliance with the License.
+// You may obtain a copy of the License at
+//
+//     http://www.apache.org/licenses/LICENSE-2.0
+//
+// Unless required by applicable law or agreed to in writing, software
+// distributed under the License is distributed on an "AS IS" BASIS,
+// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+// See the License for the specific language governing permissions and
+// limitations under the License.
+
+use api::v1::alter_expr::Kind;
+use api::v1::region::region_request::Body;
+use api::v1::region::{
+    alter_request, AddColumn, AddColumns, AlterRequest, DropColumn, DropColumns, RegionColumnDef,
+    RegionRequest, RegionRequestHeader,
+};
+use common_telemetry::tracing_context::TracingContext;
+use snafu::OptionExt;
+use store_api::storage::RegionId;
+use table::metadata::RawTableInfo;
+
+use crate::ddl::alter_table::AlterTableProcedure;
+use crate::error::{InvalidProtoMsgSnafu, Result};
+
+impl AlterTableProcedure {
+    /// Makes alter region request.
+    pub(crate) fn make_alter_region_request(&self, region_id: RegionId) -> Result<RegionRequest> {
+        // Safety: Checked in `AlterTableProcedure::new`.
+        let alter_kind = self.data.task.alter_table.kind.as_ref().unwrap();
+        // Safety: checked
+        let table_info = self.data.table_info().unwrap();
+        let kind = create_proto_alter_kind(table_info, alter_kind)?;
+
+        Ok(RegionRequest {
+            header: Some(RegionRequestHeader {
+                tracing_context: TracingContext::from_current_span().to_w3c(),
+                ..Default::default()
+            }),
+            body: Some(Body::Alter(AlterRequest {
+                region_id: region_id.as_u64(),
+                schema_version: table_info.ident.version,
+                kind,
+            })),
+        })
+    }
+}
+
+/// Creates region proto alter kind from `table_info` and `alter_kind`.
+///
+/// Returns the kind and next column id if it adds new columns.
+fn create_proto_alter_kind(
+    table_info: &RawTableInfo,
+    alter_kind: &Kind,
+) -> Result<Option<alter_request::Kind>> {
+    match alter_kind {
+        Kind::AddColumns(x) => {
+            let mut next_column_id = table_info.meta.next_column_id;
+
+            let add_columns = x
+                .add_columns
+                .iter()
+                .map(|add_column| {
+                    let column_def =
+                        add_column
+                            .column_def
+                            .as_ref()
+                            .context(InvalidProtoMsgSnafu {
+                                err_msg: "'column_def' is absent",
+                            })?;
+
+                    let column_id = next_column_id;
+                    next_column_id += 1;
+
+                    let column_def = RegionColumnDef {
+                        column_def: Some(column_def.clone()),
+                        column_id,
+                    };
+
+                    Ok(AddColumn {
+                        column_def: Some(column_def),
+                        location: add_column.location.clone(),
+                    })
+                })
+                .collect::<Result<Vec<_>>>()?;
+
+            Ok(Some(alter_request::Kind::AddColumns(AddColumns {
+                add_columns,
+            })))
+        }
+        Kind::DropColumns(x) => {
+            let drop_columns = x
+                .drop_columns
+                .iter()
+                .map(|x| DropColumn {
+                    name: x.name.clone(),
+                })
+                .collect::<Vec<_>>();
+
+            Ok(Some(alter_request::Kind::DropColumns(DropColumns {
+                drop_columns,
+            })))
+        }
+        Kind::RenameTable(_) => Ok(None),
+    }
+}
+
+#[cfg(test)]
+mod tests {
+    use std::collections::HashMap;
+    use std::sync::Arc;
+
+    use api::v1::add_column_location::LocationType;
+    use api::v1::alter_expr::Kind;
+    use api::v1::region::region_request::Body;
+    use api::v1::region::RegionColumnDef;
+    use api::v1::{
+        region, AddColumn, AddColumnLocation, AddColumns, AlterExpr, ColumnDataType,
+        ColumnDef as PbColumnDef, SemanticType,
+    };
+    use common_catalog::consts::{DEFAULT_CATALOG_NAME, DEFAULT_SCHEMA_NAME};
+    use store_api::storage::RegionId;
+
+    use crate::ddl::alter_table::AlterTableProcedure;
+    use crate::ddl::test_util::columns::TestColumnDefBuilder;
+    use crate::ddl::test_util::create_table::{
+        build_raw_table_info_from_expr, TestCreateTableExprBuilder,
+    };
+    use crate::key::table_route::TableRouteValue;
+    use crate::peer::Peer;
+    use crate::rpc::ddl::AlterTableTask;
+    use crate::rpc::router::{Region, RegionRoute};
+    use crate::test_util::{new_ddl_context, MockDatanodeManager};
+
+    #[tokio::test]
+    async fn test_make_alter_region_request() {
+        let datanode_manager = Arc::new(MockDatanodeManager::new(()));
+        let ddl_context = new_ddl_context(datanode_manager);
+        let cluster_id = 1;
+        let table_id = 1024;
+        let region_id = RegionId::new(table_id, 1);
+        let table_name = "foo";
+
+        let create_table = TestCreateTableExprBuilder::default()
+            .column_defs([
+                TestColumnDefBuilder::default()
+                    .name("ts")
+                    .data_type(ColumnDataType::TimestampMillisecond)
+                    .semantic_type(SemanticType::Timestamp)
+                    .build()
+                    .unwrap()
+                    .into(),
+                TestColumnDefBuilder::default()
+                    .name("host")
+                    .data_type(ColumnDataType::String)
+                    .semantic_type(SemanticType::Tag)
+                    .build()
+                    .unwrap()
+                    .into(),
+                TestColumnDefBuilder::default()
+                    .name("cpu")
+                    .data_type(ColumnDataType::Float64)
+                    .semantic_type(SemanticType::Field)
+                    .build()
+                    .unwrap()
+                    .into(),
+            ])
+            .table_id(table_id)
+            .time_index("ts")
+            .primary_keys(["host".into()])
+            .table_name(table_name)
+            .build()
+            .unwrap()
+            .into();
+        let table_info = build_raw_table_info_from_expr(&create_table);
+
+        // Puts a value to table name key.
+        ddl_context
+            .table_metadata_manager
+            .create_table_metadata(
+                table_info,
+                TableRouteValue::physical(vec![RegionRoute {
+                    region: Region::new_test(region_id),
+                    leader_peer: Some(Peer::empty(1)),
+                    follower_peers: vec![],
+                    leader_status: None,
+                    leader_down_since: None,
+                }]),
+                HashMap::new(),
+            )
+            .await
+            .unwrap();
+
+        let task = AlterTableTask {
+            alter_table: AlterExpr {
+                catalog_name: DEFAULT_CATALOG_NAME.to_string(),
+                schema_name: DEFAULT_SCHEMA_NAME.to_string(),
+                table_name: table_name.to_string(),
+                kind: Some(Kind::AddColumns(AddColumns {
+                    add_columns: vec![AddColumn {
+                        column_def: Some(PbColumnDef {
+                            name: "my_tag3".to_string(),
+                            data_type: ColumnDataType::String as i32,
+                            is_nullable: true,
+                            default_constraint: b"hello".to_vec(),
+                            semantic_type: SemanticType::Tag as i32,
+                            comment: String::new(),
+                            ..Default::default()
+                        }),
+                        location: Some(AddColumnLocation {
+                            location_type: LocationType::After as i32,
+                            after_column_name: "my_tag2".to_string(),
+                        }),
+                    }],
+                })),
+            },
+        };
+
+        let mut procedure =
+            AlterTableProcedure::new(cluster_id, table_id, task, ddl_context).unwrap();
+        procedure.on_prepare().await.unwrap();
+        let Some(Body::Alter(alter_region_request)) =
+            procedure.make_alter_region_request(region_id).unwrap().body
+        else {
+            unreachable!()
+        };
+        assert_eq!(alter_region_request.region_id, region_id.as_u64());
+        assert_eq!(alter_region_request.schema_version, 1);
+        assert_eq!(
+            alter_region_request.kind,
+            Some(region::alter_request::Kind::AddColumns(
+                region::AddColumns {
+                    add_columns: vec![region::AddColumn {
+                        column_def: Some(RegionColumnDef {
+                            column_def: Some(PbColumnDef {
+                                name: "my_tag3".to_string(),
+                                data_type: ColumnDataType::String as i32,
+                                is_nullable: true,
+                                default_constraint: b"hello".to_vec(),
+                                semantic_type: SemanticType::Tag as i32,
+                                comment: String::new(),
+                                ..Default::default()
+                            }),
+                            column_id: 3,
+                        }),
+                        location: Some(AddColumnLocation {
+                            location_type: LocationType::After as i32,
+                            after_column_name: "my_tag2".to_string(),
+                        }),
+                    }]
+                }
+            ))
+        );
+    }
+}
--- a/src/common/meta/src/ddl/alter_table/update_metadata.rs
+++ b/src/common/meta/src/ddl/alter_table/update_metadata.rs
@@ -0,0 +1,87 @@
+// Copyright 2023 Greptime Team
+//
+// Licensed under the Apache License, Version 2.0 (the "License");
+// you may not use this file except in compliance with the License.
+// You may obtain a copy of the License at
+//
+//     http://www.apache.org/licenses/LICENSE-2.0
+//
+// Unless required by applicable law or agreed to in writing, software
+// distributed under the License is distributed on an "AS IS" BASIS,
+// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+// See the License for the specific language governing permissions and
+// limitations under the License.
+
+use common_grpc_expr::alter_expr_to_request;
+use snafu::ResultExt;
+use table::metadata::{RawTableInfo, TableInfo};
+use table::requests::AlterKind;
+
+use crate::ddl::alter_table::AlterTableProcedure;
+use crate::error::{self, Result};
+use crate::key::table_info::TableInfoValue;
+use crate::key::DeserializedValueWithBytes;
+
+impl AlterTableProcedure {
+    /// Builds new_meta
+    pub(crate) fn build_new_table_info(&self, table_info: &RawTableInfo) -> Result<TableInfo> {
+        let table_info =
+            TableInfo::try_from(table_info.clone()).context(error::ConvertRawTableInfoSnafu)?;
+        let table_ref = self.data.table_ref();
+        let alter_expr = self.data.task.alter_table.clone();
+        let request = alter_expr_to_request(self.data.table_id(), alter_expr)
+            .context(error::ConvertAlterTableRequestSnafu)?;
+
+        let new_meta = table_info
+            .meta
+            .builder_with_alter_kind(table_ref.table, &request.alter_kind, false)
+            .context(error::TableSnafu)?
+            .build()
+            .with_context(|_| error::BuildTableMetaSnafu {
+                table_name: table_ref.table,
+            })?;
+
+        let mut new_info = table_info.clone();
+        new_info.meta = new_meta;
+        new_info.ident.version = table_info.ident.version + 1;
+        match request.alter_kind {
+            AlterKind::AddColumns { columns } => {
+                new_info.meta.next_column_id += columns.len() as u32;
+            }
+            AlterKind::RenameTable { new_table_name } => {
+                new_info.name = new_table_name.to_string();
+            }
+            AlterKind::DropColumns { .. } | AlterKind::ChangeColumnTypes { .. } => {}
+        }
+
+        Ok(new_info)
+    }
+
+    /// Updates table metadata for rename table operation.
+    pub(crate) async fn on_update_metadata_for_rename(
+        &self,
+        new_table_name: String,
+        current_table_info_value: &DeserializedValueWithBytes<TableInfoValue>,
+    ) -> Result<()> {
+        let table_metadata_manager = &self.context.table_metadata_manager;
+        table_metadata_manager
+            .rename_table(current_table_info_value, new_table_name)
+            .await?;
+
+        Ok(())
+    }
+
+    /// Updates table metadata for alter table operation.
+    pub(crate) async fn on_update_metadata_for_alter(
+        &self,
+        new_table_info: RawTableInfo,
+        current_table_info_value: &DeserializedValueWithBytes<TableInfoValue>,
+    ) -> Result<()> {
+        let table_metadata_manager = &self.context.table_metadata_manager;
+        table_metadata_manager
+            .update_table_info(current_table_info_value, new_table_info)
+            .await?;
+
+        Ok(())
+    }
+}
--- a/src/common/meta/src/ddl/create_database.rs
+++ b/src/common/meta/src/ddl/create_database.rs
@@ -0,0 +1,152 @@
+// Copyright 2023 Greptime Team
+//
+// Licensed under the Apache License, Version 2.0 (the "License");
+// you may not use this file except in compliance with the License.
+// You may obtain a copy of the License at
+//
+//     http://www.apache.org/licenses/LICENSE-2.0
+//
+// Unless required by applicable law or agreed to in writing, software
+// distributed under the License is distributed on an "AS IS" BASIS,
+// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+// See the License for the specific language governing permissions and
+// limitations under the License.
+
+use std::collections::HashMap;
+
+use async_trait::async_trait;
+use common_procedure::error::{FromJsonSnafu, Result as ProcedureResult, ToJsonSnafu};
+use common_procedure::{Context as ProcedureContext, LockKey, Procedure, Status};
+use serde::{Deserialize, Serialize};
+use snafu::{ensure, ResultExt};
+use strum::AsRefStr;
+
+use crate::ddl::utils::handle_retry_error;
+use crate::ddl::DdlContext;
+use crate::error::{self, Result};
+use crate::key::schema_name::{SchemaNameKey, SchemaNameValue};
+use crate::lock_key::{CatalogLock, SchemaLock};
+
+pub struct CreateDatabaseProcedure {
+    pub context: DdlContext,
+    pub data: CreateDatabaseData,
+}
+
+impl CreateDatabaseProcedure {
+    pub const TYPE_NAME: &'static str = "metasrv-procedure::CreateDatabase";
+
+    pub fn new(
+        catalog: String,
+        schema: String,
+        create_if_not_exists: bool,
+        options: Option<HashMap<String, String>>,
+        context: DdlContext,
+    ) -> Self {
+        Self {
+            context,
+            data: CreateDatabaseData {
+                state: CreateDatabaseState::Prepare,
+                catalog,
+                schema,
+                create_if_not_exists,
+                options,
+            },
+        }
+    }
+
+    pub fn from_json(json: &str, context: DdlContext) -> ProcedureResult<Self> {
+        let data = serde_json::from_str(json).context(FromJsonSnafu)?;
+
+        Ok(Self { context, data })
+    }
+
+    pub async fn on_prepare(&mut self) -> Result<Status> {
+        let exists = self
+            .context
+            .table_metadata_manager
+            .schema_manager()
+            .exists(SchemaNameKey::new(&self.data.catalog, &self.data.schema))
+            .await?;
+
+        if exists && self.data.create_if_not_exists {
+            return Ok(Status::done());
+        }
+
+        ensure!(
+            !exists,
+            error::SchemaAlreadyExistsSnafu {
+                catalog: &self.data.catalog,
+                schema: &self.data.schema,
+            }
+        );
+
+        self.data.state = CreateDatabaseState::CreateMetadata;
+        Ok(Status::executing(true))
+    }
+
+    pub async fn on_create_metadata(&mut self) -> Result<Status> {
+        let value: Option<SchemaNameValue> = self
+            .data
+            .options
+            .as_ref()
+            .map(|hash_map_ref| hash_map_ref.try_into())
+            .transpose()?;
+
+        self.context
+            .table_metadata_manager
+            .schema_manager()
+            .create(
+                SchemaNameKey::new(&self.data.catalog, &self.data.schema),
+                value,
+                self.data.create_if_not_exists,
+            )
+            .await?;
+
+        Ok(Status::done())
+    }
+}
+
+#[async_trait]
+impl Procedure for CreateDatabaseProcedure {
+    fn type_name(&self) -> &str {
+        Self::TYPE_NAME
+    }
+
+    async fn execute(&mut self, _ctx: &ProcedureContext) -> ProcedureResult<Status> {
+        let state = &self.data.state;
+
+        match state {
+            CreateDatabaseState::Prepare => self.on_prepare().await,
+            CreateDatabaseState::CreateMetadata => self.on_create_metadata().await,
+        }
+        .map_err(handle_retry_error)
+    }
+
+    fn dump(&self) -> ProcedureResult<String> {
+        serde_json::to_string(&self.data).context(ToJsonSnafu)
+    }
+
+    fn lock_key(&self) -> LockKey {
+        let lock_key = vec![
+            CatalogLock::Read(&self.data.catalog).into(),
+            SchemaLock::write(&self.data.catalog, &self.data.schema).into(),
+        ];
+
+        LockKey::new(lock_key)
+    }
+}
+
+#[derive(Debug, Clone, Serialize, Deserialize, AsRefStr)]
+pub enum CreateDatabaseState {
+    Prepare,
+    CreateMetadata,
+}
+
+#[derive(Debug, Serialize, Deserialize)]
+pub struct CreateDatabaseData {
+    pub state: CreateDatabaseState,
+    pub catalog: String,
+    pub schema: String,
+    pub create_if_not_exists: bool,
+    pub options: Option<HashMap<String, String>>,
+}
--- a/src/common/meta/src/ddl/create_logical_tables.rs
+++ b/src/common/meta/src/ddl/create_logical_tables.rs
@@ -12,45 +12,32 @@
 // See the License for the specific language governing permissions and
 // limitations under the License.

-use std::collections::HashMap;
-use std::ops::Deref;
+mod check;
+mod metadata;
+mod region_request;
+mod update_metadata;

-use api::v1::region::region_request::Body as PbRegionRequest;
-use api::v1::region::{CreateRequests, RegionRequest, RegionRequestHeader};
 use api::v1::CreateTableExpr;
 use async_trait::async_trait;
 use common_procedure::error::{FromJsonSnafu, Result as ProcedureResult, ToJsonSnafu};
 use common_procedure::{Context as ProcedureContext, LockKey, Procedure, Status};
-use common_telemetry::tracing_context::TracingContext;
-use common_telemetry::{info, warn};
+use common_telemetry::warn;
 use futures_util::future::join_all;
-use itertools::Itertools;
 use serde::{Deserialize, Serialize};
-use snafu::{ensure, OptionExt, ResultExt};
+use snafu::{ensure, ResultExt};
 use store_api::metadata::ColumnMetadata;
 use store_api::metric_engine_consts::ALTER_PHYSICAL_EXTENSION_KEY;
 use store_api::storage::{RegionId, RegionNumber};
 use strum::AsRefStr;
 use table::metadata::{RawTableInfo, TableId};

-use crate::cache_invalidator::Context;
-use crate::ddl::create_table_template::{build_template, CreateRequestBuilder};
-use crate::ddl::utils::{add_peer_context_if_needed, handle_retry_error, region_storage_path};
-use crate::ddl::{physical_table_metadata, DdlContext};
-use crate::error::{
-    DecodeJsonSnafu, MetadataCorruptionSnafu, Result, TableAlreadyExistsSnafu,
-    TableInfoNotFoundSnafu,
-};
-use crate::instruction::CacheIdent;
-use crate::key::table_info::TableInfoValue;
-use crate::key::table_name::TableNameKey;
+use crate::ddl::utils::{add_peer_context_if_needed, handle_retry_error};
+use crate::ddl::DdlContext;
+use crate::error::{DecodeJsonSnafu, MetadataCorruptionSnafu, Result};
 use crate::key::table_route::TableRouteValue;
-use crate::key::DeserializedValueWithBytes;
 use crate::lock_key::{CatalogLock, SchemaLock, TableLock, TableNameLock};
-use crate::peer::Peer;
 use crate::rpc::ddl::CreateTableTask;
-use crate::rpc::router::{find_leader_regions, find_leaders, RegionRoute};
-use crate::table_name::TableName;
+use crate::rpc::router::{find_leaders, RegionRoute};
 use crate::{metrics, ClusterId};

 pub struct CreateLogicalTablesProcedure {
@@ -67,17 +54,18 @@ impl CreateLogicalTablesProcedure {
        physical_table_id: TableId,
        context: DdlContext,
    ) -> Self {
-        let len = tasks.len();
-        let data = CreateTablesData {
-            cluster_id,
-            state: CreateTablesState::Prepare,
-            tasks,
-            table_ids_already_exists: vec![None; len],
-            physical_table_id,
-            physical_region_numbers: vec![],
-            physical_columns: vec![],
-        };
-        Self { context, data }
+        Self {
+            context,
+            data: CreateTablesData {
+                cluster_id,
+                state: CreateTablesState::Prepare,
+                tasks,
+                table_ids_already_exists: vec![],
+                physical_table_id,
+                physical_region_numbers: vec![],
+                physical_columns: vec![],
+            },
+        }
    }

    pub fn from_json(json: &str, context: DdlContext) -> ProcedureResult<Self> {
@@ -96,91 +84,45 @@ impl CreateLogicalTablesProcedure {
    /// - Failed to check whether tables exist.
    /// - One of logical tables has existing, and the table creation task without setting `create_if_not_exists`.
    pub(crate) async fn on_prepare(&mut self) -> Result<Status> {
-        let manager = &self.context.table_metadata_manager;
-
+        self.check_input_tasks()?;
        // Sets physical region numbers
-        let physical_table_id = self.data.physical_table_id();
-        let physical_region_numbers = manager
-            .table_route_manager()
-            .get_physical_table_route(physical_table_id)
-            .await
-            .map(|(_, route)| TableRouteValue::Physical(route).region_numbers())?;
-        self.data
-            .set_physical_region_numbers(physical_region_numbers);
-
+        self.fill_physical_table_info().await?;
        // Checks if the tables exist
-        let table_name_keys = self
-            .data
-            .all_create_table_exprs()
-            .iter()
-            .map(|expr| TableNameKey::new(&expr.catalog_name, &expr.schema_name, &expr.table_name))
-            .collect::<Vec<_>>();
-        let already_exists_tables_ids = manager
-            .table_name_manager()
-            .batch_get(table_name_keys)
-            .await?
-            .iter()
-            .map(|x| x.map(|x| x.table_id()))
-            .collect::<Vec<_>>();
-
-        // Validates the tasks
-        let tasks = &mut self.data.tasks;
-        for (task, table_id) in tasks.iter().zip(already_exists_tables_ids.iter()) {
-            if table_id.is_some() {
-                // If a table already exists, we just ignore it.
-                ensure!(
-                    task.create_table.create_if_not_exists,
-                    TableAlreadyExistsSnafu {
-                        table_name: task.create_table.table_name.to_string(),
-                    }
-                );
-                continue;
-            }
-        }
+        self.check_tables_already_exist().await?;

        // If all tables already exist, returns the table_ids.
-        if already_exists_tables_ids.iter().all(Option::is_some) {
+        if self
+            .data
+            .table_ids_already_exists
+            .iter()
+            .all(Option::is_some)
+        {
            return Ok(Status::done_with_output(
-                already_exists_tables_ids
-                    .into_iter()
+                self.data
+                    .table_ids_already_exists
+                    .drain(..)
                    .flatten()
                    .collect::<Vec<_>>(),
            ));
        }

        // Allocates table ids and sort columns on their names.
-        for (task, table_id) in tasks.iter_mut().zip(already_exists_tables_ids.iter()) {
-            let table_id = if let Some(table_id) = table_id {
-                *table_id
-            } else {
-                self.context
-                    .table_metadata_allocator
-                    .allocate_table_id(task)
-                    .await?
-            };
-            task.set_table_id(table_id);
+        self.allocate_table_ids().await?;

-            // sort columns in task
-            task.sort_columns();
-        }
-
-        self.data
-            .set_table_ids_already_exists(already_exists_tables_ids);
        self.data.state = CreateTablesState::DatanodeCreateRegions;
        Ok(Status::executing(true))
    }

    pub async fn on_datanode_create_regions(&mut self) -> Result<Status> {
-        let physical_table_id = self.data.physical_table_id();
        let (_, physical_table_route) = self
            .context
            .table_metadata_manager
            .table_route_manager()
-            .get_physical_table_route(physical_table_id)
+            .get_physical_table_route(self.data.physical_table_id)
            .await?;
-        let region_routes = &physical_table_route.region_routes;

-        self.create_regions(region_routes).await
+        self.create_regions(&physical_table_route.region_routes)
+            .await
    }

    /// Creates table metadata for logical tables and update corresponding physical
@@ -189,179 +131,54 @@ impl CreateLogicalTablesProcedure {
    /// Abort(not-retry):
    /// - Failed to create table metadata.
    pub async fn on_create_metadata(&mut self) -> Result<Status> {
-        let manager = &self.context.table_metadata_manager;
-        let physical_table_id = self.data.physical_table_id();
-        let remaining_tasks = self.data.remaining_tasks();
-        let num_tables = remaining_tasks.len();
-
-        if num_tables > 0 {
-            let chunk_size = manager.create_logical_tables_metadata_chunk_size();
-            if num_tables > chunk_size {
-                let chunks = remaining_tasks
-                    .into_iter()
-                    .chunks(chunk_size)
-                    .into_iter()
-                    .map(|chunk| chunk.collect::<Vec<_>>())
-                    .collect::<Vec<_>>();
-                for chunk in chunks {
-                    manager.create_logical_tables_metadata(chunk).await?;
-                }
-            } else {
-                manager
-                    .create_logical_tables_metadata(remaining_tasks)
-                    .await?;
-            }
-        }
-
-        // The `table_id` MUST be collected after the [Prepare::Prepare],
-        // ensures the all `table_id`s have been allocated.
-        let table_ids = self
-            .data
-            .tasks
-            .iter()
-            .map(|task| task.table_info.ident.table_id)
-            .collect::<Vec<_>>();
-
-        if !self.data.physical_columns.is_empty() {
-            // fetch old physical table's info
-            let physical_table_info = self
-                .context
-                .table_metadata_manager
-                .table_info_manager()
-                .get(self.data.physical_table_id)
-                .await?
-                .with_context(|| TableInfoNotFoundSnafu {
-                    table: format!("table id - {}", self.data.physical_table_id),
-                })?;
-
-            // generate new table info
-            let new_table_info = self
-                .data
-                .build_new_physical_table_info(&physical_table_info);
-
-            let physical_table_name = TableName::new(
-                &new_table_info.catalog_name,
-                &new_table_info.schema_name,
-                &new_table_info.name,
-            );
-
-            // update physical table's metadata
-            self.context
-                .table_metadata_manager
-                .update_table_info(physical_table_info, new_table_info)
-                .await?;
-
-            // invalid table cache
-            self.context
-                .cache_invalidator
-                .invalidate(
-                    &Context::default(),
-                    vec![
-                        CacheIdent::TableId(self.data.physical_table_id),
-                        CacheIdent::TableName(physical_table_name),
-                    ],
-                )
-                .await?;
-        } else {
-            warn!("No physical columns found, leaving the physical table's schema unchanged");
-        }
-
-        info!("Created {num_tables} tables {table_ids:?} metadata for physical table {physical_table_id}");
+        self.update_physical_table_metadata().await?;
+        let table_ids = self.create_logical_tables_metadata().await?;

        Ok(Status::done_with_output(table_ids))
    }

-    fn create_region_request_builder(
-        &self,
-        physical_table_id: TableId,
-        task: &CreateTableTask,
-    ) -> Result<CreateRequestBuilder> {
-        let create_expr = &task.create_table;
-        let template = build_template(create_expr)?;
-        Ok(CreateRequestBuilder::new(template, Some(physical_table_id)))
-    }
-
-    fn one_datanode_region_requests(
-        &self,
-        datanode: &Peer,
-        region_routes: &[RegionRoute],
-    ) -> Result<CreateRequests> {
-        let create_tables_data = &self.data;
-        let tasks = &create_tables_data.tasks;
-        let physical_table_id = create_tables_data.physical_table_id();
-        let regions = find_leader_regions(region_routes, datanode);
-        let mut requests = Vec::with_capacity(tasks.len() * regions.len());
-
-        for task in tasks {
-            let create_table_expr = &task.create_table;
-            let catalog = &create_table_expr.catalog_name;
-            let schema = &create_table_expr.schema_name;
-            let logical_table_id = task.table_info.ident.table_id;
-            let storage_path = region_storage_path(catalog, schema);
-            let request_builder = self.create_region_request_builder(physical_table_id, task)?;
-
-            for region_number in &regions {
-                let region_id = RegionId::new(logical_table_id, *region_number);
-                let create_region_request =
-                    request_builder.build_one(region_id, storage_path.clone(), &HashMap::new())?;
-                requests.push(create_region_request);
-            }
-        }
-
-        Ok(CreateRequests { requests })
-    }
-
    async fn create_regions(&mut self, region_routes: &[RegionRoute]) -> Result<Status> {
        let leaders = find_leaders(region_routes);
        let mut create_region_tasks = Vec::with_capacity(leaders.len());

-        for datanode in leaders {
-            let requester = self.context.datanode_manager.datanode(&datanode).await;
-            let creates = self.one_datanode_region_requests(&datanode, region_routes)?;
-            let request = RegionRequest {
-                header: Some(RegionRequestHeader {
-                    tracing_context: TracingContext::from_current_span().to_w3c(),
-                    ..Default::default()
-                }),
-                body: Some(PbRegionRequest::Creates(creates)),
-            };
+        for peer in leaders {
+            let requester = self.context.datanode_manager.datanode(&peer).await;
+            let request = self.make_request(&peer, region_routes)?;
+
            create_region_tasks.push(async move {
                requester
                    .handle(request)
                    .await
-                    .map_err(add_peer_context_if_needed(datanode))
+                    .map_err(add_peer_context_if_needed(peer))
            });
        }

-        // collect response from datanodes
-        let raw_schemas = join_all(create_region_tasks)
+        // Collects response from datanodes.
+        let phy_raw_schemas = join_all(create_region_tasks)
            .await
            .into_iter()
-            .map(|response| {
-                response.map(|mut response| response.extension.remove(ALTER_PHYSICAL_EXTENSION_KEY))
-            })
+            .map(|res| res.map(|mut res| res.extension.remove(ALTER_PHYSICAL_EXTENSION_KEY)))
            .collect::<Result<Vec<_>>>()?;

-        if raw_schemas.is_empty() {
+        if phy_raw_schemas.is_empty() {
            self.data.state = CreateTablesState::CreateMetadata;
            return Ok(Status::executing(false));
        }

-        // verify all datanodes return the same raw schemas
-        // Safety: previous check ensures this vector is not empty.
-        let first = raw_schemas.first().unwrap();
+        // Verify all the physical schemas are the same
+        // Safety: previous check ensures this vec is not empty
+        let first = phy_raw_schemas.first().unwrap();
        ensure!(
-            raw_schemas.iter().all(|x| x == first),
+            phy_raw_schemas.iter().all(|x| x == first),
            MetadataCorruptionSnafu {
-                err_msg: "Raw schemas from datanodes are not the same"
+                err_msg: "The physical schemas from datanodes are not the same."
            }
        );

-        // decode raw schemas and store it
-        if let Some(raw_schema) = first {
-            let physical_columns =
-                ColumnMetadata::decode_list(raw_schema).context(DecodeJsonSnafu)?;
-            self.data.physical_columns = physical_columns;
+        // Decodes the physical raw schemas
+        if let Some(phy_raw_schemas) = first {
+            self.data.physical_columns =
+                ColumnMetadata::decode_list(phy_raw_schemas).context(DecodeJsonSnafu)?;
        } else {
            warn!("creating logical table result doesn't contains extension key `{ALTER_PHYSICAL_EXTENSION_KEY}`,leaving the physical table's schema unchanged");
        }
@@ -405,7 +222,7 @@ impl Procedure for CreateLogicalTablesProcedure {
        let table_ref = self.data.tasks[0].table_ref();
        lock_key.push(CatalogLock::Read(table_ref.catalog).into());
        lock_key.push(SchemaLock::read(table_ref.catalog, table_ref.schema).into());
-        lock_key.push(TableLock::Write(self.data.physical_table_id()).into());
+        lock_key.push(TableLock::Write(self.data.physical_table_id).into());

        for task in &self.data.tasks {
            lock_key.push(
@@ -437,18 +254,6 @@ impl CreateTablesData {
        &self.state
    }

-    fn physical_table_id(&self) -> TableId {
-        self.physical_table_id
-    }
-
-    fn set_physical_region_numbers(&mut self, physical_region_numbers: Vec<RegionNumber>) {
-        self.physical_region_numbers = physical_region_numbers;
-    }
-
-    fn set_table_ids_already_exists(&mut self, table_ids_already_exists: Vec<Option<TableId>>) {
-        self.table_ids_already_exists = table_ids_already_exists;
-    }
-
    fn all_create_table_exprs(&self) -> Vec<&CreateTableExpr> {
        self.tasks
            .iter()
@@ -480,21 +285,6 @@ impl CreateTablesData {
            })
            .collect::<Vec<_>>()
    }
-
-    /// Generate the new physical table info.
-    ///
-    /// This method will consumes the physical columns.
-    fn build_new_physical_table_info(
-        &mut self,
-        old_table_info: &DeserializedValueWithBytes<TableInfoValue>,
-    ) -> RawTableInfo {
-        let raw_table_info = old_table_info.deref().table_info.clone();
-
-        physical_table_metadata::build_new_physical_table_info(
-            raw_table_info,
-            &self.physical_columns,
-        )
-    }
 }

 #[derive(Debug, Clone, Serialize, Deserialize, AsRefStr)]
--- a/src/common/meta/src/ddl/create_logical_tables/check.rs
+++ b/src/common/meta/src/ddl/create_logical_tables/check.rs
@@ -0,0 +1,81 @@
+// Copyright 2023 Greptime Team
+//
+// Licensed under the Apache License, Version 2.0 (the "License");
+// you may not use this file except in compliance with the License.
+// You may obtain a copy of the License at
+//
+//     http://www.apache.org/licenses/LICENSE-2.0
+//
+// Unless required by applicable law or agreed to in writing, software
+// distributed under the License is distributed on an "AS IS" BASIS,
+// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+// See the License for the specific language governing permissions and
+// limitations under the License.
+
+use snafu::ensure;
+
+use crate::ddl::create_logical_tables::CreateLogicalTablesProcedure;
+use crate::error::{CreateLogicalTablesInvalidArgumentsSnafu, Result, TableAlreadyExistsSnafu};
+use crate::key::table_name::TableNameKey;
+
+impl CreateLogicalTablesProcedure {
+    pub(crate) fn check_input_tasks(&self) -> Result<()> {
+        self.check_schema()?;
+
+        Ok(())
+    }
+
+    pub(crate) async fn check_tables_already_exist(&mut self) -> Result<()> {
+        let table_name_keys = self
+            .data
+            .all_create_table_exprs()
+            .iter()
+            .map(|expr| TableNameKey::new(&expr.catalog_name, &expr.schema_name, &expr.table_name))
+            .collect::<Vec<_>>();
+        let table_ids_already_exists = self
+            .context
+            .table_metadata_manager
+            .table_name_manager()
+            .batch_get(table_name_keys)
+            .await?
+            .iter()
+            .map(|x| x.map(|x| x.table_id()))
+            .collect::<Vec<_>>();
+
+        self.data.table_ids_already_exists = table_ids_already_exists;
+
+        // Validates the tasks
+        let tasks = &mut self.data.tasks;
+        for (task, table_id) in tasks.iter().zip(self.data.table_ids_already_exists.iter()) {
+            if table_id.is_some() {
+                // If a table already exists, we just ignore it.
+                ensure!(
+                    task.create_table.create_if_not_exists,
+                    TableAlreadyExistsSnafu {
+                        table_name: task.create_table.table_name.to_string(),
+                    }
+                );
+                continue;
+            }
+        }
+
+        Ok(())
+    }
+
+    // Checks if the schemas of the tasks are the same
+    fn check_schema(&self) -> Result<()> {
+        let is_same_schema = self.data.tasks.windows(2).all(|pair| {
+            pair[0].create_table.catalog_name == pair[1].create_table.catalog_name
+                && pair[0].create_table.schema_name == pair[1].create_table.schema_name
+        });
+
+        ensure!(
+            is_same_schema,
+            CreateLogicalTablesInvalidArgumentsSnafu {
+                err_msg: "Schemas of the tasks are not the same"
+            }
+        );
+
+        Ok(())
+    }
+}
--- a/src/common/meta/src/ddl/create_logical_tables/metadata.rs
+++ b/src/common/meta/src/ddl/create_logical_tables/metadata.rs
@@ -0,0 +1,57 @@
+// Copyright 2023 Greptime Team
+//
+// Licensed under the Apache License, Version 2.0 (the "License");
+// you may not use this file except in compliance with the License.
+// You may obtain a copy of the License at
+//
+//     http://www.apache.org/licenses/LICENSE-2.0
+//
+// Unless required by applicable law or agreed to in writing, software
+// distributed under the License is distributed on an "AS IS" BASIS,
+// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+// See the License for the specific language governing permissions and
+// limitations under the License.
+
+use crate::ddl::create_logical_tables::CreateLogicalTablesProcedure;
+use crate::error::Result;
+use crate::key::table_route::TableRouteValue;
+
+impl CreateLogicalTablesProcedure {
+    pub(crate) async fn fill_physical_table_info(&mut self) -> Result<()> {
+        let physical_region_numbers = self
+            .context
+            .table_metadata_manager
+            .table_route_manager()
+            .get_physical_table_route(self.data.physical_table_id)
+            .await
+            .map(|(_, route)| TableRouteValue::Physical(route).region_numbers())?;
+
+        self.data.physical_region_numbers = physical_region_numbers;
+
+        Ok(())
+    }
+
+    pub(crate) async fn allocate_table_ids(&mut self) -> Result<()> {
+        for (task, table_id) in self
+            .data
+            .tasks
+            .iter_mut()
+            .zip(self.data.table_ids_already_exists.iter())
+        {
+            let table_id = if let Some(table_id) = table_id {
+                *table_id
+            } else {
+                self.context
+                    .table_metadata_allocator
+                    .allocate_table_id(task)
+                    .await?
+            };
+            task.set_table_id(table_id);
+
+            // sort columns in task
+            task.sort_columns();
+        }
+
+        Ok(())
+    }
+}
--- a/src/common/meta/src/ddl/create_logical_tables/region_request.rs
+++ b/src/common/meta/src/ddl/create_logical_tables/region_request.rs
@@ -0,0 +1,74 @@
+// Copyright 2023 Greptime Team
+//
+// Licensed under the Apache License, Version 2.0 (the "License");
+// you may not use this file except in compliance with the License.
+// You may obtain a copy of the License at
+//
+//     http://www.apache.org/licenses/LICENSE-2.0
+//
+// Unless required by applicable law or agreed to in writing, software
+// distributed under the License is distributed on an "AS IS" BASIS,
+// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+// See the License for the specific language governing permissions and
+// limitations under the License.
+
+use std::collections::HashMap;
+
+use api::v1::region::{region_request, CreateRequests, RegionRequest, RegionRequestHeader};
+use common_telemetry::tracing_context::TracingContext;
+use store_api::storage::RegionId;
+
+use crate::ddl::create_logical_tables::CreateLogicalTablesProcedure;
+use crate::ddl::create_table_template::{build_template, CreateRequestBuilder};
+use crate::ddl::utils::region_storage_path;
+use crate::error::Result;
+use crate::peer::Peer;
+use crate::rpc::ddl::CreateTableTask;
+use crate::rpc::router::{find_leader_regions, RegionRoute};
+
+impl CreateLogicalTablesProcedure {
+    pub(crate) fn make_request(
+        &self,
+        peer: &Peer,
+        region_routes: &[RegionRoute],
+    ) -> Result<RegionRequest> {
+        let tasks = &self.data.tasks;
+        let regions_on_this_peer = find_leader_regions(region_routes, peer);
+        let mut requests = Vec::with_capacity(tasks.len() * regions_on_this_peer.len());
+        for task in tasks {
+            let create_table_expr = &task.create_table;
+            let catalog = &create_table_expr.catalog_name;
+            let schema = &create_table_expr.schema_name;
+            let logical_table_id = task.table_info.ident.table_id;
+            let storage_path = region_storage_path(catalog, schema);
+            let request_builder = self.create_region_request_builder(task)?;
+
+            for region_number in &regions_on_this_peer {
+                let region_id = RegionId::new(logical_table_id, *region_number);
+                let one_region_request =
+                    request_builder.build_one(region_id, storage_path.clone(), &HashMap::new())?;
+                requests.push(one_region_request);
+            }
+        }
+
+        Ok(RegionRequest {
+            header: Some(RegionRequestHeader {
+                tracing_context: TracingContext::from_current_span().to_w3c(),
+                ..Default::default()
+            }),
+            body: Some(region_request::Body::Creates(CreateRequests { requests })),
+        })
+    }
+
+    fn create_region_request_builder(
+        &self,
+        task: &CreateTableTask,
+    ) -> Result<CreateRequestBuilder> {
+        let create_expr = &task.create_table;
+        let template = build_template(create_expr)?;
+        Ok(CreateRequestBuilder::new(
+            template,
+            Some(self.data.physical_table_id),
+        ))
+    }
+}
--- a/src/common/meta/src/ddl/create_logical_tables/update_metadata.rs
+++ b/src/common/meta/src/ddl/create_logical_tables/update_metadata.rs
@@ -0,0 +1,128 @@
+// Copyright 2023 Greptime Team
+//
+// Licensed under the Apache License, Version 2.0 (the "License");
+// you may not use this file except in compliance with the License.
+// You may obtain a copy of the License at
+//
+//     http://www.apache.org/licenses/LICENSE-2.0
+//
+// Unless required by applicable law or agreed to in writing, software
+// distributed under the License is distributed on an "AS IS" BASIS,
+// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+// See the License for the specific language governing permissions and
+// limitations under the License.
+
+use std::ops::Deref;
+
+use common_telemetry::{info, warn};
+use itertools::Itertools;
+use snafu::OptionExt;
+use table::metadata::TableId;
+
+use crate::cache_invalidator::Context;
+use crate::ddl::create_logical_tables::CreateLogicalTablesProcedure;
+use crate::ddl::physical_table_metadata;
+use crate::error::{Result, TableInfoNotFoundSnafu};
+use crate::instruction::CacheIdent;
+use crate::table_name::TableName;
+
+impl CreateLogicalTablesProcedure {
+    pub(crate) async fn update_physical_table_metadata(&mut self) -> Result<()> {
+        if self.data.physical_columns.is_empty() {
+            warn!("No physical columns found, leaving the physical table's schema unchanged when creating logical tables");
+            return Ok(());
+        }
+
+        // Fetches old physical table's info
+        let physical_table_info = self
+            .context
+            .table_metadata_manager
+            .table_info_manager()
+            .get(self.data.physical_table_id)
+            .await?
+            .with_context(|| TableInfoNotFoundSnafu {
+                table: format!("table id - {}", self.data.physical_table_id),
+            })?;
+
+        // Generates new table info
+        let raw_table_info = physical_table_info.deref().table_info.clone();
+
+        let new_table_info = physical_table_metadata::build_new_physical_table_info(
+            raw_table_info,
+            &self.data.physical_columns,
+        );
+
+        let physical_table_name = TableName::new(
+            &new_table_info.catalog_name,
+            &new_table_info.schema_name,
+            &new_table_info.name,
+        );
+
+        // Update physical table's metadata
+        self.context
+            .table_metadata_manager
+            .update_table_info(&physical_table_info, new_table_info)
+            .await?;
+
+        // Invalid physical table cache
+        self.context
+            .cache_invalidator
+            .invalidate(
+                &Context::default(),
+                vec![
+                    CacheIdent::TableId(self.data.physical_table_id),
+                    CacheIdent::TableName(physical_table_name),
+                ],
+            )
+            .await?;
+
+        Ok(())
+    }
+
+    pub(crate) async fn create_logical_tables_metadata(&mut self) -> Result<Vec<TableId>> {
+        let remaining_tasks = self.data.remaining_tasks();
+        let num_tables = remaining_tasks.len();
+
+        if num_tables > 0 {
+            let chunk_size = self
+                .context
+                .table_metadata_manager
+                .create_logical_tables_metadata_chunk_size();
+            if num_tables > chunk_size {
+                let chunks = remaining_tasks
+                    .into_iter()
+                    .chunks(chunk_size)
+                    .into_iter()
+                    .map(|chunk| chunk.collect::<Vec<_>>())
+                    .collect::<Vec<_>>();
+                for chunk in chunks {
+                    self.context
+                        .table_metadata_manager
+                        .create_logical_tables_metadata(chunk)
+                        .await?;
+                }
+            } else {
+                self.context
+                    .table_metadata_manager
+                    .create_logical_tables_metadata(remaining_tasks)
+                    .await?;
+            }
+        }
+
+        // The `table_id` MUST be collected after the [Prepare::Prepare],
+        // ensures the all `table_id`s have been allocated.
+        let table_ids = self
+            .data
+            .tasks
+            .iter()
+            .map(|task| task.table_info.ident.table_id)
+            .collect::<Vec<_>>();
+
+        info!(
+            "Created {num_tables} tables {table_ids:?} metadata for physical table {}",
+            self.data.physical_table_id
+        );
+
+        Ok(table_ids)
+    }
+}
--- a/Show More
+++ b/Show More