feat: update dashboard to v0.4.10 (#3663 )

Co-authored-by: ZonaHex <ZonaHex@users.noreply.github.com>
feat: cluster information (#3631 )
2025-12-22 22:20:02 +00:00 · 2024-04-08 16:35:41 +08:00 · 2024-04-08 07:48:36 +00:00 · 2024-04-08 07:28:55 +00:00 · 2024-04-08 07:05:55 +00:00 · 2024-04-08 06:33:29 +00:00
589 changed files with 30783 additions and 7788 deletions
--- a/.editorconfig
+++ b/.editorconfig
@@ -0,0 +1,10 @@
+root = true
+
+[*]
+end_of_line = lf
+indent_style = space
+insert_final_newline = true
+trim_trailing_whitespace = true
+
+[{Makefile,**.mk}]
+indent_style = tab
--- a/.env.example
+++ b/.env.example
@@ -21,3 +21,6 @@ GT_GCS_CREDENTIAL_PATH = GCS credential path
 GT_GCS_ENDPOINT = GCS end point
 # Settings for kafka wal test
 GT_KAFKA_ENDPOINTS = localhost:9092
+
+# Setting for fuzz tests
+GT_MYSQL_ADDR = localhost:4002
--- a/.github/actions/build-windows-artifacts/action.yml
+++ b/.github/actions/build-windows-artifacts/action.yml
@@ -70,7 +70,7 @@ runs:

    - name: Build greptime binary
      shell: pwsh
-      run: cargo build --profile ${{ inputs.cargo-profile }} --features ${{ inputs.features }} --target ${{ inputs.arch }}
+      run: cargo build --profile ${{ inputs.cargo-profile }} --features ${{ inputs.features }} --target ${{ inputs.arch }} --bin greptime

    - name: Upload artifacts
      uses: ./.github/actions/upload-artifacts
--- a/.github/actions/fuzz-test/action.yaml
+++ b/.github/actions/fuzz-test/action.yaml
@@ -0,0 +1,13 @@
+name: Fuzz Test
+description: 'Fuzz test given setup and service'
+inputs:
+  target:
+    description: "The fuzz target to test"
+runs:
+  using: composite
+  steps:
+  - name: Run Fuzz Test
+    shell: bash
+    run: cargo fuzz run ${{ inputs.target }} --fuzz-dir tests-fuzz -D -s none -- -max_total_time=120
+    env:
+      GT_MYSQL_ADDR: 127.0.0.1:4002
--- a/.github/workflows/apidoc.yml
+++ b/.github/workflows/apidoc.yml
@@ -40,3 +40,4 @@ jobs:
      uses: JamesIves/github-pages-deploy-action@v4
      with:
        folder: target/doc
+        single-commit: true
--- a/.github/workflows/develop.yml
+++ b/.github/workflows/develop.yml
@@ -102,7 +102,7 @@ jobs:
          shared-key: "build-binaries"
      - name: Build greptime binaries
        shell: bash
-        run: cargo build
+        run: cargo build --bin greptime --bin sqlness-runner
      - name: Pack greptime binaries
        shell: bash
        run: |
@@ -117,6 +117,46 @@ jobs:
          artifacts-dir: bins
          version: current

+  fuzztest:
+    name: Fuzz Test
+    needs: build
+    runs-on: ubuntu-latest
+    strategy:
+      matrix:
+        target: [ "fuzz_create_table", "fuzz_alter_table" ]
+    steps:
+      - uses: actions/checkout@v4
+      - uses: arduino/setup-protoc@v3
+      - uses: dtolnay/rust-toolchain@master
+        with:
+          toolchain: ${{ env.RUST_TOOLCHAIN }}
+      - name: Rust Cache
+        uses: Swatinem/rust-cache@v2
+        with:
+          # Shares across multiple jobs
+          shared-key: "fuzz-test-targets"
+      - name: Set Rust Fuzz
+        shell: bash
+        run: |
+          sudo apt update && sudo apt install -y libfuzzer-14-dev
+          cargo install cargo-fuzz
+      - name: Download pre-built binaries
+        uses: actions/download-artifact@v4
+        with:
+          name: bins
+          path: .
+      - name: Unzip binaries
+        run: tar -xvf ./bins.tar.gz
+      - name: Run GreptimeDB
+        run: |
+          ./bins/greptime standalone start&
+      - name: Fuzz Test
+        uses: ./.github/actions/fuzz-test
+        env:
+          CUSTOM_LIBFUZZER_PATH: /usr/lib/llvm-14/lib/libFuzzer.a
+        with:
+          target: ${{ matrix.target }}
+
  sqlness:
    name: Sqlness Test
    needs: build
@@ -239,6 +279,10 @@ jobs:
        with:
          # Shares cross multiple jobs
          shared-key: "coverage-test"
+      - name: Docker Cache
+        uses: ScribeMD/docker-cache@0.3.7
+        with:
+          key: docker-${{ runner.os }}-coverage
      - name: Install latest nextest release
        uses: taiki-e/install-action@nextest
      - name: Install cargo-llvm-cov
--- a/.github/workflows/license.yaml
+++ b/.github/workflows/license.yaml
@@ -13,4 +13,4 @@ jobs:
    steps:
    - uses: actions/checkout@v4
    - name: Check License Header
-      uses: korandoru/hawkeye@v4
+      uses: korandoru/hawkeye@v5
--- a/.github/workflows/release.yml
+++ b/.github/workflows/release.yml
@@ -91,7 +91,7 @@ env:
  # The scheduled version is '${{ env.NEXT_RELEASE_VERSION }}-nightly-YYYYMMDD', like v0.2.0-nigthly-20230313;
  NIGHTLY_RELEASE_PREFIX: nightly
  # Note: The NEXT_RELEASE_VERSION should be modified manually by every formal release.
-  NEXT_RELEASE_VERSION: v0.7.0
+  NEXT_RELEASE_VERSION: v0.8.0

 jobs:
  allocate-runners:
@@ -288,7 +288,7 @@ jobs:
      - name: Set build windows result
        id: set-build-windows-result
        run: |
-          echo "build-windows-result=success" >> $GITHUB_OUTPUT    
+          echo "build-windows-result=success" >> $Env:GITHUB_OUTPUT

  release-images-to-dockerhub:
    name: Build and push images to DockerHub
--- a/.github/workflows/unassign.yml
+++ b/.github/workflows/unassign.yml
@@ -0,0 +1,21 @@
+name: Auto Unassign
+on:
+  schedule:
+    - cron: '4 2 * * *'
+  workflow_dispatch:
+
+permissions:
+  contents: read
+  issues: write
+  pull-requests: write
+
+jobs:
+  auto-unassign:
+    name: Auto Unassign
+    runs-on: ubuntu-latest
+    steps:
+      - name: Auto Unassign
+        uses: tisonspieces/auto-unassign@main
+        with:
+          token: ${{ secrets.GH_PERSONAL_ACCESS_TOKEN }}
+          repository: ${{ github.repository }}
--- a/.gitignore
+++ b/.gitignore
@@ -46,3 +46,7 @@ benchmarks/data
 *.code-workspace

 venv/
+
+# Fuzz tests 
+tests-fuzz/artifacts/
+tests-fuzz/corpus/
--- a/CODE_OF_CONDUCT.md
+++ b/CODE_OF_CONDUCT.md
@@ -1,132 +0,0 @@
-# Contributor Covenant Code of Conduct
-
-## Our Pledge
-
-We as members, contributors, and leaders pledge to make participation in our
-community a harassment-free experience for everyone, regardless of age, body
-size, visible or invisible disability, ethnicity, sex characteristics, gender
-identity and expression, level of experience, education, socio-economic status,
-nationality, personal appearance, race, caste, color, religion, or sexual
-identity and orientation.
-
-We pledge to act and interact in ways that contribute to an open, welcoming,
-diverse, inclusive, and healthy community.
-
-## Our Standards
-
-Examples of behavior that contributes to a positive environment for our
-community include:
-
-* Demonstrating empathy and kindness toward other people
-* Being respectful of differing opinions, viewpoints, and experiences
-* Giving and gracefully accepting constructive feedback
-* Accepting responsibility and apologizing to those affected by our mistakes,
-  and learning from the experience
-* Focusing on what is best not just for us as individuals, but for the overall
-  community
-
-Examples of unacceptable behavior include:
-
-* The use of sexualized language or imagery, and sexual attention or advances of
-  any kind
-* Trolling, insulting or derogatory comments, and personal or political attacks
-* Public or private harassment
-* Publishing others' private information, such as a physical or email address,
-  without their explicit permission
-* Other conduct which could reasonably be considered inappropriate in a
-  professional setting
-
-## Enforcement Responsibilities
-
-Community leaders are responsible for clarifying and enforcing our standards of
-acceptable behavior and will take appropriate and fair corrective action in
-response to any behavior that they deem inappropriate, threatening, offensive,
-or harmful.
-
-Community leaders have the right and responsibility to remove, edit, or reject
-comments, commits, code, wiki edits, issues, and other contributions that are
-not aligned to this Code of Conduct, and will communicate reasons for moderation
-decisions when appropriate.
-
-## Scope
-
-This Code of Conduct applies within all community spaces, and also applies when
-an individual is officially representing the community in public spaces.
-Examples of representing our community include using an official e-mail address,
-posting via an official social media account, or acting as an appointed
-representative at an online or offline event.
-
-## Enforcement
-
-Instances of abusive, harassing, or otherwise unacceptable behavior may be
-reported to the community leaders responsible for enforcement at
-info@greptime.com.
-All complaints will be reviewed and investigated promptly and fairly.
-
-All community leaders are obligated to respect the privacy and security of the
-reporter of any incident.
-
-## Enforcement Guidelines
-
-Community leaders will follow these Community Impact Guidelines in determining
-the consequences for any action they deem in violation of this Code of Conduct:
-
-### 1. Correction
-
-**Community Impact**: Use of inappropriate language or other behavior deemed
-unprofessional or unwelcome in the community.
-
-**Consequence**: A private, written warning from community leaders, providing
-clarity around the nature of the violation and an explanation of why the
-behavior was inappropriate. A public apology may be requested.
-
-### 2. Warning
-
-**Community Impact**: A violation through a single incident or series of
-actions.
-
-**Consequence**: A warning with consequences for continued behavior. No
-interaction with the people involved, including unsolicited interaction with
-those enforcing the Code of Conduct, for a specified period of time. This
-includes avoiding interactions in community spaces as well as external channels
-like social media. Violating these terms may lead to a temporary or permanent
-ban.
-
-### 3. Temporary Ban
-
-**Community Impact**: A serious violation of community standards, including
-sustained inappropriate behavior.
-
-**Consequence**: A temporary ban from any sort of interaction or public
-communication with the community for a specified period of time. No public or
-private interaction with the people involved, including unsolicited interaction
-with those enforcing the Code of Conduct, is allowed during this period.
-Violating these terms may lead to a permanent ban.
-
-### 4. Permanent Ban
-
-**Community Impact**: Demonstrating a pattern of violation of community
-standards, including sustained inappropriate behavior, harassment of an
-individual, or aggression toward or disparagement of classes of individuals.
-
-**Consequence**: A permanent ban from any sort of public interaction within the
-community.
-
-## Attribution
-
-This Code of Conduct is adapted from the [Contributor Covenant][homepage],
-version 2.1, available at
-[https://www.contributor-covenant.org/version/2/1/code_of_conduct.html][v2.1].
-
-Community Impact Guidelines were inspired by
-[Mozilla's code of conduct enforcement ladder][Mozilla CoC].
-
-For answers to common questions about this code of conduct, see the FAQ at
-[https://www.contributor-covenant.org/faq][FAQ]. Translations are available at
-[https://www.contributor-covenant.org/translations][translations].
-
-[homepage]: https://www.contributor-covenant.org
-[v2.1]: https://www.contributor-covenant.org/version/2/1/code_of_conduct.html
-[Mozilla CoC]: https://github.com/mozilla/diversity
-[FAQ]: https://www.contributor-covenant.org/faq
-[translations]: https://www.contributor-covenant.org/translations
--- a/Cargo.lock
+++ b/Cargo.lock
--- a/Cargo.toml
+++ b/Cargo.toml
@@ -62,7 +62,7 @@ members = [
 resolver = "2"

 [workspace.package]
-version = "0.6.0"
+version = "0.7.2"
 edition = "2021"
 license = "Apache-2.0"

@@ -99,17 +99,20 @@ datafusion-physical-expr = { git = "https://github.com/apache/arrow-datafusion.g
 datafusion-sql = { git = "https://github.com/apache/arrow-datafusion.git", rev = "26e43acac3a96cec8dd4c8365f22dfb1a84306e9" }
 datafusion-substrait = { git = "https://github.com/apache/arrow-datafusion.git", rev = "26e43acac3a96cec8dd4c8365f22dfb1a84306e9" }
 derive_builder = "0.12"
+dotenv = "0.15"
 etcd-client = "0.12"
 fst = "0.4.7"
 futures = "0.3"
 futures-util = "0.3"
-greptime-proto = { git = "https://github.com/GreptimeTeam/greptime-proto.git", rev = "96f1f0404f421ee560a4310c73c5071e49168168" }
+greptime-proto = { git = "https://github.com/GreptimeTeam/greptime-proto.git", rev = "1bd2398b686e5ac6c1eef6daf615867ce27f75c1" }
+humantime = "2.1"
 humantime-serde = "1.1"
 itertools = "0.10"
 lazy_static = "1.4"
 meter-core = { git = "https://github.com/GreptimeTeam/greptime-meter.git", rev = "80b72716dcde47ec4161478416a5c6c21343364d" }
 mockall = "0.11.4"
 moka = "0.12"
+notify = "6.1"
 num_cpus = "1.16"
 once_cell = "1.18"
 opentelemetry-proto = { git = "https://github.com/waynexia/opentelemetry-rust.git", rev = "33841b38dda79b15f2024952be5f32533325ca02", features = [
@@ -125,7 +128,7 @@ prost = "0.12"
 raft-engine = { version = "0.4.1", default-features = false }
 rand = "0.8"
 regex = "1.8"
-regex-automata = { version = "0.2", features = ["transducer"] }
+regex-automata = { version = "0.4" }
 reqwest = { version = "0.11", default-features = false, features = [
    "json",
    "rustls-tls-native-roots",
@@ -133,8 +136,9 @@ reqwest = { version = "0.11", default-features = false, features = [
 ] }
 rskafka = "0.5"
 rust_decimal = "1.33"
+schemars = "0.8"
 serde = { version = "1.0", features = ["derive"] }
-serde_json = "1.0"
+serde_json = { version = "1.0", features = ["float_roundtrip"] }
 serde_with = "3"
 smallvec = { version = "1", features = ["serde"] }
 snafu = "0.7"
@@ -151,6 +155,7 @@ tokio-util = { version = "0.7", features = ["io-util", "compat"] }
 toml = "0.8.8"
 tonic = { version = "0.10", features = ["tls"] }
 uuid = { version = "1", features = ["serde", "v4", "fast-rng"] }
+zstd = "0.13"

 ## workspaces members
 api = { path = "src/api" }
--- a/5
+++ b/5
@@ -3,6 +3,7 @@ CARGO_PROFILE ?=
 FEATURES ?=
 TARGET_DIR ?=
 TARGET ?=
+BUILD_BIN ?= greptime
 CARGO_BUILD_OPTS := --locked
 IMAGE_REGISTRY ?= docker.io
 IMAGE_NAMESPACE ?= greptime
@@ -45,6 +46,10 @@ ifneq ($(strip $(TARGET)),)
 	CARGO_BUILD_OPTS += --target ${TARGET}
 endif

+ifneq ($(strip $(BUILD_BIN)),)
+	CARGO_BUILD_OPTS += --bin ${BUILD_BIN}
+endif
+
 ifneq ($(strip $(RELEASE)),)
 	CARGO_BUILD_OPTS += --release
 endif
--- a/README.md
+++ b/README.md
@@ -6,145 +6,154 @@
  </picture>
 </p>

+<h1 align="center">Cloud-scale, Fast and Efficient Time Series Database</h1>

+<div align="center">
 <h3 align="center">
-    The next-generation hybrid time-series/analytics processing database in the cloud
-</h3>
+  <a href="https://greptime.com/product/cloud">GreptimeCloud</a> |
+  <a href="https://docs.greptime.com/">User guide</a> |
+  <a href="https://greptimedb.rs/">API Docs</a> |
+  <a href="https://github.com/GreptimeTeam/greptimedb/issues/3412">Roadmap 2024</a>
+</h4>

-<p align="center">
-    <a href="https://codecov.io/gh/GrepTimeTeam/greptimedb"><img src="https://codecov.io/gh/GrepTimeTeam/greptimedb/branch/main/graph/badge.svg?token=FITFDI3J3C"></img></a>
-    &nbsp;
-    <a href="https://github.com/GreptimeTeam/greptimedb/actions/workflows/develop.yml"><img src="https://github.com/GreptimeTeam/greptimedb/actions/workflows/develop.yml/badge.svg" alt="CI"></img></a>
-    &nbsp;
-    <a href="https://github.com/greptimeTeam/greptimedb/blob/main/LICENSE"><img src="https://img.shields.io/github/license/greptimeTeam/greptimedb"></a>
-</p>
+<a href="https://github.com/GreptimeTeam/greptimedb/releases/latest">
+<img src="https://img.shields.io/github/v/release/GreptimeTeam/greptimedb.svg" alt="Version"/>
+</a>
+<a href="https://github.com/GreptimeTeam/greptimedb/releases/latest">
+<img src="https://img.shields.io/github/release-date/GreptimeTeam/greptimedb.svg" alt="Releases"/>
+</a>
+<a href="https://hub.docker.com/r/greptime/greptimedb/">
+<img src="https://img.shields.io/docker/pulls/greptime/greptimedb.svg" alt="Docker Pulls"/>
+</a>
+<a href="https://github.com/GreptimeTeam/greptimedb/actions/workflows/develop.yml">
+<img src="https://github.com/GreptimeTeam/greptimedb/actions/workflows/develop.yml/badge.svg" alt="GitHub Actions"/>
+</a>
+<a href="https://codecov.io/gh/GrepTimeTeam/greptimedb">
+<img src="https://codecov.io/gh/GrepTimeTeam/greptimedb/branch/main/graph/badge.svg?token=FITFDI3J3C" alt="Codecov"/>
+</a>
+<a href="https://github.com/greptimeTeam/greptimedb/blob/main/LICENSE">
+<img src="https://img.shields.io/github/license/greptimeTeam/greptimedb" alt="License"/>
+</a>

-<p align="center">
-    <a href="https://twitter.com/greptime"><img src="https://img.shields.io/badge/twitter-follow_us-1d9bf0.svg"></a>
-    &nbsp;
-    <a href="https://www.linkedin.com/company/greptime/"><img src="https://img.shields.io/badge/linkedin-connect_with_us-0a66c2.svg"></a>
-    &nbsp;
-    <a href="https://greptime.com/slack"><img src="https://img.shields.io/badge/slack-GreptimeDB-0abd59?logo=slack" alt="slack" /></a>
-</p>
+<br/>

-## What is GreptimeDB
+<a href="https://greptime.com/slack">
+<img src="https://img.shields.io/badge/slack-GreptimeDB-0abd59?logo=slack&style=for-the-badge" alt="Slack"/>
+</a>
+<a href="https://twitter.com/greptime">
+<img src="https://img.shields.io/badge/twitter-follow_us-1d9bf0.svg?style=for-the-badge" alt="Twitter"/>
+</a>
+<a href="https://www.linkedin.com/company/greptime/">
+<img src="https://img.shields.io/badge/linkedin-connect_with_us-0a66c2.svg?style=for-the-badge" alt="LinkedIn"/>
+</a>
+</div>

-GreptimeDB is an open-source time-series database focusing on efficiency, scalability, and analytical capabilities.
-It's designed to work on infrastructure of the cloud era, and users benefit from its elasticity and commodity storage.
+## Introduction

-Our core developers have been building time-series data platforms for years. Based on their best-practices, GreptimeDB is born to give you:
+**GreptimeDB** is an open-source time-series database focusing on efficiency, scalability, and analytical capabilities.
+Designed to work on infrastructure of the cloud era, GreptimeDB benefits users with its elasticity and commodity storage, offering a fast and cost-effective **alternative to InfluxDB** and a **long-term storage for Prometheus**.

- Optimized columnar layout for handling time-series data; compacted, compressed, and stored on various storage backends, particularly cloud object storage with 50x cost efficiency.
- Fully open-source distributed cluster architecture that harnesses the power of cloud-native elastic computing resources.
- Seamless scalability from a standalone binary at edge to a robust, highly available distributed cluster in cloud, with a transparent experience for both developers and administrators.
- Native SQL and PromQL for queries, and Python scripting to facilitate complex analytical tasks.
- Flexible indexing capabilities and distributed, parallel-processing query engine, tackling high cardinality issues down.
- Widely adopted database protocols and APIs, including MySQL, PostgreSQL, and Prometheus Remote Storage, etc.
+## Why GreptimeDB

-## Quick Start
+Our core developers have been building time-series data platforms for years. Based on our best-practices, GreptimeDB is born to give you:

-### [GreptimePlay](https://greptime.com/playground)
+* **Easy horizontal scaling**
+
+  Seamless scalability from a standalone binary at edge to a robust, highly available distributed cluster in cloud, with a transparent experience for both developers and administrators.
+
+* **Analyzing time-series data**
+
+  Query your time-series data with SQL and PromQL. Use Python scripts to facilitate complex analytical tasks.
+
+* **Cloud-native distributed database**
+
+  Fully open-source distributed cluster architecture that harnesses the power of cloud-native elastic computing resources.
+
+* **Performance and Cost-effective**
+
+  Flexible indexing capabilities and distributed, parallel-processing query engine, tackling high cardinality issues down. Optimized columnar layout for handling time-series data; compacted, compressed, and stored on various storage backends, particularly cloud object storage with 50x cost efficiency.
+
+* **Compatible with InfluxDB, Prometheus and more protocols**
+
+  Widely adopted database protocols and APIs, including MySQL, PostgreSQL, and Prometheus Remote Storage, etc. [Read more](https://docs.greptime.com/user-guide/clients/overview).
+
+## Try GreptimeDB
+
+### 1. [GreptimePlay](https://greptime.com/playground)

 Try out the features of GreptimeDB right from your browser.

-### Build
+### 2. [GreptimeCloud](https://console.greptime.cloud/)

-#### Build from Source
+Start instantly with a free cluster.

-To compile GreptimeDB from source, you'll need:
+### 3. Docker Image

- C/C++ Toolchain: provides basic tools for compiling and linking. This is
-  available as `build-essential` on ubuntu and similar name on other platforms.
- Rust: the easiest way to install Rust is to use
-  [`rustup`](https://rustup.rs/), which will check our `rust-toolchain` file and
-  install correct Rust version for you.
- Protobuf: `protoc` is required for compiling `.proto` files. `protobuf` is
-  available from major package manager on macos and linux distributions. You can
-  find an installation instructions [here](https://grpc.io/docs/protoc-installation/).
-  **Note that `protoc` version needs to be >= 3.15** because we have used the `optional`
-  keyword. You can check it with `protoc --version`.
- python3-dev or python3-devel(Optional feature, only needed if you want to run scripts
-  in CPython, and also need to enable `pyo3_backend` feature when compiling(by `cargo run -F pyo3_backend` or add `pyo3_backend` to src/script/Cargo.toml 's `features.default` like `default = ["python", "pyo3_backend]`)): this install a Python shared library required for running Python
-  scripting engine(In CPython Mode). This is available as `python3-dev` on
-  ubuntu, you can install it with `sudo apt install python3-dev`, or
-  `python3-devel` on RPM based distributions (e.g. Fedora, Red Hat, SuSE). Mac's
-  `Python3` package should have this shared library by default. More detail for compiling with PyO3 can be found in [PyO3](https://pyo3.rs/v0.18.1/building_and_distribution#configuring-the-python-version)'s documentation.
+To install GreptimeDB locally, the recommended way is via Docker:

-#### Build with Docker
-
-A docker image with necessary dependencies is provided:
-
-```
-docker build --network host -f docker/Dockerfile -t greptimedb .
+```shell
+docker pull greptime/greptimedb
 ```

-### Run
-
-Start GreptimeDB from source code, in standalone mode:
+Start a GreptimeDB container with:

+```shell
+docker run --rm --name greptime --net=host greptime/greptimedb standalone start
 ```
+
+Read more about [Installation](https://docs.greptime.com/getting-started/installation/overview) on docs.
+
+## Getting Started
+
+* [Quickstart](https://docs.greptime.com/getting-started/quick-start/overview)
+* [Write Data](https://docs.greptime.com/user-guide/clients/overview)
+* [Query Data](https://docs.greptime.com/user-guide/query-data/overview)
+* [Operations](https://docs.greptime.com/user-guide/operations/overview)
+
+## Build
+
+Check the prerequisite:
+
+* [Rust toolchain](https://www.rust-lang.org/tools/install) (nightly)
+* [Protobuf compiler](https://grpc.io/docs/protoc-installation/) (>= 3.15)
+* Python toolchain (optional): Required only if built with PyO3 backend. More detail for compiling with PyO3 can be found in its [documentation](https://pyo3.rs/v0.18.1/building_and_distribution#configuring-the-python-version).
+
+Build GreptimeDB binary:
+
+```shell
+make
+```
+
+Run a standalone server:
+
+```shell
 cargo run -- standalone start
 ```

-Or if you built from docker:
-
-```
-docker run -p 4002:4002 -v "$(pwd):/tmp/greptimedb" greptime/greptimedb standalone start
-```
-
-Please see the online document site for more installation options and [operations info](https://docs.greptime.com/user-guide/operations/overview).
-
-### Get started
-
-Read the [complete getting started guide](https://docs.greptime.com/getting-started/overview) on our [official document site](https://docs.greptime.com/).
-
-To write and query data, GreptimeDB is compatible with multiple [protocols and clients](https://docs.greptime.com/user-guide/clients/overview).
-
-## Resources
-
-### Installation
-
- [Pre-built Binaries](https://greptime.com/download):
-  For Linux and macOS, you can easily download pre-built binaries including official releases and nightly builds that are ready to use.
-  In most cases, downloading the version without PyO3 is sufficient. However, if you plan to run scripts in CPython (and use Python packages like NumPy and Pandas), you will need to download the version with PyO3 and install a Python with the same version as the Python in the PyO3 version.
-  We recommend using virtualenv for the installation process to manage multiple Python versions.
- [Docker Images](https://hub.docker.com/r/greptime/greptimedb)(**recommended**): pre-built
-  Docker images, this is the easiest way to try GreptimeDB. By default it runs CPython script with `pyo3_backend` enabled.
- [`gtctl`](https://github.com/GreptimeTeam/gtctl): the command-line tool for
-  Kubernetes deployment
-
-### Documentation
-
- GreptimeDB [User Guide](https://docs.greptime.com/user-guide/concepts/overview)
- GreptimeDB [Developer
-  Guide](https://docs.greptime.com/developer-guide/overview.html)
- GreptimeDB [internal code document](https://greptimedb.rs)
+## Extension

 ### Dashboard
+
 - [The dashboard UI for GreptimeDB](https://github.com/GreptimeTeam/dashboard)

 ### SDK

- [GreptimeDB C++ Client](https://github.com/GreptimeTeam/greptimedb-client-cpp)
- [GreptimeDB Erlang Client](https://github.com/GreptimeTeam/greptimedb-client-erl)
 - [GreptimeDB Go Ingester](https://github.com/GreptimeTeam/greptimedb-ingester-go)
 - [GreptimeDB Java Ingester](https://github.com/GreptimeTeam/greptimedb-ingester-java)
- [GreptimeDB Python Client](https://github.com/GreptimeTeam/greptimedb-client-py) (WIP)
- [GreptimeDB Rust Client](https://github.com/GreptimeTeam/greptimedb-client-rust)
- [GreptimeDB JavaScript Client](https://github.com/GreptimeTeam/greptime-js-sdk)
+- [GreptimeDB C++ Ingester](https://github.com/GreptimeTeam/greptimedb-ingester-cpp)
+- [GreptimeDB Erlang Ingester](https://github.com/GreptimeTeam/greptimedb-ingester-erl)
+- [GreptimeDB Rust Ingester](https://github.com/GreptimeTeam/greptimedb-ingester-rust)
+- [GreptimeDB JavaScript Ingester](https://github.com/GreptimeTeam/greptime-ingester-js)

 ### Grafana Dashboard

-Our official Grafana dashboard is available at [grafana](./grafana/README.md) directory.
+Our official Grafana dashboard is available at [grafana](grafana/README.md) directory.

 ## Project Status

-This project is in its early stage and under heavy development. We move fast and
-break things. Benchmark on development branch may not represent its potential
-performance. We release pre-built binaries constantly for functional
-evaluation. Do not use it in production at the moment.
-
-For future plans, check out [GreptimeDB roadmap](https://github.com/GreptimeTeam/greptimedb/issues/669).
+The current version has not yet reached General Availability version standards.
+In line with our Greptime 2024 Roadmap, we plan to achieve a production-level
+version with the update to v1.0 in August. [[Join Force]](https://github.com/GreptimeTeam/greptimedb/issues/3412)

 ## Community

@@ -154,12 +163,12 @@ and what went wrong. If you have any questions or if you would like to get invol
 community, please check out:

 - GreptimeDB Community on [Slack](https://greptime.com/slack)
- GreptimeDB GitHub [Discussions](https://github.com/GreptimeTeam/greptimedb/discussions)
- Greptime official [Website](https://greptime.com)
+- GreptimeDB [GitHub Discussions forum](https://github.com/GreptimeTeam/greptimedb/discussions)
+- Greptime official [website](https://greptime.com)

 In addition, you may:

- View our official [Blog](https://greptime.com/blogs/index)
+- View our official [Blog](https://greptime.com/blogs/)
 - Connect us with [Linkedin](https://www.linkedin.com/company/greptime/)
 - Follow us on [Twitter](https://twitter.com/greptime)

@@ -170,7 +179,7 @@ open contributions and allowing you to use the software however you want.

 ## Contributing

-Please refer to [contribution guidelines](CONTRIBUTING.md) for more information.
+Please refer to [contribution guidelines](CONTRIBUTING.md) and [internal concepts docs](https://docs.greptime.com/contributor-guide/overview.html) for more information.

 ## Acknowledgement

--- a/benchmarks/Cargo.toml
+++ b/benchmarks/Cargo.toml
@@ -8,12 +8,31 @@ license.workspace = true
 workspace = true

 [dependencies]
+api.workspace = true
 arrow.workspace = true
 chrono.workspace = true
 clap.workspace = true
 client.workspace = true
+common-base.workspace = true
+common-telemetry.workspace = true
+common-wal.workspace = true
+dotenv.workspace = true
+futures.workspace = true
 futures-util.workspace = true
+humantime.workspace = true
+humantime-serde.workspace = true
 indicatif = "0.17.1"
 itertools.workspace = true
+lazy_static.workspace = true
+log-store.workspace = true
+mito2.workspace = true
+num_cpus.workspace = true
 parquet.workspace = true
+prometheus.workspace = true
+rand.workspace = true
+rskafka.workspace = true
+serde.workspace = true
+store-api.workspace = true
 tokio.workspace = true
+toml.workspace = true
+uuid.workspace = true
--- a/benchmarks/README.md
+++ b/benchmarks/README.md
@@ -0,0 +1,11 @@
+Benchmarkers for GreptimeDB
+--------------------------------
+
+## Wal Benchmarker
+The wal benchmarker serves to evaluate the performance of GreptimeDB's Write-Ahead Log (WAL) component. It meticulously assesses the read/write performance of the WAL under diverse workloads generated by the benchmarker. 
+
+
+### How to use
+To compile the benchmarker, navigate to the `greptimedb/benchmarks` directory and execute `cargo build --release`. Subsequently, you'll find the compiled target located at `greptimedb/target/release/wal_bench`.
+
+The `./wal_bench -h` command reveals numerous arguments that the target accepts. Among these, a notable one is the `cfg-file` argument. By utilizing a configuration file in the TOML format, you can bypass the need to repeatedly specify cumbersome arguments.
--- a/benchmarks/config/wal_bench.example.toml
+++ b/benchmarks/config/wal_bench.example.toml
@@ -0,0 +1,21 @@
+# Refers to the documents of `Args` in benchmarks/src/wal.rs`.
+wal_provider = "kafka"
+bootstrap_brokers = ["localhost:9092"]
+num_workers = 10
+num_topics = 32
+num_regions = 1000
+num_scrapes = 1000
+num_rows = 5
+col_types = "ifs"
+max_batch_size = "512KB"
+linger = "1ms"
+backoff_init = "10ms"
+backoff_max = "1ms"
+backoff_base = 2
+backoff_deadline = "3s"
+compression = "zstd"
+rng_seed = 42
+skip_read = false
+skip_write = false
+random_topics = true
+report_metrics = false
--- a/benchmarks/src/bin/nyc-taxi.rs
+++ b/benchmarks/src/bin/nyc-taxi.rs
@@ -29,7 +29,7 @@ use client::api::v1::column::Values;
 use client::api::v1::{
    Column, ColumnDataType, ColumnDef, CreateTableExpr, InsertRequest, InsertRequests, SemanticType,
 };
-use client::{Client, Database, Output, DEFAULT_CATALOG_NAME, DEFAULT_SCHEMA_NAME};
+use client::{Client, Database, OutputData, DEFAULT_CATALOG_NAME, DEFAULT_SCHEMA_NAME};
 use futures_util::TryStreamExt;
 use indicatif::{MultiProgress, ProgressBar, ProgressStyle};
 use parquet::arrow::arrow_reader::ParquetRecordBatchReaderBuilder;
@@ -502,9 +502,9 @@ async fn do_query(num_iter: usize, db: &Database, table_name: &str) {
        for i in 0..num_iter {
            let now = Instant::now();
            let res = db.sql(&query).await.unwrap();
-            match res {
-                Output::AffectedRows(_) | Output::RecordBatches(_) => (),
-                Output::Stream(stream, _) => {
+            match res.data {
+                OutputData::AffectedRows(_) | OutputData::RecordBatches(_) => (),
+                OutputData::Stream(stream) => {
                    stream.try_collect::<Vec<_>>().await.unwrap();
                }
            }
--- a/benchmarks/src/bin/wal_bench.rs
+++ b/benchmarks/src/bin/wal_bench.rs
@@ -0,0 +1,326 @@
+// Copyright 2023 Greptime Team
+//
+// Licensed under the Apache License, Version 2.0 (the "License");
+// you may not use this file except in compliance with the License.
+// You may obtain a copy of the License at
+//
+//     http://www.apache.org/licenses/LICENSE-2.0
+//
+// Unless required by applicable law or agreed to in writing, software
+// distributed under the License is distributed on an "AS IS" BASIS,
+// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+// See the License for the specific language governing permissions and
+// limitations under the License.
+
+#![feature(int_roundings)]
+
+use std::fs;
+use std::sync::Arc;
+use std::time::Instant;
+
+use api::v1::{ColumnDataType, ColumnSchema, SemanticType};
+use benchmarks::metrics;
+use benchmarks::wal_bench::{Args, Config, Region, WalProvider};
+use clap::Parser;
+use common_telemetry::info;
+use common_wal::config::kafka::common::BackoffConfig;
+use common_wal::config::kafka::DatanodeKafkaConfig as KafkaConfig;
+use common_wal::config::raft_engine::RaftEngineConfig;
+use common_wal::options::{KafkaWalOptions, WalOptions};
+use itertools::Itertools;
+use log_store::kafka::log_store::KafkaLogStore;
+use log_store::raft_engine::log_store::RaftEngineLogStore;
+use mito2::wal::Wal;
+use prometheus::{Encoder, TextEncoder};
+use rand::distributions::{Alphanumeric, DistString};
+use rand::rngs::SmallRng;
+use rand::SeedableRng;
+use rskafka::client::partition::Compression;
+use rskafka::client::ClientBuilder;
+use store_api::logstore::LogStore;
+use store_api::storage::RegionId;
+
+async fn run_benchmarker<S: LogStore>(cfg: &Config, topics: &[String], wal: Arc<Wal<S>>) {
+    let chunk_size = cfg.num_regions.div_ceil(cfg.num_workers);
+    let region_chunks = (0..cfg.num_regions)
+        .map(|id| {
+            build_region(
+                id as u64,
+                topics,
+                &mut SmallRng::seed_from_u64(cfg.rng_seed),
+                cfg,
+            )
+        })
+        .chunks(chunk_size as usize)
+        .into_iter()
+        .map(|chunk| Arc::new(chunk.collect::<Vec<_>>()))
+        .collect::<Vec<_>>();
+
+    let mut write_elapsed = 0;
+    let mut read_elapsed = 0;
+
+    if !cfg.skip_write {
+        info!("Benchmarking write ...");
+
+        let num_scrapes = cfg.num_scrapes;
+        let timer = Instant::now();
+        futures::future::join_all((0..cfg.num_workers).map(|i| {
+            let wal = wal.clone();
+            let regions = region_chunks[i as usize].clone();
+            tokio::spawn(async move {
+                for _ in 0..num_scrapes {
+                    let mut wal_writer = wal.writer();
+                    regions
+                        .iter()
+                        .for_each(|region| region.add_wal_entry(&mut wal_writer));
+                    wal_writer.write_to_wal().await.unwrap();
+                }
+            })
+        }))
+        .await;
+        write_elapsed += timer.elapsed().as_millis();
+    }
+
+    if !cfg.skip_read {
+        info!("Benchmarking read ...");
+
+        let timer = Instant::now();
+        futures::future::join_all((0..cfg.num_workers).map(|i| {
+            let wal = wal.clone();
+            let regions = region_chunks[i as usize].clone();
+            tokio::spawn(async move {
+                for region in regions.iter() {
+                    region.replay(&wal).await;
+                }
+            })
+        }))
+        .await;
+        read_elapsed = timer.elapsed().as_millis();
+    }
+
+    dump_report(cfg, write_elapsed, read_elapsed);
+}
+
+fn build_region(id: u64, topics: &[String], rng: &mut SmallRng, cfg: &Config) -> Region {
+    let wal_options = match cfg.wal_provider {
+        WalProvider::Kafka => {
+            assert!(!topics.is_empty());
+            WalOptions::Kafka(KafkaWalOptions {
+                topic: topics.get(id as usize % topics.len()).cloned().unwrap(),
+            })
+        }
+        WalProvider::RaftEngine => WalOptions::RaftEngine,
+    };
+    Region::new(
+        RegionId::from_u64(id),
+        build_schema(&parse_col_types(&cfg.col_types), rng),
+        wal_options,
+        cfg.num_rows,
+        cfg.rng_seed,
+    )
+}
+
+fn build_schema(col_types: &[ColumnDataType], mut rng: &mut SmallRng) -> Vec<ColumnSchema> {
+    col_types
+        .iter()
+        .map(|col_type| ColumnSchema {
+            column_name: Alphanumeric.sample_string(&mut rng, 5),
+            datatype: *col_type as i32,
+            semantic_type: SemanticType::Field as i32,
+            datatype_extension: None,
+        })
+        .chain(vec![ColumnSchema {
+            column_name: "ts".to_string(),
+            datatype: ColumnDataType::TimestampMillisecond as i32,
+            semantic_type: SemanticType::Tag as i32,
+            datatype_extension: None,
+        }])
+        .collect()
+}
+
+fn dump_report(cfg: &Config, write_elapsed: u128, read_elapsed: u128) {
+    let cost_report = format!(
+        "write costs: {} ms, read costs: {} ms",
+        write_elapsed, read_elapsed,
+    );
+
+    let total_written_bytes = metrics::METRIC_WAL_WRITE_BYTES_TOTAL.get() as u128;
+    let write_throughput = if write_elapsed > 0 {
+        (total_written_bytes * 1000).div_floor(write_elapsed)
+    } else {
+        0
+    };
+    let total_read_bytes = metrics::METRIC_WAL_READ_BYTES_TOTAL.get() as u128;
+    let read_throughput = if read_elapsed > 0 {
+        (total_read_bytes * 1000).div_floor(read_elapsed)
+    } else {
+        0
+    };
+
+    let throughput_report = format!(
+        "total written bytes: {} bytes, total read bytes: {} bytes, write throuput: {} bytes/s ({} mb/s), read throughput: {} bytes/s ({} mb/s)",
+        total_written_bytes,
+        total_read_bytes,
+        write_throughput,
+        write_throughput.div_floor(1 << 20),
+        read_throughput,
+        read_throughput.div_floor(1 << 20),
+    );
+
+    let metrics_report = if cfg.report_metrics {
+        let mut buffer = Vec::new();
+        let encoder = TextEncoder::new();
+        let metrics = prometheus::gather();
+        encoder.encode(&metrics, &mut buffer).unwrap();
+        String::from_utf8(buffer).unwrap()
+    } else {
+        String::new()
+    };
+
+    info!(
+        r#"
+Benchmark config: 
+{cfg:?}
+
+Benchmark report:
+{cost_report}
+{throughput_report}
+{metrics_report}"#
+    );
+}
+
+async fn create_topics(cfg: &Config) -> Vec<String> {
+    // Creates topics.
+    let client = ClientBuilder::new(cfg.bootstrap_brokers.clone())
+        .build()
+        .await
+        .unwrap();
+    let ctrl_client = client.controller_client().unwrap();
+    let (topics, tasks): (Vec<_>, Vec<_>) = (0..cfg.num_topics)
+        .map(|i| {
+            let topic = if cfg.random_topics {
+                format!(
+                    "greptime_wal_bench_topic_{}_{}",
+                    uuid::Uuid::new_v4().as_u128(),
+                    i
+                )
+            } else {
+                format!("greptime_wal_bench_topic_{}", i)
+            };
+            let task = ctrl_client.create_topic(
+                topic.clone(),
+                1,
+                cfg.bootstrap_brokers.len() as i16,
+                2000,
+            );
+            (topic, task)
+        })
+        .unzip();
+    // Must ignore errors since we allow topics being created more than once.
+    let _ = futures::future::try_join_all(tasks).await;
+
+    topics
+}
+
+fn parse_compression(comp: &str) -> Compression {
+    match comp {
+        "no" => Compression::NoCompression,
+        "gzip" => Compression::Gzip,
+        "lz4" => Compression::Lz4,
+        "snappy" => Compression::Snappy,
+        "zstd" => Compression::Zstd,
+        other => unreachable!("Unrecognized compression {other}"),
+    }
+}
+
+fn parse_col_types(col_types: &str) -> Vec<ColumnDataType> {
+    let parts = col_types.split('x').collect::<Vec<_>>();
+    assert!(parts.len() <= 2);
+
+    let pattern = parts[0];
+    let repeat = parts
+        .get(1)
+        .map(|r| r.parse::<usize>().unwrap())
+        .unwrap_or(1);
+
+    pattern
+        .chars()
+        .map(|c| match c {
+            'i' | 'I' => ColumnDataType::Int64,
+            'f' | 'F' => ColumnDataType::Float64,
+            's' | 'S' => ColumnDataType::String,
+            other => unreachable!("Cannot parse {other} as a column data type"),
+        })
+        .cycle()
+        .take(pattern.len() * repeat)
+        .collect()
+}
+
+fn main() {
+    // Sets the global logging to INFO and suppress loggings from rskafka other than ERROR and upper ones.
+    std::env::set_var("UNITTEST_LOG_LEVEL", "info,rskafka=error");
+    common_telemetry::init_default_ut_logging();
+
+    let args = Args::parse();
+    let cfg = if !args.cfg_file.is_empty() {
+        toml::from_str(&fs::read_to_string(&args.cfg_file).unwrap()).unwrap()
+    } else {
+        Config::from(args)
+    };
+
+    // Validates arguments.
+    if cfg.num_regions < cfg.num_workers {
+        panic!("num_regions must be greater than or equal to num_workers");
+    }
+    if cfg
+        .num_workers
+        .min(cfg.num_topics)
+        .min(cfg.num_regions)
+        .min(cfg.num_scrapes)
+        .min(cfg.max_batch_size.as_bytes() as u32)
+        .min(cfg.bootstrap_brokers.len() as u32)
+        == 0
+    {
+        panic!("Invalid arguments");
+    }
+
+    tokio::runtime::Builder::new_multi_thread()
+        .enable_all()
+        .build()
+        .unwrap()
+        .block_on(async {
+            match cfg.wal_provider {
+                WalProvider::Kafka => {
+                    let topics = create_topics(&cfg).await;
+                    let kafka_cfg = KafkaConfig {
+                        broker_endpoints: cfg.bootstrap_brokers.clone(),
+                        max_batch_size: cfg.max_batch_size,
+                        linger: cfg.linger,
+                        backoff: BackoffConfig {
+                            init: cfg.backoff_init,
+                            max: cfg.backoff_max,
+                            base: cfg.backoff_base,
+                            deadline: Some(cfg.backoff_deadline),
+                        },
+                        compression: parse_compression(&cfg.compression),
+                        ..Default::default()
+                    };
+                    let store = Arc::new(KafkaLogStore::try_new(&kafka_cfg).await.unwrap());
+                    let wal = Arc::new(Wal::new(store));
+                    run_benchmarker(&cfg, &topics, wal).await;
+                }
+                WalProvider::RaftEngine => {
+                    // The benchmarker assumes the raft engine directory exists.
+                    let store = RaftEngineLogStore::try_new(
+                        "/tmp/greptimedb/raft-engine-wal".to_string(),
+                        RaftEngineConfig::default(),
+                    )
+                    .await
+                    .map(Arc::new)
+                    .unwrap();
+                    let wal = Arc::new(Wal::new(store));
+                    run_benchmarker(&cfg, &[], wal).await;
+                }
+            }
+        });
+}
--- a/benchmarks/src/lib.rs
+++ b/benchmarks/src/lib.rs
@@ -0,0 +1,16 @@
+// Copyright 2023 Greptime Team
+//
+// Licensed under the Apache License, Version 2.0 (the "License");
+// you may not use this file except in compliance with the License.
+// You may obtain a copy of the License at
+//
+//     http://www.apache.org/licenses/LICENSE-2.0
+//
+// Unless required by applicable law or agreed to in writing, software
+// distributed under the License is distributed on an "AS IS" BASIS,
+// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+// See the License for the specific language governing permissions and
+// limitations under the License.
+
+pub mod metrics;
+pub mod wal_bench;
--- a/benchmarks/src/metrics.rs
+++ b/benchmarks/src/metrics.rs
@@ -0,0 +1,39 @@
+// Copyright 2023 Greptime Team
+//
+// Licensed under the Apache License, Version 2.0 (the "License");
+// you may not use this file except in compliance with the License.
+// You may obtain a copy of the License at
+//
+//     http://www.apache.org/licenses/LICENSE-2.0
+//
+// Unless required by applicable law or agreed to in writing, software
+// distributed under the License is distributed on an "AS IS" BASIS,
+// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+// See the License for the specific language governing permissions and
+// limitations under the License.
+
+use lazy_static::lazy_static;
+use prometheus::*;
+
+/// Logstore label.
+pub const LOGSTORE_LABEL: &str = "logstore";
+/// Operation type label.
+pub const OPTYPE_LABEL: &str = "optype";
+
+lazy_static! {
+    /// Counters of bytes of each operation on a logstore.
+    pub static ref METRIC_WAL_OP_BYTES_TOTAL: IntCounterVec = register_int_counter_vec!(
+        "greptime_bench_wal_op_bytes_total",
+        "wal operation bytes total",
+        &[OPTYPE_LABEL],
+    )
+    .unwrap();
+    /// Counter of bytes of the append_batch operation.
+    pub static ref METRIC_WAL_WRITE_BYTES_TOTAL: IntCounter = METRIC_WAL_OP_BYTES_TOTAL.with_label_values(
+        &["write"],
+    );
+    /// Counter of bytes of the read operation.
+    pub static ref METRIC_WAL_READ_BYTES_TOTAL: IntCounter = METRIC_WAL_OP_BYTES_TOTAL.with_label_values(
+        &["read"],
+    );
+}
--- a/benchmarks/src/wal_bench.rs
+++ b/benchmarks/src/wal_bench.rs
@@ -0,0 +1,361 @@
+// Copyright 2023 Greptime Team
+//
+// Licensed under the Apache License, Version 2.0 (the "License");
+// you may not use this file except in compliance with the License.
+// You may obtain a copy of the License at
+//
+//     http://www.apache.org/licenses/LICENSE-2.0
+//
+// Unless required by applicable law or agreed to in writing, software
+// distributed under the License is distributed on an "AS IS" BASIS,
+// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+// See the License for the specific language governing permissions and
+// limitations under the License.
+
+use std::mem::size_of;
+use std::sync::atomic::{AtomicI64, AtomicU64, Ordering};
+use std::sync::{Arc, Mutex};
+use std::time::Duration;
+
+use api::v1::value::ValueData;
+use api::v1::{ColumnDataType, ColumnSchema, Mutation, OpType, Row, Rows, Value, WalEntry};
+use clap::{Parser, ValueEnum};
+use common_base::readable_size::ReadableSize;
+use common_wal::options::WalOptions;
+use futures::StreamExt;
+use mito2::wal::{Wal, WalWriter};
+use rand::distributions::{Alphanumeric, DistString, Uniform};
+use rand::rngs::SmallRng;
+use rand::{Rng, SeedableRng};
+use serde::{Deserialize, Serialize};
+use store_api::logstore::LogStore;
+use store_api::storage::RegionId;
+
+use crate::metrics;
+
+/// The wal provider.
+#[derive(Clone, ValueEnum, Default, Debug, PartialEq, Serialize, Deserialize)]
+#[serde(rename_all = "snake_case")]
+pub enum WalProvider {
+    #[default]
+    RaftEngine,
+    Kafka,
+}
+
+#[derive(Parser)]
+pub struct Args {
+    /// The provided configuration file.
+    /// The example configuration file can be found at `greptimedb/benchmarks/config/wal_bench.example.toml`.
+    #[clap(long, short = 'c')]
+    pub cfg_file: String,
+
+    /// The wal provider.
+    #[clap(long, value_enum, default_value_t = WalProvider::default())]
+    pub wal_provider: WalProvider,
+
+    /// The advertised addresses of the kafka brokers.
+    /// If there're multiple bootstrap brokers, their addresses should be separated by comma, for e.g. "localhost:9092,localhost:9093".
+    #[clap(long, short = 'b', default_value = "localhost:9092")]
+    pub bootstrap_brokers: String,
+
+    /// The number of workers each running in a dedicated thread.
+    #[clap(long, default_value_t = num_cpus::get() as u32)]
+    pub num_workers: u32,
+
+    /// The number of kafka topics to be created.
+    #[clap(long, default_value_t = 32)]
+    pub num_topics: u32,
+
+    /// The number of regions.
+    #[clap(long, default_value_t = 1000)]
+    pub num_regions: u32,
+
+    /// The number of times each region is scraped.
+    #[clap(long, default_value_t = 1000)]
+    pub num_scrapes: u32,
+
+    /// The number of rows in each wal entry.
+    /// Each time a region is scraped, a wal entry containing will be produced.
+    #[clap(long, default_value_t = 5)]
+    pub num_rows: u32,
+
+    /// The column types of the schema for each region.
+    /// Currently, three column types are supported:
+    /// - i = ColumnDataType::Int64
+    /// - f = ColumnDataType::Float64
+    /// - s = ColumnDataType::String  
+    /// For e.g., "ifs" will be parsed as three columns: i64, f64, and string.
+    ///
+    /// Additionally, a "x" sign can be provided to repeat the column types for a given number of times.
+    /// For e.g., "iix2" will be parsed as 4 columns: i64, i64, i64, and i64.
+    /// This feature is useful if you want to specify many columns.
+    #[clap(long, default_value = "ifs")]
+    pub col_types: String,
+
+    /// The maximum size of a batch of kafka records.
+    /// The default value is 1mb.
+    #[clap(long, default_value = "512KB")]
+    pub max_batch_size: ReadableSize,
+
+    /// The minimum latency the kafka client issues a batch of kafka records.
+    /// However, a batch of kafka records would be immediately issued if a record cannot be fit into the batch.
+    #[clap(long, default_value = "1ms")]
+    pub linger: String,
+
+    /// The initial backoff delay of the kafka consumer.
+    #[clap(long, default_value = "10ms")]
+    pub backoff_init: String,
+
+    /// The maximum backoff delay of the kafka consumer.
+    #[clap(long, default_value = "1s")]
+    pub backoff_max: String,
+
+    /// The exponential backoff rate of the kafka consumer. The next back off = base * the current backoff.
+    #[clap(long, default_value_t = 2)]
+    pub backoff_base: u32,
+
+    /// The deadline of backoff. The backoff ends if the total backoff delay reaches the deadline.
+    #[clap(long, default_value = "3s")]
+    pub backoff_deadline: String,
+
+    /// The client-side compression algorithm for kafka records.
+    #[clap(long, default_value = "zstd")]
+    pub compression: String,
+
+    /// The seed of random number generators.
+    #[clap(long, default_value_t = 42)]
+    pub rng_seed: u64,
+
+    /// Skips the read phase, aka. region replay, if set to true.
+    #[clap(long, default_value_t = false)]
+    pub skip_read: bool,
+
+    /// Skips the write phase if set to true.
+    #[clap(long, default_value_t = false)]
+    pub skip_write: bool,
+
+    /// Randomly generates topic names if set to true.
+    /// Useful when you want to run the benchmarker without worrying about the topics created before.
+    #[clap(long, default_value_t = false)]
+    pub random_topics: bool,
+
+    /// Logs out the gathered prometheus metrics when the benchmarker ends.
+    #[clap(long, default_value_t = false)]
+    pub report_metrics: bool,
+}
+
+/// Benchmarker config.
+#[derive(Debug, Clone, Serialize, Deserialize)]
+pub struct Config {
+    pub wal_provider: WalProvider,
+    pub bootstrap_brokers: Vec<String>,
+    pub num_workers: u32,
+    pub num_topics: u32,
+    pub num_regions: u32,
+    pub num_scrapes: u32,
+    pub num_rows: u32,
+    pub col_types: String,
+    pub max_batch_size: ReadableSize,
+    #[serde(with = "humantime_serde")]
+    pub linger: Duration,
+    #[serde(with = "humantime_serde")]
+    pub backoff_init: Duration,
+    #[serde(with = "humantime_serde")]
+    pub backoff_max: Duration,
+    pub backoff_base: u32,
+    #[serde(with = "humantime_serde")]
+    pub backoff_deadline: Duration,
+    pub compression: String,
+    pub rng_seed: u64,
+    pub skip_read: bool,
+    pub skip_write: bool,
+    pub random_topics: bool,
+    pub report_metrics: bool,
+}
+
+impl From<Args> for Config {
+    fn from(args: Args) -> Self {
+        let cfg = Self {
+            wal_provider: args.wal_provider,
+            bootstrap_brokers: args
+                .bootstrap_brokers
+                .split(',')
+                .map(ToString::to_string)
+                .collect::<Vec<_>>(),
+            num_workers: args.num_workers.min(num_cpus::get() as u32),
+            num_topics: args.num_topics,
+            num_regions: args.num_regions,
+            num_scrapes: args.num_scrapes,
+            num_rows: args.num_rows,
+            col_types: args.col_types,
+            max_batch_size: args.max_batch_size,
+            linger: humantime::parse_duration(&args.linger).unwrap(),
+            backoff_init: humantime::parse_duration(&args.backoff_init).unwrap(),
+            backoff_max: humantime::parse_duration(&args.backoff_max).unwrap(),
+            backoff_base: args.backoff_base,
+            backoff_deadline: humantime::parse_duration(&args.backoff_deadline).unwrap(),
+            compression: args.compression,
+            rng_seed: args.rng_seed,
+            skip_read: args.skip_read,
+            skip_write: args.skip_write,
+            random_topics: args.random_topics,
+            report_metrics: args.report_metrics,
+        };
+
+        cfg
+    }
+}
+
+/// The region used for wal benchmarker.
+pub struct Region {
+    id: RegionId,
+    schema: Vec<ColumnSchema>,
+    wal_options: WalOptions,
+    next_sequence: AtomicU64,
+    next_entry_id: AtomicU64,
+    next_timestamp: AtomicI64,
+    rng: Mutex<Option<SmallRng>>,
+    num_rows: u32,
+}
+
+impl Region {
+    /// Creates a new region.
+    pub fn new(
+        id: RegionId,
+        schema: Vec<ColumnSchema>,
+        wal_options: WalOptions,
+        num_rows: u32,
+        rng_seed: u64,
+    ) -> Self {
+        Self {
+            id,
+            schema,
+            wal_options,
+            next_sequence: AtomicU64::new(1),
+            next_entry_id: AtomicU64::new(1),
+            next_timestamp: AtomicI64::new(1655276557000),
+            rng: Mutex::new(Some(SmallRng::seed_from_u64(rng_seed))),
+            num_rows,
+        }
+    }
+
+    /// Scrapes the region and adds the generated entry to wal.
+    pub fn add_wal_entry<S: LogStore>(&self, wal_writer: &mut WalWriter<S>) {
+        let mutation = Mutation {
+            op_type: OpType::Put as i32,
+            sequence: self
+                .next_sequence
+                .fetch_add(self.num_rows as u64, Ordering::Relaxed),
+            rows: Some(self.build_rows()),
+        };
+        let entry = WalEntry {
+            mutations: vec![mutation],
+        };
+        metrics::METRIC_WAL_WRITE_BYTES_TOTAL.inc_by(Self::entry_estimated_size(&entry) as u64);
+
+        wal_writer
+            .add_entry(
+                self.id,
+                self.next_entry_id.fetch_add(1, Ordering::Relaxed),
+                &entry,
+                &self.wal_options,
+            )
+            .unwrap();
+    }
+
+    /// Replays the region.
+    pub async fn replay<S: LogStore>(&self, wal: &Arc<Wal<S>>) {
+        let mut wal_stream = wal.scan(self.id, 0, &self.wal_options).unwrap();
+        while let Some(res) = wal_stream.next().await {
+            let (_, entry) = res.unwrap();
+            metrics::METRIC_WAL_READ_BYTES_TOTAL.inc_by(Self::entry_estimated_size(&entry) as u64);
+        }
+    }
+
+    /// Computes the estimated size in bytes of the entry.
+    pub fn entry_estimated_size(entry: &WalEntry) -> usize {
+        let wrapper_size = size_of::<WalEntry>()
+            + entry.mutations.capacity() * size_of::<Mutation>()
+            + size_of::<Rows>();
+
+        let rows = entry.mutations[0].rows.as_ref().unwrap();
+
+        let schema_size = rows.schema.capacity() * size_of::<ColumnSchema>()
+            + rows
+                .schema
+                .iter()
+                .map(|s| s.column_name.capacity())
+                .sum::<usize>();
+        let values_size = (rows.rows.capacity() * size_of::<Row>())
+            + rows
+                .rows
+                .iter()
+                .map(|r| r.values.capacity() * size_of::<Value>())
+                .sum::<usize>();
+
+        wrapper_size + schema_size + values_size
+    }
+
+    fn build_rows(&self) -> Rows {
+        let cols = self
+            .schema
+            .iter()
+            .map(|col_schema| {
+                let col_data_type = ColumnDataType::try_from(col_schema.datatype).unwrap();
+                self.build_col(&col_data_type, self.num_rows)
+            })
+            .collect::<Vec<_>>();
+
+        let rows = (0..self.num_rows)
+            .map(|i| {
+                let values = cols.iter().map(|col| col[i as usize].clone()).collect();
+                Row { values }
+            })
+            .collect();
+
+        Rows {
+            schema: self.schema.clone(),
+            rows,
+        }
+    }
+
+    fn build_col(&self, col_data_type: &ColumnDataType, num_rows: u32) -> Vec<Value> {
+        let mut rng_guard = self.rng.lock().unwrap();
+        let rng = rng_guard.as_mut().unwrap();
+        match col_data_type {
+            ColumnDataType::TimestampMillisecond => (0..num_rows)
+                .map(|_| {
+                    let ts = self.next_timestamp.fetch_add(1000, Ordering::Relaxed);
+                    Value {
+                        value_data: Some(ValueData::TimestampMillisecondValue(ts)),
+                    }
+                })
+                .collect(),
+            ColumnDataType::Int64 => (0..num_rows)
+                .map(|_| {
+                    let v = rng.sample(Uniform::new(0, 10_000));
+                    Value {
+                        value_data: Some(ValueData::I64Value(v)),
+                    }
+                })
+                .collect(),
+            ColumnDataType::Float64 => (0..num_rows)
+                .map(|_| {
+                    let v = rng.sample(Uniform::new(0.0, 5000.0));
+                    Value {
+                        value_data: Some(ValueData::F64Value(v)),
+                    }
+                })
+                .collect(),
+            ColumnDataType::String => (0..num_rows)
+                .map(|_| {
+                    let v = Alphanumeric.sample_string(rng, 10);
+                    Value {
+                        value_data: Some(ValueData::StringValue(v)),
+                    }
+                })
+                .collect(),
+            _ => unreachable!(),
+        }
+    }
+}
--- a/cliff.toml
+++ b/cliff.toml
@@ -0,0 +1,117 @@
+# https://git-cliff.org/docs/configuration
+
+[remote.github]
+owner = "GreptimeTeam"
+repo = "greptimedb"
+
+[changelog]
+header = ""
+footer = ""
+# template for the changelog body
+# https://keats.github.io/tera/docs/#introduction
+body = """
+# {{ version }}
+
+Release date: {{ timestamp | date(format="%B %d, %Y") }}
+
+{%- set breakings = commits | filter(attribute="breaking", value=true) -%}
+{%- if breakings | length > 0 %}
+
+## Breaking changes
+    {% for commit in breakings %}
+      * {{ commit.github.pr_title }}\
+        {% if commit.github.username %} by \
+          {% set author = commit.github.username -%}
+          [@{{ author }}](https://github.com/{{ author }})
+        {%- endif -%}
+        {% if commit.github.pr_number %} in \
+          {% set number = commit.github.pr_number -%}
+          [#{{ number }}]({{ self::remote_url() }}/pull/{{ number }})
+        {%- endif %}
+    {%- endfor %}
+{%- endif -%}
+
+{%- set grouped_commits = commits | filter(attribute="breaking", value=false) | group_by(attribute="group") -%}
+{% for group, commits in grouped_commits %}
+
+    ### {{ group | striptags | trim | upper_first }}
+    {% for commit in commits %}
+        * {{ commit.github.pr_title }}\
+            {% if commit.github.username %} by \
+              {% set author = commit.github.username -%}
+              [@{{ author }}](https://github.com/{{ author }})
+            {%- endif -%}
+            {% if commit.github.pr_number %} in \
+              {% set number = commit.github.pr_number -%}
+              [#{{ number }}]({{ self::remote_url() }}/pull/{{ number }})
+            {%- endif %}
+    {%- endfor -%}
+{% endfor %}
+
+{%- if github.contributors | filter(attribute="is_first_time", value=true) | length != 0 %}
+  {% raw %}\n{% endraw -%}
+  ## New Contributors
+{% endif -%}
+{% for contributor in github.contributors | filter(attribute="is_first_time", value=true) %}
+  * @{{ contributor.username }} made their first contribution
+    {%- if contributor.pr_number %} in \
+      [#{{ contributor.pr_number }}]({{ self::remote_url() }}/pull/{{ contributor.pr_number }}) \
+    {%- endif %}
+{%- endfor -%}
+
+{% if github.contributors | length != 0 %}
+  {% raw %}\n{% endraw -%}
+## All Contributors
+
+We would like to thank the following contributors from the GreptimeDB community:
+
+{{ github.contributors | map(attribute="username") | join(sep=", ") }}
+{%- endif %}
+{% raw %}\n{% endraw %}
+
+{%- macro remote_url() -%}
+  https://github.com/{{ remote.github.owner }}/{{ remote.github.repo }}
+{%- endmacro -%}
+"""
+trim = true
+
+[git]
+# parse the commits based on https://www.conventionalcommits.org
+conventional_commits = true
+# filter out the commits that are not conventional
+filter_unconventional = true
+# process each line of a commit as an individual commit
+split_commits = false
+# regex for parsing and grouping commits
+commit_parsers = [
+  { message = "^feat", group = "<!-- 0 -->🚀 Features" },
+  { message = "^fix", group = "<!-- 1 -->🐛 Bug Fixes" },
+  { message = "^doc", group = "<!-- 3 -->📚 Documentation" },
+  { message = "^perf", group = "<!-- 4 -->⚡ Performance" },
+  { message = "^refactor", group = "<!-- 2 -->🚜 Refactor" },
+  { message = "^style", group = "<!-- 5 -->🎨 Styling" },
+  { message = "^test", group = "<!-- 6 -->🧪 Testing" },
+  { message = "^chore\\(release\\): prepare for", skip = true },
+  { message = "^chore\\(deps.*\\)", skip = true },
+  { message = "^chore\\(pr\\)", skip = true },
+  { message = "^chore\\(pull\\)", skip = true },
+  { message = "^chore|^ci", group = "<!-- 7 -->⚙️ Miscellaneous Tasks" },
+  { body = ".*security", group = "<!-- 8 -->🛡️ Security" },
+  { message = "^revert", group = "<!-- 9 -->◀️ Revert" },
+]
+# protect breaking changes from being skipped due to matching a skipping commit_parser
+protect_breaking_commits = false
+# filter out the commits that are not matched by commit parsers
+filter_commits = false
+# regex for matching git tags
+# tag_pattern = "v[0-9].*"
+# regex for skipping tags
+# skip_tags = ""
+# regex for ignoring tags
+ignore_tags = ".*-nightly-.*"
+# sort the tags topologically
+topo_order = false
+# sort the commits inside sections by oldest/newest order
+sort_commits = "oldest"
+# limit the number of commits included in the changelog.
+# limit_commits = 42
--- a/config/datanode.example.toml
+++ b/config/datanode.example.toml
@@ -138,6 +138,18 @@ mem_threshold_on_create = "64M"
 # File system path to store intermediate files for external sorting (default `{data_home}/index_intermediate`).
 intermediate_path = ""

+[region_engine.mito.memtable]
+# Memtable type.
+# - "partition_tree": partition tree memtable
+# - "time_series": time-series memtable (deprecated)
+type = "partition_tree"
+# The max number of keys in one shard.
+index_max_keys_per_shard = 8192
+# The max rows of data inside the actively writing buffer in one shard.
+data_freeze_threshold = 32768
+# Max dictionary bytes.
+fork_dictionary_bytes = "1GiB"
+
 # Log options, see `standalone.example.toml`
 # [logging]
 # dir = "/tmp/greptimedb/logs"
--- a/config/metasrv.example.toml
+++ b/config/metasrv.example.toml
@@ -29,6 +29,12 @@ store_key_prefix = ""
 max_retry_times = 12
 # Initial retry delay of procedures, increases exponentially
 retry_delay = "500ms"
+# Auto split large value
+# GreptimeDB procedure uses etcd as the default metadata storage backend.
+# The etcd the maximum size of any request is 1.5 MiB
+# 1500KiB = 1536KiB (1.5MiB) - 36KiB (reserved size of key)
+# Comments out the `max_metadata_value_size`, for don't split large value (no limit).
+max_metadata_value_size = "1500KiB"

 # Failure detectors options.
 [failure_detector]
--- a/config/standalone.example.toml
+++ b/config/standalone.example.toml
@@ -244,6 +244,18 @@ mem_threshold_on_create = "64M"
 # File system path to store intermediate files for external sorting (default `{data_home}/index_intermediate`).
 intermediate_path = ""

+[region_engine.mito.memtable]
+# Memtable type.
+# - "partition_tree": partition tree memtable
+# - "time_series": time-series memtable (deprecated)
+type = "partition_tree"
+# The max number of keys in one shard.
+index_max_keys_per_shard = 8192
+# The max rows of data inside the actively writing buffer in one shard.
+data_freeze_threshold = 32768
+# Max dictionary bytes.
+fork_dictionary_bytes = "1GiB"
+
 # Log options
 # [logging]
 # Specify logs directory.
@@ -254,10 +266,11 @@ intermediate_path = ""
 # enable_otlp_tracing = false
 # tracing exporter endpoint with format `ip:port`, we use grpc oltp as exporter, default endpoint is `localhost:4317`
 # otlp_endpoint = "localhost:4317"
-# The percentage of tracing will be sampled and exported. Valid range `[0, 1]`, 1 means all traces are sampled, 0 means all traces are not sampled, the default value is 1. ratio > 1 are treated as 1. Fractions < 0 are treated as 0
-# tracing_sample_ratio = 1.0
 # Whether to append logs to stdout. Defaults to true.
 # append_stdout = true
+# The percentage of tracing will be sampled and exported. Valid range `[0, 1]`, 1 means all traces are sampled, 0 means all traces are not sampled, the default value is 1. ratio > 1 are treated as 1. Fractions < 0 are treated as 0
+# [logging.tracing_sample_ratio]
+# default_ratio = 0.0

 # Standalone export the metrics generated by itself
 # encoded to Prometheus remote-write format
--- a/docs/benchmarks/tsbs/v0.7.0.md
+++ b/docs/benchmarks/tsbs/v0.7.0.md
@@ -0,0 +1,50 @@
+# TSBS benchmark - v0.7.0
+
+## Environment
+
+### Local
+|        |                                    |
+| ------ | ---------------------------------- |
+| CPU    | AMD Ryzen 7 7735HS (8 core 3.2GHz) |
+| Memory | 32GB                               |
+| Disk   | SOLIDIGM SSDPFKNU010TZ             |
+| OS     | Ubuntu 22.04.2 LTS                 |
+
+### Amazon EC2
+
+|         |                |
+| ------- | -------------- |
+| Machine | c5d.2xlarge    |
+| CPU     | 8 core         |
+| Memory  | 16GB           |
+| Disk    | 50GB (GP3)     |
+| OS      | Ubuntu 22.04.1 |
+
+
+## Write performance
+
+| Environment        | Ingest rate (rows/s)  |
+| ------------------ | --------------------- |
+| Local              | 3695814.64            |
+| EC2 c5d.2xlarge    | 2987166.64            |
+
+
+## Query performance
+
+| Query type            | Local (ms) | EC2 c5d.2xlarge (ms)   |
+| --------------------- | ---------- | ---------------------- |
+| cpu-max-all-1         | 30.56      | 54.74                  |
+| cpu-max-all-8         | 52.69      | 70.50                  |
+| double-groupby-1      | 664.30     | 1366.63                |
+| double-groupby-5      | 1391.26    | 2141.71                |
+| double-groupby-all    | 2828.94    | 3389.59                |
+| groupby-orderby-limit | 718.92     | 1213.90                |
+| high-cpu-1            | 29.21      | 52.98                  |
+| high-cpu-all          | 5514.12    | 7194.91                |
+| lastpoint             | 7571.40    | 9423.41                |
+| single-groupby-1-1-1  | 19.09      | 7.77                   |
+| single-groupby-1-1-12 | 27.28      | 51.64                  |
+| single-groupby-1-8-1  | 31.85      | 11.64                  |
+| single-groupby-5-1-1  | 16.14      | 9.67                   |
+| single-groupby-5-1-12 | 27.21      | 53.62                  |
+| single-groupby-5-8-1  | 39.62      | 14.96                  |
--- a/docs/rfcs/2023-05-09-distributed-planner.md
+++ b/docs/rfcs/2023-05-09-distributed-planner.md
@@ -79,7 +79,7 @@ This RFC proposes to add a new expression node `MergeScan` to merge result from
 │               │    │                             │
 └─Frontend──────┘    └─Remote-Sources──────────────┘
 ```
-This merge operation simply chains all the the underlying remote data sources and return `RecordBatch`, just like a coalesce op. And each remote sources is a gRPC query to datanode via the substrait logical plan interface. The plan is transformed and divided from the original query that comes to frontend.
+This merge operation simply chains all the underlying remote data sources and return `RecordBatch`, just like a coalesce op. And each remote sources is a gRPC query to datanode via the substrait logical plan interface. The plan is transformed and divided from the original query that comes to frontend.

 ## Commutativity of MergeScan

--- a/docs/rfcs/2024-02-21-multi-dimension-partition-rule/2d-example.png
+++ b/docs/rfcs/2024-02-21-multi-dimension-partition-rule/2d-example.png
--- a/docs/rfcs/2024-02-21-multi-dimension-partition-rule/rfc.md
+++ b/docs/rfcs/2024-02-21-multi-dimension-partition-rule/rfc.md
@@ -0,0 +1,101 @@
+---
+Feature Name:  Multi-dimension Partition Rule
+Tracking Issue: https://github.com/GreptimeTeam/greptimedb/issues/3351
+Date: 2024-02-21
+Author: "Ruihang Xia <waynestxia@gmail.com>"
+---
+
+# Summary
+
+A new region partition scheme that runs on multiple dimensions of the key space. The partition rule is defined by a set of simple expressions on the partition key columns.
+
+# Motivation
+
+The current partition rule is from MySQL's [`RANGE Partition`](https://dev.mysql.com/doc/refman/8.0/en/partitioning-range.html), which is based on a single dimension. It is sort of a [Hilbert Curve](https://en.wikipedia.org/wiki/Hilbert_curve) and pick several point on the curve to divide the space. It is neither easy to understand how the data get partitioned nor flexible enough to handle complex partitioning requirements.
+
+Considering the future requirements like region repartitioning or autonomous rebalancing, where both workload and partition may change frequently. Here proposes a new region partition scheme that uses a set of simple expressions on the partition key columns to divide the key space.
+
+# Details
+
+## Partition rule
+
+First, we define a simple expression that can be used to define the partition rule. The simple expression is a binary expression expression on the partition key columns that can be evaluated to a boolean value. The binary operator is limited to comparison operators only, like `=`, `!=`, `>`, `>=`, `<`, `<=`. And the operands are limited either literal value or partition column.
+
+Example of valid simple expressions are $`col_A = 10`$, $`col_A \gt 10 \& col_B \gt 20`$ or $`col_A \ne 10`$.
+
+Those expressions can be used as predicates to divide the key space into different regions. The following example have two partition columns `Col A` and `Col B`, and four partitioned regions.
+
+```math
+\left\{\begin{aligned}
+ 
+&col_A \le 10 &Region_1 \\
+&10 \lt col_A \& col_A \le 20 &Region_2 \\
+&20 \lt col_A \space \& \space col_B \lt 100 &Region_3 \\
+&20 \lt col_A \space \& \space col_B \ge 100 &Region_4
+
+\end{aligned}\right\}
+```
+
+An advantage of this scheme is that it is easy to understand how the data get partitioned. The above example can be visualized in a 2D space (two partition column is involved in the example).
+
+![example](2d-example.png)
+
+Here each expression draws a line in the 2D space. Managing data partitioning becomes a matter of drawing lines in the key space.
+
+To make it easy to use, there is a "default region" which catches all the data that doesn't match any of previous expressions. The default region exist by default and do not need to specify. It is also possible to remove this default region if the DB finds it is not necessary.
+
+## SQL interface
+
+The SQL interface is in response to two parts: specifying the partition columns and the partition rule. Thouth we are targeting an autonomous system, it's still allowed to give some bootstrap rules or hints on creating table.
+
+Partition column is specified by `PARTITION ON COLUMNS` sub-clause in `CREATE TABLE`:
+
+```sql
+CREATE TABLE t (...)
+PARTITION ON COLUMNS (...) ();
+```
+
+Two following brackets are for partition columns and partition rule respectively.
+
+Columns provided here are only used as an allow-list of how the partition rule can be defined. Which means (a) the sequence between columns doesn't matter, (b) the columns provided here are not necessarily being used in the partition rule.
+
+The partition rule part is a list of comma-separated simple expressions. Expressions here are not corresponding to region, as they might be changed by system to fit various workload.
+
+A full example of `CREATE TABLE` with partition rule is:
+
+```sql
+CREATE TABLE IF NOT EXISTS demo (
+  a STRING,
+  b STRING,
+  c STRING,
+  d STRING,
+  ts TIMESTAMP,
+  memory DOUBLE,
+  TIME INDEX (ts),
+  PRIMARY KEY (a, b, c, d)
+)
+PARTITION ON COLUMNS (c, b, a) (
+  a < 10,
+  10 >= a AND a < 20,
+  20 >= a AND b < 100,
+  20 >= a AND b > 100
+)
+```
+
+## Combine with storage
+
+Examining columns separately suits our columnar storage very well in two aspects.
+
+1. The simple expression can be pushed down to storage and file format, and is likely to hit existing index. Makes pruning operation very efficient.
+
+2. Columns in columnar storage are not tightly coupled like in the traditional row storages, which means we can easily add or remove columns from partition rule without much impact (like a global reshuffle) on data.
+
+The data file itself can be "projected" to the key space as a polyhedron, it is guaranteed that each plane is in parallel with some coordinate planes (in a 2D scenario, this is saying that all the files can be projected to a rectangle). Thus partition or repartition also only need to consider related columns.
+
+![sst-project](sst-project.png)
+
+An additional limitation is that considering how the index works and how we organize the primary keys at present, the partition columns are limited to be a subset of primary keys for better performance.
+
+# Drawbacks
+
+This is a breaking change.
--- a/docs/rfcs/2024-02-21-multi-dimension-partition-rule/sst-project.png
+++ b/docs/rfcs/2024-02-21-multi-dimension-partition-rule/sst-project.png
--- a/licenserc.toml
+++ b/licenserc.toml
@@ -19,6 +19,12 @@ includes = [
    "*.py",
 ]

+excludes = [
+    # copied sources
+    "src/common/base/src/readable_size.rs",
+    "src/servers/src/repeated_field.rs",
+]
+
 [properties]
 inceptionYear = 2023
 copyrightOwner = "Greptime Team"
--- a/scripts/fetch-dashboard-assets.sh
+++ b/scripts/fetch-dashboard-assets.sh
@@ -27,7 +27,7 @@ function retry_fetch() {
        echo "Failed to download $url"
        echo "You may try to set http_proxy and https_proxy environment variables."
        if [[ -z "$GITHUB_PROXY_URL" ]]; then
-          echo "You may try to set GITHUB_PROXY_URL=http://mirror.ghproxy.com/"
+          echo "You may try to set GITHUB_PROXY_URL=http://mirror.ghproxy.com/https://github.com/"
        fi
        exit 1
     }
@@ -39,7 +39,7 @@ function retry_fetch() {
 retry_fetch "${GITHUB_URL}/GreptimeTeam/dashboard/releases/download/${RELEASE_VERSION}/sha256.txt" sha256.txt

 # Download the tar file containing the built dashboard assets.
-retry_fetch "${GITHUB_URL}/GreptimeTeam/dashboard/releases/download/$RELEASE_VERSION/build.tar.gz" build.tar.gz
+retry_fetch "${GITHUB_URL}/GreptimeTeam/dashboard/releases/download/${RELEASE_VERSION}/build.tar.gz" build.tar.gz

 # Verify the checksums match; exit if they don't.
 case "$(uname -s)" in
--- a/src/api/Cargo.toml
+++ b/src/api/Cargo.toml
@@ -18,7 +18,6 @@ greptime-proto.workspace = true
 paste = "1.0"
 prost.workspace = true
 snafu.workspace = true
-tonic.workspace = true

 [build-dependencies]
 tonic-build = "0.9"
--- a/src/api/src/helper.rs
+++ b/src/api/src/helper.rs
@@ -707,7 +707,6 @@ pub fn pb_values_to_vector_ref(data_type: &ConcreteDataType, values: Values) ->
 }

 pub fn pb_values_to_values(data_type: &ConcreteDataType, values: Values) -> Vec<Value> {
-    // TODO(fys): use macros to optimize code
    match data_type {
        ConcreteDataType::Int64(_) => values
            .i64_values
--- a/src/auth/Cargo.toml
+++ b/src/auth/Cargo.toml
@@ -16,8 +16,9 @@ api.workspace = true
 async-trait.workspace = true
 common-error.workspace = true
 common-macro.workspace = true
+common-telemetry.workspace = true
 digest = "0.10"
-hex = { version = "0.4" }
+notify.workspace = true
 secrecy = { version = "0.8", features = ["serde", "alloc"] }
 sha1 = "0.10"
 snafu.workspace = true
--- a/src/auth/src/common.rs
+++ b/src/auth/src/common.rs
@@ -22,6 +22,9 @@ use snafu::{ensure, OptionExt};
 use crate::error::{IllegalParamSnafu, InvalidConfigSnafu, Result, UserPasswordMismatchSnafu};
 use crate::user_info::DefaultUserInfo;
 use crate::user_provider::static_user_provider::{StaticUserProvider, STATIC_USER_PROVIDER};
+use crate::user_provider::watch_file_user_provider::{
+    WatchFileUserProvider, WATCH_FILE_USER_PROVIDER,
+};
 use crate::{UserInfoRef, UserProviderRef};

 pub(crate) const DEFAULT_USERNAME: &str = "greptime";
@@ -40,9 +43,12 @@ pub fn user_provider_from_option(opt: &String) -> Result<UserProviderRef> {
    match name {
        STATIC_USER_PROVIDER => {
            let provider =
-                StaticUserProvider::try_from(content).map(|p| Arc::new(p) as UserProviderRef)?;
+                StaticUserProvider::new(content).map(|p| Arc::new(p) as UserProviderRef)?;
            Ok(provider)
        }
+        WATCH_FILE_USER_PROVIDER => {
+            WatchFileUserProvider::new(content).map(|p| Arc::new(p) as UserProviderRef)
+        }
        _ => InvalidConfigSnafu {
            value: name.to_string(),
            msg: "Invalid UserProviderOption",
--- a/src/auth/src/error.rs
+++ b/src/auth/src/error.rs
@@ -64,6 +64,13 @@ pub enum Error {
        username: String,
    },

+    #[snafu(display("Failed to initialize a watcher for file {}", path))]
+    FileWatch {
+        path: String,
+        #[snafu(source)]
+        error: notify::Error,
+    },
+
    #[snafu(display("User is not authorized to perform this action"))]
    PermissionDenied { location: Location },
 }
@@ -73,6 +80,7 @@ impl ErrorExt for Error {
        match self {
            Error::InvalidConfig { .. } => StatusCode::InvalidArguments,
            Error::IllegalParam { .. } => StatusCode::InvalidArguments,
+            Error::FileWatch { .. } => StatusCode::InvalidArguments,
            Error::InternalState { .. } => StatusCode::Unexpected,
            Error::Io { .. } => StatusCode::Internal,
            Error::AuthBackend { .. } => StatusCode::Internal,
--- a/src/auth/src/user_provider.rs
+++ b/src/auth/src/user_provider.rs
@@ -13,10 +13,24 @@
 // limitations under the License.

 pub(crate) mod static_user_provider;
+pub(crate) mod watch_file_user_provider;
+
+use std::collections::HashMap;
+use std::fs::File;
+use std::io;
+use std::io::BufRead;
+use std::path::Path;
+
+use secrecy::ExposeSecret;
+use snafu::{ensure, OptionExt, ResultExt};

 use crate::common::{Identity, Password};
-use crate::error::Result;
-use crate::UserInfoRef;
+use crate::error::{
+    IllegalParamSnafu, InvalidConfigSnafu, IoSnafu, Result, UnsupportedPasswordTypeSnafu,
+    UserNotFoundSnafu, UserPasswordMismatchSnafu,
+};
+use crate::user_info::DefaultUserInfo;
+use crate::{auth_mysql, UserInfoRef};

 #[async_trait::async_trait]
 pub trait UserProvider: Send + Sync {
@@ -44,3 +58,88 @@ pub trait UserProvider: Send + Sync {
        Ok(user_info)
    }
 }
+
+fn load_credential_from_file(filepath: &str) -> Result<Option<HashMap<String, Vec<u8>>>> {
+    // check valid path
+    let path = Path::new(filepath);
+    if !path.exists() {
+        return Ok(None);
+    }
+
+    ensure!(
+        path.is_file(),
+        InvalidConfigSnafu {
+            value: filepath,
+            msg: "UserProvider file must be a file",
+        }
+    );
+    let file = File::open(path).context(IoSnafu)?;
+    let credential = io::BufReader::new(file)
+        .lines()
+        .map_while(std::result::Result::ok)
+        .filter_map(|line| {
+            if let Some((k, v)) = line.split_once('=') {
+                Some((k.to_string(), v.as_bytes().to_vec()))
+            } else {
+                None
+            }
+        })
+        .collect::<HashMap<String, Vec<u8>>>();
+
+    ensure!(
+        !credential.is_empty(),
+        InvalidConfigSnafu {
+            value: filepath,
+            msg: "UserProvider's file must contains at least one valid credential",
+        }
+    );
+
+    Ok(Some(credential))
+}
+
+fn authenticate_with_credential(
+    users: &HashMap<String, Vec<u8>>,
+    input_id: Identity<'_>,
+    input_pwd: Password<'_>,
+) -> Result<UserInfoRef> {
+    match input_id {
+        Identity::UserId(username, _) => {
+            ensure!(
+                !username.is_empty(),
+                IllegalParamSnafu {
+                    msg: "blank username"
+                }
+            );
+            let save_pwd = users.get(username).context(UserNotFoundSnafu {
+                username: username.to_string(),
+            })?;
+
+            match input_pwd {
+                Password::PlainText(pwd) => {
+                    ensure!(
+                        !pwd.expose_secret().is_empty(),
+                        IllegalParamSnafu {
+                            msg: "blank password"
+                        }
+                    );
+                    if save_pwd == pwd.expose_secret().as_bytes() {
+                        Ok(DefaultUserInfo::with_name(username))
+                    } else {
+                        UserPasswordMismatchSnafu {
+                            username: username.to_string(),
+                        }
+                        .fail()
+                    }
+                }
+                Password::MysqlNativePassword(auth_data, salt) => {
+                    auth_mysql(auth_data, salt, username, save_pwd)
+                        .map(|_| DefaultUserInfo::with_name(username))
+                }
+                Password::PgMD5(_, _) => UnsupportedPasswordTypeSnafu {
+                    password_type: "pg_md5",
+                }
+                .fail(),
+            }
+        }
+    }
+}
--- a/src/auth/src/user_provider/static_user_provider.rs
+++ b/src/auth/src/user_provider/static_user_provider.rs
@@ -13,60 +13,34 @@
 // limitations under the License.

 use std::collections::HashMap;
-use std::fs::File;
-use std::io;
-use std::io::BufRead;
-use std::path::Path;

 use async_trait::async_trait;
-use secrecy::ExposeSecret;
-use snafu::{ensure, OptionExt, ResultExt};
+use snafu::OptionExt;

-use crate::error::{
-    Error, IllegalParamSnafu, InvalidConfigSnafu, IoSnafu, Result, UnsupportedPasswordTypeSnafu,
-    UserNotFoundSnafu, UserPasswordMismatchSnafu,
-};
-use crate::user_info::DefaultUserInfo;
-use crate::{auth_mysql, Identity, Password, UserInfoRef, UserProvider};
+use crate::error::{InvalidConfigSnafu, Result};
+use crate::user_provider::{authenticate_with_credential, load_credential_from_file};
+use crate::{Identity, Password, UserInfoRef, UserProvider};

 pub(crate) const STATIC_USER_PROVIDER: &str = "static_user_provider";

-impl TryFrom<&str> for StaticUserProvider {
-    type Error = Error;
+pub(crate) struct StaticUserProvider {
+    users: HashMap<String, Vec<u8>>,
+}

-    fn try_from(value: &str) -> Result<Self> {
+impl StaticUserProvider {
+    pub(crate) fn new(value: &str) -> Result<Self> {
        let (mode, content) = value.split_once(':').context(InvalidConfigSnafu {
            value: value.to_string(),
            msg: "StaticUserProviderOption must be in format `<option>:<value>`",
        })?;
        return match mode {
            "file" => {
-                // check valid path
-                let path = Path::new(content);
-                ensure!(path.exists() && path.is_file(), InvalidConfigSnafu {
-                    value: content.to_string(),
-                    msg: "StaticUserProviderOption file must be a valid file path",
-                });
-
-                let file = File::open(path).context(IoSnafu)?;
-                let credential = io::BufReader::new(file)
-                    .lines()
-                    .map_while(std::result::Result::ok)
-                    .filter_map(|line| {
-                        if let Some((k, v)) = line.split_once('=') {
-                            Some((k.to_string(), v.as_bytes().to_vec()))
-                        } else {
-                            None
-                        }
-                    })
-                    .collect::<HashMap<String, Vec<u8>>>();
-
-                ensure!(!credential.is_empty(), InvalidConfigSnafu {
-                    value: content.to_string(),
-                    msg: "StaticUserProviderOption file must contains at least one valid credential",
-                });
-
-                Ok(StaticUserProvider { users: credential, })
+                let users = load_credential_from_file(content)?
+                    .context(InvalidConfigSnafu {
+                        value: content.to_string(),
+                        msg: "StaticFileUserProvider must be a valid file path",
+                    })?;
+                Ok(StaticUserProvider { users })
            }
            "cmd" => content
                .split(',')
@@ -83,66 +57,19 @@ impl TryFrom<&str> for StaticUserProvider {
                value: mode.to_string(),
                msg: "StaticUserProviderOption must be in format `file:<path>` or `cmd:<values>`",
            }
-            .fail(),
+                .fail(),
        };
    }
 }

-pub(crate) struct StaticUserProvider {
-    users: HashMap<String, Vec<u8>>,
-}
-
 #[async_trait]
 impl UserProvider for StaticUserProvider {
    fn name(&self) -> &str {
        STATIC_USER_PROVIDER
    }

-    async fn authenticate(
-        &self,
-        input_id: Identity<'_>,
-        input_pwd: Password<'_>,
-    ) -> Result<UserInfoRef> {
-        match input_id {
-            Identity::UserId(username, _) => {
-                ensure!(
-                    !username.is_empty(),
-                    IllegalParamSnafu {
-                        msg: "blank username"
-                    }
-                );
-                let save_pwd = self.users.get(username).context(UserNotFoundSnafu {
-                    username: username.to_string(),
-                })?;
-
-                match input_pwd {
-                    Password::PlainText(pwd) => {
-                        ensure!(
-                            !pwd.expose_secret().is_empty(),
-                            IllegalParamSnafu {
-                                msg: "blank password"
-                            }
-                        );
-                        return if save_pwd == pwd.expose_secret().as_bytes() {
-                            Ok(DefaultUserInfo::with_name(username))
-                        } else {
-                            UserPasswordMismatchSnafu {
-                                username: username.to_string(),
-                            }
-                            .fail()
-                        };
-                    }
-                    Password::MysqlNativePassword(auth_data, salt) => {
-                        auth_mysql(auth_data, salt, username, save_pwd)
-                            .map(|_| DefaultUserInfo::with_name(username))
-                    }
-                    Password::PgMD5(_, _) => UnsupportedPasswordTypeSnafu {
-                        password_type: "pg_md5",
-                    }
-                    .fail(),
-                }
-            }
-        }
+    async fn authenticate(&self, id: Identity<'_>, pwd: Password<'_>) -> Result<UserInfoRef> {
+        authenticate_with_credential(&self.users, id, pwd)
    }

    async fn authorize(
@@ -181,7 +108,7 @@ pub mod test {
    #[tokio::test]
    async fn test_authorize() {
        let user_info = DefaultUserInfo::with_name("root");
-        let provider = StaticUserProvider::try_from("cmd:root=123456,admin=654321").unwrap();
+        let provider = StaticUserProvider::new("cmd:root=123456,admin=654321").unwrap();
        provider
            .authorize("catalog", "schema", &user_info)
            .await
@@ -190,7 +117,7 @@ pub mod test {

    #[tokio::test]
    async fn test_inline_provider() {
-        let provider = StaticUserProvider::try_from("cmd:root=123456,admin=654321").unwrap();
+        let provider = StaticUserProvider::new("cmd:root=123456,admin=654321").unwrap();
        test_authenticate(&provider, "root", "123456").await;
        test_authenticate(&provider, "admin", "654321").await;
    }
@@ -214,7 +141,7 @@ admin=654321",
        }

        let param = format!("file:{file_path}");
-        let provider = StaticUserProvider::try_from(param.as_str()).unwrap();
+        let provider = StaticUserProvider::new(param.as_str()).unwrap();
        test_authenticate(&provider, "root", "123456").await;
        test_authenticate(&provider, "admin", "654321").await;
    }
--- a/src/auth/src/user_provider/watch_file_user_provider.rs
+++ b/src/auth/src/user_provider/watch_file_user_provider.rs
@@ -0,0 +1,215 @@
+// Copyright 2023 Greptime Team
+//
+// Licensed under the Apache License, Version 2.0 (the "License");
+// you may not use this file except in compliance with the License.
+// You may obtain a copy of the License at
+//
+//     http://www.apache.org/licenses/LICENSE-2.0
+//
+// Unless required by applicable law or agreed to in writing, software
+// distributed under the License is distributed on an "AS IS" BASIS,
+// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+// See the License for the specific language governing permissions and
+// limitations under the License.
+
+use std::collections::HashMap;
+use std::path::Path;
+use std::sync::mpsc::channel;
+use std::sync::{Arc, Mutex};
+
+use async_trait::async_trait;
+use common_telemetry::{info, warn};
+use notify::{EventKind, RecursiveMode, Watcher};
+use snafu::{ensure, ResultExt};
+
+use crate::error::{FileWatchSnafu, InvalidConfigSnafu, Result};
+use crate::user_info::DefaultUserInfo;
+use crate::user_provider::{authenticate_with_credential, load_credential_from_file};
+use crate::{Identity, Password, UserInfoRef, UserProvider};
+
+pub(crate) const WATCH_FILE_USER_PROVIDER: &str = "watch_file_user_provider";
+
+type WatchedCredentialRef = Arc<Mutex<Option<HashMap<String, Vec<u8>>>>>;
+
+/// A user provider that reads user credential from a file and watches the file for changes.
+///
+/// Empty file is invalid; but file not exist means every user can be authenticated.
+pub(crate) struct WatchFileUserProvider {
+    users: WatchedCredentialRef,
+}
+
+impl WatchFileUserProvider {
+    pub fn new(filepath: &str) -> Result<Self> {
+        let credential = load_credential_from_file(filepath)?;
+        let users = Arc::new(Mutex::new(credential));
+        let this = WatchFileUserProvider {
+            users: users.clone(),
+        };
+
+        let (tx, rx) = channel::<notify::Result<notify::Event>>();
+        let mut debouncer =
+            notify::recommended_watcher(tx).context(FileWatchSnafu { path: "<none>" })?;
+        let mut dir = Path::new(filepath).to_path_buf();
+        ensure!(
+            dir.pop(),
+            InvalidConfigSnafu {
+                value: filepath,
+                msg: "UserProvider path must be a file path",
+            }
+        );
+        debouncer
+            .watch(&dir, RecursiveMode::NonRecursive)
+            .context(FileWatchSnafu { path: filepath })?;
+
+        let filepath = filepath.to_string();
+        std::thread::spawn(move || {
+            let filename = Path::new(&filepath).file_name();
+            let _hold = debouncer;
+            while let Ok(res) = rx.recv() {
+                if let Ok(event) = res {
+                    let is_this_file = event.paths.iter().any(|p| p.file_name() == filename);
+                    let is_relevant_event = matches!(
+                        event.kind,
+                        EventKind::Modify(_) | EventKind::Create(_) | EventKind::Remove(_)
+                    );
+                    if is_this_file && is_relevant_event {
+                        info!(?event.kind, "User provider file {} changed", &filepath);
+                        match load_credential_from_file(&filepath) {
+                            Ok(credential) => {
+                                let mut users =
+                                    users.lock().expect("users credential must be valid");
+                                #[cfg(not(test))]
+                                info!("User provider file {filepath} reloaded");
+                                #[cfg(test)]
+                                info!("User provider file {filepath} reloaded: {credential:?}");
+                                *users = credential;
+                            }
+                            Err(err) => {
+                                warn!(
+                                    ?err,
+                                    "Fail to load credential from file {filepath}; keep the old one",
+                                )
+                            }
+                        }
+                    }
+                }
+            }
+        });
+
+        Ok(this)
+    }
+}
+
+#[async_trait]
+impl UserProvider for WatchFileUserProvider {
+    fn name(&self) -> &str {
+        WATCH_FILE_USER_PROVIDER
+    }
+
+    async fn authenticate(&self, id: Identity<'_>, password: Password<'_>) -> Result<UserInfoRef> {
+        let users = self.users.lock().expect("users credential must be valid");
+        if let Some(users) = users.as_ref() {
+            authenticate_with_credential(users, id, password)
+        } else {
+            match id {
+                Identity::UserId(id, _) => {
+                    warn!(id, "User provider file not exist, allow all users");
+                    Ok(DefaultUserInfo::with_name(id))
+                }
+            }
+        }
+    }
+
+    async fn authorize(&self, _: &str, _: &str, _: &UserInfoRef) -> Result<()> {
+        // default allow all
+        Ok(())
+    }
+}
+
+#[cfg(test)]
+pub mod test {
+    use std::time::{Duration, Instant};
+
+    use common_test_util::temp_dir::create_temp_dir;
+    use tokio::time::sleep;
+
+    use crate::user_provider::watch_file_user_provider::WatchFileUserProvider;
+    use crate::user_provider::{Identity, Password};
+    use crate::UserProvider;
+
+    async fn test_authenticate(
+        provider: &dyn UserProvider,
+        username: &str,
+        password: &str,
+        ok: bool,
+        timeout: Option<Duration>,
+    ) {
+        if let Some(timeout) = timeout {
+            let deadline = Instant::now().checked_add(timeout).unwrap();
+            loop {
+                let re = provider
+                    .authenticate(
+                        Identity::UserId(username, None),
+                        Password::PlainText(password.to_string().into()),
+                    )
+                    .await;
+                if re.is_ok() == ok {
+                    break;
+                } else if Instant::now() < deadline {
+                    sleep(Duration::from_millis(100)).await;
+                } else {
+                    panic!("timeout (username: {username}, password: {password}, expected: {ok})");
+                }
+            }
+        } else {
+            let re = provider
+                .authenticate(
+                    Identity::UserId(username, None),
+                    Password::PlainText(password.to_string().into()),
+                )
+                .await;
+            assert_eq!(
+                re.is_ok(),
+                ok,
+                "username: {}, password: {}",
+                username,
+                password
+            );
+        }
+    }
+
+    #[tokio::test]
+    async fn test_file_provider() {
+        common_telemetry::init_default_ut_logging();
+
+        let dir = create_temp_dir("test_file_provider");
+        let file_path = format!("{}/test_file_provider", dir.path().to_str().unwrap());
+
+        // write a tmp file
+        assert!(std::fs::write(&file_path, "root=123456\nadmin=654321\n").is_ok());
+        let provider = WatchFileUserProvider::new(file_path.as_str()).unwrap();
+        let timeout = Duration::from_secs(60);
+
+        test_authenticate(&provider, "root", "123456", true, None).await;
+        test_authenticate(&provider, "admin", "654321", true, None).await;
+        test_authenticate(&provider, "root", "654321", false, None).await;
+
+        // update the tmp file
+        assert!(std::fs::write(&file_path, "root=654321\n").is_ok());
+        test_authenticate(&provider, "root", "123456", false, Some(timeout)).await;
+        test_authenticate(&provider, "root", "654321", true, Some(timeout)).await;
+        test_authenticate(&provider, "admin", "654321", false, Some(timeout)).await;
+
+        // remove the tmp file
+        assert!(std::fs::remove_file(&file_path).is_ok());
+        test_authenticate(&provider, "root", "123456", true, Some(timeout)).await;
+        test_authenticate(&provider, "root", "654321", true, Some(timeout)).await;
+        test_authenticate(&provider, "admin", "654321", true, Some(timeout)).await;
+
+        // recreate the tmp file
+        assert!(std::fs::write(&file_path, "root=123456\n").is_ok());
+        test_authenticate(&provider, "root", "123456", true, Some(timeout)).await;
+        test_authenticate(&provider, "root", "654321", false, Some(timeout)).await;
+        test_authenticate(&provider, "admin", "654321", false, Some(timeout)).await;
+    }
+}
--- a/src/catalog/Cargo.toml
+++ b/src/catalog/Cargo.toml
@@ -12,19 +12,16 @@ workspace = true

 [dependencies]
 api.workspace = true
-arc-swap = "1.0"
 arrow.workspace = true
 arrow-schema.workspace = true
 async-stream.workspace = true
 async-trait = "0.1"
 common-catalog.workspace = true
 common-error.workspace = true
-common-grpc.workspace = true
 common-macro.workspace = true
 common-meta.workspace = true
 common-query.workspace = true
 common-recordbatch.workspace = true
-common-runtime.workspace = true
 common-telemetry.workspace = true
 common-time.workspace = true
 common-version.workspace = true
@@ -37,15 +34,13 @@ itertools.workspace = true
 lazy_static.workspace = true
 meta-client.workspace = true
 moka = { workspace = true, features = ["future", "sync"] }
-parking_lot = "0.12"
 partition.workspace = true
 paste = "1.0"
 prometheus.workspace = true
-regex.workspace = true
-serde.workspace = true
 serde_json.workspace = true
 session.workspace = true
 snafu.workspace = true
+sql.workspace = true
 store-api.workspace = true
 table.workspace = true
 tokio.workspace = true
--- a/src/catalog/src/information_schema.rs
+++ b/src/catalog/src/information_schema.rs
@@ -12,16 +12,16 @@
 // See the License for the specific language governing permissions and
 // limitations under the License.

-mod columns;
-mod key_column_usage;
+pub mod columns;
+pub mod key_column_usage;
 mod memory_table;
 mod partitions;
 mod predicate;
 mod region_peers;
 mod runtime_metrics;
-mod schemata;
+pub mod schemata;
 mod table_names;
-mod tables;
+pub mod tables;

 use std::collections::HashMap;
 use std::sync::{Arc, Weak};
@@ -41,8 +41,7 @@ use table::error::{SchemaConversionSnafu, TablesRecordBatchSnafu};
 use table::metadata::{
    FilterPushDownType, TableInfoBuilder, TableInfoRef, TableMetaBuilder, TableType,
 };
-use table::thin_table::{ThinTable, ThinTableAdapter};
-use table::TableRef;
+use table::{Table, TableRef};
 pub use table_names::*;

 use self::columns::InformationSchemaColumns;
@@ -187,10 +186,9 @@ impl InformationSchemaProvider {
        self.information_table(name).map(|table| {
            let table_info = Self::table_info(self.catalog_name.clone(), &table);
            let filter_pushdown = FilterPushDownType::Inexact;
-            let thin_table = ThinTable::new(table_info, filter_pushdown);
-
            let data_source = Arc::new(InformationTableDataSource::new(table));
-            Arc::new(ThinTableAdapter::new(thin_table, data_source)) as _
+            let table = Table::new(table_info, filter_pushdown, data_source);
+            Arc::new(table)
        })
    }

--- a/src/catalog/src/information_schema/columns.rs
+++ b/src/catalog/src/information_schema/columns.rs
@@ -26,13 +26,16 @@ use common_recordbatch::{RecordBatch, SendableRecordBatchStream};
 use datafusion::physical_plan::stream::RecordBatchStreamAdapter as DfRecordBatchStreamAdapter;
 use datafusion::physical_plan::streaming::PartitionStream as DfPartitionStream;
 use datafusion::physical_plan::SendableRecordBatchStream as DfSendableRecordBatchStream;
-use datatypes::prelude::{ConcreteDataType, DataType};
+use datatypes::prelude::{ConcreteDataType, DataType, MutableVector};
 use datatypes::scalars::ScalarVectorBuilder;
 use datatypes::schema::{ColumnSchema, Schema, SchemaRef};
 use datatypes::value::Value;
-use datatypes::vectors::{StringVectorBuilder, VectorRef};
+use datatypes::vectors::{
+    ConstantVector, Int64Vector, Int64VectorBuilder, StringVector, StringVectorBuilder, VectorRef,
+};
 use futures::TryStreamExt;
 use snafu::{OptionExt, ResultExt};
+use sql::statements;
 use store_api::storage::{ScanRequest, TableId};

 use super::{InformationTable, COLUMNS};
@@ -48,18 +51,42 @@ pub(super) struct InformationSchemaColumns {
    catalog_manager: Weak<dyn CatalogManager>,
 }

-const TABLE_CATALOG: &str = "table_catalog";
-const TABLE_SCHEMA: &str = "table_schema";
-const TABLE_NAME: &str = "table_name";
-const COLUMN_NAME: &str = "column_name";
-const DATA_TYPE: &str = "data_type";
-const SEMANTIC_TYPE: &str = "semantic_type";
-const COLUMN_DEFAULT: &str = "column_default";
-const IS_NULLABLE: &str = "is_nullable";
+pub const TABLE_CATALOG: &str = "table_catalog";
+pub const TABLE_SCHEMA: &str = "table_schema";
+pub const TABLE_NAME: &str = "table_name";
+pub const COLUMN_NAME: &str = "column_name";
+const ORDINAL_POSITION: &str = "ordinal_position";
+const CHARACTER_MAXIMUM_LENGTH: &str = "character_maximum_length";
+const CHARACTER_OCTET_LENGTH: &str = "character_octet_length";
+const NUMERIC_PRECISION: &str = "numeric_precision";
+const NUMERIC_SCALE: &str = "numeric_scale";
+const DATETIME_PRECISION: &str = "datetime_precision";
+const CHARACTER_SET_NAME: &str = "character_set_name";
+pub const COLLATION_NAME: &str = "collation_name";
+pub const COLUMN_KEY: &str = "column_key";
+pub const EXTRA: &str = "extra";
+pub const PRIVILEGES: &str = "privileges";
+const GENERATION_EXPRESSION: &str = "generation_expression";
+// Extension field to keep greptime data type name
+pub const GREPTIME_DATA_TYPE: &str = "greptime_data_type";
+pub const DATA_TYPE: &str = "data_type";
+pub const SEMANTIC_TYPE: &str = "semantic_type";
+pub const COLUMN_DEFAULT: &str = "column_default";
+pub const IS_NULLABLE: &str = "is_nullable";
 const COLUMN_TYPE: &str = "column_type";
-const COLUMN_COMMENT: &str = "column_comment";
+pub const COLUMN_COMMENT: &str = "column_comment";
+const SRS_ID: &str = "srs_id";
 const INIT_CAPACITY: usize = 42;

+// The maximum length of string type
+const MAX_STRING_LENGTH: i64 = 2147483647;
+const UTF8_CHARSET_NAME: &str = "utf8";
+const UTF8_COLLATE_NAME: &str = "utf8_bin";
+const PRI_COLUMN_KEY: &str = "PRI";
+const TIME_INDEX_COLUMN_KEY: &str = "TIME INDEX";
+const DEFAULT_PRIVILEGES: &str = "select,insert";
+const EMPTY_STR: &str = "";
+
 impl InformationSchemaColumns {
    pub(super) fn new(catalog_name: String, catalog_manager: Weak<dyn CatalogManager>) -> Self {
        Self {
@@ -75,12 +102,46 @@ impl InformationSchemaColumns {
            ColumnSchema::new(TABLE_SCHEMA, ConcreteDataType::string_datatype(), false),
            ColumnSchema::new(TABLE_NAME, ConcreteDataType::string_datatype(), false),
            ColumnSchema::new(COLUMN_NAME, ConcreteDataType::string_datatype(), false),
+            ColumnSchema::new(ORDINAL_POSITION, ConcreteDataType::int64_datatype(), false),
+            ColumnSchema::new(
+                CHARACTER_MAXIMUM_LENGTH,
+                ConcreteDataType::int64_datatype(),
+                true,
+            ),
+            ColumnSchema::new(
+                CHARACTER_OCTET_LENGTH,
+                ConcreteDataType::int64_datatype(),
+                true,
+            ),
+            ColumnSchema::new(NUMERIC_PRECISION, ConcreteDataType::int64_datatype(), true),
+            ColumnSchema::new(NUMERIC_SCALE, ConcreteDataType::int64_datatype(), true),
+            ColumnSchema::new(DATETIME_PRECISION, ConcreteDataType::int64_datatype(), true),
+            ColumnSchema::new(
+                CHARACTER_SET_NAME,
+                ConcreteDataType::string_datatype(),
+                true,
+            ),
+            ColumnSchema::new(COLLATION_NAME, ConcreteDataType::string_datatype(), true),
+            ColumnSchema::new(COLUMN_KEY, ConcreteDataType::string_datatype(), false),
+            ColumnSchema::new(EXTRA, ConcreteDataType::string_datatype(), false),
+            ColumnSchema::new(PRIVILEGES, ConcreteDataType::string_datatype(), false),
+            ColumnSchema::new(
+                GENERATION_EXPRESSION,
+                ConcreteDataType::string_datatype(),
+                false,
+            ),
+            ColumnSchema::new(
+                GREPTIME_DATA_TYPE,
+                ConcreteDataType::string_datatype(),
+                false,
+            ),
            ColumnSchema::new(DATA_TYPE, ConcreteDataType::string_datatype(), false),
            ColumnSchema::new(SEMANTIC_TYPE, ConcreteDataType::string_datatype(), false),
            ColumnSchema::new(COLUMN_DEFAULT, ConcreteDataType::string_datatype(), true),
            ColumnSchema::new(IS_NULLABLE, ConcreteDataType::string_datatype(), false),
            ColumnSchema::new(COLUMN_TYPE, ConcreteDataType::string_datatype(), false),
            ColumnSchema::new(COLUMN_COMMENT, ConcreteDataType::string_datatype(), true),
+            ColumnSchema::new(SRS_ID, ConcreteDataType::int64_datatype(), true),
        ]))
    }

@@ -136,9 +197,18 @@ struct InformationSchemaColumnsBuilder {
    schema_names: StringVectorBuilder,
    table_names: StringVectorBuilder,
    column_names: StringVectorBuilder,
+    ordinal_positions: Int64VectorBuilder,
+    character_maximum_lengths: Int64VectorBuilder,
+    character_octet_lengths: Int64VectorBuilder,
+    numeric_precisions: Int64VectorBuilder,
+    numeric_scales: Int64VectorBuilder,
+    datetime_precisions: Int64VectorBuilder,
+    character_set_names: StringVectorBuilder,
+    collation_names: StringVectorBuilder,
+    column_keys: StringVectorBuilder,
+    greptime_data_types: StringVectorBuilder,
    data_types: StringVectorBuilder,
    semantic_types: StringVectorBuilder,
-
    column_defaults: StringVectorBuilder,
    is_nullables: StringVectorBuilder,
    column_types: StringVectorBuilder,
@@ -159,6 +229,16 @@ impl InformationSchemaColumnsBuilder {
            schema_names: StringVectorBuilder::with_capacity(INIT_CAPACITY),
            table_names: StringVectorBuilder::with_capacity(INIT_CAPACITY),
            column_names: StringVectorBuilder::with_capacity(INIT_CAPACITY),
+            ordinal_positions: Int64VectorBuilder::with_capacity(INIT_CAPACITY),
+            character_maximum_lengths: Int64VectorBuilder::with_capacity(INIT_CAPACITY),
+            character_octet_lengths: Int64VectorBuilder::with_capacity(INIT_CAPACITY),
+            numeric_precisions: Int64VectorBuilder::with_capacity(INIT_CAPACITY),
+            numeric_scales: Int64VectorBuilder::with_capacity(INIT_CAPACITY),
+            datetime_precisions: Int64VectorBuilder::with_capacity(INIT_CAPACITY),
+            character_set_names: StringVectorBuilder::with_capacity(INIT_CAPACITY),
+            collation_names: StringVectorBuilder::with_capacity(INIT_CAPACITY),
+            column_keys: StringVectorBuilder::with_capacity(INIT_CAPACITY),
+            greptime_data_types: StringVectorBuilder::with_capacity(INIT_CAPACITY),
            data_types: StringVectorBuilder::with_capacity(INIT_CAPACITY),
            semantic_types: StringVectorBuilder::with_capacity(INIT_CAPACITY),
            column_defaults: StringVectorBuilder::with_capacity(INIT_CAPACITY),
@@ -194,6 +274,7 @@ impl InformationSchemaColumnsBuilder {
                    };

                    self.add_column(
+                        idx,
                        &predicates,
                        &catalog_name,
                        &schema_name,
@@ -208,8 +289,10 @@ impl InformationSchemaColumnsBuilder {
        self.finish()
    }

+    #[allow(clippy::too_many_arguments)]
    fn add_column(
        &mut self,
+        index: usize,
        predicates: &Predicates,
        catalog_name: &str,
        schema_name: &str,
@@ -217,7 +300,16 @@ impl InformationSchemaColumnsBuilder {
        semantic_type: &str,
        column_schema: &ColumnSchema,
    ) {
-        let data_type = &column_schema.data_type.name();
+        // Use sql data type name
+        let data_type = statements::concrete_data_type_to_sql_data_type(&column_schema.data_type)
+            .map(|dt| dt.to_string().to_lowercase())
+            .unwrap_or_else(|_| column_schema.data_type.name());
+
+        let column_key = match semantic_type {
+            SEMANTIC_TYPE_PRIMARY_KEY => PRI_COLUMN_KEY,
+            SEMANTIC_TYPE_TIME_INDEX => TIME_INDEX_COLUMN_KEY,
+            _ => EMPTY_STR,
+        };

        let row = [
            (TABLE_CATALOG, &Value::from(catalog_name)),
@@ -226,6 +318,8 @@ impl InformationSchemaColumnsBuilder {
            (COLUMN_NAME, &Value::from(column_schema.name.as_str())),
            (DATA_TYPE, &Value::from(data_type.as_str())),
            (SEMANTIC_TYPE, &Value::from(semantic_type)),
+            (ORDINAL_POSITION, &Value::from((index + 1) as i64)),
+            (COLUMN_KEY, &Value::from(column_key)),
        ];

        if !predicates.eval(&row) {
@@ -236,7 +330,63 @@ impl InformationSchemaColumnsBuilder {
        self.schema_names.push(Some(schema_name));
        self.table_names.push(Some(table_name));
        self.column_names.push(Some(&column_schema.name));
-        self.data_types.push(Some(data_type));
+        // Starts from 1
+        self.ordinal_positions.push(Some((index + 1) as i64));
+
+        if column_schema.data_type.is_string() {
+            self.character_maximum_lengths.push(Some(MAX_STRING_LENGTH));
+            self.character_octet_lengths.push(Some(MAX_STRING_LENGTH));
+            self.numeric_precisions.push(None);
+            self.numeric_scales.push(None);
+            self.datetime_precisions.push(None);
+            self.character_set_names.push(Some(UTF8_CHARSET_NAME));
+            self.collation_names.push(Some(UTF8_COLLATE_NAME));
+        } else if column_schema.data_type.is_numeric() || column_schema.data_type.is_decimal() {
+            self.character_maximum_lengths.push(None);
+            self.character_octet_lengths.push(None);
+
+            self.numeric_precisions.push(
+                column_schema
+                    .data_type
+                    .numeric_precision()
+                    .map(|x| x as i64),
+            );
+            self.numeric_scales
+                .push(column_schema.data_type.numeric_scale().map(|x| x as i64));
+
+            self.datetime_precisions.push(None);
+            self.character_set_names.push(None);
+            self.collation_names.push(None);
+        } else {
+            self.character_maximum_lengths.push(None);
+            self.character_octet_lengths.push(None);
+            self.numeric_precisions.push(None);
+            self.numeric_scales.push(None);
+
+            match &column_schema.data_type {
+                ConcreteDataType::DateTime(datetime_type) => {
+                    self.datetime_precisions
+                        .push(Some(datetime_type.precision() as i64));
+                }
+                ConcreteDataType::Timestamp(ts_type) => {
+                    self.datetime_precisions
+                        .push(Some(ts_type.precision() as i64));
+                }
+                ConcreteDataType::Time(time_type) => {
+                    self.datetime_precisions
+                        .push(Some(time_type.precision() as i64));
+                }
+                _ => self.datetime_precisions.push(None),
+            }
+
+            self.character_set_names.push(None);
+            self.collation_names.push(None);
+        }
+
+        self.column_keys.push(Some(column_key));
+        self.greptime_data_types
+            .push(Some(&column_schema.data_type.name()));
+        self.data_types.push(Some(&data_type));
        self.semantic_types.push(Some(semantic_type));
        self.column_defaults.push(
            column_schema
@@ -249,23 +399,52 @@ impl InformationSchemaColumnsBuilder {
        } else {
            self.is_nullables.push(Some("No"));
        }
-        self.column_types.push(Some(data_type));
+        self.column_types.push(Some(&data_type));
        self.column_comments
            .push(column_schema.column_comment().map(|x| x.as_ref()));
    }

    fn finish(&mut self) -> Result<RecordBatch> {
+        let rows_num = self.collation_names.len();
+
+        let privileges = Arc::new(ConstantVector::new(
+            Arc::new(StringVector::from(vec![DEFAULT_PRIVILEGES])),
+            rows_num,
+        ));
+        let empty_string = Arc::new(ConstantVector::new(
+            Arc::new(StringVector::from(vec![EMPTY_STR])),
+            rows_num,
+        ));
+        let srs_ids = Arc::new(ConstantVector::new(
+            Arc::new(Int64Vector::from(vec![None])),
+            rows_num,
+        ));
+
        let columns: Vec<VectorRef> = vec![
            Arc::new(self.catalog_names.finish()),
            Arc::new(self.schema_names.finish()),
            Arc::new(self.table_names.finish()),
            Arc::new(self.column_names.finish()),
+            Arc::new(self.ordinal_positions.finish()),
+            Arc::new(self.character_maximum_lengths.finish()),
+            Arc::new(self.character_octet_lengths.finish()),
+            Arc::new(self.numeric_precisions.finish()),
+            Arc::new(self.numeric_scales.finish()),
+            Arc::new(self.datetime_precisions.finish()),
+            Arc::new(self.character_set_names.finish()),
+            Arc::new(self.collation_names.finish()),
+            Arc::new(self.column_keys.finish()),
+            empty_string.clone(),
+            privileges,
+            empty_string,
+            Arc::new(self.greptime_data_types.finish()),
            Arc::new(self.data_types.finish()),
            Arc::new(self.semantic_types.finish()),
            Arc::new(self.column_defaults.finish()),
            Arc::new(self.is_nullables.finish()),
            Arc::new(self.column_types.finish()),
            Arc::new(self.column_comments.finish()),
+            srs_ids,
        ];

        RecordBatch::new(self.schema.clone(), columns).context(CreateRecordBatchSnafu)
--- a/src/catalog/src/information_schema/key_column_usage.rs
+++ b/src/catalog/src/information_schema/key_column_usage.rs
@@ -37,13 +37,16 @@ use crate::error::{
 use crate::information_schema::{InformationTable, Predicates};
 use crate::CatalogManager;

-const CONSTRAINT_SCHEMA: &str = "constraint_schema";
-const CONSTRAINT_NAME: &str = "constraint_name";
-const TABLE_CATALOG: &str = "table_catalog";
-const TABLE_SCHEMA: &str = "table_schema";
-const TABLE_NAME: &str = "table_name";
-const COLUMN_NAME: &str = "column_name";
-const ORDINAL_POSITION: &str = "ordinal_position";
+pub const CONSTRAINT_SCHEMA: &str = "constraint_schema";
+pub const CONSTRAINT_NAME: &str = "constraint_name";
+// It's always `def` in MySQL
+pub const TABLE_CATALOG: &str = "table_catalog";
+// The real catalog name for this key column.
+pub const REAL_TABLE_CATALOG: &str = "real_table_catalog";
+pub const TABLE_SCHEMA: &str = "table_schema";
+pub const TABLE_NAME: &str = "table_name";
+pub const COLUMN_NAME: &str = "column_name";
+pub const ORDINAL_POSITION: &str = "ordinal_position";
 const INIT_CAPACITY: usize = 42;

 /// The virtual table implementation for `information_schema.KEY_COLUMN_USAGE`.
@@ -76,6 +79,11 @@ impl InformationSchemaKeyColumnUsage {
            ),
            ColumnSchema::new(CONSTRAINT_NAME, ConcreteDataType::string_datatype(), false),
            ColumnSchema::new(TABLE_CATALOG, ConcreteDataType::string_datatype(), false),
+            ColumnSchema::new(
+                REAL_TABLE_CATALOG,
+                ConcreteDataType::string_datatype(),
+                false,
+            ),
            ColumnSchema::new(TABLE_SCHEMA, ConcreteDataType::string_datatype(), false),
            ColumnSchema::new(TABLE_NAME, ConcreteDataType::string_datatype(), false),
            ColumnSchema::new(COLUMN_NAME, ConcreteDataType::string_datatype(), false),
@@ -158,6 +166,7 @@ struct InformationSchemaKeyColumnUsageBuilder {
    constraint_schema: StringVectorBuilder,
    constraint_name: StringVectorBuilder,
    table_catalog: StringVectorBuilder,
+    real_table_catalog: StringVectorBuilder,
    table_schema: StringVectorBuilder,
    table_name: StringVectorBuilder,
    column_name: StringVectorBuilder,
@@ -179,6 +188,7 @@ impl InformationSchemaKeyColumnUsageBuilder {
            constraint_schema: StringVectorBuilder::with_capacity(INIT_CAPACITY),
            constraint_name: StringVectorBuilder::with_capacity(INIT_CAPACITY),
            table_catalog: StringVectorBuilder::with_capacity(INIT_CAPACITY),
+            real_table_catalog: StringVectorBuilder::with_capacity(INIT_CAPACITY),
            table_schema: StringVectorBuilder::with_capacity(INIT_CAPACITY),
            table_name: StringVectorBuilder::with_capacity(INIT_CAPACITY),
            column_name: StringVectorBuilder::with_capacity(INIT_CAPACITY),
@@ -223,6 +233,7 @@ impl InformationSchemaKeyColumnUsageBuilder {
                                &predicates,
                                &schema_name,
                                "TIME INDEX",
+                                &catalog_name,
                                &schema_name,
                                &table_name,
                                &column.name,
@@ -231,6 +242,7 @@ impl InformationSchemaKeyColumnUsageBuilder {
                        }
                        if keys.contains(&idx) {
                            primary_constraints.push((
+                                catalog_name.clone(),
                                schema_name.clone(),
                                table_name.clone(),
                                column.name.clone(),
@@ -244,13 +256,14 @@ impl InformationSchemaKeyColumnUsageBuilder {
            }
        }

-        for (i, (schema_name, table_name, column_name)) in
+        for (i, (catalog_name, schema_name, table_name, column_name)) in
            primary_constraints.into_iter().enumerate()
        {
            self.add_key_column_usage(
                &predicates,
                &schema_name,
                "PRIMARY",
+                &catalog_name,
                &schema_name,
                &table_name,
                &column_name,
@@ -269,6 +282,7 @@ impl InformationSchemaKeyColumnUsageBuilder {
        predicates: &Predicates,
        constraint_schema: &str,
        constraint_name: &str,
+        table_catalog: &str,
        table_schema: &str,
        table_name: &str,
        column_name: &str,
@@ -277,6 +291,7 @@ impl InformationSchemaKeyColumnUsageBuilder {
        let row = [
            (CONSTRAINT_SCHEMA, &Value::from(constraint_schema)),
            (CONSTRAINT_NAME, &Value::from(constraint_name)),
+            (REAL_TABLE_CATALOG, &Value::from(table_catalog)),
            (TABLE_SCHEMA, &Value::from(table_schema)),
            (TABLE_NAME, &Value::from(table_name)),
            (COLUMN_NAME, &Value::from(column_name)),
@@ -291,6 +306,7 @@ impl InformationSchemaKeyColumnUsageBuilder {
        self.constraint_schema.push(Some(constraint_schema));
        self.constraint_name.push(Some(constraint_name));
        self.table_catalog.push(Some("def"));
+        self.real_table_catalog.push(Some(table_catalog));
        self.table_schema.push(Some(table_schema));
        self.table_name.push(Some(table_name));
        self.column_name.push(Some(column_name));
@@ -310,6 +326,7 @@ impl InformationSchemaKeyColumnUsageBuilder {
            Arc::new(self.constraint_schema.finish()),
            Arc::new(self.constraint_name.finish()),
            Arc::new(self.table_catalog.finish()),
+            Arc::new(self.real_table_catalog.finish()),
            Arc::new(self.table_schema.finish()),
            Arc::new(self.table_name.finish()),
            Arc::new(self.column_name.finish()),
--- a/src/catalog/src/information_schema/memory_table/tables.rs
+++ b/src/catalog/src/information_schema/memory_table/tables.rs
@@ -14,13 +14,15 @@

 use std::sync::Arc;

-use common_catalog::consts::MITO_ENGINE;
+use common_catalog::consts::{METRIC_ENGINE, MITO_ENGINE};
 use datatypes::prelude::{ConcreteDataType, VectorRef};
 use datatypes::schema::{ColumnSchema, Schema, SchemaRef};
 use datatypes::vectors::{Int64Vector, StringVector};

 use crate::information_schema::table_names::*;

+const NO_VALUE: &str = "NO";
+
 /// Find the schema and columns by the table_name, only valid for memory tables.
 /// Safety: the user MUST ensure the table schema exists, panic otherwise.
 pub fn get_schema_columns(table_name: &str) -> (SchemaRef, Vec<VectorRef>) {
@@ -59,14 +61,15 @@ pub fn get_schema_columns(table_name: &str) -> (SchemaRef, Vec<VectorRef>) {
                "SAVEPOINTS",
            ]),
            vec![
-                Arc::new(StringVector::from(vec![MITO_ENGINE])),
-                Arc::new(StringVector::from(vec!["DEFAULT"])),
+                Arc::new(StringVector::from(vec![MITO_ENGINE, METRIC_ENGINE])),
+                Arc::new(StringVector::from(vec!["DEFAULT", "YES"])),
                Arc::new(StringVector::from(vec![
                    "Storage engine for time-series data",
+                    "Storage engine for observability scenarios, which is adept at handling a large number of small tables, making it particularly suitable for cloud-native monitoring",
                ])),
-                Arc::new(StringVector::from(vec!["NO"])),
-                Arc::new(StringVector::from(vec!["NO"])),
-                Arc::new(StringVector::from(vec!["NO"])),
+                Arc::new(StringVector::from(vec![NO_VALUE, NO_VALUE])),
+                Arc::new(StringVector::from(vec![NO_VALUE, NO_VALUE])),
+                Arc::new(StringVector::from(vec![NO_VALUE, NO_VALUE])),
            ],
        ),

--- a/src/catalog/src/information_schema/schemata.rs
+++ b/src/catalog/src/information_schema/schemata.rs
@@ -37,8 +37,8 @@ use crate::error::{
 use crate::information_schema::{InformationTable, Predicates};
 use crate::CatalogManager;

-const CATALOG_NAME: &str = "catalog_name";
-const SCHEMA_NAME: &str = "schema_name";
+pub const CATALOG_NAME: &str = "catalog_name";
+pub const SCHEMA_NAME: &str = "schema_name";
 const DEFAULT_CHARACTER_SET_NAME: &str = "default_character_set_name";
 const DEFAULT_COLLATION_NAME: &str = "default_collation_name";
 const INIT_CAPACITY: usize = 42;
--- a/src/catalog/src/information_schema/tables.rs
+++ b/src/catalog/src/information_schema/tables.rs
@@ -39,10 +39,10 @@ use crate::error::{
 use crate::information_schema::{InformationTable, Predicates};
 use crate::CatalogManager;

-const TABLE_CATALOG: &str = "table_catalog";
-const TABLE_SCHEMA: &str = "table_schema";
-const TABLE_NAME: &str = "table_name";
-const TABLE_TYPE: &str = "table_type";
+pub const TABLE_CATALOG: &str = "table_catalog";
+pub const TABLE_SCHEMA: &str = "table_schema";
+pub const TABLE_NAME: &str = "table_name";
+pub const TABLE_TYPE: &str = "table_type";
 const TABLE_ID: &str = "table_id";
 const ENGINE: &str = "engine";
 const INIT_CAPACITY: usize = 42;
--- a/src/catalog/src/kvbackend/client.rs
+++ b/src/catalog/src/kvbackend/client.rs
@@ -364,6 +364,10 @@ impl KvBackend for MetaKvBackend {
        "MetaKvBackend"
    }

+    fn as_any(&self) -> &dyn Any {
+        self
+    }
+
    async fn range(&self, req: RangeRequest) -> Result<RangeResponse> {
        self.client
            .range(req)
@@ -372,27 +376,6 @@ impl KvBackend for MetaKvBackend {
            .context(ExternalSnafu)
    }

-    async fn get(&self, key: &[u8]) -> Result<Option<KeyValue>> {
-        let mut response = self
-            .client
-            .range(RangeRequest::new().with_key(key))
-            .await
-            .map_err(BoxedError::new)
-            .context(ExternalSnafu)?;
-        Ok(response.take_kvs().get_mut(0).map(|kv| KeyValue {
-            key: kv.take_key(),
-            value: kv.take_value(),
-        }))
-    }
-
-    async fn batch_put(&self, req: BatchPutRequest) -> Result<BatchPutResponse> {
-        self.client
-            .batch_put(req)
-            .await
-            .map_err(BoxedError::new)
-            .context(ExternalSnafu)
-    }
-
    async fn put(&self, req: PutRequest) -> Result<PutResponse> {
        self.client
            .put(req)
@@ -401,17 +384,9 @@ impl KvBackend for MetaKvBackend {
            .context(ExternalSnafu)
    }

-    async fn delete_range(&self, req: DeleteRangeRequest) -> Result<DeleteRangeResponse> {
+    async fn batch_put(&self, req: BatchPutRequest) -> Result<BatchPutResponse> {
        self.client
-            .delete_range(req)
-            .await
-            .map_err(BoxedError::new)
-            .context(ExternalSnafu)
-    }
-
-    async fn batch_delete(&self, req: BatchDeleteRequest) -> Result<BatchDeleteResponse> {
-        self.client
-            .batch_delete(req)
+            .batch_put(req)
            .await
            .map_err(BoxedError::new)
            .context(ExternalSnafu)
@@ -436,8 +411,33 @@ impl KvBackend for MetaKvBackend {
            .context(ExternalSnafu)
    }

-    fn as_any(&self) -> &dyn Any {
-        self
+    async fn delete_range(&self, req: DeleteRangeRequest) -> Result<DeleteRangeResponse> {
+        self.client
+            .delete_range(req)
+            .await
+            .map_err(BoxedError::new)
+            .context(ExternalSnafu)
+    }
+
+    async fn batch_delete(&self, req: BatchDeleteRequest) -> Result<BatchDeleteResponse> {
+        self.client
+            .batch_delete(req)
+            .await
+            .map_err(BoxedError::new)
+            .context(ExternalSnafu)
+    }
+
+    async fn get(&self, key: &[u8]) -> Result<Option<KeyValue>> {
+        let mut response = self
+            .client
+            .range(RangeRequest::new().with_key(key))
+            .await
+            .map_err(BoxedError::new)
+            .context(ExternalSnafu)?;
+        Ok(response.take_kvs().get_mut(0).map(|kv| KeyValue {
+            key: kv.take_key(),
+            value: kv.take_value(),
+        }))
    }
 }

--- a/src/catalog/src/kvbackend/manager.rs
+++ b/src/catalog/src/kvbackend/manager.rs
@@ -23,15 +23,14 @@ use common_catalog::consts::{
 };
 use common_catalog::format_full_table_name;
 use common_error::ext::BoxedError;
-use common_meta::cache_invalidator::{CacheInvalidator, CacheInvalidatorRef, Context};
-use common_meta::error::Result as MetaResult;
+use common_meta::cache_invalidator::{CacheInvalidator, Context, MultiCacheInvalidator};
+use common_meta::instruction::CacheIdent;
 use common_meta::key::catalog_name::CatalogNameKey;
 use common_meta::key::schema_name::SchemaNameKey;
 use common_meta::key::table_info::TableInfoValue;
 use common_meta::key::table_name::TableNameKey;
 use common_meta::key::{TableMetadataManager, TableMetadataManagerRef};
 use common_meta::kv_backend::KvBackendRef;
-use common_meta::table_name::TableName;
 use futures_util::stream::BoxStream;
 use futures_util::{StreamExt, TryStreamExt};
 use moka::future::{Cache as AsyncCache, CacheBuilder};
@@ -39,14 +38,13 @@ use moka::sync::Cache;
 use partition::manager::{PartitionRuleManager, PartitionRuleManagerRef};
 use snafu::prelude::*;
 use table::dist_table::DistTable;
-use table::metadata::TableId;
 use table::table::numbers::{NumbersTable, NUMBERS_TABLE_NAME};
 use table::TableRef;

 use crate::error::Error::{GetTableCache, TableCacheNotGet};
 use crate::error::{
-    self as catalog_err, ListCatalogsSnafu, ListSchemasSnafu, ListTablesSnafu,
-    Result as CatalogResult, TableCacheNotGetSnafu, TableMetadataManagerSnafu,
+    InvalidTableInfoInCatalogSnafu, ListCatalogsSnafu, ListSchemasSnafu, ListTablesSnafu, Result,
+    TableCacheNotGetSnafu, TableMetadataManagerSnafu,
 };
 use crate::information_schema::InformationSchemaProvider;
 use crate::CatalogManager;
@@ -58,10 +56,6 @@ use crate::CatalogManager;
 /// comes from `SystemCatalog`, which is static and read-only.
 #[derive(Clone)]
 pub struct KvBackendCatalogManager {
-    // TODO(LFC): Maybe use a real implementation for Standalone mode.
-    // Now we use `NoopKvCacheInvalidator` for Standalone mode. In Standalone mode, the KV backend
-    // is implemented by RaftEngine. Maybe we need a cache for it?
-    cache_invalidator: CacheInvalidatorRef,
    partition_manager: PartitionRuleManagerRef,
    table_metadata_manager: TableMetadataManagerRef,
    /// A sub-CatalogManager that handles system tables
@@ -69,33 +63,33 @@ pub struct KvBackendCatalogManager {
    table_cache: AsyncCache<String, TableRef>,
 }

-fn make_table(table_info_value: TableInfoValue) -> CatalogResult<TableRef> {
-    let table_info = table_info_value
-        .table_info
-        .try_into()
-        .context(catalog_err::InvalidTableInfoInCatalogSnafu)?;
-    Ok(DistTable::table(Arc::new(table_info)))
+struct TableCacheInvalidator {
+    table_cache: AsyncCache<String, TableRef>,
+}
+
+impl TableCacheInvalidator {
+    pub fn new(table_cache: AsyncCache<String, TableRef>) -> Self {
+        Self { table_cache }
+    }
 }

 #[async_trait::async_trait]
-impl CacheInvalidator for KvBackendCatalogManager {
-    async fn invalidate_table_id(&self, ctx: &Context, table_id: TableId) -> MetaResult<()> {
-        self.cache_invalidator
-            .invalidate_table_id(ctx, table_id)
-            .await
-    }
-
-    async fn invalidate_table_name(&self, ctx: &Context, table_name: TableName) -> MetaResult<()> {
-        let table_cache_key = format_full_table_name(
-            &table_name.catalog_name,
-            &table_name.schema_name,
-            &table_name.table_name,
-        );
-        self.cache_invalidator
-            .invalidate_table_name(ctx, table_name)
-            .await?;
-        self.table_cache.invalidate(&table_cache_key).await;
-
+impl CacheInvalidator for TableCacheInvalidator {
+    async fn invalidate(
+        &self,
+        _ctx: &Context,
+        caches: Vec<CacheIdent>,
+    ) -> common_meta::error::Result<()> {
+        for cache in caches {
+            if let CacheIdent::TableName(table_name) = cache {
+                let table_cache_key = format_full_table_name(
+                    &table_name.catalog_name,
+                    &table_name.schema_name,
+                    &table_name.table_name,
+                );
+                self.table_cache.invalidate(&table_cache_key).await;
+            }
+        }
        Ok(())
    }
 }
@@ -106,11 +100,21 @@ const TABLE_CACHE_TTL: Duration = Duration::from_secs(10 * 60);
 const TABLE_CACHE_TTI: Duration = Duration::from_secs(5 * 60);

 impl KvBackendCatalogManager {
-    pub fn new(backend: KvBackendRef, cache_invalidator: CacheInvalidatorRef) -> Arc<Self> {
+    pub async fn new(
+        backend: KvBackendRef,
+        multi_cache_invalidator: Arc<MultiCacheInvalidator>,
+    ) -> Arc<Self> {
+        let table_cache: AsyncCache<String, TableRef> = CacheBuilder::new(TABLE_CACHE_MAX_CAPACITY)
+            .time_to_live(TABLE_CACHE_TTL)
+            .time_to_idle(TABLE_CACHE_TTI)
+            .build();
+        multi_cache_invalidator
+            .add_invalidator(Arc::new(TableCacheInvalidator::new(table_cache.clone())))
+            .await;
+
        Arc::new_cyclic(|me| Self {
            partition_manager: Arc::new(PartitionRuleManager::new(backend.clone())),
            table_metadata_manager: Arc::new(TableMetadataManager::new(backend)),
-            cache_invalidator,
            system_catalog: SystemCatalog {
                catalog_manager: me.clone(),
                catalog_cache: Cache::new(CATALOG_CACHE_MAX_CAPACITY),
@@ -119,10 +123,7 @@ impl KvBackendCatalogManager {
                    me.clone(),
                )),
            },
-            table_cache: CacheBuilder::new(TABLE_CACHE_MAX_CAPACITY)
-                .time_to_live(TABLE_CACHE_TTL)
-                .time_to_idle(TABLE_CACHE_TTI)
-                .build(),
+            table_cache,
        })
    }

@@ -141,12 +142,11 @@ impl CatalogManager for KvBackendCatalogManager {
        self
    }

-    async fn catalog_names(&self) -> CatalogResult<Vec<String>> {
+    async fn catalog_names(&self) -> Result<Vec<String>> {
        let stream = self
            .table_metadata_manager
            .catalog_manager()
-            .catalog_names()
-            .await;
+            .catalog_names();

        let keys = stream
            .try_collect::<Vec<_>>()
@@ -157,12 +157,11 @@ impl CatalogManager for KvBackendCatalogManager {
        Ok(keys)
    }

-    async fn schema_names(&self, catalog: &str) -> CatalogResult<Vec<String>> {
+    async fn schema_names(&self, catalog: &str) -> Result<Vec<String>> {
        let stream = self
            .table_metadata_manager
            .schema_manager()
-            .schema_names(catalog)
-            .await;
+            .schema_names(catalog);
        let mut keys = stream
            .try_collect::<BTreeSet<_>>()
            .await
@@ -174,12 +173,11 @@ impl CatalogManager for KvBackendCatalogManager {
        Ok(keys.into_iter().collect())
    }

-    async fn table_names(&self, catalog: &str, schema: &str) -> CatalogResult<Vec<String>> {
+    async fn table_names(&self, catalog: &str, schema: &str) -> Result<Vec<String>> {
        let stream = self
            .table_metadata_manager
            .table_name_manager()
-            .tables(catalog, schema)
-            .await;
+            .tables(catalog, schema);
        let mut tables = stream
            .try_collect::<Vec<_>>()
            .await
@@ -193,7 +191,7 @@ impl CatalogManager for KvBackendCatalogManager {
        Ok(tables.into_iter().collect())
    }

-    async fn catalog_exists(&self, catalog: &str) -> CatalogResult<bool> {
+    async fn catalog_exists(&self, catalog: &str) -> Result<bool> {
        self.table_metadata_manager
            .catalog_manager()
            .exists(CatalogNameKey::new(catalog))
@@ -201,7 +199,7 @@ impl CatalogManager for KvBackendCatalogManager {
            .context(TableMetadataManagerSnafu)
    }

-    async fn schema_exists(&self, catalog: &str, schema: &str) -> CatalogResult<bool> {
+    async fn schema_exists(&self, catalog: &str, schema: &str) -> Result<bool> {
        if self.system_catalog.schema_exist(schema) {
            return Ok(true);
        }
@@ -213,7 +211,7 @@ impl CatalogManager for KvBackendCatalogManager {
            .context(TableMetadataManagerSnafu)
    }

-    async fn table_exists(&self, catalog: &str, schema: &str, table: &str) -> CatalogResult<bool> {
+    async fn table_exists(&self, catalog: &str, schema: &str, table: &str) -> Result<bool> {
        if self.system_catalog.table_exist(schema, table) {
            return Ok(true);
        }
@@ -232,7 +230,7 @@ impl CatalogManager for KvBackendCatalogManager {
        catalog: &str,
        schema: &str,
        table_name: &str,
-    ) -> CatalogResult<Option<TableRef>> {
+    ) -> Result<Option<TableRef>> {
        if let Some(table) = self.system_catalog.table(catalog, schema, table_name) {
            return Ok(Some(table));
        }
@@ -266,7 +264,7 @@ impl CatalogManager for KvBackendCatalogManager {
                }
                .fail();
            };
-            make_table(table_info_value)
+            build_table(table_info_value)
        };

        match self
@@ -289,7 +287,7 @@ impl CatalogManager for KvBackendCatalogManager {
        &'a self,
        catalog: &'a str,
        schema: &'a str,
-    ) -> BoxStream<'a, CatalogResult<TableRef>> {
+    ) -> BoxStream<'a, Result<TableRef>> {
        let sys_tables = try_stream!({
            // System tables
            let sys_table_names = self.system_catalog.table_names(schema);
@@ -304,7 +302,6 @@ impl CatalogManager for KvBackendCatalogManager {
            .table_metadata_manager
            .table_name_manager()
            .tables(catalog, schema)
-            .await
            .map_ok(|(_, v)| v.table_id());
        const BATCH_SIZE: usize = 128;
        let user_tables = try_stream!({
@@ -314,7 +311,7 @@ impl CatalogManager for KvBackendCatalogManager {
            while let Some(table_ids) = table_id_chunks.next().await {
                let table_ids = table_ids
                    .into_iter()
-                    .collect::<Result<Vec<_>, _>>()
+                    .collect::<std::result::Result<Vec<_>, _>>()
                    .map_err(BoxedError::new)
                    .context(ListTablesSnafu { catalog, schema })?;

@@ -326,7 +323,7 @@ impl CatalogManager for KvBackendCatalogManager {
                    .context(TableMetadataManagerSnafu)?;

                for table_info_value in table_info_values.into_values() {
-                    yield make_table(table_info_value)?;
+                    yield build_table(table_info_value)?;
                }
            }
        });
@@ -335,6 +332,14 @@ impl CatalogManager for KvBackendCatalogManager {
    }
 }

+fn build_table(table_info_value: TableInfoValue) -> Result<TableRef> {
+    let table_info = table_info_value
+        .table_info
+        .try_into()
+        .context(InvalidTableInfoInCatalogSnafu)?;
+    Ok(DistTable::table(Arc::new(table_info)))
+}
+
 // TODO: This struct can hold a static map of all system tables when
 // the upper layer (e.g., procedure) can inform the catalog manager
 // a new catalog is created.
--- a/src/catalog/src/lib.rs
+++ b/src/catalog/src/lib.rs
@@ -19,10 +19,10 @@ use std::any::Any;
 use std::fmt::{Debug, Formatter};
 use std::sync::Arc;

+use api::v1::CreateTableExpr;
 use futures::future::BoxFuture;
 use futures_util::stream::BoxStream;
 use table::metadata::TableId;
-use table::requests::CreateTableRequest;
 use table::TableRef;

 use crate::error::Result;
@@ -75,9 +75,9 @@ pub type OpenSystemTableHook =
 /// Register system table request:
 /// - When system table is already created and registered, the hook will be called
 ///     with table ref after opening the system table
-/// - When system table is not exists, create and register the table by create_table_request and calls open_hook with the created table.
+/// - When system table is not exists, create and register the table by `create_table_expr` and calls `open_hook` with the created table.
 pub struct RegisterSystemTableRequest {
-    pub create_table_request: CreateTableRequest,
+    pub create_table_expr: CreateTableExpr,
    pub open_hook: Option<OpenSystemTableHook>,
 }

--- a/src/client/Cargo.toml
+++ b/src/client/Cargo.toml
@@ -16,7 +16,6 @@ arc-swap = "1.6"
 arrow-flight.workspace = true
 async-stream.workspace = true
 async-trait.workspace = true
-common-base.workspace = true
 common-catalog.workspace = true
 common-error.workspace = true
 common-grpc.workspace = true
@@ -25,10 +24,6 @@ common-meta.workspace = true
 common-query.workspace = true
 common-recordbatch.workspace = true
 common-telemetry.workspace = true
-common-time.workspace = true
-datafusion.workspace = true
-datatypes.workspace = true
-derive_builder.workspace = true
 enum_dispatch = "0.3"
 futures-util.workspace = true
 lazy_static.workspace = true
@@ -37,9 +32,7 @@ parking_lot = "0.12"
 prometheus.workspace = true
 prost.workspace = true
 rand.workspace = true
-serde.workspace = true
 serde_json.workspace = true
-session.workspace = true
 snafu.workspace = true
 tokio.workspace = true
 tokio-stream = { workspace = true, features = ["net"] }
--- a/src/client/src/database.rs
+++ b/src/client/src/database.rs
@@ -307,7 +307,7 @@ impl Database {
                        reason: "Expect 'AffectedRows' Flight messages to be the one and the only!"
                    }
                );
-                Ok(Output::AffectedRows(rows))
+                Ok(Output::new_with_affected_rows(rows))
            }
            FlightMessage::Recordbatch(_) | FlightMessage::Metrics(_) => {
                IllegalFlightMessagesSnafu {
@@ -340,7 +340,7 @@ impl Database {
                    output_ordering: None,
                    metrics: Default::default(),
                };
-                Ok(Output::new_stream(Box::pin(record_batch_stream)))
+                Ok(Output::new_with_stream(Box::pin(record_batch_stream)))
            }
        }
    }
--- a/src/client/src/lib.rs
+++ b/src/client/src/lib.rs
@@ -26,7 +26,7 @@ use api::v1::greptime_response::Response;
 use api::v1::{AffectedRows, GreptimeResponse};
 pub use common_catalog::consts::{DEFAULT_CATALOG_NAME, DEFAULT_SCHEMA_NAME};
 use common_error::status_code::StatusCode;
-pub use common_query::Output;
+pub use common_query::{Output, OutputData, OutputMeta};
 pub use common_recordbatch::{RecordBatches, SendableRecordBatchStream};
 use snafu::OptionExt;

--- a/src/client/src/region.rs
+++ b/src/client/src/region.rs
@@ -14,7 +14,7 @@

 use std::sync::Arc;

-use api::v1::region::{QueryRequest, RegionRequest, RegionResponse};
+use api::v1::region::{QueryRequest, RegionRequest};
 use api::v1::ResponseHeader;
 use arc_swap::ArcSwapOption;
 use arrow_flight::Ticket;
@@ -23,7 +23,7 @@ use async_trait::async_trait;
 use common_error::ext::{BoxedError, ErrorExt};
 use common_error::status_code::StatusCode;
 use common_grpc::flight::{FlightDecoder, FlightMessage};
-use common_meta::datanode_manager::{AffectedRows, Datanode};
+use common_meta::datanode_manager::{Datanode, HandleResponse};
 use common_meta::error::{self as meta_error, Result as MetaResult};
 use common_recordbatch::error::ExternalSnafu;
 use common_recordbatch::{RecordBatchStreamWrapper, SendableRecordBatchStream};
@@ -46,7 +46,7 @@ pub struct RegionRequester {

 #[async_trait]
 impl Datanode for RegionRequester {
-    async fn handle(&self, request: RegionRequest) -> MetaResult<AffectedRows> {
+    async fn handle(&self, request: RegionRequest) -> MetaResult<HandleResponse> {
        self.handle_inner(request).await.map_err(|err| {
            if err.should_retry() {
                meta_error::Error::RetryLater {
@@ -165,7 +165,7 @@ impl RegionRequester {
        Ok(Box::pin(record_batch_stream))
    }

-    async fn handle_inner(&self, request: RegionRequest) -> Result<AffectedRows> {
+    async fn handle_inner(&self, request: RegionRequest) -> Result<HandleResponse> {
        let request_type = request
            .body
            .as_ref()
@@ -178,10 +178,7 @@ impl RegionRequester {

        let mut client = self.client.raw_region_client()?;

-        let RegionResponse {
-            header,
-            affected_rows,
-        } = client
+        let response = client
            .handle(request)
            .await
            .map_err(|e| {
@@ -195,19 +192,20 @@ impl RegionRequester {
            })?
            .into_inner();

-        check_response_header(header)?;
+        check_response_header(&response.header)?;

-        Ok(affected_rows as _)
+        Ok(HandleResponse::from_region_response(response))
    }

-    pub async fn handle(&self, request: RegionRequest) -> Result<AffectedRows> {
+    pub async fn handle(&self, request: RegionRequest) -> Result<HandleResponse> {
        self.handle_inner(request).await
    }
 }

-pub fn check_response_header(header: Option<ResponseHeader>) -> Result<()> {
+pub fn check_response_header(header: &Option<ResponseHeader>) -> Result<()> {
    let status = header
-        .and_then(|header| header.status)
+        .as_ref()
+        .and_then(|header| header.status.as_ref())
        .context(IllegalDatabaseResponseSnafu {
            err_msg: "either response header or status is missing",
        })?;
@@ -221,7 +219,7 @@ pub fn check_response_header(header: Option<ResponseHeader>) -> Result<()> {
            })?;
        ServerSnafu {
            code,
-            msg: status.err_msg,
+            msg: status.err_msg.clone(),
        }
        .fail()
    }
@@ -236,19 +234,19 @@ mod test {

    #[test]
    fn test_check_response_header() {
-        let result = check_response_header(None);
+        let result = check_response_header(&None);
        assert!(matches!(
            result.unwrap_err(),
            IllegalDatabaseResponse { .. }
        ));

-        let result = check_response_header(Some(ResponseHeader { status: None }));
+        let result = check_response_header(&Some(ResponseHeader { status: None }));
        assert!(matches!(
            result.unwrap_err(),
            IllegalDatabaseResponse { .. }
        ));

-        let result = check_response_header(Some(ResponseHeader {
+        let result = check_response_header(&Some(ResponseHeader {
            status: Some(PbStatus {
                status_code: StatusCode::Success as u32,
                err_msg: String::default(),
@@ -256,7 +254,7 @@ mod test {
        }));
        assert!(result.is_ok());

-        let result = check_response_header(Some(ResponseHeader {
+        let result = check_response_header(&Some(ResponseHeader {
            status: Some(PbStatus {
                status_code: u32::MAX,
                err_msg: String::default(),
@@ -267,7 +265,7 @@ mod test {
            IllegalDatabaseResponse { .. }
        ));

-        let result = check_response_header(Some(ResponseHeader {
+        let result = check_response_header(&Some(ResponseHeader {
            status: Some(PbStatus {
                status_code: StatusCode::Internal as u32,
                err_msg: "blabla".to_string(),
--- a/src/cmd/Cargo.toml
+++ b/src/cmd/Cargo.toml
@@ -16,7 +16,6 @@ tokio-console = ["common-telemetry/tokio-console"]
 workspace = true

 [dependencies]
-anymap = "1.0.0-beta.2"
 async-trait.workspace = true
 auth.workspace = true
 catalog.workspace = true
@@ -52,7 +51,6 @@ meta-client.workspace = true
 meta-srv.workspace = true
 mito2.workspace = true
 nu-ansi-term = "0.46"
-partition.workspace = true
 plugins.workspace = true
 prometheus.workspace = true
 prost.workspace = true
--- a/src/cmd/build.rs
+++ b/src/cmd/build.rs
@@ -13,5 +13,8 @@
 // limitations under the License.

 fn main() {
+    // Trigger this script if the git branch/commit changes
+    println!("cargo:rerun-if-changed=.git/refs/heads");
+
    common_version::setup_build_info();
 }
--- a/src/cmd/src/cli/bench.rs
+++ b/src/cmd/src/cli/bench.rs
@@ -62,7 +62,9 @@ pub struct BenchTableMetadataCommand {

 impl BenchTableMetadataCommand {
    pub async fn build(&self) -> Result<Instance> {
-        let etcd_store = EtcdStore::with_endpoints([&self.etcd_addr]).await.unwrap();
+        let etcd_store = EtcdStore::with_endpoints([&self.etcd_addr], 128)
+            .await
+            .unwrap();

        let table_metadata_manager = Arc::new(TableMetadataManager::new(etcd_store));

--- a/src/cmd/src/cli/bench/metadata.rs
+++ b/src/cmd/src/cli/bench/metadata.rs
@@ -106,9 +106,15 @@ impl TableMetadataBencher {
                    .await
                    .unwrap();
                let start = Instant::now();
+                let table_info = table_info.unwrap();
+                let table_id = table_info.table_info.ident.table_id;
                let _ = self
                    .table_metadata_manager
-                    .delete_table_metadata(&table_info.unwrap(), &table_route.unwrap())
+                    .delete_table_metadata(
+                        table_id,
+                        &table_info.table_name(),
+                        table_route.unwrap().region_routes().unwrap(),
+                    )
                    .await;
                start.elapsed()
            },
--- a/src/cmd/src/cli/export.rs
+++ b/src/cmd/src/cli/export.rs
@@ -19,8 +19,7 @@ use async_trait::async_trait;
 use clap::{Parser, ValueEnum};
 use client::api::v1::auth_header::AuthScheme;
 use client::api::v1::Basic;
-use client::{Client, Database, DEFAULT_SCHEMA_NAME};
-use common_query::Output;
+use client::{Client, Database, OutputData, DEFAULT_SCHEMA_NAME};
 use common_recordbatch::util::collect;
 use common_telemetry::{debug, error, info, warn};
 use datatypes::scalars::ScalarVector;
@@ -142,7 +141,7 @@ impl Export {
                    .with_context(|_| RequestDatabaseSnafu {
                        sql: "show databases".to_string(),
                    })?;
-            let Output::Stream(stream, _) = result else {
+            let OutputData::Stream(stream) = result.data else {
                NotDataFromOutputSnafu.fail()?
            };
            let record_batch = collect(stream)
@@ -183,7 +182,7 @@ impl Export {
            .sql(&sql)
            .await
            .with_context(|_| RequestDatabaseSnafu { sql })?;
-        let Output::Stream(stream, _) = result else {
+        let OutputData::Stream(stream) = result.data else {
            NotDataFromOutputSnafu.fail()?
        };
        let Some(record_batch) = collect(stream)
@@ -235,7 +234,7 @@ impl Export {
            .sql(&sql)
            .await
            .with_context(|_| RequestDatabaseSnafu { sql })?;
-        let Output::Stream(stream, _) = result else {
+        let OutputData::Stream(stream) = result.data else {
            NotDataFromOutputSnafu.fail()?
        };
        let record_batch = collect(stream)
--- a/src/cmd/src/cli/repl.rs
+++ b/src/cmd/src/cli/repl.rs
@@ -19,9 +19,10 @@ use std::time::Instant;
 use catalog::kvbackend::{
    CachedMetaKvBackend, CachedMetaKvBackendBuilder, KvBackendCatalogManager,
 };
-use client::{Client, Database, DEFAULT_CATALOG_NAME, DEFAULT_SCHEMA_NAME};
+use client::{Client, Database, OutputData, DEFAULT_CATALOG_NAME, DEFAULT_SCHEMA_NAME};
 use common_base::Plugins;
 use common_error::ext::ErrorExt;
+use common_meta::cache_invalidator::MultiCacheInvalidator;
 use common_query::Output;
 use common_recordbatch::RecordBatches;
 use common_telemetry::logging;
@@ -184,15 +185,15 @@ impl Repl {
        }
        .context(RequestDatabaseSnafu { sql: &sql })?;

-        let either = match output {
-            Output::Stream(s, _) => {
+        let either = match output.data {
+            OutputData::Stream(s) => {
                let x = RecordBatches::try_collect(s)
                    .await
                    .context(CollectRecordBatchesSnafu)?;
                Either::Left(x)
            }
-            Output::RecordBatches(x) => Either::Left(x),
-            Output::AffectedRows(rows) => Either::Right(rows),
+            OutputData::RecordBatches(x) => Either::Left(x),
+            OutputData::AffectedRows(rows) => Either::Right(rows),
        };

        let end = Instant::now();
@@ -252,9 +253,11 @@ async fn create_query_engine(meta_addr: &str) -> Result<DatafusionQueryEngine> {

    let cached_meta_backend =
        Arc::new(CachedMetaKvBackendBuilder::new(meta_client.clone()).build());
-
+    let multi_cache_invalidator = Arc::new(MultiCacheInvalidator::with_invalidators(vec![
+        cached_meta_backend.clone(),
+    ]));
    let catalog_list =
-        KvBackendCatalogManager::new(cached_meta_backend.clone(), cached_meta_backend);
+        KvBackendCatalogManager::new(cached_meta_backend.clone(), multi_cache_invalidator).await;
    let plugins: Plugins = Default::default();
    let state = Arc::new(QueryEngineState::new(
        catalog_list,
--- a/src/cmd/src/cli/upgrade.rs
+++ b/src/cmd/src/cli/upgrade.rs
@@ -70,7 +70,7 @@ impl UpgradeCommand {
                etcd_addr: &self.etcd_addr,
            })?;
        let tool = MigrateTableMetadata {
-            etcd_store: EtcdStore::with_etcd_client(client),
+            etcd_store: EtcdStore::with_etcd_client(client, 128),
            dryrun: self.dryrun,
            skip_catalog_keys: self.skip_catalog_keys,
            skip_table_global_keys: self.skip_table_global_keys,
--- a/src/cmd/src/frontend.rs
+++ b/src/cmd/src/frontend.rs
@@ -16,9 +16,10 @@ use std::sync::Arc;
 use std::time::Duration;

 use async_trait::async_trait;
-use catalog::kvbackend::CachedMetaKvBackendBuilder;
+use catalog::kvbackend::{CachedMetaKvBackendBuilder, KvBackendCatalogManager};
 use clap::Parser;
 use client::client_manager::DatanodeClients;
+use common_meta::cache_invalidator::MultiCacheInvalidator;
 use common_meta::heartbeat::handler::parse_mailbox_message::ParseMailboxMessageHandler;
 use common_meta::heartbeat::handler::HandlerGroupExecutor;
 use common_telemetry::logging;
@@ -247,11 +248,19 @@ impl StartCommand {
            .cache_tti(cache_tti)
            .build();
        let cached_meta_backend = Arc::new(cached_meta_backend);
+        let multi_cache_invalidator = Arc::new(MultiCacheInvalidator::with_invalidators(vec![
+            cached_meta_backend.clone(),
+        ]));
+        let catalog_manager = KvBackendCatalogManager::new(
+            cached_meta_backend.clone(),
+            multi_cache_invalidator.clone(),
+        )
+        .await;

        let executor = HandlerGroupExecutor::new(vec![
            Arc::new(ParseMailboxMessageHandler),
            Arc::new(InvalidateTableCacheHandler::new(
-                cached_meta_backend.clone(),
+                multi_cache_invalidator.clone(),
            )),
        ]);

@@ -263,11 +272,12 @@ impl StartCommand {

        let mut instance = FrontendBuilder::new(
            cached_meta_backend.clone(),
+            catalog_manager,
            Arc::new(DatanodeClients::default()),
            meta_client,
        )
-        .with_cache_invalidator(cached_meta_backend)
        .with_plugin(plugins.clone())
+        .with_cache_invalidator(multi_cache_invalidator)
        .with_heartbeat_task(heartbeat_task)
        .try_build()
        .await
--- a/src/cmd/src/metasrv.rs
+++ b/src/cmd/src/metasrv.rs
@@ -117,10 +117,12 @@ struct StartCommand {
    /// The working home directory of this metasrv instance.
    #[clap(long)]
    data_home: Option<String>,
-
    /// If it's not empty, the metasrv will store all data with this key prefix.
    #[clap(long, default_value = "")]
    store_key_prefix: String,
+    /// The max operations per txn
+    #[clap(long)]
+    max_txn_ops: Option<usize>,
 }

 impl StartCommand {
@@ -181,6 +183,10 @@ impl StartCommand {
            opts.store_key_prefix = self.store_key_prefix.clone()
        }

+        if let Some(max_txn_ops) = self.max_txn_ops {
+            opts.max_txn_ops = max_txn_ops;
+        }
+
        // Disable dashboard in metasrv.
        opts.http.disable_dashboard = true;

@@ -212,6 +218,7 @@ impl StartCommand {
 mod tests {
    use std::io::Write;

+    use common_base::readable_size::ReadableSize;
    use common_test_util::temp_dir::create_named_temp_file;
    use meta_srv::selector::SelectorType;

@@ -291,6 +298,10 @@ mod tests {
                .first_heartbeat_estimate
                .as_millis()
        );
+        assert_eq!(
+            options.procedure.max_metadata_value_size,
+            Some(ReadableSize::kb(1500))
+        );
    }

    #[test]
--- a/src/cmd/src/standalone.rs
+++ b/src/cmd/src/standalone.rs
@@ -16,10 +16,11 @@ use std::sync::Arc;
 use std::{fs, path};

 use async_trait::async_trait;
+use catalog::kvbackend::KvBackendCatalogManager;
 use clap::Parser;
 use common_catalog::consts::MIN_USER_TABLE_ID;
 use common_config::{metadata_store_dir, KvBackendConfig};
-use common_meta::cache_invalidator::DummyCacheInvalidator;
+use common_meta::cache_invalidator::{CacheInvalidatorRef, MultiCacheInvalidator};
 use common_meta::datanode_manager::DatanodeManagerRef;
 use common_meta::ddl::table_meta::{TableMetadataAllocator, TableMetadataAllocatorRef};
 use common_meta::ddl::ProcedureExecutorRef;
@@ -399,6 +400,10 @@ impl StartCommand {
        .await
        .context(StartFrontendSnafu)?;

+        let multi_cache_invalidator = Arc::new(MultiCacheInvalidator::default());
+        let catalog_manager =
+            KvBackendCatalogManager::new(kv_backend.clone(), multi_cache_invalidator.clone()).await;
+
        let builder =
            DatanodeBuilder::new(dn_opts, fe_plugins.clone()).with_kv_backend(kv_backend.clone());
        let datanode = builder.build().await.context(StartDatanodeSnafu)?;
@@ -422,22 +427,27 @@ impl StartCommand {
        let table_meta_allocator = Arc::new(TableMetadataAllocator::new(
            table_id_sequence,
            wal_options_allocator.clone(),
-            table_metadata_manager.table_name_manager().clone(),
        ));

        let ddl_task_executor = Self::create_ddl_task_executor(
            table_metadata_manager,
            procedure_manager.clone(),
            datanode_manager.clone(),
+            multi_cache_invalidator,
            table_meta_allocator,
        )
        .await?;

-        let mut frontend = FrontendBuilder::new(kv_backend, datanode_manager, ddl_task_executor)
-            .with_plugin(fe_plugins.clone())
-            .try_build()
-            .await
-            .context(StartFrontendSnafu)?;
+        let mut frontend = FrontendBuilder::new(
+            kv_backend,
+            catalog_manager,
+            datanode_manager,
+            ddl_task_executor,
+        )
+        .with_plugin(fe_plugins.clone())
+        .try_build()
+        .await
+        .context(StartFrontendSnafu)?;

        let servers = Services::new(fe_opts.clone(), Arc::new(frontend.clone()), fe_plugins)
            .build()
@@ -459,16 +469,18 @@ impl StartCommand {
        table_metadata_manager: TableMetadataManagerRef,
        procedure_manager: ProcedureManagerRef,
        datanode_manager: DatanodeManagerRef,
+        cache_invalidator: CacheInvalidatorRef,
        table_meta_allocator: TableMetadataAllocatorRef,
    ) -> Result<ProcedureExecutorRef> {
        let procedure_executor: ProcedureExecutorRef = Arc::new(
            DdlManager::try_new(
                procedure_manager,
                datanode_manager,
-                Arc::new(DummyCacheInvalidator),
+                cache_invalidator,
                table_metadata_manager,
                table_meta_allocator,
                Arc::new(MemoryRegionKeeper::default()),
+                true,
            )
            .context(InitDdlManagerSnafu)?,
        );
--- a/src/common/base/src/readable_size.rs
+++ b/src/common/base/src/readable_size.rs
@@ -1,20 +1,6 @@
 // Copyright (c) 2017-present, PingCAP, Inc. Licensed under Apache-2.0.

-// Copyright 2023 Greptime Team
-//
-// Licensed under the Apache License, Version 2.0 (the "License");
-// you may not use this file except in compliance with the License.
-// You may obtain a copy of the License at
-//
-//     http://www.apache.org/licenses/LICENSE-2.0
-//
-// Unless required by applicable law or agreed to in writing, software
-// distributed under the License is distributed on an "AS IS" BASIS,
-// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
-// See the License for the specific language governing permissions and
-// limitations under the License.
-
-// This file is copied from https://github.com/tikv/raft-engine/blob/8dd2a39f359ff16f5295f35343f626e0c10132fa/src/util.rs
+// This file is copied from https://github.com/tikv/raft-engine/blob/0.3.0/src/util.rs

 use std::fmt::{self, Debug, Display, Write};
 use std::ops::{Div, Mul};
--- a/src/common/catalog/src/lib.rs
+++ b/src/common/catalog/src/lib.rs
@@ -55,10 +55,10 @@ pub fn build_db_string(catalog: &str, schema: &str) -> String {
 /// schema name
 /// - if `[<catalog>-]` is provided, we split database name with `-` and use
 /// `<catalog>` and `<schema>`.
-pub fn parse_catalog_and_schema_from_db_string(db: &str) -> (&str, &str) {
+pub fn parse_catalog_and_schema_from_db_string(db: &str) -> (String, String) {
    match parse_optional_catalog_and_schema_from_db_string(db) {
        (Some(catalog), schema) => (catalog, schema),
-        (None, schema) => (DEFAULT_CATALOG_NAME, schema),
+        (None, schema) => (DEFAULT_CATALOG_NAME.to_string(), schema),
    }
 }

@@ -66,12 +66,12 @@ pub fn parse_catalog_and_schema_from_db_string(db: &str) -> (&str, &str) {
 ///
 /// Similar to [`parse_catalog_and_schema_from_db_string`] but returns an optional
 /// catalog if it's not provided in the database name.
-pub fn parse_optional_catalog_and_schema_from_db_string(db: &str) -> (Option<&str>, &str) {
+pub fn parse_optional_catalog_and_schema_from_db_string(db: &str) -> (Option<String>, String) {
    let parts = db.splitn(2, '-').collect::<Vec<&str>>();
    if parts.len() == 2 {
-        (Some(parts[0]), parts[1])
+        (Some(parts[0].to_lowercase()), parts[1].to_lowercase())
    } else {
-        (None, db)
+        (None, db.to_lowercase())
    }
 }

@@ -88,32 +88,37 @@ mod tests {
    #[test]
    fn test_parse_catalog_and_schema() {
        assert_eq!(
-            (DEFAULT_CATALOG_NAME, "fullschema"),
+            (DEFAULT_CATALOG_NAME.to_string(), "fullschema".to_string()),
            parse_catalog_and_schema_from_db_string("fullschema")
        );

        assert_eq!(
-            ("catalog", "schema"),
+            ("catalog".to_string(), "schema".to_string()),
            parse_catalog_and_schema_from_db_string("catalog-schema")
        );

        assert_eq!(
-            ("catalog", "schema1-schema2"),
+            ("catalog".to_string(), "schema1-schema2".to_string()),
            parse_catalog_and_schema_from_db_string("catalog-schema1-schema2")
        );

        assert_eq!(
-            (None, "fullschema"),
+            (None, "fullschema".to_string()),
            parse_optional_catalog_and_schema_from_db_string("fullschema")
        );

        assert_eq!(
-            (Some("catalog"), "schema"),
+            (Some("catalog".to_string()), "schema".to_string()),
            parse_optional_catalog_and_schema_from_db_string("catalog-schema")
        );

        assert_eq!(
-            (Some("catalog"), "schema1-schema2"),
+            (Some("catalog".to_string()), "schema".to_string()),
+            parse_optional_catalog_and_schema_from_db_string("CATALOG-SCHEMA")
+        );
+
+        assert_eq!(
+            (Some("catalog".to_string()), "schema1-schema2".to_string()),
            parse_optional_catalog_and_schema_from_db_string("catalog-schema1-schema2")
        );
    }
--- a/src/common/config/Cargo.toml
+++ b/src/common/config/Cargo.toml
@@ -9,7 +9,6 @@ workspace = true

 [dependencies]
 common-base.workspace = true
-humantime-serde.workspace = true
 num_cpus.workspace = true
 serde.workspace = true
 sysinfo.workspace = true
--- a/src/common/datasource/src/object_store/s3.rs
+++ b/src/common/datasource/src/object_store/s3.rs
@@ -28,12 +28,15 @@ const REGION: &str = "region";
 const ENABLE_VIRTUAL_HOST_STYLE: &str = "enable_virtual_host_style";

 pub fn is_supported_in_s3(key: &str) -> bool {
-    key == ENDPOINT
-        || key == ACCESS_KEY_ID
-        || key == SECRET_ACCESS_KEY
-        || key == SESSION_TOKEN
-        || key == REGION
-        || key == ENABLE_VIRTUAL_HOST_STYLE
+    [
+        ENDPOINT,
+        ACCESS_KEY_ID,
+        SECRET_ACCESS_KEY,
+        SESSION_TOKEN,
+        REGION,
+        ENABLE_VIRTUAL_HOST_STYLE,
+    ]
+    .contains(&key)
 }

 pub fn build_s3_backend(
--- a/src/common/decimal/Cargo.toml
+++ b/src/common/decimal/Cargo.toml
@@ -8,7 +8,6 @@ license.workspace = true
 workspace = true

 [dependencies]
-arrow.workspace = true
 bigdecimal.workspace = true
 common-error.workspace = true
 common-macro.workspace = true
--- a/src/common/function/Cargo.toml
+++ b/src/common/function/Cargo.toml
@@ -11,7 +11,6 @@ workspace = true
 api.workspace = true
 arc-swap = "1.0"
 async-trait.workspace = true
-chrono-tz = "0.6"
 common-base.workspace = true
 common-catalog.workspace = true
 common-error.workspace = true
@@ -24,7 +23,6 @@ common-time.workspace = true
 common-version.workspace = true
 datafusion.workspace = true
 datatypes.workspace = true
-libc = "0.2"
 num = "0.4"
 num-traits = "0.2"
 once_cell.workspace = true
--- a/src/common/function/src/handlers.rs
+++ b/src/common/function/src/handlers.rs
@@ -18,6 +18,7 @@ use async_trait::async_trait;
 use common_base::AffectedRows;
 use common_meta::rpc::procedure::{MigrateRegionRequest, ProcedureStateResponse};
 use common_query::error::Result;
+use common_query::Output;
 use session::context::QueryContextRef;
 use store_api::storage::RegionId;
 use table::requests::{CompactTableRequest, DeleteRequest, FlushTableRequest, InsertRequest};
@@ -26,7 +27,7 @@ use table::requests::{CompactTableRequest, DeleteRequest, FlushTableRequest, Ins
 #[async_trait]
 pub trait TableMutationHandler: Send + Sync {
    /// Inserts rows into the table.
-    async fn insert(&self, request: InsertRequest, ctx: QueryContextRef) -> Result<AffectedRows>;
+    async fn insert(&self, request: InsertRequest, ctx: QueryContextRef) -> Result<Output>;

    /// Delete rows from the table.
    async fn delete(&self, request: DeleteRequest, ctx: QueryContextRef) -> Result<AffectedRows>;
--- a/src/common/function/src/scalars/math.rs
+++ b/src/common/function/src/scalars/math.rs
@@ -12,6 +12,7 @@
 // See the License for the specific language governing permissions and
 // limitations under the License.

+mod clamp;
 mod modulo;
 mod pow;
 mod rate;
@@ -19,6 +20,7 @@ mod rate;
 use std::fmt;
 use std::sync::Arc;

+pub use clamp::ClampFunction;
 use common_query::error::{GeneralDataFusionSnafu, Result};
 use common_query::prelude::Signature;
 use datafusion::error::DataFusionError;
@@ -40,7 +42,8 @@ impl MathFunction {
        registry.register(Arc::new(ModuloFunction));
        registry.register(Arc::new(PowFunction));
        registry.register(Arc::new(RateFunction));
-        registry.register(Arc::new(RangeFunction))
+        registry.register(Arc::new(RangeFunction));
+        registry.register(Arc::new(ClampFunction));
    }
 }

--- a/src/common/function/src/scalars/math/clamp.rs
+++ b/src/common/function/src/scalars/math/clamp.rs
@@ -0,0 +1,403 @@
+// Copyright 2023 Greptime Team
+//
+// Licensed under the Apache License, Version 2.0 (the "License");
+// you may not use this file except in compliance with the License.
+// You may obtain a copy of the License at
+//
+//     http://www.apache.org/licenses/LICENSE-2.0
+//
+// Unless required by applicable law or agreed to in writing, software
+// distributed under the License is distributed on an "AS IS" BASIS,
+// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+// See the License for the specific language governing permissions and
+// limitations under the License.
+
+use std::fmt::{self, Display};
+use std::sync::Arc;
+
+use common_query::error::{InvalidFuncArgsSnafu, Result};
+use common_query::prelude::Signature;
+use datafusion::arrow::array::{ArrayIter, PrimitiveArray};
+use datafusion::logical_expr::Volatility;
+use datatypes::data_type::{ConcreteDataType, DataType};
+use datatypes::prelude::VectorRef;
+use datatypes::types::LogicalPrimitiveType;
+use datatypes::value::TryAsPrimitive;
+use datatypes::vectors::PrimitiveVector;
+use datatypes::with_match_primitive_type_id;
+use snafu::{ensure, OptionExt};
+
+use crate::function::Function;
+
+#[derive(Clone, Debug, Default)]
+pub struct ClampFunction;
+
+const CLAMP_NAME: &str = "clamp";
+
+impl Function for ClampFunction {
+    fn name(&self) -> &str {
+        CLAMP_NAME
+    }
+
+    fn return_type(&self, input_types: &[ConcreteDataType]) -> Result<ConcreteDataType> {
+        // Type check is done by `signature`
+        Ok(input_types[0].clone())
+    }
+
+    fn signature(&self) -> Signature {
+        // input, min, max
+        Signature::uniform(3, ConcreteDataType::numerics(), Volatility::Immutable)
+    }
+
+    fn eval(
+        &self,
+        _func_ctx: crate::function::FunctionContext,
+        columns: &[VectorRef],
+    ) -> Result<VectorRef> {
+        ensure!(
+            columns.len() == 3,
+            InvalidFuncArgsSnafu {
+                err_msg: format!(
+                    "The length of the args is not correct, expect exactly 3, have: {}",
+                    columns.len()
+                ),
+            }
+        );
+        ensure!(
+            columns[0].data_type().is_numeric(),
+            InvalidFuncArgsSnafu {
+                err_msg: format!(
+                    "The first arg's type is not numeric, have: {}",
+                    columns[0].data_type()
+                ),
+            }
+        );
+        ensure!(
+            columns[0].data_type() == columns[1].data_type()
+                && columns[1].data_type() == columns[2].data_type(),
+            InvalidFuncArgsSnafu {
+                err_msg: format!(
+                    "Arguments don't have identical types: {}, {}, {}",
+                    columns[0].data_type(),
+                    columns[1].data_type(),
+                    columns[2].data_type()
+                ),
+            }
+        );
+        ensure!(
+            columns[1].len() == 1 && columns[2].len() == 1,
+            InvalidFuncArgsSnafu {
+                err_msg: format!(
+                    "The second and third args should be scalar, have: {:?}, {:?}",
+                    columns[1], columns[2]
+                ),
+            }
+        );
+
+        with_match_primitive_type_id!(columns[0].data_type().logical_type_id(), |$S| {
+            let input_array = columns[0].to_arrow_array();
+            let input = input_array
+                    .as_any()
+                    .downcast_ref::<PrimitiveArray<<$S as LogicalPrimitiveType>::ArrowPrimitive>>()
+                    .unwrap();
+
+            let min = TryAsPrimitive::<$S>::try_as_primitive(&columns[1].get(0))
+                .with_context(|| {
+                    InvalidFuncArgsSnafu {
+                        err_msg: "The second arg should not be none",
+                    }
+                })?;
+            let max = TryAsPrimitive::<$S>::try_as_primitive(&columns[2].get(0))
+                .with_context(|| {
+                    InvalidFuncArgsSnafu {
+                        err_msg: "The third arg should not be none",
+                    }
+                })?;
+
+            // ensure min <= max
+            ensure!(
+                min <= max,
+                    InvalidFuncArgsSnafu {
+                        err_msg: format!(
+                        "The second arg should be less than or equal to the third arg, have: {:?}, {:?}",
+                        columns[1], columns[2]
+                    ),
+                }
+            );
+
+            clamp_impl::<$S, true, true>(input, min, max)
+        },{
+            unreachable!()
+        })
+    }
+}
+
+impl Display for ClampFunction {
+    fn fmt(&self, f: &mut fmt::Formatter) -> fmt::Result {
+        write!(f, "{}", CLAMP_NAME.to_ascii_uppercase())
+    }
+}
+
+fn clamp_impl<T: LogicalPrimitiveType, const CLAMP_MIN: bool, const CLAMP_MAX: bool>(
+    input: &PrimitiveArray<T::ArrowPrimitive>,
+    min: T::Native,
+    max: T::Native,
+) -> Result<VectorRef> {
+    common_telemetry::info!("[DEBUG] min {min:?}, max {max:?}");
+
+    let iter = ArrayIter::new(input);
+    let result = iter.map(|x| {
+        x.map(|x| {
+            if CLAMP_MIN && x < min {
+                min
+            } else if CLAMP_MAX && x > max {
+                max
+            } else {
+                x
+            }
+        })
+    });
+    let result = PrimitiveArray::<T::ArrowPrimitive>::from_iter(result);
+    Ok(Arc::new(PrimitiveVector::<T>::from(result)))
+}
+
+#[cfg(test)]
+mod test {
+
+    use std::sync::Arc;
+
+    use datatypes::prelude::ScalarVector;
+    use datatypes::vectors::{
+        ConstantVector, Float64Vector, Int64Vector, StringVector, UInt64Vector,
+    };
+
+    use super::*;
+    use crate::function::FunctionContext;
+
+    #[test]
+    fn clamp_i64() {
+        let inputs = [
+            (
+                vec![Some(-3), Some(-2), Some(-1), Some(0), Some(1), Some(2)],
+                -1,
+                10,
+                vec![Some(-1), Some(-1), Some(-1), Some(0), Some(1), Some(2)],
+            ),
+            (
+                vec![Some(-3), Some(-2), Some(-1), Some(0), Some(1), Some(2)],
+                0,
+                0,
+                vec![Some(0), Some(0), Some(0), Some(0), Some(0), Some(0)],
+            ),
+            (
+                vec![Some(-3), None, Some(-1), None, None, Some(2)],
+                -2,
+                1,
+                vec![Some(-2), None, Some(-1), None, None, Some(1)],
+            ),
+            (
+                vec![None, None, None, None, None],
+                0,
+                1,
+                vec![None, None, None, None, None],
+            ),
+        ];
+
+        let func = ClampFunction;
+        for (in_data, min, max, expected) in inputs {
+            let args = [
+                Arc::new(Int64Vector::from(in_data)) as _,
+                Arc::new(Int64Vector::from_vec(vec![min])) as _,
+                Arc::new(Int64Vector::from_vec(vec![max])) as _,
+            ];
+            let result = func
+                .eval(FunctionContext::default(), args.as_slice())
+                .unwrap();
+            let expected: VectorRef = Arc::new(Int64Vector::from(expected));
+            assert_eq!(expected, result);
+        }
+    }
+
+    #[test]
+    fn clamp_u64() {
+        let inputs = [
+            (
+                vec![Some(0), Some(1), Some(2), Some(3), Some(4), Some(5)],
+                1,
+                3,
+                vec![Some(1), Some(1), Some(2), Some(3), Some(3), Some(3)],
+            ),
+            (
+                vec![Some(0), Some(1), Some(2), Some(3), Some(4), Some(5)],
+                0,
+                0,
+                vec![Some(0), Some(0), Some(0), Some(0), Some(0), Some(0)],
+            ),
+            (
+                vec![Some(0), None, Some(2), None, None, Some(5)],
+                1,
+                3,
+                vec![Some(1), None, Some(2), None, None, Some(3)],
+            ),
+            (
+                vec![None, None, None, None, None],
+                0,
+                1,
+                vec![None, None, None, None, None],
+            ),
+        ];
+
+        let func = ClampFunction;
+        for (in_data, min, max, expected) in inputs {
+            let args = [
+                Arc::new(UInt64Vector::from(in_data)) as _,
+                Arc::new(UInt64Vector::from_vec(vec![min])) as _,
+                Arc::new(UInt64Vector::from_vec(vec![max])) as _,
+            ];
+            let result = func
+                .eval(FunctionContext::default(), args.as_slice())
+                .unwrap();
+            let expected: VectorRef = Arc::new(UInt64Vector::from(expected));
+            assert_eq!(expected, result);
+        }
+    }
+
+    #[test]
+    fn clamp_f64() {
+        let inputs = [
+            (
+                vec![Some(-3.0), Some(-2.0), Some(-1.0), Some(0.0), Some(1.0)],
+                -1.0,
+                10.0,
+                vec![Some(-1.0), Some(-1.0), Some(-1.0), Some(0.0), Some(1.0)],
+            ),
+            (
+                vec![Some(-2.0), Some(-1.0), Some(0.0), Some(1.0)],
+                0.0,
+                0.0,
+                vec![Some(0.0), Some(0.0), Some(0.0), Some(0.0)],
+            ),
+            (
+                vec![Some(-3.0), None, Some(-1.0), None, None, Some(2.0)],
+                -2.0,
+                1.0,
+                vec![Some(-2.0), None, Some(-1.0), None, None, Some(1.0)],
+            ),
+            (
+                vec![None, None, None, None, None],
+                0.0,
+                1.0,
+                vec![None, None, None, None, None],
+            ),
+        ];
+
+        let func = ClampFunction;
+        for (in_data, min, max, expected) in inputs {
+            let args = [
+                Arc::new(Float64Vector::from(in_data)) as _,
+                Arc::new(Float64Vector::from_vec(vec![min])) as _,
+                Arc::new(Float64Vector::from_vec(vec![max])) as _,
+            ];
+            let result = func
+                .eval(FunctionContext::default(), args.as_slice())
+                .unwrap();
+            let expected: VectorRef = Arc::new(Float64Vector::from(expected));
+            assert_eq!(expected, result);
+        }
+    }
+
+    #[test]
+    fn clamp_const_i32() {
+        let input = vec![Some(5)];
+        let min = 2;
+        let max = 4;
+
+        let func = ClampFunction;
+        let args = [
+            Arc::new(ConstantVector::new(Arc::new(Int64Vector::from(input)), 1)) as _,
+            Arc::new(Int64Vector::from_vec(vec![min])) as _,
+            Arc::new(Int64Vector::from_vec(vec![max])) as _,
+        ];
+        let result = func
+            .eval(FunctionContext::default(), args.as_slice())
+            .unwrap();
+        let expected: VectorRef = Arc::new(Int64Vector::from(vec![Some(4)]));
+        assert_eq!(expected, result);
+    }
+
+    #[test]
+    fn clamp_invalid_min_max() {
+        let input = vec![Some(-3.0), Some(-2.0), Some(-1.0), Some(0.0), Some(1.0)];
+        let min = 10.0;
+        let max = -1.0;
+
+        let func = ClampFunction;
+        let args = [
+            Arc::new(Float64Vector::from(input)) as _,
+            Arc::new(Float64Vector::from_vec(vec![min])) as _,
+            Arc::new(Float64Vector::from_vec(vec![max])) as _,
+        ];
+        let result = func.eval(FunctionContext::default(), args.as_slice());
+        assert!(result.is_err());
+    }
+
+    #[test]
+    fn clamp_type_not_match() {
+        let input = vec![Some(-3.0), Some(-2.0), Some(-1.0), Some(0.0), Some(1.0)];
+        let min = -1;
+        let max = 10;
+
+        let func = ClampFunction;
+        let args = [
+            Arc::new(Float64Vector::from(input)) as _,
+            Arc::new(Int64Vector::from_vec(vec![min])) as _,
+            Arc::new(UInt64Vector::from_vec(vec![max])) as _,
+        ];
+        let result = func.eval(FunctionContext::default(), args.as_slice());
+        assert!(result.is_err());
+    }
+
+    #[test]
+    fn clamp_min_is_not_scalar() {
+        let input = vec![Some(-3.0), Some(-2.0), Some(-1.0), Some(0.0), Some(1.0)];
+        let min = -10.0;
+        let max = 1.0;
+
+        let func = ClampFunction;
+        let args = [
+            Arc::new(Float64Vector::from(input)) as _,
+            Arc::new(Float64Vector::from_vec(vec![min, min])) as _,
+            Arc::new(Float64Vector::from_vec(vec![max])) as _,
+        ];
+        let result = func.eval(FunctionContext::default(), args.as_slice());
+        assert!(result.is_err());
+    }
+
+    #[test]
+    fn clamp_no_max() {
+        let input = vec![Some(-3.0), Some(-2.0), Some(-1.0), Some(0.0), Some(1.0)];
+        let min = -10.0;
+
+        let func = ClampFunction;
+        let args = [
+            Arc::new(Float64Vector::from(input)) as _,
+            Arc::new(Float64Vector::from_vec(vec![min])) as _,
+        ];
+        let result = func.eval(FunctionContext::default(), args.as_slice());
+        assert!(result.is_err());
+    }
+
+    #[test]
+    fn clamp_on_string() {
+        let input = vec![Some("foo"), Some("foo"), Some("foo"), Some("foo")];
+
+        let func = ClampFunction;
+        let args = [
+            Arc::new(StringVector::from(input)) as _,
+            Arc::new(StringVector::from_vec(vec!["bar"])) as _,
+            Arc::new(StringVector::from_vec(vec!["baz"])) as _,
+        ];
+        let result = func.eval(FunctionContext::default(), args.as_slice());
+        assert!(result.is_err());
+    }
+}
--- a/src/common/function/src/scalars/timestamp.rs
+++ b/src/common/function/src/scalars/timestamp.rs
@@ -14,9 +14,11 @@

 use std::sync::Arc;
 mod greatest;
+mod to_timezone;
 mod to_unixtime;

 use greatest::GreatestFunction;
+use to_timezone::ToTimezoneFunction;
 use to_unixtime::ToUnixtimeFunction;

 use crate::function_registry::FunctionRegistry;
@@ -25,6 +27,7 @@ pub(crate) struct TimestampFunction;

 impl TimestampFunction {
    pub fn register(registry: &FunctionRegistry) {
+        registry.register(Arc::new(ToTimezoneFunction));
        registry.register(Arc::new(ToUnixtimeFunction));
        registry.register(Arc::new(GreatestFunction));
    }
--- a/src/common/function/src/scalars/timestamp/to_timezone.rs
+++ b/src/common/function/src/scalars/timestamp/to_timezone.rs
@@ -0,0 +1,313 @@
+// Copyright 2023 Greptime Team
+//
+// Licensed under the Apache License, Version 2.0 (the "License");
+// you may not use this file except in compliance with the License.
+// You may obtain a copy of the License at
+//
+//     http://www.apache.org/licenses/LICENSE-2.0
+//
+// Unless required by applicable law or agreed to in writing, software
+// distributed under the License is distributed on an "AS IS" BASIS,
+// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+// See the License for the specific language governing permissions and
+// limitations under the License.
+
+use std::fmt;
+use std::sync::Arc;
+
+use common_query::error::{InvalidFuncArgsSnafu, Result, UnsupportedInputDataTypeSnafu};
+use common_query::prelude::Signature;
+use common_time::{Timestamp, Timezone};
+use datatypes::data_type::ConcreteDataType;
+use datatypes::prelude::VectorRef;
+use datatypes::types::TimestampType;
+use datatypes::value::Value;
+use datatypes::vectors::{
+    Int64Vector, StringVector, TimestampMicrosecondVector, TimestampMillisecondVector,
+    TimestampNanosecondVector, TimestampSecondVector, Vector,
+};
+use snafu::{ensure, OptionExt};
+
+use crate::function::{Function, FunctionContext};
+use crate::helper;
+
+#[derive(Clone, Debug, Default)]
+pub struct ToTimezoneFunction;
+
+const NAME: &str = "to_timezone";
+
+fn convert_to_timezone(arg: &str) -> Option<Timezone> {
+    Timezone::from_tz_string(arg).ok()
+}
+
+fn convert_to_timestamp(arg: &Value) -> Option<Timestamp> {
+    match arg {
+        Value::Timestamp(ts) => Some(*ts),
+        Value::Int64(i) => Some(Timestamp::new_millisecond(*i)),
+        _ => None,
+    }
+}
+
+impl fmt::Display for ToTimezoneFunction {
+    fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result {
+        write!(f, "TO_TIMEZONE")
+    }
+}
+
+impl Function for ToTimezoneFunction {
+    fn name(&self) -> &str {
+        NAME
+    }
+
+    fn return_type(&self, input_types: &[ConcreteDataType]) -> Result<ConcreteDataType> {
+        // type checked by signature - MUST BE timestamp
+        Ok(input_types[0].clone())
+    }
+
+    fn signature(&self) -> Signature {
+        helper::one_of_sigs2(
+            vec![
+                ConcreteDataType::int32_datatype(),
+                ConcreteDataType::int64_datatype(),
+                ConcreteDataType::timestamp_second_datatype(),
+                ConcreteDataType::timestamp_millisecond_datatype(),
+                ConcreteDataType::timestamp_microsecond_datatype(),
+                ConcreteDataType::timestamp_nanosecond_datatype(),
+            ],
+            vec![ConcreteDataType::string_datatype()],
+        )
+    }
+
+    fn eval(&self, _ctx: FunctionContext, columns: &[VectorRef]) -> Result<VectorRef> {
+        ensure!(
+            columns.len() == 2,
+            InvalidFuncArgsSnafu {
+                err_msg: format!(
+                    "The length of the args is not correct, expect exactly 2, have: {}",
+                    columns.len()
+                ),
+            }
+        );
+
+        let array = columns[0].to_arrow_array();
+        let times = match columns[0].data_type() {
+            ConcreteDataType::Int64(_) | ConcreteDataType::Int32(_) => {
+                let vector = Int64Vector::try_from_arrow_array(array).unwrap();
+                (0..vector.len())
+                    .map(|i| convert_to_timestamp(&vector.get(i)))
+                    .collect::<Vec<_>>()
+            }
+            ConcreteDataType::Timestamp(ts) => match ts {
+                TimestampType::Second(_) => {
+                    let vector = TimestampSecondVector::try_from_arrow_array(array).unwrap();
+                    (0..vector.len())
+                        .map(|i| convert_to_timestamp(&vector.get(i)))
+                        .collect::<Vec<_>>()
+                }
+                TimestampType::Millisecond(_) => {
+                    let vector = TimestampMillisecondVector::try_from_arrow_array(array).unwrap();
+                    (0..vector.len())
+                        .map(|i| convert_to_timestamp(&vector.get(i)))
+                        .collect::<Vec<_>>()
+                }
+                TimestampType::Microsecond(_) => {
+                    let vector = TimestampMicrosecondVector::try_from_arrow_array(array).unwrap();
+                    (0..vector.len())
+                        .map(|i| convert_to_timestamp(&vector.get(i)))
+                        .collect::<Vec<_>>()
+                }
+                TimestampType::Nanosecond(_) => {
+                    let vector = TimestampNanosecondVector::try_from_arrow_array(array).unwrap();
+                    (0..vector.len())
+                        .map(|i| convert_to_timestamp(&vector.get(i)))
+                        .collect::<Vec<_>>()
+                }
+            },
+            _ => UnsupportedInputDataTypeSnafu {
+                function: NAME,
+                datatypes: columns.iter().map(|c| c.data_type()).collect::<Vec<_>>(),
+            }
+            .fail()?,
+        };
+
+        let tzs = {
+            let array = columns[1].to_arrow_array();
+            let vector = StringVector::try_from_arrow_array(&array)
+                .ok()
+                .with_context(|| UnsupportedInputDataTypeSnafu {
+                    function: NAME,
+                    datatypes: columns.iter().map(|c| c.data_type()).collect::<Vec<_>>(),
+                })?;
+            (0..vector.len())
+                .map(|i| convert_to_timezone(&vector.get(i).to_string()))
+                .collect::<Vec<_>>()
+        };
+
+        let result = times
+            .iter()
+            .zip(tzs.iter())
+            .map(|(time, tz)| match (time, tz) {
+                (Some(time), _) => Some(time.to_timezone_aware_string(tz.as_ref())),
+                _ => None,
+            })
+            .collect::<Vec<Option<String>>>();
+        Ok(Arc::new(StringVector::from(result)))
+    }
+}
+
+#[cfg(test)]
+mod tests {
+
+    use datatypes::scalars::ScalarVector;
+    use datatypes::timestamp::{
+        TimestampMicrosecond, TimestampMillisecond, TimestampNanosecond, TimestampSecond,
+    };
+    use datatypes::vectors::{Int64Vector, StringVector};
+
+    use super::*;
+
+    #[test]
+    fn test_timestamp_to_timezone() {
+        let f = ToTimezoneFunction;
+        assert_eq!("to_timezone", f.name());
+
+        let results = vec![
+            Some("1969-12-31 19:00:01"),
+            None,
+            Some("1970-01-01 03:00:01"),
+            None,
+        ];
+        let times: Vec<Option<TimestampSecond>> = vec![
+            Some(TimestampSecond::new(1)),
+            None,
+            Some(TimestampSecond::new(1)),
+            None,
+        ];
+        let ts_vector: TimestampSecondVector =
+            TimestampSecondVector::from_owned_iterator(times.into_iter());
+        let tzs = vec![Some("America/New_York"), None, Some("Europe/Moscow"), None];
+        let args: Vec<VectorRef> = vec![
+            Arc::new(ts_vector),
+            Arc::new(StringVector::from(tzs.clone())),
+        ];
+        let vector = f.eval(FunctionContext::default(), &args).unwrap();
+        assert_eq!(4, vector.len());
+        let expect_times: VectorRef = Arc::new(StringVector::from(results));
+        assert_eq!(expect_times, vector);
+
+        let results = vec![
+            Some("1969-12-31 19:00:00.001"),
+            None,
+            Some("1970-01-01 03:00:00.001"),
+            None,
+        ];
+        let times: Vec<Option<TimestampMillisecond>> = vec![
+            Some(TimestampMillisecond::new(1)),
+            None,
+            Some(TimestampMillisecond::new(1)),
+            None,
+        ];
+        let ts_vector: TimestampMillisecondVector =
+            TimestampMillisecondVector::from_owned_iterator(times.into_iter());
+        let args: Vec<VectorRef> = vec![
+            Arc::new(ts_vector),
+            Arc::new(StringVector::from(tzs.clone())),
+        ];
+        let vector = f.eval(FunctionContext::default(), &args).unwrap();
+        assert_eq!(4, vector.len());
+        let expect_times: VectorRef = Arc::new(StringVector::from(results));
+        assert_eq!(expect_times, vector);
+
+        let results = vec![
+            Some("1969-12-31 19:00:00.000001"),
+            None,
+            Some("1970-01-01 03:00:00.000001"),
+            None,
+        ];
+        let times: Vec<Option<TimestampMicrosecond>> = vec![
+            Some(TimestampMicrosecond::new(1)),
+            None,
+            Some(TimestampMicrosecond::new(1)),
+            None,
+        ];
+        let ts_vector: TimestampMicrosecondVector =
+            TimestampMicrosecondVector::from_owned_iterator(times.into_iter());
+
+        let args: Vec<VectorRef> = vec![
+            Arc::new(ts_vector),
+            Arc::new(StringVector::from(tzs.clone())),
+        ];
+        let vector = f.eval(FunctionContext::default(), &args).unwrap();
+        assert_eq!(4, vector.len());
+        let expect_times: VectorRef = Arc::new(StringVector::from(results));
+        assert_eq!(expect_times, vector);
+
+        let results = vec![
+            Some("1969-12-31 19:00:00.000000001"),
+            None,
+            Some("1970-01-01 03:00:00.000000001"),
+            None,
+        ];
+        let times: Vec<Option<TimestampNanosecond>> = vec![
+            Some(TimestampNanosecond::new(1)),
+            None,
+            Some(TimestampNanosecond::new(1)),
+            None,
+        ];
+        let ts_vector: TimestampNanosecondVector =
+            TimestampNanosecondVector::from_owned_iterator(times.into_iter());
+
+        let args: Vec<VectorRef> = vec![
+            Arc::new(ts_vector),
+            Arc::new(StringVector::from(tzs.clone())),
+        ];
+        let vector = f.eval(FunctionContext::default(), &args).unwrap();
+        assert_eq!(4, vector.len());
+        let expect_times: VectorRef = Arc::new(StringVector::from(results));
+        assert_eq!(expect_times, vector);
+    }
+
+    #[test]
+    fn test_numerical_to_timezone() {
+        let f = ToTimezoneFunction;
+        let results = vec![
+            Some("1969-12-31 19:00:00.001"),
+            None,
+            Some("1970-01-01 03:00:00.001"),
+            None,
+            Some("2024-03-26 23:01:50"),
+            None,
+            Some("2024-03-27 06:02:00"),
+            None,
+        ];
+        let times: Vec<Option<i64>> = vec![
+            Some(1),
+            None,
+            Some(1),
+            None,
+            Some(1711508510000),
+            None,
+            Some(1711508520000),
+            None,
+        ];
+        let ts_vector: Int64Vector = Int64Vector::from_owned_iterator(times.into_iter());
+        let tzs = vec![
+            Some("America/New_York"),
+            None,
+            Some("Europe/Moscow"),
+            None,
+            Some("America/New_York"),
+            None,
+            Some("Europe/Moscow"),
+            None,
+        ];
+        let args: Vec<VectorRef> = vec![
+            Arc::new(ts_vector),
+            Arc::new(StringVector::from(tzs.clone())),
+        ];
+        let vector = f.eval(FunctionContext::default(), &args).unwrap();
+        assert_eq!(8, vector.len());
+        let expect_times: VectorRef = Arc::new(StringVector::from(results));
+        assert_eq!(expect_times, vector);
+    }
+}
--- a/src/common/function/src/state.rs
+++ b/src/common/function/src/state.rs
@@ -35,6 +35,7 @@ impl FunctionState {
        use common_base::AffectedRows;
        use common_meta::rpc::procedure::{MigrateRegionRequest, ProcedureStateResponse};
        use common_query::error::Result;
+        use common_query::Output;
        use session::context::QueryContextRef;
        use store_api::storage::RegionId;
        use table::requests::{
@@ -70,8 +71,8 @@ impl FunctionState {
                &self,
                _request: InsertRequest,
                _ctx: QueryContextRef,
-            ) -> Result<AffectedRows> {
-                Ok(ROWS)
+            ) -> Result<Output> {
+                Ok(Output::new_with_affected_rows(ROWS))
            }

            async fn delete(
--- a/src/common/greptimedb-telemetry/Cargo.toml
+++ b/src/common/greptimedb-telemetry/Cargo.toml
@@ -9,12 +9,10 @@ workspace = true

 [dependencies]
 async-trait.workspace = true
-common-error.workspace = true
 common-runtime.workspace = true
 common-telemetry.workspace = true
 reqwest.workspace = true
 serde.workspace = true
-serde_json.workspace = true
 tokio.workspace = true
 uuid.workspace = true

--- a/src/common/grpc-expr/Cargo.toml
+++ b/src/common/grpc-expr/Cargo.toml
@@ -9,13 +9,11 @@ workspace = true

 [dependencies]
 api.workspace = true
-async-trait.workspace = true
 common-base.workspace = true
 common-catalog.workspace = true
 common-error.workspace = true
 common-macro.workspace = true
 common-query.workspace = true
-common-telemetry.workspace = true
 common-time.workspace = true
 datatypes.workspace = true
 snafu.workspace = true
--- a/src/common/grpc/Cargo.toml
+++ b/src/common/grpc/Cargo.toml
@@ -10,8 +10,6 @@ workspace = true
 [dependencies]
 api.workspace = true
 arrow-flight.workspace = true
-async-trait = "0.1"
-backtrace = "0.3"
 common-base.workspace = true
 common-error.workspace = true
 common-macro.workspace = true
@@ -20,10 +18,8 @@ common-runtime.workspace = true
 common-telemetry.workspace = true
 common-time.workspace = true
 dashmap.workspace = true
-datafusion.workspace = true
 datatypes.workspace = true
 flatbuffers = "23.1"
-futures = "0.3"
 lazy_static.workspace = true
 prost.workspace = true
 snafu.workspace = true
--- a/src/common/macro/src/admin_fn.rs
+++ b/src/common/macro/src/admin_fn.rs
@@ -32,7 +32,7 @@ macro_rules! ok {
    };
 }

-/// Internal util macro to to create an error.
+/// Internal util macro to create an error.
 macro_rules! error {
    ($span:expr, $msg: expr) => {
        Err(syn::Error::new($span, $msg))
--- a/src/common/macro/src/range_fn.rs
+++ b/src/common/macro/src/range_fn.rs
@@ -56,6 +56,18 @@ pub(crate) fn process_range_fn(args: TokenStream, input: TokenStream) -> TokenSt
    } = &sig;
    let arg_types = ok!(extract_input_types(inputs));

+    // with format like Float64Array
+    let array_types = arg_types
+        .iter()
+        .map(|ty| {
+            if let Type::Reference(TypeReference { elem, .. }) = ty {
+                elem.as_ref().clone()
+            } else {
+                ty.clone()
+            }
+        })
+        .collect::<Vec<_>>();
+
    // build the struct and its impl block
    // only do this when `display_name` is specified
    if let Ok(display_name) = get_ident(&arg_map, "display_name", arg_span) {
@@ -64,6 +76,8 @@ pub(crate) fn process_range_fn(args: TokenStream, input: TokenStream) -> TokenSt
            vis,
            ok!(get_ident(&arg_map, "name", arg_span)),
            display_name,
+            array_types,
+            ok!(get_ident(&arg_map, "ret", arg_span)),
        );
        result.extend(struct_code);
    }
@@ -90,6 +104,8 @@ fn build_struct(
    vis: Visibility,
    name: Ident,
    display_name_ident: Ident,
+    array_types: Vec<Type>,
+    return_array_type: Ident,
 ) -> TokenStream {
    let display_name = display_name_ident.to_string();
    quote! {
@@ -114,18 +130,12 @@ fn build_struct(
                }
            }

-            // TODO(ruihang): this should be parameterized
-            // time index column and value column
            fn input_type() -> Vec<DataType> {
-                vec![
-                    RangeArray::convert_data_type(DataType::Timestamp(TimeUnit::Millisecond, None)),
-                    RangeArray::convert_data_type(DataType::Float64),
-                ]
+                vec![#( RangeArray::convert_data_type(#array_types::new_null(0).data_type().clone()), )*]
            }

-            // TODO(ruihang): this should be parameterized
            fn return_type() -> DataType {
-                DataType::Float64
+                #return_array_type::new_null(0).data_type().clone()
            }
        }
    }
@@ -160,6 +170,7 @@ fn build_calc_fn(
        .map(|name| Ident::new(&format!("{}_range_array", name), name.span()))
        .collect::<Vec<_>>();
    let first_range_array_name = range_array_names.first().unwrap().clone();
+    let first_param_name = param_names.first().unwrap().clone();

    quote! {
        impl #name {
@@ -168,13 +179,29 @@ fn build_calc_fn(

                #( let #range_array_names = RangeArray::try_new(extract_array(&input[#param_numbers])?.to_data().into())?; )*

-                // TODO(ruihang): add ensure!()
+                // check arrays len
+                {
+                    let len_first = #first_range_array_name.len();
+                    #(
+                        if len_first != #range_array_names.len() {
+                            return Err(DataFusionError::Execution(format!("RangeArray have different lengths in PromQL function {}: array1={}, array2={}", #name::name(), len_first, #range_array_names.len())));
+                        }
+                    )*
+                }

                let mut result_array = Vec::new();
                for index in 0..#first_range_array_name.len(){
                    #( let #param_names = #range_array_names.get(index).unwrap().as_any().downcast_ref::<#unref_param_types>().unwrap().clone(); )*

-                    // TODO(ruihang): add ensure!() to check length
+                    // check element len
+                    {
+                        let len_first = #first_param_name.len();
+                        #(
+                            if len_first != #param_names.len() {
+                                return Err(DataFusionError::Execution(format!("RangeArray's element {} have different lengths in PromQL function {}: array1={}, array2={}", index, #name::name(), len_first, #param_names.len())));
+                            }
+                        )*
+                    }

                    let result = #fn_name(#( &#param_names, )*);
                    result_array.push(result);
--- a/src/common/meta/Cargo.toml
+++ b/src/common/meta/Cargo.toml
@@ -13,7 +13,6 @@ workspace = true
 [dependencies]
 api.workspace = true
 async-recursion = "1.0"
-async-stream.workspace = true
 async-trait.workspace = true
 base64.workspace = true
 bytes.workspace = true
@@ -26,7 +25,6 @@ common-macro.workspace = true
 common-procedure.workspace = true
 common-procedure-test.workspace = true
 common-recordbatch.workspace = true
-common-runtime.workspace = true
 common-telemetry.workspace = true
 common-time.workspace = true
 common-wal.workspace = true
@@ -53,6 +51,7 @@ strum.workspace = true
 table.workspace = true
 tokio.workspace = true
 tonic.workspace = true
+typetag = "0.2"

 [dev-dependencies]
 chrono.workspace = true
--- a/src/common/meta/src/cache_invalidator.rs
+++ b/src/common/meta/src/cache_invalidator.rs
@@ -14,14 +14,14 @@

 use std::sync::Arc;

-use table::metadata::TableId;
+use tokio::sync::RwLock;

 use crate::error::Result;
+use crate::instruction::CacheIdent;
 use crate::key::table_info::TableInfoKey;
 use crate::key::table_name::TableNameKey;
 use crate::key::table_route::TableRouteKey;
 use crate::key::TableMetaKey;
-use crate::table_name::TableName;

 /// KvBackend cache invalidator
 #[async_trait::async_trait]
@@ -46,10 +46,7 @@ pub struct Context {

 #[async_trait::async_trait]
 pub trait CacheInvalidator: Send + Sync {
-    // Invalidates table cache
-    async fn invalidate_table_id(&self, ctx: &Context, table_id: TableId) -> Result<()>;
-
-    async fn invalidate_table_name(&self, ctx: &Context, table_name: TableName) -> Result<()>;
+    async fn invalidate(&self, ctx: &Context, caches: Vec<CacheIdent>) -> Result<()>;
 }

 pub type CacheInvalidatorRef = Arc<dyn CacheInvalidator>;
@@ -58,11 +55,35 @@ pub struct DummyCacheInvalidator;

 #[async_trait::async_trait]
 impl CacheInvalidator for DummyCacheInvalidator {
-    async fn invalidate_table_id(&self, _ctx: &Context, _table_id: TableId) -> Result<()> {
+    async fn invalidate(&self, _ctx: &Context, _caches: Vec<CacheIdent>) -> Result<()> {
        Ok(())
    }
+}

-    async fn invalidate_table_name(&self, _ctx: &Context, _table_name: TableName) -> Result<()> {
+#[derive(Default)]
+pub struct MultiCacheInvalidator {
+    invalidators: RwLock<Vec<CacheInvalidatorRef>>,
+}
+
+impl MultiCacheInvalidator {
+    pub fn with_invalidators(invalidators: Vec<CacheInvalidatorRef>) -> Self {
+        Self {
+            invalidators: RwLock::new(invalidators),
+        }
+    }
+
+    pub async fn add_invalidator(&self, invalidator: CacheInvalidatorRef) {
+        self.invalidators.write().await.push(invalidator);
+    }
+}
+
+#[async_trait::async_trait]
+impl CacheInvalidator for MultiCacheInvalidator {
+    async fn invalidate(&self, ctx: &Context, caches: Vec<CacheIdent>) -> Result<()> {
+        let invalidators = self.invalidators.read().await;
+        for invalidator in invalidators.iter() {
+            invalidator.invalidate(ctx, caches.clone()).await?;
+        }
        Ok(())
    }
 }
@@ -72,21 +93,22 @@ impl<T> CacheInvalidator for T
 where
    T: KvCacheInvalidator,
 {
-    async fn invalidate_table_name(&self, _ctx: &Context, table_name: TableName) -> Result<()> {
-        let key: TableNameKey = (&table_name).into();
-
-        self.invalidate_key(&key.as_raw_key()).await;
-
-        Ok(())
-    }
-
-    async fn invalidate_table_id(&self, _ctx: &Context, table_id: TableId) -> Result<()> {
-        let key = TableInfoKey::new(table_id);
-        self.invalidate_key(&key.as_raw_key()).await;
-
-        let key = &TableRouteKey { table_id };
-        self.invalidate_key(&key.as_raw_key()).await;
+    async fn invalidate(&self, _ctx: &Context, caches: Vec<CacheIdent>) -> Result<()> {
+        for cache in caches {
+            match cache {
+                CacheIdent::TableId(table_id) => {
+                    let key = TableInfoKey::new(table_id);
+                    self.invalidate_key(&key.as_raw_key()).await;

+                    let key = &TableRouteKey { table_id };
+                    self.invalidate_key(&key.as_raw_key()).await;
+                }
+                CacheIdent::TableName(table_name) => {
+                    let key: TableNameKey = (&table_name).into();
+                    self.invalidate_key(&key.as_raw_key()).await
+                }
+            }
+        }
        Ok(())
    }
 }
--- a/src/common/meta/src/cluster.rs
+++ b/src/common/meta/src/cluster.rs
@@ -0,0 +1,300 @@
+// Copyright 2023 Greptime Team
+//
+// Licensed under the Apache License, Version 2.0 (the "License");
+// you may not use this file except in compliance with the License.
+// You may obtain a copy of the License at
+//
+//     http://www.apache.org/licenses/LICENSE-2.0
+//
+// Unless required by applicable law or agreed to in writing, software
+// distributed under the License is distributed on an "AS IS" BASIS,
+// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+// See the License for the specific language governing permissions and
+// limitations under the License.
+
+use std::str::FromStr;
+
+use common_error::ext::ErrorExt;
+use lazy_static::lazy_static;
+use regex::Regex;
+use serde::{Deserialize, Serialize};
+use snafu::{ensure, OptionExt, ResultExt};
+
+use crate::error::{
+    DecodeJsonSnafu, EncodeJsonSnafu, Error, FromUtf8Snafu, InvalidNodeInfoKeySnafu,
+    InvalidRoleSnafu, ParseNumSnafu, Result,
+};
+use crate::peer::Peer;
+
+const CLUSTER_NODE_INFO_PREFIX: &str = "__meta_cluster_node_info";
+
+lazy_static! {
+    static ref CLUSTER_NODE_INFO_PREFIX_PATTERN: Regex = Regex::new(&format!(
+        "^{CLUSTER_NODE_INFO_PREFIX}-([0-9]+)-([0-9]+)-([0-9]+)$"
+    ))
+    .unwrap();
+}
+
+/// [ClusterInfo] provides information about the cluster.
+#[async_trait::async_trait]
+pub trait ClusterInfo {
+    type Error: ErrorExt;
+
+    /// List all nodes by role in the cluster. If `role` is `None`, list all nodes.
+    async fn list_nodes(
+        &self,
+        role: Option<Role>,
+    ) -> std::result::Result<Vec<NodeInfo>, Self::Error>;
+
+    // TODO(jeremy): Other info, like region status, etc.
+}
+
+/// The key of [NodeInfo] in the storage. The format is `__meta_cluster_node_info-{cluster_id}-{role}-{node_id}`.
+#[derive(Debug, Clone, Eq, Hash, PartialEq, Serialize, Deserialize)]
+pub struct NodeInfoKey {
+    /// The cluster id.
+    pub cluster_id: u64,
+    /// The role of the node. It can be [Role::Datanode], [Role::Frontend], or [Role::Metasrv].
+    pub role: Role,
+    /// The node id.
+    pub node_id: u64,
+}
+
+impl NodeInfoKey {
+    pub fn key_prefix_with_cluster_id(cluster_id: u64) -> String {
+        format!("{}-{}-", CLUSTER_NODE_INFO_PREFIX, cluster_id)
+    }
+
+    pub fn key_prefix_with_role(cluster_id: u64, role: Role) -> String {
+        format!(
+            "{}-{}-{}-",
+            CLUSTER_NODE_INFO_PREFIX,
+            cluster_id,
+            i32::from(role)
+        )
+    }
+}
+
+/// The information of a node in the cluster.
+#[derive(Debug, Serialize, Deserialize)]
+pub struct NodeInfo {
+    /// The peer information. [node_id, address]
+    pub peer: Peer,
+    /// Last activity time in milliseconds.
+    pub last_activity_ts: i64,
+    /// The status of the node. Different roles have different node status.
+    pub status: NodeStatus,
+}
+
+#[derive(Debug, Clone, Eq, Hash, PartialEq, Serialize, Deserialize)]
+pub enum Role {
+    Datanode,
+    Frontend,
+    Metasrv,
+}
+
+#[derive(Debug, Serialize, Deserialize)]
+pub enum NodeStatus {
+    Datanode(DatanodeStatus),
+    Frontend(FrontendStatus),
+    Metasrv(MetasrvStatus),
+}
+
+/// The status of a datanode.
+#[derive(Debug, Serialize, Deserialize)]
+pub struct DatanodeStatus {
+    /// The read capacity units during this period.
+    pub rcus: i64,
+    /// The write capacity units during this period.
+    pub wcus: i64,
+    /// How many leader regions on this node.
+    pub leader_regions: usize,
+    /// How many follower regions on this node.
+    pub follower_regions: usize,
+}
+
+/// The status of a frontend.
+#[derive(Debug, Serialize, Deserialize)]
+pub struct FrontendStatus {}
+
+/// The status of a metasrv.
+#[derive(Debug, Serialize, Deserialize)]
+pub struct MetasrvStatus {
+    pub is_leader: bool,
+}
+
+impl FromStr for NodeInfoKey {
+    type Err = Error;
+
+    fn from_str(key: &str) -> Result<Self> {
+        let caps = CLUSTER_NODE_INFO_PREFIX_PATTERN
+            .captures(key)
+            .context(InvalidNodeInfoKeySnafu { key })?;
+
+        ensure!(caps.len() == 4, InvalidNodeInfoKeySnafu { key });
+
+        let cluster_id = caps[1].to_string();
+        let role = caps[2].to_string();
+        let node_id = caps[3].to_string();
+        let cluster_id: u64 = cluster_id.parse().context(ParseNumSnafu {
+            err_msg: format!("invalid cluster_id: {cluster_id}"),
+        })?;
+        let role: i32 = role.parse().context(ParseNumSnafu {
+            err_msg: format!("invalid role {role}"),
+        })?;
+        let role = Role::try_from(role)?;
+        let node_id: u64 = node_id.parse().context(ParseNumSnafu {
+            err_msg: format!("invalid node_id: {node_id}"),
+        })?;
+
+        Ok(Self {
+            cluster_id,
+            role,
+            node_id,
+        })
+    }
+}
+
+impl TryFrom<Vec<u8>> for NodeInfoKey {
+    type Error = Error;
+
+    fn try_from(bytes: Vec<u8>) -> Result<Self> {
+        String::from_utf8(bytes)
+            .context(FromUtf8Snafu {
+                name: "NodeInfoKey",
+            })
+            .map(|x| x.parse())?
+    }
+}
+
+impl From<NodeInfoKey> for Vec<u8> {
+    fn from(key: NodeInfoKey) -> Self {
+        format!(
+            "{}-{}-{}-{}",
+            CLUSTER_NODE_INFO_PREFIX,
+            key.cluster_id,
+            i32::from(key.role),
+            key.node_id
+        )
+        .into_bytes()
+    }
+}
+
+impl FromStr for NodeInfo {
+    type Err = Error;
+
+    fn from_str(value: &str) -> Result<Self> {
+        serde_json::from_str(value).context(DecodeJsonSnafu)
+    }
+}
+
+impl TryFrom<Vec<u8>> for NodeInfo {
+    type Error = Error;
+
+    fn try_from(bytes: Vec<u8>) -> Result<Self> {
+        String::from_utf8(bytes)
+            .context(FromUtf8Snafu { name: "NodeInfo" })
+            .map(|x| x.parse())?
+    }
+}
+
+impl TryFrom<NodeInfo> for Vec<u8> {
+    type Error = Error;
+
+    fn try_from(info: NodeInfo) -> Result<Self> {
+        Ok(serde_json::to_string(&info)
+            .context(EncodeJsonSnafu)?
+            .into_bytes())
+    }
+}
+
+impl From<Role> for i32 {
+    fn from(role: Role) -> Self {
+        match role {
+            Role::Datanode => 0,
+            Role::Frontend => 1,
+            Role::Metasrv => 2,
+        }
+    }
+}
+
+impl TryFrom<i32> for Role {
+    type Error = Error;
+
+    fn try_from(role: i32) -> Result<Self> {
+        match role {
+            0 => Ok(Self::Datanode),
+            1 => Ok(Self::Frontend),
+            2 => Ok(Self::Metasrv),
+            _ => InvalidRoleSnafu { role }.fail(),
+        }
+    }
+}
+
+#[cfg(test)]
+mod tests {
+    use std::assert_matches::assert_matches;
+
+    use crate::cluster::Role::{Datanode, Frontend};
+    use crate::cluster::{DatanodeStatus, NodeInfo, NodeInfoKey, NodeStatus};
+    use crate::peer::Peer;
+
+    #[test]
+    fn test_node_info_key_round_trip() {
+        let key = NodeInfoKey {
+            cluster_id: 1,
+            role: Datanode,
+            node_id: 2,
+        };
+
+        let key_bytes: Vec<u8> = key.into();
+        let new_key: NodeInfoKey = key_bytes.try_into().unwrap();
+
+        assert_eq!(1, new_key.cluster_id);
+        assert_eq!(Datanode, new_key.role);
+        assert_eq!(2, new_key.node_id);
+    }
+
+    #[test]
+    fn test_node_info_round_trip() {
+        let node_info = NodeInfo {
+            peer: Peer {
+                id: 1,
+                addr: "127.0.0.1".to_string(),
+            },
+            last_activity_ts: 123,
+            status: NodeStatus::Datanode(DatanodeStatus {
+                rcus: 1,
+                wcus: 2,
+                leader_regions: 3,
+                follower_regions: 4,
+            }),
+        };
+
+        let node_info_bytes: Vec<u8> = node_info.try_into().unwrap();
+        let new_node_info: NodeInfo = node_info_bytes.try_into().unwrap();
+
+        assert_matches!(
+            new_node_info,
+            NodeInfo {
+                peer: Peer { id: 1, .. },
+                last_activity_ts: 123,
+                status: NodeStatus::Datanode(DatanodeStatus {
+                    rcus: 1,
+                    wcus: 2,
+                    leader_regions: 3,
+                    follower_regions: 4,
+                }),
+            }
+        );
+    }
+
+    #[test]
+    fn test_node_info_key_prefix() {
+        let prefix = NodeInfoKey::key_prefix_with_cluster_id(1);
+        assert_eq!(prefix, "__meta_cluster_node_info-1-");
+
+        let prefix = NodeInfoKey::key_prefix_with_role(2, Frontend);
+        assert_eq!(prefix, "__meta_cluster_node_info-2-1-");
+    }
+}
--- a/src/common/meta/src/datanode_manager.rs
+++ b/src/common/meta/src/datanode_manager.rs
@@ -12,9 +12,10 @@
 // See the License for the specific language governing permissions and
 // limitations under the License.

+use std::collections::HashMap;
 use std::sync::Arc;

-use api::v1::region::{QueryRequest, RegionRequest};
+use api::v1::region::{QueryRequest, RegionRequest, RegionResponse};
 pub use common_base::AffectedRows;
 use common_recordbatch::SendableRecordBatchStream;

@@ -25,7 +26,7 @@ use crate::peer::Peer;
 #[async_trait::async_trait]
 pub trait Datanode: Send + Sync {
    /// Handles DML, and DDL requests.
-    async fn handle(&self, request: RegionRequest) -> Result<AffectedRows>;
+    async fn handle(&self, request: RegionRequest) -> Result<HandleResponse>;

    /// Handles query requests
    async fn handle_query(&self, request: QueryRequest) -> Result<SendableRecordBatchStream>;
@@ -41,3 +42,27 @@ pub trait DatanodeManager: Send + Sync {
 }

 pub type DatanodeManagerRef = Arc<dyn DatanodeManager>;
+
+/// This result struct is derived from [RegionResponse]
+#[derive(Debug)]
+pub struct HandleResponse {
+    pub affected_rows: AffectedRows,
+    pub extension: HashMap<String, Vec<u8>>,
+}
+
+impl HandleResponse {
+    pub fn from_region_response(region_response: RegionResponse) -> Self {
+        Self {
+            affected_rows: region_response.affected_rows as _,
+            extension: region_response.extension,
+        }
+    }
+
+    /// Creates one response without extension
+    pub fn new(affected_rows: AffectedRows) -> Self {
+        Self {
+            affected_rows,
+            extension: Default::default(),
+        }
+    }
+}
--- a/src/common/meta/src/ddl.rs
+++ b/src/common/meta/src/ddl.rs
@@ -22,17 +22,21 @@ use self::table_meta::TableMetadataAllocatorRef;
 use crate::cache_invalidator::CacheInvalidatorRef;
 use crate::datanode_manager::DatanodeManagerRef;
 use crate::error::Result;
-use crate::key::table_route::TableRouteValue;
+use crate::key::table_route::PhysicalTableRouteValue;
 use crate::key::TableMetadataManagerRef;
 use crate::region_keeper::MemoryRegionKeeperRef;
 use crate::rpc::ddl::{SubmitDdlTaskRequest, SubmitDdlTaskResponse};
 use crate::rpc::procedure::{MigrateRegionRequest, MigrateRegionResponse, ProcedureStateResponse};

+pub mod alter_logical_tables;
 pub mod alter_table;
+pub mod create_database;
 pub mod create_logical_tables;
 pub mod create_table;
 mod create_table_template;
+pub mod drop_database;
 pub mod drop_table;
+mod physical_table_metadata;
 pub mod table_meta;
 #[cfg(any(test, feature = "testing"))]
 pub mod test_util;
@@ -83,7 +87,7 @@ pub struct TableMetadata {
    /// Table id.
    pub table_id: TableId,
    /// Route information for each region of the table.
-    pub table_route: TableRouteValue,
+    pub table_route: PhysicalTableRouteValue,
    /// The encoded wal options for regions of the table.
    // If a region does not have an associated wal options, no key for the region would be found in the map.
    pub region_wal_options: HashMap<RegionNumber, String>,
--- a/src/common/meta/src/ddl/alter_logical_tables.rs
+++ b/src/common/meta/src/ddl/alter_logical_tables.rs
@@ -0,0 +1,265 @@
+// Copyright 2023 Greptime Team
+//
+// Licensed under the Apache License, Version 2.0 (the "License");
+// you may not use this file except in compliance with the License.
+// You may obtain a copy of the License at
+//
+//     http://www.apache.org/licenses/LICENSE-2.0
+//
+// Unless required by applicable law or agreed to in writing, software
+// distributed under the License is distributed on an "AS IS" BASIS,
+// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+// See the License for the specific language governing permissions and
+// limitations under the License.
+
+mod check;
+mod metadata;
+mod region_request;
+mod table_cache_keys;
+mod update_metadata;
+
+use async_trait::async_trait;
+use common_procedure::error::{FromJsonSnafu, Result as ProcedureResult, ToJsonSnafu};
+use common_procedure::{Context, LockKey, Procedure, Status};
+use common_telemetry::{info, warn};
+use futures_util::future;
+use serde::{Deserialize, Serialize};
+use snafu::{ensure, ResultExt};
+use store_api::metadata::ColumnMetadata;
+use store_api::metric_engine_consts::ALTER_PHYSICAL_EXTENSION_KEY;
+use strum::AsRefStr;
+use table::metadata::TableId;
+
+use crate::ddl::utils::add_peer_context_if_needed;
+use crate::ddl::DdlContext;
+use crate::error::{DecodeJsonSnafu, Error, MetadataCorruptionSnafu, Result};
+use crate::key::table_info::TableInfoValue;
+use crate::key::table_route::PhysicalTableRouteValue;
+use crate::lock_key::{CatalogLock, SchemaLock, TableLock};
+use crate::rpc::ddl::AlterTableTask;
+use crate::rpc::router::find_leaders;
+use crate::{cache_invalidator, metrics, ClusterId};
+
+pub struct AlterLogicalTablesProcedure {
+    pub context: DdlContext,
+    pub data: AlterTablesData,
+}
+
+impl AlterLogicalTablesProcedure {
+    pub const TYPE_NAME: &'static str = "metasrv-procedure::AlterLogicalTables";
+
+    pub fn new(
+        cluster_id: ClusterId,
+        tasks: Vec<AlterTableTask>,
+        physical_table_id: TableId,
+        context: DdlContext,
+    ) -> Self {
+        Self {
+            context,
+            data: AlterTablesData {
+                cluster_id,
+                state: AlterTablesState::Prepare,
+                tasks,
+                table_info_values: vec![],
+                physical_table_id,
+                physical_table_info: None,
+                physical_table_route: None,
+                physical_columns: vec![],
+            },
+        }
+    }
+
+    pub fn from_json(json: &str, context: DdlContext) -> ProcedureResult<Self> {
+        let data = serde_json::from_str(json).context(FromJsonSnafu)?;
+        Ok(Self { context, data })
+    }
+
+    pub(crate) async fn on_prepare(&mut self) -> Result<Status> {
+        // Checks all the tasks
+        self.check_input_tasks()?;
+        // Fills the table info values
+        self.fill_table_info_values().await?;
+        // Checks the physical table, must after [fill_table_info_values]
+        self.check_physical_table().await?;
+        // Fills the physical table info
+        self.fill_physical_table_info().await?;
+        // Filter the finished tasks
+        let finished_tasks = self.check_finished_tasks()?;
+        let already_finished_count = finished_tasks
+            .iter()
+            .map(|x| if *x { 1 } else { 0 })
+            .sum::<usize>();
+        let apply_tasks_count = self.data.tasks.len();
+        if already_finished_count == apply_tasks_count {
+            info!("All the alter tasks are finished, will skip the procedure.");
+            // Re-invalidate the table cache
+            self.data.state = AlterTablesState::InvalidateTableCache;
+            return Ok(Status::executing(true));
+        } else if already_finished_count > 0 {
+            info!(
+                "There are {} alter tasks, {} of them were already finished.",
+                apply_tasks_count, already_finished_count
+            );
+        }
+        self.filter_task(&finished_tasks)?;
+
+        // Next state
+        self.data.state = AlterTablesState::SubmitAlterRegionRequests;
+        Ok(Status::executing(true))
+    }
+
+    pub(crate) async fn on_submit_alter_region_requests(&mut self) -> Result<Status> {
+        // Safety: we have checked the state in on_prepare
+        let physical_table_route = &self.data.physical_table_route.as_ref().unwrap();
+        let leaders = find_leaders(&physical_table_route.region_routes);
+        let mut alter_region_tasks = Vec::with_capacity(leaders.len());
+
+        for peer in leaders {
+            let requester = self.context.datanode_manager.datanode(&peer).await;
+            let request = self.make_request(&peer, &physical_table_route.region_routes)?;
+
+            alter_region_tasks.push(async move {
+                requester
+                    .handle(request)
+                    .await
+                    .map_err(add_peer_context_if_needed(peer))
+            });
+        }
+
+        // Collects responses from datanodes.
+        let phy_raw_schemas = future::join_all(alter_region_tasks)
+            .await
+            .into_iter()
+            .map(|res| res.map(|mut res| res.extension.remove(ALTER_PHYSICAL_EXTENSION_KEY)))
+            .collect::<Result<Vec<_>>>()?;
+
+        if phy_raw_schemas.is_empty() {
+            self.data.state = AlterTablesState::UpdateMetadata;
+            return Ok(Status::executing(true));
+        }
+
+        // Verify all the physical schemas are the same
+        // Safety: previous check ensures this vec is not empty
+        let first = phy_raw_schemas.first().unwrap();
+        ensure!(
+            phy_raw_schemas.iter().all(|x| x == first),
+            MetadataCorruptionSnafu {
+                err_msg: "The physical schemas from datanodes are not the same."
+            }
+        );
+
+        // Decodes the physical raw schemas
+        if let Some(phy_raw_schema) = first {
+            self.data.physical_columns =
+                ColumnMetadata::decode_list(phy_raw_schema).context(DecodeJsonSnafu)?;
+        } else {
+            warn!("altering logical table result doesn't contains extension key `{ALTER_PHYSICAL_EXTENSION_KEY}`,leaving the physical table's schema unchanged");
+        }
+
+        self.data.state = AlterTablesState::UpdateMetadata;
+        Ok(Status::executing(true))
+    }
+
+    pub(crate) async fn on_update_metadata(&mut self) -> Result<Status> {
+        self.update_physical_table_metadata().await?;
+        self.update_logical_tables_metadata().await?;
+
+        self.data.state = AlterTablesState::InvalidateTableCache;
+        Ok(Status::executing(true))
+    }
+
+    pub(crate) async fn on_invalidate_table_cache(&mut self) -> Result<Status> {
+        let ctx = cache_invalidator::Context::default();
+        let to_invalidate = self.build_table_cache_keys_to_invalidate();
+
+        self.context
+            .cache_invalidator
+            .invalidate(&ctx, to_invalidate)
+            .await?;
+        Ok(Status::done())
+    }
+}
+
+#[async_trait]
+impl Procedure for AlterLogicalTablesProcedure {
+    fn type_name(&self) -> &str {
+        Self::TYPE_NAME
+    }
+
+    async fn execute(&mut self, _ctx: &Context) -> ProcedureResult<Status> {
+        let error_handler = |e: Error| {
+            if e.is_retry_later() {
+                common_procedure::Error::retry_later(e)
+            } else {
+                common_procedure::Error::external(e)
+            }
+        };
+
+        let state = &self.data.state;
+
+        let step = state.as_ref();
+
+        let _timer = metrics::METRIC_META_PROCEDURE_ALTER_TABLE
+            .with_label_values(&[step])
+            .start_timer();
+
+        match state {
+            AlterTablesState::Prepare => self.on_prepare().await,
+            AlterTablesState::SubmitAlterRegionRequests => {
+                self.on_submit_alter_region_requests().await
+            }
+            AlterTablesState::UpdateMetadata => self.on_update_metadata().await,
+            AlterTablesState::InvalidateTableCache => self.on_invalidate_table_cache().await,
+        }
+        .map_err(error_handler)
+    }
+
+    fn dump(&self) -> ProcedureResult<String> {
+        serde_json::to_string(&self.data).context(ToJsonSnafu)
+    }
+
+    fn lock_key(&self) -> LockKey {
+        // CatalogLock, SchemaLock,
+        // TableLock
+        // TableNameLock(s)
+        let mut lock_key = Vec::with_capacity(2 + 1 + self.data.tasks.len());
+        let table_ref = self.data.tasks[0].table_ref();
+        lock_key.push(CatalogLock::Read(table_ref.catalog).into());
+        lock_key.push(SchemaLock::read(table_ref.catalog, table_ref.schema).into());
+        lock_key.push(TableLock::Write(self.data.physical_table_id).into());
+        lock_key.extend(
+            self.data
+                .table_info_values
+                .iter()
+                .map(|table| TableLock::Write(table.table_info.ident.table_id).into()),
+        );
+
+        LockKey::new(lock_key)
+    }
+}
+
+#[derive(Debug, Serialize, Deserialize)]
+pub struct AlterTablesData {
+    cluster_id: ClusterId,
+    state: AlterTablesState,
+    tasks: Vec<AlterTableTask>,
+    /// Table info values before the alter operation.
+    /// Corresponding one-to-one with the AlterTableTask in tasks.
+    table_info_values: Vec<TableInfoValue>,
+    /// Physical table info
+    physical_table_id: TableId,
+    physical_table_info: Option<TableInfoValue>,
+    physical_table_route: Option<PhysicalTableRouteValue>,
+    physical_columns: Vec<ColumnMetadata>,
+}
+
+#[derive(Debug, Serialize, Deserialize, AsRefStr)]
+enum AlterTablesState {
+    /// Prepares to alter the table
+    Prepare,
+    SubmitAlterRegionRequests,
+    /// Updates table metadata.
+    UpdateMetadata,
+    /// Broadcasts the invalidating table cache instruction.
+    InvalidateTableCache,
+}
--- a/src/common/meta/src/ddl/alter_logical_tables/check.rs
+++ b/src/common/meta/src/ddl/alter_logical_tables/check.rs
@@ -0,0 +1,136 @@
+// Copyright 2023 Greptime Team
+//
+// Licensed under the Apache License, Version 2.0 (the "License");
+// you may not use this file except in compliance with the License.
+// You may obtain a copy of the License at
+//
+//     http://www.apache.org/licenses/LICENSE-2.0
+//
+// Unless required by applicable law or agreed to in writing, software
+// distributed under the License is distributed on an "AS IS" BASIS,
+// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+// See the License for the specific language governing permissions and
+// limitations under the License.
+
+use std::collections::HashSet;
+
+use api::v1::alter_expr::Kind;
+use snafu::{ensure, OptionExt};
+
+use crate::ddl::alter_logical_tables::AlterLogicalTablesProcedure;
+use crate::error::{AlterLogicalTablesInvalidArgumentsSnafu, Result};
+use crate::key::table_info::TableInfoValue;
+use crate::key::table_route::TableRouteValue;
+use crate::rpc::ddl::AlterTableTask;
+
+impl AlterLogicalTablesProcedure {
+    pub(crate) fn check_input_tasks(&self) -> Result<()> {
+        self.check_schema()?;
+        self.check_alter_kind()?;
+        Ok(())
+    }
+
+    pub(crate) async fn check_physical_table(&self) -> Result<()> {
+        let table_route_manager = self.context.table_metadata_manager.table_route_manager();
+        let table_ids = self
+            .data
+            .table_info_values
+            .iter()
+            .map(|v| v.table_info.ident.table_id)
+            .collect::<Vec<_>>();
+        let table_routes = table_route_manager
+            .table_route_storage()
+            .batch_get(&table_ids)
+            .await?;
+        let physical_table_id = self.data.physical_table_id;
+        let is_same_physical_table = table_routes.iter().all(|r| {
+            if let Some(TableRouteValue::Logical(r)) = r {
+                r.physical_table_id() == physical_table_id
+            } else {
+                false
+            }
+        });
+
+        ensure!(
+            is_same_physical_table,
+            AlterLogicalTablesInvalidArgumentsSnafu {
+                err_msg: "All the tasks should have the same physical table id"
+            }
+        );
+
+        Ok(())
+    }
+
+    pub(crate) fn check_finished_tasks(&self) -> Result<Vec<bool>> {
+        let task = &self.data.tasks;
+        let table_info_values = &self.data.table_info_values;
+
+        Ok(task
+            .iter()
+            .zip(table_info_values.iter())
+            .map(|(task, table)| Self::check_finished_task(task, table))
+            .collect())
+    }
+
+    // Checks if the schemas of the tasks are the same
+    fn check_schema(&self) -> Result<()> {
+        let is_same_schema = self.data.tasks.windows(2).all(|pair| {
+            pair[0].alter_table.catalog_name == pair[1].alter_table.catalog_name
+                && pair[0].alter_table.schema_name == pair[1].alter_table.schema_name
+        });
+
+        ensure!(
+            is_same_schema,
+            AlterLogicalTablesInvalidArgumentsSnafu {
+                err_msg: "Schemas of the tasks are not the same"
+            }
+        );
+
+        Ok(())
+    }
+
+    fn check_alter_kind(&self) -> Result<()> {
+        for task in &self.data.tasks {
+            let kind = task.alter_table.kind.as_ref().context(
+                AlterLogicalTablesInvalidArgumentsSnafu {
+                    err_msg: "Alter kind is missing",
+                },
+            )?;
+            let Kind::AddColumns(_) = kind else {
+                return AlterLogicalTablesInvalidArgumentsSnafu {
+                    err_msg: "Only support add columns operation",
+                }
+                .fail();
+            };
+        }
+
+        Ok(())
+    }
+
+    fn check_finished_task(task: &AlterTableTask, table: &TableInfoValue) -> bool {
+        let columns = table
+            .table_info
+            .meta
+            .schema
+            .column_schemas
+            .iter()
+            .map(|c| &c.name)
+            .collect::<HashSet<_>>();
+
+        let Some(kind) = task.alter_table.kind.as_ref() else {
+            return true; // Never get here since we have checked it in `check_alter_kind`
+        };
+        let Kind::AddColumns(add_columns) = kind else {
+            return true; // Never get here since we have checked it in `check_alter_kind`
+        };
+
+        // We only check that all columns have been finished. That is to say,
+        // if one part is finished but another part is not, it will be considered
+        // unfinished.
+        add_columns
+            .add_columns
+            .iter()
+            .map(|add_column| add_column.column_def.as_ref().map(|c| &c.name))
+            .all(|column| column.map(|c| columns.contains(c)).unwrap_or(false))
+    }
+}
--- a/src/common/meta/src/ddl/alter_logical_tables/metadata.rs
+++ b/src/common/meta/src/ddl/alter_logical_tables/metadata.rs
@@ -0,0 +1,159 @@
+// Copyright 2023 Greptime Team
+//
+// Licensed under the Apache License, Version 2.0 (the "License");
+// you may not use this file except in compliance with the License.
+// You may obtain a copy of the License at
+//
+//     http://www.apache.org/licenses/LICENSE-2.0
+//
+// Unless required by applicable law or agreed to in writing, software
+// distributed under the License is distributed on an "AS IS" BASIS,
+// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+// See the License for the specific language governing permissions and
+// limitations under the License.
+
+use common_catalog::format_full_table_name;
+use snafu::OptionExt;
+use table::metadata::TableId;
+
+use crate::ddl::alter_logical_tables::AlterLogicalTablesProcedure;
+use crate::error::{
+    AlterLogicalTablesInvalidArgumentsSnafu, Result, TableInfoNotFoundSnafu, TableNotFoundSnafu,
+    TableRouteNotFoundSnafu,
+};
+use crate::key::table_info::TableInfoValue;
+use crate::key::table_name::TableNameKey;
+use crate::key::table_route::TableRouteValue;
+use crate::rpc::ddl::AlterTableTask;
+
+impl AlterLogicalTablesProcedure {
+    pub(crate) fn filter_task(&mut self, finished_tasks: &[bool]) -> Result<()> {
+        debug_assert_eq!(finished_tasks.len(), self.data.tasks.len());
+        debug_assert_eq!(finished_tasks.len(), self.data.table_info_values.len());
+        self.data.tasks = self
+            .data
+            .tasks
+            .drain(..)
+            .zip(finished_tasks.iter())
+            .filter_map(|(task, finished)| if *finished { None } else { Some(task) })
+            .collect();
+        self.data.table_info_values = self
+            .data
+            .table_info_values
+            .drain(..)
+            .zip(finished_tasks.iter())
+            .filter_map(|(table_info_value, finished)| {
+                if *finished {
+                    None
+                } else {
+                    Some(table_info_value)
+                }
+            })
+            .collect();
+
+        Ok(())
+    }
+
+    pub(crate) async fn fill_physical_table_info(&mut self) -> Result<()> {
+        let (physical_table_info, physical_table_route) = self
+            .context
+            .table_metadata_manager
+            .get_full_table_info(self.data.physical_table_id)
+            .await?;
+
+        let physical_table_info = physical_table_info
+            .with_context(|| TableInfoNotFoundSnafu {
+                table: format!("table id - {}", self.data.physical_table_id),
+            })?
+            .into_inner();
+        let physical_table_route = physical_table_route
+            .context(TableRouteNotFoundSnafu {
+                table_id: self.data.physical_table_id,
+            })?
+            .into_inner();
+
+        self.data.physical_table_info = Some(physical_table_info);
+        let TableRouteValue::Physical(physical_table_route) = physical_table_route else {
+            return AlterLogicalTablesInvalidArgumentsSnafu {
+                err_msg: format!(
+                    "expected a physical table but got a logical table: {:?}",
+                    self.data.physical_table_id
+                ),
+            }
+            .fail();
+        };
+        self.data.physical_table_route = Some(physical_table_route);
+
+        Ok(())
+    }
+
+    pub(crate) async fn fill_table_info_values(&mut self) -> Result<()> {
+        let table_ids = self.get_all_table_ids().await?;
+        let table_info_values = self.get_all_table_info_values(&table_ids).await?;
+        debug_assert_eq!(table_info_values.len(), self.data.tasks.len());
+        self.data.table_info_values = table_info_values;
+
+        Ok(())
+    }
+
+    async fn get_all_table_info_values(
+        &self,
+        table_ids: &[TableId],
+    ) -> Result<Vec<TableInfoValue>> {
+        let table_info_manager = self.context.table_metadata_manager.table_info_manager();
+        let mut table_info_map = table_info_manager.batch_get(table_ids).await?;
+        let mut table_info_values = Vec::with_capacity(table_ids.len());
+        for (table_id, task) in table_ids.iter().zip(self.data.tasks.iter()) {
+            let table_info_value =
+                table_info_map
+                    .remove(table_id)
+                    .with_context(|| TableInfoNotFoundSnafu {
+                        table: extract_table_name(task),
+                    })?;
+            table_info_values.push(table_info_value);
+        }
+
+        Ok(table_info_values)
+    }
+
+    async fn get_all_table_ids(&self) -> Result<Vec<TableId>> {
+        let table_name_manager = self.context.table_metadata_manager.table_name_manager();
+        let table_name_keys = self
+            .data
+            .tasks
+            .iter()
+            .map(|task| extract_table_name_key(task))
+            .collect();
+
+        let table_name_values = table_name_manager.batch_get(table_name_keys).await?;
+        let mut table_ids = Vec::with_capacity(table_name_values.len());
+        for (value, task) in table_name_values.into_iter().zip(self.data.tasks.iter()) {
+            let table_id = value
+                .with_context(|| TableNotFoundSnafu {
+                    table_name: extract_table_name(task),
+                })?
+                .table_id();
+            table_ids.push(table_id);
+        }
+
+        Ok(table_ids)
+    }
+}
+
+#[inline]
+fn extract_table_name(task: &AlterTableTask) -> String {
+    format_full_table_name(
+        &task.alter_table.catalog_name,
+        &task.alter_table.schema_name,
+        &task.alter_table.table_name,
+    )
+}
+
+#[inline]
+fn extract_table_name_key(task: &AlterTableTask) -> TableNameKey {
+    TableNameKey::new(
+        &task.alter_table.catalog_name,
+        &task.alter_table.schema_name,
+        &task.alter_table.table_name,
+    )
+}
--- a/src/common/meta/src/ddl/alter_logical_tables/region_request.rs
+++ b/src/common/meta/src/ddl/alter_logical_tables/region_request.rs
@@ -0,0 +1,112 @@
+// Copyright 2023 Greptime Team
+//
+// Licensed under the Apache License, Version 2.0 (the "License");
+// you may not use this file except in compliance with the License.
+// You may obtain a copy of the License at
+//
+//     http://www.apache.org/licenses/LICENSE-2.0
+//
+// Unless required by applicable law or agreed to in writing, software
+// distributed under the License is distributed on an "AS IS" BASIS,
+// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+// See the License for the specific language governing permissions and
+// limitations under the License.
+
+use api::v1;
+use api::v1::alter_expr::Kind;
+use api::v1::region::{
+    alter_request, region_request, AddColumn, AddColumns, AlterRequest, AlterRequests,
+    RegionColumnDef, RegionRequest, RegionRequestHeader,
+};
+use common_telemetry::tracing_context::TracingContext;
+use store_api::storage::RegionId;
+
+use crate::ddl::alter_logical_tables::AlterLogicalTablesProcedure;
+use crate::error::Result;
+use crate::key::table_info::TableInfoValue;
+use crate::peer::Peer;
+use crate::rpc::ddl::AlterTableTask;
+use crate::rpc::router::{find_leader_regions, RegionRoute};
+
+impl AlterLogicalTablesProcedure {
+    pub(crate) fn make_request(
+        &self,
+        peer: &Peer,
+        region_routes: &[RegionRoute],
+    ) -> Result<RegionRequest> {
+        let alter_requests = self.make_alter_region_requests(peer, region_routes)?;
+        let request = RegionRequest {
+            header: Some(RegionRequestHeader {
+                tracing_context: TracingContext::from_current_span().to_w3c(),
+                ..Default::default()
+            }),
+            body: Some(region_request::Body::Alters(alter_requests)),
+        };
+
+        Ok(request)
+    }
+
+    fn make_alter_region_requests(
+        &self,
+        peer: &Peer,
+        region_routes: &[RegionRoute],
+    ) -> Result<AlterRequests> {
+        let tasks = &self.data.tasks;
+        let regions_on_this_peer = find_leader_regions(region_routes, peer);
+        let mut requests = Vec::with_capacity(tasks.len() * regions_on_this_peer.len());
+        for (task, table) in self
+            .data
+            .tasks
+            .iter()
+            .zip(self.data.table_info_values.iter())
+        {
+            for region_number in &regions_on_this_peer {
+                let region_id = RegionId::new(table.table_info.ident.table_id, *region_number);
+                let request = self.make_alter_region_request(region_id, task, table)?;
+                requests.push(request);
+            }
+        }
+
+        Ok(AlterRequests { requests })
+    }
+
+    fn make_alter_region_request(
+        &self,
+        region_id: RegionId,
+        task: &AlterTableTask,
+        table: &TableInfoValue,
+    ) -> Result<AlterRequest> {
+        let region_id = region_id.as_u64();
+        let schema_version = table.table_info.ident.version;
+        let kind = match &task.alter_table.kind {
+            Some(Kind::AddColumns(add_columns)) => Some(alter_request::Kind::AddColumns(
+                to_region_add_columns(add_columns),
+            )),
+            _ => unreachable!(), // Safety: we have checked the kind in check_input_tasks
+        };
+
+        Ok(AlterRequest {
+            region_id,
+            schema_version,
+            kind,
+        })
+    }
+}
+
+fn to_region_add_columns(add_columns: &v1::AddColumns) -> AddColumns {
+    let add_columns = add_columns
+        .add_columns
+        .iter()
+        .map(|add_column| {
+            let region_column_def = RegionColumnDef {
+                column_def: add_column.column_def.clone(),
+                ..Default::default() // other fields are not used in alter logical table
+            };
+            AddColumn {
+                column_def: Some(region_column_def),
+                ..Default::default() // other fields are not used in alter logical table
+            }
+        })
+        .collect();
+    AddColumns { add_columns }
+}
--- a/src/common/meta/src/ddl/alter_logical_tables/table_cache_keys.rs
+++ b/src/common/meta/src/ddl/alter_logical_tables/table_cache_keys.rs
@@ -0,0 +1,51 @@
+// Copyright 2023 Greptime Team
+//
+// Licensed under the Apache License, Version 2.0 (the "License");
+// you may not use this file except in compliance with the License.
+// You may obtain a copy of the License at
+//
+//     http://www.apache.org/licenses/LICENSE-2.0
+//
+// Unless required by applicable law or agreed to in writing, software
+// distributed under the License is distributed on an "AS IS" BASIS,
+// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+// See the License for the specific language governing permissions and
+// limitations under the License.
+
+use table::metadata::RawTableInfo;
+
+use crate::ddl::alter_logical_tables::AlterLogicalTablesProcedure;
+use crate::instruction::CacheIdent;
+use crate::table_name::TableName;
+
+impl AlterLogicalTablesProcedure {
+    pub(crate) fn build_table_cache_keys_to_invalidate(&self) -> Vec<CacheIdent> {
+        let mut cache_keys = self
+            .data
+            .table_info_values
+            .iter()
+            .flat_map(|table| {
+                vec![
+                    CacheIdent::TableId(table.table_info.ident.table_id),
+                    CacheIdent::TableName(extract_table_name(&table.table_info)),
+                ]
+            })
+            .collect::<Vec<_>>();
+        cache_keys.push(CacheIdent::TableId(self.data.physical_table_id));
+        // Safety: physical_table_info already filled in previous steps
+        let physical_table_info = &self.data.physical_table_info.as_ref().unwrap().table_info;
+        cache_keys.push(CacheIdent::TableName(extract_table_name(
+            physical_table_info,
+        )));
+
+        cache_keys
+    }
+}
+
+fn extract_table_name(table_info: &RawTableInfo) -> TableName {
+    TableName::new(
+        &table_info.catalog_name,
+        &table_info.schema_name,
+        &table_info.name,
+    )
+}
--- a/src/common/meta/src/ddl/alter_logical_tables/update_metadata.rs
+++ b/src/common/meta/src/ddl/alter_logical_tables/update_metadata.rs
@@ -0,0 +1,124 @@
+// Copyright 2023 Greptime Team
+//
+// Licensed under the Apache License, Version 2.0 (the "License");
+// you may not use this file except in compliance with the License.
+// You may obtain a copy of the License at
+//
+//     http://www.apache.org/licenses/LICENSE-2.0
+//
+// Unless required by applicable law or agreed to in writing, software
+// distributed under the License is distributed on an "AS IS" BASIS,
+// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+// See the License for the specific language governing permissions and
+// limitations under the License.
+
+use common_grpc_expr::alter_expr_to_request;
+use common_telemetry::warn;
+use itertools::Itertools;
+use snafu::ResultExt;
+use table::metadata::{RawTableInfo, TableInfo};
+
+use crate::ddl::alter_logical_tables::AlterLogicalTablesProcedure;
+use crate::ddl::physical_table_metadata;
+use crate::error;
+use crate::error::{ConvertAlterTableRequestSnafu, Result};
+use crate::key::table_info::TableInfoValue;
+use crate::key::DeserializedValueWithBytes;
+use crate::rpc::ddl::AlterTableTask;
+
+impl AlterLogicalTablesProcedure {
+    pub(crate) async fn update_physical_table_metadata(&mut self) -> Result<()> {
+        if self.data.physical_columns.is_empty() {
+            warn!("No physical columns found, leaving the physical table's schema unchanged when altering logical tables");
+            return Ok(());
+        }
+
+        let physical_table_info = self.data.physical_table_info.as_ref().unwrap();
+
+        // Generates new table info
+        let old_raw_table_info = physical_table_info.table_info.clone();
+        let new_raw_table_info = physical_table_metadata::build_new_physical_table_info(
+            old_raw_table_info,
+            &self.data.physical_columns,
+        );
+
+        // Updates physical table's metadata
+        self.context
+            .table_metadata_manager
+            .update_table_info(
+                DeserializedValueWithBytes::from_inner(physical_table_info.clone()),
+                new_raw_table_info,
+            )
+            .await?;
+
+        Ok(())
+    }
+
+    pub(crate) async fn update_logical_tables_metadata(&mut self) -> Result<()> {
+        let table_info_values = self.build_update_metadata()?;
+        let manager = &self.context.table_metadata_manager;
+        let chunk_size = manager.batch_update_table_info_value_chunk_size();
+        if table_info_values.len() > chunk_size {
+            let chunks = table_info_values
+                .into_iter()
+                .chunks(chunk_size)
+                .into_iter()
+                .map(|check| check.collect::<Vec<_>>())
+                .collect::<Vec<_>>();
+            for chunk in chunks {
+                manager.batch_update_table_info_values(chunk).await?;
+            }
+        } else {
+            manager
+                .batch_update_table_info_values(table_info_values)
+                .await?;
+        }
+
+        Ok(())
+    }
+
+    pub(crate) fn build_update_metadata(&self) -> Result<Vec<(TableInfoValue, RawTableInfo)>> {
+        let mut table_info_values_to_update = Vec::with_capacity(self.data.tasks.len());
+        for (task, table) in self
+            .data
+            .tasks
+            .iter()
+            .zip(self.data.table_info_values.iter())
+        {
+            table_info_values_to_update.push(self.build_new_table_info(task, table)?);
+        }
+
+        Ok(table_info_values_to_update)
+    }
+
+    fn build_new_table_info(
+        &self,
+        task: &AlterTableTask,
+        table: &TableInfoValue,
+    ) -> Result<(TableInfoValue, RawTableInfo)> {
+        // Builds new_meta
+        let table_info = TableInfo::try_from(table.table_info.clone())
+            .context(error::ConvertRawTableInfoSnafu)?;
+        let table_ref = task.table_ref();
+        let request =
+            alter_expr_to_request(table.table_info.ident.table_id, task.alter_table.clone())
+                .context(ConvertAlterTableRequestSnafu)?;
+        let new_meta = table_info
+            .meta
+            .builder_with_alter_kind(table_ref.table, &request.alter_kind, true)
+            .context(error::TableSnafu)?
+            .build()
+            .with_context(|_| error::BuildTableMetaSnafu {
+                table_name: table_ref.table,
+            })?;
+        let version = table_info.ident.version + 1;
+        let mut new_table = table_info;
+        new_table.meta = new_meta;
+        new_table.ident.version = version;
+
+        let mut raw_table_info = RawTableInfo::from(new_table);
+        raw_table_info.sort_columns();
+
+        Ok((table.clone(), raw_table_info))
+    }
+}
--- a/src/common/meta/src/ddl/alter_table.rs
+++ b/src/common/meta/src/ddl/alter_table.rs
@@ -43,6 +43,7 @@ use crate::cache_invalidator::Context;
 use crate::ddl::utils::add_peer_context_if_needed;
 use crate::ddl::DdlContext;
 use crate::error::{self, ConvertAlterTableRequestSnafu, Error, InvalidProtoMsgSnafu, Result};
+use crate::instruction::CacheIdent;
 use crate::key::table_info::TableInfoValue;
 use crate::key::table_name::TableNameKey;
 use crate::key::DeserializedValueWithBytes;
@@ -66,7 +67,6 @@ impl AlterTableProcedure {
        cluster_id: u64,
        task: AlterTableTask,
        table_info_value: DeserializedValueWithBytes<TableInfoValue>,
-        physical_table_info: Option<(TableId, TableName)>,
        context: DdlContext,
    ) -> Result<Self> {
        let alter_kind = task
@@ -86,13 +86,7 @@ impl AlterTableProcedure {

        Ok(Self {
            context,
-            data: AlterTableData::new(
-                task,
-                table_info_value,
-                physical_table_info,
-                cluster_id,
-                next_column_id,
-            ),
+            data: AlterTableData::new(task, table_info_value, cluster_id, next_column_id),
            kind,
        })
    }
@@ -280,7 +274,7 @@ impl AlterTableProcedure {

        let new_meta = table_info
            .meta
-            .builder_with_alter_kind(table_ref.table, &request.alter_kind)
+            .builder_with_alter_kind(table_ref.table, &request.alter_kind, false)
            .context(error::TableSnafu)?
            .build()
            .with_context(|_| error::BuildTableMetaSnafu {
@@ -330,35 +324,24 @@ impl AlterTableProcedure {
    async fn on_broadcast(&mut self) -> Result<Status> {
        let alter_kind = self.alter_kind()?;
        let cache_invalidator = &self.context.cache_invalidator;
-
-        if matches!(alter_kind, Kind::RenameTable { .. }) {
-            cache_invalidator
-                .invalidate_table_name(&Context::default(), self.data.table_ref().into())
-                .await?;
+        let cache_keys = if matches!(alter_kind, Kind::RenameTable { .. }) {
+            vec![CacheIdent::TableName(self.data.table_ref().into())]
        } else {
-            cache_invalidator
-                .invalidate_table_id(&Context::default(), self.data.table_id())
-                .await?;
+            vec![
+                CacheIdent::TableId(self.data.table_id()),
+                CacheIdent::TableName(self.data.table_ref().into()),
+            ]
        };

+        cache_invalidator
+            .invalidate(&Context::default(), cache_keys)
+            .await?;
+
        Ok(Status::done())
    }

    fn lock_key_inner(&self) -> Vec<StringKey> {
        let mut lock_key = vec![];
-
-        if let Some((physical_table_id, physical_table_name)) = self.data.physical_table_info() {
-            lock_key.push(CatalogLock::Read(&physical_table_name.catalog_name).into());
-            lock_key.push(
-                SchemaLock::read(
-                    &physical_table_name.catalog_name,
-                    &physical_table_name.schema_name,
-                )
-                .into(),
-            );
-            lock_key.push(TableLock::Read(*physical_table_id).into())
-        }
-
        let table_ref = self.data.table_ref();
        let table_id = self.data.table_id();
        lock_key.push(CatalogLock::Read(table_ref.catalog).into());
@@ -436,8 +419,6 @@ pub struct AlterTableData {
    task: AlterTableTask,
    /// Table info value before alteration.
    table_info_value: DeserializedValueWithBytes<TableInfoValue>,
-    /// Physical table name, if the table to alter is a logical table.
-    physical_table_info: Option<(TableId, TableName)>,
    /// Next column id of the table if the task adds columns to the table.
    next_column_id: Option<ColumnId>,
 }
@@ -446,7 +427,6 @@ impl AlterTableData {
    pub fn new(
        task: AlterTableTask,
        table_info_value: DeserializedValueWithBytes<TableInfoValue>,
-        physical_table_info: Option<(TableId, TableName)>,
        cluster_id: u64,
        next_column_id: Option<ColumnId>,
    ) -> Self {
@@ -454,7 +434,6 @@ impl AlterTableData {
            state: AlterTableState::Prepare,
            task,
            table_info_value,
-            physical_table_info,
            cluster_id,
            next_column_id,
        }
@@ -471,10 +450,6 @@ impl AlterTableData {
    fn table_info(&self) -> &RawTableInfo {
        &self.table_info_value.table_info
    }
-
-    fn physical_table_info(&self) -> Option<&(TableId, TableName)> {
-        self.physical_table_info.as_ref()
-    }
 }

 /// Creates region proto alter kind from `table_info` and `alter_kind`.
--- a/src/common/meta/src/ddl/create_database.rs
+++ b/src/common/meta/src/ddl/create_database.rs
@@ -0,0 +1,152 @@
+// Copyright 2023 Greptime Team
+//
+// Licensed under the Apache License, Version 2.0 (the "License");
+// you may not use this file except in compliance with the License.
+// You may obtain a copy of the License at
+//
+//     http://www.apache.org/licenses/LICENSE-2.0
+//
+// Unless required by applicable law or agreed to in writing, software
+// distributed under the License is distributed on an "AS IS" BASIS,
+// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+// See the License for the specific language governing permissions and
+// limitations under the License.
+
+use std::collections::HashMap;
+
+use async_trait::async_trait;
+use common_procedure::error::{FromJsonSnafu, Result as ProcedureResult, ToJsonSnafu};
+use common_procedure::{Context as ProcedureContext, LockKey, Procedure, Status};
+use serde::{Deserialize, Serialize};
+use snafu::{ensure, ResultExt};
+use strum::AsRefStr;
+
+use crate::ddl::utils::handle_retry_error;
+use crate::ddl::DdlContext;
+use crate::error::{self, Result};
+use crate::key::schema_name::{SchemaNameKey, SchemaNameValue};
+use crate::lock_key::{CatalogLock, SchemaLock};
+
+pub struct CreateDatabaseProcedure {
+    pub context: DdlContext,
+    pub data: CreateDatabaseData,
+}
+
+impl CreateDatabaseProcedure {
+    pub const TYPE_NAME: &'static str = "metasrv-procedure::CreateDatabase";
+
+    pub fn new(
+        catalog: String,
+        schema: String,
+        create_if_not_exists: bool,
+        options: Option<HashMap<String, String>>,
+        context: DdlContext,
+    ) -> Self {
+        Self {
+            context,
+            data: CreateDatabaseData {
+                state: CreateDatabaseState::Prepare,
+                catalog,
+                schema,
+                create_if_not_exists,
+                options,
+            },
+        }
+    }
+
+    pub fn from_json(json: &str, context: DdlContext) -> ProcedureResult<Self> {
+        let data = serde_json::from_str(json).context(FromJsonSnafu)?;
+
+        Ok(Self { context, data })
+    }
+
+    pub async fn on_prepare(&mut self) -> Result<Status> {
+        let exists = self
+            .context
+            .table_metadata_manager
+            .schema_manager()
+            .exists(SchemaNameKey::new(&self.data.catalog, &self.data.schema))
+            .await?;
+
+        if exists && self.data.create_if_not_exists {
+            return Ok(Status::done());
+        }
+
+        ensure!(
+            !exists,
+            error::SchemaAlreadyExistsSnafu {
+                catalog: &self.data.catalog,
+                schema: &self.data.schema,
+            }
+        );
+
+        self.data.state = CreateDatabaseState::CreateMetadata;
+        Ok(Status::executing(true))
+    }
+
+    pub async fn on_create_metadata(&mut self) -> Result<Status> {
+        let value: Option<SchemaNameValue> = self
+            .data
+            .options
+            .as_ref()
+            .map(|hash_map_ref| hash_map_ref.try_into())
+            .transpose()?;
+
+        self.context
+            .table_metadata_manager
+            .schema_manager()
+            .create(
+                SchemaNameKey::new(&self.data.catalog, &self.data.schema),
+                value,
+                self.data.create_if_not_exists,
+            )
+            .await?;
+
+        Ok(Status::done())
+    }
+}
+
+#[async_trait]
+impl Procedure for CreateDatabaseProcedure {
+    fn type_name(&self) -> &str {
+        Self::TYPE_NAME
+    }
+
+    async fn execute(&mut self, _ctx: &ProcedureContext) -> ProcedureResult<Status> {
+        let state = &self.data.state;
+
+        match state {
+            CreateDatabaseState::Prepare => self.on_prepare().await,
+            CreateDatabaseState::CreateMetadata => self.on_create_metadata().await,
+        }
+        .map_err(handle_retry_error)
+    }
+
+    fn dump(&self) -> ProcedureResult<String> {
+        serde_json::to_string(&self.data).context(ToJsonSnafu)
+    }
+
+    fn lock_key(&self) -> LockKey {
+        let lock_key = vec![
+            CatalogLock::Read(&self.data.catalog).into(),
+            SchemaLock::write(&self.data.catalog, &self.data.schema).into(),
+        ];
+
+        LockKey::new(lock_key)
+    }
+}
+
+#[derive(Debug, Clone, Serialize, Deserialize, AsRefStr)]
+pub enum CreateDatabaseState {
+    Prepare,
+    CreateMetadata,
+}
+
+#[derive(Debug, Serialize, Deserialize)]
+pub struct CreateDatabaseData {
+    pub state: CreateDatabaseState,
+    pub catalog: String,
+    pub schema: String,
+    pub create_if_not_exists: bool,
+    pub options: Option<HashMap<String, String>>,
+}
--- a/src/common/meta/src/ddl/create_logical_tables.rs
+++ b/src/common/meta/src/ddl/create_logical_tables.rs
@@ -12,39 +12,37 @@
 // See the License for the specific language governing permissions and
 // limitations under the License.

-use std::collections::HashMap;
+mod check;
+mod metadata;
+mod region_request;
+mod update_metadata;

-use api::v1::region::region_request::Body as PbRegionRequest;
-use api::v1::region::{CreateRequests, RegionRequest, RegionRequestHeader};
 use api::v1::CreateTableExpr;
 use async_trait::async_trait;
 use common_procedure::error::{FromJsonSnafu, Result as ProcedureResult, ToJsonSnafu};
 use common_procedure::{Context as ProcedureContext, LockKey, Procedure, Status};
-use common_telemetry::info;
-use common_telemetry::tracing_context::TracingContext;
+use common_telemetry::warn;
 use futures_util::future::join_all;
-use itertools::Itertools;
 use serde::{Deserialize, Serialize};
 use snafu::{ensure, ResultExt};
+use store_api::metadata::ColumnMetadata;
+use store_api::metric_engine_consts::ALTER_PHYSICAL_EXTENSION_KEY;
 use store_api::storage::{RegionId, RegionNumber};
 use strum::AsRefStr;
 use table::metadata::{RawTableInfo, TableId};

-use crate::ddl::create_table_template::{build_template, CreateRequestBuilder};
-use crate::ddl::utils::{add_peer_context_if_needed, handle_retry_error, region_storage_path};
+use crate::ddl::utils::{add_peer_context_if_needed, handle_retry_error};
 use crate::ddl::DdlContext;
-use crate::error::{Result, TableAlreadyExistsSnafu};
-use crate::key::table_name::TableNameKey;
+use crate::error::{DecodeJsonSnafu, MetadataCorruptionSnafu, Result};
 use crate::key::table_route::TableRouteValue;
-use crate::lock_key::{TableLock, TableNameLock};
-use crate::peer::Peer;
+use crate::lock_key::{CatalogLock, SchemaLock, TableLock, TableNameLock};
 use crate::rpc::ddl::CreateTableTask;
-use crate::rpc::router::{find_leader_regions, find_leaders, RegionRoute};
+use crate::rpc::router::{find_leaders, RegionRoute};
 use crate::{metrics, ClusterId};

 pub struct CreateLogicalTablesProcedure {
    pub context: DdlContext,
-    pub creator: TablesCreator,
+    pub data: CreateTablesData,
 }

 impl CreateLogicalTablesProcedure {
@@ -56,228 +54,138 @@ impl CreateLogicalTablesProcedure {
        physical_table_id: TableId,
        context: DdlContext,
    ) -> Self {
-        let creator = TablesCreator::new(cluster_id, tasks, physical_table_id);
-        Self { context, creator }
+        Self {
+            context,
+            data: CreateTablesData {
+                cluster_id,
+                state: CreateTablesState::Prepare,
+                tasks,
+                table_ids_already_exists: vec![],
+                physical_table_id,
+                physical_region_numbers: vec![],
+                physical_columns: vec![],
+            },
+        }
    }

    pub fn from_json(json: &str, context: DdlContext) -> ProcedureResult<Self> {
        let data = serde_json::from_str(json).context(FromJsonSnafu)?;
-        let creator = TablesCreator { data };
-        Ok(Self { context, creator })
+        Ok(Self { context, data })
    }

    /// On the prepares step, it performs:
    /// - Checks whether physical table exists.
    /// - Checks whether logical tables exist.
    /// - Allocates the table ids.
+    /// - Modify tasks to sort logical columns on their names.
    ///
    /// Abort(non-retry):
    /// - The physical table does not exist.
    /// - Failed to check whether tables exist.
    /// - One of logical tables has existing, and the table creation task without setting `create_if_not_exists`.
    pub(crate) async fn on_prepare(&mut self) -> Result<Status> {
-        let manager = &self.context.table_metadata_manager;
-
+        self.check_input_tasks()?;
        // Sets physical region numbers
-        let physical_table_id = self.creator.data.physical_table_id();
-        let physical_region_numbers = manager
-            .table_route_manager()
-            .get_physical_table_route(physical_table_id)
-            .await
-            .map(|(_, route)| TableRouteValue::Physical(route).region_numbers())?;
-        self.creator
-            .data
-            .set_physical_region_numbers(physical_region_numbers);
-
+        self.fill_physical_table_info().await?;
        // Checks if the tables exist
-        let table_name_keys = self
-            .creator
-            .data
-            .all_create_table_exprs()
-            .iter()
-            .map(|expr| TableNameKey::new(&expr.catalog_name, &expr.schema_name, &expr.table_name))
-            .collect::<Vec<_>>();
-        let already_exists_tables_ids = manager
-            .table_name_manager()
-            .batch_get(table_name_keys)
-            .await?
-            .iter()
-            .map(|x| x.map(|x| x.table_id()))
-            .collect::<Vec<_>>();
-
-        // Validates the tasks
-        let tasks = &mut self.creator.data.tasks;
-        for (task, table_id) in tasks.iter().zip(already_exists_tables_ids.iter()) {
-            if table_id.is_some() {
-                // If a table already exists, we just ignore it.
-                ensure!(
-                    task.create_table.create_if_not_exists,
-                    TableAlreadyExistsSnafu {
-                        table_name: task.create_table.table_name.to_string(),
-                    }
-                );
-                continue;
-            }
-        }
+        self.check_tables_already_exist().await?;

        // If all tables already exist, returns the table_ids.
-        if already_exists_tables_ids.iter().all(Option::is_some) {
+        if self
+            .data
+            .table_ids_already_exists
+            .iter()
+            .all(Option::is_some)
+        {
            return Ok(Status::done_with_output(
-                already_exists_tables_ids
-                    .into_iter()
+                self.data
+                    .table_ids_already_exists
+                    .drain(..)
                    .flatten()
                    .collect::<Vec<_>>(),
            ));
        }

-        // Allocates table ids
-        for (task, table_id) in tasks.iter_mut().zip(already_exists_tables_ids.iter()) {
-            let table_id = if let Some(table_id) = table_id {
-                *table_id
-            } else {
-                self.context
-                    .table_metadata_allocator
-                    .allocate_table_id(task)
-                    .await?
-            };
-            task.set_table_id(table_id);
-        }
+        // Allocates table ids and sort columns on their names.
+        self.allocate_table_ids().await?;

-        self.creator
-            .data
-            .set_table_ids_already_exists(already_exists_tables_ids);
-        self.creator.data.state = CreateTablesState::DatanodeCreateRegions;
+        self.data.state = CreateTablesState::DatanodeCreateRegions;
        Ok(Status::executing(true))
    }

    pub async fn on_datanode_create_regions(&mut self) -> Result<Status> {
-        let physical_table_id = self.creator.data.physical_table_id();
        let (_, physical_table_route) = self
            .context
            .table_metadata_manager
            .table_route_manager()
-            .get_physical_table_route(physical_table_id)
+            .get_physical_table_route(self.data.physical_table_id)
            .await?;
-        let region_routes = &physical_table_route.region_routes;

-        self.create_regions(region_routes).await
+        self.create_regions(&physical_table_route.region_routes)
+            .await
    }

-    /// Creates table metadata
+    /// Creates table metadata for logical tables and update corresponding physical
+    /// table's metadata.
    ///
    /// Abort(not-retry):
    /// - Failed to create table metadata.
-    pub async fn on_create_metadata(&self) -> Result<Status> {
-        let manager = &self.context.table_metadata_manager;
-        let physical_table_id = self.creator.data.physical_table_id();
-        let remaining_tasks = self.creator.data.remaining_tasks();
-        let num_tables = remaining_tasks.len();
-
-        if num_tables > 0 {
-            let chunk_size = manager.max_logical_tables_per_batch();
-            if num_tables > chunk_size {
-                let chunks = remaining_tasks
-                    .into_iter()
-                    .chunks(chunk_size)
-                    .into_iter()
-                    .map(|chunk| chunk.collect::<Vec<_>>())
-                    .collect::<Vec<_>>();
-                for chunk in chunks {
-                    manager.create_logical_tables_metadata(chunk).await?;
-                }
-            } else {
-                manager
-                    .create_logical_tables_metadata(remaining_tasks)
-                    .await?;
-            }
-        }
-
-        // The `table_id` MUST be collected after the [Prepare::Prepare],
-        // ensures the all `table_id`s have been allocated.
-        let table_ids = self
-            .creator
-            .data
-            .tasks
-            .iter()
-            .map(|task| task.table_info.ident.table_id)
-            .collect::<Vec<_>>();
-
-        info!("Created {num_tables} tables {table_ids:?} metadata for physical table {physical_table_id}");
+    pub async fn on_create_metadata(&mut self) -> Result<Status> {
+        self.update_physical_table_metadata().await?;
+        let table_ids = self.create_logical_tables_metadata().await?;

        Ok(Status::done_with_output(table_ids))
    }

-    fn create_region_request_builder(
-        &self,
-        physical_table_id: TableId,
-        task: &CreateTableTask,
-    ) -> Result<CreateRequestBuilder> {
-        let create_expr = &task.create_table;
-        let template = build_template(create_expr)?;
-        Ok(CreateRequestBuilder::new(template, Some(physical_table_id)))
-    }
-
-    fn one_datanode_region_requests(
-        &self,
-        datanode: &Peer,
-        region_routes: &[RegionRoute],
-    ) -> Result<CreateRequests> {
-        let create_tables_data = &self.creator.data;
-        let tasks = &create_tables_data.tasks;
-        let physical_table_id = create_tables_data.physical_table_id();
-        let regions = find_leader_regions(region_routes, datanode);
-        let mut requests = Vec::with_capacity(tasks.len() * regions.len());
-
-        for task in tasks {
-            let create_table_expr = &task.create_table;
-            let catalog = &create_table_expr.catalog_name;
-            let schema = &create_table_expr.schema_name;
-            let logical_table_id = task.table_info.ident.table_id;
-            let storage_path = region_storage_path(catalog, schema);
-            let request_builder = self.create_region_request_builder(physical_table_id, task)?;
-
-            for region_number in &regions {
-                let region_id = RegionId::new(logical_table_id, *region_number);
-                let create_region_request =
-                    request_builder.build_one(region_id, storage_path.clone(), &HashMap::new())?;
-                requests.push(create_region_request);
-            }
-        }
-
-        Ok(CreateRequests { requests })
-    }
-
    async fn create_regions(&mut self, region_routes: &[RegionRoute]) -> Result<Status> {
        let leaders = find_leaders(region_routes);
        let mut create_region_tasks = Vec::with_capacity(leaders.len());

-        for datanode in leaders {
-            let requester = self.context.datanode_manager.datanode(&datanode).await;
-            let creates = self.one_datanode_region_requests(&datanode, region_routes)?;
-            let request = RegionRequest {
-                header: Some(RegionRequestHeader {
-                    tracing_context: TracingContext::from_current_span().to_w3c(),
-                    ..Default::default()
-                }),
-                body: Some(PbRegionRequest::Creates(creates)),
-            };
+        for peer in leaders {
+            let requester = self.context.datanode_manager.datanode(&peer).await;
+            let request = self.make_request(&peer, region_routes)?;
+
            create_region_tasks.push(async move {
                requester
                    .handle(request)
                    .await
-                    .map_err(add_peer_context_if_needed(datanode))
+                    .map_err(add_peer_context_if_needed(peer))
            });
        }

-        join_all(create_region_tasks)
+        // Collects response from datanodes.
+        let phy_raw_schemas = join_all(create_region_tasks)
            .await
            .into_iter()
+            .map(|res| res.map(|mut res| res.extension.remove(ALTER_PHYSICAL_EXTENSION_KEY)))
            .collect::<Result<Vec<_>>>()?;

-        self.creator.data.state = CreateTablesState::CreateMetadata;
+        if phy_raw_schemas.is_empty() {
+            self.data.state = CreateTablesState::CreateMetadata;
+            return Ok(Status::executing(false));
+        }

-        // Ensures the procedures after the crash start from the `DatanodeCreateRegions` stage.
-        Ok(Status::executing(false))
+        // Verify all the physical schemas are the same
+        // Safety: previous check ensures this vec is not empty
+        let first = phy_raw_schemas.first().unwrap();
+        ensure!(
+            phy_raw_schemas.iter().all(|x| x == first),
+            MetadataCorruptionSnafu {
+                err_msg: "The physical schemas from datanodes are not the same."
+            }
+        );
+
+        // Decodes the physical raw schemas
+        if let Some(phy_raw_schemas) = first {
+            self.data.physical_columns =
+                ColumnMetadata::decode_list(phy_raw_schemas).context(DecodeJsonSnafu)?;
+        } else {
+            warn!("creating logical table result doesn't contains extension key `{ALTER_PHYSICAL_EXTENSION_KEY}`,leaving the physical table's schema unchanged");
+        }
+
+        self.data.state = CreateTablesState::CreateMetadata;
+
+        Ok(Status::executing(true))
    }
 }

@@ -288,7 +196,7 @@ impl Procedure for CreateLogicalTablesProcedure {
    }

    async fn execute(&mut self, _ctx: &ProcedureContext) -> ProcedureResult<Status> {
-        let state = &self.creator.data.state;
+        let state = &self.data.state;

        let _timer = metrics::METRIC_META_PROCEDURE_CREATE_TABLES
            .with_label_values(&[state.as_ref()])
@@ -303,13 +211,20 @@ impl Procedure for CreateLogicalTablesProcedure {
    }

    fn dump(&self) -> ProcedureResult<String> {
-        serde_json::to_string(&self.creator.data).context(ToJsonSnafu)
+        serde_json::to_string(&self.data).context(ToJsonSnafu)
    }

    fn lock_key(&self) -> LockKey {
-        let mut lock_key = Vec::with_capacity(1 + self.creator.data.tasks.len());
-        lock_key.push(TableLock::Write(self.creator.data.physical_table_id()).into());
-        for task in &self.creator.data.tasks {
+        // CatalogLock, SchemaLock,
+        // TableLock
+        // TableNameLock(s)
+        let mut lock_key = Vec::with_capacity(2 + 1 + self.data.tasks.len());
+        let table_ref = self.data.tasks[0].table_ref();
+        lock_key.push(CatalogLock::Read(table_ref.catalog).into());
+        lock_key.push(SchemaLock::read(table_ref.catalog, table_ref.schema).into());
+        lock_key.push(TableLock::Write(self.data.physical_table_id).into());
+
+        for task in &self.data.tasks {
            lock_key.push(
                TableNameLock::new(
                    &task.create_table.catalog_name,
@@ -323,32 +238,6 @@ impl Procedure for CreateLogicalTablesProcedure {
    }
 }

-pub struct TablesCreator {
-    /// The serializable data.
-    pub data: CreateTablesData,
-}
-
-impl TablesCreator {
-    pub fn new(
-        cluster_id: ClusterId,
-        tasks: Vec<CreateTableTask>,
-        physical_table_id: TableId,
-    ) -> Self {
-        let len = tasks.len();
-
-        Self {
-            data: CreateTablesData {
-                cluster_id,
-                state: CreateTablesState::Prepare,
-                tasks,
-                table_ids_already_exists: vec![None; len],
-                physical_table_id,
-                physical_region_numbers: vec![],
-            },
-        }
-    }
-}
-
 #[derive(Debug, Serialize, Deserialize)]
 pub struct CreateTablesData {
    cluster_id: ClusterId,
@@ -357,6 +246,7 @@ pub struct CreateTablesData {
    table_ids_already_exists: Vec<Option<TableId>>,
    physical_table_id: TableId,
    physical_region_numbers: Vec<RegionNumber>,
+    physical_columns: Vec<ColumnMetadata>,
 }

 impl CreateTablesData {
@@ -364,18 +254,6 @@ impl CreateTablesData {
        &self.state
    }

-    fn physical_table_id(&self) -> TableId {
-        self.physical_table_id
-    }
-
-    fn set_physical_region_numbers(&mut self, physical_region_numbers: Vec<RegionNumber>) {
-        self.physical_region_numbers = physical_region_numbers;
-    }
-
-    fn set_table_ids_already_exists(&mut self, table_ids_already_exists: Vec<Option<TableId>>) {
-        self.table_ids_already_exists = table_ids_already_exists;
-    }
-
    fn all_create_table_exprs(&self) -> Vec<&CreateTableExpr> {
        self.tasks
            .iter()
--- a/src/common/meta/src/ddl/create_logical_tables/check.rs
+++ b/src/common/meta/src/ddl/create_logical_tables/check.rs
@@ -0,0 +1,81 @@
+// Copyright 2023 Greptime Team
+//
+// Licensed under the Apache License, Version 2.0 (the "License");
+// you may not use this file except in compliance with the License.
+// You may obtain a copy of the License at
+//
+//     http://www.apache.org/licenses/LICENSE-2.0
+//
+// Unless required by applicable law or agreed to in writing, software
+// distributed under the License is distributed on an "AS IS" BASIS,
+// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+// See the License for the specific language governing permissions and
+// limitations under the License.
+
+use snafu::ensure;
+
+use crate::ddl::create_logical_tables::CreateLogicalTablesProcedure;
+use crate::error::{CreateLogicalTablesInvalidArgumentsSnafu, Result, TableAlreadyExistsSnafu};
+use crate::key::table_name::TableNameKey;
+
+impl CreateLogicalTablesProcedure {
+    pub(crate) fn check_input_tasks(&self) -> Result<()> {
+        self.check_schema()?;
+
+        Ok(())
+    }
+
+    pub(crate) async fn check_tables_already_exist(&mut self) -> Result<()> {
+        let table_name_keys = self
+            .data
+            .all_create_table_exprs()
+            .iter()
+            .map(|expr| TableNameKey::new(&expr.catalog_name, &expr.schema_name, &expr.table_name))
+            .collect::<Vec<_>>();
+        let table_ids_already_exists = self
+            .context
+            .table_metadata_manager
+            .table_name_manager()
+            .batch_get(table_name_keys)
+            .await?
+            .iter()
+            .map(|x| x.map(|x| x.table_id()))
+            .collect::<Vec<_>>();
+
+        self.data.table_ids_already_exists = table_ids_already_exists;
+
+        // Validates the tasks
+        let tasks = &mut self.data.tasks;
+        for (task, table_id) in tasks.iter().zip(self.data.table_ids_already_exists.iter()) {
+            if table_id.is_some() {
+                // If a table already exists, we just ignore it.
+                ensure!(
+                    task.create_table.create_if_not_exists,
+                    TableAlreadyExistsSnafu {
+                        table_name: task.create_table.table_name.to_string(),
+                    }
+                );
+                continue;
+            }
+        }
+
+        Ok(())
+    }
+
+    // Checks if the schemas of the tasks are the same
+    fn check_schema(&self) -> Result<()> {
+        let is_same_schema = self.data.tasks.windows(2).all(|pair| {
+            pair[0].create_table.catalog_name == pair[1].create_table.catalog_name
+                && pair[0].create_table.schema_name == pair[1].create_table.schema_name
+        });
+
+        ensure!(
+            is_same_schema,
+            CreateLogicalTablesInvalidArgumentsSnafu {
+                err_msg: "Schemas of the tasks are not the same"
+            }
+        );
+
+        Ok(())
+    }
+}
--- a/src/common/meta/src/ddl/create_logical_tables/metadata.rs
+++ b/src/common/meta/src/ddl/create_logical_tables/metadata.rs
@@ -0,0 +1,57 @@
+// Copyright 2023 Greptime Team
+//
+// Licensed under the Apache License, Version 2.0 (the "License");
+// you may not use this file except in compliance with the License.
+// You may obtain a copy of the License at
+//
+//     http://www.apache.org/licenses/LICENSE-2.0
+//
+// Unless required by applicable law or agreed to in writing, software
+// distributed under the License is distributed on an "AS IS" BASIS,
+// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+// See the License for the specific language governing permissions and
+// limitations under the License.
+
+use crate::ddl::create_logical_tables::CreateLogicalTablesProcedure;
+use crate::error::Result;
+use crate::key::table_route::TableRouteValue;
+
+impl CreateLogicalTablesProcedure {
+    pub(crate) async fn fill_physical_table_info(&mut self) -> Result<()> {
+        let physical_region_numbers = self
+            .context
+            .table_metadata_manager
+            .table_route_manager()
+            .get_physical_table_route(self.data.physical_table_id)
+            .await
+            .map(|(_, route)| TableRouteValue::Physical(route).region_numbers())?;
+
+        self.data.physical_region_numbers = physical_region_numbers;
+
+        Ok(())
+    }
+
+    pub(crate) async fn allocate_table_ids(&mut self) -> Result<()> {
+        for (task, table_id) in self
+            .data
+            .tasks
+            .iter_mut()
+            .zip(self.data.table_ids_already_exists.iter())
+        {
+            let table_id = if let Some(table_id) = table_id {
+                *table_id
+            } else {
+                self.context
+                    .table_metadata_allocator
+                    .allocate_table_id(task)
+                    .await?
+            };
+            task.set_table_id(table_id);
+
+            // sort columns in task
+            task.sort_columns();
+        }
+
+        Ok(())
+    }
+}
--- a/src/common/meta/src/ddl/create_logical_tables/region_request.rs
+++ b/src/common/meta/src/ddl/create_logical_tables/region_request.rs
@@ -0,0 +1,74 @@
+// Copyright 2023 Greptime Team
+//
+// Licensed under the Apache License, Version 2.0 (the "License");
+// you may not use this file except in compliance with the License.
+// You may obtain a copy of the License at
+//
+//     http://www.apache.org/licenses/LICENSE-2.0
+//
+// Unless required by applicable law or agreed to in writing, software
+// distributed under the License is distributed on an "AS IS" BASIS,
+// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+// See the License for the specific language governing permissions and
+// limitations under the License.
+
+use std::collections::HashMap;
+
+use api::v1::region::{region_request, CreateRequests, RegionRequest, RegionRequestHeader};
+use common_telemetry::tracing_context::TracingContext;
+use store_api::storage::RegionId;
+
+use crate::ddl::create_logical_tables::CreateLogicalTablesProcedure;
+use crate::ddl::create_table_template::{build_template, CreateRequestBuilder};
+use crate::ddl::utils::region_storage_path;
+use crate::error::Result;
+use crate::peer::Peer;
+use crate::rpc::ddl::CreateTableTask;
+use crate::rpc::router::{find_leader_regions, RegionRoute};
+
+impl CreateLogicalTablesProcedure {
+    pub(crate) fn make_request(
+        &self,
+        peer: &Peer,
+        region_routes: &[RegionRoute],
+    ) -> Result<RegionRequest> {
+        let tasks = &self.data.tasks;
+        let regions_on_this_peer = find_leader_regions(region_routes, peer);
+        let mut requests = Vec::with_capacity(tasks.len() * regions_on_this_peer.len());
+        for task in tasks {
+            let create_table_expr = &task.create_table;
+            let catalog = &create_table_expr.catalog_name;
+            let schema = &create_table_expr.schema_name;
+            let logical_table_id = task.table_info.ident.table_id;
+            let storage_path = region_storage_path(catalog, schema);
+            let request_builder = self.create_region_request_builder(task)?;
+
+            for region_number in &regions_on_this_peer {
+                let region_id = RegionId::new(logical_table_id, *region_number);
+                let one_region_request =
+                    request_builder.build_one(region_id, storage_path.clone(), &HashMap::new())?;
+                requests.push(one_region_request);
+            }
+        }
+
+        Ok(RegionRequest {
+            header: Some(RegionRequestHeader {
+                tracing_context: TracingContext::from_current_span().to_w3c(),
+                ..Default::default()
+            }),
+            body: Some(region_request::Body::Creates(CreateRequests { requests })),
+        })
+    }
+
+    fn create_region_request_builder(
+        &self,
+        task: &CreateTableTask,
+    ) -> Result<CreateRequestBuilder> {
+        let create_expr = &task.create_table;
+        let template = build_template(create_expr)?;
+        Ok(CreateRequestBuilder::new(
+            template,
+            Some(self.data.physical_table_id),
+        ))
+    }
+}
--- a/Show More
+++ b/Show More