chore: Merge branch 'main' into chore/bench-metrics

feat: bump opendal and switch prometheus layer to the upstream impl (#5179 )
* feat: bump opendal and switch prometheus layer to the upstream impl Signed-off-by: Ruihang Xia <waynestxia@gmail.com> * remove unused files Signed-off-by: Ruihang Xia <waynestxia@gmail.com> * fix tests Signed-off-by: Ruihang Xia <waynestxia@gmail.com> * remove unused things Signed-off-by: Ruihang Xia <waynestxia@gmail.com> * remove root dir on recovering cache Signed-off-by: Ruihang Xia <waynestxia@gmail.com> * filter out non-files entry in test Signed-off-by: Ruihang Xia <waynestxia@gmail.com> --------- Signed-off-by: Ruihang Xia <waynestxia@gmail.com>
2025-12-25 07:30:02 +00:00 · 2024-12-19 16:07:43 +08:00 · 2024-12-19 03:42:05 +00:00 · 2024-12-19 03:29:34 +00:00 · 2024-12-18 15:15:55 +00:00 · 2024-12-18 12:41:24 +00:00
444 changed files with 17708 additions and 6785 deletions
--- a/.github/actions/setup-greptimedb-cluster/action.yml
+++ b/.github/actions/setup-greptimedb-cluster/action.yml
@@ -8,7 +8,7 @@ inputs:
    default: 2
    description: "Number of Datanode replicas"
  meta-replicas:
-    default: 3
+    default: 1
    description: "Number of Metasrv replicas"
  image-registry: 
    default: "docker.io"
@@ -58,7 +58,7 @@ runs:
        --set image.tag=${{ inputs.image-tag }} \
        --set base.podTemplate.main.resources.requests.cpu=50m \
        --set base.podTemplate.main.resources.requests.memory=256Mi \
-        --set base.podTemplate.main.resources.limits.cpu=1000m \
+        --set base.podTemplate.main.resources.limits.cpu=2000m \
        --set base.podTemplate.main.resources.limits.memory=2Gi \
        --set frontend.replicas=${{ inputs.frontend-replicas }} \
        --set datanode.replicas=${{ inputs.datanode-replicas }} \
--- a/.github/actions/setup-kafka-cluster/action.yml
+++ b/.github/actions/setup-kafka-cluster/action.yml
@@ -18,6 +18,8 @@ runs:
        --set controller.replicaCount=${{ inputs.controller-replicas }} \
        --set controller.resources.requests.cpu=50m \
        --set controller.resources.requests.memory=128Mi \
+        --set controller.resources.limits.cpu=2000m \
+        --set controller.resources.limits.memory=2Gi \
        --set listeners.controller.protocol=PLAINTEXT \
        --set listeners.client.protocol=PLAINTEXT \
        --create-namespace \
--- a/.github/cargo-blacklist.txt
+++ b/.github/cargo-blacklist.txt
@@ -0,0 +1,3 @@
+native-tls
+openssl
+aws-lc-sys
--- a/.github/workflows/dependency-check.yml
+++ b/.github/workflows/dependency-check.yml
@@ -0,0 +1,36 @@
+name: Check Dependencies
+
+on:
+  push:
+    branches:
+      - main
+  pull_request:
+    branches:
+      - main
+
+jobs:
+  check-dependencies:
+    runs-on: ubuntu-latest
+
+    steps:
+    - name: Checkout code
+      uses: actions/checkout@v4
+
+    - name: Set up Rust
+      uses: actions-rust-lang/setup-rust-toolchain@v1
+
+    - name: Run cargo tree
+      run: cargo tree --prefix none > dependencies.txt
+
+    - name: Extract dependency names
+      run: awk '{print $1}' dependencies.txt > dependency_names.txt
+
+    - name: Check for blacklisted crates
+      run: |
+        while read -r dep; do
+          if grep -qFx "$dep" dependency_names.txt; then
+            echo "Blacklisted crate '$dep' found in dependencies."
+            exit 1
+          fi
+        done < .github/cargo-blacklist.txt
+        echo "No blacklisted crates found."
--- a/.github/workflows/develop.yml
+++ b/.github/workflows/develop.yml
@@ -269,13 +269,6 @@ jobs:
      - name: Install cargo-gc-bin
        shell: bash
        run: cargo install cargo-gc-bin
-      - name: Check aws-lc-sys will not build
-        shell: bash
-        run: |
-             if cargo tree -i aws-lc-sys -e features | grep -q aws-lc-sys; then
-               echo "Found aws-lc-sys, which has compilation problems on older gcc versions. Please replace it with ring until its building experience improves."
-               exit 1
-             fi
      - name: Build greptime bianry
        shell: bash
        # `cargo gc` will invoke `cargo build` with specified args
@@ -330,8 +323,6 @@ jobs:
        uses: ./.github/actions/setup-kafka-cluster
      - name: Setup Etcd cluser
        uses: ./.github/actions/setup-etcd-cluster
-      - name: Setup Postgres cluser
-        uses: ./.github/actions/setup-postgres-cluster
      # Prepares for fuzz tests
      - uses: arduino/setup-protoc@v3
        with:
@@ -481,8 +472,6 @@ jobs:
        uses: ./.github/actions/setup-kafka-cluster
      - name: Setup Etcd cluser
        uses: ./.github/actions/setup-etcd-cluster
-      - name: Setup Postgres cluser
-        uses: ./.github/actions/setup-postgres-cluster
      # Prepares for fuzz tests
      - uses: arduino/setup-protoc@v3
        with:
--- a/.github/workflows/nightly-build.yml
+++ b/.github/workflows/nightly-build.yml
@@ -12,7 +12,7 @@ on:
      linux_amd64_runner:
        type: choice
        description: The runner uses to build linux-amd64 artifacts
-        default: ec2-c6i.2xlarge-amd64
+        default: ec2-c6i.4xlarge-amd64
        options:
          - ubuntu-20.04
          - ubuntu-20.04-8-cores
@@ -27,7 +27,7 @@ on:
      linux_arm64_runner:
        type: choice
        description: The runner uses to build linux-arm64 artifacts
-        default: ec2-c6g.2xlarge-arm64
+        default: ec2-c6g.4xlarge-arm64
        options:
          - ec2-c6g.xlarge-arm64 # 4C8G
          - ec2-c6g.2xlarge-arm64 # 8C16G
--- a/.github/workflows/nightly-ci.yml
+++ b/.github/workflows/nightly-ci.yml
@@ -114,6 +114,17 @@ jobs:
          GT_S3_REGION: ${{ vars.AWS_CI_TEST_BUCKET_REGION }}
          UNITTEST_LOG_DIR: "__unittest_logs"

+  cleanbuild-linux-nix:
+    runs-on: ubuntu-latest-8-cores
+    timeout-minutes: 60
+    needs: [coverage, fmt, clippy, check]
+    steps:
+      - uses: actions/checkout@v4
+      - uses: cachix/install-nix-action@v27
+        with:
+          nix_path: nixpkgs=channel:nixos-unstable
+      - run: nix-shell --pure --run "cargo build"
+
  check-status:
    name: Check status
    needs: [sqlness-test, sqlness-windows, test-on-windows]
--- a/.github/workflows/release.yml
+++ b/.github/workflows/release.yml
@@ -91,7 +91,7 @@ env:
  # The scheduled version is '${{ env.NEXT_RELEASE_VERSION }}-nightly-YYYYMMDD', like v0.2.0-nigthly-20230313;
  NIGHTLY_RELEASE_PREFIX: nightly
  # Note: The NEXT_RELEASE_VERSION should be modified manually by every formal release.
-  NEXT_RELEASE_VERSION: v0.11.0
+  NEXT_RELEASE_VERSION: v0.12.0

 # Permission reference: https://docs.github.com/en/actions/using-jobs/assigning-permissions-to-jobs
 permissions:
--- a/.gitignore
+++ b/.gitignore
@@ -47,6 +47,10 @@ benchmarks/data

 venv/

-# Fuzz tests 
+# Fuzz tests
 tests-fuzz/artifacts/
 tests-fuzz/corpus/
+
+# Nix
+.direnv
+.envrc
--- a/AUTHOR.md
+++ b/AUTHOR.md
@@ -7,6 +7,8 @@
 * [NiwakaDev](https://github.com/NiwakaDev)
 * [etolbakov](https://github.com/etolbakov)
 * [irenjj](https://github.com/irenjj)
+* [tisonkun](https://github.com/tisonkun)
+* [Lanqing Yang](https://github.com/lyang24)

 ## Team Members (in alphabetical order)

@@ -30,7 +32,6 @@
 * [shuiyisong](https://github.com/shuiyisong)
 * [sunchanglong](https://github.com/sunchanglong)
 * [sunng87](https://github.com/sunng87)
-* [tisonkun](https://github.com/tisonkun)
 * [v0y4g3r](https://github.com/v0y4g3r)
 * [waynexia](https://github.com/waynexia)
 * [xtang](https://github.com/xtang)
--- a/Cargo.lock
+++ b/Cargo.lock
--- a/Cargo.toml
+++ b/Cargo.toml
@@ -4,6 +4,7 @@ members = [
    "src/auth",
    "src/cache",
    "src/catalog",
+    "src/cli",
    "src/client",
    "src/cmd",
    "src/common/base",
@@ -40,6 +41,7 @@ members = [
    "src/flow",
    "src/frontend",
    "src/index",
+    "src/log-query",
    "src/log-store",
    "src/meta-client",
    "src/meta-srv",
@@ -66,7 +68,7 @@ members = [
 resolver = "2"

 [workspace.package]
-version = "0.10.2"
+version = "0.12.0"
 edition = "2021"
 license = "Apache-2.0"

@@ -167,7 +169,6 @@ rstest = "0.21"
 rstest_reuse = "0.7"
 rust_decimal = "1.33"
 rustc-hash = "2.0"
-schemars = "0.8"
 serde = { version = "1.0", features = ["derive"] }
 serde_json = { version = "1.0", features = ["float_roundtrip"] }
 serde_with = "3"
@@ -179,6 +180,7 @@ sysinfo = "0.30"
 # on branch v0.44.x
 sqlparser = { git = "https://github.com/GreptimeTeam/sqlparser-rs.git", rev = "54a267ac89c09b11c0c88934690530807185d3e7", features = [
    "visitor",
+    "serde",
 ] }
 strum = { version = "0.25", features = ["derive"] }
 tempfile = "3"
@@ -200,6 +202,7 @@ api = { path = "src/api" }
 auth = { path = "src/auth" }
 cache = { path = "src/cache" }
 catalog = { path = "src/catalog" }
+cli = { path = "src/cli" }
 client = { path = "src/client" }
 cmd = { path = "src/cmd", default-features = false }
 common-base = { path = "src/common/base" }
--- a/README.md
+++ b/README.md
@@ -56,7 +56,7 @@
 - [Project Status](#project-status)
 - [Join the community](#community)
  - [Contributing](#contributing)
- [Extension](#extension )
+- [Tools & Extensions](#tools--extensions)
 - [License](#license)
 - [Acknowledgement](#acknowledgement)

@@ -66,31 +66,33 @@

 ## Why GreptimeDB

-Our core developers have been building time-series data platforms for years. Based on our best-practices, GreptimeDB is born to give you:
+Our core developers have been building time-series data platforms for years. Based on our best practices, GreptimeDB was born to give you:

-* **Unified all kinds of time series**
+* **Unified Processing of Metrics, Logs, and Events**

-  GreptimeDB treats all time series as contextual events with timestamp, and thus unifies the processing of metrics, logs, and events. It supports analyzing metrics, logs, and events with SQL and PromQL, and doing streaming with continuous aggregation.
+GreptimeDB unifies time series data processing by treating all data - whether metrics, logs, or events - as timestamped events with context. Users can analyze this data using either [SQL](https://docs.greptime.com/user-guide/query-data/sql) or [PromQL](https://docs.greptime.com/user-guide/query-data/promql) and leverage stream processing ([Flow](https://docs.greptime.com/user-guide/continuous-aggregation/overview)) to enable continuous aggregation. [Read more](https://docs.greptime.com/user-guide/concepts/data-model).

-* **Cloud-Edge collaboration**
+* **Cloud-native Distributed Database**

-  GreptimeDB can be deployed on ARM architecture-compatible Android/Linux systems as well as cloud environments from various vendors. Both sides run the same software, providing identical APIs and control planes, so your application can run at the edge or on the cloud without modification, and data synchronization also becomes extremely easy and efficient.
-
-* **Cloud-native distributed database**
-
-  By leveraging object storage (S3 and others), separating compute and storage, scaling stateless compute nodes arbitrarily, GreptimeDB implements seamless scalability. It also supports cross-cloud deployment with a built-in unified data access layer over different object storages.
+Built for [Kubernetes](https://docs.greptime.com/user-guide/deployments/deploy-on-kubernetes/greptimedb-operator-management). GreptimeDB achieves seamless scalability with its [cloud-native architecture](https://docs.greptime.com/user-guide/concepts/architecture) of separated compute and storage, built on object storage (AWS S3, Azure Blob Storage, etc.) while enabling cross-cloud deployment through a unified data access layer.

 * **Performance and Cost-effective**

-  Flexible indexing capabilities and distributed, parallel-processing query engine, tackling high cardinality issues down. Optimized columnar layout for handling time-series data; compacted, compressed, and stored on various storage backends, particularly cloud object storage with 50x cost efficiency.
+Written in pure Rust for superior performance and reliability. GreptimeDB features a distributed query engine with intelligent indexing to handle high cardinality data efficiently. Its optimized columnar storage achieves 50x cost efficiency on cloud object storage through advanced compression. [Benchmark reports](https://www.greptime.com/blogs/2024-09-09-report-summary).

-* **Compatible with InfluxDB, Prometheus and more protocols**
+* **Cloud-Edge Collaboration**

-  Widely adopted database protocols and APIs, including MySQL, PostgreSQL, and Prometheus Remote Storage, etc. [Read more](https://docs.greptime.com/user-guide/protocols/overview).
+GreptimeDB seamlessly operates across cloud and edge (ARM/Android/Linux), providing consistent APIs and control plane for unified data management and efficient synchronization. [Learn how to run on Android](https://docs.greptime.com/user-guide/deployments/run-on-android/).
+
+* **Multi-protocol Ingestion, SQL & PromQL Ready**
+
+Widely adopted database protocols and APIs, including MySQL, PostgreSQL, InfluxDB, OpenTelemetry, Loki and Prometheus, etc.  Effortless Adoption & Seamless Migration. [Supported Protocols Overview](https://docs.greptime.com/user-guide/protocols/overview).
+
+For more detailed info please read  [Why GreptimeDB](https://docs.greptime.com/user-guide/concepts/why-greptimedb).

 ## Try GreptimeDB

-### 1. [GreptimePlay](https://greptime.com/playground)
+### 1. [Live Demo](https://greptime.com/playground)

 Try out the features of GreptimeDB right from your browser.

@@ -109,9 +111,18 @@ docker pull greptime/greptimedb
 Start a GreptimeDB container with:

 ```shell
-docker run --rm --name greptime --net=host greptime/greptimedb standalone start
+docker run -p 127.0.0.1:4000-4003:4000-4003 \
+  -v "$(pwd)/greptimedb:/tmp/greptimedb" \
+  --name greptime --rm \
+  greptime/greptimedb:latest standalone start \
+  --http-addr 0.0.0.0:4000 \
+  --rpc-addr 0.0.0.0:4001 \
+  --mysql-addr 0.0.0.0:4002 \
+  --postgres-addr 0.0.0.0:4003
 ```

+Access the dashboard via `http://localhost:4000/dashboard`.
+
 Read more about [Installation](https://docs.greptime.com/getting-started/installation/overview) on docs.

 ## Getting Started
@@ -141,7 +152,7 @@ Run a standalone server:
 cargo run -- standalone start
 ```

-## Extension
+## Tools & Extensions

 ### Dashboard

@@ -158,14 +169,19 @@ cargo run -- standalone start

 ### Grafana Dashboard

-Our official Grafana dashboard is available at [grafana](grafana/README.md) directory.
+Our official Grafana dashboard for monitoring GreptimeDB is available at [grafana](grafana/README.md) directory.

 ## Project Status

-The current version has not yet reached the standards for General Availability.
-According to our Greptime 2024 Roadmap, we aim to achieve a production-level version with the release of v1.0 by the end of 2024. [Join Us](https://github.com/GreptimeTeam/greptimedb/issues/3412)
+GreptimeDB is currently in Beta. We are targeting GA (General Availability) with v1.0 release by Early 2025. 

-We welcome you to test and use GreptimeDB. Some users have already adopted it in their production environments. If you're interested in trying it out, please use the latest stable release available.
+While in Beta, GreptimeDB is already:
+
+* Being used in production by early adopters
+* Actively maintained with regular releases, [about version number](https://docs.greptime.com/nightly/reference/about-greptimedb-version)
+* Suitable for testing and evaluation
+
+For production use, we recommend using the latest stable release.

 ## Community

@@ -184,12 +200,12 @@ In addition, you may:
 - Connect us with [Linkedin](https://www.linkedin.com/company/greptime/)
 - Follow us on [Twitter](https://twitter.com/greptime)

-## Commerial Support
+## Commercial Support

 If you are running GreptimeDB OSS in your organization, we offer additional
-enterprise addons, installation service, training and consulting. [Contact
+enterprise add-ons, installation services, training, and consulting. [Contact
 us](https://greptime.com/contactus) and we will reach out to you with more
-detail of our commerial license.
+detail of our commercial license.

 ## License

--- a/config/config.md
+++ b/config/config.md
@@ -13,11 +13,11 @@
 | Key | Type | Default | Descriptions |
 | --- | -----| ------- | ----------- |
 | `mode` | String | `standalone` | The running mode of the datanode. It can be `standalone` or `distributed`. |
-| `enable_telemetry` | Bool | `true` | Enable telemetry to collect anonymous usage data. |
 | `default_timezone` | String | Unset | The default timezone of the server. |
 | `init_regions_in_background` | Bool | `false` | Initialize all regions in the background during the startup.<br/>By default, it provides services after all regions have been initialized. |
 | `init_regions_parallelism` | Integer | `16` | Parallelism of initializing regions. |
 | `max_concurrent_queries` | Integer | `0` | The maximum current queries allowed to be executed. Zero means unlimited. |
+| `enable_telemetry` | Bool | `true` | Enable telemetry to collect anonymous usage data. Enabled by default. |
 | `runtime` | -- | -- | The runtime options. |
 | `runtime.global_rt_size` | Integer | `8` | The number of threads to execute the runtime for global read operations. |
 | `runtime.compact_rt_size` | Integer | `4` | The number of threads to execute the runtime for global write operations. |
@@ -61,9 +61,9 @@
 | `wal` | -- | -- | The WAL options. |
 | `wal.provider` | String | `raft_engine` | The provider of the WAL.<br/>- `raft_engine`: the wal is stored in the local file system by raft-engine.<br/>- `kafka`: it's remote wal that data is stored in Kafka. |
 | `wal.dir` | String | Unset | The directory to store the WAL files.<br/>**It's only used when the provider is `raft_engine`**. |
-| `wal.file_size` | String | `256MB` | The size of the WAL segment file.<br/>**It's only used when the provider is `raft_engine`**. |
-| `wal.purge_threshold` | String | `4GB` | The threshold of the WAL size to trigger a flush.<br/>**It's only used when the provider is `raft_engine`**. |
-| `wal.purge_interval` | String | `10m` | The interval to trigger a flush.<br/>**It's only used when the provider is `raft_engine`**. |
+| `wal.file_size` | String | `128MB` | The size of the WAL segment file.<br/>**It's only used when the provider is `raft_engine`**. |
+| `wal.purge_threshold` | String | `1GB` | The threshold of the WAL size to trigger a flush.<br/>**It's only used when the provider is `raft_engine`**. |
+| `wal.purge_interval` | String | `1m` | The interval to trigger a flush.<br/>**It's only used when the provider is `raft_engine`**. |
 | `wal.read_batch_size` | Integer | `128` | The read batch size.<br/>**It's only used when the provider is `raft_engine`**. |
 | `wal.sync_write` | Bool | `false` | Whether to use sync write.<br/>**It's only used when the provider is `raft_engine`**. |
 | `wal.enable_log_recycle` | Bool | `true` | Whether to reuse logically truncated log files.<br/>**It's only used when the provider is `raft_engine`**. |
@@ -93,7 +93,7 @@
 | `storage` | -- | -- | The data storage options. |
 | `storage.data_home` | String | `/tmp/greptimedb/` | The working home directory. |
 | `storage.type` | String | `File` | The storage type used to store the data.<br/>- `File`: the data is stored in the local file system.<br/>- `S3`: the data is stored in the S3 object storage.<br/>- `Gcs`: the data is stored in the Google Cloud Storage.<br/>- `Azblob`: the data is stored in the Azure Blob Storage.<br/>- `Oss`: the data is stored in the Aliyun OSS. |
-| `storage.cache_path` | String | Unset | Cache configuration for object storage such as 'S3' etc. It is recommended to configure it when using object storage for better performance.<br/>The local file cache directory. |
+| `storage.cache_path` | String | Unset | Read cache configuration for object storage such as 'S3' etc, it's configured by default when using object storage. It is recommended to configure it when using object storage for better performance.<br/>A local file directory, defaults to `{data_home}/object_cache/read`. An empty string means disabling. |
 | `storage.cache_capacity` | String | Unset | The local file cache capacity in bytes. If your disk space is sufficient, it is recommended to set it larger. |
 | `storage.bucket` | String | Unset | The S3 bucket name.<br/>**It's only used when the storage type is `S3`, `Oss` and `Gcs`**. |
 | `storage.root` | String | Unset | The S3 data will be stored in the specified prefix, for example, `s3://${bucket}/${root}`.<br/>**It's only used when the storage type is `S3`, `Oss` and `Azblob`**. |
@@ -131,12 +131,11 @@
 | `region_engine.mito.vector_cache_size` | String | Auto | Cache size for vectors and arrow arrays. Setting it to 0 to disable the cache.<br/>If not set, it's default to 1/16 of OS memory with a max limitation of 512MB. |
 | `region_engine.mito.page_cache_size` | String | Auto | Cache size for pages of SST row groups. Setting it to 0 to disable the cache.<br/>If not set, it's default to 1/8 of OS memory. |
 | `region_engine.mito.selector_result_cache_size` | String | Auto | Cache size for time series selector (e.g. `last_value()`). Setting it to 0 to disable the cache.<br/>If not set, it's default to 1/16 of OS memory with a max limitation of 512MB. |
-| `region_engine.mito.enable_experimental_write_cache` | Bool | `false` | Whether to enable the experimental write cache. It is recommended to enable it when using object storage for better performance. |
-| `region_engine.mito.experimental_write_cache_path` | String | `""` | File system path for write cache, defaults to `{data_home}/write_cache`. |
-| `region_engine.mito.experimental_write_cache_size` | String | `1GiB` | Capacity for write cache. If your disk space is sufficient, it is recommended to set it larger. |
+| `region_engine.mito.enable_experimental_write_cache` | Bool | `false` | Whether to enable the experimental write cache, it's enabled by default when using object storage. It is recommended to enable it when using object storage for better performance. |
+| `region_engine.mito.experimental_write_cache_path` | String | `""` | File system path for write cache, defaults to `{data_home}/object_cache/write`. |
+| `region_engine.mito.experimental_write_cache_size` | String | `5GiB` | Capacity for write cache. If your disk space is sufficient, it is recommended to set it larger. |
 | `region_engine.mito.experimental_write_cache_ttl` | String | Unset | TTL for write cache. |
 | `region_engine.mito.sst_write_buffer_size` | String | `8MB` | Buffer size for SST writing. |
-| `region_engine.mito.scan_parallelism` | Integer | `0` | Parallelism to scan a region (default: 1/4 of cpu cores).<br/>- `0`: using the default value (1/4 of cpu cores).<br/>- `1`: scan in current thread.<br/>- `n`: scan in parallelism n. |
 | `region_engine.mito.parallel_scan_channel_size` | Integer | `32` | Capacity of the channel to send data from parallel scan tasks to the main task. |
 | `region_engine.mito.allow_stale_entries` | Bool | `false` | Whether to allow stale WAL entries read during replay. |
 | `region_engine.mito.min_compaction_interval` | String | `0m` | Minimum time interval between two compactions.<br/>To align with the old behavior, the default value is 0 (no restrictions). |
@@ -151,6 +150,7 @@
 | `region_engine.mito.inverted_index.intermediate_path` | String | `""` | Deprecated, use `region_engine.mito.index.aux_path` instead. |
 | `region_engine.mito.inverted_index.metadata_cache_size` | String | `64MiB` | Cache size for inverted index metadata. |
 | `region_engine.mito.inverted_index.content_cache_size` | String | `128MiB` | Cache size for inverted index content. |
+| `region_engine.mito.inverted_index.content_cache_page_size` | String | `8MiB` | Page size for inverted index content cache. |
 | `region_engine.mito.fulltext_index` | -- | -- | The options for full-text index in Mito engine. |
 | `region_engine.mito.fulltext_index.create_on_flush` | String | `auto` | Whether to create the index on flush.<br/>- `auto`: automatically (default)<br/>- `disable`: never |
 | `region_engine.mito.fulltext_index.create_on_compaction` | String | `auto` | Whether to create the index on compaction.<br/>- `auto`: automatically (default)<br/>- `disable`: never |
@@ -286,13 +286,13 @@
 | `data_home` | String | `/tmp/metasrv/` | The working home directory. |
 | `bind_addr` | String | `127.0.0.1:3002` | The bind address of metasrv. |
 | `server_addr` | String | `127.0.0.1:3002` | The communication server address for frontend and datanode to connect to metasrv,  "127.0.0.1:3002" by default for localhost. |
-| `store_addr` | String | `127.0.0.1:2379` | Store server address default to etcd store. |
+| `store_addrs` | Array | -- | Store server address default to etcd store. |
+| `store_key_prefix` | String | `""` | If it's not empty, the metasrv will store all data with this key prefix. |
+| `backend` | String | `EtcdStore` | The datastore for meta server. |
 | `selector` | String | `round_robin` | Datanode selector type.<br/>- `round_robin` (default value)<br/>- `lease_based`<br/>- `load_based`<br/>For details, please see "https://docs.greptime.com/developer-guide/metasrv/selector". |
 | `use_memory_store` | Bool | `false` | Store data in memory. |
-| `enable_telemetry` | Bool | `true` | Whether to enable greptimedb telemetry. |
-| `store_key_prefix` | String | `""` | If it's not empty, the metasrv will store all data with this key prefix. |
 | `enable_region_failover` | Bool | `false` | Whether to enable region failover.<br/>This feature is only available on GreptimeDB running on cluster mode and<br/>- Using Remote WAL<br/>- Using shared storage (e.g., s3). |
-| `backend` | String | `EtcdStore` | The datastore for meta server. |
+| `enable_telemetry` | Bool | `true` | Whether to enable greptimedb telemetry. Enabled by default. |
 | `runtime` | -- | -- | The runtime options. |
 | `runtime.global_rt_size` | Integer | `8` | The number of threads to execute the runtime for global read operations. |
 | `runtime.compact_rt_size` | Integer | `4` | The number of threads to execute the runtime for global write operations. |
@@ -357,7 +357,6 @@
 | `node_id` | Integer | Unset | The datanode identifier and should be unique in the cluster. |
 | `require_lease_before_startup` | Bool | `false` | Start services after regions have obtained leases.<br/>It will block the datanode start if it can't receive leases in the heartbeat from metasrv. |
 | `init_regions_in_background` | Bool | `false` | Initialize all regions in the background during the startup.<br/>By default, it provides services after all regions have been initialized. |
-| `enable_telemetry` | Bool | `true` | Enable telemetry to collect anonymous usage data. |
 | `init_regions_parallelism` | Integer | `16` | Parallelism of initializing regions. |
 | `max_concurrent_queries` | Integer | `0` | The maximum current queries allowed to be executed. Zero means unlimited. |
 | `rpc_addr` | String | Unset | Deprecated, use `grpc.addr` instead. |
@@ -365,6 +364,7 @@
 | `rpc_runtime_size` | Integer | Unset | Deprecated, use `grpc.runtime_size` instead. |
 | `rpc_max_recv_message_size` | String | Unset | Deprecated, use `grpc.rpc_max_recv_message_size` instead. |
 | `rpc_max_send_message_size` | String | Unset | Deprecated, use `grpc.rpc_max_send_message_size` instead. |
+| `enable_telemetry` | Bool | `true` | Enable telemetry to collect anonymous usage data. Enabled by default. |
 | `http` | -- | -- | The HTTP server options. |
 | `http.addr` | String | `127.0.0.1:4000` | The address to bind the HTTP server. |
 | `http.timeout` | String | `30s` | HTTP request timeout. Set to 0 to disable timeout. |
@@ -399,9 +399,9 @@
 | `wal` | -- | -- | The WAL options. |
 | `wal.provider` | String | `raft_engine` | The provider of the WAL.<br/>- `raft_engine`: the wal is stored in the local file system by raft-engine.<br/>- `kafka`: it's remote wal that data is stored in Kafka. |
 | `wal.dir` | String | Unset | The directory to store the WAL files.<br/>**It's only used when the provider is `raft_engine`**. |
-| `wal.file_size` | String | `256MB` | The size of the WAL segment file.<br/>**It's only used when the provider is `raft_engine`**. |
-| `wal.purge_threshold` | String | `4GB` | The threshold of the WAL size to trigger a flush.<br/>**It's only used when the provider is `raft_engine`**. |
-| `wal.purge_interval` | String | `10m` | The interval to trigger a flush.<br/>**It's only used when the provider is `raft_engine`**. |
+| `wal.file_size` | String | `128MB` | The size of the WAL segment file.<br/>**It's only used when the provider is `raft_engine`**. |
+| `wal.purge_threshold` | String | `1GB` | The threshold of the WAL size to trigger a flush.<br/>**It's only used when the provider is `raft_engine`**. |
+| `wal.purge_interval` | String | `1m` | The interval to trigger a flush.<br/>**It's only used when the provider is `raft_engine`**. |
 | `wal.read_batch_size` | Integer | `128` | The read batch size.<br/>**It's only used when the provider is `raft_engine`**. |
 | `wal.sync_write` | Bool | `false` | Whether to use sync write.<br/>**It's only used when the provider is `raft_engine`**. |
 | `wal.enable_log_recycle` | Bool | `true` | Whether to reuse logically truncated log files.<br/>**It's only used when the provider is `raft_engine`**. |
@@ -421,7 +421,7 @@
 | `storage` | -- | -- | The data storage options. |
 | `storage.data_home` | String | `/tmp/greptimedb/` | The working home directory. |
 | `storage.type` | String | `File` | The storage type used to store the data.<br/>- `File`: the data is stored in the local file system.<br/>- `S3`: the data is stored in the S3 object storage.<br/>- `Gcs`: the data is stored in the Google Cloud Storage.<br/>- `Azblob`: the data is stored in the Azure Blob Storage.<br/>- `Oss`: the data is stored in the Aliyun OSS. |
-| `storage.cache_path` | String | Unset | Cache configuration for object storage such as 'S3' etc. It is recommended to configure it when using object storage for better performance.<br/>The local file cache directory. |
+| `storage.cache_path` | String | Unset | Read cache configuration for object storage such as 'S3' etc, it's configured by default when using object storage. It is recommended to configure it when using object storage for better performance.<br/>A local file directory, defaults to `{data_home}/object_cache/read`. An empty string means disabling. |
 | `storage.cache_capacity` | String | Unset | The local file cache capacity in bytes. If your disk space is sufficient, it is recommended to set it larger. |
 | `storage.bucket` | String | Unset | The S3 bucket name.<br/>**It's only used when the storage type is `S3`, `Oss` and `Gcs`**. |
 | `storage.root` | String | Unset | The S3 data will be stored in the specified prefix, for example, `s3://${bucket}/${root}`.<br/>**It's only used when the storage type is `S3`, `Oss` and `Azblob`**. |
@@ -459,12 +459,11 @@
 | `region_engine.mito.vector_cache_size` | String | Auto | Cache size for vectors and arrow arrays. Setting it to 0 to disable the cache.<br/>If not set, it's default to 1/16 of OS memory with a max limitation of 512MB. |
 | `region_engine.mito.page_cache_size` | String | Auto | Cache size for pages of SST row groups. Setting it to 0 to disable the cache.<br/>If not set, it's default to 1/8 of OS memory. |
 | `region_engine.mito.selector_result_cache_size` | String | Auto | Cache size for time series selector (e.g. `last_value()`). Setting it to 0 to disable the cache.<br/>If not set, it's default to 1/16 of OS memory with a max limitation of 512MB. |
-| `region_engine.mito.enable_experimental_write_cache` | Bool | `false` | Whether to enable the experimental write cache. It is recommended to enable it when using object storage for better performance. |
-| `region_engine.mito.experimental_write_cache_path` | String | `""` | File system path for write cache, defaults to `{data_home}/write_cache`. |
-| `region_engine.mito.experimental_write_cache_size` | String | `1GiB` | Capacity for write cache. If your disk space is sufficient, it is recommended to set it larger. |
+| `region_engine.mito.enable_experimental_write_cache` | Bool | `false` | Whether to enable the experimental write cache, it's enabled by default when using object storage. It is recommended to enable it when using object storage for better performance. |
+| `region_engine.mito.experimental_write_cache_path` | String | `""` | File system path for write cache, defaults to `{data_home}/object_cache/write`. |
+| `region_engine.mito.experimental_write_cache_size` | String | `5GiB` | Capacity for write cache. If your disk space is sufficient, it is recommended to set it larger. |
 | `region_engine.mito.experimental_write_cache_ttl` | String | Unset | TTL for write cache. |
 | `region_engine.mito.sst_write_buffer_size` | String | `8MB` | Buffer size for SST writing. |
-| `region_engine.mito.scan_parallelism` | Integer | `0` | Parallelism to scan a region (default: 1/4 of cpu cores).<br/>- `0`: using the default value (1/4 of cpu cores).<br/>- `1`: scan in current thread.<br/>- `n`: scan in parallelism n. |
 | `region_engine.mito.parallel_scan_channel_size` | Integer | `32` | Capacity of the channel to send data from parallel scan tasks to the main task. |
 | `region_engine.mito.allow_stale_entries` | Bool | `false` | Whether to allow stale WAL entries read during replay. |
 | `region_engine.mito.min_compaction_interval` | String | `0m` | Minimum time interval between two compactions.<br/>To align with the old behavior, the default value is 0 (no restrictions). |
@@ -477,6 +476,9 @@
 | `region_engine.mito.inverted_index.apply_on_query` | String | `auto` | Whether to apply the index on query<br/>- `auto`: automatically (default)<br/>- `disable`: never |
 | `region_engine.mito.inverted_index.mem_threshold_on_create` | String | `auto` | Memory threshold for performing an external sort during index creation.<br/>- `auto`: automatically determine the threshold based on the system memory size (default)<br/>- `unlimited`: no memory limit<br/>- `[size]` e.g. `64MB`: fixed memory threshold |
 | `region_engine.mito.inverted_index.intermediate_path` | String | `""` | Deprecated, use `region_engine.mito.index.aux_path` instead. |
+| `region_engine.mito.inverted_index.metadata_cache_size` | String | `64MiB` | Cache size for inverted index metadata. |
+| `region_engine.mito.inverted_index.content_cache_size` | String | `128MiB` | Cache size for inverted index content. |
+| `region_engine.mito.inverted_index.content_cache_page_size` | String | `8MiB` | Page size for inverted index content cache. |
 | `region_engine.mito.fulltext_index` | -- | -- | The options for full-text index in Mito engine. |
 | `region_engine.mito.fulltext_index.create_on_flush` | String | `auto` | Whether to create the index on flush.<br/>- `auto`: automatically (default)<br/>- `disable`: never |
 | `region_engine.mito.fulltext_index.create_on_compaction` | String | `auto` | Whether to create the index on compaction.<br/>- `auto`: automatically (default)<br/>- `disable`: never |
--- a/config/datanode.example.toml
+++ b/config/datanode.example.toml
@@ -13,9 +13,6 @@ require_lease_before_startup = false
 ## By default, it provides services after all regions have been initialized.
 init_regions_in_background = false

-## Enable telemetry to collect anonymous usage data.
-enable_telemetry = true
-
 ## Parallelism of initializing regions.
 init_regions_parallelism = 16

@@ -42,6 +39,8 @@ rpc_max_recv_message_size = "512MB"
 ## @toml2docs:none-default
 rpc_max_send_message_size = "512MB"

+## Enable telemetry to collect anonymous usage data. Enabled by default.
+#+ enable_telemetry = true

 ## The HTTP server options.
 [http]
@@ -143,15 +142,15 @@ dir = "/tmp/greptimedb/wal"

 ## The size of the WAL segment file.
 ## **It's only used when the provider is `raft_engine`**.
-file_size = "256MB"
+file_size = "128MB"

 ## The threshold of the WAL size to trigger a flush.
 ## **It's only used when the provider is `raft_engine`**.
-purge_threshold = "4GB"
+purge_threshold = "1GB"

 ## The interval to trigger a flush.
 ## **It's only used when the provider is `raft_engine`**.
-purge_interval = "10m"
+purge_interval = "1m"

 ## The read batch size.
 ## **It's only used when the provider is `raft_engine`**.
@@ -294,14 +293,14 @@ data_home = "/tmp/greptimedb/"
 ## - `Oss`: the data is stored in the Aliyun OSS.
 type = "File"

-## Cache configuration for object storage such as 'S3' etc. It is recommended to configure it when using object storage for better performance.
-## The local file cache directory.
+## Read cache configuration for object storage such as 'S3' etc, it's configured by default when using object storage. It is recommended to configure it when using object storage for better performance.
+## A local file directory, defaults to `{data_home}/object_cache/read`. An empty string means disabling.
 ## @toml2docs:none-default
-cache_path = "/path/local_cache"
+#+ cache_path = ""

 ## The local file cache capacity in bytes. If your disk space is sufficient, it is recommended to set it larger.
 ## @toml2docs:none-default
-cache_capacity = "1GiB"
+cache_capacity = "5GiB"

 ## The S3 bucket name.
 ## **It's only used when the storage type is `S3`, `Oss` and `Gcs`**.
@@ -476,14 +475,14 @@ auto_flush_interval = "1h"
 ## @toml2docs:none-default="Auto"
 #+ selector_result_cache_size = "512MB"

-## Whether to enable the experimental write cache. It is recommended to enable it when using object storage for better performance.
+## Whether to enable the experimental write cache, it's enabled by default when using object storage. It is recommended to enable it when using object storage for better performance.
 enable_experimental_write_cache = false

-## File system path for write cache, defaults to `{data_home}/write_cache`.
+## File system path for write cache, defaults to `{data_home}/object_cache/write`.
 experimental_write_cache_path = ""

 ## Capacity for write cache. If your disk space is sufficient, it is recommended to set it larger.
-experimental_write_cache_size = "1GiB"
+experimental_write_cache_size = "5GiB"

 ## TTL for write cache.
 ## @toml2docs:none-default
@@ -492,12 +491,6 @@ experimental_write_cache_ttl = "8h"
 ## Buffer size for SST writing.
 sst_write_buffer_size = "8MB"

-## Parallelism to scan a region (default: 1/4 of cpu cores).
-## - `0`: using the default value (1/4 of cpu cores).
-## - `1`: scan in current thread.
-## - `n`: scan in parallelism n.
-scan_parallelism = 0
-
 ## Capacity of the channel to send data from parallel scan tasks to the main task.
 parallel_scan_channel_size = 32

@@ -550,6 +543,15 @@ mem_threshold_on_create = "auto"
 ## Deprecated, use `region_engine.mito.index.aux_path` instead.
 intermediate_path = ""

+## Cache size for inverted index metadata.
+metadata_cache_size = "64MiB"
+
+## Cache size for inverted index content.
+content_cache_size = "128MiB"
+
+## Page size for inverted index content cache.
+content_cache_page_size = "8MiB"
+
 ## The options for full-text index in Mito engine.
 [region_engine.mito.fulltext_index]

--- a/config/metasrv.example.toml
+++ b/config/metasrv.example.toml
@@ -8,7 +8,13 @@ bind_addr = "127.0.0.1:3002"
 server_addr = "127.0.0.1:3002"

 ## Store server address default to etcd store.
-store_addr = "127.0.0.1:2379"
+store_addrs = ["127.0.0.1:2379"]
+
+## If it's not empty, the metasrv will store all data with this key prefix.
+store_key_prefix = ""
+
+## The datastore for meta server.
+backend = "EtcdStore"

 ## Datanode selector type.
 ## - `round_robin` (default value)
@@ -20,20 +26,14 @@ selector = "round_robin"
 ## Store data in memory.
 use_memory_store = false

-## Whether to enable greptimedb telemetry.
-enable_telemetry = true
-
-## If it's not empty, the metasrv will store all data with this key prefix.
-store_key_prefix = ""
-
 ## Whether to enable region failover.
 ## This feature is only available on GreptimeDB running on cluster mode and
 ## - Using Remote WAL
 ## - Using shared storage (e.g., s3).
 enable_region_failover = false

-## The datastore for meta server.
-backend = "EtcdStore"
+## Whether to enable greptimedb telemetry. Enabled by default.
+#+ enable_telemetry = true

 ## The runtime options.
 #+ [runtime]
--- a/config/standalone.example.toml
+++ b/config/standalone.example.toml
@@ -1,9 +1,6 @@
 ## The running mode of the datanode. It can be `standalone` or `distributed`.
 mode = "standalone"

-## Enable telemetry to collect anonymous usage data.
-enable_telemetry = true
-
 ## The default timezone of the server.
 ## @toml2docs:none-default
 default_timezone = "UTC"
@@ -18,6 +15,9 @@ init_regions_parallelism = 16
 ## The maximum current queries allowed to be executed. Zero means unlimited.
 max_concurrent_queries = 0

+## Enable telemetry to collect anonymous usage data. Enabled by default.
+#+ enable_telemetry = true
+
 ## The runtime options.
 #+ [runtime]
 ## The number of threads to execute the runtime for global read operations.
@@ -147,15 +147,15 @@ dir = "/tmp/greptimedb/wal"

 ## The size of the WAL segment file.
 ## **It's only used when the provider is `raft_engine`**.
-file_size = "256MB"
+file_size = "128MB"

 ## The threshold of the WAL size to trigger a flush.
 ## **It's only used when the provider is `raft_engine`**.
-purge_threshold = "4GB"
+purge_threshold = "1GB"

 ## The interval to trigger a flush.
 ## **It's only used when the provider is `raft_engine`**.
-purge_interval = "10m"
+purge_interval = "1m"

 ## The read batch size.
 ## **It's only used when the provider is `raft_engine`**.
@@ -332,14 +332,14 @@ data_home = "/tmp/greptimedb/"
 ## - `Oss`: the data is stored in the Aliyun OSS.
 type = "File"

-## Cache configuration for object storage such as 'S3' etc. It is recommended to configure it when using object storage for better performance.
-## The local file cache directory.
+## Read cache configuration for object storage such as 'S3' etc, it's configured by default when using object storage. It is recommended to configure it when using object storage for better performance.
+## A local file directory, defaults to `{data_home}/object_cache/read`. An empty string means disabling.
 ## @toml2docs:none-default
-cache_path = "/path/local_cache"
+#+ cache_path = ""

 ## The local file cache capacity in bytes. If your disk space is sufficient, it is recommended to set it larger.
 ## @toml2docs:none-default
-cache_capacity = "1GiB"
+cache_capacity = "5GiB"

 ## The S3 bucket name.
 ## **It's only used when the storage type is `S3`, `Oss` and `Gcs`**.
@@ -514,14 +514,14 @@ auto_flush_interval = "1h"
 ## @toml2docs:none-default="Auto"
 #+ selector_result_cache_size = "512MB"

-## Whether to enable the experimental write cache. It is recommended to enable it when using object storage for better performance.
+## Whether to enable the experimental write cache, it's enabled by default when using object storage. It is recommended to enable it when using object storage for better performance.
 enable_experimental_write_cache = false

-## File system path for write cache, defaults to `{data_home}/write_cache`.
+## File system path for write cache, defaults to `{data_home}/object_cache/write`.
 experimental_write_cache_path = ""

 ## Capacity for write cache. If your disk space is sufficient, it is recommended to set it larger.
-experimental_write_cache_size = "1GiB"
+experimental_write_cache_size = "5GiB"

 ## TTL for write cache.
 ## @toml2docs:none-default
@@ -530,12 +530,6 @@ experimental_write_cache_ttl = "8h"
 ## Buffer size for SST writing.
 sst_write_buffer_size = "8MB"

-## Parallelism to scan a region (default: 1/4 of cpu cores).
-## - `0`: using the default value (1/4 of cpu cores).
-## - `1`: scan in current thread.
-## - `n`: scan in parallelism n.
-scan_parallelism = 0
-
 ## Capacity of the channel to send data from parallel scan tasks to the main task.
 parallel_scan_channel_size = 32

@@ -594,6 +588,9 @@ metadata_cache_size = "64MiB"
 ## Cache size for inverted index content.
 content_cache_size = "128MiB"

+## Page size for inverted index content cache.
+content_cache_page_size = "8MiB"
+
 ## The options for full-text index in Mito engine.
 [region_engine.mito.fulltext_index]

--- a/docs/how-to/how-to-profile-cpu.md
+++ b/docs/how-to/how-to-profile-cpu.md
@@ -3,7 +3,7 @@
 ## HTTP API
 Sample at 99 Hertz, for 5 seconds, output report in [protobuf format](https://github.com/google/pprof/blob/master/proto/profile.proto).
 ```bash
-curl -s '0:4000/debug/prof/cpu' > /tmp/pprof.out
+curl -X POST -s '0:4000/debug/prof/cpu' > /tmp/pprof.out
 ```

 Then you can use `pprof` command with the protobuf file.
@@ -13,10 +13,10 @@ go tool pprof -top /tmp/pprof.out

 Sample at 99 Hertz, for 60 seconds, output report in flamegraph format.
 ```bash
-curl -s '0:4000/debug/prof/cpu?seconds=60&output=flamegraph' > /tmp/pprof.svg
+curl -X POST -s '0:4000/debug/prof/cpu?seconds=60&output=flamegraph' > /tmp/pprof.svg
 ```

 Sample at 49 Hertz, for 10 seconds, output report in text format.
 ```bash
-curl -s '0:4000/debug/prof/cpu?seconds=10&frequency=49&output=text' > /tmp/pprof.txt
+curl -X POST -s '0:4000/debug/prof/cpu?seconds=10&frequency=49&output=text' > /tmp/pprof.txt
 ```
--- a/docs/how-to/how-to-profile-memory.md
+++ b/docs/how-to/how-to-profile-memory.md
@@ -23,13 +23,13 @@ curl https://raw.githubusercontent.com/brendangregg/FlameGraph/master/flamegraph
 Start GreptimeDB instance with environment variables:

 ```bash
-MALLOC_CONF=prof:true,lg_prof_interval:28 ./target/debug/greptime standalone start
+MALLOC_CONF=prof:true ./target/debug/greptime standalone start
 ```

 Dump memory profiling data through HTTP API:

 ```bash
-curl localhost:4000/debug/prof/mem > greptime.hprof
+curl -X POST localhost:4000/debug/prof/mem > greptime.hprof
 ```

 You can periodically dump profiling data and compare them to find the delta memory usage.
--- a/grafana/greptimedb.json
+++ b/grafana/greptimedb.json
--- a/rust-toolchain.toml
+++ b/rust-toolchain.toml
@@ -1,2 +1,3 @@
 [toolchain]
 channel = "nightly-2024-10-19"
+components = ["rust-analyzer"]
--- a/scripts/check-snafu.py
+++ b/scripts/check-snafu.py
@@ -58,8 +58,10 @@ def main():
        if not check_snafu_in_files(branch_name, other_rust_files)
    ]

-    for name in unused_snafu:
-        print(name)
+    if unused_snafu:
+        print("Unused error variants:")
+        for name in unused_snafu:
+            print(name)

    if unused_snafu:
        raise SystemExit(1)
--- a/shell.nix
+++ b/shell.nix
@@ -0,0 +1,27 @@
+let
+  nixpkgs = fetchTarball "https://github.com/NixOS/nixpkgs/tarball/nixos-unstable";
+  fenix = import (fetchTarball "https://github.com/nix-community/fenix/archive/main.tar.gz") {};
+  pkgs = import nixpkgs { config = {}; overlays = []; };
+in
+
+pkgs.mkShell rec {
+  nativeBuildInputs = with pkgs; [
+    pkg-config
+    git
+    clang
+    gcc
+    protobuf
+    mold
+    (fenix.fromToolchainFile {
+      dir = ./.;
+    })
+    cargo-nextest
+    taplo
+  ];
+
+  buildInputs = with pkgs; [
+    libgit2
+  ];
+
+  LD_LIBRARY_PATH = pkgs.lib.makeLibraryPath buildInputs;
+}
--- a/src/api/src/v1/column_def.rs
+++ b/src/api/src/v1/column_def.rs
@@ -16,7 +16,7 @@ use std::collections::HashMap;

 use datatypes::schema::{
    ColumnDefaultConstraint, ColumnSchema, FulltextAnalyzer, FulltextOptions, COMMENT_KEY,
-    FULLTEXT_KEY, INVERTED_INDEX_KEY,
+    FULLTEXT_KEY, INVERTED_INDEX_KEY, SKIPPING_INDEX_KEY,
 };
 use greptime_proto::v1::Analyzer;
 use snafu::ResultExt;
@@ -29,6 +29,8 @@ use crate::v1::{ColumnDef, ColumnOptions, SemanticType};
 const FULLTEXT_GRPC_KEY: &str = "fulltext";
 /// Key used to store inverted index options in gRPC column options.
 const INVERTED_INDEX_GRPC_KEY: &str = "inverted_index";
+/// Key used to store skip index options in gRPC column options.
+const SKIPPING_INDEX_GRPC_KEY: &str = "skipping_index";

 /// Tries to construct a `ColumnSchema` from the given  `ColumnDef`.
 pub fn try_as_column_schema(column_def: &ColumnDef) -> Result<ColumnSchema> {
@@ -60,6 +62,9 @@ pub fn try_as_column_schema(column_def: &ColumnDef) -> Result<ColumnSchema> {
        if let Some(inverted_index) = options.options.get(INVERTED_INDEX_GRPC_KEY) {
            metadata.insert(INVERTED_INDEX_KEY.to_string(), inverted_index.clone());
        }
+        if let Some(skipping_index) = options.options.get(SKIPPING_INDEX_GRPC_KEY) {
+            metadata.insert(SKIPPING_INDEX_KEY.to_string(), skipping_index.clone());
+        }
    }

    ColumnSchema::new(&column_def.name, data_type.into(), column_def.is_nullable)
@@ -84,6 +89,11 @@ pub fn options_from_column_schema(column_schema: &ColumnSchema) -> Option<Column
            .options
            .insert(INVERTED_INDEX_GRPC_KEY.to_string(), inverted_index.clone());
    }
+    if let Some(skipping_index) = column_schema.metadata().get(SKIPPING_INDEX_KEY) {
+        options
+            .options
+            .insert(SKIPPING_INDEX_GRPC_KEY.to_string(), skipping_index.clone());
+    }

    (!options.options.is_empty()).then_some(options)
 }
--- a/src/cache/Cargo.toml
+++ b/src/cache/Cargo.toml
@@ -11,4 +11,3 @@ common-macro.workspace = true
 common-meta.workspace = true
 moka.workspace = true
 snafu.workspace = true
-substrait.workspace = true
--- a/src/cache/src/lib.rs
+++ b/src/cache/src/lib.rs
@@ -19,9 +19,9 @@ use std::time::Duration;

 use catalog::kvbackend::new_table_cache;
 use common_meta::cache::{
-    new_table_flownode_set_cache, new_table_info_cache, new_table_name_cache,
-    new_table_route_cache, new_view_info_cache, CacheRegistry, CacheRegistryBuilder,
-    LayeredCacheRegistryBuilder,
+    new_schema_cache, new_table_flownode_set_cache, new_table_info_cache, new_table_name_cache,
+    new_table_route_cache, new_table_schema_cache, new_view_info_cache, CacheRegistry,
+    CacheRegistryBuilder, LayeredCacheRegistryBuilder,
 };
 use common_meta::kv_backend::KvBackendRef;
 use moka::future::CacheBuilder;
@@ -37,9 +37,47 @@ pub const TABLE_INFO_CACHE_NAME: &str = "table_info_cache";
 pub const VIEW_INFO_CACHE_NAME: &str = "view_info_cache";
 pub const TABLE_NAME_CACHE_NAME: &str = "table_name_cache";
 pub const TABLE_CACHE_NAME: &str = "table_cache";
+pub const SCHEMA_CACHE_NAME: &str = "schema_cache";
+pub const TABLE_SCHEMA_NAME_CACHE_NAME: &str = "table_schema_name_cache";
 pub const TABLE_FLOWNODE_SET_CACHE_NAME: &str = "table_flownode_set_cache";
 pub const TABLE_ROUTE_CACHE_NAME: &str = "table_route_cache";

+/// Builds cache registry for datanode, including:
+/// - Schema cache.
+/// - Table id to schema name cache.
+pub fn build_datanode_cache_registry(kv_backend: KvBackendRef) -> CacheRegistry {
+    // Builds table id schema name cache that never expires.
+    let cache = CacheBuilder::new(DEFAULT_CACHE_MAX_CAPACITY).build();
+    let table_id_schema_cache = Arc::new(new_table_schema_cache(
+        TABLE_SCHEMA_NAME_CACHE_NAME.to_string(),
+        cache,
+        kv_backend.clone(),
+    ));
+
+    // Builds schema cache
+    let cache = CacheBuilder::new(DEFAULT_CACHE_MAX_CAPACITY)
+        .time_to_live(DEFAULT_CACHE_TTL)
+        .time_to_idle(DEFAULT_CACHE_TTI)
+        .build();
+    let schema_cache = Arc::new(new_schema_cache(
+        SCHEMA_CACHE_NAME.to_string(),
+        cache,
+        kv_backend.clone(),
+    ));
+
+    CacheRegistryBuilder::default()
+        .add_cache(table_id_schema_cache)
+        .add_cache(schema_cache)
+        .build()
+}
+
+/// Builds cache registry for frontend and datanode, including:
+/// - Table info cache
+/// - Table name cache
+/// - Table route cache
+/// - Table flow node cache
+/// - View cache
+/// - Schema cache
 pub fn build_fundamental_cache_registry(kv_backend: KvBackendRef) -> CacheRegistry {
    // Builds table info cache
    let cache = CacheBuilder::new(DEFAULT_CACHE_MAX_CAPACITY)
@@ -95,12 +133,30 @@ pub fn build_fundamental_cache_registry(kv_backend: KvBackendRef) -> CacheRegist
        kv_backend.clone(),
    ));

+    // Builds schema cache
+    let cache = CacheBuilder::new(DEFAULT_CACHE_MAX_CAPACITY)
+        .time_to_live(DEFAULT_CACHE_TTL)
+        .time_to_idle(DEFAULT_CACHE_TTI)
+        .build();
+    let schema_cache = Arc::new(new_schema_cache(
+        SCHEMA_CACHE_NAME.to_string(),
+        cache,
+        kv_backend.clone(),
+    ));
+
+    let table_id_schema_cache = Arc::new(new_table_schema_cache(
+        TABLE_SCHEMA_NAME_CACHE_NAME.to_string(),
+        CacheBuilder::new(DEFAULT_CACHE_MAX_CAPACITY).build(),
+        kv_backend,
+    ));
    CacheRegistryBuilder::default()
        .add_cache(table_info_cache)
        .add_cache(table_name_cache)
        .add_cache(table_route_cache)
        .add_cache(view_info_cache)
        .add_cache(table_flownode_set_cache)
+        .add_cache(schema_cache)
+        .add_cache(table_id_schema_cache)
        .build()
 }

--- a/src/catalog/Cargo.toml
+++ b/src/catalog/Cargo.toml
@@ -18,7 +18,6 @@ async-stream.workspace = true
 async-trait = "0.1"
 bytes.workspace = true
 common-catalog.workspace = true
-common-config.workspace = true
 common-error.workspace = true
 common-macro.workspace = true
 common-meta.workspace = true
@@ -58,7 +57,5 @@ catalog = { workspace = true, features = ["testing"] }
 chrono.workspace = true
 common-meta = { workspace = true, features = ["testing"] }
 common-query = { workspace = true, features = ["testing"] }
-common-test-util.workspace = true
-log-store.workspace = true
 object-store.workspace = true
 tokio.workspace = true
--- a/src/catalog/src/information_extension.rs
+++ b/src/catalog/src/information_extension.rs
@@ -0,0 +1,92 @@
+// Copyright 2023 Greptime Team
+//
+// Licensed under the Apache License, Version 2.0 (the "License");
+// you may not use this file except in compliance with the License.
+// You may obtain a copy of the License at
+//
+//     http://www.apache.org/licenses/LICENSE-2.0
+//
+// Unless required by applicable law or agreed to in writing, software
+// distributed under the License is distributed on an "AS IS" BASIS,
+// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+// See the License for the specific language governing permissions and
+// limitations under the License.
+
+use api::v1::meta::ProcedureStatus;
+use common_error::ext::BoxedError;
+use common_meta::cluster::{ClusterInfo, NodeInfo};
+use common_meta::datanode::RegionStat;
+use common_meta::ddl::{ExecutorContext, ProcedureExecutor};
+use common_meta::rpc::procedure;
+use common_procedure::{ProcedureInfo, ProcedureState};
+use meta_client::MetaClientRef;
+use snafu::ResultExt;
+
+use crate::error;
+use crate::information_schema::InformationExtension;
+
+pub struct DistributedInformationExtension {
+    meta_client: MetaClientRef,
+}
+
+impl DistributedInformationExtension {
+    pub fn new(meta_client: MetaClientRef) -> Self {
+        Self { meta_client }
+    }
+}
+
+#[async_trait::async_trait]
+impl InformationExtension for DistributedInformationExtension {
+    type Error = crate::error::Error;
+
+    async fn nodes(&self) -> std::result::Result<Vec<NodeInfo>, Self::Error> {
+        self.meta_client
+            .list_nodes(None)
+            .await
+            .map_err(BoxedError::new)
+            .context(error::ListNodesSnafu)
+    }
+
+    async fn procedures(&self) -> std::result::Result<Vec<(String, ProcedureInfo)>, Self::Error> {
+        let procedures = self
+            .meta_client
+            .list_procedures(&ExecutorContext::default())
+            .await
+            .map_err(BoxedError::new)
+            .context(error::ListProceduresSnafu)?
+            .procedures;
+        let mut result = Vec::with_capacity(procedures.len());
+        for procedure in procedures {
+            let pid = match procedure.id {
+                Some(pid) => pid,
+                None => return error::ProcedureIdNotFoundSnafu {}.fail(),
+            };
+            let pid = procedure::pb_pid_to_pid(&pid)
+                .map_err(BoxedError::new)
+                .context(error::ConvertProtoDataSnafu)?;
+            let status = ProcedureStatus::try_from(procedure.status)
+                .map(|v| v.as_str_name())
+                .unwrap_or("Unknown")
+                .to_string();
+            let procedure_info = ProcedureInfo {
+                id: pid,
+                type_name: procedure.type_name,
+                start_time_ms: procedure.start_time_ms,
+                end_time_ms: procedure.end_time_ms,
+                state: ProcedureState::Running,
+                lock_keys: procedure.lock_keys,
+            };
+            result.push((status, procedure_info));
+        }
+
+        Ok(result)
+    }
+
+    async fn region_stats(&self) -> std::result::Result<Vec<RegionStat>, Self::Error> {
+        self.meta_client
+            .list_region_stats()
+            .await
+            .map_err(BoxedError::new)
+            .context(error::ListRegionStatsSnafu)
+    }
+}
--- a/src/catalog/src/lib.rs
+++ b/src/catalog/src/lib.rs
@@ -30,6 +30,7 @@ use table::TableRef;
 use crate::error::Result;

 pub mod error;
+pub mod information_extension;
 pub mod kvbackend;
 pub mod memory;
 mod metrics;
--- a/src/catalog/src/system_schema/information_schema/key_column_usage.rs
+++ b/src/catalog/src/system_schema/information_schema/key_column_usage.rs
@@ -54,6 +54,10 @@ const INIT_CAPACITY: usize = 42;
 pub(crate) const PRI_CONSTRAINT_NAME: &str = "PRIMARY";
 /// Time index constraint name
 pub(crate) const TIME_INDEX_CONSTRAINT_NAME: &str = "TIME INDEX";
+/// Inverted index constraint name
+pub(crate) const INVERTED_INDEX_CONSTRAINT_NAME: &str = "INVERTED INDEX";
+/// Fulltext index constraint name
+pub(crate) const FULLTEXT_INDEX_CONSTRAINT_NAME: &str = "FULLTEXT INDEX";

 /// The virtual table implementation for `information_schema.KEY_COLUMN_USAGE`.
 pub(super) struct InformationSchemaKeyColumnUsage {
@@ -216,14 +220,13 @@ impl InformationSchemaKeyColumnUsageBuilder {
            let mut stream = catalog_manager.tables(&catalog_name, &schema_name, None);

            while let Some(table) = stream.try_next().await? {
-                let mut primary_constraints = vec![];
-
                let table_info = table.table_info();
                let table_name = &table_info.name;
                let keys = &table_info.meta.primary_key_indices;
                let schema = table.schema();

                for (idx, column) in schema.column_schemas().iter().enumerate() {
+                    let mut constraints = vec![];
                    if column.is_time_index() {
                        self.add_key_column_usage(
                            &predicates,
@@ -236,30 +239,31 @@ impl InformationSchemaKeyColumnUsageBuilder {
                            1, //always 1 for time index
                        );
                    }
-                    if keys.contains(&idx) {
-                        primary_constraints.push((
-                            catalog_name.clone(),
-                            schema_name.clone(),
-                            table_name.to_string(),
-                            column.name.clone(),
-                        ));
-                    }
                    // TODO(dimbtp): foreign key constraint not supported yet
-                }
+                    if keys.contains(&idx) {
+                        constraints.push(PRI_CONSTRAINT_NAME);
+                    }
+                    if column.is_inverted_indexed() {
+                        constraints.push(INVERTED_INDEX_CONSTRAINT_NAME);
+                    }

-                for (i, (catalog_name, schema_name, table_name, column_name)) in
-                    primary_constraints.into_iter().enumerate()
-                {
-                    self.add_key_column_usage(
-                        &predicates,
-                        &schema_name,
-                        PRI_CONSTRAINT_NAME,
-                        &catalog_name,
-                        &schema_name,
-                        &table_name,
-                        &column_name,
-                        i as u32 + 1,
-                    );
+                    if column.has_fulltext_index_key() {
+                        constraints.push(FULLTEXT_INDEX_CONSTRAINT_NAME);
+                    }
+
+                    if !constraints.is_empty() {
+                        let aggregated_constraints = constraints.join(", ");
+                        self.add_key_column_usage(
+                            &predicates,
+                            &schema_name,
+                            &aggregated_constraints,
+                            &catalog_name,
+                            &schema_name,
+                            table_name,
+                            &column.name,
+                            idx as u32 + 1,
+                        );
+                    }
                }
            }
        }
--- a/src/cli/Cargo.toml
+++ b/src/cli/Cargo.toml
@@ -0,0 +1,63 @@
+[package]
+name = "cli"
+version.workspace = true
+edition.workspace = true
+license.workspace = true
+
+[lints]
+workspace = true
+
+[dependencies]
+async-trait.workspace = true
+auth.workspace = true
+base64.workspace = true
+cache.workspace = true
+catalog.workspace = true
+chrono.workspace = true
+clap.workspace = true
+client.workspace = true
+common-base.workspace = true
+common-catalog.workspace = true
+common-config.workspace = true
+common-error.workspace = true
+common-grpc.workspace = true
+common-macro.workspace = true
+common-meta.workspace = true
+common-procedure.workspace = true
+common-query.workspace = true
+common-recordbatch.workspace = true
+common-runtime.workspace = true
+common-telemetry = { workspace = true, features = [
+    "deadlock_detection",
+] }
+common-time.workspace = true
+common-version.workspace = true
+common-wal.workspace = true
+datatypes.workspace = true
+either = "1.8"
+etcd-client.workspace = true
+futures.workspace = true
+humantime.workspace = true
+meta-client.workspace = true
+nu-ansi-term = "0.46"
+query.workspace = true
+rand.workspace = true
+reqwest.workspace = true
+rustyline = "10.1"
+serde.workspace = true
+serde_json.workspace = true
+servers.workspace = true
+session.workspace = true
+snafu.workspace = true
+store-api.workspace = true
+substrait.workspace = true
+table.workspace = true
+tokio.workspace = true
+tracing-appender.workspace = true
+
+[dev-dependencies]
+client = { workspace = true, features = ["testing"] }
+common-test-util.workspace = true
+common-version.workspace = true
+serde.workspace = true
+tempfile.workspace = true
--- a/src/cmd/src/cli/bench.rs
+++ b/src/cmd/src/cli/bench.rs
@@ -19,6 +19,7 @@ use std::time::Duration;

 use async_trait::async_trait;
 use clap::Parser;
+use common_error::ext::BoxedError;
 use common_meta::key::{TableMetadataManager, TableMetadataManagerRef};
 use common_meta::kv_backend::etcd::EtcdStore;
 use common_meta::peer::Peer;
@@ -30,11 +31,9 @@ use rand::Rng;
 use store_api::storage::RegionNumber;
 use table::metadata::{RawTableInfo, RawTableMeta, TableId, TableIdent, TableType};
 use table::table_name::TableName;
-use tracing_appender::non_blocking::WorkerGuard;

 use self::metadata::TableMetadataBencher;
-use crate::cli::{Instance, Tool};
-use crate::error::Result;
+use crate::Tool;

 mod metadata;

@@ -62,7 +61,7 @@ pub struct BenchTableMetadataCommand {
 }

 impl BenchTableMetadataCommand {
-    pub async fn build(&self, guard: Vec<WorkerGuard>) -> Result<Instance> {
+    pub async fn build(&self) -> std::result::Result<Box<dyn Tool>, BoxedError> {
        let etcd_store = EtcdStore::with_endpoints([&self.etcd_addr], 128)
            .await
            .unwrap();
@@ -73,7 +72,7 @@ impl BenchTableMetadataCommand {
            table_metadata_manager,
            count: self.count,
        };
-        Ok(Instance::new(Box::new(tool), guard))
+        Ok(Box::new(tool))
    }
 }

@@ -84,7 +83,7 @@ struct BenchTableMetadata {

 #[async_trait]
 impl Tool for BenchTableMetadata {
-    async fn do_work(&self) -> Result<()> {
+    async fn do_work(&self) -> std::result::Result<(), BoxedError> {
        let bencher = TableMetadataBencher::new(self.table_metadata_manager.clone(), self.count);
        bencher.bench_create().await;
        bencher.bench_get().await;
--- a/src/cmd/src/cli/bench/metadata.rs
+++ b/src/cmd/src/cli/bench/metadata.rs
@@ -18,7 +18,7 @@ use common_meta::key::table_route::TableRouteValue;
 use common_meta::key::TableMetadataManagerRef;
 use table::table_name::TableName;

-use crate::cli::bench::{
+use crate::bench::{
    bench_self_recorded, create_region_routes, create_region_wal_options, create_table_info,
 };

--- a/src/cmd/src/cli/cmd.rs
+++ b/src/cmd/src/cli/cmd.rs
--- a/src/cmd/src/cli/database.rs
+++ b/src/cmd/src/cli/database.rs
@@ -26,7 +26,8 @@ use snafu::ResultExt;

 use crate::error::{HttpQuerySqlSnafu, Result, SerdeJsonSnafu};

-pub(crate) struct DatabaseClient {
+#[derive(Debug, Clone)]
+pub struct DatabaseClient {
    addr: String,
    catalog: String,
    auth_header: Option<String>,
--- a/src/cli/src/error.rs
+++ b/src/cli/src/error.rs
@@ -0,0 +1,316 @@
+// Copyright 2023 Greptime Team
+//
+// Licensed under the Apache License, Version 2.0 (the "License");
+// you may not use this file except in compliance with the License.
+// You may obtain a copy of the License at
+//
+//     http://www.apache.org/licenses/LICENSE-2.0
+//
+// Unless required by applicable law or agreed to in writing, software
+// distributed under the License is distributed on an "AS IS" BASIS,
+// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+// See the License for the specific language governing permissions and
+// limitations under the License.
+
+use std::any::Any;
+
+use common_error::ext::{BoxedError, ErrorExt};
+use common_error::status_code::StatusCode;
+use common_macro::stack_trace_debug;
+use rustyline::error::ReadlineError;
+use snafu::{Location, Snafu};
+
+#[derive(Snafu)]
+#[snafu(visibility(pub))]
+#[stack_trace_debug]
+pub enum Error {
+    #[snafu(display("Failed to install ring crypto provider: {}", msg))]
+    InitTlsProvider {
+        #[snafu(implicit)]
+        location: Location,
+        msg: String,
+    },
+    #[snafu(display("Failed to create default catalog and schema"))]
+    InitMetadata {
+        #[snafu(implicit)]
+        location: Location,
+        source: common_meta::error::Error,
+    },
+
+    #[snafu(display("Failed to init DDL manager"))]
+    InitDdlManager {
+        #[snafu(implicit)]
+        location: Location,
+        source: common_meta::error::Error,
+    },
+
+    #[snafu(display("Failed to init default timezone"))]
+    InitTimezone {
+        #[snafu(implicit)]
+        location: Location,
+        source: common_time::error::Error,
+    },
+
+    #[snafu(display("Failed to start procedure manager"))]
+    StartProcedureManager {
+        #[snafu(implicit)]
+        location: Location,
+        source: common_procedure::error::Error,
+    },
+
+    #[snafu(display("Failed to stop procedure manager"))]
+    StopProcedureManager {
+        #[snafu(implicit)]
+        location: Location,
+        source: common_procedure::error::Error,
+    },
+
+    #[snafu(display("Failed to start wal options allocator"))]
+    StartWalOptionsAllocator {
+        #[snafu(implicit)]
+        location: Location,
+        source: common_meta::error::Error,
+    },
+
+    #[snafu(display("Missing config, msg: {}", msg))]
+    MissingConfig {
+        msg: String,
+        #[snafu(implicit)]
+        location: Location,
+    },
+
+    #[snafu(display("Illegal config: {}", msg))]
+    IllegalConfig {
+        msg: String,
+        #[snafu(implicit)]
+        location: Location,
+    },
+
+    #[snafu(display("Invalid REPL command: {reason}"))]
+    InvalidReplCommand { reason: String },
+
+    #[snafu(display("Cannot create REPL"))]
+    ReplCreation {
+        #[snafu(source)]
+        error: ReadlineError,
+        #[snafu(implicit)]
+        location: Location,
+    },
+
+    #[snafu(display("Error reading command"))]
+    Readline {
+        #[snafu(source)]
+        error: ReadlineError,
+        #[snafu(implicit)]
+        location: Location,
+    },
+
+    #[snafu(display("Failed to request database, sql: {sql}"))]
+    RequestDatabase {
+        sql: String,
+        #[snafu(source)]
+        source: client::Error,
+        #[snafu(implicit)]
+        location: Location,
+    },
+
+    #[snafu(display("Failed to collect RecordBatches"))]
+    CollectRecordBatches {
+        #[snafu(implicit)]
+        location: Location,
+        source: common_recordbatch::error::Error,
+    },
+
+    #[snafu(display("Failed to pretty print Recordbatches"))]
+    PrettyPrintRecordBatches {
+        #[snafu(implicit)]
+        location: Location,
+        source: common_recordbatch::error::Error,
+    },
+
+    #[snafu(display("Failed to start Meta client"))]
+    StartMetaClient {
+        #[snafu(implicit)]
+        location: Location,
+        source: meta_client::error::Error,
+    },
+
+    #[snafu(display("Failed to parse SQL: {}", sql))]
+    ParseSql {
+        sql: String,
+        #[snafu(implicit)]
+        location: Location,
+        source: query::error::Error,
+    },
+
+    #[snafu(display("Failed to plan statement"))]
+    PlanStatement {
+        #[snafu(implicit)]
+        location: Location,
+        source: query::error::Error,
+    },
+
+    #[snafu(display("Failed to encode logical plan in substrait"))]
+    SubstraitEncodeLogicalPlan {
+        #[snafu(implicit)]
+        location: Location,
+        source: substrait::error::Error,
+    },
+
+    #[snafu(display("Failed to load layered config"))]
+    LoadLayeredConfig {
+        #[snafu(source(from(common_config::error::Error, Box::new)))]
+        source: Box<common_config::error::Error>,
+        #[snafu(implicit)]
+        location: Location,
+    },
+
+    #[snafu(display("Failed to connect to Etcd at {etcd_addr}"))]
+    ConnectEtcd {
+        etcd_addr: String,
+        #[snafu(source)]
+        error: etcd_client::Error,
+        #[snafu(implicit)]
+        location: Location,
+    },
+
+    #[snafu(display("Failed to serde json"))]
+    SerdeJson {
+        #[snafu(source)]
+        error: serde_json::error::Error,
+        #[snafu(implicit)]
+        location: Location,
+    },
+
+    #[snafu(display("Failed to run http request: {reason}"))]
+    HttpQuerySql {
+        reason: String,
+        #[snafu(source)]
+        error: reqwest::Error,
+        #[snafu(implicit)]
+        location: Location,
+    },
+
+    #[snafu(display("Empty result from output"))]
+    EmptyResult {
+        #[snafu(implicit)]
+        location: Location,
+    },
+
+    #[snafu(display("Failed to manipulate file"))]
+    FileIo {
+        #[snafu(implicit)]
+        location: Location,
+        #[snafu(source)]
+        error: std::io::Error,
+    },
+
+    #[snafu(display("Failed to create directory {}", dir))]
+    CreateDir {
+        dir: String,
+        #[snafu(source)]
+        error: std::io::Error,
+    },
+
+    #[snafu(display("Failed to spawn thread"))]
+    SpawnThread {
+        #[snafu(source)]
+        error: std::io::Error,
+    },
+
+    #[snafu(display("Other error"))]
+    Other {
+        source: BoxedError,
+        #[snafu(implicit)]
+        location: Location,
+    },
+
+    #[snafu(display("Failed to build runtime"))]
+    BuildRuntime {
+        #[snafu(implicit)]
+        location: Location,
+        source: common_runtime::error::Error,
+    },
+
+    #[snafu(display("Failed to get cache from cache registry: {}", name))]
+    CacheRequired {
+        #[snafu(implicit)]
+        location: Location,
+        name: String,
+    },
+
+    #[snafu(display("Failed to build cache registry"))]
+    BuildCacheRegistry {
+        #[snafu(implicit)]
+        location: Location,
+        source: cache::error::Error,
+    },
+
+    #[snafu(display("Failed to initialize meta client"))]
+    MetaClientInit {
+        #[snafu(implicit)]
+        location: Location,
+        source: meta_client::error::Error,
+    },
+
+    #[snafu(display("Cannot find schema {schema} in catalog {catalog}"))]
+    SchemaNotFound {
+        catalog: String,
+        schema: String,
+        #[snafu(implicit)]
+        location: Location,
+    },
+}
+
+pub type Result<T> = std::result::Result<T, Error>;
+
+impl ErrorExt for Error {
+    fn status_code(&self) -> StatusCode {
+        match self {
+            Error::InitMetadata { source, .. } | Error::InitDdlManager { source, .. } => {
+                source.status_code()
+            }
+
+            Error::MissingConfig { .. }
+            | Error::LoadLayeredConfig { .. }
+            | Error::IllegalConfig { .. }
+            | Error::InvalidReplCommand { .. }
+            | Error::InitTimezone { .. }
+            | Error::ConnectEtcd { .. }
+            | Error::CreateDir { .. }
+            | Error::EmptyResult { .. } => StatusCode::InvalidArguments,
+
+            Error::StartProcedureManager { source, .. }
+            | Error::StopProcedureManager { source, .. } => source.status_code(),
+            Error::StartWalOptionsAllocator { source, .. } => source.status_code(),
+            Error::ReplCreation { .. } | Error::Readline { .. } | Error::HttpQuerySql { .. } => {
+                StatusCode::Internal
+            }
+            Error::RequestDatabase { source, .. } => source.status_code(),
+            Error::CollectRecordBatches { source, .. }
+            | Error::PrettyPrintRecordBatches { source, .. } => source.status_code(),
+            Error::StartMetaClient { source, .. } => source.status_code(),
+            Error::ParseSql { source, .. } | Error::PlanStatement { source, .. } => {
+                source.status_code()
+            }
+            Error::SubstraitEncodeLogicalPlan { source, .. } => source.status_code(),
+
+            Error::SerdeJson { .. }
+            | Error::FileIo { .. }
+            | Error::SpawnThread { .. }
+            | Error::InitTlsProvider { .. } => StatusCode::Unexpected,
+
+            Error::Other { source, .. } => source.status_code(),
+
+            Error::BuildRuntime { source, .. } => source.status_code(),
+
+            Error::CacheRequired { .. } | Error::BuildCacheRegistry { .. } => StatusCode::Internal,
+            Error::MetaClientInit { source, .. } => source.status_code(),
+            Error::SchemaNotFound { .. } => StatusCode::DatabaseNotFound,
+        }
+    }
+
+    fn as_any(&self) -> &dyn Any {
+        self
+    }
+}
--- a/src/cmd/src/cli/export.rs
+++ b/src/cmd/src/cli/export.rs
@@ -19,6 +19,7 @@ use std::time::Duration;

 use async_trait::async_trait;
 use clap::{Parser, ValueEnum};
+use common_error::ext::BoxedError;
 use common_telemetry::{debug, error, info};
 use serde_json::Value;
 use snafu::{OptionExt, ResultExt};
@@ -26,11 +27,10 @@ use tokio::fs::File;
 use tokio::io::{AsyncWriteExt, BufWriter};
 use tokio::sync::Semaphore;
 use tokio::time::Instant;
-use tracing_appender::non_blocking::WorkerGuard;

-use crate::cli::database::DatabaseClient;
-use crate::cli::{database, Instance, Tool};
+use crate::database::DatabaseClient;
 use crate::error::{EmptyResultSnafu, Error, FileIoSnafu, Result, SchemaNotFoundSnafu};
+use crate::{database, Tool};

 type TableReference = (String, String, String);

@@ -94,8 +94,9 @@ pub struct ExportCommand {
 }

 impl ExportCommand {
-    pub async fn build(&self, guard: Vec<WorkerGuard>) -> Result<Instance> {
-        let (catalog, schema) = database::split_database(&self.database)?;
+    pub async fn build(&self) -> std::result::Result<Box<dyn Tool>, BoxedError> {
+        let (catalog, schema) =
+            database::split_database(&self.database).map_err(BoxedError::new)?;

        let database_client = DatabaseClient::new(
            self.addr.clone(),
@@ -105,19 +106,16 @@ impl ExportCommand {
            self.timeout.unwrap_or_default(),
        );

-        Ok(Instance::new(
-            Box::new(Export {
-                catalog,
-                schema,
-                database_client,
-                output_dir: self.output_dir.clone(),
-                parallelism: self.export_jobs,
-                target: self.target.clone(),
-                start_time: self.start_time.clone(),
-                end_time: self.end_time.clone(),
-            }),
-            guard,
-        ))
+        Ok(Box::new(Export {
+            catalog,
+            schema,
+            database_client,
+            output_dir: self.output_dir.clone(),
+            parallelism: self.export_jobs,
+            target: self.target.clone(),
+            start_time: self.start_time.clone(),
+            end_time: self.end_time.clone(),
+        }))
    }
 }

@@ -465,97 +463,22 @@ impl Export {

 #[async_trait]
 impl Tool for Export {
-    async fn do_work(&self) -> Result<()> {
+    async fn do_work(&self) -> std::result::Result<(), BoxedError> {
        match self.target {
            ExportTarget::Schema => {
-                self.export_create_database().await?;
-                self.export_create_table().await
+                self.export_create_database()
+                    .await
+                    .map_err(BoxedError::new)?;
+                self.export_create_table().await.map_err(BoxedError::new)
            }
-            ExportTarget::Data => self.export_database_data().await,
+            ExportTarget::Data => self.export_database_data().await.map_err(BoxedError::new),
            ExportTarget::All => {
-                self.export_create_database().await?;
-                self.export_create_table().await?;
-                self.export_database_data().await
+                self.export_create_database()
+                    .await
+                    .map_err(BoxedError::new)?;
+                self.export_create_table().await.map_err(BoxedError::new)?;
+                self.export_database_data().await.map_err(BoxedError::new)
            }
        }
    }
 }
-
-#[cfg(test)]
-mod tests {
-    use clap::Parser;
-    use client::{Client, Database};
-    use common_catalog::consts::{DEFAULT_CATALOG_NAME, DEFAULT_SCHEMA_NAME};
-    use common_telemetry::logging::LoggingOptions;
-
-    use crate::error::Result as CmdResult;
-    use crate::options::GlobalOptions;
-    use crate::{cli, standalone, App};
-
-    #[tokio::test(flavor = "multi_thread")]
-    async fn test_export_create_table_with_quoted_names() -> CmdResult<()> {
-        let output_dir = tempfile::tempdir().unwrap();
-
-        let standalone = standalone::Command::parse_from([
-            "standalone",
-            "start",
-            "--data-home",
-            &*output_dir.path().to_string_lossy(),
-        ]);
-
-        let standalone_opts = standalone.load_options(&GlobalOptions::default()).unwrap();
-        let mut instance = standalone.build(standalone_opts).await?;
-        instance.start().await?;
-
-        let client = Client::with_urls(["127.0.0.1:4001"]);
-        let database = Database::new(DEFAULT_CATALOG_NAME, DEFAULT_SCHEMA_NAME, client);
-        database
-            .sql(r#"CREATE DATABASE "cli.export.create_table";"#)
-            .await
-            .unwrap();
-        database
-            .sql(
-                r#"CREATE TABLE "cli.export.create_table"."a.b.c"(
-                        ts TIMESTAMP,
-                        TIME INDEX (ts)
-                    ) engine=mito;
-                "#,
-            )
-            .await
-            .unwrap();
-
-        let output_dir = tempfile::tempdir().unwrap();
-        let cli = cli::Command::parse_from([
-            "cli",
-            "export",
-            "--addr",
-            "127.0.0.1:4000",
-            "--output-dir",
-            &*output_dir.path().to_string_lossy(),
-            "--target",
-            "schema",
-        ]);
-        let mut cli_app = cli.build(LoggingOptions::default()).await?;
-        cli_app.start().await?;
-
-        instance.stop().await?;
-
-        let output_file = output_dir
-            .path()
-            .join("greptime")
-            .join("cli.export.create_table")
-            .join("create_tables.sql");
-        let res = std::fs::read_to_string(output_file).unwrap();
-        let expect = r#"CREATE TABLE IF NOT EXISTS "a.b.c" (
-  "ts" TIMESTAMP(3) NOT NULL,
-  TIME INDEX ("ts")
-)
-
-ENGINE=mito
-;
-"#;
-        assert_eq!(res.trim(), expect.trim());
-
-        Ok(())
-    }
-}
--- a/src/cmd/src/cli/helper.rs
+++ b/src/cmd/src/cli/helper.rs
@@ -19,7 +19,7 @@ use rustyline::highlight::{Highlighter, MatchingBracketHighlighter};
 use rustyline::hint::{Hinter, HistoryHinter};
 use rustyline::validate::{ValidationContext, ValidationResult, Validator};

-use crate::cli::cmd::ReplCommand;
+use crate::cmd::ReplCommand;

 pub(crate) struct RustylineHelper {
    hinter: HistoryHinter,
--- a/src/cmd/src/cli/import.rs
+++ b/src/cmd/src/cli/import.rs
@@ -19,15 +19,15 @@ use std::time::Duration;
 use async_trait::async_trait;
 use clap::{Parser, ValueEnum};
 use common_catalog::consts::DEFAULT_SCHEMA_NAME;
+use common_error::ext::BoxedError;
 use common_telemetry::{error, info, warn};
 use snafu::{OptionExt, ResultExt};
 use tokio::sync::Semaphore;
 use tokio::time::Instant;
-use tracing_appender::non_blocking::WorkerGuard;

-use crate::cli::database::DatabaseClient;
-use crate::cli::{database, Instance, Tool};
+use crate::database::DatabaseClient;
 use crate::error::{Error, FileIoSnafu, Result, SchemaNotFoundSnafu};
+use crate::{database, Tool};

 #[derive(Debug, Default, Clone, ValueEnum)]
 enum ImportTarget {
@@ -79,8 +79,9 @@ pub struct ImportCommand {
 }

 impl ImportCommand {
-    pub async fn build(&self, guard: Vec<WorkerGuard>) -> Result<Instance> {
-        let (catalog, schema) = database::split_database(&self.database)?;
+    pub async fn build(&self) -> std::result::Result<Box<dyn Tool>, BoxedError> {
+        let (catalog, schema) =
+            database::split_database(&self.database).map_err(BoxedError::new)?;
        let database_client = DatabaseClient::new(
            self.addr.clone(),
            catalog.clone(),
@@ -89,17 +90,14 @@ impl ImportCommand {
            self.timeout.unwrap_or_default(),
        );

-        Ok(Instance::new(
-            Box::new(Import {
-                catalog,
-                schema,
-                database_client,
-                input_dir: self.input_dir.clone(),
-                parallelism: self.import_jobs,
-                target: self.target.clone(),
-            }),
-            guard,
-        ))
+        Ok(Box::new(Import {
+            catalog,
+            schema,
+            database_client,
+            input_dir: self.input_dir.clone(),
+            parallelism: self.import_jobs,
+            target: self.target.clone(),
+        }))
    }
 }

@@ -218,13 +216,13 @@ impl Import {

 #[async_trait]
 impl Tool for Import {
-    async fn do_work(&self) -> Result<()> {
+    async fn do_work(&self) -> std::result::Result<(), BoxedError> {
        match self.target {
-            ImportTarget::Schema => self.import_create_table().await,
-            ImportTarget::Data => self.import_database_data().await,
+            ImportTarget::Schema => self.import_create_table().await.map_err(BoxedError::new),
+            ImportTarget::Data => self.import_database_data().await.map_err(BoxedError::new),
            ImportTarget::All => {
-                self.import_create_table().await?;
-                self.import_database_data().await
+                self.import_create_table().await.map_err(BoxedError::new)?;
+                self.import_database_data().await.map_err(BoxedError::new)
            }
        }
    }
--- a/src/cli/src/lib.rs
+++ b/src/cli/src/lib.rs
@@ -0,0 +1,60 @@
+// Copyright 2023 Greptime Team
+//
+// Licensed under the Apache License, Version 2.0 (the "License");
+// you may not use this file except in compliance with the License.
+// You may obtain a copy of the License at
+//
+//     http://www.apache.org/licenses/LICENSE-2.0
+//
+// Unless required by applicable law or agreed to in writing, software
+// distributed under the License is distributed on an "AS IS" BASIS,
+// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+// See the License for the specific language governing permissions and
+// limitations under the License.
+
+mod bench;
+pub mod error;
+// Wait for https://github.com/GreptimeTeam/greptimedb/issues/2373
+#[allow(unused)]
+mod cmd;
+mod export;
+mod helper;
+
+// Wait for https://github.com/GreptimeTeam/greptimedb/issues/2373
+mod database;
+mod import;
+#[allow(unused)]
+mod repl;
+
+use async_trait::async_trait;
+use clap::Parser;
+use common_error::ext::BoxedError;
+pub use database::DatabaseClient;
+use error::Result;
+pub use repl::Repl;
+
+pub use crate::bench::BenchTableMetadataCommand;
+pub use crate::export::ExportCommand;
+pub use crate::import::ImportCommand;
+
+#[async_trait]
+pub trait Tool: Send + Sync {
+    async fn do_work(&self) -> std::result::Result<(), BoxedError>;
+}
+
+#[derive(Debug, Parser)]
+pub(crate) struct AttachCommand {
+    #[clap(long)]
+    pub(crate) grpc_addr: String,
+    #[clap(long)]
+    pub(crate) meta_addr: Option<String>,
+    #[clap(long, action)]
+    pub(crate) disable_helper: bool,
+}
+
+impl AttachCommand {
+    #[allow(dead_code)]
+    async fn build(self) -> Result<Box<dyn Tool>> {
+        unimplemented!("Wait for https://github.com/GreptimeTeam/greptimedb/issues/2373")
+    }
+}
--- a/src/cmd/src/cli/repl.rs
+++ b/src/cmd/src/cli/repl.rs
@@ -20,6 +20,7 @@ use cache::{
    build_fundamental_cache_registry, with_default_composite_cache_registry, TABLE_CACHE_NAME,
    TABLE_ROUTE_CACHE_NAME,
 };
+use catalog::information_extension::DistributedInformationExtension;
 use catalog::kvbackend::{
    CachedKvBackend, CachedKvBackendBuilder, KvBackendCatalogManager, MetaKvBackend,
 };
@@ -44,15 +45,14 @@ use session::context::QueryContext;
 use snafu::{OptionExt, ResultExt};
 use substrait::{DFLogicalSubstraitConvertor, SubstraitPlan};

-use crate::cli::cmd::ReplCommand;
-use crate::cli::helper::RustylineHelper;
-use crate::cli::AttachCommand;
+use crate::cmd::ReplCommand;
 use crate::error::{
    CollectRecordBatchesSnafu, ParseSqlSnafu, PlanStatementSnafu, PrettyPrintRecordBatchesSnafu,
    ReadlineSnafu, ReplCreationSnafu, RequestDatabaseSnafu, Result, StartMetaClientSnafu,
    SubstraitEncodeLogicalPlanSnafu,
 };
-use crate::{error, DistributedInformationExtension};
+use crate::helper::RustylineHelper;
+use crate::{error, AttachCommand};

 /// Captures the state of the repl, gathers commands and executes them one by one
 pub struct Repl {
--- a/src/client/Cargo.toml
+++ b/src/client/Cargo.toml
@@ -42,8 +42,6 @@ tonic.workspace = true

 [dev-dependencies]
 common-grpc-expr.workspace = true
-datanode.workspace = true
-derive-new = "0.5"
 tracing = "0.1"

 [dev-dependencies.substrait_proto]
--- a/src/cmd/Cargo.toml
+++ b/src/cmd/Cargo.toml
@@ -25,6 +25,7 @@ cache.workspace = true
 catalog.workspace = true
 chrono.workspace = true
 clap.workspace = true
+cli.workspace = true
 client.workspace = true
 common-base.workspace = true
 common-catalog.workspace = true
--- a/src/cmd/src/cli.rs
+++ b/src/cmd/src/cli.rs
@@ -12,39 +12,17 @@
 // See the License for the specific language governing permissions and
 // limitations under the License.

-mod bench;
-
-// Wait for https://github.com/GreptimeTeam/greptimedb/issues/2373
-#[allow(unused)]
-mod cmd;
-mod export;
-mod helper;
-
-// Wait for https://github.com/GreptimeTeam/greptimedb/issues/2373
-mod database;
-mod import;
-#[allow(unused)]
-mod repl;
-
-use async_trait::async_trait;
-use bench::BenchTableMetadataCommand;
 use clap::Parser;
+use cli::Tool;
 use common_telemetry::logging::{LoggingOptions, TracingOptions};
-pub use repl::Repl;
+use plugins::SubCommand;
+use snafu::ResultExt;
 use tracing_appender::non_blocking::WorkerGuard;

-use self::export::ExportCommand;
-use crate::cli::import::ImportCommand;
-use crate::error::Result;
 use crate::options::GlobalOptions;
-use crate::App;
-
+use crate::{error, App, Result};
 pub const APP_NAME: &str = "greptime-cli";
-
-#[async_trait]
-pub trait Tool: Send + Sync {
-    async fn do_work(&self) -> Result<()>;
-}
+use async_trait::async_trait;

 pub struct Instance {
    tool: Box<dyn Tool>,
@@ -54,12 +32,16 @@ pub struct Instance {
 }

 impl Instance {
-    fn new(tool: Box<dyn Tool>, guard: Vec<WorkerGuard>) -> Self {
+    pub fn new(tool: Box<dyn Tool>, guard: Vec<WorkerGuard>) -> Self {
        Self {
            tool,
            _guard: guard,
        }
    }
+
+    pub async fn start(&mut self) -> Result<()> {
+        self.tool.do_work().await.context(error::StartCliSnafu)
+    }
 }

 #[async_trait]
@@ -69,7 +51,8 @@ impl App for Instance {
    }

    async fn start(&mut self) -> Result<()> {
-        self.tool.do_work().await
+        self.start().await.unwrap();
+        Ok(())
    }

    fn wait_signal(&self) -> bool {
@@ -96,7 +79,12 @@ impl Command {
            None,
        );

-        self.cmd.build(guard).await
+        let tool = self.cmd.build().await.context(error::BuildCliSnafu)?;
+        let instance = Instance {
+            tool,
+            _guard: guard,
+        };
+        Ok(instance)
    }

    pub fn load_options(&self, global_options: &GlobalOptions) -> Result<LoggingOptions> {
@@ -112,38 +100,81 @@ impl Command {
    }
 }

-#[derive(Parser)]
-enum SubCommand {
-    // Attach(AttachCommand),
-    Bench(BenchTableMetadataCommand),
-    Export(ExportCommand),
-    Import(ImportCommand),
-}
+#[cfg(test)]
+mod tests {
+    use clap::Parser;
+    use client::{Client, Database};
+    use common_catalog::consts::{DEFAULT_CATALOG_NAME, DEFAULT_SCHEMA_NAME};
+    use common_telemetry::logging::LoggingOptions;

-impl SubCommand {
-    async fn build(&self, guard: Vec<WorkerGuard>) -> Result<Instance> {
-        match self {
-            // SubCommand::Attach(cmd) => cmd.build().await,
-            SubCommand::Bench(cmd) => cmd.build(guard).await,
-            SubCommand::Export(cmd) => cmd.build(guard).await,
-            SubCommand::Import(cmd) => cmd.build(guard).await,
-        }
-    }
-}
-
-#[derive(Debug, Parser)]
-pub(crate) struct AttachCommand {
-    #[clap(long)]
-    pub(crate) grpc_addr: String,
-    #[clap(long)]
-    pub(crate) meta_addr: Option<String>,
-    #[clap(long, action)]
-    pub(crate) disable_helper: bool,
-}
-
-impl AttachCommand {
-    #[allow(dead_code)]
-    async fn build(self) -> Result<Instance> {
-        unimplemented!("Wait for https://github.com/GreptimeTeam/greptimedb/issues/2373")
+    use crate::error::Result as CmdResult;
+    use crate::options::GlobalOptions;
+    use crate::{cli, standalone, App};
+
+    #[tokio::test(flavor = "multi_thread")]
+    async fn test_export_create_table_with_quoted_names() -> CmdResult<()> {
+        let output_dir = tempfile::tempdir().unwrap();
+
+        let standalone = standalone::Command::parse_from([
+            "standalone",
+            "start",
+            "--data-home",
+            &*output_dir.path().to_string_lossy(),
+        ]);
+
+        let standalone_opts = standalone.load_options(&GlobalOptions::default()).unwrap();
+        let mut instance = standalone.build(standalone_opts).await?;
+        instance.start().await?;
+
+        let client = Client::with_urls(["127.0.0.1:4001"]);
+        let database = Database::new(DEFAULT_CATALOG_NAME, DEFAULT_SCHEMA_NAME, client);
+        database
+            .sql(r#"CREATE DATABASE "cli.export.create_table";"#)
+            .await
+            .unwrap();
+        database
+            .sql(
+                r#"CREATE TABLE "cli.export.create_table"."a.b.c"(
+                        ts TIMESTAMP,
+                        TIME INDEX (ts)
+                    ) engine=mito;
+                "#,
+            )
+            .await
+            .unwrap();
+
+        let output_dir = tempfile::tempdir().unwrap();
+        let cli = cli::Command::parse_from([
+            "cli",
+            "export",
+            "--addr",
+            "127.0.0.1:4000",
+            "--output-dir",
+            &*output_dir.path().to_string_lossy(),
+            "--target",
+            "schema",
+        ]);
+        let mut cli_app = cli.build(LoggingOptions::default()).await?;
+        cli_app.start().await?;
+
+        instance.stop().await?;
+
+        let output_file = output_dir
+            .path()
+            .join("greptime")
+            .join("cli.export.create_table")
+            .join("create_tables.sql");
+        let res = std::fs::read_to_string(output_file).unwrap();
+        let expect = r#"CREATE TABLE IF NOT EXISTS "a.b.c" (
+  "ts" TIMESTAMP(3) NOT NULL,
+  TIME INDEX ("ts")
+)
+
+ENGINE=mito
+;
+"#;
+        assert_eq!(res.trim(), expect.trim());
+
+        Ok(())
    }
 }
--- a/src/cmd/src/datanode.rs
+++ b/src/cmd/src/datanode.rs
@@ -16,10 +16,12 @@ use std::sync::Arc;
 use std::time::Duration;

 use async_trait::async_trait;
+use cache::build_datanode_cache_registry;
 use catalog::kvbackend::MetaKvBackend;
 use clap::Parser;
 use common_base::Plugins;
 use common_config::Configurable;
+use common_meta::cache::LayeredCacheRegistryBuilder;
 use common_telemetry::logging::TracingOptions;
 use common_telemetry::{info, warn};
 use common_version::{short_version, version};
@@ -57,10 +59,6 @@ impl Instance {
        }
    }

-    pub fn datanode_mut(&mut self) -> &mut Datanode {
-        &mut self.datanode
-    }
-
    pub fn datanode(&self) -> &Datanode {
        &self.datanode
    }
@@ -300,9 +298,17 @@ impl StartCommand {
            client: meta_client.clone(),
        });

+        // Builds cache registry for datanode.
+        let layered_cache_registry = Arc::new(
+            LayeredCacheRegistryBuilder::default()
+                .add_cache_registry(build_datanode_cache_registry(meta_backend.clone()))
+                .build(),
+        );
+
        let mut datanode = DatanodeBuilder::new(opts.clone(), plugins)
            .with_meta_client(meta_client)
            .with_kv_backend(meta_backend)
+            .with_cache_registry(layered_cache_registry)
            .build()
            .await
            .context(StartDatanodeSnafu)?;
--- a/src/cmd/src/error.rs
+++ b/src/cmd/src/error.rs
@@ -114,6 +114,20 @@ pub enum Error {
        source: frontend::error::Error,
    },

+    #[snafu(display("Failed to build cli"))]
+    BuildCli {
+        #[snafu(implicit)]
+        location: Location,
+        source: BoxedError,
+    },
+
+    #[snafu(display("Failed to start cli"))]
+    StartCli {
+        #[snafu(implicit)]
+        location: Location,
+        source: BoxedError,
+    },
+
    #[snafu(display("Failed to build meta server"))]
    BuildMetaServer {
        #[snafu(implicit)]
@@ -346,6 +360,8 @@ impl ErrorExt for Error {
            Error::ShutdownMetaServer { source, .. } => source.status_code(),
            Error::BuildMetaServer { source, .. } => source.status_code(),
            Error::UnsupportedSelectorType { source, .. } => source.status_code(),
+            Error::BuildCli { source, .. } => source.status_code(),
+            Error::StartCli { source, .. } => source.status_code(),

            Error::InitMetadata { source, .. } | Error::InitDdlManager { source, .. } => {
                source.status_code()
--- a/src/cmd/src/flownode.rs
+++ b/src/cmd/src/flownode.rs
@@ -15,6 +15,7 @@
 use std::sync::Arc;

 use cache::{build_fundamental_cache_registry, with_default_composite_cache_registry};
+use catalog::information_extension::DistributedInformationExtension;
 use catalog::kvbackend::{CachedKvBackendBuilder, KvBackendCatalogManager, MetaKvBackend};
 use clap::Parser;
 use client::client_manager::NodeClients;
@@ -22,6 +23,7 @@ use common_base::Plugins;
 use common_config::Configurable;
 use common_grpc::channel_manager::ChannelConfig;
 use common_meta::cache::{CacheRegistryBuilder, LayeredCacheRegistryBuilder};
+use common_meta::heartbeat::handler::invalidate_table_cache::InvalidateCacheHandler;
 use common_meta::heartbeat::handler::parse_mailbox_message::ParseMailboxMessageHandler;
 use common_meta::heartbeat::handler::HandlerGroupExecutor;
 use common_meta::key::flow::FlowMetadataManager;
@@ -30,7 +32,6 @@ use common_telemetry::info;
 use common_telemetry::logging::TracingOptions;
 use common_version::{short_version, version};
 use flow::{FlownodeBuilder, FlownodeInstance, FrontendInvoker};
-use frontend::heartbeat::handler::invalidate_table_cache::InvalidateTableCacheHandler;
 use meta_client::{MetaClientOptions, MetaClientType};
 use servers::Mode;
 use snafu::{OptionExt, ResultExt};
@@ -41,7 +42,7 @@ use crate::error::{
    MissingConfigSnafu, Result, ShutdownFlownodeSnafu, StartFlownodeSnafu,
 };
 use crate::options::{GlobalOptions, GreptimeOptions};
-use crate::{log_versions, App, DistributedInformationExtension};
+use crate::{log_versions, App};

 pub const APP_NAME: &str = "greptime-flownode";

@@ -62,10 +63,6 @@ impl Instance {
        }
    }

-    pub fn flownode_mut(&mut self) -> &mut FlownodeInstance {
-        &mut self.flownode
-    }
-
    pub fn flownode(&self) -> &FlownodeInstance {
        &self.flownode
    }
@@ -288,9 +285,7 @@ impl StartCommand {

        let executor = HandlerGroupExecutor::new(vec![
            Arc::new(ParseMailboxMessageHandler),
-            Arc::new(InvalidateTableCacheHandler::new(
-                layered_cache_registry.clone(),
-            )),
+            Arc::new(InvalidateCacheHandler::new(layered_cache_registry.clone())),
        ]);

        let heartbeat_task = flow::heartbeat::HeartbeatTask::new(
--- a/src/cmd/src/frontend.rs
+++ b/src/cmd/src/frontend.rs
@@ -17,6 +17,7 @@ use std::time::Duration;

 use async_trait::async_trait;
 use cache::{build_fundamental_cache_registry, with_default_composite_cache_registry};
+use catalog::information_extension::DistributedInformationExtension;
 use catalog::kvbackend::{CachedKvBackendBuilder, KvBackendCatalogManager, MetaKvBackend};
 use clap::Parser;
 use client::client_manager::NodeClients;
@@ -24,13 +25,13 @@ use common_base::Plugins;
 use common_config::Configurable;
 use common_grpc::channel_manager::ChannelConfig;
 use common_meta::cache::{CacheRegistryBuilder, LayeredCacheRegistryBuilder};
+use common_meta::heartbeat::handler::invalidate_table_cache::InvalidateCacheHandler;
 use common_meta::heartbeat::handler::parse_mailbox_message::ParseMailboxMessageHandler;
 use common_meta::heartbeat::handler::HandlerGroupExecutor;
 use common_telemetry::info;
 use common_telemetry::logging::TracingOptions;
 use common_time::timezone::set_default_timezone;
 use common_version::{short_version, version};
-use frontend::heartbeat::handler::invalidate_table_cache::InvalidateTableCacheHandler;
 use frontend::heartbeat::HeartbeatTask;
 use frontend::instance::builder::FrontendBuilder;
 use frontend::instance::{FrontendInstance, Instance as FeInstance};
@@ -46,7 +47,7 @@ use crate::error::{
    Result, StartFrontendSnafu,
 };
 use crate::options::{GlobalOptions, GreptimeOptions};
-use crate::{log_versions, App, DistributedInformationExtension};
+use crate::{log_versions, App};

 type FrontendOptions = GreptimeOptions<frontend::frontend::FrontendOptions>;

@@ -328,9 +329,7 @@ impl StartCommand {

        let executor = HandlerGroupExecutor::new(vec![
            Arc::new(ParseMailboxMessageHandler),
-            Arc::new(InvalidateTableCacheHandler::new(
-                layered_cache_registry.clone(),
-            )),
+            Arc::new(InvalidateCacheHandler::new(layered_cache_registry.clone())),
        ]);

        let heartbeat_task = HeartbeatTask::new(
--- a/src/cmd/src/lib.rs
+++ b/src/cmd/src/lib.rs
@@ -15,17 +15,7 @@
 #![feature(assert_matches, let_chains)]

 use async_trait::async_trait;
-use catalog::information_schema::InformationExtension;
-use client::api::v1::meta::ProcedureStatus;
-use common_error::ext::BoxedError;
-use common_meta::cluster::{ClusterInfo, NodeInfo};
-use common_meta::datanode::RegionStat;
-use common_meta::ddl::{ExecutorContext, ProcedureExecutor};
-use common_meta::rpc::procedure;
-use common_procedure::{ProcedureInfo, ProcedureState};
 use common_telemetry::{error, info};
-use meta_client::MetaClientRef;
-use snafu::ResultExt;

 use crate::error::Result;

@@ -130,69 +120,3 @@ fn log_env_flags() {
        info!("argument: {}", argument);
    }
 }
-
-pub struct DistributedInformationExtension {
-    meta_client: MetaClientRef,
-}
-
-impl DistributedInformationExtension {
-    pub fn new(meta_client: MetaClientRef) -> Self {
-        Self { meta_client }
-    }
-}
-
-#[async_trait::async_trait]
-impl InformationExtension for DistributedInformationExtension {
-    type Error = catalog::error::Error;
-
-    async fn nodes(&self) -> std::result::Result<Vec<NodeInfo>, Self::Error> {
-        self.meta_client
-            .list_nodes(None)
-            .await
-            .map_err(BoxedError::new)
-            .context(catalog::error::ListNodesSnafu)
-    }
-
-    async fn procedures(&self) -> std::result::Result<Vec<(String, ProcedureInfo)>, Self::Error> {
-        let procedures = self
-            .meta_client
-            .list_procedures(&ExecutorContext::default())
-            .await
-            .map_err(BoxedError::new)
-            .context(catalog::error::ListProceduresSnafu)?
-            .procedures;
-        let mut result = Vec::with_capacity(procedures.len());
-        for procedure in procedures {
-            let pid = match procedure.id {
-                Some(pid) => pid,
-                None => return catalog::error::ProcedureIdNotFoundSnafu {}.fail(),
-            };
-            let pid = procedure::pb_pid_to_pid(&pid)
-                .map_err(BoxedError::new)
-                .context(catalog::error::ConvertProtoDataSnafu)?;
-            let status = ProcedureStatus::try_from(procedure.status)
-                .map(|v| v.as_str_name())
-                .unwrap_or("Unknown")
-                .to_string();
-            let procedure_info = ProcedureInfo {
-                id: pid,
-                type_name: procedure.type_name,
-                start_time_ms: procedure.start_time_ms,
-                end_time_ms: procedure.end_time_ms,
-                state: ProcedureState::Running,
-                lock_keys: procedure.lock_keys,
-            };
-            result.push((status, procedure_info));
-        }
-
-        Ok(result)
-    }
-
-    async fn region_stats(&self) -> std::result::Result<Vec<RegionStat>, Self::Error> {
-        self.meta_client
-            .list_region_stats()
-            .await
-            .map_err(BoxedError::new)
-            .context(catalog::error::ListRegionStatsSnafu)
-    }
-}
--- a/src/cmd/src/standalone.rs
+++ b/src/cmd/src/standalone.rs
@@ -497,6 +497,7 @@ impl StartCommand {

        let datanode = DatanodeBuilder::new(dn_opts, plugins.clone())
            .with_kv_backend(kv_backend.clone())
+            .with_cache_registry(layered_cache_registry.clone())
            .build()
            .await
            .context(StartDatanodeSnafu)?;
--- a/src/cmd/tests/load_config_test.rs
+++ b/src/cmd/tests/load_config_test.rs
@@ -69,7 +69,6 @@ fn test_load_datanode_example_config() {
            region_engine: vec![
                RegionEngineConfig::Mito(MitoConfig {
                    auto_flush_interval: Duration::from_secs(3600),
-                    scan_parallelism: 0,
                    experimental_write_cache_ttl: Some(Duration::from_secs(60 * 60 * 8)),
                    ..Default::default()
                }),
@@ -205,7 +204,6 @@ fn test_load_standalone_example_config() {
                RegionEngineConfig::Mito(MitoConfig {
                    auto_flush_interval: Duration::from_secs(3600),
                    experimental_write_cache_ttl: Some(Duration::from_secs(60 * 60 * 8)),
-                    scan_parallelism: 0,
                    ..Default::default()
                }),
                RegionEngineConfig::File(EngineConfig {}),
--- a/src/common/base/Cargo.toml
+++ b/src/common/base/Cargo.toml
@@ -17,6 +17,7 @@ common-macro.workspace = true
 futures.workspace = true
 paste = "1.0"
 pin-project.workspace = true
+rand.workspace = true
 serde = { version = "1.0", features = ["derive"] }
 snafu.workspace = true
 tokio.workspace = true
--- a/src/common/base/src/range_read.rs
+++ b/src/common/base/src/range_read.rs
@@ -36,6 +36,11 @@ pub struct Metadata {
 /// `RangeReader` reads a range of bytes from a source.
 #[async_trait]
 pub trait RangeReader: Send + Unpin {
+    /// Sets the file size hint for the reader.
+    ///
+    /// It's used to optimize the reading process by reducing the number of remote requests.
+    fn with_file_size_hint(&mut self, file_size_hint: u64);
+
    /// Returns the metadata of the source.
    async fn metadata(&mut self) -> io::Result<Metadata>;

@@ -70,6 +75,10 @@ pub trait RangeReader: Send + Unpin {

 #[async_trait]
 impl<R: ?Sized + RangeReader> RangeReader for &mut R {
+    fn with_file_size_hint(&mut self, file_size_hint: u64) {
+        (*self).with_file_size_hint(file_size_hint)
+    }
+
    async fn metadata(&mut self) -> io::Result<Metadata> {
        (*self).metadata().await
    }
@@ -186,15 +195,17 @@ impl<R: RangeReader + 'static> AsyncRead for AsyncReadAdapter<R> {

 #[async_trait]
 impl RangeReader for Vec<u8> {
+    fn with_file_size_hint(&mut self, _file_size_hint: u64) {
+        // do nothing
+    }
+
    async fn metadata(&mut self) -> io::Result<Metadata> {
        Ok(Metadata {
            content_length: self.len() as u64,
        })
    }

-    async fn read(&mut self, mut range: Range<u64>) -> io::Result<Bytes> {
-        range.end = range.end.min(self.len() as u64);
-
+    async fn read(&mut self, range: Range<u64>) -> io::Result<Bytes> {
        let bytes = Bytes::copy_from_slice(&self[range.start as usize..range.end as usize]);
        Ok(bytes)
    }
@@ -222,6 +233,10 @@ impl FileReader {

 #[async_trait]
 impl RangeReader for FileReader {
+    fn with_file_size_hint(&mut self, _file_size_hint: u64) {
+        // do nothing
+    }
+
    async fn metadata(&mut self) -> io::Result<Metadata> {
        Ok(Metadata {
            content_length: self.content_length,
--- a/src/common/base/src/readable_size.rs
+++ b/src/common/base/src/readable_size.rs
@@ -19,7 +19,7 @@ pub const GIB: u64 = MIB * BINARY_DATA_MAGNITUDE;
 pub const TIB: u64 = GIB * BINARY_DATA_MAGNITUDE;
 pub const PIB: u64 = TIB * BINARY_DATA_MAGNITUDE;

-#[derive(Clone, Copy, PartialEq, Eq, Ord, PartialOrd)]
+#[derive(Clone, Copy, PartialEq, Eq, Ord, PartialOrd, Default)]
 pub struct ReadableSize(pub u64);

 impl ReadableSize {
--- a/src/common/catalog/Cargo.toml
+++ b/src/common/catalog/Cargo.toml
@@ -8,10 +8,5 @@ license.workspace = true
 workspace = true

 [dependencies]
-common-error.workspace = true
-common-macro.workspace = true
-snafu.workspace = true

 [dev-dependencies]
-chrono.workspace = true
-tokio.workspace = true
--- a/src/common/datasource/Cargo.toml
+++ b/src/common/datasource/Cargo.toml
@@ -48,5 +48,4 @@ url = "2.3"
 [dev-dependencies]
 common-telemetry.workspace = true
 common-test-util.workspace = true
-dotenv.workspace = true
 uuid.workspace = true
--- a/src/common/datasource/src/object_store/fs.rs
+++ b/src/common/datasource/src/object_store/fs.rs
@@ -27,7 +27,7 @@ pub fn build_fs_backend(root: &str) -> Result<ObjectStore> {
            DefaultLoggingInterceptor,
        ))
        .layer(object_store::layers::TracingLayer)
-        .layer(object_store::layers::PrometheusMetricsLayer::new(true))
+        .layer(object_store::layers::build_prometheus_metrics_layer(true))
        .finish();
    Ok(object_store)
 }
--- a/src/common/datasource/src/object_store/s3.rs
+++ b/src/common/datasource/src/object_store/s3.rs
@@ -89,7 +89,7 @@ pub fn build_s3_backend(
            DefaultLoggingInterceptor,
        ))
        .layer(object_store::layers::TracingLayer)
-        .layer(object_store::layers::PrometheusMetricsLayer::new(true))
+        .layer(object_store::layers::build_prometheus_metrics_layer(true))
        .finish())
 }

--- a/src/common/frontend/Cargo.toml
+++ b/src/common/frontend/Cargo.toml
@@ -5,12 +5,7 @@ edition.workspace = true
 license.workspace = true

 [dependencies]
-api.workspace = true
 async-trait.workspace = true
-common-base.workspace = true
 common-error.workspace = true
 common-macro.workspace = true
-common-query.workspace = true
-session.workspace = true
 snafu.workspace = true
-sql.workspace = true
--- a/src/common/function/Cargo.toml
+++ b/src/common/function/Cargo.toml
@@ -51,6 +51,5 @@ wkt = { version = "0.11", optional = true }

 [dev-dependencies]
 approx = "0.5"
-ron = "0.7"
 serde = { version = "1.0", features = ["derive"] }
 tokio.workspace = true
--- a/src/common/function/src/scalars/vector.rs
+++ b/src/common/function/src/scalars/vector.rs
@@ -14,6 +14,9 @@

 mod convert;
 mod distance;
+pub(crate) mod impl_conv;
+mod scalar_add;
+mod scalar_mul;

 use std::sync::Arc;

@@ -31,5 +34,9 @@ impl VectorFunction {
        registry.register(Arc::new(distance::CosDistanceFunction));
        registry.register(Arc::new(distance::DotProductFunction));
        registry.register(Arc::new(distance::L2SqDistanceFunction));
+
+        // scalar calculation
+        registry.register(Arc::new(scalar_add::ScalarAddFunction));
+        registry.register(Arc::new(scalar_mul::ScalarMulFunction));
    }
 }
--- a/src/common/function/src/scalars/vector/distance.rs
+++ b/src/common/function/src/scalars/vector/distance.rs
@@ -18,18 +18,17 @@ mod l2sq;

 use std::borrow::Cow;
 use std::fmt::Display;
-use std::sync::Arc;

 use common_query::error::{InvalidFuncArgsSnafu, Result};
 use common_query::prelude::Signature;
 use datatypes::prelude::ConcreteDataType;
 use datatypes::scalars::ScalarVectorBuilder;
-use datatypes::value::ValueRef;
-use datatypes::vectors::{Float32VectorBuilder, MutableVector, Vector, VectorRef};
+use datatypes::vectors::{Float32VectorBuilder, MutableVector, VectorRef};
 use snafu::ensure;

 use crate::function::{Function, FunctionContext};
 use crate::helper;
+use crate::scalars::vector::impl_conv::{as_veclit, as_veclit_if_const};

 macro_rules! define_distance_function {
    ($StructName:ident, $display_name:expr, $similarity_method:path) => {
@@ -80,17 +79,17 @@ macro_rules! define_distance_function {
                    return Ok(result.to_vector());
                }

-                let arg0_const = parse_if_constant_string(arg0)?;
-                let arg1_const = parse_if_constant_string(arg1)?;
+                let arg0_const = as_veclit_if_const(arg0)?;
+                let arg1_const = as_veclit_if_const(arg1)?;

                for i in 0..size {
                    let vec0 = match arg0_const.as_ref() {
-                        Some(a) => Some(Cow::Borrowed(a.as_slice())),
-                        None => as_vector(arg0.get_ref(i))?,
+                        Some(a) => Some(Cow::Borrowed(a.as_ref())),
+                        None => as_veclit(arg0.get_ref(i))?,
                    };
                    let vec1 = match arg1_const.as_ref() {
-                        Some(b) => Some(Cow::Borrowed(b.as_slice())),
-                        None => as_vector(arg1.get_ref(i))?,
+                        Some(b) => Some(Cow::Borrowed(b.as_ref())),
+                        None => as_veclit(arg1.get_ref(i))?,
                    };

                    if let (Some(vec0), Some(vec1)) = (vec0, vec1) {
@@ -129,98 +128,6 @@ define_distance_function!(CosDistanceFunction, "vec_cos_distance", cos::cos);
 define_distance_function!(L2SqDistanceFunction, "vec_l2sq_distance", l2sq::l2sq);
 define_distance_function!(DotProductFunction, "vec_dot_product", dot::dot);

-/// Parse a vector value if the value is a constant string.
-fn parse_if_constant_string(arg: &Arc<dyn Vector>) -> Result<Option<Vec<f32>>> {
-    if !arg.is_const() {
-        return Ok(None);
-    }
-    if arg.data_type() != ConcreteDataType::string_datatype() {
-        return Ok(None);
-    }
-    arg.get_ref(0)
-        .as_string()
-        .unwrap() // Safe: checked if it is a string
-        .map(parse_f32_vector_from_string)
-        .transpose()
-}
-
-/// Convert a value to a vector value.
-/// Supported data types are binary and string.
-fn as_vector(arg: ValueRef<'_>) -> Result<Option<Cow<'_, [f32]>>> {
-    match arg.data_type() {
-        ConcreteDataType::Binary(_) => arg
-            .as_binary()
-            .unwrap() // Safe: checked if it is a binary
-            .map(binary_as_vector)
-            .transpose(),
-        ConcreteDataType::String(_) => arg
-            .as_string()
-            .unwrap() // Safe: checked if it is a string
-            .map(|s| Ok(Cow::Owned(parse_f32_vector_from_string(s)?)))
-            .transpose(),
-        ConcreteDataType::Null(_) => Ok(None),
-        _ => InvalidFuncArgsSnafu {
-            err_msg: format!("Unsupported data type: {:?}", arg.data_type()),
-        }
-        .fail(),
-    }
-}
-
-/// Convert a u8 slice to a vector value.
-fn binary_as_vector(bytes: &[u8]) -> Result<Cow<'_, [f32]>> {
-    if bytes.len() % std::mem::size_of::<f32>() != 0 {
-        return InvalidFuncArgsSnafu {
-            err_msg: format!("Invalid binary length of vector: {}", bytes.len()),
-        }
-        .fail();
-    }
-
-    if cfg!(target_endian = "little") {
-        Ok(unsafe {
-            let vec = std::slice::from_raw_parts(
-                bytes.as_ptr() as *const f32,
-                bytes.len() / std::mem::size_of::<f32>(),
-            );
-            Cow::Borrowed(vec)
-        })
-    } else {
-        let v = bytes
-            .chunks_exact(std::mem::size_of::<f32>())
-            .map(|chunk| f32::from_le_bytes(chunk.try_into().unwrap()))
-            .collect::<Vec<f32>>();
-        Ok(Cow::Owned(v))
-    }
-}
-
-/// Parse a string to a vector value.
-/// Valid inputs are strings like "[1.0, 2.0, 3.0]".
-fn parse_f32_vector_from_string(s: &str) -> Result<Vec<f32>> {
-    let trimmed = s.trim();
-    if !trimmed.starts_with('[') || !trimmed.ends_with(']') {
-        return InvalidFuncArgsSnafu {
-            err_msg: format!(
-                "Failed to parse {s} to Vector value: not properly enclosed in brackets"
-            ),
-        }
-        .fail();
-    }
-    let content = trimmed[1..trimmed.len() - 1].trim();
-    if content.is_empty() {
-        return Ok(Vec::new());
-    }
-
-    content
-        .split(',')
-        .map(|s| s.trim().parse::<f32>())
-        .collect::<std::result::Result<_, _>>()
-        .map_err(|e| {
-            InvalidFuncArgsSnafu {
-                err_msg: format!("Failed to parse {s} to Vector value: {e}"),
-            }
-            .build()
-        })
-}
-
 #[cfg(test)]
 mod tests {
    use std::sync::Arc;
@@ -456,27 +363,4 @@ mod tests {
            assert!(result.is_err());
        }
    }
-
-    #[test]
-    fn test_parse_vector_from_string() {
-        let result = parse_f32_vector_from_string("[1.0, 2.0, 3.0]").unwrap();
-        assert_eq!(result, vec![1.0, 2.0, 3.0]);
-
-        let result = parse_f32_vector_from_string("[]").unwrap();
-        assert_eq!(result, Vec::<f32>::new());
-
-        let result = parse_f32_vector_from_string("[1.0, a, 3.0]");
-        assert!(result.is_err());
-    }
-
-    #[test]
-    fn test_binary_as_vector() {
-        let bytes = [0, 0, 128, 63];
-        let result = binary_as_vector(&bytes).unwrap();
-        assert_eq!(result.as_ref(), &[1.0]);
-
-        let invalid_bytes = [0, 0, 128];
-        let result = binary_as_vector(&invalid_bytes);
-        assert!(result.is_err());
-    }
 }
--- a/src/common/function/src/scalars/vector/impl_conv.rs
+++ b/src/common/function/src/scalars/vector/impl_conv.rs
@@ -0,0 +1,155 @@
+// Copyright 2023 Greptime Team
+//
+// Licensed under the Apache License, Version 2.0 (the "License");
+// you may not use this file except in compliance with the License.
+// You may obtain a copy of the License at
+//
+//     http://www.apache.org/licenses/LICENSE-2.0
+//
+// Unless required by applicable law or agreed to in writing, software
+// distributed under the License is distributed on an "AS IS" BASIS,
+// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+// See the License for the specific language governing permissions and
+// limitations under the License.
+
+use std::borrow::Cow;
+use std::sync::Arc;
+
+use common_query::error::{InvalidFuncArgsSnafu, Result};
+use datatypes::prelude::ConcreteDataType;
+use datatypes::value::ValueRef;
+use datatypes::vectors::Vector;
+
+/// Convert a constant string or binary literal to a vector literal.
+pub fn as_veclit_if_const(arg: &Arc<dyn Vector>) -> Result<Option<Cow<'_, [f32]>>> {
+    if !arg.is_const() {
+        return Ok(None);
+    }
+    if arg.data_type() != ConcreteDataType::string_datatype()
+        && arg.data_type() != ConcreteDataType::binary_datatype()
+    {
+        return Ok(None);
+    }
+    as_veclit(arg.get_ref(0))
+}
+
+/// Convert a string or binary literal to a vector literal.
+pub fn as_veclit(arg: ValueRef<'_>) -> Result<Option<Cow<'_, [f32]>>> {
+    match arg.data_type() {
+        ConcreteDataType::Binary(_) => arg
+            .as_binary()
+            .unwrap() // Safe: checked if it is a binary
+            .map(binlit_as_veclit)
+            .transpose(),
+        ConcreteDataType::String(_) => arg
+            .as_string()
+            .unwrap() // Safe: checked if it is a string
+            .map(|s| Ok(Cow::Owned(parse_veclit_from_strlit(s)?)))
+            .transpose(),
+        ConcreteDataType::Null(_) => Ok(None),
+        _ => InvalidFuncArgsSnafu {
+            err_msg: format!("Unsupported data type: {:?}", arg.data_type()),
+        }
+        .fail(),
+    }
+}
+
+/// Convert a u8 slice to a vector literal.
+pub fn binlit_as_veclit(bytes: &[u8]) -> Result<Cow<'_, [f32]>> {
+    if bytes.len() % std::mem::size_of::<f32>() != 0 {
+        return InvalidFuncArgsSnafu {
+            err_msg: format!("Invalid binary length of vector: {}", bytes.len()),
+        }
+        .fail();
+    }
+
+    if cfg!(target_endian = "little") {
+        Ok(unsafe {
+            let vec = std::slice::from_raw_parts(
+                bytes.as_ptr() as *const f32,
+                bytes.len() / std::mem::size_of::<f32>(),
+            );
+            Cow::Borrowed(vec)
+        })
+    } else {
+        let v = bytes
+            .chunks_exact(std::mem::size_of::<f32>())
+            .map(|chunk| f32::from_le_bytes(chunk.try_into().unwrap()))
+            .collect::<Vec<f32>>();
+        Ok(Cow::Owned(v))
+    }
+}
+
+/// Parse a string literal to a vector literal.
+/// Valid inputs are strings like "[1.0, 2.0, 3.0]".
+pub fn parse_veclit_from_strlit(s: &str) -> Result<Vec<f32>> {
+    let trimmed = s.trim();
+    if !trimmed.starts_with('[') || !trimmed.ends_with(']') {
+        return InvalidFuncArgsSnafu {
+            err_msg: format!(
+                "Failed to parse {s} to Vector value: not properly enclosed in brackets"
+            ),
+        }
+        .fail();
+    }
+    let content = trimmed[1..trimmed.len() - 1].trim();
+    if content.is_empty() {
+        return Ok(Vec::new());
+    }
+
+    content
+        .split(',')
+        .map(|s| s.trim().parse::<f32>())
+        .collect::<std::result::Result<_, _>>()
+        .map_err(|e| {
+            InvalidFuncArgsSnafu {
+                err_msg: format!("Failed to parse {s} to Vector value: {e}"),
+            }
+            .build()
+        })
+}
+
+/// Convert a vector literal to a binary literal.
+pub fn veclit_to_binlit(vec: &[f32]) -> Vec<u8> {
+    if cfg!(target_endian = "little") {
+        unsafe {
+            std::slice::from_raw_parts(vec.as_ptr() as *const u8, std::mem::size_of_val(vec))
+                .to_vec()
+        }
+    } else {
+        let mut bytes = Vec::with_capacity(std::mem::size_of_val(vec));
+        for e in vec {
+            bytes.extend_from_slice(&e.to_le_bytes());
+        }
+        bytes
+    }
+}
+
+#[cfg(test)]
+mod tests {
+    use super::*;
+
+    #[test]
+    fn test_parse_veclit_from_strlit() {
+        let result = parse_veclit_from_strlit("[1.0, 2.0, 3.0]").unwrap();
+        assert_eq!(result, vec![1.0, 2.0, 3.0]);
+
+        let result = parse_veclit_from_strlit("[]").unwrap();
+        assert_eq!(result, Vec::<f32>::new());
+
+        let result = parse_veclit_from_strlit("[1.0, a, 3.0]");
+        assert!(result.is_err());
+    }
+
+    #[test]
+    fn test_binlit_as_veclit() {
+        let vec = &[1.0, 2.0, 3.0];
+        let bytes = veclit_to_binlit(vec);
+        let result = binlit_as_veclit(&bytes).unwrap();
+        assert_eq!(result.as_ref(), vec);
+
+        let invalid_bytes = [0, 0, 128];
+        let result = binlit_as_veclit(&invalid_bytes);
+        assert!(result.is_err());
+    }
+}
--- a/src/common/function/src/scalars/vector/scalar_add.rs
+++ b/src/common/function/src/scalars/vector/scalar_add.rs
@@ -0,0 +1,173 @@
+// Copyright 2023 Greptime Team
+//
+// Licensed under the Apache License, Version 2.0 (the "License");
+// you may not use this file except in compliance with the License.
+// You may obtain a copy of the License at
+//
+//     http://www.apache.org/licenses/LICENSE-2.0
+//
+// Unless required by applicable law or agreed to in writing, software
+// distributed under the License is distributed on an "AS IS" BASIS,
+// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+// See the License for the specific language governing permissions and
+// limitations under the License.
+
+use std::borrow::Cow;
+use std::fmt::Display;
+
+use common_query::error::{InvalidFuncArgsSnafu, Result};
+use common_query::prelude::Signature;
+use datatypes::prelude::ConcreteDataType;
+use datatypes::scalars::ScalarVectorBuilder;
+use datatypes::vectors::{BinaryVectorBuilder, MutableVector, VectorRef};
+use nalgebra::DVectorView;
+use snafu::ensure;
+
+use crate::function::{Function, FunctionContext};
+use crate::helper;
+use crate::scalars::vector::impl_conv::{as_veclit, as_veclit_if_const, veclit_to_binlit};
+
+const NAME: &str = "vec_scalar_add";
+
+/// Adds a scalar to each element of a vector.
+///
+/// # Example
+///
+/// ```sql
+/// SELECT vec_to_string(vec_scalar_add(1, "[1, 2, 3]")) as result;
+///
+/// +---------+
+/// | result  |
+/// +---------+
+/// | [2,3,4] |
+/// +---------+
+///
+/// -- Negative scalar to simulate subtraction
+/// SELECT vec_to_string(vec_scalar_add(-1, "[1, 2, 3]")) as result;
+///
+/// +---------+
+/// | result  |
+/// +---------+
+/// | [0,1,2] |
+/// +---------+
+/// ```
+#[derive(Debug, Clone, Default)]
+pub struct ScalarAddFunction;
+
+impl Function for ScalarAddFunction {
+    fn name(&self) -> &str {
+        NAME
+    }
+
+    fn return_type(&self, _input_types: &[ConcreteDataType]) -> Result<ConcreteDataType> {
+        Ok(ConcreteDataType::binary_datatype())
+    }
+
+    fn signature(&self) -> Signature {
+        helper::one_of_sigs2(
+            vec![ConcreteDataType::float64_datatype()],
+            vec![
+                ConcreteDataType::string_datatype(),
+                ConcreteDataType::binary_datatype(),
+            ],
+        )
+    }
+
+    fn eval(&self, _func_ctx: FunctionContext, columns: &[VectorRef]) -> Result<VectorRef> {
+        ensure!(
+            columns.len() == 2,
+            InvalidFuncArgsSnafu {
+                err_msg: format!(
+                    "The length of the args is not correct, expect exactly two, have: {}",
+                    columns.len()
+                ),
+            }
+        );
+        let arg0 = &columns[0];
+        let arg1 = &columns[1];
+
+        let len = arg0.len();
+        let mut result = BinaryVectorBuilder::with_capacity(len);
+        if len == 0 {
+            return Ok(result.to_vector());
+        }
+
+        let arg1_const = as_veclit_if_const(arg1)?;
+
+        for i in 0..len {
+            let arg0 = arg0.get(i).as_f64_lossy();
+            let Some(arg0) = arg0 else {
+                result.push_null();
+                continue;
+            };
+
+            let arg1 = match arg1_const.as_ref() {
+                Some(arg1) => Some(Cow::Borrowed(arg1.as_ref())),
+                None => as_veclit(arg1.get_ref(i))?,
+            };
+            let Some(arg1) = arg1 else {
+                result.push_null();
+                continue;
+            };
+
+            let vec = DVectorView::from_slice(&arg1, arg1.len());
+            let vec_res = vec.add_scalar(arg0 as _);
+
+            let veclit = vec_res.as_slice();
+            let binlit = veclit_to_binlit(veclit);
+            result.push(Some(&binlit));
+        }
+
+        Ok(result.to_vector())
+    }
+}
+
+impl Display for ScalarAddFunction {
+    fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result {
+        write!(f, "{}", NAME.to_ascii_uppercase())
+    }
+}
+
+#[cfg(test)]
+mod tests {
+    use std::sync::Arc;
+
+    use datatypes::vectors::{Float32Vector, StringVector};
+
+    use super::*;
+
+    #[test]
+    fn test_scalar_add() {
+        let func = ScalarAddFunction;
+
+        let input0 = Arc::new(Float32Vector::from(vec![
+            Some(1.0),
+            Some(-1.0),
+            None,
+            Some(3.0),
+        ]));
+        let input1 = Arc::new(StringVector::from(vec![
+            Some("[1.0,2.0,3.0]".to_string()),
+            Some("[4.0,5.0,6.0]".to_string()),
+            Some("[7.0,8.0,9.0]".to_string()),
+            None,
+        ]));
+
+        let result = func
+            .eval(FunctionContext::default(), &[input0, input1])
+            .unwrap();
+
+        let result = result.as_ref();
+        assert_eq!(result.len(), 4);
+        assert_eq!(
+            result.get_ref(0).as_binary().unwrap(),
+            Some(veclit_to_binlit(&[2.0, 3.0, 4.0]).as_slice())
+        );
+        assert_eq!(
+            result.get_ref(1).as_binary().unwrap(),
+            Some(veclit_to_binlit(&[3.0, 4.0, 5.0]).as_slice())
+        );
+        assert!(result.get_ref(2).is_null());
+        assert!(result.get_ref(3).is_null());
+    }
+}
--- a/src/common/function/src/scalars/vector/scalar_mul.rs
+++ b/src/common/function/src/scalars/vector/scalar_mul.rs
@@ -0,0 +1,173 @@
+// Copyright 2023 Greptime Team
+//
+// Licensed under the Apache License, Version 2.0 (the "License");
+// you may not use this file except in compliance with the License.
+// You may obtain a copy of the License at
+//
+//     http://www.apache.org/licenses/LICENSE-2.0
+//
+// Unless required by applicable law or agreed to in writing, software
+// distributed under the License is distributed on an "AS IS" BASIS,
+// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+// See the License for the specific language governing permissions and
+// limitations under the License.
+
+use std::borrow::Cow;
+use std::fmt::Display;
+
+use common_query::error::{InvalidFuncArgsSnafu, Result};
+use common_query::prelude::Signature;
+use datatypes::prelude::ConcreteDataType;
+use datatypes::scalars::ScalarVectorBuilder;
+use datatypes::vectors::{BinaryVectorBuilder, MutableVector, VectorRef};
+use nalgebra::DVectorView;
+use snafu::ensure;
+
+use crate::function::{Function, FunctionContext};
+use crate::helper;
+use crate::scalars::vector::impl_conv::{as_veclit, as_veclit_if_const, veclit_to_binlit};
+
+const NAME: &str = "vec_scalar_mul";
+
+/// Multiples a scalar to each element of a vector.
+///
+/// # Example
+///
+/// ```sql
+/// SELECT vec_to_string(vec_scalar_mul(2, "[1, 2, 3]")) as result;
+///
+/// +---------+
+/// | result  |
+/// +---------+
+/// | [2,4,6] |
+/// +---------+
+///
+/// -- 1/scalar to simulate division
+/// SELECT vec_to_string(vec_scalar_mul(0.5, "[2, 4, 6]")) as result;
+///
+/// +---------+
+/// | result  |
+/// +---------+
+/// | [1,2,3] |
+/// +---------+
+/// ```
+#[derive(Debug, Clone, Default)]
+pub struct ScalarMulFunction;
+
+impl Function for ScalarMulFunction {
+    fn name(&self) -> &str {
+        NAME
+    }
+
+    fn return_type(&self, _input_types: &[ConcreteDataType]) -> Result<ConcreteDataType> {
+        Ok(ConcreteDataType::binary_datatype())
+    }
+
+    fn signature(&self) -> Signature {
+        helper::one_of_sigs2(
+            vec![ConcreteDataType::float64_datatype()],
+            vec![
+                ConcreteDataType::string_datatype(),
+                ConcreteDataType::binary_datatype(),
+            ],
+        )
+    }
+
+    fn eval(&self, _func_ctx: FunctionContext, columns: &[VectorRef]) -> Result<VectorRef> {
+        ensure!(
+            columns.len() == 2,
+            InvalidFuncArgsSnafu {
+                err_msg: format!(
+                    "The length of the args is not correct, expect exactly two, have: {}",
+                    columns.len()
+                ),
+            }
+        );
+        let arg0 = &columns[0];
+        let arg1 = &columns[1];
+
+        let len = arg0.len();
+        let mut result = BinaryVectorBuilder::with_capacity(len);
+        if len == 0 {
+            return Ok(result.to_vector());
+        }
+
+        let arg1_const = as_veclit_if_const(arg1)?;
+
+        for i in 0..len {
+            let arg0 = arg0.get(i).as_f64_lossy();
+            let Some(arg0) = arg0 else {
+                result.push_null();
+                continue;
+            };
+
+            let arg1 = match arg1_const.as_ref() {
+                Some(arg1) => Some(Cow::Borrowed(arg1.as_ref())),
+                None => as_veclit(arg1.get_ref(i))?,
+            };
+            let Some(arg1) = arg1 else {
+                result.push_null();
+                continue;
+            };
+
+            let vec = DVectorView::from_slice(&arg1, arg1.len());
+            let vec_res = vec.scale(arg0 as _);
+
+            let veclit = vec_res.as_slice();
+            let binlit = veclit_to_binlit(veclit);
+            result.push(Some(&binlit));
+        }
+
+        Ok(result.to_vector())
+    }
+}
+
+impl Display for ScalarMulFunction {
+    fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result {
+        write!(f, "{}", NAME.to_ascii_uppercase())
+    }
+}
+
+#[cfg(test)]
+mod tests {
+    use std::sync::Arc;
+
+    use datatypes::vectors::{Float32Vector, StringVector};
+
+    use super::*;
+
+    #[test]
+    fn test_scalar_mul() {
+        let func = ScalarMulFunction;
+
+        let input0 = Arc::new(Float32Vector::from(vec![
+            Some(2.0),
+            Some(-0.5),
+            None,
+            Some(3.0),
+        ]));
+        let input1 = Arc::new(StringVector::from(vec![
+            Some("[1.0,2.0,3.0]".to_string()),
+            Some("[8.0,10.0,12.0]".to_string()),
+            Some("[7.0,8.0,9.0]".to_string()),
+            None,
+        ]));
+
+        let result = func
+            .eval(FunctionContext::default(), &[input0, input1])
+            .unwrap();
+
+        let result = result.as_ref();
+        assert_eq!(result.len(), 4);
+        assert_eq!(
+            result.get_ref(0).as_binary().unwrap(),
+            Some(veclit_to_binlit(&[2.0, 4.0, 6.0]).as_slice())
+        );
+        assert_eq!(
+            result.get_ref(1).as_binary().unwrap(),
+            Some(veclit_to_binlit(&[-4.0, -5.0, -6.0]).as_slice())
+        );
+        assert!(result.get_ref(2).is_null());
+        assert!(result.get_ref(3).is_null());
+    }
+}
--- a/src/common/meta/src/cache.rs
+++ b/src/common/meta/src/cache.rs
@@ -24,7 +24,8 @@ pub use registry::{
    LayeredCacheRegistryBuilder, LayeredCacheRegistryRef,
 };
 pub use table::{
-    new_table_info_cache, new_table_name_cache, new_table_route_cache, new_view_info_cache,
-    TableInfoCache, TableInfoCacheRef, TableNameCache, TableNameCacheRef, TableRoute,
-    TableRouteCache, TableRouteCacheRef, ViewInfoCache, ViewInfoCacheRef,
+    new_schema_cache, new_table_info_cache, new_table_name_cache, new_table_route_cache,
+    new_table_schema_cache, new_view_info_cache, SchemaCache, SchemaCacheRef, TableInfoCache,
+    TableInfoCacheRef, TableNameCache, TableNameCacheRef, TableRoute, TableRouteCache,
+    TableRouteCacheRef, TableSchemaCache, TableSchemaCacheRef, ViewInfoCache, ViewInfoCacheRef,
 };
--- a/src/common/meta/src/cache/table.rs
+++ b/src/common/meta/src/cache/table.rs
@@ -12,12 +12,16 @@
 // See the License for the specific language governing permissions and
 // limitations under the License.

+mod schema;
 mod table_info;
 mod table_name;
 mod table_route;
+mod table_schema;
 mod view_info;

+pub use schema::{new_schema_cache, SchemaCache, SchemaCacheRef};
 pub use table_info::{new_table_info_cache, TableInfoCache, TableInfoCacheRef};
 pub use table_name::{new_table_name_cache, TableNameCache, TableNameCacheRef};
 pub use table_route::{new_table_route_cache, TableRoute, TableRouteCache, TableRouteCacheRef};
+pub use table_schema::{new_table_schema_cache, TableSchemaCache, TableSchemaCacheRef};
 pub use view_info::{new_view_info_cache, ViewInfoCache, ViewInfoCacheRef};
--- a/src/common/meta/src/cache/table/schema.rs
+++ b/src/common/meta/src/cache/table/schema.rs
@@ -0,0 +1,73 @@
+// Copyright 2023 Greptime Team
+//
+// Licensed under the Apache License, Version 2.0 (the "License");
+// you may not use this file except in compliance with the License.
+// You may obtain a copy of the License at
+//
+//     http://www.apache.org/licenses/LICENSE-2.0
+//
+// Unless required by applicable law or agreed to in writing, software
+// distributed under the License is distributed on an "AS IS" BASIS,
+// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+// See the License for the specific language governing permissions and
+// limitations under the License.
+
+use std::sync::Arc;
+
+use futures_util::future::BoxFuture;
+use moka::future::Cache;
+use snafu::OptionExt;
+
+use crate::cache::{CacheContainer, Initializer};
+use crate::error::ValueNotExistSnafu;
+use crate::instruction::CacheIdent;
+use crate::key::schema_name::{SchemaManager, SchemaName, SchemaNameKey, SchemaNameValue};
+use crate::kv_backend::KvBackendRef;
+
+pub type SchemaCache = CacheContainer<SchemaName, Arc<SchemaNameValue>, CacheIdent>;
+pub type SchemaCacheRef = Arc<SchemaCache>;
+
+/// Constructs a [SchemaCache].
+pub fn new_schema_cache(
+    name: String,
+    cache: Cache<SchemaName, Arc<SchemaNameValue>>,
+    kv_backend: KvBackendRef,
+) -> SchemaCache {
+    let schema_manager = SchemaManager::new(kv_backend.clone());
+    let init = init_factory(schema_manager);
+
+    CacheContainer::new(name, cache, Box::new(invalidator), init, Box::new(filter))
+}
+
+fn init_factory(schema_manager: SchemaManager) -> Initializer<SchemaName, Arc<SchemaNameValue>> {
+    Arc::new(move |schema_name| {
+        let manager = schema_manager.clone();
+        Box::pin(async move {
+            let schema_value = manager
+                .get(SchemaNameKey {
+                    catalog: &schema_name.catalog_name,
+                    schema: &schema_name.schema_name,
+                })
+                .await?
+                .context(ValueNotExistSnafu)?
+                .into_inner();
+            Ok(Some(Arc::new(schema_value)))
+        })
+    })
+}
+
+fn invalidator<'a>(
+    cache: &'a Cache<SchemaName, Arc<SchemaNameValue>>,
+    ident: &'a CacheIdent,
+) -> BoxFuture<'a, crate::error::Result<()>> {
+    Box::pin(async move {
+        if let CacheIdent::SchemaName(schema_name) = ident {
+            cache.invalidate(schema_name).await
+        }
+        Ok(())
+    })
+}
+
+fn filter(ident: &CacheIdent) -> bool {
+    matches!(ident, CacheIdent::SchemaName(_))
+}
--- a/src/common/meta/src/cache/table/table_route.rs
+++ b/src/common/meta/src/cache/table/table_route.rs
@@ -49,14 +49,6 @@ impl TableRoute {
            TableRoute::Logical(_) => None,
        }
    }
-
-    /// Returns [LogicalTableRouteValue] reference if it's [TableRoute::Logical]; Otherwise it returns [None].
-    pub fn as_logical_table_route_ref(&self) -> Option<&Arc<LogicalTableRouteValue>> {
-        match self {
-            TableRoute::Physical(_) => None,
-            TableRoute::Logical(table_route) => Some(table_route),
-        }
-    }
 }

 /// [TableRouteCache] caches the [TableId] to [TableRoute] mapping.
--- a/src/common/meta/src/cache/table/table_schema.rs
+++ b/src/common/meta/src/cache/table/table_schema.rs
@@ -0,0 +1,76 @@
+// Copyright 2023 Greptime Team
+//
+// Licensed under the Apache License, Version 2.0 (the "License");
+// you may not use this file except in compliance with the License.
+// You may obtain a copy of the License at
+//
+//     http://www.apache.org/licenses/LICENSE-2.0
+//
+// Unless required by applicable law or agreed to in writing, software
+// distributed under the License is distributed on an "AS IS" BASIS,
+// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+// See the License for the specific language governing permissions and
+// limitations under the License.
+
+//! Cache for table id to schema name mapping.
+
+use std::sync::Arc;
+
+use futures_util::future::BoxFuture;
+use moka::future::Cache;
+use snafu::OptionExt;
+use store_api::storage::TableId;
+
+use crate::cache::{CacheContainer, Initializer};
+use crate::error;
+use crate::instruction::CacheIdent;
+use crate::key::schema_name::SchemaName;
+use crate::key::table_info::TableInfoManager;
+use crate::kv_backend::KvBackendRef;
+
+pub type TableSchemaCache = CacheContainer<TableId, Arc<SchemaName>, CacheIdent>;
+pub type TableSchemaCacheRef = Arc<TableSchemaCache>;
+
+/// Constructs a [TableSchemaCache].
+pub fn new_table_schema_cache(
+    name: String,
+    cache: Cache<TableId, Arc<SchemaName>>,
+    kv_backend: KvBackendRef,
+) -> TableSchemaCache {
+    let table_info_manager = TableInfoManager::new(kv_backend);
+    let init = init_factory(table_info_manager);
+
+    CacheContainer::new(name, cache, Box::new(invalidator), init, Box::new(filter))
+}
+
+fn init_factory(table_info_manager: TableInfoManager) -> Initializer<TableId, Arc<SchemaName>> {
+    Arc::new(move |table_id| {
+        let table_info_manager = table_info_manager.clone();
+        Box::pin(async move {
+            let raw_table_info = table_info_manager
+                .get(*table_id)
+                .await?
+                .context(error::ValueNotExistSnafu)?
+                .into_inner()
+                .table_info;
+
+            Ok(Some(Arc::new(SchemaName {
+                catalog_name: raw_table_info.catalog_name,
+                schema_name: raw_table_info.schema_name,
+            })))
+        })
+    })
+}
+
+/// Never invalidates table id schema cache.
+fn invalidator<'a>(
+    _cache: &'a Cache<TableId, Arc<SchemaName>>,
+    _ident: &'a CacheIdent,
+) -> BoxFuture<'a, error::Result<()>> {
+    Box::pin(std::future::ready(Ok(())))
+}
+
+/// Never invalidates table id schema cache.
+fn filter(_ident: &CacheIdent) -> bool {
+    false
+}
--- a/src/common/meta/src/ddl/alter_database.rs
+++ b/src/common/meta/src/ddl/alter_database.rs
@@ -46,11 +46,7 @@ fn build_new_schema_value(
            for option in options.0.iter() {
                match option {
                    SetDatabaseOption::Ttl(ttl) => {
-                        if ttl.is_zero() {
-                            value.ttl = None;
-                        } else {
-                            value.ttl = Some(*ttl);
-                        }
+                        value.ttl = Some(*ttl);
                    }
                }
            }
@@ -230,12 +226,12 @@ mod tests {
    #[test]
    fn test_build_new_schema_value() {
        let set_ttl = AlterDatabaseKind::SetDatabaseOptions(SetDatabaseOptions(vec![
-            SetDatabaseOption::Ttl(Duration::from_secs(10)),
+            SetDatabaseOption::Ttl(Duration::from_secs(10).into()),
        ]));
        let current_schema_value = SchemaNameValue::default();
        let new_schema_value =
            build_new_schema_value(current_schema_value.clone(), &set_ttl).unwrap();
-        assert_eq!(new_schema_value.ttl, Some(Duration::from_secs(10)));
+        assert_eq!(new_schema_value.ttl, Some(Duration::from_secs(10).into()));

        let unset_ttl_alter_kind =
            AlterDatabaseKind::UnsetDatabaseOptions(UnsetDatabaseOptions(vec![
--- a/src/common/meta/src/error.rs
+++ b/src/common/meta/src/error.rs
@@ -425,6 +425,13 @@ pub enum Error {
        source: BoxedError,
    },

+    #[snafu(display("The response exceeded size limit"))]
+    ResponseExceededSizeLimit {
+        #[snafu(implicit)]
+        location: Location,
+        source: BoxedError,
+    },
+
    #[snafu(display("Invalid heartbeat response"))]
    InvalidHeartbeatResponse {
        #[snafu(implicit)]
@@ -763,6 +770,7 @@ impl ErrorExt for Error {
            | StopProcedureManager { source, .. } => source.status_code(),
            RegisterProcedureLoader { source, .. } => source.status_code(),
            External { source, .. } => source.status_code(),
+            ResponseExceededSizeLimit { source, .. } => source.status_code(),
            OperateDatanode { source, .. } => source.status_code(),
            Table { source, .. } => source.status_code(),
            RetryLater { source, .. } => source.status_code(),
@@ -805,13 +813,13 @@ impl Error {

    /// Returns true if the response exceeds the size limit.
    pub fn is_exceeded_size_limit(&self) -> bool {
-        if let Error::EtcdFailed {
-            error: etcd_client::Error::GRpcStatus(status),
-            ..
-        } = self
-        {
-            return status.code() == tonic::Code::OutOfRange;
+        match self {
+            Error::EtcdFailed {
+                error: etcd_client::Error::GRpcStatus(status),
+                ..
+            } => status.code() == tonic::Code::OutOfRange,
+            Error::ResponseExceededSizeLimit { .. } => true,
+            _ => false,
        }
-        false
    }
 }
--- a/src/common/meta/src/heartbeat/handler.rs
+++ b/src/common/meta/src/heartbeat/handler.rs
@@ -21,6 +21,7 @@ use common_telemetry::error;
 use crate::error::Result;
 use crate::heartbeat::mailbox::{IncomingMessage, MailboxRef};

+pub mod invalidate_table_cache;
 pub mod parse_mailbox_message;
 #[cfg(test)]
 mod tests;
--- a/src/common/meta/src/heartbeat/handler/invalidate_table_cache.rs
+++ b/src/common/meta/src/heartbeat/handler/invalidate_table_cache.rs
@@ -13,21 +13,22 @@
 // limitations under the License.

 use async_trait::async_trait;
-use common_meta::cache_invalidator::{CacheInvalidatorRef, Context};
-use common_meta::error::Result as MetaResult;
-use common_meta::heartbeat::handler::{
-    HandleControl, HeartbeatResponseHandler, HeartbeatResponseHandlerContext,
-};
-use common_meta::instruction::Instruction;
 use common_telemetry::debug;

+use crate::cache_invalidator::{CacheInvalidatorRef, Context};
+use crate::error::Result as MetaResult;
+use crate::heartbeat::handler::{
+    HandleControl, HeartbeatResponseHandler, HeartbeatResponseHandlerContext,
+};
+use crate::instruction::Instruction;
+
 #[derive(Clone)]
-pub struct InvalidateTableCacheHandler {
+pub struct InvalidateCacheHandler {
    cache_invalidator: CacheInvalidatorRef,
 }

 #[async_trait]
-impl HeartbeatResponseHandler for InvalidateTableCacheHandler {
+impl HeartbeatResponseHandler for InvalidateCacheHandler {
    fn is_acceptable(&self, ctx: &HeartbeatResponseHandlerContext) -> bool {
        matches!(
            ctx.incoming_message.as_ref(),
@@ -37,13 +38,10 @@ impl HeartbeatResponseHandler for InvalidateTableCacheHandler {

    async fn handle(&self, ctx: &mut HeartbeatResponseHandlerContext) -> MetaResult<HandleControl> {
        let Some((_, Instruction::InvalidateCaches(caches))) = ctx.incoming_message.take() else {
-            unreachable!("InvalidateTableCacheHandler: should be guarded by 'is_acceptable'")
+            unreachable!("InvalidateCacheHandler: should be guarded by 'is_acceptable'")
        };

-        debug!(
-            "InvalidateTableCacheHandler: invalidating caches: {:?}",
-            caches
-        );
+        debug!("InvalidateCacheHandler: invalidating caches: {:?}", caches);

        // Invalidate local cache always success
        let _ = self
@@ -55,7 +53,7 @@ impl HeartbeatResponseHandler for InvalidateTableCacheHandler {
    }
 }

-impl InvalidateTableCacheHandler {
+impl InvalidateCacheHandler {
    pub fn new(cache_invalidator: CacheInvalidatorRef) -> Self {
        Self { cache_invalidator }
    }
--- a/src/common/meta/src/key.rs
+++ b/src/common/meta/src/key.rs
@@ -149,7 +149,7 @@ use crate::DatanodeId;
 pub const NAME_PATTERN: &str = r"[a-zA-Z_:-][a-zA-Z0-9_:\-\.@#]*";
 pub const MAINTENANCE_KEY: &str = "__maintenance";

-const DATANODE_TABLE_KEY_PREFIX: &str = "__dn_table";
+pub const DATANODE_TABLE_KEY_PREFIX: &str = "__dn_table";
 pub const TABLE_INFO_KEY_PREFIX: &str = "__table_info";
 pub const VIEW_INFO_KEY_PREFIX: &str = "__view_info";
 pub const TABLE_NAME_KEY_PREFIX: &str = "__table_name";
--- a/src/common/meta/src/key/schema_metadata_manager.rs
+++ b/src/common/meta/src/key/schema_metadata_manager.rs
@@ -19,41 +19,39 @@ use std::sync::Arc;
 use snafu::OptionExt;
 use store_api::storage::TableId;

+use crate::cache::{SchemaCacheRef, TableSchemaCacheRef};
 use crate::error::TableInfoNotFoundSnafu;
-use crate::key::schema_name::{SchemaManager, SchemaNameKey};
-use crate::key::table_info::{TableInfoManager, TableInfoManagerRef};
-use crate::kv_backend::KvBackendRef;
 use crate::{error, SchemaOptions};

 pub type SchemaMetadataManagerRef = Arc<SchemaMetadataManager>;

 pub struct SchemaMetadataManager {
-    table_info_manager: TableInfoManagerRef,
-    schema_manager: SchemaManager,
+    table_id_schema_cache: TableSchemaCacheRef,
+    schema_cache: SchemaCacheRef,
    #[cfg(any(test, feature = "testing"))]
-    kv_backend: KvBackendRef,
+    kv_backend: crate::kv_backend::KvBackendRef,
 }

 impl SchemaMetadataManager {
    /// Creates a new database meta
    #[cfg(not(any(test, feature = "testing")))]
-    pub fn new(kv_backend: KvBackendRef) -> Self {
-        let table_info_manager = Arc::new(TableInfoManager::new(kv_backend.clone()));
-        let schema_manager = SchemaManager::new(kv_backend);
+    pub fn new(table_id_schema_cache: TableSchemaCacheRef, schema_cache: SchemaCacheRef) -> Self {
        Self {
-            table_info_manager,
-            schema_manager,
+            table_id_schema_cache,
+            schema_cache,
        }
    }

    /// Creates a new database meta
    #[cfg(any(test, feature = "testing"))]
-    pub fn new(kv_backend: KvBackendRef) -> Self {
-        let table_info_manager = Arc::new(TableInfoManager::new(kv_backend.clone()));
-        let schema_manager = SchemaManager::new(kv_backend.clone());
+    pub fn new(
+        kv_backend: crate::kv_backend::KvBackendRef,
+        table_id_schema_cache: TableSchemaCacheRef,
+        schema_cache: SchemaCacheRef,
+    ) -> Self {
        Self {
-            table_info_manager,
-            schema_manager,
+            table_id_schema_cache,
+            schema_cache,
            kv_backend,
        }
    }
@@ -62,23 +60,16 @@ impl SchemaMetadataManager {
    pub async fn get_schema_options_by_table_id(
        &self,
        table_id: TableId,
-    ) -> error::Result<Option<SchemaOptions>> {
-        let table_info = self
-            .table_info_manager
+    ) -> error::Result<Option<Arc<SchemaOptions>>> {
+        let schema_name = self
+            .table_id_schema_cache
            .get(table_id)
            .await?
            .with_context(|| TableInfoNotFoundSnafu {
                table: format!("table id: {}", table_id),
            })?;

-        let key = SchemaNameKey::new(
-            &table_info.table_info.catalog_name,
-            &table_info.table_info.schema_name,
-        );
-        self.schema_manager
-            .get(key)
-            .await
-            .map(|v| v.map(|v| v.into_inner()))
+        self.schema_cache.get_by_ref(&schema_name).await
    }

    #[cfg(any(test, feature = "testing"))]
@@ -100,17 +91,19 @@ impl SchemaMetadataManager {
            meta: Default::default(),
            table_type: TableType::Base,
        });
-        let (txn, _) = self
-            .table_info_manager
+        let table_info_manager =
+            crate::key::table_info::TableInfoManager::new(self.kv_backend.clone());
+        let (txn, _) = table_info_manager
            .build_create_txn(table_id, &value)
            .unwrap();
        let resp = self.kv_backend.txn(txn).await.unwrap();
        assert!(resp.succeeded, "Failed to create table metadata");
-        let key = SchemaNameKey {
+        let key = crate::key::schema_name::SchemaNameKey {
            catalog: catalog_name,
            schema: schema_name,
        };
-        self.schema_manager
+
+        crate::key::schema_name::SchemaManager::new(self.kv_backend.clone())
            .create(key, schema_value, false)
            .await
            .expect("Failed to create schema metadata");
--- a/src/common/meta/src/key/schema_name.rs
+++ b/src/common/meta/src/key/schema_name.rs
@@ -15,9 +15,9 @@
 use std::collections::HashMap;
 use std::fmt::Display;
 use std::sync::Arc;
-use std::time::Duration;

 use common_catalog::consts::{DEFAULT_CATALOG_NAME, DEFAULT_SCHEMA_NAME};
+use common_time::DatabaseTimeToLive;
 use futures::stream::BoxStream;
 use humantime_serde::re::humantime;
 use serde::{Deserialize, Serialize};
@@ -57,15 +57,13 @@ impl Default for SchemaNameKey<'_> {
 #[derive(Debug, Default, Clone, PartialEq, Serialize, Deserialize)]
 pub struct SchemaNameValue {
    #[serde(default)]
-    #[serde(with = "humantime_serde")]
-    pub ttl: Option<Duration>,
+    pub ttl: Option<DatabaseTimeToLive>,
 }

 impl Display for SchemaNameValue {
    fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result {
-        if let Some(ttl) = self.ttl {
-            let ttl = humantime::format_duration(ttl);
-            write!(f, "ttl='{ttl}'")?;
+        if let Some(ttl) = self.ttl.map(|i| i.to_string()) {
+            write!(f, "ttl='{}'", ttl)?;
        }

        Ok(())
@@ -96,11 +94,8 @@ impl TryFrom<&HashMap<String, String>> for SchemaNameValue {
 impl From<SchemaNameValue> for HashMap<String, String> {
    fn from(value: SchemaNameValue) -> Self {
        let mut opts = HashMap::new();
-        if let Some(ttl) = value.ttl {
-            opts.insert(
-                OPT_KEY_TTL.to_string(),
-                format!("{}", humantime::format_duration(ttl)),
-            );
+        if let Some(ttl) = value.ttl.map(|ttl| ttl.to_string()) {
+            opts.insert(OPT_KEY_TTL.to_string(), ttl);
        }
        opts
    }
@@ -171,6 +166,7 @@ impl<'a> TryFrom<&'a str> for SchemaNameKey<'a> {
    }
 }

+#[derive(Clone)]
 pub struct SchemaManager {
    kv_backend: KvBackendRef,
 }
@@ -312,6 +308,7 @@ impl<'a> From<&'a SchemaName> for SchemaNameKey<'a> {

 #[cfg(test)]
 mod tests {
+    use std::time::Duration;

    use super::*;
    use crate::kv_backend::memory::MemoryKvBackend;
@@ -322,9 +319,14 @@ mod tests {
        assert_eq!("", schema_value.to_string());

        let schema_value = SchemaNameValue {
-            ttl: Some(Duration::from_secs(9)),
+            ttl: Some(Duration::from_secs(9).into()),
        };
        assert_eq!("ttl='9s'", schema_value.to_string());
+
+        let schema_value = SchemaNameValue {
+            ttl: Some(Duration::from_secs(0).into()),
+        };
+        assert_eq!("ttl='forever'", schema_value.to_string());
    }

    #[test]
@@ -337,17 +339,36 @@ mod tests {
        assert_eq!(key, parsed);

        let value = SchemaNameValue {
-            ttl: Some(Duration::from_secs(10)),
+            ttl: Some(Duration::from_secs(10).into()),
        };
        let mut opts: HashMap<String, String> = HashMap::new();
        opts.insert("ttl".to_string(), "10s".to_string());
        let from_value = SchemaNameValue::try_from(&opts).unwrap();
        assert_eq!(value, from_value);

-        let parsed = SchemaNameValue::try_from_raw_value("{\"ttl\":\"10s\"}".as_bytes()).unwrap();
+        let parsed = SchemaNameValue::try_from_raw_value(
+            serde_json::json!({"ttl": "10s"}).to_string().as_bytes(),
+        )
+        .unwrap();
        assert_eq!(Some(value), parsed);
+
+        let forever = SchemaNameValue {
+            ttl: Some(Default::default()),
+        };
+        let parsed = SchemaNameValue::try_from_raw_value(
+            serde_json::json!({"ttl": "forever"}).to_string().as_bytes(),
+        )
+        .unwrap();
+        assert_eq!(Some(forever), parsed);
+
+        let instant_err = SchemaNameValue::try_from_raw_value(
+            serde_json::json!({"ttl": "instant"}).to_string().as_bytes(),
+        );
+        assert!(instant_err.is_err());
+
        let none = SchemaNameValue::try_from_raw_value("null".as_bytes()).unwrap();
        assert!(none.is_none());
+
        let err_empty = SchemaNameValue::try_from_raw_value("".as_bytes());
        assert!(err_empty.is_err());
    }
@@ -373,7 +394,7 @@ mod tests {

        let current_schema_value = manager.get(schema_key).await.unwrap().unwrap();
        let new_schema_value = SchemaNameValue {
-            ttl: Some(Duration::from_secs(10)),
+            ttl: Some(Duration::from_secs(10).into()),
        };
        manager
            .update(schema_key, &current_schema_value, &new_schema_value)
@@ -387,10 +408,10 @@ mod tests {
            .unwrap();

        let new_schema_value = SchemaNameValue {
-            ttl: Some(Duration::from_secs(40)),
+            ttl: Some(Duration::from_secs(40).into()),
        };
        let incorrect_schema_value = SchemaNameValue {
-            ttl: Some(Duration::from_secs(20)),
+            ttl: Some(Duration::from_secs(20).into()),
        }
        .try_as_raw_value()
        .unwrap();
@@ -401,5 +422,15 @@ mod tests {
            .update(schema_key, &incorrect_schema_value, &new_schema_value)
            .await
            .unwrap_err();
+
+        let current_schema_value = manager.get(schema_key).await.unwrap().unwrap();
+        let new_schema_value = SchemaNameValue { ttl: None };
+        manager
+            .update(schema_key, &current_schema_value, &new_schema_value)
+            .await
+            .unwrap();
+
+        let current_schema_value = manager.get(schema_key).await.unwrap().unwrap();
+        assert_eq!(new_schema_value, *current_schema_value);
    }
 }
--- a/src/common/meta/src/key/table_route.rs
+++ b/src/common/meta/src/key/table_route.rs
@@ -290,28 +290,6 @@ impl TableRouteManager {
        }
    }

-    /// Returns the [`PhysicalTableRouteValue`] in the first level,
-    /// It won't follow the [`LogicalTableRouteValue`] to find the next level [`PhysicalTableRouteValue`].
-    ///
-    /// Returns an error if the first level value is not a [`PhysicalTableRouteValue`].
-    pub async fn try_get_physical_table_route(
-        &self,
-        table_id: TableId,
-    ) -> Result<Option<PhysicalTableRouteValue>> {
-        match self.storage.get(table_id).await? {
-            Some(route) => {
-                ensure!(
-                    route.is_physical(),
-                    UnexpectedLogicalRouteTableSnafu {
-                        err_msg: format!("{route:?} is a non-physical TableRouteValue.")
-                    }
-                );
-                Ok(Some(route.into_physical_table_route()))
-            }
-            None => Ok(None),
-        }
-    }
-
    /// Returns the [TableId] recursively.
    ///
    /// Returns a [TableRouteNotFound](crate::error::Error::TableRouteNotFound) Error if:
@@ -569,37 +547,6 @@ impl TableRouteStorage {
            .transpose()
    }

-    /// Returns the physical `DeserializedValueWithBytes<TableRouteValue>` recursively.
-    ///
-    /// Returns a [TableRouteNotFound](crate::error::Error::TableRouteNotFound) Error if:
-    /// - the physical table(`logical_or_physical_table_id`) does not exist
-    /// - the corresponding physical table of the logical table(`logical_or_physical_table_id`) does not exist.
-    pub async fn get_physical_table_route_with_raw_bytes(
-        &self,
-        logical_or_physical_table_id: TableId,
-    ) -> Result<(TableId, DeserializedValueWithBytes<TableRouteValue>)> {
-        let table_route = self
-            .get_with_raw_bytes(logical_or_physical_table_id)
-            .await?
-            .context(TableRouteNotFoundSnafu {
-                table_id: logical_or_physical_table_id,
-            })?;
-
-        match table_route.get_inner_ref() {
-            TableRouteValue::Physical(_) => Ok((logical_or_physical_table_id, table_route)),
-            TableRouteValue::Logical(x) => {
-                let physical_table_id = x.physical_table_id();
-                let physical_table_route = self
-                    .get_with_raw_bytes(physical_table_id)
-                    .await?
-                    .context(TableRouteNotFoundSnafu {
-                        table_id: physical_table_id,
-                    })?;
-                Ok((physical_table_id, physical_table_route))
-            }
-        }
-    }
-
    /// Returns batch of [`TableRouteValue`] that respects the order of `table_ids`.
    pub async fn batch_get(&self, table_ids: &[TableId]) -> Result<Vec<Option<TableRouteValue>>> {
        let mut table_routes = self.batch_get_inner(table_ids).await?;
--- a/src/common/meta/src/kv_backend/etcd.rs
+++ b/src/common/meta/src/kv_backend/etcd.rs
@@ -15,6 +15,7 @@
 use std::any::Any;
 use std::sync::Arc;

+use common_telemetry::info;
 use etcd_client::{
    Client, DeleteOptions, GetOptions, PutOptions, Txn, TxnOp, TxnOpResponse, TxnResponse,
 };
@@ -55,6 +56,7 @@ impl EtcdStore {
    }

    pub fn with_etcd_client(client: Client, max_txn_ops: usize) -> KvBackendRef {
+        info!("Connected to etcd");
        Arc::new(Self {
            client,
            max_txn_ops,
--- a/src/common/meta/src/kv_backend/txn.rs
+++ b/src/common/meta/src/kv_backend/txn.rs
@@ -136,6 +136,13 @@ pub struct Txn {
    c_else: bool,
 }

+#[cfg(any(test, feature = "testing"))]
+impl Txn {
+    pub fn req(&self) -> &TxnRequest {
+        &self.req
+    }
+}
+
 impl Txn {
    pub fn merge_all<T: IntoIterator<Item = Txn>>(values: T) -> Self {
        values
--- a/src/common/meta/src/rpc/ddl.rs
+++ b/src/common/meta/src/rpc/ddl.rs
@@ -14,7 +14,6 @@

 use std::collections::{HashMap, HashSet};
 use std::result;
-use std::time::Duration;

 use api::v1::alter_database_expr::Kind as PbAlterDatabaseKind;
 use api::v1::meta::ddl_task_request::Task;
@@ -36,7 +35,7 @@ use api::v1::{
 };
 use base64::engine::general_purpose;
 use base64::Engine as _;
-use humantime_serde::re::humantime;
+use common_time::DatabaseTimeToLive;
 use prost::Message;
 use serde::{Deserialize, Serialize};
 use serde_with::{serde_as, DefaultOnNull};
@@ -1009,12 +1008,8 @@ impl TryFrom<PbOption> for SetDatabaseOption {
    fn try_from(PbOption { key, value }: PbOption) -> Result<Self> {
        match key.to_ascii_lowercase().as_str() {
            TTL_KEY => {
-                let ttl = if value.is_empty() {
-                    Duration::from_secs(0)
-                } else {
-                    humantime::parse_duration(&value)
-                        .map_err(|_| InvalidSetDatabaseOptionSnafu { key, value }.build())?
-                };
+                let ttl = DatabaseTimeToLive::from_humantime_or_str(&value)
+                    .map_err(|_| InvalidSetDatabaseOptionSnafu { key, value }.build())?;

                Ok(SetDatabaseOption::Ttl(ttl))
            }
@@ -1025,7 +1020,7 @@ impl TryFrom<PbOption> for SetDatabaseOption {

 #[derive(Debug, PartialEq, Clone, Serialize, Deserialize)]
 pub enum SetDatabaseOption {
-    Ttl(Duration),
+    Ttl(DatabaseTimeToLive),
 }

 #[derive(Debug, PartialEq, Clone, Serialize, Deserialize)]
--- a/src/common/meta/src/rpc/router.rs
+++ b/src/common/meta/src/rpc/router.rs
@@ -89,39 +89,6 @@ pub fn convert_to_region_leader_map(region_routes: &[RegionRoute]) -> HashMap<Re
        .collect::<HashMap<_, _>>()
 }

-/// Returns the HashMap<[RegionNumber], HashSet<DatanodeId>>
-pub fn convert_to_region_peer_map(
-    region_routes: &[RegionRoute],
-) -> HashMap<RegionNumber, HashSet<u64>> {
-    region_routes
-        .iter()
-        .map(|x| {
-            let set = x
-                .follower_peers
-                .iter()
-                .map(|p| p.id)
-                .chain(x.leader_peer.as_ref().map(|p| p.id))
-                .collect::<HashSet<_>>();
-
-            (x.region.id.region_number(), set)
-        })
-        .collect::<HashMap<_, _>>()
-}
-
-/// Returns the HashMap<[RegionNumber], [LeaderState]>;
-pub fn convert_to_region_leader_state_map(
-    region_routes: &[RegionRoute],
-) -> HashMap<RegionNumber, LeaderState> {
-    region_routes
-        .iter()
-        .filter_map(|x| {
-            x.leader_state
-                .as_ref()
-                .map(|state| (x.region.id.region_number(), *state))
-        })
-        .collect::<HashMap<_, _>>()
-}
-
 pub fn find_region_leader(
    region_routes: &[RegionRoute],
    region_number: RegionNumber,
@@ -147,19 +114,6 @@ pub fn find_leader_regions(region_routes: &[RegionRoute], datanode: &Peer) -> Ve
        .collect()
 }

-pub fn extract_all_peers(region_routes: &[RegionRoute]) -> Vec<Peer> {
-    let mut peers = region_routes
-        .iter()
-        .flat_map(|x| x.leader_peer.iter().chain(x.follower_peers.iter()))
-        .collect::<HashSet<_>>()
-        .into_iter()
-        .cloned()
-        .collect::<Vec<_>>();
-    peers.sort_by_key(|x| x.id);
-
-    peers
-}
-
 impl TableRoute {
    pub fn new(table: Table, region_routes: Vec<RegionRoute>) -> Self {
        let region_leaders = region_routes
--- a/src/common/procedure/src/local/runner.rs
+++ b/src/common/procedure/src/local/runner.rs
@@ -544,7 +544,7 @@ mod tests {
    use common_test_util::temp_dir::create_temp_dir;
    use futures_util::future::BoxFuture;
    use futures_util::FutureExt;
-    use object_store::ObjectStore;
+    use object_store::{EntryMode, ObjectStore};
    use tokio::sync::mpsc;

    use super::*;
@@ -578,7 +578,11 @@ mod tests {
    ) {
        let dir = proc_path!(procedure_store, "{procedure_id}/");
        let lister = object_store.list(&dir).await.unwrap();
-        let mut files_in_dir: Vec<_> = lister.into_iter().map(|de| de.name().to_string()).collect();
+        let mut files_in_dir: Vec<_> = lister
+            .into_iter()
+            .filter(|x| x.metadata().mode() == EntryMode::FILE)
+            .map(|de| de.name().to_string())
+            .collect();
        files_in_dir.sort_unstable();
        assert_eq!(files, files_in_dir);
    }
--- a/src/common/recordbatch/src/cursor.rs
+++ b/src/common/recordbatch/src/cursor.rs
@@ -0,0 +1,173 @@
+// Copyright 2023 Greptime Team
+//
+// Licensed under the Apache License, Version 2.0 (the "License");
+// you may not use this file except in compliance with the License.
+// You may obtain a copy of the License at
+//
+//     http://www.apache.org/licenses/LICENSE-2.0
+//
+// Unless required by applicable law or agreed to in writing, software
+// distributed under the License is distributed on an "AS IS" BASIS,
+// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+// See the License for the specific language governing permissions and
+// limitations under the License.
+
+use futures::StreamExt;
+use tokio::sync::Mutex;
+
+use crate::error::Result;
+use crate::recordbatch::merge_record_batches;
+use crate::{RecordBatch, SendableRecordBatchStream};
+
+struct Inner {
+    stream: SendableRecordBatchStream,
+    current_row_index: usize,
+    current_batch: Option<RecordBatch>,
+    total_rows_in_current_batch: usize,
+}
+
+/// A cursor on RecordBatchStream that fetches data batch by batch
+pub struct RecordBatchStreamCursor {
+    inner: Mutex<Inner>,
+}
+
+impl RecordBatchStreamCursor {
+    pub fn new(stream: SendableRecordBatchStream) -> RecordBatchStreamCursor {
+        Self {
+            inner: Mutex::new(Inner {
+                stream,
+                current_row_index: 0,
+                current_batch: None,
+                total_rows_in_current_batch: 0,
+            }),
+        }
+    }
+
+    /// Take `size` of row from the `RecordBatchStream` and create a new
+    /// `RecordBatch` for these rows.
+    pub async fn take(&self, size: usize) -> Result<RecordBatch> {
+        let mut remaining_rows_to_take = size;
+        let mut accumulated_rows = Vec::new();
+
+        let mut inner = self.inner.lock().await;
+
+        while remaining_rows_to_take > 0 {
+            // Ensure we have a current batch or fetch the next one
+            if inner.current_batch.is_none()
+                || inner.current_row_index >= inner.total_rows_in_current_batch
+            {
+                match inner.stream.next().await {
+                    Some(Ok(batch)) => {
+                        inner.total_rows_in_current_batch = batch.num_rows();
+                        inner.current_batch = Some(batch);
+                        inner.current_row_index = 0;
+                    }
+                    Some(Err(e)) => return Err(e),
+                    None => {
+                        // Stream is exhausted
+                        break;
+                    }
+                }
+            }
+
+            // If we still have no batch after attempting to fetch
+            let current_batch = match &inner.current_batch {
+                Some(batch) => batch,
+                None => break,
+            };
+
+            // Calculate how many rows we can take from this batch
+            let rows_to_take_from_batch = remaining_rows_to_take
+                .min(inner.total_rows_in_current_batch - inner.current_row_index);
+
+            // Slice the current batch to get the desired rows
+            let taken_batch =
+                current_batch.slice(inner.current_row_index, rows_to_take_from_batch)?;
+
+            // Add the taken batch to accumulated rows
+            accumulated_rows.push(taken_batch);
+
+            // Update cursor and remaining rows
+            inner.current_row_index += rows_to_take_from_batch;
+            remaining_rows_to_take -= rows_to_take_from_batch;
+        }
+
+        // If no rows were accumulated, return empty
+        if accumulated_rows.is_empty() {
+            return Ok(RecordBatch::new_empty(inner.stream.schema()));
+        }
+
+        // If only one batch was accumulated, return it directly
+        if accumulated_rows.len() == 1 {
+            return Ok(accumulated_rows.remove(0));
+        }
+
+        // Merge multiple batches
+        merge_record_batches(inner.stream.schema(), &accumulated_rows)
+    }
+}
+
+#[cfg(test)]
+mod tests {
+    use std::sync::Arc;
+
+    use datatypes::prelude::ConcreteDataType;
+    use datatypes::schema::{ColumnSchema, Schema};
+    use datatypes::vectors::StringVector;
+
+    use super::*;
+    use crate::RecordBatches;
+
+    #[tokio::test]
+    async fn test_cursor() {
+        let schema = Arc::new(Schema::new(vec![ColumnSchema::new(
+            "a",
+            ConcreteDataType::string_datatype(),
+            false,
+        )]));
+
+        let rbs = RecordBatches::try_from_columns(
+            schema.clone(),
+            vec![Arc::new(StringVector::from(vec!["hello", "world"])) as _],
+        )
+        .unwrap();
+
+        let cursor = RecordBatchStreamCursor::new(rbs.as_stream());
+        let result_rb = cursor.take(1).await.expect("take from cursor failed");
+        assert_eq!(result_rb.num_rows(), 1);
+
+        let result_rb = cursor.take(1).await.expect("take from cursor failed");
+        assert_eq!(result_rb.num_rows(), 1);
+
+        let result_rb = cursor.take(1).await.expect("take from cursor failed");
+        assert_eq!(result_rb.num_rows(), 0);
+
+        let rb = RecordBatch::new(
+            schema.clone(),
+            vec![Arc::new(StringVector::from(vec!["hello", "world"])) as _],
+        )
+        .unwrap();
+        let rbs2 =
+            RecordBatches::try_new(schema.clone(), vec![rb.clone(), rb.clone(), rb]).unwrap();
+        let cursor = RecordBatchStreamCursor::new(rbs2.as_stream());
+        let result_rb = cursor.take(3).await.expect("take from cursor failed");
+        assert_eq!(result_rb.num_rows(), 3);
+        let result_rb = cursor.take(2).await.expect("take from cursor failed");
+        assert_eq!(result_rb.num_rows(), 2);
+        let result_rb = cursor.take(2).await.expect("take from cursor failed");
+        assert_eq!(result_rb.num_rows(), 1);
+        let result_rb = cursor.take(2).await.expect("take from cursor failed");
+        assert_eq!(result_rb.num_rows(), 0);
+
+        let rb = RecordBatch::new(
+            schema.clone(),
+            vec![Arc::new(StringVector::from(vec!["hello", "world"])) as _],
+        )
+        .unwrap();
+        let rbs3 =
+            RecordBatches::try_new(schema.clone(), vec![rb.clone(), rb.clone(), rb]).unwrap();
+        let cursor = RecordBatchStreamCursor::new(rbs3.as_stream());
+        let result_rb = cursor.take(10).await.expect("take from cursor failed");
+        assert_eq!(result_rb.num_rows(), 6);
+    }
+}
--- a/src/common/recordbatch/src/error.rs
+++ b/src/common/recordbatch/src/error.rs
@@ -168,6 +168,13 @@ pub enum Error {
        #[snafu(source)]
        error: tokio::time::error::Elapsed,
    },
+    #[snafu(display("RecordBatch slice index overflow: {visit_index} > {size}"))]
+    RecordBatchSliceIndexOverflow {
+        #[snafu(implicit)]
+        location: Location,
+        size: usize,
+        visit_index: usize,
+    },
 }

 impl ErrorExt for Error {
@@ -182,7 +189,8 @@ impl ErrorExt for Error {
            | Error::Format { .. }
            | Error::ToArrowScalar { .. }
            | Error::ProjectArrowRecordBatch { .. }
-            | Error::PhysicalExpr { .. } => StatusCode::Internal,
+            | Error::PhysicalExpr { .. }
+            | Error::RecordBatchSliceIndexOverflow { .. } => StatusCode::Internal,

            Error::PollStream { .. } => StatusCode::EngineExecuteQuery,

--- a/src/common/recordbatch/src/lib.rs
+++ b/src/common/recordbatch/src/lib.rs
@@ -15,6 +15,7 @@
 #![feature(never_type)]

 pub mod adapter;
+pub mod cursor;
 pub mod error;
 pub mod filter;
 mod recordbatch;
@@ -25,7 +26,6 @@ use std::sync::Arc;

 use adapter::RecordBatchMetrics;
 use arc_swap::ArcSwapOption;
-use datafusion::physical_plan::memory::MemoryStream;
 pub use datafusion::physical_plan::SendableRecordBatchStream as DfSendableRecordBatchStream;
 use datatypes::arrow::compute::SortOptions;
 pub use datatypes::arrow::record_batch::RecordBatch as DfRecordBatch;
@@ -169,19 +169,6 @@ impl RecordBatches {
            index: 0,
        })
    }
-
-    pub fn into_df_stream(self) -> DfSendableRecordBatchStream {
-        let df_record_batches = self
-            .batches
-            .into_iter()
-            .map(|batch| batch.into_df_record_batch())
-            .collect();
-        // unwrap safety: `MemoryStream::try_new` won't fail
-        Box::pin(
-            MemoryStream::try_new(df_record_batches, self.schema.arrow_schema().clone(), None)
-                .unwrap(),
-        )
-    }
 }

 impl IntoIterator for RecordBatches {
--- a/src/common/recordbatch/src/recordbatch.rs
+++ b/src/common/recordbatch/src/recordbatch.rs
@@ -23,7 +23,7 @@ use datatypes::value::Value;
 use datatypes::vectors::{Helper, VectorRef};
 use serde::ser::{Error, SerializeStruct};
 use serde::{Serialize, Serializer};
-use snafu::{OptionExt, ResultExt};
+use snafu::{ensure, OptionExt, ResultExt};

 use crate::error::{
    self, CastVectorSnafu, ColumnNotExistsSnafu, DataTypesSnafu, ProjectArrowRecordBatchSnafu,
@@ -194,6 +194,19 @@ impl RecordBatch {
            .map(|t| t.to_string())
            .unwrap_or("failed to pretty display a record batch".to_string())
    }
+
+    /// Return a slice record batch starts from offset, with len rows
+    pub fn slice(&self, offset: usize, len: usize) -> Result<RecordBatch> {
+        ensure!(
+            offset + len <= self.num_rows(),
+            error::RecordBatchSliceIndexOverflowSnafu {
+                size: self.num_rows(),
+                visit_index: offset + len
+            }
+        );
+        let columns = self.columns.iter().map(|vector| vector.slice(offset, len));
+        RecordBatch::new(self.schema.clone(), columns)
+    }
 }

 impl Serialize for RecordBatch {
@@ -256,6 +269,36 @@ impl Iterator for RecordBatchRowIterator<'_> {
    }
 }

+/// merge multiple recordbatch into a single
+pub fn merge_record_batches(schema: SchemaRef, batches: &[RecordBatch]) -> Result<RecordBatch> {
+    let batches_len = batches.len();
+    if batches_len == 0 {
+        return Ok(RecordBatch::new_empty(schema));
+    }
+
+    let n_rows = batches.iter().map(|b| b.num_rows()).sum();
+    let n_columns = schema.num_columns();
+    // Collect arrays from each batch
+    let mut merged_columns = Vec::with_capacity(n_columns);
+
+    for col_idx in 0..n_columns {
+        let mut acc = schema.column_schemas()[col_idx]
+            .data_type
+            .create_mutable_vector(n_rows);
+
+        for batch in batches {
+            let column = batch.column(col_idx);
+            acc.extend_slice_of(column.as_ref(), 0, column.len())
+                .context(error::DataTypesSnafu)?;
+        }
+
+        merged_columns.push(acc.to_vector());
+    }
+
+    // Create a new RecordBatch with merged columns
+    RecordBatch::new(schema, merged_columns)
+}
+
 #[cfg(test)]
 mod tests {
    use std::sync::Arc;
@@ -375,4 +418,80 @@ mod tests {

        assert!(record_batch_iter.next().is_none());
    }
+
+    #[test]
+    fn test_record_batch_slice() {
+        let column_schemas = vec![
+            ColumnSchema::new("numbers", ConcreteDataType::uint32_datatype(), false),
+            ColumnSchema::new("strings", ConcreteDataType::string_datatype(), true),
+        ];
+        let schema = Arc::new(Schema::new(column_schemas));
+        let columns: Vec<VectorRef> = vec![
+            Arc::new(UInt32Vector::from_slice(vec![1, 2, 3, 4])),
+            Arc::new(StringVector::from(vec![
+                None,
+                Some("hello"),
+                Some("greptime"),
+                None,
+            ])),
+        ];
+        let recordbatch = RecordBatch::new(schema, columns).unwrap();
+        let recordbatch = recordbatch.slice(1, 2).expect("recordbatch slice");
+        let mut record_batch_iter = recordbatch.rows();
+        assert_eq!(
+            vec![Value::UInt32(2), Value::String("hello".into())],
+            record_batch_iter
+                .next()
+                .unwrap()
+                .into_iter()
+                .collect::<Vec<Value>>()
+        );
+
+        assert_eq!(
+            vec![Value::UInt32(3), Value::String("greptime".into())],
+            record_batch_iter
+                .next()
+                .unwrap()
+                .into_iter()
+                .collect::<Vec<Value>>()
+        );
+
+        assert!(record_batch_iter.next().is_none());
+
+        assert!(recordbatch.slice(1, 5).is_err());
+    }
+
+    #[test]
+    fn test_merge_record_batch() {
+        let column_schemas = vec![
+            ColumnSchema::new("numbers", ConcreteDataType::uint32_datatype(), false),
+            ColumnSchema::new("strings", ConcreteDataType::string_datatype(), true),
+        ];
+        let schema = Arc::new(Schema::new(column_schemas));
+        let columns: Vec<VectorRef> = vec![
+            Arc::new(UInt32Vector::from_slice(vec![1, 2, 3, 4])),
+            Arc::new(StringVector::from(vec![
+                None,
+                Some("hello"),
+                Some("greptime"),
+                None,
+            ])),
+        ];
+        let recordbatch = RecordBatch::new(schema.clone(), columns).unwrap();
+
+        let columns: Vec<VectorRef> = vec![
+            Arc::new(UInt32Vector::from_slice(vec![1, 2, 3, 4])),
+            Arc::new(StringVector::from(vec![
+                None,
+                Some("hello"),
+                Some("greptime"),
+                None,
+            ])),
+        ];
+        let recordbatch2 = RecordBatch::new(schema.clone(), columns).unwrap();
+
+        let merged = merge_record_batches(schema.clone(), &[recordbatch, recordbatch2])
+            .expect("merge recordbatch");
+        assert_eq!(merged.num_rows(), 8);
+    }
 }
--- a/src/common/runtime/Cargo.toml
+++ b/src/common/runtime/Cargo.toml
@@ -35,8 +35,6 @@ serde_json.workspace = true
 snafu.workspace = true
 tempfile.workspace = true
 tokio.workspace = true
-tokio-metrics = "0.3"
-tokio-metrics-collector = { git = "https://github.com/MichaelScofield/tokio-metrics-collector.git", rev = "89d692d5753d28564a7aac73c6ac5aba22243ba0" }
 tokio-util.workspace = true

 [dev-dependencies]
--- a/src/common/time/Cargo.toml
+++ b/src/common/time/Cargo.toml
@@ -13,6 +13,8 @@ chrono.workspace = true
 chrono-tz = "0.8"
 common-error.workspace = true
 common-macro.workspace = true
+humantime.workspace = true
+humantime-serde.workspace = true
 once_cell.workspace = true
 serde = { version = "1.0", features = ["derive"] }
 serde_json.workspace = true
--- a/src/common/time/src/error.rs
+++ b/src/common/time/src/error.rs
@@ -93,12 +93,28 @@ pub enum Error {
        #[snafu(implicit)]
        location: Location,
    },
+
+    #[snafu(display("Failed to parse duration"))]
+    ParseDuration {
+        #[snafu(source)]
+        error: humantime::DurationError,
+        #[snafu(implicit)]
+        location: Location,
+    },
+
+    #[snafu(display("Database's TTL can't be `instant`"))]
+    InvalidDatabaseTtl {
+        #[snafu(implicit)]
+        location: Location,
+    },
 }

 impl ErrorExt for Error {
    fn status_code(&self) -> StatusCode {
        match self {
            Error::ParseDateStr { .. }
+            | Error::ParseDuration { .. }
+            | Error::InvalidDatabaseTtl { .. }
            | Error::ParseTimestamp { .. }
            | Error::InvalidTimezoneOffset { .. }
            | Error::Format { .. }
--- a/src/common/time/src/lib.rs
+++ b/src/common/time/src/lib.rs
@@ -22,6 +22,7 @@ pub mod time;
 pub mod timestamp;
 pub mod timestamp_millis;
 pub mod timezone;
+pub mod ttl;
 pub mod util;

 pub use date::Date;
@@ -32,3 +33,4 @@ pub use range::RangeMillis;
 pub use timestamp::Timestamp;
 pub use timestamp_millis::TimestampMillis;
 pub use timezone::Timezone;
+pub use ttl::{DatabaseTimeToLive, TimeToLive, FOREVER, INSTANT};
--- a/src/common/time/src/ttl.rs
+++ b/src/common/time/src/ttl.rs
@@ -0,0 +1,266 @@
+// Copyright 2023 Greptime Team
+//
+// Licensed under the Apache License, Version 2.0 (the "License");
+// you may not use this file except in compliance with the License.
+// You may obtain a copy of the License at
+//
+//     http://www.apache.org/licenses/LICENSE-2.0
+//
+// Unless required by applicable law or agreed to in writing, software
+// distributed under the License is distributed on an "AS IS" BASIS,
+// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+// See the License for the specific language governing permissions and
+// limitations under the License.
+
+use std::fmt::Display;
+use std::time::Duration;
+
+use serde::{Deserialize, Serialize};
+use snafu::ResultExt;
+
+use crate::error::{Error, InvalidDatabaseTtlSnafu, ParseDurationSnafu};
+use crate::Timestamp;
+
+pub const INSTANT: &str = "instant";
+pub const FOREVER: &str = "forever";
+
+/// Time To Live for database, which can be `Forever`, or a `Duration`, but can't be `Instant`.
+///
+/// unlike `TimeToLive` which can be `Instant`, `Forever`, or a `Duration`
+#[derive(Debug, PartialEq, Eq, PartialOrd, Ord, Clone, Copy, Default, Serialize, Deserialize)]
+#[serde(rename_all = "snake_case")]
+pub enum DatabaseTimeToLive {
+    /// Keep the data forever
+    #[default]
+    Forever,
+    /// Duration to keep the data, this duration should be non-zero
+    #[serde(untagged, with = "humantime_serde")]
+    Duration(Duration),
+}
+
+impl Display for DatabaseTimeToLive {
+    fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result {
+        match self {
+            DatabaseTimeToLive::Forever => write!(f, "{}", FOREVER),
+            DatabaseTimeToLive::Duration(d) => write!(f, "{}", humantime::Duration::from(*d)),
+        }
+    }
+}
+
+impl DatabaseTimeToLive {
+    /// Parse a string that is either `forever`, or a duration to `TimeToLive`
+    ///
+    /// note that an empty string or a zero duration(a duration that spans no time) is treat as `forever` too
+    pub fn from_humantime_or_str(s: &str) -> Result<Self, Error> {
+        let ttl = match s.to_lowercase().as_ref() {
+            INSTANT => InvalidDatabaseTtlSnafu.fail()?,
+            FOREVER | "" => Self::Forever,
+            _ => {
+                let d = humantime::parse_duration(s).context(ParseDurationSnafu)?;
+                Self::from(d)
+            }
+        };
+        Ok(ttl)
+    }
+}
+
+impl TryFrom<TimeToLive> for DatabaseTimeToLive {
+    type Error = Error;
+    fn try_from(value: TimeToLive) -> Result<Self, Self::Error> {
+        match value {
+            TimeToLive::Instant => InvalidDatabaseTtlSnafu.fail()?,
+            TimeToLive::Forever => Ok(Self::Forever),
+            TimeToLive::Duration(d) => Ok(Self::from(d)),
+        }
+    }
+}
+
+impl From<DatabaseTimeToLive> for TimeToLive {
+    fn from(value: DatabaseTimeToLive) -> Self {
+        match value {
+            DatabaseTimeToLive::Forever => TimeToLive::Forever,
+            DatabaseTimeToLive::Duration(d) => TimeToLive::from(d),
+        }
+    }
+}
+
+impl From<Duration> for DatabaseTimeToLive {
+    fn from(duration: Duration) -> Self {
+        if duration.is_zero() {
+            Self::Forever
+        } else {
+            Self::Duration(duration)
+        }
+    }
+}
+
+impl From<humantime::Duration> for DatabaseTimeToLive {
+    fn from(duration: humantime::Duration) -> Self {
+        Self::from(*duration)
+    }
+}
+
+/// Time To Live
+#[derive(Debug, PartialEq, Eq, PartialOrd, Ord, Clone, Copy, Default, Serialize, Deserialize)]
+#[serde(rename_all = "snake_case")]
+pub enum TimeToLive {
+    /// Instantly discard upon insert
+    Instant,
+    /// Keep the data forever
+    #[default]
+    Forever,
+    /// Duration to keep the data, this duration should be non-zero
+    #[serde(untagged, with = "humantime_serde")]
+    Duration(Duration),
+}
+
+impl Display for TimeToLive {
+    fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result {
+        match self {
+            TimeToLive::Instant => write!(f, "{}", INSTANT),
+            TimeToLive::Duration(d) => write!(f, "{}", humantime::Duration::from(*d)),
+            TimeToLive::Forever => write!(f, "{}", FOREVER),
+        }
+    }
+}
+
+impl TimeToLive {
+    /// Parse a string that is either `instant`, `forever`, or a duration to `TimeToLive`
+    ///
+    /// note that an empty string or a zero duration(a duration that spans no time) is treat as `forever` too
+    pub fn from_humantime_or_str(s: &str) -> Result<Self, Error> {
+        match s.to_lowercase().as_ref() {
+            INSTANT => Ok(TimeToLive::Instant),
+            FOREVER | "" => Ok(TimeToLive::Forever),
+            _ => {
+                let d = humantime::parse_duration(s).context(ParseDurationSnafu)?;
+                Ok(TimeToLive::from(d))
+            }
+        }
+    }
+
+    /// Check if the TimeToLive is expired
+    /// with the given `created_at` and `now` timestamp
+    pub fn is_expired(
+        &self,
+        created_at: &Timestamp,
+        now: &Timestamp,
+    ) -> crate::error::Result<bool> {
+        Ok(match self {
+            TimeToLive::Instant => true,
+            TimeToLive::Forever => false,
+            TimeToLive::Duration(d) => now.sub_duration(*d)? > *created_at,
+        })
+    }
+
+    /// is instant variant
+    pub fn is_instant(&self) -> bool {
+        matches!(self, TimeToLive::Instant)
+    }
+
+    /// Is the default value, which is `Forever`
+    pub fn is_forever(&self) -> bool {
+        matches!(self, TimeToLive::Forever)
+    }
+}
+
+impl From<Duration> for TimeToLive {
+    fn from(duration: Duration) -> Self {
+        if duration.is_zero() {
+            // compatibility with old code, and inline with cassandra's behavior when ttl set to 0
+            TimeToLive::Forever
+        } else {
+            TimeToLive::Duration(duration)
+        }
+    }
+}
+
+impl From<humantime::Duration> for TimeToLive {
+    fn from(duration: humantime::Duration) -> Self {
+        Self::from(*duration)
+    }
+}
+
+#[cfg(test)]
+mod test {
+    use super::*;
+
+    #[test]
+    fn test_db_ttl_table_ttl() {
+        // test from ttl to db ttl
+        let ttl = TimeToLive::from(Duration::from_secs(10));
+        let db_ttl: DatabaseTimeToLive = ttl.try_into().unwrap();
+        assert_eq!(db_ttl, DatabaseTimeToLive::from(Duration::from_secs(10)));
+        assert_eq!(TimeToLive::from(db_ttl), ttl);
+
+        let ttl = TimeToLive::from(Duration::from_secs(0));
+        let db_ttl: DatabaseTimeToLive = ttl.try_into().unwrap();
+        assert_eq!(db_ttl, DatabaseTimeToLive::Forever);
+        assert_eq!(TimeToLive::from(db_ttl), ttl);
+
+        let ttl = TimeToLive::Instant;
+        let err_instant = DatabaseTimeToLive::try_from(ttl);
+        assert!(err_instant.is_err());
+
+        // test 0 duration
+        let ttl = Duration::from_secs(0);
+        let db_ttl: DatabaseTimeToLive = ttl.into();
+        assert_eq!(db_ttl, DatabaseTimeToLive::Forever);
+
+        let ttl = Duration::from_secs(10);
+        let db_ttl: DatabaseTimeToLive = ttl.into();
+        assert_eq!(
+            db_ttl,
+            DatabaseTimeToLive::Duration(Duration::from_secs(10))
+        );
+
+        let ttl = DatabaseTimeToLive::from_humantime_or_str("10s").unwrap();
+        let ttl: TimeToLive = ttl.into();
+        assert_eq!(ttl, TimeToLive::from(Duration::from_secs(10)));
+
+        let ttl = DatabaseTimeToLive::from_humantime_or_str("forever").unwrap();
+        let ttl: TimeToLive = ttl.into();
+        assert_eq!(ttl, TimeToLive::Forever);
+
+        assert!(DatabaseTimeToLive::from_humantime_or_str("instant").is_err());
+
+        // test 0s
+        let ttl = DatabaseTimeToLive::from_humantime_or_str("0s").unwrap();
+        let ttl: TimeToLive = ttl.into();
+        assert_eq!(ttl, TimeToLive::Forever);
+    }
+
+    #[test]
+    fn test_serde() {
+        let cases = vec![
+            ("\"instant\"", TimeToLive::Instant),
+            ("\"forever\"", TimeToLive::Forever),
+            ("\"10d\"", Duration::from_secs(86400 * 10).into()),
+            (
+                "\"10000 years\"",
+                humantime::parse_duration("10000 years").unwrap().into(),
+            ),
+        ];
+
+        for (s, expected) in cases {
+            let serialized = serde_json::to_string(&expected).unwrap();
+            let deserialized: TimeToLive = serde_json::from_str(&serialized).unwrap();
+            assert_eq!(deserialized, expected);
+
+            let deserialized: TimeToLive = serde_json::from_str(s).unwrap_or_else(|err| {
+                panic!("Actual serialized: {}, s=`{s}`, err: {:?}", serialized, err)
+            });
+            assert_eq!(deserialized, expected);
+
+            // test db ttl too
+            if s == "\"instant\"" {
+                assert!(serde_json::from_str::<DatabaseTimeToLive>(s).is_err());
+                continue;
+            }
+
+            let db_ttl: DatabaseTimeToLive = serde_json::from_str(s).unwrap();
+            let re_serialized = serde_json::to_string(&db_ttl).unwrap();
+            assert_eq!(re_serialized, serialized);
+        }
+    }
+}
--- a/src/common/time/src/util.rs
+++ b/src/common/time/src/util.rs
@@ -29,10 +29,6 @@ pub fn format_utc_datetime(utc: &NaiveDateTime, pattern: &str) -> String {
    }
 }

-pub fn system_datetime_to_utc(local: &NaiveDateTime) -> LocalResult<NaiveDateTime> {
-    datetime_to_utc(local, get_timezone(None))
-}
-
 /// Cast a [`NaiveDateTime`] with the given timezone.
 pub fn datetime_to_utc(
    datetime: &NaiveDateTime,
--- a/src/common/version/Cargo.toml
+++ b/src/common/version/Cargo.toml
@@ -8,11 +8,10 @@ license.workspace = true
 workspace = true

 [features]
-codec = ["dep:serde", "dep:schemars"]
+codec = ["dep:serde"]

 [dependencies]
 const_format = "0.2"
-schemars = { workspace = true, optional = true }
 serde = { workspace = true, optional = true }
 shadow-rs.workspace = true

--- a/src/common/version/src/lib.rs
+++ b/src/common/version/src/lib.rs
@@ -49,10 +49,7 @@ impl Display for BuildInfo {
 }

 #[derive(Clone, Debug, PartialEq)]
-#[cfg_attr(
-    feature = "codec",
-    derive(serde::Serialize, serde::Deserialize, schemars::JsonSchema)
-)]
+#[cfg_attr(feature = "codec", derive(serde::Serialize, serde::Deserialize))]
 pub struct OwnedBuildInfo {
    pub branch: String,
    pub commit: String,
--- a/src/common/wal/src/config/raft_engine.rs
+++ b/src/common/wal/src/config/raft_engine.rs
@@ -49,9 +49,9 @@ impl Default for RaftEngineConfig {
    fn default() -> Self {
        Self {
            dir: None,
-            file_size: ReadableSize::mb(256),
-            purge_threshold: ReadableSize::gb(4),
-            purge_interval: Duration::from_secs(600),
+            file_size: ReadableSize::mb(128),
+            purge_threshold: ReadableSize::gb(1),
+            purge_interval: Duration::from_secs(60),
            read_batch_size: 128,
            sync_write: false,
            enable_log_recycle: true,
--- a/src/datanode/Cargo.toml
+++ b/src/datanode/Cargo.toml
@@ -65,6 +65,7 @@ toml.workspace = true
 tonic.workspace = true

 [dev-dependencies]
+cache.workspace = true
 client.workspace = true
 common-meta = { workspace = true, features = ["testing"] }
 common-query.workspace = true
--- a/src/datanode/src/config.rs
+++ b/src/datanode/src/config.rs
@@ -32,7 +32,7 @@ use servers::heartbeat_options::HeartbeatOptions;
 use servers::http::HttpOptions;
 use servers::Mode;

-pub const DEFAULT_OBJECT_STORE_CACHE_SIZE: ReadableSize = ReadableSize::gb(1);
+pub const DEFAULT_OBJECT_STORE_CACHE_SIZE: ReadableSize = ReadableSize::gb(5);

 /// Default data home in file storage
 const DEFAULT_DATA_HOME: &str = "/tmp/greptimedb";
@@ -60,6 +60,11 @@ impl ObjectStoreConfig {
        }
    }

+    /// Returns true when it's a remote object storage such as AWS s3 etc.
+    pub fn is_object_storage(&self) -> bool {
+        !matches!(self, Self::File(_))
+    }
+
    /// Returns the object storage configuration name, return the provider name if it's empty.
    pub fn config_name(&self) -> &str {
        let name = match self {
@@ -91,6 +96,13 @@ pub struct StorageConfig {
    pub providers: Vec<ObjectStoreConfig>,
 }

+impl StorageConfig {
+    /// Returns true when the default storage config is a remote object storage service such as AWS S3, etc.
+    pub fn is_object_storage(&self) -> bool {
+        self.store.is_object_storage()
+    }
+}
+
 impl Default for StorageConfig {
    fn default() -> Self {
        Self {
@@ -452,6 +464,20 @@ mod tests {
        assert_eq!("S3", s3_config.provider_name());
    }

+    #[test]
+    fn test_is_object_storage() {
+        let store = ObjectStoreConfig::default();
+        assert!(!store.is_object_storage());
+        let s3_config = ObjectStoreConfig::S3(S3Config::default());
+        assert!(s3_config.is_object_storage());
+        let oss_config = ObjectStoreConfig::Oss(OssConfig::default());
+        assert!(oss_config.is_object_storage());
+        let gcs_config = ObjectStoreConfig::Gcs(GcsConfig::default());
+        assert!(gcs_config.is_object_storage());
+        let azblob_config = ObjectStoreConfig::Azblob(AzblobConfig::default());
+        assert!(azblob_config.is_object_storage());
+    }
+
    #[test]
    fn test_secstr() {
        let toml_str = r#"
--- a/src/datanode/src/datanode.rs
+++ b/src/datanode/src/datanode.rs
@@ -18,11 +18,11 @@ use std::path::Path;
 use std::sync::Arc;
 use std::time::Duration;

-use catalog::kvbackend::CachedKvBackendBuilder;
 use catalog::memory::MemoryCatalogManager;
 use common_base::Plugins;
 use common_error::ext::BoxedError;
 use common_greptimedb_telemetry::GreptimeDBTelemetryTask;
+use common_meta::cache::{LayeredCacheRegistry, SchemaCacheRef, TableSchemaCacheRef};
 use common_meta::key::datanode_table::{DatanodeTableManager, DatanodeTableValue};
 use common_meta::key::{SchemaMetadataManager, SchemaMetadataManagerRef};
 use common_meta::kv_backend::KvBackendRef;
@@ -57,9 +57,9 @@ use tokio::sync::Notify;

 use crate::config::{DatanodeOptions, RegionEngineConfig, StorageConfig};
 use crate::error::{
-    self, BuildMitoEngineSnafu, CreateDirSnafu, GetMetadataSnafu, MissingKvBackendSnafu,
-    MissingNodeIdSnafu, OpenLogStoreSnafu, Result, ShutdownInstanceSnafu, ShutdownServerSnafu,
-    StartServerSnafu,
+    self, BuildMitoEngineSnafu, CreateDirSnafu, GetMetadataSnafu, MissingCacheSnafu,
+    MissingKvBackendSnafu, MissingNodeIdSnafu, OpenLogStoreSnafu, Result, ShutdownInstanceSnafu,
+    ShutdownServerSnafu, StartServerSnafu,
 };
 use crate::event_listener::{
    new_region_server_event_channel, NoopRegionServerEventListener, RegionServerEventListenerRef,
@@ -160,6 +160,7 @@ pub struct DatanodeBuilder {
    plugins: Plugins,
    meta_client: Option<MetaClientRef>,
    kv_backend: Option<KvBackendRef>,
+    cache_registry: Option<Arc<LayeredCacheRegistry>>,
 }

 impl DatanodeBuilder {
@@ -171,6 +172,7 @@ impl DatanodeBuilder {
            plugins,
            meta_client: None,
            kv_backend: None,
+            cache_registry: None,
        }
    }

@@ -181,6 +183,13 @@ impl DatanodeBuilder {
        }
    }

+    pub fn with_cache_registry(self, cache_registry: Arc<LayeredCacheRegistry>) -> Self {
+        Self {
+            cache_registry: Some(cache_registry),
+            ..self
+        }
+    }
+
    pub fn with_kv_backend(self, kv_backend: KvBackendRef) -> Self {
        Self {
            kv_backend: Some(kv_backend),
@@ -209,10 +218,16 @@ impl DatanodeBuilder {
            (Box::new(NoopRegionServerEventListener) as _, None)
        };

-        let cached_kv_backend = Arc::new(CachedKvBackendBuilder::new(kv_backend.clone()).build());
+        let cache_registry = self.cache_registry.take().context(MissingCacheSnafu)?;
+        let schema_cache: SchemaCacheRef = cache_registry.get().context(MissingCacheSnafu)?;
+        let table_id_schema_cache: TableSchemaCacheRef =
+            cache_registry.get().context(MissingCacheSnafu)?;

-        let schema_metadata_manager =
-            Arc::new(SchemaMetadataManager::new(cached_kv_backend.clone()));
+        let schema_metadata_manager = Arc::new(SchemaMetadataManager::new(
+            kv_backend.clone(),
+            table_id_schema_cache,
+            schema_cache,
+        ));
        let region_server = self
            .new_region_server(schema_metadata_manager, region_event_listener)
            .await?;
@@ -248,7 +263,7 @@ impl DatanodeBuilder {
                    &self.opts,
                    region_server.clone(),
                    meta_client,
-                    cached_kv_backend,
+                    cache_registry,
                )
                .await?,
            )
@@ -413,10 +428,16 @@ impl DatanodeBuilder {
    async fn build_mito_engine(
        opts: &DatanodeOptions,
        object_store_manager: ObjectStoreManagerRef,
-        config: MitoConfig,
+        mut config: MitoConfig,
        schema_metadata_manager: SchemaMetadataManagerRef,
        plugins: Plugins,
    ) -> Result<MitoEngine> {
+        if opts.storage.is_object_storage() {
+            // Enable the write cache when setting object storage
+            config.enable_experimental_write_cache = true;
+            info!("Configured 'enable_experimental_write_cache=true' for mito engine.");
+        }
+
        let mito_engine = match &opts.wal {
            DatanodeWalConfig::RaftEngine(raft_engine_config) => MitoEngine::new(
                &opts.storage.data_home,
@@ -591,7 +612,9 @@ mod tests {
    use std::collections::{BTreeMap, HashMap};
    use std::sync::Arc;

+    use cache::build_datanode_cache_registry;
    use common_base::Plugins;
+    use common_meta::cache::LayeredCacheRegistryBuilder;
    use common_meta::key::datanode_table::DatanodeTableManager;
    use common_meta::kv_backend::memory::MemoryKvBackend;
    use common_meta::kv_backend::KvBackendRef;
@@ -628,13 +651,21 @@ mod tests {

        mock_region_server.register_engine(mock_region.clone());

+        let kv_backend = Arc::new(MemoryKvBackend::new());
+        let layered_cache_registry = Arc::new(
+            LayeredCacheRegistryBuilder::default()
+                .add_cache_registry(build_datanode_cache_registry(kv_backend))
+                .build(),
+        );
+
        let builder = DatanodeBuilder::new(
            DatanodeOptions {
                node_id: Some(0),
                ..Default::default()
            },
            Plugins::default(),
-        );
+        )
+        .with_cache_registry(layered_cache_registry);

        let kv = Arc::new(MemoryKvBackend::default()) as _;
        setup_table_datanode(&kv).await;
--- a/src/datanode/src/error.rs
+++ b/src/datanode/src/error.rs
@@ -193,6 +193,14 @@ pub enum Error {
        location: Location,
    },

+    #[snafu(display("Failed to build http client"))]
+    BuildHttpClient {
+        #[snafu(implicit)]
+        location: Location,
+        #[snafu(source)]
+        error: reqwest::Error,
+    },
+
    #[snafu(display("Missing required field: {}", name))]
    MissingRequiredField {
        name: String,
@@ -364,6 +372,12 @@ pub enum Error {
        #[snafu(implicit)]
        location: Location,
    },
+
+    #[snafu(display("Cache not found in registry"))]
+    MissingCache {
+        #[snafu(implicit)]
+        location: Location,
+    },
 }

 pub type Result<T> = std::result::Result<T, Error>;
@@ -400,9 +414,10 @@ impl ErrorExt for Error {
            | MissingKvBackend { .. }
            | TomlFormat { .. } => StatusCode::InvalidArguments,

-            PayloadNotExist { .. } | Unexpected { .. } | WatchAsyncTaskChange { .. } => {
-                StatusCode::Unexpected
-            }
+            PayloadNotExist { .. }
+            | Unexpected { .. }
+            | WatchAsyncTaskChange { .. }
+            | BuildHttpClient { .. } => StatusCode::Unexpected,

            AsyncTaskExecute { source, .. } => source.status_code(),

@@ -431,6 +446,7 @@ impl ErrorExt for Error {
            ConcurrentQueryLimiterClosed { .. } | ConcurrentQueryLimiterTimeout { .. } => {
                StatusCode::RegionBusy
            }
+            MissingCache { .. } => StatusCode::Internal,
        }
    }

--- a/Show More
+++ b/Show More