Revert "build(deps): upgrade opendal to 0.46 (#4037 )"

This reverts commit f9db5ff0d6.
Merge branch 'main' into avoid-query-meta
2025-12-23 06:30:05 +00:00 · 2024-06-03 20:28:59 +08:00 · 2024-05-31 18:16:36 +08:00 · 2024-05-31 18:16:08 +08:00
1922 changed files with 67507 additions and 212166 deletions
--- a/.cargo/config.toml
+++ b/.cargo/config.toml
@@ -3,12 +3,3 @@ linker = "aarch64-linux-gnu-gcc"

 [alias]
 sqlness = "run --bin sqlness-runner --"
-
-[unstable.git]
-shallow_index = true
-shallow_deps = true
-[unstable.gitoxide]
-fetch = true
-checkout = true
-list_files = true
-internal_use_git2 = false
--- a/.coderabbit.yaml
+++ b/.coderabbit.yaml
@@ -1,15 +0,0 @@
-# yaml-language-server: $schema=https://coderabbit.ai/integrations/schema.v2.json
-language: "en-US"
-early_access: false
-reviews:
-  profile: "chill"
-  request_changes_workflow: false
-  high_level_summary: true
-  poem: true
-  review_status: true
-  collapse_walkthrough: false
-  auto_review:
-    enabled: false
-    drafts: false
-chat:
-  auto_reply: true
--- a/.env.example
+++ b/.env.example
@@ -14,11 +14,10 @@ GT_AZBLOB_CONTAINER=AZBLOB container
 GT_AZBLOB_ACCOUNT_NAME=AZBLOB account name
 GT_AZBLOB_ACCOUNT_KEY=AZBLOB account key
 GT_AZBLOB_ENDPOINT=AZBLOB endpoint
-# Settings for gcs test
-GT_GCS_BUCKET = GCS bucket
+# Settings for gcs test 
+GT_GCS_BUCKET = GCS bucket 
 GT_GCS_SCOPE  = GCS scope
-GT_GCS_CREDENTIAL_PATH = GCS credential path
-GT_GCS_CREDENTIAL = GCS credential
+GT_GCS_CREDENTIAL_PATH = GCS credential path 
 GT_GCS_ENDPOINT = GCS end point
 # Settings for kafka wal test
 GT_KAFKA_ENDPOINTS = localhost:9092
@@ -29,8 +28,3 @@ GT_MYSQL_ADDR = localhost:4002
 # Setting for unstable fuzz tests
 GT_FUZZ_BINARY_PATH=/path/to/
 GT_FUZZ_INSTANCE_ROOT_DIR=/tmp/unstable_greptime
-GT_FUZZ_INPUT_MAX_ROWS=2048
-GT_FUZZ_INPUT_MAX_TABLES=32
-GT_FUZZ_INPUT_MAX_COLUMNS=32
-GT_FUZZ_INPUT_MAX_ALTER_ACTIONS=256
-GT_FUZZ_INPUT_MAX_INSERT_ACTIONS=8
--- a/.github/actions/build-dev-builder-images/action.yml
+++ b/.github/actions/build-dev-builder-images/action.yml
@@ -41,14 +41,7 @@ runs:
        username: ${{ inputs.dockerhub-image-registry-username }}
        password: ${{ inputs.dockerhub-image-registry-token }}

-    - name: Set up qemu for multi-platform builds
-      uses: docker/setup-qemu-action@v3
-      with:
-        platforms: linux/amd64,linux/arm64
-        # The latest version will lead to segmentation fault.
-        image: tonistiigi/binfmt:qemu-v7.0.0-28
-
-    - name: Build and push dev-builder-ubuntu image # Build image for amd64 and arm64 platform.
+    - name: Build and push dev-builder-ubuntu image
      shell: bash
      if: ${{ inputs.build-dev-builder-ubuntu == 'true' }}
      run: |
@@ -57,9 +50,9 @@ runs:
          BUILDX_MULTI_PLATFORM_BUILD=all \
          IMAGE_REGISTRY=${{ inputs.dockerhub-image-registry }} \
          IMAGE_NAMESPACE=${{ inputs.dockerhub-image-namespace }} \
-          DEV_BUILDER_IMAGE_TAG=${{ inputs.version }}
+          IMAGE_TAG=${{ inputs.version }}

-    - name: Build and push dev-builder-centos image # Only build image for amd64 platform.
+    - name: Build and push dev-builder-centos image
      shell: bash
      if: ${{ inputs.build-dev-builder-centos == 'true' }}
      run: |
@@ -68,7 +61,7 @@ runs:
          BUILDX_MULTI_PLATFORM_BUILD=amd64 \
          IMAGE_REGISTRY=${{ inputs.dockerhub-image-registry }} \
          IMAGE_NAMESPACE=${{ inputs.dockerhub-image-namespace }} \
-          DEV_BUILDER_IMAGE_TAG=${{ inputs.version }}
+          IMAGE_TAG=${{ inputs.version }}

    - name: Build and push dev-builder-android image # Only build image for amd64 platform.
      shell: bash
@@ -76,7 +69,8 @@ runs:
      run: |
        make dev-builder \
          BASE_IMAGE=android \
-          BUILDX_MULTI_PLATFORM_BUILD=amd64 \
          IMAGE_REGISTRY=${{ inputs.dockerhub-image-registry }} \
          IMAGE_NAMESPACE=${{ inputs.dockerhub-image-namespace }} \
-          DEV_BUILDER_IMAGE_TAG=${{ inputs.version }}
+          IMAGE_TAG=${{ inputs.version }} && \
+
+        docker push ${{ inputs.dockerhub-image-registry }}/${{ inputs.dockerhub-image-namespace }}/dev-builder-android:${{ inputs.version }}
--- a/.github/actions/build-greptime-binary/action.yml
+++ b/.github/actions/build-greptime-binary/action.yml
@@ -24,14 +24,6 @@ inputs:
    description: Build android artifacts
    required: false
    default: 'false'
-  image-namespace:
-    description: Image Namespace
-    required: false
-    default: 'greptime'
-  image-registry:
-    description: Image Registry
-    required: false
-    default: 'docker.io'
 runs:
  using: composite
  steps:
@@ -43,9 +35,7 @@ runs:
        make build-by-dev-builder \
          CARGO_PROFILE=${{ inputs.cargo-profile }} \
          FEATURES=${{ inputs.features }} \
-          BASE_IMAGE=${{ inputs.base-image }} \
-          IMAGE_NAMESPACE=${{ inputs.image-namespace }} \
-          IMAGE_REGISTRY=${{ inputs.image-registry }}
+          BASE_IMAGE=${{ inputs.base-image }}

    - name: Upload artifacts
      uses: ./.github/actions/upload-artifacts
@@ -54,7 +44,7 @@ runs:
        PROFILE_TARGET: ${{ inputs.cargo-profile == 'dev' && 'debug' || inputs.cargo-profile }}
      with:
        artifacts-dir: ${{ inputs.artifacts-dir }}
-        target-files: ./target/$PROFILE_TARGET/greptime
+        target-file: ./target/$PROFILE_TARGET/greptime
        version: ${{ inputs.version }}
        working-dir: ${{ inputs.working-dir }}

@@ -63,15 +53,13 @@ runs:
      shell: bash
      if: ${{ inputs.build-android-artifacts == 'true' }}
      run: |
-        cd ${{ inputs.working-dir }} && make strip-android-bin \
-          IMAGE_NAMESPACE=${{ inputs.image-namespace }} \
-          IMAGE_REGISTRY=${{ inputs.image-registry }}
+        cd ${{ inputs.working-dir }} && make strip-android-bin

    - name: Upload android artifacts
      uses: ./.github/actions/upload-artifacts
      if: ${{ inputs.build-android-artifacts == 'true' }}
      with:
        artifacts-dir: ${{ inputs.artifacts-dir }}
-        target-files: ./target/aarch64-linux-android/release/greptime
+        target-file: ./target/aarch64-linux-android/release/greptime
        version: ${{ inputs.version }}
        working-dir: ${{ inputs.working-dir }}
--- a/.github/actions/build-greptime-images/action.yml
+++ b/.github/actions/build-greptime-images/action.yml
@@ -34,8 +34,8 @@ inputs:
    required: true
  push-latest-tag:
    description: Whether to push the latest tag
-    required: true
-    default: 'false'
+    required: false
+    default: 'true'
 runs:
  using: composite
  steps:
@@ -47,11 +47,7 @@ runs:
        password: ${{ inputs.image-registry-password }}

    - name: Set up qemu for multi-platform builds
-      uses: docker/setup-qemu-action@v3
-      with:
-        platforms: linux/amd64,linux/arm64
-        # The latest version will lead to segmentation fault.
-        image: tonistiigi/binfmt:qemu-v7.0.0-28
+      uses: docker/setup-qemu-action@v2

    - name: Set up buildx
      uses: docker/setup-buildx-action@v2
--- a/.github/actions/build-images/action.yml
+++ b/.github/actions/build-images/action.yml
@@ -22,8 +22,8 @@ inputs:
    required: true
  push-latest-tag:
    description: Whether to push the latest tag
-    required: true
-    default: 'false'
+    required: false
+    default: 'true'
  dev-mode:
    description: Enable dev mode, only build standard greptime
    required: false
@@ -41,8 +41,8 @@ runs:
        image-name: ${{ inputs.image-name }}
        image-tag: ${{ inputs.version }}
        docker-file: docker/ci/ubuntu/Dockerfile
-        amd64-artifact-name: greptime-linux-amd64-${{ inputs.version }}
-        arm64-artifact-name: greptime-linux-arm64-${{ inputs.version }}
+        amd64-artifact-name: greptime-linux-amd64-pyo3-${{ inputs.version }}
+        arm64-artifact-name: greptime-linux-arm64-pyo3-${{ inputs.version }}
        platforms: linux/amd64,linux/arm64
        push-latest-tag: ${{ inputs.push-latest-tag }}

--- a/.github/actions/build-linux-artifacts/action.yml
+++ b/.github/actions/build-linux-artifacts/action.yml
@@ -17,12 +17,6 @@ inputs:
    description: Enable dev mode, only build standard greptime
    required: false
    default: "false"
-  image-namespace:
-    description: Image Namespace
-    required: true
-  image-registry:
-    description: Image Registry
-    required: true
  working-dir:
    description: Working directory to build the artifacts
    required: false
@@ -36,9 +30,7 @@ runs:
      # NOTE: If the BUILD_JOBS > 4, it's always OOM in EC2 instance.
      run: |
        cd ${{ inputs.working-dir }} && \
-        make run-it-in-container BUILD_JOBS=4 \
-        IMAGE_NAMESPACE=${{ inputs.image-namespace }} \
-        IMAGE_REGISTRY=${{ inputs.image-registry }}
+        make run-it-in-container BUILD_JOBS=4

    - name: Upload sqlness logs
      if: ${{ failure() && inputs.disable-run-tests == 'false' }} # Only upload logs when the integration tests failed.
@@ -48,17 +40,26 @@ runs:
        path: /tmp/greptime-*.log
        retention-days: 3

-    - name: Build greptime # Builds standard greptime binary
+    - name: Build standard greptime
      uses: ./.github/actions/build-greptime-binary
      with:
        base-image: ubuntu
-        features: servers/dashboard,pg_kvbackend,mysql_kvbackend
+        features: pyo3_backend,servers/dashboard
+        cargo-profile: ${{ inputs.cargo-profile }}
+        artifacts-dir: greptime-linux-${{ inputs.arch }}-pyo3-${{ inputs.version }}
+        version: ${{ inputs.version }}
+        working-dir: ${{ inputs.working-dir }}
+
+    - name: Build greptime without pyo3
+      if: ${{ inputs.dev-mode == 'false' }}
+      uses: ./.github/actions/build-greptime-binary
+      with:
+        base-image: ubuntu
+        features: servers/dashboard
        cargo-profile: ${{ inputs.cargo-profile }}
        artifacts-dir: greptime-linux-${{ inputs.arch }}-${{ inputs.version }}
        version: ${{ inputs.version }}
        working-dir: ${{ inputs.working-dir }}
-        image-registry: ${{ inputs.image-registry }}
-        image-namespace: ${{ inputs.image-namespace }}

    - name: Clean up the target directory # Clean up the target directory for the centos7 base image, or it will still use the objects of last build.
      shell: bash
@@ -70,13 +71,11 @@ runs:
      if: ${{ inputs.arch == 'amd64' && inputs.dev-mode == 'false' }} # Builds greptime for centos if the host machine is amd64.
      with:
        base-image: centos
-        features: servers/dashboard,pg_kvbackend,mysql_kvbackend
+        features: servers/dashboard
        cargo-profile: ${{ inputs.cargo-profile }}
        artifacts-dir: greptime-linux-${{ inputs.arch }}-centos-${{ inputs.version }}
        version: ${{ inputs.version }}
        working-dir: ${{ inputs.working-dir }}
-        image-registry: ${{ inputs.image-registry }}
-        image-namespace: ${{ inputs.image-namespace }}

    - name: Build greptime on android base image
      uses: ./.github/actions/build-greptime-binary
@@ -87,5 +86,3 @@ runs:
        version: ${{ inputs.version }}
        working-dir: ${{ inputs.working-dir }}
        build-android-artifacts: true
-        image-registry: ${{ inputs.image-registry }}
-        image-namespace: ${{ inputs.image-namespace }}
--- a/.github/actions/build-macos-artifacts/action.yml
+++ b/.github/actions/build-macos-artifacts/action.yml
@@ -4,6 +4,9 @@ inputs:
  arch:
    description: Architecture to build
    required: true
+  rust-toolchain:
+    description: Rust toolchain to use
+    required: true
  cargo-profile:
    description: Cargo profile to build
    required: true
@@ -40,9 +43,10 @@ runs:
        brew install protobuf

    - name: Install rust toolchain
-      uses: actions-rust-lang/setup-rust-toolchain@v1
+      uses: dtolnay/rust-toolchain@master
      with:
-        target: ${{ inputs.arch }}
+        toolchain: ${{ inputs.rust-toolchain }}
+        targets: ${{ inputs.arch }}

    - name: Start etcd # For integration tests.
      if: ${{ inputs.disable-run-tests == 'false' }}
@@ -55,16 +59,9 @@ runs:
      if: ${{ inputs.disable-run-tests == 'false' }}
      uses: taiki-e/install-action@nextest

-    # Get proper backtraces in mac Sonoma. Currently there's an issue with the new
-    # linker that prevents backtraces from getting printed correctly.
-    #
-    # <https://github.com/rust-lang/rust/issues/113783>
    - name: Run integration tests
      if: ${{ inputs.disable-run-tests == 'false' }}
      shell: bash
-      env:
-        CARGO_BUILD_RUSTFLAGS: "-Clink-arg=-Wl,-ld_classic"
-        SQLNESS_OPTS: "--preserve-state"
      run: |
        make test sqlness-test

@@ -78,8 +75,6 @@ runs:

    - name: Build greptime binary
      shell: bash
-      env:
-        CARGO_BUILD_RUSTFLAGS: "-Clink-arg=-Wl,-ld_classic"
      run: |
        make build \
        CARGO_PROFILE=${{ inputs.cargo-profile }} \
@@ -90,5 +85,5 @@ runs:
      uses: ./.github/actions/upload-artifacts
      with:
        artifacts-dir: ${{ inputs.artifacts-dir }}
-        target-files: target/${{ inputs.arch }}/${{ inputs.cargo-profile }}/greptime
+        target-file: target/${{ inputs.arch }}/${{ inputs.cargo-profile }}/greptime
        version: ${{ inputs.version }}
--- a/.github/actions/build-windows-artifacts/action.yml
+++ b/.github/actions/build-windows-artifacts/action.yml
@@ -4,6 +4,9 @@ inputs:
  arch:
    description: Architecture to build
    required: true
+  rust-toolchain:
+    description: Rust toolchain to use
+    required: true
  cargo-profile:
    description: Cargo profile to build
    required: true
@@ -25,14 +28,24 @@ runs:
    - uses: arduino/setup-protoc@v3

    - name: Install rust toolchain
-      uses: actions-rust-lang/setup-rust-toolchain@v1
+      uses: dtolnay/rust-toolchain@master
      with:
-        target: ${{ inputs.arch }}
+        toolchain: ${{ inputs.rust-toolchain }}
+        targets: ${{ inputs.arch }}
        components: llvm-tools-preview

    - name: Rust Cache
      uses: Swatinem/rust-cache@v2

+    - name: Install Python
+      uses: actions/setup-python@v5
+      with:
+        python-version: '3.10'
+
+    - name: Install PyArrow Package
+      shell: pwsh
+      run: pip install pyarrow
+
    - name: Install WSL distribution
      uses: Vampire/setup-wsl@v2
      with:
@@ -47,15 +60,15 @@ runs:
      shell: pwsh
      run: make test sqlness-test
      env:
+        RUSTUP_WINDOWS_PATH_ADD_BIN: 1 # Workaround for https://github.com/nextest-rs/nextest/issues/1493
        RUST_BACKTRACE: 1
-        SQLNESS_OPTS: "--preserve-state"

    - name: Upload sqlness logs
      if: ${{ failure() }} # Only upload logs when the integration tests failed.
      uses: actions/upload-artifact@v4
      with:
        name: sqlness-logs
-        path: C:\Users\RUNNER~1\AppData\Local\Temp\sqlness*
+        path: /tmp/greptime-*.log
        retention-days: 3

    - name: Build greptime binary
@@ -66,5 +79,5 @@ runs:
      uses: ./.github/actions/upload-artifacts
      with:
        artifacts-dir: ${{ inputs.artifacts-dir }}
-        target-files: target/${{ inputs.arch }}/${{ inputs.cargo-profile }}/greptime,target/${{ inputs.arch }}/${{ inputs.cargo-profile }}/greptime.pdb
+        target-file: target/${{ inputs.arch }}/${{ inputs.cargo-profile }}/greptime
        version: ${{ inputs.version }}
--- a/.github/actions/publish-github-release/action.yml
+++ b/.github/actions/publish-github-release/action.yml
@@ -9,8 +9,8 @@ runs:
  steps:
    # Download artifacts from previous jobs, the artifacts will be downloaded to:
    # ${WORKING_DIR}
-    #   |- greptime-darwin-amd64-v0.5.0/greptime-darwin-amd64-v0.5.0.tar.gz
-    #   |- greptime-darwin-amd64-v0.5.0.sha256sum/greptime-darwin-amd64-v0.5.0.sha256sum
+    #   |- greptime-darwin-amd64-pyo3-v0.5.0/greptime-darwin-amd64-pyo3-v0.5.0.tar.gz
+    #   |- greptime-darwin-amd64-pyo3-v0.5.0.sha256sum/greptime-darwin-amd64-pyo3-v0.5.0.sha256sum
    #   |- greptime-darwin-amd64-v0.5.0/greptime-darwin-amd64-v0.5.0.tar.gz
    #   |- greptime-darwin-amd64-v0.5.0.sha256sum/greptime-darwin-amd64-v0.5.0.sha256sum
    #   ...
--- a/.github/actions/release-cn-artifacts/action.yaml
+++ b/.github/actions/release-cn-artifacts/action.yaml
@@ -51,8 +51,8 @@ inputs:
    required: true
  upload-to-s3:
    description: Upload to S3
-    required: true
-    default: 'false'
+    required: false
+    default: 'true'
  artifacts-dir:
    description: Directory to store artifacts
    required: false
@@ -77,21 +77,13 @@ runs:
      with:
        path: ${{ inputs.artifacts-dir }}

-    - name: Install s5cmd
-      shell: bash
-      run: |
-        wget https://github.com/peak/s5cmd/releases/download/v2.3.0/s5cmd_2.3.0_Linux-64bit.tar.gz
-        tar -xzf s5cmd_2.3.0_Linux-64bit.tar.gz
-        sudo mv s5cmd /usr/local/bin/
-        sudo chmod +x /usr/local/bin/s5cmd
-
    - name: Release artifacts to cn region
      uses: nick-invision/retry@v2
      if: ${{ inputs.upload-to-s3 == 'true' }}
      env:
        AWS_ACCESS_KEY_ID: ${{ inputs.aws-cn-access-key-id }}
        AWS_SECRET_ACCESS_KEY: ${{ inputs.aws-cn-secret-access-key }}
-        AWS_REGION: ${{ inputs.aws-cn-region }}
+        AWS_DEFAULT_REGION: ${{ inputs.aws-cn-region }}
        UPDATE_VERSION_INFO: ${{ inputs.update-version-info }}
      with:
        max_attempts: ${{ inputs.upload-max-retry-times }}
@@ -131,10 +123,10 @@ runs:
        DST_REGISTRY_PASSWORD: ${{ inputs.dst-image-registry-password }}
      run: |
        ./.github/scripts/copy-image.sh \
-         ${{ inputs.src-image-registry }}/${{ inputs.src-image-namespace }}/${{ inputs.src-image-name }}-centos:${{ inputs.version }} \
+         ${{ inputs.src-image-registry }}/${{ inputs.src-image-namespace }}/${{ inputs.src-image-name }}-centos:latest \
         ${{ inputs.dst-image-registry }}/${{ inputs.dst-image-namespace }}

-    - name: Push latest greptimedb-centos image from DockerHub to ACR
+    - name: Push greptimedb-centos image from DockerHub to ACR
      shell: bash
      if: ${{ inputs.dev-mode == 'false' && inputs.push-latest-tag == 'true' }}
      env:
--- a/.github/actions/setup-chaos/action.yml
+++ b/.github/actions/setup-chaos/action.yml
@@ -1,17 +0,0 @@
-name: Setup Kind
-description: Deploy Kind
-runs:
-  using: composite
-  steps:
-  - uses: actions/checkout@v4
-  - name: Create kind cluster
-    shell: bash
-    run: |
-      helm repo add chaos-mesh https://charts.chaos-mesh.org
-      kubectl create ns chaos-mesh
-      helm install chaos-mesh chaos-mesh/chaos-mesh -n=chaos-mesh --version 2.6.3
-  - name: Print Chaos-mesh
-    if: always()
-    shell: bash
-    run: | 
-      kubectl get po -n chaos-mesh
--- a/.github/actions/setup-etcd-cluster/action.yml
+++ b/.github/actions/setup-etcd-cluster/action.yml
@@ -2,7 +2,7 @@ name: Setup Etcd cluster
 description: Deploy Etcd cluster on Kubernetes
 inputs:
  etcd-replicas:
-    default: 1
+    default: 3
    description: "Etcd replicas"
  namespace:
    default: "etcd-cluster"
@@ -18,8 +18,6 @@ runs:
        --set replicaCount=${{ inputs.etcd-replicas }} \
        --set resources.requests.cpu=50m \
        --set resources.requests.memory=128Mi \
-        --set resources.limits.cpu=1500m \
-        --set resources.limits.memory=2Gi \
        --set auth.rbac.create=false \
        --set auth.rbac.token.enabled=false \
        --set persistence.size=2Gi \
--- a/.github/actions/setup-greptimedb-cluster/action.yml
+++ b/.github/actions/setup-greptimedb-cluster/action.yml
@@ -8,7 +8,7 @@ inputs:
    default: 2
    description: "Number of Datanode replicas"
  meta-replicas:
-    default: 1
+    default: 3
    description: "Number of Metasrv replicas"
  image-registry: 
    default: "docker.io"
@@ -22,43 +22,34 @@ inputs:
  etcd-endpoints:
    default: "etcd.etcd-cluster.svc.cluster.local:2379"
    description: "Etcd endpoints"
-  values-filename:
-    default: "with-minio.yaml"
-  enable-region-failover:
-    default: false

 runs:
  using: composite
  steps:
  - name: Install GreptimeDB operator
-    uses: nick-fields/retry@v3
-    with: 
-      timeout_minutes: 3
-      max_attempts: 3
-      shell: bash
-      command: |
-        helm repo add greptime https://greptimeteam.github.io/helm-charts/ 
-        helm repo update
-        helm upgrade \
-          --install \
-          --create-namespace \
-          greptimedb-operator greptime/greptimedb-operator \
-          -n greptimedb-admin \
-          --wait \
-          --wait-for-jobs
+    shell: bash
+    run: |
+      helm repo add greptime https://greptimeteam.github.io/helm-charts/ 
+      helm repo update
+      helm upgrade \
+        --install \
+        --create-namespace \
+        greptimedb-operator greptime/greptimedb-operator \
+        -n greptimedb-admin \
+        --wait \
+        --wait-for-jobs
  - name: Install GreptimeDB cluster
    shell: bash
    run: | 
      helm upgrade \
        --install my-greptimedb \
        --set meta.etcdEndpoints=${{ inputs.etcd-endpoints }} \
-        --set meta.enableRegionFailover=${{ inputs.enable-region-failover }} \
        --set image.registry=${{ inputs.image-registry }} \
        --set image.repository=${{ inputs.image-repository }}  \
        --set image.tag=${{ inputs.image-tag }} \
        --set base.podTemplate.main.resources.requests.cpu=50m \
        --set base.podTemplate.main.resources.requests.memory=256Mi \
-        --set base.podTemplate.main.resources.limits.cpu=2000m \
+        --set base.podTemplate.main.resources.limits.cpu=1000m \
        --set base.podTemplate.main.resources.limits.memory=2Gi \
        --set frontend.replicas=${{ inputs.frontend-replicas }} \
        --set datanode.replicas=${{ inputs.datanode-replicas }} \
@@ -66,7 +57,6 @@ runs:
        greptime/greptimedb-cluster \
        --create-namespace \
        -n my-greptimedb \
-        --values ./.github/actions/setup-greptimedb-cluster/${{ inputs.values-filename }} \
        --wait \
        --wait-for-jobs
  - name: Wait for GreptimeDB
--- a/.github/actions/setup-greptimedb-cluster/with-disk.yaml
+++ b/.github/actions/setup-greptimedb-cluster/with-disk.yaml
@@ -1,13 +0,0 @@
-meta:
-  configData: |-
-    [runtime]
-    global_rt_size = 4
-datanode:
-  configData: |-
-    [runtime]
-    global_rt_size = 4
-    compact_rt_size = 2
-frontend:
-  configData: |-
-    [runtime]
-    global_rt_size = 4
--- a/.github/actions/setup-greptimedb-cluster/with-minio-and-cache.yaml
+++ b/.github/actions/setup-greptimedb-cluster/with-minio-and-cache.yaml
@@ -1,33 +0,0 @@
-meta:
-  configData: |-
-    [runtime]
-    global_rt_size = 4
-
-    [datanode]
-    [datanode.client]
-    timeout = "120s"
-datanode:
-  configData: |-
-    [runtime]
-    global_rt_size = 4
-    compact_rt_size = 2
-    
-    [storage]
-    cache_path = "/data/greptimedb/s3cache"
-    cache_capacity = "256MB"
-frontend:
-  configData: |-
-    [runtime]
-    global_rt_size = 4
-
-    [meta_client]
-    ddl_timeout = "120s"
-objectStorage:
-  s3:
-    bucket: default
-    region: us-west-2
-    root: test-root
-    endpoint: http://minio.minio.svc.cluster.local 
-  credentials:
-    accessKeyId: rootuser
-    secretAccessKey: rootpass123
--- a/.github/actions/setup-greptimedb-cluster/with-minio.yaml
+++ b/.github/actions/setup-greptimedb-cluster/with-minio.yaml
@@ -1,29 +0,0 @@
-meta:
-  configData: |-
-    [runtime]
-    global_rt_size = 4
-    
-    [datanode]
-    [datanode.client]
-    timeout = "120s"
-datanode:
-  configData: |-
-    [runtime]
-    global_rt_size = 4
-    compact_rt_size = 2
-frontend:
-  configData: |-
-    [runtime]
-    global_rt_size = 4
-
-    [meta_client]
-    ddl_timeout = "120s"
-objectStorage:
-  s3:
-    bucket: default
-    region: us-west-2
-    root: test-root
-    endpoint: http://minio.minio.svc.cluster.local 
-  credentials:
-    accessKeyId: rootuser
-    secretAccessKey: rootpass123
--- a/.github/actions/setup-greptimedb-cluster/with-remote-wal.yaml
+++ b/.github/actions/setup-greptimedb-cluster/with-remote-wal.yaml
@@ -1,45 +0,0 @@
-meta:
-  configData: |-
-    [runtime]
-    global_rt_size = 4
-    
-    [wal]
-    provider = "kafka"
-    broker_endpoints = ["kafka.kafka-cluster.svc.cluster.local:9092"]
-    num_topics = 3
-
-        
-    [datanode]
-    [datanode.client]
-    timeout = "120s"
-datanode:
-  configData: |-
-    [runtime]
-    global_rt_size = 4
-    compact_rt_size = 2
-
-    [wal]
-    provider = "kafka"
-    broker_endpoints = ["kafka.kafka-cluster.svc.cluster.local:9092"]
-    linger = "2ms"
-frontend:
-  configData: |-
-    [runtime]
-    global_rt_size = 4
-
-    [meta_client]
-    ddl_timeout = "120s"
-objectStorage:
-  s3:
-    bucket: default
-    region: us-west-2
-    root: test-root
-    endpoint: http://minio.minio.svc.cluster.local 
-  credentials:
-    accessKeyId: rootuser
-    secretAccessKey: rootpass123
-remoteWal:
-   enabled: true
-   kafka:
-     brokerEndpoints: 
-      - "kafka.kafka-cluster.svc.cluster.local:9092"
--- a/.github/actions/setup-kafka-cluster/action.yml
+++ b/.github/actions/setup-kafka-cluster/action.yml
@@ -1,26 +0,0 @@
-name: Setup Kafka cluster
-description: Deploy Kafka cluster on Kubernetes
-inputs:
-  controller-replicas:
-    default: 3
-    description: "Kafka controller replicas"
-  namespace:
-    default: "kafka-cluster"
-
-runs:
-  using: composite
-  steps:
-  - name: Install Kafka cluster
-    shell: bash
-    run: | 
-      helm upgrade \
-        --install kafka oci://registry-1.docker.io/bitnamicharts/kafka \
-        --set controller.replicaCount=${{ inputs.controller-replicas }} \
-        --set controller.resources.requests.cpu=50m \
-        --set controller.resources.requests.memory=128Mi \
-        --set controller.resources.limits.cpu=2000m \
-        --set controller.resources.limits.memory=2Gi \
-        --set listeners.controller.protocol=PLAINTEXT \
-        --set listeners.client.protocol=PLAINTEXT \
-        --create-namespace \
-        -n ${{ inputs.namespace }}
--- a/.github/actions/setup-minio/action.yml
+++ b/.github/actions/setup-minio/action.yml
@@ -1,24 +0,0 @@
-name: Setup Minio cluster
-description: Deploy Minio cluster on Kubernetes
-inputs:
-  replicas:
-    default: 1
-    description: "replicas"
-
-runs:
-  using: composite
-  steps:
-  - name: Install Etcd cluster
-    shell: bash
-    run: | 
-      helm repo add minio https://charts.min.io/
-      helm upgrade --install minio \
-      --set resources.requests.memory=128Mi \
-      --set replicas=${{ inputs.replicas }} \
-      --set mode=standalone \
-      --set rootUser=rootuser,rootPassword=rootpass123 \
-      --set buckets[0].name=default \
-      --set service.port=80,service.targetPort=9000 \
-      minio/minio \
-      --create-namespace \
-      -n minio
--- a/.github/actions/setup-postgres-cluster/action.yml
+++ b/.github/actions/setup-postgres-cluster/action.yml
@@ -1,30 +0,0 @@
-name: Setup PostgreSQL
-description: Deploy PostgreSQL on Kubernetes
-inputs:
-  postgres-replicas:
-    default: 1
-    description: "Number of PostgreSQL replicas"
-  namespace:
-    default: "postgres-namespace"
-  postgres-version:
-    default: "14.2"
-    description: "PostgreSQL version"
-  storage-size:
-    default: "1Gi"
-    description: "Storage size for PostgreSQL"
-
-runs:
-  using: composite
-  steps:
-  - name: Install PostgreSQL
-    shell: bash
-    run: |
-      helm upgrade \
-        --install postgresql oci://registry-1.docker.io/bitnamicharts/postgresql \
-        --set replicaCount=${{ inputs.postgres-replicas }} \
-        --set image.tag=${{ inputs.postgres-version }} \
-        --set persistence.size=${{ inputs.storage-size }} \
-        --set postgresql.username=greptimedb \
-        --set postgresql.password=admin \
-        --create-namespace \
-        -n ${{ inputs.namespace }}
--- a/.github/actions/start-runner/action.yml
+++ b/.github/actions/start-runner/action.yml
@@ -38,7 +38,7 @@ runs:
  steps:
    - name: Configure AWS credentials
      if: startsWith(inputs.runner, 'ec2')
-      uses: aws-actions/configure-aws-credentials@v4
+      uses: aws-actions/configure-aws-credentials@v2
      with:
        aws-access-key-id: ${{ inputs.aws-access-key-id }}
        aws-secret-access-key: ${{ inputs.aws-secret-access-key }}
@@ -56,7 +56,7 @@ runs:

    - name: Start EC2 runner
      if: startsWith(inputs.runner, 'ec2')
-      uses: machulav/ec2-github-runner@v2.3.8
+      uses: machulav/ec2-github-runner@v2
      id: start-linux-arm64-ec2-runner
      with:
        mode: start
--- a/.github/actions/stop-runner/action.yml
+++ b/.github/actions/stop-runner/action.yml
@@ -25,7 +25,7 @@ runs:
  steps:
    - name: Configure AWS credentials
      if: ${{ inputs.label && inputs.ec2-instance-id }}
-      uses: aws-actions/configure-aws-credentials@v4
+      uses: aws-actions/configure-aws-credentials@v2
      with:
        aws-access-key-id: ${{ inputs.aws-access-key-id }}
        aws-secret-access-key: ${{ inputs.aws-secret-access-key }}
@@ -33,7 +33,7 @@ runs:

    - name: Stop EC2 runner
      if: ${{ inputs.label && inputs.ec2-instance-id }}
-      uses: machulav/ec2-github-runner@v2.3.8
+      uses: machulav/ec2-github-runner@v2
      with:
        mode: stop
        label: ${{ inputs.label }}
--- a/.github/actions/upload-artifacts/action.yml
+++ b/.github/actions/upload-artifacts/action.yml
@@ -4,8 +4,8 @@ inputs:
  artifacts-dir:
    description: Directory to store artifacts
    required: true
-  target-files:
-    description: The multiple target files to upload, separated by comma
+  target-file:
+    description: The path of the target artifact
    required: false
  version:
    description: Version of the artifact
@@ -18,21 +18,17 @@ runs:
  using: composite
  steps:
    - name: Create artifacts directory
-      if: ${{ inputs.target-files != '' }}
+      if: ${{ inputs.target-file != '' }}
      working-directory: ${{ inputs.working-dir }}
      shell: bash
      run: |
-        set -e
-        mkdir -p ${{ inputs.artifacts-dir }}
-        IFS=',' read -ra FILES <<< "${{ inputs.target-files }}"
-        for file in "${FILES[@]}"; do
-          cp "$file" ${{ inputs.artifacts-dir }}/
-        done
+        mkdir -p ${{ inputs.artifacts-dir }} && \
+        cp ${{ inputs.target-file }} ${{ inputs.artifacts-dir }}

    # The compressed artifacts will use the following layout:
-    # greptime-linux-amd64-v0.3.0sha256sum
-    # greptime-linux-amd64-v0.3.0.tar.gz
-    #   greptime-linux-amd64-v0.3.0
+    # greptime-linux-amd64-pyo3-v0.3.0sha256sum
+    # greptime-linux-amd64-pyo3-v0.3.0.tar.gz
+    #   greptime-linux-amd64-pyo3-v0.3.0
    #   └── greptime
    - name: Compress artifacts and calculate checksum
      working-directory: ${{ inputs.working-dir }}
--- a/.github/cargo-blacklist.txt
+++ b/.github/cargo-blacklist.txt
@@ -1,3 +0,0 @@
-native-tls
-openssl
-aws-lc-sys
--- a/.github/pull_request_template.md
+++ b/.github/pull_request_template.md
@@ -4,8 +4,7 @@ I hereby agree to the terms of the [GreptimeDB CLA](https://github.com/GreptimeT

 ## What's changed and what's your intention?

-<!--    
- __!!! DO NOT LEAVE THIS BLOCK EMPTY !!!__
+__!!! DO NOT LEAVE THIS BLOCK EMPTY !!!__

 Please explain IN DETAIL what the changes are in this PR and why they are needed:

@@ -13,14 +12,9 @@ Please explain IN DETAIL what the changes are in this PR and why they are needed
 - How does this PR work? Need a brief introduction for the changed logic (optional)
 - Describe clearly one logical change and avoid lazy messages (optional)
 - Describe any limitations of the current code (optional)
- Describe if this PR will break **API or data compatibility**  (optional)
-->

-## PR Checklist
-Please convert it to a draft if some of the following conditions are not met.
+## Checklist

 - [ ] I have written the necessary rustdoc comments.
 - [ ] I have added the necessary unit tests and integration tests.
 - [ ] This PR requires documentation updates.
- [ ] API changes are backward compatible.
- [ ] Schema or data changes are backward compatible.
--- a/.github/scripts/check-install-script.sh
+++ b/.github/scripts/check-install-script.sh
@@ -1,14 +0,0 @@
-#!/bin/sh
-
-set -e
-
-# Get the latest version of github.com/GreptimeTeam/greptimedb
-VERSION=$(curl -s https://api.github.com/repos/GreptimeTeam/greptimedb/releases/latest | jq -r '.tag_name')
-
-echo "Downloading the latest version: $VERSION"
-
-# Download the install script
-curl -fsSL https://raw.githubusercontent.com/greptimeteam/greptimedb/main/scripts/install.sh | sh -s $VERSION
-
-# Execute the `greptime` command
-./greptime --version
--- a/.github/scripts/upload-artifacts-to-s3.sh
+++ b/.github/scripts/upload-artifacts-to-s3.sh
@@ -27,13 +27,13 @@ function upload_artifacts() {
  # ├── latest-version.txt
  # ├── latest-nightly-version.txt
  # ├── v0.1.0
-  # │   ├── greptime-darwin-amd64-v0.1.0.sha256sum
-  # │   └── greptime-darwin-amd64-v0.1.0.tar.gz
+  # │   ├── greptime-darwin-amd64-pyo3-v0.1.0.sha256sum
+  # │   └── greptime-darwin-amd64-pyo3-v0.1.0.tar.gz
  # └── v0.2.0
-  #    ├── greptime-darwin-amd64-v0.2.0.sha256sum
-  #    └── greptime-darwin-amd64-v0.2.0.tar.gz
+  #    ├── greptime-darwin-amd64-pyo3-v0.2.0.sha256sum
+  #    └── greptime-darwin-amd64-pyo3-v0.2.0.tar.gz
  find "$ARTIFACTS_DIR" -type f \( -name "*.tar.gz" -o -name "*.sha256sum" \) | while IFS= read -r file; do
-    s5cmd cp \
+    aws s3 cp \
      "$file" "s3://$AWS_S3_BUCKET/$RELEASE_DIRS/$VERSION/$(basename "$file")"
  done
 }
@@ -45,7 +45,7 @@ function update_version_info() {
    if [[ "$VERSION" =~ ^v[0-9]+\.[0-9]+\.[0-9]+$ ]]; then
      echo "Updating latest-version.txt"
      echo "$VERSION" > latest-version.txt
-      s5cmd cp \
+      aws s3 cp \
        latest-version.txt "s3://$AWS_S3_BUCKET/$RELEASE_DIRS/latest-version.txt"
    fi

@@ -53,7 +53,7 @@ function update_version_info() {
    if [[ "$VERSION" == *"nightly"* ]]; then
      echo "Updating latest-nightly-version.txt"
      echo "$VERSION" > latest-nightly-version.txt
-      s5cmd cp \
+      aws s3 cp \
        latest-nightly-version.txt "s3://$AWS_S3_BUCKET/$RELEASE_DIRS/latest-nightly-version.txt"
    fi
  fi
--- a/.github/workflows/apidoc.yml
+++ b/.github/workflows/apidoc.yml
@@ -12,17 +12,20 @@ on:

 name: Build API docs

+env:
+  RUST_TOOLCHAIN: nightly-2024-04-20
+
 jobs:
  apidoc:
-    runs-on: ubuntu-latest
+    runs-on: ubuntu-20.04
    steps:
    - uses: actions/checkout@v4
-      with:
-        persist-credentials: false
    - uses: arduino/setup-protoc@v3
      with:
        repo-token: ${{ secrets.GITHUB_TOKEN }}
-    - uses: actions-rust-lang/setup-rust-toolchain@v1
+    - uses: dtolnay/rust-toolchain@master
+      with:
+        toolchain: ${{ env.RUST_TOOLCHAIN }}
    - run: cargo doc --workspace --no-deps --document-private-items
    - run: |
        cat <<EOF > target/doc/index.html
--- a/.github/workflows/dependency-check.yml
+++ b/.github/workflows/dependency-check.yml
@@ -1,35 +0,0 @@
-name: Check Dependencies
-
-on:
-  pull_request:
-    branches:
-      - main
-
-jobs:
-  check-dependencies:
-    runs-on: ubuntu-latest
-
-    steps:
-    - name: Checkout code
-      uses: actions/checkout@v4
-      with:
-        persist-credentials: false
-
-    - name: Set up Rust
-      uses: actions-rust-lang/setup-rust-toolchain@v1
-
-    - name: Run cargo tree
-      run: cargo tree --prefix none > dependencies.txt
-
-    - name: Extract dependency names
-      run: awk '{print $1}' dependencies.txt > dependency_names.txt
-
-    - name: Check for blacklisted crates
-      run: |
-        while read -r dep; do
-          if grep -qFx "$dep" dependency_names.txt; then
-            echo "Blacklisted crate '$dep' found in dependencies."
-            exit 1
-          fi
-        done < .github/cargo-blacklist.txt
-        echo "No blacklisted crates found."
--- a/.github/workflows/dev-build.yml
+++ b/.github/workflows/dev-build.yml
@@ -16,11 +16,11 @@ on:
        description: The runner uses to build linux-amd64 artifacts
        default: ec2-c6i.4xlarge-amd64
        options:
-          - ubuntu-22.04
-          - ubuntu-22.04-8-cores
-          - ubuntu-22.04-16-cores
-          - ubuntu-22.04-32-cores
-          - ubuntu-22.04-64-cores
+          - ubuntu-20.04
+          - ubuntu-20.04-8-cores
+          - ubuntu-20.04-16-cores
+          - ubuntu-20.04-32-cores
+          - ubuntu-20.04-64-cores
          - ec2-c6i.xlarge-amd64 # 4C8G
          - ec2-c6i.2xlarge-amd64 # 8C16G
          - ec2-c6i.4xlarge-amd64 # 16C32G
@@ -76,14 +76,20 @@ env:

  NIGHTLY_RELEASE_PREFIX: nightly

+  # Use the different image name to avoid conflict with the release images.
+  IMAGE_NAME: greptimedb-dev
+
  # The source code will check out in the following path: '${WORKING_DIR}/dev/greptime'.
  CHECKOUT_GREPTIMEDB_PATH: dev/greptimedb

+permissions:
+  issues: write
+
 jobs:
  allocate-runners:
    name: Allocate runners
    if: ${{ github.repository == 'GreptimeTeam/greptimedb' }}
-    runs-on: ubuntu-latest
+    runs-on: ubuntu-20.04
    outputs:
      linux-amd64-runner: ${{ steps.start-linux-amd64-runner.outputs.label }}
      linux-arm64-runner: ${{ steps.start-linux-arm64-runner.outputs.label }}
@@ -101,7 +107,6 @@ jobs:
        uses: actions/checkout@v4
        with:
          fetch-depth: 0
-          persist-credentials: false

      - name: Create version
        id: create-version
@@ -156,7 +161,6 @@ jobs:
        uses: actions/checkout@v4
        with:
          fetch-depth: 0
-          persist-credentials: false

      - name: Checkout greptimedb
        uses: actions/checkout@v4
@@ -164,7 +168,6 @@ jobs:
          repository: ${{ inputs.repository }}
          ref: ${{ inputs.commit }}
          path: ${{ env.CHECKOUT_GREPTIMEDB_PATH }}
-          persist-credentials: true

      - uses: ./.github/actions/build-linux-artifacts
        with:
@@ -174,8 +177,6 @@ jobs:
          disable-run-tests: ${{ env.DISABLE_RUN_TESTS }}
          dev-mode: true # Only build the standard greptime binary.
          working-dir: ${{ env.CHECKOUT_GREPTIMEDB_PATH }}
-          image-registry: ${{ vars.ECR_IMAGE_REGISTRY }}
-          image-namespace: ${{ vars.ECR_IMAGE_NAMESPACE }}

  build-linux-arm64-artifacts:
    name: Build linux-arm64 artifacts
@@ -189,7 +190,6 @@ jobs:
        uses: actions/checkout@v4
        with:
          fetch-depth: 0
-          persist-credentials: false

      - name: Checkout greptimedb
        uses: actions/checkout@v4
@@ -197,7 +197,6 @@ jobs:
          repository: ${{ inputs.repository }}
          ref: ${{ inputs.commit }}
          path: ${{ env.CHECKOUT_GREPTIMEDB_PATH }}
-          persist-credentials: true

      - uses: ./.github/actions/build-linux-artifacts
        with:
@@ -207,8 +206,6 @@ jobs:
          disable-run-tests: ${{ env.DISABLE_RUN_TESTS }}
          dev-mode: true # Only build the standard greptime binary.
          working-dir: ${{ env.CHECKOUT_GREPTIMEDB_PATH }}
-          image-registry: ${{ vars.ECR_IMAGE_REGISTRY }}
-          image-namespace: ${{ vars.ECR_IMAGE_NAMESPACE }}

  release-images-to-dockerhub:
    name: Build and push images to DockerHub
@@ -218,33 +215,25 @@ jobs:
      build-linux-amd64-artifacts,
      build-linux-arm64-artifacts,
    ]
-    runs-on: ubuntu-latest
+    runs-on: ubuntu-20.04
    outputs:
      build-result: ${{ steps.set-build-result.outputs.build-result }}
    steps:
      - uses: actions/checkout@v4
        with:
          fetch-depth: 0
-          persist-credentials: false

      - name: Build and push images to dockerhub
        uses: ./.github/actions/build-images
        with:
          image-registry: docker.io
          image-namespace: ${{ vars.IMAGE_NAMESPACE }}
-          image-name: ${{ vars.DEV_BUILD_IMAGE_NAME }}
+          image-name: ${{ env.IMAGE_NAME }}
          image-registry-username: ${{ secrets.DOCKERHUB_USERNAME }}
          image-registry-password: ${{ secrets.DOCKERHUB_TOKEN }}
          version: ${{ needs.allocate-runners.outputs.version }}
          push-latest-tag: false # Don't push the latest tag to registry.
          dev-mode: true # Only build the standard images.
-          
-      - name: Echo Docker image tag to step summary
-        run: |
-          echo "## Docker Image Tag" >> $GITHUB_STEP_SUMMARY
-          echo "Image Tag: \`${{ needs.allocate-runners.outputs.version }}\`" >> $GITHUB_STEP_SUMMARY
-          echo "Full Image Name: \`docker.io/${{ vars.IMAGE_NAMESPACE }}/${{ vars.DEV_BUILD_IMAGE_NAME }}:${{ needs.allocate-runners.outputs.version }}\`" >> $GITHUB_STEP_SUMMARY
-          echo "Pull Command: \`docker pull docker.io/${{ vars.IMAGE_NAMESPACE }}/${{ vars.DEV_BUILD_IMAGE_NAME }}:${{ needs.allocate-runners.outputs.version }}\`" >> $GITHUB_STEP_SUMMARY

      - name: Set build result
        id: set-build-result
@@ -258,20 +247,19 @@ jobs:
      allocate-runners,
      release-images-to-dockerhub,
    ]
-    runs-on: ubuntu-latest
+    runs-on: ubuntu-20.04
    continue-on-error: true
    steps:
      - uses: actions/checkout@v4
        with:
          fetch-depth: 0
-          persist-credentials: false

      - name: Release artifacts to CN region
        uses: ./.github/actions/release-cn-artifacts
        with:
          src-image-registry: docker.io
          src-image-namespace: ${{ vars.IMAGE_NAMESPACE }}
-          src-image-name: ${{ vars.DEV_BUILD_IMAGE_NAME }}
+          src-image-name: ${{ env.IMAGE_NAME }}
          dst-image-registry-username: ${{ secrets.ALICLOUD_USERNAME }}
          dst-image-registry-password: ${{ secrets.ALICLOUD_PASSWORD }}
          dst-image-registry: ${{ vars.ACR_IMAGE_REGISTRY }}
@@ -281,7 +269,6 @@ jobs:
          aws-cn-access-key-id: ${{ secrets.AWS_CN_ACCESS_KEY_ID }}
          aws-cn-secret-access-key: ${{ secrets.AWS_CN_SECRET_ACCESS_KEY }}
          aws-cn-region: ${{ vars.AWS_RELEASE_BUCKET_REGION }}
-          upload-to-s3: false
          dev-mode: true                     # Only build the standard images(exclude centos images).
          push-latest-tag: false             # Don't push the latest tag to registry.
          update-version-info: false         # Don't update the version info in S3.
@@ -290,7 +277,7 @@ jobs:
    name: Stop linux-amd64 runner
    # Only run this job when the runner is allocated.
    if: ${{ always() }}
-    runs-on: ubuntu-latest
+    runs-on: ubuntu-20.04
    needs: [
      allocate-runners,
      build-linux-amd64-artifacts,
@@ -300,7 +287,6 @@ jobs:
        uses: actions/checkout@v4
        with:
          fetch-depth: 0
-          persist-credentials: false

      - name: Stop EC2 runner
        uses: ./.github/actions/stop-runner
@@ -316,7 +302,7 @@ jobs:
    name: Stop linux-arm64 runner
    # Only run this job when the runner is allocated.
    if: ${{ always() }}
-    runs-on: ubuntu-latest
+    runs-on: ubuntu-20.04
    needs: [
      allocate-runners,
      build-linux-arm64-artifacts,
@@ -326,7 +312,6 @@ jobs:
        uses: actions/checkout@v4
        with:
          fetch-depth: 0
-          persist-credentials: false

      - name: Stop EC2 runner
        uses: ./.github/actions/stop-runner
@@ -344,17 +329,11 @@ jobs:
    needs: [
      release-images-to-dockerhub
    ]
-    runs-on: ubuntu-latest
-    permissions:
-      issues: write
-
+    runs-on: ubuntu-20.04
    env:
      SLACK_WEBHOOK_URL: ${{ secrets.SLACK_WEBHOOK_URL_DEVELOP_CHANNEL }}
    steps:
      - uses: actions/checkout@v4
-        with:
-          fetch-depth: 0
-          persist-credentials: false
      - uses: ./.github/actions/setup-cyborg
      - name: Report CI status
        id: report-ci-status
--- a/.github/workflows/develop.yml
+++ b/.github/workflows/develop.yml
@@ -1,6 +1,4 @@
 on:
-  schedule:
-    - cron: "0 15 * * 1-5"
  merge_group:
  pull_request:
    types: [ opened, synchronize, reopened, ready_for_review ]
@@ -12,6 +10,17 @@ on:
      - 'docker/**'
      - '.gitignore'
      - 'grafana/**'
+  push:
+    branches:
+      - main
+    paths-ignore:
+      - 'docs/**'
+      - 'config/**'
+      - '**.md'
+      - '.dockerignore'
+      - 'docker/**'
+      - '.gitignore'
+      - 'grafana/**'
  workflow_dispatch:

 name: CI
@@ -20,14 +29,15 @@ concurrency:
  group: ${{ github.workflow }}-${{ github.head_ref || github.run_id }}
  cancel-in-progress: true

+env:
+  RUST_TOOLCHAIN: nightly-2024-04-20
+
 jobs:
  check-typos-and-docs:
    name: Check typos and docs
-    runs-on: ubuntu-latest
+    runs-on: ubuntu-20.04
    steps:
      - uses: actions/checkout@v4
-        with:
-          persist-credentials: false
      - uses: crate-ci/typos@master
      - name: Check the config docs
        run: |
@@ -36,12 +46,10 @@ jobs:
          || (echo "'config/config.md' is not up-to-date, please run 'make config-docs'." && exit 1)

  license-header-check:
-    runs-on: ubuntu-latest
+    runs-on: ubuntu-20.04
    name: Check License Header
    steps:
      - uses: actions/checkout@v4
-        with:
-          persist-credentials: false
      - uses: korandoru/hawkeye@v5

  check:
@@ -49,38 +57,41 @@ jobs:
    runs-on: ${{ matrix.os }}
    strategy:
      matrix:
-        os: [ ubuntu-latest ]
+        os: [ windows-2022, ubuntu-20.04 ]
    timeout-minutes: 60
    steps:
      - uses: actions/checkout@v4
-        with:
-          persist-credentials: false
      - uses: arduino/setup-protoc@v3
        with:
          repo-token: ${{ secrets.GITHUB_TOKEN }}
-      - uses: actions-rust-lang/setup-rust-toolchain@v1
+      - uses: dtolnay/rust-toolchain@master
+        with:
+          toolchain: ${{ env.RUST_TOOLCHAIN }}
      - name: Rust Cache
        uses: Swatinem/rust-cache@v2
        with:
          # Shares across multiple jobs
          # Shares with `Clippy` job
          shared-key: "check-lint"
-          cache-all-crates: "true"
-          save-if: ${{ github.ref == 'refs/heads/main' }}
      - name: Run cargo check
        run: cargo check --locked --workspace --all-targets

  toml:
    name: Toml Check
-    runs-on: ubuntu-latest
+    runs-on: ubuntu-20.04
    timeout-minutes: 60
    steps:
      - uses: actions/checkout@v4
+      - uses: dtolnay/rust-toolchain@master
        with:
-          persist-credentials: false
-      - uses: actions-rust-lang/setup-rust-toolchain@v1
+          toolchain: stable
+      - name: Rust Cache
+        uses: Swatinem/rust-cache@v2
+        with:
+          # Shares across multiple jobs
+          shared-key: "check-toml"
      - name: Install taplo
-        run: cargo +stable install taplo-cli --version ^0.9 --locked --force
+        run: cargo +stable install taplo-cli --version ^0.9 --locked
      - name: Run taplo
        run: taplo format --check

@@ -89,29 +100,27 @@ jobs:
    runs-on: ${{ matrix.os }}
    strategy:
      matrix:
-        os: [ ubuntu-latest ]
+        os: [ ubuntu-20.04 ]
    timeout-minutes: 60
    steps:
      - uses: actions/checkout@v4
-        with:
-          persist-credentials: false
      - uses: arduino/setup-protoc@v3
        with:
          repo-token: ${{ secrets.GITHUB_TOKEN }}
-      - uses: actions-rust-lang/setup-rust-toolchain@v1
+      - uses: dtolnay/rust-toolchain@master
+        with:
+          toolchain: ${{ env.RUST_TOOLCHAIN }}
      - uses: Swatinem/rust-cache@v2
        with:
          # Shares across multiple jobs
          shared-key: "build-binaries"
-          cache-all-crates: "true"
-          save-if: ${{ github.ref == 'refs/heads/main' }}
      - name: Install cargo-gc-bin
        shell: bash
-        run: cargo install cargo-gc-bin --force
+        run: cargo install cargo-gc-bin
      - name: Build greptime binaries
        shell: bash
        # `cargo gc` will invoke `cargo build` with specified args
-        run: cargo gc -- --bin greptime --bin sqlness-runner --features "pg_kvbackend,mysql_kvbackend"
+        run: cargo gc -- --bin greptime --bin sqlness-runner
      - name: Pack greptime binaries
        shell: bash
        run: |
@@ -130,46 +139,35 @@ jobs:
    name: Fuzz Test
    needs: build
    runs-on: ubuntu-latest
-    timeout-minutes: 60
    strategy:
-      fail-fast: false
      matrix:
        target: [ "fuzz_create_table", "fuzz_alter_table", "fuzz_create_database", "fuzz_create_logical_table", "fuzz_alter_logical_table", "fuzz_insert", "fuzz_insert_logical_table" ]
    steps:
-      - name: Remove unused software
-        run: |
-          echo "Disk space before:"
-          df -h
-          [[ -d /usr/share/dotnet ]] && sudo rm -rf /usr/share/dotnet
-          [[ -d /usr/local/lib/android ]] && sudo rm -rf /usr/local/lib/android
-          [[ -d /opt/ghc ]] && sudo rm -rf /opt/ghc
-          [[ -d /opt/hostedtoolcache/CodeQL ]] && sudo rm -rf /opt/hostedtoolcache/CodeQL
-          sudo docker image prune --all --force
-          sudo docker builder prune -a
-          echo "Disk space after:"
-          df -h
      - uses: actions/checkout@v4
-        with:
-          persist-credentials: false
      - uses: arduino/setup-protoc@v3
        with:
          repo-token: ${{ secrets.GITHUB_TOKEN }}
-      - uses: actions-rust-lang/setup-rust-toolchain@v1
+      - uses: dtolnay/rust-toolchain@master
+        with:
+          toolchain: ${{ env.RUST_TOOLCHAIN }}
+      - name: Rust Cache
+        uses: Swatinem/rust-cache@v2
+        with:
+          # Shares across multiple jobs
+          shared-key: "fuzz-test-targets"
      - name: Set Rust Fuzz
        shell: bash
        run: |
          sudo apt-get install -y libfuzzer-14-dev
          rustup install nightly
-          cargo +nightly install cargo-fuzz cargo-gc-bin --force
+          cargo +nightly install cargo-fuzz
      - name: Download pre-built binaries
        uses: actions/download-artifact@v4
        with:
          name: bins
          path: .
      - name: Unzip binaries
-        run: |
-          tar -xvf ./bins.tar.gz
-          rm ./bins.tar.gz
+        run: tar -xvf ./bins.tar.gz
      - name: Run GreptimeDB
        run: |
          ./bins/greptime standalone start&
@@ -184,52 +182,49 @@ jobs:

  unstable-fuzztest:
    name: Unstable Fuzz Test
-    needs: build-greptime-ci
+    needs: build
    runs-on: ubuntu-latest
-    timeout-minutes: 60
    strategy:
      matrix:
        target: [ "unstable_fuzz_create_table_standalone" ]
    steps:
-      - name: Remove unused software
-        run: |
-          echo "Disk space before:"
-          df -h
-          [[ -d /usr/share/dotnet ]] && sudo rm -rf /usr/share/dotnet
-          [[ -d /usr/local/lib/android ]] && sudo rm -rf /usr/local/lib/android
-          [[ -d /opt/ghc ]] && sudo rm -rf /opt/ghc
-          [[ -d /opt/hostedtoolcache/CodeQL ]] && sudo rm -rf /opt/hostedtoolcache/CodeQL
-          sudo docker image prune --all --force
-          sudo docker builder prune -a
-          echo "Disk space after:"
-          df -h
      - uses: actions/checkout@v4
-        with:
-          persist-credentials: false
      - uses: arduino/setup-protoc@v3
        with:
          repo-token: ${{ secrets.GITHUB_TOKEN }}
-      - uses: actions-rust-lang/setup-rust-toolchain@v1
+      - uses: dtolnay/rust-toolchain@master
+        with:
+          toolchain: ${{ env.RUST_TOOLCHAIN }}
+      - name: Rust Cache
+        uses: Swatinem/rust-cache@v2
+        with:
+          # Shares across multiple jobs
+          shared-key: "fuzz-test-targets"
      - name: Set Rust Fuzz
        shell: bash
        run: |
          sudo apt update && sudo apt install -y libfuzzer-14-dev
-          cargo install cargo-fuzz cargo-gc-bin --force
-      - name: Download pre-built binariy
+          cargo install cargo-fuzz
+      - name: Download pre-built binaries
        uses: actions/download-artifact@v4
        with:
-          name: bin
+          name: bins
          path: .
-      - name: Unzip bianry
+      - name: Unzip binaries
+        run: tar -xvf ./bins.tar.gz
+      - name: Build Fuzz Test
+        shell: bash
        run: |
-          tar -xvf ./bin.tar.gz
-          rm ./bin.tar.gz
+          cd tests-fuzz &
+          cargo install cargo-gc-bin &
+          cargo gc &
+          cd ..
      - name: Run Fuzz Test
        uses: ./.github/actions/fuzz-test
        env:
          CUSTOM_LIBFUZZER_PATH: /usr/lib/llvm-14/lib/libFuzzer.a
          GT_MYSQL_ADDR: 127.0.0.1:4002
-          GT_FUZZ_BINARY_PATH: ./bin/greptime
+          GT_FUZZ_BINARY_PATH: ./bins/greptime
          GT_FUZZ_INSTANCE_ROOT_DIR: /tmp/unstable-greptime/
        with:
          target: ${{ matrix.target }}
@@ -248,29 +243,27 @@ jobs:
    runs-on: ${{ matrix.os }}
    strategy:
      matrix:
-        os: [ ubuntu-latest ]
+        os: [ ubuntu-20.04 ]
    timeout-minutes: 60
    steps:
      - uses: actions/checkout@v4
-        with:
-          persist-credentials: false
      - uses: arduino/setup-protoc@v3
        with:
          repo-token: ${{ secrets.GITHUB_TOKEN }}
-      - uses: actions-rust-lang/setup-rust-toolchain@v1
+      - uses: dtolnay/rust-toolchain@master
+        with:
+          toolchain: ${{ env.RUST_TOOLCHAIN }}
      - uses: Swatinem/rust-cache@v2
        with:
          # Shares across multiple jobs
          shared-key: "build-greptime-ci"
-          cache-all-crates: "true"
-          save-if: ${{ github.ref == 'refs/heads/main' }}
      - name: Install cargo-gc-bin
        shell: bash
-        run: cargo install cargo-gc-bin --force
+        run: cargo install cargo-gc-bin
      - name: Build greptime bianry
        shell: bash
        # `cargo gc` will invoke `cargo build` with specified args
-        run: cargo gc --profile ci -- --bin greptime --features "pg_kvbackend,mysql_kvbackend"
+        run: cargo build --bin greptime --profile ci
      - name: Pack greptime binary
        shell: bash
        run: |
@@ -284,56 +277,37 @@ jobs:
          artifacts-dir: bin
          version: current

-  distributed-fuzztest:
-    name: Fuzz Test (Distributed, ${{ matrix.mode.name }}, ${{ matrix.target }})
+  distributed-fuzztest: 
+    name: Fuzz Test (Distributed, Disk)
    runs-on: ubuntu-latest
    needs:  build-greptime-ci
-    timeout-minutes: 60
    strategy:
      matrix:
        target: [ "fuzz_create_table", "fuzz_alter_table", "fuzz_create_database", "fuzz_create_logical_table", "fuzz_alter_logical_table", "fuzz_insert", "fuzz_insert_logical_table" ]
-        mode:
-          - name: "Remote WAL"
-            minio: true
-            kafka: true
-            values: "with-remote-wal.yaml"
    steps:
-      - name: Remove unused software
-        run: |
-          echo "Disk space before:"
-          df -h
-          [[ -d /usr/share/dotnet ]] && sudo rm -rf /usr/share/dotnet
-          [[ -d /usr/local/lib/android ]] && sudo rm -rf /usr/local/lib/android
-          [[ -d /opt/ghc ]] && sudo rm -rf /opt/ghc
-          [[ -d /opt/hostedtoolcache/CodeQL ]] && sudo rm -rf /opt/hostedtoolcache/CodeQL
-          sudo docker image prune --all --force
-          sudo docker builder prune -a
-          echo "Disk space after:"
-          df -h
      - uses: actions/checkout@v4
-        with:
-          persist-credentials: false
      - name: Setup Kind
        uses: ./.github/actions/setup-kind
-      - if: matrix.mode.minio
-        name: Setup Minio
-        uses: ./.github/actions/setup-minio
-      - if: matrix.mode.kafka
-        name: Setup Kafka cluser
-        uses: ./.github/actions/setup-kafka-cluster
      - name: Setup Etcd cluser
        uses: ./.github/actions/setup-etcd-cluster
      # Prepares for fuzz tests
      - uses: arduino/setup-protoc@v3
        with:
          repo-token: ${{ secrets.GITHUB_TOKEN }}
-      - uses: actions-rust-lang/setup-rust-toolchain@v1
+      - uses: dtolnay/rust-toolchain@master
+        with:
+          toolchain: ${{ env.RUST_TOOLCHAIN }}
+      - name: Rust Cache
+        uses: Swatinem/rust-cache@v2
+        with:
+          # Shares across multiple jobs
+          shared-key: "fuzz-test-targets"
      - name: Set Rust Fuzz
        shell: bash
        run: |
          sudo apt-get install -y libfuzzer-14-dev
          rustup install nightly
-          cargo +nightly install cargo-fuzz cargo-gc-bin --force
+          cargo +nightly install cargo-fuzz
      # Downloads ci image
      - name: Download pre-built binariy
        uses: actions/download-artifact@v4
@@ -341,9 +315,7 @@ jobs:
          name: bin
          path: .
      - name: Unzip binary
-        run: |
-          tar -xvf ./bin.tar.gz
-          rm ./bin.tar.gz
+        run: tar -xvf ./bin.tar.gz
      - name: Build and push GreptimeDB image
        uses: ./.github/actions/build-and-push-ci-image
      - name: Wait for etcd
@@ -353,22 +325,6 @@ jobs:
            pod -l app.kubernetes.io/instance=etcd \
            --timeout=120s \
            -n etcd-cluster
-      - if: matrix.mode.minio
-        name: Wait for minio
-        run: |
-          kubectl wait \
-            --for=condition=Ready \
-            pod -l app=minio \
-            --timeout=120s \
-            -n minio
-      - if: matrix.mode.kafka
-        name: Wait for kafka
-        run: |
-          kubectl wait \
-            --for=condition=Ready \
-            pod -l app.kubernetes.io/instance=kafka \
-            --timeout=120s \
-            -n kafka-cluster
      - name: Print etcd info
        shell: bash
        run: kubectl get all --show-labels -n etcd-cluster
@@ -377,7 +333,6 @@ jobs:
        uses: ./.github/actions/setup-greptimedb-cluster
        with:
          image-registry: localhost:5001
-          values-filename: ${{ matrix.mode.values }}
      - name: Port forward (mysql)
        run: |
          kubectl port-forward service/my-greptimedb-frontend 4002:4002 -n my-greptimedb&
@@ -392,202 +347,32 @@ jobs:
      - name: Describe Nodes
        if: failure()
        shell: bash
-        run: |
-          kubectl describe nodes
+        run: | 
+          kubectl describe nodes      
      - name: Export kind logs
        if: failure()
        shell: bash
-        run: |
+        run: | 
          kind export logs /tmp/kind
      - name: Upload logs
        if: failure()
        uses: actions/upload-artifact@v4
        with:
-          name: fuzz-tests-kind-logs-${{ matrix.mode.name }}-${{ matrix.target }}
+          name: fuzz-tests-kind-logs-${{ matrix.target }}
          path: /tmp/kind
          retention-days: 3
-      - name: Delete cluster
-        if: success()
-        shell: bash
-        run: |
-          kind delete cluster
-          docker stop $(docker ps -a -q)
-          docker rm $(docker ps -a -q)
-          docker system prune -f
-
-  distributed-fuzztest-with-chaos:
-    name: Fuzz Test with Chaos (Distributed, ${{ matrix.mode.name }}, ${{ matrix.target }})
-    runs-on: ubuntu-latest
-    needs:  build-greptime-ci
-    timeout-minutes: 60
-    strategy:
-      matrix:
-        target: ["fuzz_migrate_mito_regions", "fuzz_migrate_metric_regions", "fuzz_failover_mito_regions", "fuzz_failover_metric_regions"]
-        mode:
-          - name: "Remote WAL"
-            minio: true
-            kafka: true
-            values: "with-remote-wal.yaml"
-        include:
-          - target: "fuzz_migrate_mito_regions"
-            mode:
-              name: "Local WAL"
-              minio: true
-              kafka: false
-              values: "with-minio.yaml"
-          - target: "fuzz_migrate_metric_regions"
-            mode:
-              name: "Local WAL"
-              minio: true
-              kafka: false
-              values: "with-minio.yaml"
-    steps:
-      - name: Remove unused software
-        run: |
-          echo "Disk space before:"
-          df -h
-          [[ -d /usr/share/dotnet ]] && sudo rm -rf /usr/share/dotnet
-          [[ -d /usr/local/lib/android ]] && sudo rm -rf /usr/local/lib/android
-          [[ -d /opt/ghc ]] && sudo rm -rf /opt/ghc
-          [[ -d /opt/hostedtoolcache/CodeQL ]] && sudo rm -rf /opt/hostedtoolcache/CodeQL
-          sudo docker image prune --all --force
-          sudo docker builder prune -a
-          echo "Disk space after:"
-          df -h
-      - uses: actions/checkout@v4
-        with:
-          persist-credentials: false
-      - name: Setup Kind
-        uses: ./.github/actions/setup-kind
-      - name: Setup Chaos Mesh
-        uses: ./.github/actions/setup-chaos
-      - if: matrix.mode.minio
-        name: Setup Minio
-        uses: ./.github/actions/setup-minio
-      - if: matrix.mode.kafka
-        name: Setup Kafka cluser
-        uses: ./.github/actions/setup-kafka-cluster
-      - name: Setup Etcd cluser
-        uses: ./.github/actions/setup-etcd-cluster
-      # Prepares for fuzz tests
-      - uses: arduino/setup-protoc@v3
-        with:
-          repo-token: ${{ secrets.GITHUB_TOKEN }}
-      - uses: actions-rust-lang/setup-rust-toolchain@v1
-      - name: Set Rust Fuzz
-        shell: bash
-        run: |
-          sudo apt-get install -y libfuzzer-14-dev
-          rustup install nightly
-          cargo +nightly install cargo-fuzz cargo-gc-bin --force
-      # Downloads ci image
-      - name: Download pre-built binariy
-        uses: actions/download-artifact@v4
-        with:
-          name: bin
-          path: .
-      - name: Unzip binary
-        run: |
-          tar -xvf ./bin.tar.gz
-          rm ./bin.tar.gz
-      - name: Build and push GreptimeDB image
-        uses: ./.github/actions/build-and-push-ci-image
-      - name: Wait for etcd
-        run: |
-          kubectl wait \
-            --for=condition=Ready \
-            pod -l app.kubernetes.io/instance=etcd \
-            --timeout=120s \
-            -n etcd-cluster
-      - if: matrix.mode.minio
-        name: Wait for minio
-        run: |
-          kubectl wait \
-            --for=condition=Ready \
-            pod -l app=minio \
-            --timeout=120s \
-            -n minio
-      - if: matrix.mode.kafka
-        name: Wait for kafka
-        run: |
-          kubectl wait \
-            --for=condition=Ready \
-            pod -l app.kubernetes.io/instance=kafka \
-            --timeout=120s \
-            -n kafka-cluster
-      - name: Print etcd info
-        shell: bash
-        run: kubectl get all --show-labels -n etcd-cluster
-      # Setup cluster for test
-      - name: Setup GreptimeDB cluster
-        uses: ./.github/actions/setup-greptimedb-cluster
-        with:
-          image-registry: localhost:5001
-          values-filename: ${{ matrix.mode.values }}
-          enable-region-failover: ${{ matrix.mode.kafka }}
-      - name: Port forward (mysql)
-        run: |
-          kubectl port-forward service/my-greptimedb-frontend 4002:4002 -n my-greptimedb&
-      - name: Fuzz Test
-        uses: ./.github/actions/fuzz-test
-        env:
-          CUSTOM_LIBFUZZER_PATH: /usr/lib/llvm-14/lib/libFuzzer.a
-          GT_MYSQL_ADDR: 127.0.0.1:4002
-        with:
-          target: ${{ matrix.target }}
-          max-total-time: 120
-      - name: Describe Nodes
-        if: failure()
-        shell: bash
-        run: |
-          kubectl describe nodes
-      - name: Export kind logs
-        if: failure()
-        shell: bash
-        run: |
-          kind export logs /tmp/kind
-      - name: Upload logs
-        if: failure()
-        uses: actions/upload-artifact@v4
-        with:
-          name: fuzz-tests-kind-logs-${{ matrix.mode.name }}-${{ matrix.target }}
-          path: /tmp/kind
-          retention-days: 3
-      - name: Delete cluster
-        if: success()
-        shell: bash
-        run: |
-          kind delete cluster
-          docker stop $(docker ps -a -q)
-          docker rm $(docker ps -a -q)
-          docker system prune -f
+        

  sqlness:
-    name: Sqlness Test (${{ matrix.mode.name }})
+    name: Sqlness Test
    needs: build
    runs-on: ${{ matrix.os }}
    strategy:
      matrix:
-        os: [ ubuntu-latest ]
-        mode:
-          - name: "Basic"
-            opts: ""
-            kafka: false
-          - name: "Remote WAL"
-            opts: "-w kafka -k 127.0.0.1:9092"
-            kafka: true
-          - name: "Pg Kvbackend"
-            opts: "--setup-pg"
-            kafka: false
+        os: [ ubuntu-20.04 ]
    timeout-minutes: 60
    steps:
      - uses: actions/checkout@v4
-        with:
-          persist-credentials: false
-      - if: matrix.mode.kafka
-        name: Setup kafka server
-        working-directory: tests-integration/fixtures
-        run: docker compose up -d --wait kafka
      - name: Download pre-built binaries
        uses: actions/download-artifact@v4
        with:
@@ -596,45 +381,78 @@ jobs:
      - name: Unzip binaries
        run: tar -xvf ./bins.tar.gz
      - name: Run sqlness
-        run: RUST_BACKTRACE=1 ./bins/sqlness-runner ${{ matrix.mode.opts }} -c ./tests/cases --bins-dir ./bins --preserve-state
+        run: RUST_BACKTRACE=1 ./bins/sqlness-runner -c ./tests/cases --bins-dir ./bins --preserve-state
      - name: Upload sqlness logs
-        if: failure()
+        if: always()
        uses: actions/upload-artifact@v4
        with:
-          name: sqlness-logs-${{ matrix.mode.name }}
+          name: sqlness-logs
+          path: /tmp/sqlness*
+          retention-days: 3
+
+  sqlness-kafka-wal:
+    name: Sqlness Test with Kafka Wal
+    needs: build
+    runs-on: ${{ matrix.os }}
+    strategy:
+      matrix:
+        os: [ ubuntu-20.04 ]
+    timeout-minutes: 60
+    steps:
+      - uses: actions/checkout@v4
+      - name: Download pre-built binaries
+        uses: actions/download-artifact@v4
+        with:
+          name: bins
+          path: .
+      - name: Unzip binaries
+        run: tar -xvf ./bins.tar.gz
+      - name: Setup kafka server
+        working-directory: tests-integration/fixtures/kafka
+        run: docker compose -f docker-compose-standalone.yml up -d --wait
+      - name: Run sqlness
+        run: RUST_BACKTRACE=1 ./bins/sqlness-runner -w kafka -k 127.0.0.1:9092 -c ./tests/cases --bins-dir ./bins --preserve-state
+      - name: Upload sqlness logs
+        if: always()
+        uses: actions/upload-artifact@v4
+        with:
+          name: sqlness-logs-with-kafka-wal
          path: /tmp/sqlness*
          retention-days: 3

  fmt:
    name: Rustfmt
-    runs-on: ubuntu-latest
+    runs-on: ubuntu-20.04
    timeout-minutes: 60
    steps:
      - uses: actions/checkout@v4
-        with:
-          persist-credentials: false
      - uses: arduino/setup-protoc@v3
        with:
          repo-token: ${{ secrets.GITHUB_TOKEN }}
-      - uses: actions-rust-lang/setup-rust-toolchain@v1
+      - uses: dtolnay/rust-toolchain@master
        with:
+          toolchain: ${{ env.RUST_TOOLCHAIN }}
          components: rustfmt
-      - name: Check format
-        run: make fmt-check
+      - name: Rust Cache
+        uses: Swatinem/rust-cache@v2
+        with:
+          # Shares across multiple jobs
+          shared-key: "check-rust-fmt"
+      - name: Run cargo fmt
+        run: cargo fmt --all -- --check

  clippy:
    name: Clippy
-    runs-on: ubuntu-latest
+    runs-on: ubuntu-20.04
    timeout-minutes: 60
    steps:
      - uses: actions/checkout@v4
-        with:
-          persist-credentials: false
      - uses: arduino/setup-protoc@v3
        with:
          repo-token: ${{ secrets.GITHUB_TOKEN }}
-      - uses: actions-rust-lang/setup-rust-toolchain@v1
+      - uses: dtolnay/rust-toolchain@master
        with:
+          toolchain: ${{ env.RUST_TOOLCHAIN }}
          components: clippy
      - name: Rust Cache
        uses: Swatinem/rust-cache@v2
@@ -642,123 +460,63 @@ jobs:
          # Shares across multiple jobs
          # Shares with `Check` job
          shared-key: "check-lint"
-          cache-all-crates: "true"
-          save-if: ${{ github.ref == 'refs/heads/main' }}
      - name: Run cargo clippy
        run: make clippy

-  conflict-check:
-    name: Check for conflict
-    runs-on: ubuntu-latest
-    steps:
-      - uses: actions/checkout@v4
-        with:
-          persist-credentials: false
-      - name: Merge Conflict Finder
-        uses: olivernybroe/action-conflict-finder@v4.0
-
-  test:
-    if: github.event_name != 'merge_group'
-    runs-on: ubuntu-22.04-arm
-    timeout-minutes: 60
-    needs:  [conflict-check, clippy, fmt]
-    steps:
-      - uses: actions/checkout@v4
-        with:
-          persist-credentials: false
-      - uses: arduino/setup-protoc@v3
-        with:
-          repo-token: ${{ secrets.GITHUB_TOKEN }}
-      - uses: rui314/setup-mold@v1
-      - name: Install toolchain
-        uses: actions-rust-lang/setup-rust-toolchain@v1
-        with:
-            cache: false
-      - name: Rust Cache
-        uses: Swatinem/rust-cache@v2
-        with:
-          # Shares cross multiple jobs
-          shared-key: "coverage-test"
-          cache-all-crates: "true"
-          save-if: ${{ github.ref == 'refs/heads/main' }}
-      - name: Install latest nextest release
-        uses: taiki-e/install-action@nextest
-      - name: Setup external services
-        working-directory: tests-integration/fixtures
-        run: docker compose up -d --wait
-      - name: Run nextest cases
-        run: cargo nextest run --workspace -F dashboard -F pg_kvbackend -F mysql_kvbackend
-        env:
-          CARGO_BUILD_RUSTFLAGS: "-C link-arg=-fuse-ld=mold"
-          RUST_BACKTRACE: 1
-          RUST_MIN_STACK: 8388608 # 8MB
-          CARGO_INCREMENTAL: 0
-          GT_S3_BUCKET: ${{ vars.AWS_CI_TEST_BUCKET }}
-          GT_S3_ACCESS_KEY_ID: ${{ secrets.AWS_CI_TEST_ACCESS_KEY_ID }}
-          GT_S3_ACCESS_KEY: ${{ secrets.AWS_CI_TEST_SECRET_ACCESS_KEY }}
-          GT_S3_REGION: ${{ vars.AWS_CI_TEST_BUCKET_REGION }}
-          GT_MINIO_BUCKET: greptime
-          GT_MINIO_ACCESS_KEY_ID: superpower_ci_user
-          GT_MINIO_ACCESS_KEY: superpower_password
-          GT_MINIO_REGION: us-west-2
-          GT_MINIO_ENDPOINT_URL: http://127.0.0.1:9000
-          GT_ETCD_ENDPOINTS: http://127.0.0.1:2379
-          GT_POSTGRES_ENDPOINTS: postgres://greptimedb:admin@127.0.0.1:5432/postgres
-          GT_MYSQL_ENDPOINTS: mysql://greptimedb:admin@127.0.0.1:3306/mysql
-          GT_KAFKA_ENDPOINTS: 127.0.0.1:9092
-          GT_KAFKA_SASL_ENDPOINTS: 127.0.0.1:9093
-          UNITTEST_LOG_DIR: "__unittest_logs"
-
  coverage:
-    if: github.event_name == 'merge_group'
-    runs-on: ubuntu-22.04-8-cores
+    if: github.event.pull_request.draft == false
+    runs-on: ubuntu-20.04-8-cores
    timeout-minutes: 60
    steps:
      - uses: actions/checkout@v4
-        with:
-          persist-credentials: false
      - uses: arduino/setup-protoc@v3
        with:
          repo-token: ${{ secrets.GITHUB_TOKEN }}
-      - uses: rui314/setup-mold@v1
-      - name: Install toolchain
-        uses: actions-rust-lang/setup-rust-toolchain@v1
+      - uses: KyleMayes/install-llvm-action@v1
        with:
-          components: llvm-tools
-          cache: false
+          version: "14.0"
+      - name: Install toolchain
+        uses: dtolnay/rust-toolchain@master
+        with:
+          toolchain: ${{ env.RUST_TOOLCHAIN }}
+          components: llvm-tools-preview
      - name: Rust Cache
        uses: Swatinem/rust-cache@v2
        with:
          # Shares cross multiple jobs
          shared-key: "coverage-test"
-          save-if: ${{ github.ref == 'refs/heads/main' }}
+      - name: Docker Cache
+        uses: ScribeMD/docker-cache@0.3.7
+        with:
+          key: docker-${{ runner.os }}-coverage
      - name: Install latest nextest release
        uses: taiki-e/install-action@nextest
      - name: Install cargo-llvm-cov
        uses: taiki-e/install-action@cargo-llvm-cov
-      - name: Setup external services
-        working-directory: tests-integration/fixtures
-        run: docker compose up -d --wait
+      - name: Install Python
+        uses: actions/setup-python@v5
+        with:
+          python-version: '3.10'
+      - name: Install PyArrow Package
+        run: pip install pyarrow
+      - name: Setup etcd server
+        working-directory: tests-integration/fixtures/etcd
+        run: docker compose -f docker-compose-standalone.yml up -d --wait
+      - name: Setup kafka server
+        working-directory: tests-integration/fixtures/kafka
+        run: docker compose -f docker-compose-standalone.yml up -d --wait
      - name: Run nextest cases
-        run: cargo llvm-cov nextest --workspace --lcov --output-path lcov.info -F dashboard -F pg_kvbackend -F mysql_kvbackend
+        run: cargo llvm-cov nextest --workspace --lcov --output-path lcov.info -F pyo3_backend -F dashboard
        env:
-          CARGO_BUILD_RUSTFLAGS: "-C link-arg=-fuse-ld=mold"
+          CARGO_BUILD_RUSTFLAGS: "-C link-arg=-fuse-ld=lld"
          RUST_BACKTRACE: 1
          CARGO_INCREMENTAL: 0
          GT_S3_BUCKET: ${{ vars.AWS_CI_TEST_BUCKET }}
          GT_S3_ACCESS_KEY_ID: ${{ secrets.AWS_CI_TEST_ACCESS_KEY_ID }}
          GT_S3_ACCESS_KEY: ${{ secrets.AWS_CI_TEST_SECRET_ACCESS_KEY }}
          GT_S3_REGION: ${{ vars.AWS_CI_TEST_BUCKET_REGION }}
-          GT_MINIO_BUCKET: greptime
-          GT_MINIO_ACCESS_KEY_ID: superpower_ci_user
-          GT_MINIO_ACCESS_KEY: superpower_password
-          GT_MINIO_REGION: us-west-2
-          GT_MINIO_ENDPOINT_URL: http://127.0.0.1:9000
          GT_ETCD_ENDPOINTS: http://127.0.0.1:2379
-          GT_POSTGRES_ENDPOINTS: postgres://greptimedb:admin@127.0.0.1:5432/postgres
-          GT_MYSQL_ENDPOINTS: mysql://greptimedb:admin@127.0.0.1:3306/mysql
          GT_KAFKA_ENDPOINTS: 127.0.0.1:9092
-          GT_KAFKA_SASL_ENDPOINTS: 127.0.0.1:9093
          UNITTEST_LOG_DIR: "__unittest_logs"
      - name: Codecov upload
        uses: codecov/codecov-action@v4
@@ -772,7 +530,7 @@ jobs:
  # compat:
  #   name: Compatibility Test
  #   needs: build
-  #   runs-on: ubuntu-22.04
+  #   runs-on: ubuntu-20.04
  #   timeout-minutes: 60
  #   steps:
  #     - uses: actions/checkout@v4
--- a/.github/workflows/docbot.yml
+++ b/.github/workflows/docbot.yml
@@ -3,21 +3,16 @@ on:
  pull_request_target:
    types: [opened, edited]

-concurrency:
-  group: ${{ github.workflow }}-${{ github.head_ref || github.run_id }}
-  cancel-in-progress: true
+permissions:
+  pull-requests: write
+  contents: read

 jobs:
  docbot:
-    runs-on: ubuntu-latest
-    permissions:
-      pull-requests: write
-      contents: read
+    runs-on: ubuntu-20.04
    timeout-minutes: 10
    steps:
      - uses: actions/checkout@v4
-        with:
-          persist-credentials: false
      - uses: ./.github/actions/setup-cyborg
      - name: Maybe Follow Up Docs Issue
        working-directory: cyborg
--- a/.github/workflows/docs.yml
+++ b/.github/workflows/docs.yml
@@ -31,58 +31,55 @@ name: CI
 jobs:
  typos:
    name: Spell Check with Typos
-    runs-on: ubuntu-latest
+    runs-on: ubuntu-20.04
    steps:
      - uses: actions/checkout@v4
-        with:
-          persist-credentials: false
      - uses: crate-ci/typos@master

  license-header-check:
-    runs-on: ubuntu-latest
+    runs-on: ubuntu-20.04
    name: Check License Header
    steps:
      - uses: actions/checkout@v4
-        with:
-          persist-credentials: false
      - uses: korandoru/hawkeye@v5

  check:
    name: Check
-    runs-on: ubuntu-latest
+    runs-on: ubuntu-20.04
    steps:
      - run: 'echo "No action required"'

  fmt:
    name: Rustfmt
-    runs-on: ubuntu-latest
+    runs-on: ubuntu-20.04
    steps:
      - run: 'echo "No action required"'

  clippy:
    name: Clippy
-    runs-on: ubuntu-latest
+    runs-on: ubuntu-20.04
    steps:
      - run: 'echo "No action required"'

  coverage:
-    runs-on: ubuntu-latest
-    steps:
-      - run: 'echo "No action required"'
-
-  test:
-    runs-on: ubuntu-latest
+    runs-on: ubuntu-20.04
    steps:
      - run: 'echo "No action required"'

  sqlness:
-    name: Sqlness Test (${{ matrix.mode.name }})
+    name: Sqlness Test
    runs-on: ${{ matrix.os }}
    strategy:
      matrix:
-        os: [ ubuntu-latest ]
-        mode:
-          - name: "Basic"
-          - name: "Remote WAL"
+        os: [ ubuntu-20.04 ]
+    steps:
+      - run: 'echo "No action required"'
+
+  sqlness-kafka-wal:
+    name: Sqlness Test with Kafka Wal
+    runs-on: ${{ matrix.os }}
+    strategy:
+      matrix:
+        os: [ ubuntu-20.04 ]
    steps:
      - run: 'echo "No action required"'
--- a/.github/workflows/grafana.yml
+++ b/.github/workflows/grafana.yml
@@ -1,52 +0,0 @@
-name: Check Grafana Panels
-
-on:
-  pull_request:
-    branches:
-      - main
-    paths:
-      - 'grafana/**'  # Trigger only when files under the grafana/ directory change
-
-jobs:
-  check-panels:
-    runs-on: ubuntu-latest
-
-    steps:
-      # Check out the repository
-      - name: Checkout repository
-        uses: actions/checkout@v4
-
-      # Install jq (required for the script)
-      - name: Install jq
-        run: sudo apt-get install -y jq
-
-      # Make the check.sh script executable
-      - name: Make check.sh executable
-        run: chmod +x grafana/check.sh
-
-      # Run the check.sh script
-      - name: Run check.sh
-        run: ./grafana/check.sh
-
-      # Only run summary.sh for pull_request events (not for merge queues or final pushes)
-      - name: Check if this is a pull request
-        id: check-pr
-        run: |
-          if [[ "${{ github.event_name }}" == "pull_request" ]]; then
-            echo "is_pull_request=true" >> $GITHUB_OUTPUT
-          else
-            echo "is_pull_request=false" >> $GITHUB_OUTPUT
-          fi
-
-      # Make the summary.sh script executable
-      - name: Make summary.sh executable
-        if: steps.check-pr.outputs.is_pull_request == 'true'
-        run: chmod +x grafana/summary.sh
-
-      # Run the summary.sh script and add its output to the GitHub Job Summary
-      - name: Run summary.sh and add to Job Summary
-        if: steps.check-pr.outputs.is_pull_request == 'true'
-        run: |
-          SUMMARY=$(./grafana/summary.sh)
-          echo "### Summary of Grafana Panels" >> $GITHUB_STEP_SUMMARY
-          echo "$SUMMARY" >> $GITHUB_STEP_SUMMARY
--- a/.github/workflows/nightly-build.yml
+++ b/.github/workflows/nightly-build.yml
@@ -12,13 +12,13 @@ on:
      linux_amd64_runner:
        type: choice
        description: The runner uses to build linux-amd64 artifacts
-        default: ec2-c6i.4xlarge-amd64
+        default: ec2-c6i.2xlarge-amd64
        options:
-          - ubuntu-22.04
-          - ubuntu-22.04-8-cores
-          - ubuntu-22.04-16-cores
-          - ubuntu-22.04-32-cores
-          - ubuntu-22.04-64-cores
+          - ubuntu-20.04
+          - ubuntu-20.04-8-cores
+          - ubuntu-20.04-16-cores
+          - ubuntu-20.04-32-cores
+          - ubuntu-20.04-64-cores
          - ec2-c6i.xlarge-amd64 # 4C8G
          - ec2-c6i.2xlarge-amd64 # 8C16G
          - ec2-c6i.4xlarge-amd64 # 16C32G
@@ -27,7 +27,7 @@ on:
      linux_arm64_runner:
        type: choice
        description: The runner uses to build linux-arm64 artifacts
-        default: ec2-c6g.4xlarge-arm64
+        default: ec2-c6g.2xlarge-arm64
        options:
          - ec2-c6g.xlarge-arm64 # 4C8G
          - ec2-c6g.2xlarge-arm64 # 8C16G
@@ -66,11 +66,18 @@ env:

  NIGHTLY_RELEASE_PREFIX: nightly

+  # Use the different image name to avoid conflict with the release images.
+  # The DockerHub image will be greptime/greptimedb-nightly.
+  IMAGE_NAME: greptimedb-nightly
+
+permissions:
+  issues: write
+
 jobs:
  allocate-runners:
    name: Allocate runners
    if: ${{ github.repository == 'GreptimeTeam/greptimedb' }}
-    runs-on: ubuntu-latest
+    runs-on: ubuntu-20.04
    outputs:
      linux-amd64-runner: ${{ steps.start-linux-amd64-runner.outputs.label }}
      linux-arm64-runner: ${{ steps.start-linux-arm64-runner.outputs.label }}
@@ -88,7 +95,6 @@ jobs:
        uses: actions/checkout@v4
        with:
          fetch-depth: 0
-          persist-credentials: false

      - name: Create version
        id: create-version
@@ -141,7 +147,6 @@ jobs:
      - uses: actions/checkout@v4
        with:
          fetch-depth: 0
-          persist-credentials: false

      - uses: ./.github/actions/build-linux-artifacts
        with:
@@ -149,8 +154,6 @@ jobs:
          cargo-profile: ${{ env.CARGO_PROFILE }}
          version: ${{ needs.allocate-runners.outputs.version }}
          disable-run-tests: ${{ env.DISABLE_RUN_TESTS }}
-          image-registry: ${{ vars.ECR_IMAGE_REGISTRY }}
-          image-namespace: ${{ vars.ECR_IMAGE_NAMESPACE }}

  build-linux-arm64-artifacts:
    name: Build linux-arm64 artifacts
@@ -163,7 +166,6 @@ jobs:
      - uses: actions/checkout@v4
        with:
          fetch-depth: 0
-          persist-credentials: false

      - uses: ./.github/actions/build-linux-artifacts
        with:
@@ -171,8 +173,6 @@ jobs:
          cargo-profile: ${{ env.CARGO_PROFILE }}
          version: ${{ needs.allocate-runners.outputs.version }}
          disable-run-tests: ${{ env.DISABLE_RUN_TESTS }}
-          image-registry: ${{ vars.ECR_IMAGE_REGISTRY }}
-          image-namespace: ${{ vars.ECR_IMAGE_NAMESPACE }}

  release-images-to-dockerhub:
    name: Build and push images to DockerHub
@@ -182,25 +182,24 @@ jobs:
      build-linux-amd64-artifacts,
      build-linux-arm64-artifacts,
    ]
-    runs-on: ubuntu-latest
+    runs-on: ubuntu-20.04
    outputs:
      nightly-build-result: ${{ steps.set-nightly-build-result.outputs.nightly-build-result }}
    steps:
      - uses: actions/checkout@v4
        with:
          fetch-depth: 0
-          persist-credentials: false

      - name: Build and push images to dockerhub
        uses: ./.github/actions/build-images
        with:
          image-registry: docker.io
          image-namespace: ${{ vars.IMAGE_NAMESPACE }}
-          image-name: ${{ vars.NIGHTLY_BUILD_IMAGE_NAME }}
+          image-name: ${{ env.IMAGE_NAME }}
          image-registry-username: ${{ secrets.DOCKERHUB_USERNAME }}
          image-registry-password: ${{ secrets.DOCKERHUB_TOKEN }}
          version: ${{ needs.allocate-runners.outputs.version }}
-          push-latest-tag: false
+          push-latest-tag: false # Don't push the latest tag to registry.

      - name: Set nightly build result
        id: set-nightly-build-result
@@ -214,7 +213,7 @@ jobs:
      allocate-runners,
      release-images-to-dockerhub,
    ]
-    runs-on: ubuntu-latest
+    runs-on: ubuntu-20.04
    # When we push to ACR, it's easy to fail due to some unknown network issues.
    # However, we don't want to fail the whole workflow because of this.
    # The ACR have daily sync with DockerHub, so don't worry about the image not being updated.
@@ -223,14 +222,13 @@ jobs:
      - uses: actions/checkout@v4
        with:
          fetch-depth: 0
-          persist-credentials: false

      - name: Release artifacts to CN region
        uses: ./.github/actions/release-cn-artifacts
        with:
          src-image-registry: docker.io
          src-image-namespace: ${{ vars.IMAGE_NAMESPACE }}
-          src-image-name: ${{ vars.NIGHTLY_BUILD_IMAGE_NAME }}
+          src-image-name: ${{ env.IMAGE_NAME }}
          dst-image-registry-username: ${{ secrets.ALICLOUD_USERNAME }}
          dst-image-registry-password: ${{ secrets.ALICLOUD_PASSWORD }}
          dst-image-registry: ${{ vars.ACR_IMAGE_REGISTRY }}
@@ -240,16 +238,15 @@ jobs:
          aws-cn-access-key-id: ${{ secrets.AWS_CN_ACCESS_KEY_ID }}
          aws-cn-secret-access-key: ${{ secrets.AWS_CN_SECRET_ACCESS_KEY }}
          aws-cn-region: ${{ vars.AWS_RELEASE_BUCKET_REGION }}
-          upload-to-s3: false
          dev-mode: false
          update-version-info: false  # Don't update version info in S3.
-          push-latest-tag: false
+          push-latest-tag: false      # Don't push the latest tag to registry.

  stop-linux-amd64-runner: # It's always run as the last job in the workflow to make sure that the runner is released.
    name: Stop linux-amd64 runner
    # Only run this job when the runner is allocated.
    if: ${{ always() }}
-    runs-on: ubuntu-latest
+    runs-on: ubuntu-20.04
    needs: [
      allocate-runners,
      build-linux-amd64-artifacts,
@@ -259,7 +256,6 @@ jobs:
        uses: actions/checkout@v4
        with:
          fetch-depth: 0
-          persist-credentials: false

      - name: Stop EC2 runner
        uses: ./.github/actions/stop-runner
@@ -275,7 +271,7 @@ jobs:
    name: Stop linux-arm64 runner
    # Only run this job when the runner is allocated.
    if: ${{ always() }}
-    runs-on: ubuntu-latest
+    runs-on: ubuntu-20.04
    needs: [
      allocate-runners,
      build-linux-arm64-artifacts,
@@ -285,7 +281,6 @@ jobs:
        uses: actions/checkout@v4
        with:
          fetch-depth: 0
-          persist-credentials: false

      - name: Stop EC2 runner
        uses: ./.github/actions/stop-runner
@@ -303,15 +298,11 @@ jobs:
    needs: [
      release-images-to-dockerhub
    ]
-    runs-on: ubuntu-latest
-    permissions:
-      issues: write
+    runs-on: ubuntu-20.04
    env:
      SLACK_WEBHOOK_URL: ${{ secrets.SLACK_WEBHOOK_URL_DEVELOP_CHANNEL }}
    steps:
      - uses: actions/checkout@v4
-        with:
-          persist-credentials: false
      - uses: ./.github/actions/setup-cyborg
      - name: Report CI status
        id: report-ci-status
--- a/.github/workflows/nightly-ci.yml
+++ b/.github/workflows/nightly-ci.yml
@@ -1,6 +1,6 @@
 on:
  schedule:
-    - cron: "0 23 * * 1-4"
+    - cron: "0 23 * * 1-5"
  workflow_dispatch:

 name: Nightly CI
@@ -9,21 +9,22 @@ concurrency:
  group: ${{ github.workflow }}-${{ github.head_ref || github.run_id }}
  cancel-in-progress: true

+env:
+  RUST_TOOLCHAIN: nightly-2024-04-20
+
+permissions:
+  issues: write
+
 jobs:
  sqlness-test:
    name: Run sqlness test
    if: ${{ github.repository == 'GreptimeTeam/greptimedb' }}
-    runs-on: ubuntu-latest
+    runs-on: ubuntu-22.04
    steps:
      - name: Checkout
        uses: actions/checkout@v4
        with:
          fetch-depth: 0
-          persist-credentials: false
-
-      - name: Check install.sh
-        run: ./.github/scripts/check-install-script.sh
-
      - name: Run sqlness test
        uses: ./.github/actions/sqlness-test
        with:
@@ -32,43 +33,31 @@ jobs:
          aws-region: ${{ vars.AWS_CI_TEST_BUCKET_REGION }}
          aws-access-key-id: ${{ secrets.AWS_CI_TEST_ACCESS_KEY_ID }}
          aws-secret-access-key: ${{ secrets.AWS_CI_TEST_SECRET_ACCESS_KEY }}
-      - name: Upload sqlness logs
-        if: failure()
-        uses: actions/upload-artifact@v4
-        with:
-          name: sqlness-logs-kind
-          path: /tmp/kind/
-          retention-days: 3

  sqlness-windows:
    name: Sqlness tests on Windows
    if: ${{ github.repository == 'GreptimeTeam/greptimedb' }}
    runs-on: windows-2022-8-cores
-    permissions:
-      issues: write
    timeout-minutes: 60
    steps:
      - uses: actions/checkout@v4
-        with:
-          fetch-depth: 0
-          persist-credentials: false
      - uses: ./.github/actions/setup-cyborg
      - uses: arduino/setup-protoc@v3
        with:
          repo-token: ${{ secrets.GITHUB_TOKEN }}
-      - uses: actions-rust-lang/setup-rust-toolchain@v1
+      - uses: dtolnay/rust-toolchain@master
+        with:
+          toolchain: ${{ env.RUST_TOOLCHAIN }}
      - name: Rust Cache
        uses: Swatinem/rust-cache@v2
      - name: Run sqlness
-        run: make sqlness-test
-        env:
-          SQLNESS_OPTS: "--preserve-state"
+        run: cargo sqlness
      - name: Upload sqlness logs
-        if: failure()
+        if: always()
        uses: actions/upload-artifact@v4
        with:
          name: sqlness-logs
-          path: C:\Users\RUNNER~1\AppData\Local\Temp\sqlness*
+          path: /tmp/greptime-*.log
          retention-days: 3

  test-on-windows:
@@ -79,9 +68,6 @@ jobs:
    steps:
      - run: git config --global core.autocrlf false
      - uses: actions/checkout@v4
-        with:
-          fetch-depth: 0
-          persist-credentials: false
      - uses: ./.github/actions/setup-cyborg
      - uses: arduino/setup-protoc@v3
        with:
@@ -90,49 +76,46 @@ jobs:
        with:
          version: "14.0"
      - name: Install Rust toolchain
-        uses: actions-rust-lang/setup-rust-toolchain@v1
+        uses: dtolnay/rust-toolchain@master
        with:
+          toolchain: ${{ env.RUST_TOOLCHAIN }}
          components: llvm-tools-preview
      - name: Rust Cache
        uses: Swatinem/rust-cache@v2
      - name: Install Cargo Nextest
        uses: taiki-e/install-action@nextest
+      - name: Install Python
+        uses: actions/setup-python@v5
+        with:
+          python-version: "3.10"
+      - name: Install PyArrow Package
+        run: pip install pyarrow
      - name: Install WSL distribution
        uses: Vampire/setup-wsl@v2
        with:
          distribution: Ubuntu-22.04
      - name: Running tests
-        run: cargo nextest run -F dashboard
+        run: cargo nextest run -F pyo3_backend,dashboard
        env:
          CARGO_BUILD_RUSTFLAGS: "-C linker=lld-link"
          RUST_BACKTRACE: 1
          CARGO_INCREMENTAL: 0
+          RUSTUP_WINDOWS_PATH_ADD_BIN: 1 # Workaround for https://github.com/nextest-rs/nextest/issues/1493
          GT_S3_BUCKET: ${{ vars.AWS_CI_TEST_BUCKET }}
          GT_S3_ACCESS_KEY_ID: ${{ secrets.AWS_CI_TEST_ACCESS_KEY_ID }}
          GT_S3_ACCESS_KEY: ${{ secrets.AWS_CI_TEST_SECRET_ACCESS_KEY }}
          GT_S3_REGION: ${{ vars.AWS_CI_TEST_BUCKET_REGION }}
          UNITTEST_LOG_DIR: "__unittest_logs"

-  cleanbuild-linux-nix:
-    name: Run clean build on Linux
-    runs-on: ubuntu-latest
-    if: ${{ github.repository == 'GreptimeTeam/greptimedb' }}
-    timeout-minutes: 60
-    steps:
-      - uses: actions/checkout@v4
-        with:
-          fetch-depth: 0
-          persist-credentials: false
-      - uses: cachix/install-nix-action@v27
-        with:
-          nix_path: nixpkgs=channel:nixos-24.11
-      - run: nix develop --command cargo build
-
  check-status:
    name: Check status
-    needs: [sqlness-test, sqlness-windows, test-on-windows]
+    needs: [
+      sqlness-test,
+      sqlness-windows,
+      test-on-windows,
+    ]
    if: ${{ github.repository == 'GreptimeTeam/greptimedb' }}
-    runs-on: ubuntu-latest
+    runs-on: ubuntu-20.04
    outputs:
      check-result: ${{ steps.set-check-result.outputs.check-result }}
    steps:
@@ -144,15 +127,14 @@ jobs:
  notification:
    if: ${{ github.repository == 'GreptimeTeam/greptimedb' && always() }} # Not requiring successful dependent jobs, always run.
    name: Send notification to Greptime team
-    needs: [check-status]
-    runs-on: ubuntu-latest
+    needs: [
+      check-status
+    ]
+    runs-on: ubuntu-20.04
    env:
      SLACK_WEBHOOK_URL: ${{ secrets.SLACK_WEBHOOK_URL_DEVELOP_CHANNEL }}
    steps:
      - uses: actions/checkout@v4
-        with:
-          fetch-depth: 0
-          persist-credentials: false
      - uses: ./.github/actions/setup-cyborg
      - name: Report CI status
        id: report-ci-status
--- a/.github/workflows/release-dev-builder-images.yaml
+++ b/.github/workflows/release-dev-builder-images.yaml
@@ -1,14 +1,12 @@
 name: Release dev-builder images

 on:
-  push:
-    branches:
-      - main
-    paths:
-      - rust-toolchain.toml
-      - 'docker/dev-builder/**'
  workflow_dispatch: # Allows you to run this workflow manually.
    inputs:
+      version:
+        description: Version of the dev-builder
+        required: false
+        default: latest
      release_dev_builder_ubuntu_image:
        type: boolean
        description: Release dev-builder-ubuntu image
@@ -29,175 +27,59 @@ jobs:
  release-dev-builder-images:
    name: Release dev builder images
    if: ${{ inputs.release_dev_builder_ubuntu_image || inputs.release_dev_builder_centos_image || inputs.release_dev_builder_android_image }} # Only manually trigger this job.
-    runs-on: ubuntu-latest
-    outputs:
-      version: ${{ steps.set-version.outputs.version }}
+    runs-on: ubuntu-20.04-16-cores
    steps:
      - name: Checkout
        uses: actions/checkout@v4
        with:
          fetch-depth: 0
-          persist-credentials: false
-
-      - name: Configure build image version
-        id: set-version
-        shell: bash
-        run: |
-          commitShortSHA=`echo ${{ github.sha }} | cut -c1-8`
-          buildTime=`date +%Y%m%d%H%M%S`
-          BUILD_VERSION="$commitShortSHA-$buildTime"
-          RUST_TOOLCHAIN_VERSION=$(cat rust-toolchain.toml | grep -Eo '[0-9]{4}-[0-9]{2}-[0-9]{2}')
-          IMAGE_VERSION="${RUST_TOOLCHAIN_VERSION}-${BUILD_VERSION}"
-          echo "VERSION=${IMAGE_VERSION}" >> $GITHUB_ENV
-          echo "version=$IMAGE_VERSION" >> $GITHUB_OUTPUT

      - name: Build and push dev builder images
        uses: ./.github/actions/build-dev-builder-images
        with:
-          version: ${{ env.VERSION }}
+          version: ${{ inputs.version }}
          dockerhub-image-registry-username: ${{ secrets.DOCKERHUB_USERNAME }}
          dockerhub-image-registry-token: ${{ secrets.DOCKERHUB_TOKEN }}
          build-dev-builder-ubuntu: ${{ inputs.release_dev_builder_ubuntu_image }}
          build-dev-builder-centos: ${{ inputs.release_dev_builder_centos_image }}
          build-dev-builder-android: ${{ inputs.release_dev_builder_android_image }}

-  release-dev-builder-images-ecr:
-    name: Release dev builder images to AWS ECR
-    runs-on: ubuntu-latest
-    needs: [
-      release-dev-builder-images
-    ]
-    steps:
-      - name: Configure AWS credentials
-        uses: aws-actions/configure-aws-credentials@v4
-        with:
-          aws-access-key-id: ${{ secrets.AWS_ECR_ACCESS_KEY_ID }}
-          aws-secret-access-key: ${{ secrets.AWS_ECR_SECRET_ACCESS_KEY }}
-          aws-region: ${{ vars.ECR_REGION }}
-
-      - name: Login to Amazon ECR
-        id: login-ecr-public
-        uses: aws-actions/amazon-ecr-login@v2
-        env:
-          AWS_REGION: ${{ vars.ECR_REGION }}
-        with:
-          registry-type: public
-
-      - name: Push dev-builder-ubuntu image
-        shell: bash
-        if: ${{ inputs.release_dev_builder_ubuntu_image }}
-        env:
-          IMAGE_VERSION: ${{ needs.release-dev-builder-images.outputs.version }}
-          IMAGE_NAMESPACE: ${{ vars.IMAGE_NAMESPACE }}
-          ECR_IMAGE_REGISTRY: ${{ vars.ECR_IMAGE_REGISTRY }}
-          ECR_IMAGE_NAMESPACE: ${{ vars.ECR_IMAGE_NAMESPACE }}
-        run: |
-          docker run -v "${DOCKER_CONFIG:-$HOME/.docker}:/root/.docker:ro" \
-            -e "REGISTRY_AUTH_FILE=/root/.docker/config.json" \
-            quay.io/skopeo/stable:latest \
-            copy -a docker://docker.io/$IMAGE_NAMESPACE/dev-builder-ubuntu:$IMAGE_VERSION \
-            docker://$ECR_IMAGE_REGISTRY/$ECR_IMAGE_NAMESPACE/dev-builder-ubuntu:$IMAGE_VERSION
-
-          docker run -v "${DOCKER_CONFIG:-$HOME/.docker}:/root/.docker:ro" \
-            -e "REGISTRY_AUTH_FILE=/root/.docker/config.json" \
-            quay.io/skopeo/stable:latest \
-            copy -a docker://docker.io/$IMAGE_NAMESPACE/dev-builder-ubuntu:latest \
-            docker://$ECR_IMAGE_REGISTRY/$ECR_IMAGE_NAMESPACE/dev-builder-ubuntu:latest
-
-      - name: Push dev-builder-centos image
-        shell: bash
-        if: ${{ inputs.release_dev_builder_centos_image }}
-        env:
-          IMAGE_VERSION: ${{ needs.release-dev-builder-images.outputs.version }}
-          IMAGE_NAMESPACE: ${{ vars.IMAGE_NAMESPACE }}
-          ECR_IMAGE_REGISTRY: ${{ vars.ECR_IMAGE_REGISTRY }}
-          ECR_IMAGE_NAMESPACE: ${{ vars.ECR_IMAGE_NAMESPACE }}
-        run: |
-          docker run -v "${DOCKER_CONFIG:-$HOME/.docker}:/root/.docker:ro" \
-            -e "REGISTRY_AUTH_FILE=/root/.docker/config.json" \
-            quay.io/skopeo/stable:latest \
-            copy -a docker://docker.io/$IMAGE_NAMESPACE/dev-builder-centos:$IMAGE_VERSION \
-            docker://$ECR_IMAGE_REGISTRY/$ECR_IMAGE_NAMESPACE/dev-builder-centos:$IMAGE_VERSION
-
-          docker run -v "${DOCKER_CONFIG:-$HOME/.docker}:/root/.docker:ro" \
-            -e "REGISTRY_AUTH_FILE=/root/.docker/config.json" \
-            quay.io/skopeo/stable:latest \
-            copy -a docker://docker.io/$IMAGE_NAMESPACE/dev-builder-centos:latest \
-            docker://$ECR_IMAGE_REGISTRY/$ECR_IMAGE_NAMESPACE/dev-builder-centos:latest
-
-      - name: Push dev-builder-android image
-        shell: bash
-        if: ${{ inputs.release_dev_builder_android_image }}
-        env:
-          IMAGE_VERSION: ${{ needs.release-dev-builder-images.outputs.version }}
-          IMAGE_NAMESPACE: ${{ vars.IMAGE_NAMESPACE }}
-          ECR_IMAGE_REGISTRY: ${{ vars.ECR_IMAGE_REGISTRY }}
-          ECR_IMAGE_NAMESPACE: ${{ vars.ECR_IMAGE_NAMESPACE }}
-        run: |
-          docker run -v "${DOCKER_CONFIG:-$HOME/.docker}:/root/.docker:ro" \
-            -e "REGISTRY_AUTH_FILE=/root/.docker/config.json" \
-            quay.io/skopeo/stable:latest \
-            copy -a docker://docker.io/$IMAGE_NAMESPACE/dev-builder-android:$IMAGE_VERSION \
-            docker://$ECR_IMAGE_REGISTRY/$ECR_IMAGE_NAMESPACE/dev-builder-android:$IMAGE_VERSION
-
-          docker run -v "${DOCKER_CONFIG:-$HOME/.docker}:/root/.docker:ro" \
-            -e "REGISTRY_AUTH_FILE=/root/.docker/config.json" \
-            quay.io/skopeo/stable:latest \
-            copy -a docker://docker.io/$IMAGE_NAMESPACE/dev-builder-android:latest \
-            docker://$ECR_IMAGE_REGISTRY/$ECR_IMAGE_NAMESPACE/dev-builder-android:latest
-
  release-dev-builder-images-cn: # Note: Be careful issue: https://github.com/containers/skopeo/issues/1874 and we decide to use the latest stable skopeo container.
    name: Release dev builder images to CN region
-    runs-on: ubuntu-latest
+    runs-on: ubuntu-20.04
    needs: [
      release-dev-builder-images
    ]
    steps:
-      - name: Login to AliCloud Container Registry
-        uses: docker/login-action@v3
-        with:
-          registry: ${{ vars.ACR_IMAGE_REGISTRY }}
-          username: ${{ secrets.ALICLOUD_USERNAME }}
-          password: ${{ secrets.ALICLOUD_PASSWORD }}
-
      - name: Push dev-builder-ubuntu image
        shell: bash
        if: ${{ inputs.release_dev_builder_ubuntu_image }}
        env:
-          IMAGE_VERSION: ${{ needs.release-dev-builder-images.outputs.version }}
-          IMAGE_NAMESPACE: ${{ vars.IMAGE_NAMESPACE }}
-          ACR_IMAGE_REGISTRY: ${{ vars.ACR_IMAGE_REGISTRY }}
+          DST_REGISTRY_USERNAME: ${{ secrets.ALICLOUD_USERNAME }}
+          DST_REGISTRY_PASSWORD: ${{ secrets.ALICLOUD_PASSWORD }}
        run: |
-          docker run -v "${DOCKER_CONFIG:-$HOME/.docker}:/root/.docker:ro" \
-            -e "REGISTRY_AUTH_FILE=/root/.docker/config.json" \
-            quay.io/skopeo/stable:latest \
-            copy -a docker://docker.io/$IMAGE_NAMESPACE/dev-builder-ubuntu:$IMAGE_VERSION \
-            docker://$ACR_IMAGE_REGISTRY/$IMAGE_NAMESPACE/dev-builder-ubuntu:$IMAGE_VERSION
+          docker run quay.io/skopeo/stable:latest copy -a docker://docker.io/${{ vars.IMAGE_NAMESPACE }}/dev-builder-ubuntu:${{ inputs.version }} \
+            --dest-creds "$DST_REGISTRY_USERNAME":"$DST_REGISTRY_PASSWORD" \
+            docker://${{ vars.ACR_IMAGE_REGISTRY }}/${{ vars.IMAGE_NAMESPACE }}/dev-builder-ubuntu:${{ inputs.version }}

      - name: Push dev-builder-centos image
        shell: bash
        if: ${{ inputs.release_dev_builder_centos_image }}
        env:
-          IMAGE_VERSION: ${{ needs.release-dev-builder-images.outputs.version }}
-          IMAGE_NAMESPACE: ${{ vars.IMAGE_NAMESPACE }}
-          ACR_IMAGE_REGISTRY: ${{ vars.ACR_IMAGE_REGISTRY }}
+          DST_REGISTRY_USERNAME: ${{ secrets.ALICLOUD_USERNAME }}
+          DST_REGISTRY_PASSWORD: ${{ secrets.ALICLOUD_PASSWORD }}
        run: |
-          docker run -v "${DOCKER_CONFIG:-$HOME/.docker}:/root/.docker:ro" \
-            -e "REGISTRY_AUTH_FILE=/root/.docker/config.json" \
-            quay.io/skopeo/stable:latest \
-            copy -a docker://docker.io/$IMAGE_NAMESPACE/dev-builder-centos:$IMAGE_VERSION \
-            docker://$ACR_IMAGE_REGISTRY/$IMAGE_NAMESPACE/dev-builder-centos:$IMAGE_VERSION
+          docker run quay.io/skopeo/stable:latest copy -a docker://docker.io/${{ vars.IMAGE_NAMESPACE }}/dev-builder-centos:${{ inputs.version }} \
+            --dest-creds "$DST_REGISTRY_USERNAME":"$DST_REGISTRY_PASSWORD" \
+            docker://${{ vars.ACR_IMAGE_REGISTRY }}/${{ vars.IMAGE_NAMESPACE }}/dev-builder-centos:${{ inputs.version }}

      - name: Push dev-builder-android image
        shell: bash
        if: ${{ inputs.release_dev_builder_android_image }}
        env:
-          IMAGE_VERSION: ${{ needs.release-dev-builder-images.outputs.version }}
-          IMAGE_NAMESPACE: ${{ vars.IMAGE_NAMESPACE }}
-          ACR_IMAGE_REGISTRY: ${{ vars.ACR_IMAGE_REGISTRY }}
+          DST_REGISTRY_USERNAME: ${{ secrets.ALICLOUD_USERNAME }}
+          DST_REGISTRY_PASSWORD: ${{ secrets.ALICLOUD_PASSWORD }}
        run: |
-          docker run -v "${DOCKER_CONFIG:-$HOME/.docker}:/root/.docker:ro" \
-            -e "REGISTRY_AUTH_FILE=/root/.docker/config.json" \
-            quay.io/skopeo/stable:latest \
-            copy -a docker://docker.io/$IMAGE_NAMESPACE/dev-builder-android:$IMAGE_VERSION \
-            docker://$ACR_IMAGE_REGISTRY/$IMAGE_NAMESPACE/dev-builder-android:$IMAGE_VERSION
+          docker run quay.io/skopeo/stable:latest copy -a docker://docker.io/${{ vars.IMAGE_NAMESPACE }}/dev-builder-android:${{ inputs.version }} \
+            --dest-creds "$DST_REGISTRY_USERNAME":"$DST_REGISTRY_PASSWORD" \
+            docker://${{ vars.ACR_IMAGE_REGISTRY }}/${{ vars.IMAGE_NAMESPACE }}/dev-builder-android:${{ inputs.version }}
--- a/.github/workflows/release.yml
+++ b/.github/workflows/release.yml
@@ -18,11 +18,11 @@ on:
        description: The runner uses to build linux-amd64 artifacts
        default: ec2-c6i.4xlarge-amd64
        options:
-          - ubuntu-22.04
-          - ubuntu-22.04-8-cores
-          - ubuntu-22.04-16-cores
-          - ubuntu-22.04-32-cores
-          - ubuntu-22.04-64-cores
+          - ubuntu-20.04
+          - ubuntu-20.04-8-cores
+          - ubuntu-20.04-16-cores
+          - ubuntu-20.04-32-cores
+          - ubuntu-20.04-64-cores
          - ec2-c6i.xlarge-amd64 # 4C8G
          - ec2-c6i.2xlarge-amd64 # 8C16G
          - ec2-c6i.4xlarge-amd64 # 16C32G
@@ -31,9 +31,8 @@ on:
      linux_arm64_runner:
        type: choice
        description: The runner uses to build linux-arm64 artifacts
-        default: ec2-c6g.8xlarge-arm64
+        default: ec2-c6g.4xlarge-arm64
        options:
-          - ubuntu-2204-32-cores-arm
          - ec2-c6g.xlarge-arm64 # 4C8G
          - ec2-c6g.2xlarge-arm64 # 8C16G
          - ec2-c6g.4xlarge-arm64 # 16C32G
@@ -83,6 +82,7 @@ on:
 # Use env variables to control all the release process.
 env:
  # The arguments of building greptime.
+  RUST_TOOLCHAIN: nightly-2024-04-20
  CARGO_PROFILE: nightly

  # Controls whether to run tests, include unit-test, integration-test and sqlness.
@@ -91,13 +91,18 @@ env:
  # The scheduled version is '${{ env.NEXT_RELEASE_VERSION }}-nightly-YYYYMMDD', like v0.2.0-nigthly-20230313;
  NIGHTLY_RELEASE_PREFIX: nightly
  # Note: The NEXT_RELEASE_VERSION should be modified manually by every formal release.
-  NEXT_RELEASE_VERSION: v0.13.0
+  NEXT_RELEASE_VERSION: v0.9.0
+
+# Permission reference: https://docs.github.com/en/actions/using-jobs/assigning-permissions-to-jobs
+permissions:
+  issues: write # Allows the action to create issues for cyborg.
+  contents: write # Allows the action to create a release.

 jobs:
  allocate-runners:
    name: Allocate runners
    if: ${{ github.repository == 'GreptimeTeam/greptimedb' }}
-    runs-on: ubuntu-latest
+    runs-on: ubuntu-20.04
    outputs:
      linux-amd64-runner: ${{ steps.start-linux-amd64-runner.outputs.label }}
      linux-arm64-runner: ${{ steps.start-linux-arm64-runner.outputs.label }}
@@ -117,12 +122,6 @@ jobs:
        uses: actions/checkout@v4
        with:
          fetch-depth: 0
-          persist-credentials: false
-
-      - name: Check Rust toolchain version
-        shell: bash
-        run: |
-          ./scripts/check-builder-rust-version.sh

      # The create-version will create a global variable named 'version' in the global workflows.
      # - If it's a tag push release, the version is the tag name(${{ github.ref_name }});
@@ -177,7 +176,6 @@ jobs:
      - uses: actions/checkout@v4
        with:
          fetch-depth: 0
-          persist-credentials: false

      - uses: ./.github/actions/build-linux-artifacts
        with:
@@ -185,8 +183,6 @@ jobs:
          cargo-profile: ${{ env.CARGO_PROFILE }}
          version: ${{ needs.allocate-runners.outputs.version }}
          disable-run-tests: ${{ env.DISABLE_RUN_TESTS }}
-          image-registry: ${{ vars.ECR_IMAGE_REGISTRY }}
-          image-namespace: ${{ vars.ECR_IMAGE_NAMESPACE }}

  build-linux-arm64-artifacts:
    name: Build linux-arm64 artifacts
@@ -199,7 +195,6 @@ jobs:
      - uses: actions/checkout@v4
        with:
          fetch-depth: 0
-          persist-credentials: false

      - uses: ./.github/actions/build-linux-artifacts
        with:
@@ -207,8 +202,6 @@ jobs:
          cargo-profile: ${{ env.CARGO_PROFILE }}
          version: ${{ needs.allocate-runners.outputs.version }}
          disable-run-tests: ${{ env.DISABLE_RUN_TESTS }}
-          image-registry: ${{ vars.ECR_IMAGE_REGISTRY }}
-          image-namespace: ${{ vars.ECR_IMAGE_NAMESPACE }}

  build-macos-artifacts:
    name: Build macOS artifacts
@@ -220,10 +213,18 @@ jobs:
            arch: aarch64-apple-darwin
            features: servers/dashboard
            artifacts-dir-prefix: greptime-darwin-arm64
+          - os: ${{ needs.allocate-runners.outputs.macos-runner }}
+            arch: aarch64-apple-darwin
+            features: pyo3_backend,servers/dashboard
+            artifacts-dir-prefix: greptime-darwin-arm64-pyo3
          - os: ${{ needs.allocate-runners.outputs.macos-runner }}
            features: servers/dashboard
            arch: x86_64-apple-darwin
            artifacts-dir-prefix: greptime-darwin-amd64
+          - os: ${{ needs.allocate-runners.outputs.macos-runner }}
+            features: pyo3_backend,servers/dashboard
+            arch: x86_64-apple-darwin
+            artifacts-dir-prefix: greptime-darwin-amd64-pyo3
    runs-on: ${{ matrix.os }}
    outputs:
      build-macos-result: ${{ steps.set-build-macos-result.outputs.build-macos-result }}
@@ -235,16 +236,15 @@ jobs:
      - uses: actions/checkout@v4
        with:
          fetch-depth: 0
-          persist-credentials: false

      - uses: ./.github/actions/build-macos-artifacts
        with:
          arch: ${{ matrix.arch }}
+          rust-toolchain: ${{ env.RUST_TOOLCHAIN }}
          cargo-profile: ${{ env.CARGO_PROFILE }}
          features: ${{ matrix.features }}
          version: ${{ needs.allocate-runners.outputs.version }}
-          # We decide to disable the integration tests on macOS because it's unnecessary and time-consuming.
-          disable-run-tests: true
+          disable-run-tests: ${{ env.DISABLE_RUN_TESTS }}
          artifacts-dir: ${{ matrix.artifacts-dir-prefix }}-${{ needs.allocate-runners.outputs.version }}

      - name: Set build macos result
@@ -262,6 +262,10 @@ jobs:
            arch: x86_64-pc-windows-msvc
            features: servers/dashboard
            artifacts-dir-prefix: greptime-windows-amd64
+          - os: ${{ needs.allocate-runners.outputs.windows-runner }}
+            arch: x86_64-pc-windows-msvc
+            features: pyo3_backend,servers/dashboard
+            artifacts-dir-prefix: greptime-windows-amd64-pyo3
    runs-on: ${{ matrix.os }}
    outputs:
      build-windows-result: ${{ steps.set-build-windows-result.outputs.build-windows-result }}
@@ -275,11 +279,11 @@ jobs:
      - uses: actions/checkout@v4
        with:
          fetch-depth: 0
-          persist-credentials: false

      - uses: ./.github/actions/build-windows-artifacts
        with:
          arch: ${{ matrix.arch }}
+          rust-toolchain: ${{ env.RUST_TOOLCHAIN }}
          cargo-profile: ${{ env.CARGO_PROFILE }}
          features: ${{ matrix.features }}
          version: ${{ needs.allocate-runners.outputs.version }}
@@ -299,25 +303,22 @@ jobs:
      build-linux-amd64-artifacts,
      build-linux-arm64-artifacts,
    ]
-    runs-on: ubuntu-latest
+    runs-on: ubuntu-2004-16-cores
    outputs:
      build-image-result: ${{ steps.set-build-image-result.outputs.build-image-result }}
    steps:
      - uses: actions/checkout@v4
        with:
          fetch-depth: 0
-          persist-credentials: false

      - name: Build and push images to dockerhub
        uses: ./.github/actions/build-images
        with:
          image-registry: docker.io
          image-namespace: ${{ vars.IMAGE_NAMESPACE }}
-          image-name: ${{ vars.GREPTIMEDB_IMAGE_NAME }}
          image-registry-username: ${{ secrets.DOCKERHUB_USERNAME }}
          image-registry-password: ${{ secrets.DOCKERHUB_TOKEN }}
          version: ${{ needs.allocate-runners.outputs.version }}
-          push-latest-tag: true

      - name: Set build image result
        id: set-build-image-result
@@ -335,7 +336,7 @@ jobs:
      build-windows-artifacts,
      release-images-to-dockerhub,
    ]
-    runs-on: ubuntu-latest
+    runs-on: ubuntu-20.04
    # When we push to ACR, it's easy to fail due to some unknown network issues.
    # However, we don't want to fail the whole workflow because of this.
    # The ACR have daily sync with DockerHub, so don't worry about the image not being updated.
@@ -344,14 +345,13 @@ jobs:
      - uses: actions/checkout@v4
        with:
          fetch-depth: 0
-          persist-credentials: false

      - name: Release artifacts to CN region
        uses: ./.github/actions/release-cn-artifacts
        with:
          src-image-registry: docker.io
          src-image-namespace: ${{ vars.IMAGE_NAMESPACE }}
-          src-image-name: ${{ vars.GREPTIMEDB_IMAGE_NAME }}
+          src-image-name: greptimedb
          dst-image-registry-username: ${{ secrets.ALICLOUD_USERNAME }}
          dst-image-registry-password: ${{ secrets.ALICLOUD_PASSWORD }}
          dst-image-registry: ${{ vars.ACR_IMAGE_REGISTRY }}
@@ -362,7 +362,6 @@ jobs:
          aws-cn-secret-access-key: ${{ secrets.AWS_CN_SECRET_ACCESS_KEY }}
          aws-cn-region: ${{ vars.AWS_RELEASE_BUCKET_REGION }}
          dev-mode: false
-          upload-to-s3: true
          update-version-info: true
          push-latest-tag: true

@@ -377,12 +376,11 @@ jobs:
      build-windows-artifacts,
      release-images-to-dockerhub,
    ]
-    runs-on: ubuntu-latest
+    runs-on: ubuntu-20.04
    steps:
      - uses: actions/checkout@v4
        with:
          fetch-depth: 0
-          persist-credentials: false

      - name: Publish GitHub release
        uses: ./.github/actions/publish-github-release
@@ -396,7 +394,7 @@ jobs:
    name: Stop linux-amd64 runner
    # Only run this job when the runner is allocated.
    if: ${{ always() }}
-    runs-on: ubuntu-latest
+    runs-on: ubuntu-20.04
    needs: [
      allocate-runners,
      build-linux-amd64-artifacts,
@@ -406,7 +404,6 @@ jobs:
        uses: actions/checkout@v4
        with:
          fetch-depth: 0
-          persist-credentials: false

      - name: Stop EC2 runner
        uses: ./.github/actions/stop-runner
@@ -422,7 +419,7 @@ jobs:
    name: Stop linux-arm64 runner
    # Only run this job when the runner is allocated.
    if: ${{ always() }}
-    runs-on: ubuntu-latest
+    runs-on: ubuntu-20.04
    needs: [
      allocate-runners,
      build-linux-arm64-artifacts,
@@ -432,7 +429,6 @@ jobs:
        uses: actions/checkout@v4
        with:
          fetch-depth: 0
-          persist-credentials: false

      - name: Stop EC2 runner
        uses: ./.github/actions/stop-runner
@@ -444,29 +440,6 @@ jobs:
          aws-region: ${{ vars.EC2_RUNNER_REGION }}
          github-token: ${{ secrets.GH_PERSONAL_ACCESS_TOKEN }}

-  bump-doc-version:
-    name: Bump doc version
-    if: ${{ github.event_name == 'push' || github.event_name == 'schedule' }}
-    needs: [allocate-runners]
-    runs-on: ubuntu-latest
-    # Permission reference: https://docs.github.com/en/actions/using-jobs/assigning-permissions-to-jobs
-    permissions:
-      issues: write # Allows the action to create issues for cyborg.
-      contents: write # Allows the action to create a release.
-    steps:
-      - uses: actions/checkout@v4
-        with:
-          fetch-depth: 0
-          persist-credentials: false
-      - uses: ./.github/actions/setup-cyborg
-      - name: Bump doc version
-        working-directory: cyborg
-        run: pnpm tsx bin/bump-doc-version.ts
-        env:
-          VERSION: ${{ needs.allocate-runners.outputs.version }}
-          GITHUB_TOKEN: ${{ secrets.GITHUB_TOKEN }}
-          DOCS_REPO_TOKEN: ${{ secrets.DOCS_REPO_TOKEN }}
-
  notification:
    if: ${{ github.repository == 'GreptimeTeam/greptimedb' && (github.event_name == 'push' || github.event_name == 'schedule') && always() }}
    name: Send notification to Greptime team
@@ -475,18 +448,11 @@ jobs:
      build-macos-artifacts,
      build-windows-artifacts,
    ]
-    runs-on: ubuntu-latest
-    # Permission reference: https://docs.github.com/en/actions/using-jobs/assigning-permissions-to-jobs
-    permissions:
-      issues: write # Allows the action to create issues for cyborg.
-      contents: write # Allows the action to create a release.
+    runs-on: ubuntu-20.04
    env:
      SLACK_WEBHOOK_URL: ${{ secrets.SLACK_WEBHOOK_URL_DEVELOP_CHANNEL }}
    steps:
      - uses: actions/checkout@v4
-        with:
-          fetch-depth: 0
-          persist-credentials: false
      - uses: ./.github/actions/setup-cyborg
      - name: Report CI status
        id: report-ci-status
--- a/.github/workflows/schedule.yml
+++ b/.github/workflows/schedule.yml
@@ -4,20 +4,18 @@ on:
    - cron: '4 2 * * *'
  workflow_dispatch:

+permissions:
+  contents: read
+  issues: write
+  pull-requests: write

 jobs:
  maintenance:
    name: Periodic Maintenance
    runs-on: ubuntu-latest
-    permissions:
-      contents: read
-      issues: write
-      pull-requests: write
    if: ${{ github.repository == 'GreptimeTeam/greptimedb' }}
    steps:
      - uses: actions/checkout@v4
-        with:
-          persist-credentials: false
      - uses: ./.github/actions/setup-cyborg
      - name: Do Maintenance
        working-directory: cyborg
--- a/.github/workflows/semantic-pull-request.yml
+++ b/.github/workflows/semantic-pull-request.yml
@@ -1,24 +1,18 @@
 name: "Semantic Pull Request"

 on:
-  pull_request:
+  pull_request_target:
    types:
      - opened
      - reopened
      - edited

-concurrency:
-  group: ${{ github.workflow }}-${{ github.head_ref || github.run_id }}
-  cancel-in-progress: true
-
 jobs:
  check:
-    runs-on: ubuntu-latest
+    runs-on: ubuntu-20.04
    timeout-minutes: 10
    steps:
      - uses: actions/checkout@v4
-        with:
-          persist-credentials: false
      - uses: ./.github/actions/setup-cyborg
      - name: Check Pull Request
        working-directory: cyborg
--- a/.gitignore
+++ b/.gitignore
@@ -47,10 +47,6 @@ benchmarks/data

 venv/

-# Fuzz tests
+# Fuzz tests 
 tests-fuzz/artifacts/
 tests-fuzz/corpus/
-
-# Nix
-.direnv
-.envrc
--- a/.pre-commit-config.yaml
+++ b/.pre-commit-config.yaml
@@ -16,7 +16,6 @@ repos:
    hooks:
    -    id: fmt
    -    id: clippy
-         args: ["--workspace", "--all-targets", "--all-features", "--", "-D", "warnings"]
-         stages: [pre-push]
+         args: ["--workspace", "--all-targets", "--", "-D", "warnings", "-D", "clippy::print_stdout", "-D", "clippy::print_stderr"]
+         stages: [push]
    -    id: cargo-check
-         args: ["--workspace", "--all-targets", "--all-features"]
--- a/AUTHOR.md
+++ b/AUTHOR.md
@@ -1,45 +0,0 @@
-# GreptimeDB Authors
-
-## Individual Committers (in alphabetical order)
-
-* [CookiePieWw](https://github.com/CookiePieWw)
-* [etolbakov](https://github.com/etolbakov)
-* [irenjj](https://github.com/irenjj)
-* [KKould](https://github.com/KKould)
-* [Lanqing Yang](https://github.com/lyang24)
-* [NiwakaDev](https://github.com/NiwakaDev)
-* [tisonkun](https://github.com/tisonkun)
-
-
-## Team Members (in alphabetical order)
-
-* [apdong2022](https://github.com/apdong2022)
-* [beryl678](https://github.com/beryl678)
-* [Breeze-P](https://github.com/Breeze-P)
-* [daviderli614](https://github.com/daviderli614)
-* [discord9](https://github.com/discord9)
-* [evenyag](https://github.com/evenyag)
-* [fengjiachun](https://github.com/fengjiachun)
-* [fengys1996](https://github.com/fengys1996)
-* [GrepTime](https://github.com/GrepTime)
-* [holalengyu](https://github.com/holalengyu)
-* [killme2008](https://github.com/killme2008)
-* [MichaelScofield](https://github.com/MichaelScofield)
-* [nicecui](https://github.com/nicecui)
-* [paomian](https://github.com/paomian)
-* [shuiyisong](https://github.com/shuiyisong)
-* [sunchanglong](https://github.com/sunchanglong)
-* [sunng87](https://github.com/sunng87)
-* [v0y4g3r](https://github.com/v0y4g3r)
-* [waynexia](https://github.com/waynexia)
-* [Wenjie0329](https://github.com/Wenjie0329)
-* [WenyXu](https://github.com/WenyXu)
-* [xtang](https://github.com/xtang)
-* [zhaoyingnan01](https://github.com/zhaoyingnan01)
-* [zhongzc](https://github.com/zhongzc)
-* [ZonaHex](https://github.com/ZonaHex)
-* [zyy17](https://github.com/zyy17)
-
-## All Contributors
-
-To see the full list of contributors, please visit our [Contributors page](https://github.com/GreptimeTeam/greptimedb/graphs/contributors)
--- a/CONTRIBUTING.md
+++ b/CONTRIBUTING.md
@@ -2,11 +2,7 @@

 Thanks a lot for considering contributing to GreptimeDB. We believe people like you would make GreptimeDB a great product. We intend to build a community where individuals can have open talks, show respect for one another, and speak with true ❤️. Meanwhile, we are to keep transparency and make your effort count here.

-You can find our contributors at https://github.com/GreptimeTeam/greptimedb/graphs/contributors. When you dedicate to GreptimeDB for a few months and keep bringing high-quality contributions (code, docs, advocate, etc.), you will be a candidate of a committer.
-
-A committer will be granted both read & write access to GreptimeDB repos. Check the [AUTHOR.md](AUTHOR.md) file for all current individual committers.
-
-Please read the guidelines, and they can help you get started. Communicate respectfully with the developers maintaining and developing the project. In return, they should reciprocate that respect by addressing your issue, reviewing changes, as well as helping finalize and merge your pull requests.
+Please read the guidelines, and they can help you get started. Communicate with respect to developers maintaining and developing the project. In return, they should reciprocate that respect by addressing your issue, reviewing changes, as well as helping finalize and merge your pull requests.

 Follow our [README](https://github.com/GreptimeTeam/greptimedb#readme) to get the whole picture of the project. To learn about the design of GreptimeDB, please refer to the [design docs](https://github.com/GrepTimeTeam/docs).

@@ -14,7 +10,7 @@ Follow our [README](https://github.com/GreptimeTeam/greptimedb#readme) to get th

 It can feel intimidating to contribute to a complex project, but it can also be exciting and fun. These general notes will help everyone participate in this communal activity.

- Follow the [Code of Conduct](https://github.com/GreptimeTeam/.github/blob/main/.github/CODE_OF_CONDUCT.md)
+- Follow the [Code of Conduct](https://github.com/GreptimeTeam/greptimedb/blob/main/CODE_OF_CONDUCT.md)
 - Small changes make huge differences. We will happily accept a PR making a single character change if it helps move forward. Don't wait to have everything working.
 - Check the closed issues before opening your issue.
 - Try to follow the existing style of the code.
@@ -30,7 +26,7 @@ Pull requests are great, but we accept all kinds of other help if you like. Such

 ## Code of Conduct

-Also, there are things that we are not looking for because they don't match the goals of the product or benefit the community. Please read [Code of Conduct](https://github.com/GreptimeTeam/.github/blob/main/.github/CODE_OF_CONDUCT.md); we hope everyone can keep good manners and become an honored member.
+Also, there are things that we are not looking for because they don't match the goals of the product or benefit the community. Please read [Code of Conduct](https://github.com/GreptimeTeam/greptimedb/blob/main/CODE_OF_CONDUCT.md); we hope everyone can keep good manners and become an honored member.

 ## License

@@ -55,7 +51,7 @@ GreptimeDB uses the [Apache 2.0 license](https://github.com/GreptimeTeam/greptim
 - To ensure that community is free and confident in its ability to use your contributions, please sign the Contributor License Agreement (CLA) which will be incorporated in the pull request process.
 - Make sure all files have proper license header (running `docker run --rm -v $(pwd):/github/workspace ghcr.io/korandoru/hawkeye-native:v3 format` from the project root).
 - Make sure all your codes are formatted and follow the [coding style](https://pingcap.github.io/style-guide/rust/) and [style guide](docs/style-guide.md).
- Make sure all unit tests are passed using [nextest](https://nexte.st/index.html) `cargo nextest run`.
+- Make sure all unit tests are passed (using `cargo test --workspace` or [nextest](https://nexte.st/index.html) `cargo nextest run`).
 - Make sure all clippy warnings are fixed (you can check it locally by running `cargo clippy --workspace --all-targets -- -D warnings`).

 #### `pre-commit` Hooks
--- a/Cargo.lock
+++ b/Cargo.lock
--- a/Cargo.toml
+++ b/Cargo.toml
@@ -1,29 +1,26 @@
 [workspace]
 members = [
+    "benchmarks",
    "src/api",
    "src/auth",
-    "src/cache",
    "src/catalog",
-    "src/cli",
+    "src/cache",
    "src/client",
    "src/cmd",
    "src/common/base",
    "src/common/catalog",
    "src/common/config",
    "src/common/datasource",
-    "src/common/decimal",
    "src/common/error",
    "src/common/frontend",
    "src/common/function",
+    "src/common/macro",
    "src/common/greptimedb-telemetry",
    "src/common/grpc",
    "src/common/grpc-expr",
-    "src/common/macro",
    "src/common/mem-prof",
    "src/common/meta",
-    "src/common/options",
    "src/common/plugins",
-    "src/common/pprof",
    "src/common/procedure",
    "src/common/procedure-test",
    "src/common/query",
@@ -33,6 +30,7 @@ members = [
    "src/common/telemetry",
    "src/common/test-util",
    "src/common/time",
+    "src/common/decimal",
    "src/common/version",
    "src/common/wal",
    "src/datanode",
@@ -40,8 +38,6 @@ members = [
    "src/file-engine",
    "src/flow",
    "src/frontend",
-    "src/index",
-    "src/log-query",
    "src/log-store",
    "src/meta-client",
    "src/meta-srv",
@@ -50,16 +46,17 @@ members = [
    "src/object-store",
    "src/operator",
    "src/partition",
-    "src/pipeline",
    "src/plugins",
    "src/promql",
    "src/puffin",
    "src/query",
+    "src/script",
    "src/servers",
    "src/session",
    "src/sql",
    "src/store-api",
    "src/table",
+    "src/index",
    "tests-fuzz",
    "tests-integration",
    "tests/runner",
@@ -67,7 +64,7 @@ members = [
 resolver = "2"

 [workspace.package]
-version = "0.13.0"
+version = "0.8.1"
 edition = "2021"
 license = "Apache-2.0"

@@ -78,10 +75,10 @@ clippy.dbg_macro = "warn"
 clippy.implicit_clone = "warn"
 clippy.readonly_write_lock = "allow"
 rust.unknown_lints = "deny"
-rust.unexpected_cfgs = { level = "warn", check-cfg = ['cfg(tokio_unstable)'] }
+# Remove this after https://github.com/PyO3/pyo3/issues/4094
+rust.non_local_definitions = "allow"

 [workspace.dependencies]
-# DO_NOT_REMOVE_THIS: BEGIN_OF_EXTERNAL_DEPENDENCIES
 # We turn off default-features for some dependencies here so the workspaces which inherit them can
 # selectively turn them on if needed, since we can override default-features = true (from false)
 # for the inherited dependency but cannot do the reverse (override from true to false).
@@ -89,140 +86,103 @@ rust.unexpected_cfgs = { level = "warn", check-cfg = ['cfg(tokio_unstable)'] }
 # See for more detaiils: https://github.com/rust-lang/cargo/issues/11329
 ahash = { version = "0.8", features = ["compile-time-rng"] }
 aquamarine = "0.3"
-arrow = { version = "53.0.0", features = ["prettyprint"] }
-arrow-array = { version = "53.0.0", default-features = false, features = ["chrono-tz"] }
-arrow-flight = "53.0"
-arrow-ipc = { version = "53.0.0", default-features = false, features = ["lz4", "zstd"] }
-arrow-schema = { version = "53.0", features = ["serde"] }
+arrow = { version = "51.0.0", features = ["prettyprint"] }
+arrow-array = { version = "51.0.0", default-features = false, features = ["chrono-tz"] }
+arrow-flight = "51.0"
+arrow-ipc = { version = "51.0.0", default-features = false, features = ["lz4"] }
+arrow-schema = { version = "51.0", features = ["serde"] }
 async-stream = "0.3"
 async-trait = "0.1"
-# Remember to update axum-extra, axum-macros when updating axum
-axum = "0.8"
-axum-extra = "0.10"
-axum-macros = "0.4"
-backon = "1"
+axum = { version = "0.6", features = ["headers"] }
 base64 = "0.21"
 bigdecimal = "0.4.2"
 bitflags = "2.4.1"
 bytemuck = "1.12"
-bytes = { version = "1.7", features = ["serde"] }
+bytes = { version = "1.5", features = ["serde"] }
 chrono = { version = "0.4", features = ["serde"] }
-chrono-tz = "0.10.1"
 clap = { version = "4.4", features = ["derive"] }
 config = "0.13.0"
 crossbeam-utils = "0.8"
 dashmap = "5.4"
-datafusion = { git = "https://github.com/apache/datafusion.git", rev = "2464703c84c400a09cc59277018813f0e797bb4e" }
-datafusion-common = { git = "https://github.com/apache/datafusion.git", rev = "2464703c84c400a09cc59277018813f0e797bb4e" }
-datafusion-expr = { git = "https://github.com/apache/datafusion.git", rev = "2464703c84c400a09cc59277018813f0e797bb4e" }
-datafusion-functions = { git = "https://github.com/apache/datafusion.git", rev = "2464703c84c400a09cc59277018813f0e797bb4e" }
-datafusion-optimizer = { git = "https://github.com/apache/datafusion.git", rev = "2464703c84c400a09cc59277018813f0e797bb4e" }
-datafusion-physical-expr = { git = "https://github.com/apache/datafusion.git", rev = "2464703c84c400a09cc59277018813f0e797bb4e" }
-datafusion-physical-plan = { git = "https://github.com/apache/datafusion.git", rev = "2464703c84c400a09cc59277018813f0e797bb4e" }
-datafusion-sql = { git = "https://github.com/apache/datafusion.git", rev = "2464703c84c400a09cc59277018813f0e797bb4e" }
-datafusion-substrait = { git = "https://github.com/apache/datafusion.git", rev = "2464703c84c400a09cc59277018813f0e797bb4e" }
-deadpool = "0.10"
-deadpool-postgres = "0.12"
+datafusion = { git = "https://github.com/apache/arrow-datafusion.git", rev = "34eda15b73a9e278af8844b30ed2f1c21c10359c" }
+datafusion-common = { git = "https://github.com/apache/arrow-datafusion.git", rev = "34eda15b73a9e278af8844b30ed2f1c21c10359c" }
+datafusion-expr = { git = "https://github.com/apache/arrow-datafusion.git", rev = "34eda15b73a9e278af8844b30ed2f1c21c10359c" }
+datafusion-functions = { git = "https://github.com/apache/arrow-datafusion.git", rev = "34eda15b73a9e278af8844b30ed2f1c21c10359c" }
+datafusion-optimizer = { git = "https://github.com/apache/arrow-datafusion.git", rev = "34eda15b73a9e278af8844b30ed2f1c21c10359c" }
+datafusion-physical-expr = { git = "https://github.com/apache/arrow-datafusion.git", rev = "34eda15b73a9e278af8844b30ed2f1c21c10359c" }
+datafusion-physical-plan = { git = "https://github.com/apache/arrow-datafusion.git", rev = "34eda15b73a9e278af8844b30ed2f1c21c10359c" }
+datafusion-sql = { git = "https://github.com/apache/arrow-datafusion.git", rev = "34eda15b73a9e278af8844b30ed2f1c21c10359c" }
+datafusion-substrait = { git = "https://github.com/apache/arrow-datafusion.git", rev = "34eda15b73a9e278af8844b30ed2f1c21c10359c" }
 derive_builder = "0.12"
 dotenv = "0.15"
-etcd-client = "0.14"
+# TODO(LFC): Wait for https://github.com/etcdv3/etcd-client/pull/76
+etcd-client = { git = "https://github.com/MichaelScofield/etcd-client.git", rev = "4c371e9b3ea8e0a8ee2f9cbd7ded26e54a45df3b" }
 fst = "0.4.7"
 futures = "0.3"
 futures-util = "0.3"
-greptime-proto = { git = "https://github.com/GreptimeTeam/greptime-proto.git", rev = "2be0f36b3264e28ab0e1c22a980d0bb634eb3a77" }
-hex = "0.4"
-http = "1"
+greptime-proto = { git = "https://github.com/GreptimeTeam/greptime-proto.git", rev = "ae26136accd82fbdf8be540cd502f2e94951077e" }
 humantime = "2.1"
 humantime-serde = "1.1"
-hyper = "1.1"
-hyper-util = "0.1"
 itertools = "0.10"
-jsonb = { git = "https://github.com/databendlabs/jsonb.git", rev = "8c8d2fc294a39f3ff08909d60f718639cfba3875", default-features = false }
 lazy_static = "1.4"
-local-ip-address = "0.6"
-loki-proto = { git = "https://github.com/GreptimeTeam/loki-proto.git", rev = "1434ecf23a2654025d86188fb5205e7a74b225d3" }
-meter-core = { git = "https://github.com/GreptimeTeam/greptime-meter.git", rev = "5618e779cf2bb4755b499c630fba4c35e91898cb" }
+meter-core = { git = "https://github.com/GreptimeTeam/greptime-meter.git", rev = "80b72716dcde47ec4161478416a5c6c21343364d" }
 mockall = "0.11.4"
 moka = "0.12"
-nalgebra = "0.33"
 notify = "6.1"
 num_cpus = "1.16"
 once_cell = "1.18"
-opentelemetry-proto = { version = "0.27", features = [
+opentelemetry-proto = { version = "0.5", features = [
    "gen-tonic",
    "metrics",
    "trace",
-    "with-serde",
-    "logs",
 ] }
-parking_lot = "0.12"
-parquet = { version = "53.0.0", default-features = false, features = ["arrow", "async", "object_store"] }
+parquet = { version = "51.0.0", default-features = false, features = ["arrow", "async", "object_store"] }
 paste = "1.0"
 pin-project = "1.0"
 prometheus = { version = "0.13.3", features = ["process"] }
-promql-parser = { version = "0.5", features = ["ser"] }
-prost = "0.13"
+promql-parser = { version = "0.4" }
+prost = "0.12"
 raft-engine = { version = "0.4.1", default-features = false }
 rand = "0.8"
-ratelimit = "0.9"
 regex = "1.8"
-regex-automata = "0.4"
-reqwest = { version = "0.12", default-features = false, features = [
+regex-automata = { version = "0.4" }
+reqwest = { version = "0.11", default-features = false, features = [
    "json",
    "rustls-tls-native-roots",
    "stream",
    "multipart",
 ] }
-rskafka = { git = "https://github.com/influxdata/rskafka.git", rev = "75535b5ad9bae4a5dbb582c82e44dfd81ec10105", features = [
-    "transport-tls",
-] }
-rstest = "0.21"
-rstest_reuse = "0.7"
+rskafka = "0.5"
 rust_decimal = "1.33"
-rustc-hash = "2.0"
-rustls = { version = "0.23.20", default-features = false } # override by patch, see [patch.crates-io]
+schemars = "0.8"
 serde = { version = "1.0", features = ["derive"] }
 serde_json = { version = "1.0", features = ["float_roundtrip"] }
 serde_with = "3"
-shadow-rs = "0.38"
-similar-asserts = "1.6.0"
 smallvec = { version = "1", features = ["serde"] }
 snafu = "0.8"
-sqlx = { version = "0.8", features = [
-    "runtime-tokio-rustls",
-    "mysql",
-] }
 sysinfo = "0.30"
-# on branch v0.52.x
-sqlparser = { git = "https://github.com/GreptimeTeam/sqlparser-rs.git", rev = "71dd86058d2af97b9925093d40c4e03360403170", features = [
+# on branch v0.44.x
+sqlparser = { git = "https://github.com/GreptimeTeam/sqlparser-rs.git", rev = "e4e496b8d62416ad50ce70a1b460c7313610cf5d", features = [
    "visitor",
-    "serde",
-] } # on branch v0.44.x
+] }
 strum = { version = "0.25", features = ["derive"] }
 tempfile = "3"
-tokio = { version = "1.40", features = ["full"] }
-tokio-postgres = "0.7"
-tokio-rustls = { version = "0.26.0", default-features = false } # override by patch, see [patch.crates-io]
-tokio-stream = "0.1"
+tokio = { version = "1.36", features = ["full"] }
+tokio-stream = { version = "0.1" }
 tokio-util = { version = "0.7", features = ["io-util", "compat"] }
 toml = "0.8.8"
-tonic = { version = "0.12", features = ["tls", "gzip", "zstd"] }
-tower = "0.5"
-tracing-appender = "0.2"
-tracing-subscriber = { version = "0.3", features = ["env-filter", "json", "fmt"] }
-typetag = "0.2"
+tonic = { version = "0.11", features = ["tls", "gzip", "zstd"] }
+tower = { version = "0.4" }
 uuid = { version = "1.7", features = ["serde", "v4", "fast-rng"] }
 zstd = "0.13"
-# DO_NOT_REMOVE_THIS: END_OF_EXTERNAL_DEPENDENCIES

 ## workspaces members
 api = { path = "src/api" }
 auth = { path = "src/auth" }
 cache = { path = "src/cache" }
 catalog = { path = "src/catalog" }
-cli = { path = "src/cli" }
 client = { path = "src/client" }
-cmd = { path = "src/cmd", default-features = false }
+cmd = { path = "src/cmd" }
 common-base = { path = "src/common/base" }
 common-catalog = { path = "src/common/catalog" }
 common-config = { path = "src/common/config" }
@@ -237,9 +197,7 @@ common-grpc-expr = { path = "src/common/grpc-expr" }
 common-macro = { path = "src/common/macro" }
 common-mem-prof = { path = "src/common/mem-prof" }
 common-meta = { path = "src/common/meta" }
-common-options = { path = "src/common/options" }
 common-plugins = { path = "src/common/plugins" }
-common-pprof = { path = "src/common/pprof" }
 common-procedure = { path = "src/common/procedure" }
 common-procedure-test = { path = "src/common/procedure-test" }
 common-query = { path = "src/common/query" }
@@ -254,9 +212,8 @@ datanode = { path = "src/datanode" }
 datatypes = { path = "src/datatypes" }
 file-engine = { path = "src/file-engine" }
 flow = { path = "src/flow" }
-frontend = { path = "src/frontend", default-features = false }
+frontend = { path = "src/frontend" }
 index = { path = "src/index" }
-log-query = { path = "src/log-query" }
 log-store = { path = "src/log-store" }
 meta-client = { path = "src/meta-client" }
 meta-srv = { path = "src/meta-srv" }
@@ -265,11 +222,11 @@ mito2 = { path = "src/mito2" }
 object-store = { path = "src/object-store" }
 operator = { path = "src/operator" }
 partition = { path = "src/partition" }
-pipeline = { path = "src/pipeline" }
 plugins = { path = "src/plugins" }
 promql = { path = "src/promql" }
 puffin = { path = "src/puffin" }
 query = { path = "src/query" }
+script = { path = "src/script" }
 servers = { path = "src/servers" }
 session = { path = "src/session" }
 sql = { path = "src/sql" }
@@ -277,37 +234,25 @@ store-api = { path = "src/store-api" }
 substrait = { path = "src/common/substrait" }
 table = { path = "src/table" }

-[patch.crates-io]
-# change all rustls dependencies to use our fork to default to `ring` to make it "just work"
-hyper-rustls = { git = "https://github.com/GreptimeTeam/hyper-rustls", rev = "a951e03" } # version = "0.27.5" with ring patch
-rustls = { git = "https://github.com/GreptimeTeam/rustls", rev = "34fd0c6" }             # version = "0.23.20" with ring patch
-tokio-rustls = { git = "https://github.com/GreptimeTeam/tokio-rustls", rev = "4604ca6" } # version = "0.26.0" with ring patch
-# This is commented, since we are not using aws-lc-sys, if we need to use it, we need to uncomment this line or use a release after this commit, or it wouldn't compile with gcc < 8.1
-# see https://github.com/aws/aws-lc-rs/pull/526
-# aws-lc-sys = { git ="https://github.com/aws/aws-lc-rs", rev = "556558441e3494af4b156ae95ebc07ebc2fd38aa" }
-
 [workspace.dependencies.meter-macros]
 git = "https://github.com/GreptimeTeam/greptime-meter.git"
-rev = "5618e779cf2bb4755b499c630fba4c35e91898cb"
+rev = "80b72716dcde47ec4161478416a5c6c21343364d"

 [profile.release]
 debug = 1

 [profile.nightly]
 inherits = "release"
-strip = "debuginfo"
+strip = true
 lto = "thin"
 debug = false
 incremental = false

 [profile.ci]
 inherits = "dev"
+debug = false
 strip = true

 [profile.dev.package.sqlness-runner]
 debug = false
 strip = true
-
-[profile.dev.package.tests-fuzz]
-debug = false
-strip = true
--- a/Cross.toml
+++ b/Cross.toml
@@ -1,6 +1,3 @@
-[target.aarch64-unknown-linux-gnu]
-image = "ghcr.io/cross-rs/aarch64-unknown-linux-gnu:0.2.5"
-
 [build]
 pre-build = [
    "dpkg --add-architecture $CROSS_DEB_ARCH",
@@ -8,8 +5,3 @@ pre-build = [
    "curl -LO https://github.com/protocolbuffers/protobuf/releases/download/v3.15.8/protoc-3.15.8-linux-x86_64.zip && unzip protoc-3.15.8-linux-x86_64.zip -d /usr/",
    "chmod a+x /usr/bin/protoc && chmod -R a+rx /usr/include/google",
 ]
-
-[build.env]
-passthrough = [
-    "JEMALLOC_SYS_WITH_LG_PAGE",
-]
--- a/39
+++ b/39
@@ -8,7 +8,6 @@ CARGO_BUILD_OPTS := --locked
 IMAGE_REGISTRY ?= docker.io
 IMAGE_NAMESPACE ?= greptime
 IMAGE_TAG ?= latest
-DEV_BUILDER_IMAGE_TAG ?= 2024-12-25-a71b93dd-20250305072908
 BUILDX_MULTI_PLATFORM_BUILD ?= false
 BUILDX_BUILDER_NAME ?= gtbuilder
 BASE_IMAGE ?= ubuntu
@@ -16,7 +15,6 @@ RUST_TOOLCHAIN ?= $(shell cat rust-toolchain.toml | grep channel | cut -d'"' -f2
 CARGO_REGISTRY_CACHE ?= ${HOME}/.cargo/registry
 ARCH := $(shell uname -m | sed 's/x86_64/amd64/' | sed 's/aarch64/arm64/')
 OUTPUT_DIR := $(shell if [ "$(RELEASE)" = "true" ]; then echo "release"; elif [ ! -z "$(CARGO_PROFILE)" ]; then echo "$(CARGO_PROFILE)" ; else echo "debug"; fi)
-SQLNESS_OPTS ?=

 # The arguments for running integration tests.
 ETCD_VERSION ?= v3.5.9
@@ -60,8 +58,6 @@ ifeq ($(BUILDX_MULTI_PLATFORM_BUILD), all)
 	BUILDX_MULTI_PLATFORM_BUILD_OPTS := --platform linux/amd64,linux/arm64 --push
 else ifeq ($(BUILDX_MULTI_PLATFORM_BUILD), amd64)
 	BUILDX_MULTI_PLATFORM_BUILD_OPTS := --platform linux/amd64 --push
-else ifeq ($(BUILDX_MULTI_PLATFORM_BUILD), arm64)
-	BUILDX_MULTI_PLATFORM_BUILD_OPTS := --platform linux/arm64 --push
 else
 	BUILDX_MULTI_PLATFORM_BUILD_OPTS := -o type=docker
 endif
@@ -80,7 +76,7 @@ build: ## Build debug version greptime.
 build-by-dev-builder: ## Build greptime by dev-builder.
 	docker run --network=host \
 	-v ${PWD}:/greptimedb -v ${CARGO_REGISTRY_CACHE}:/root/.cargo/registry \
-	-w /greptimedb ${IMAGE_REGISTRY}/${IMAGE_NAMESPACE}/dev-builder-${BASE_IMAGE}:${DEV_BUILDER_IMAGE_TAG} \
+	-w /greptimedb ${IMAGE_REGISTRY}/${IMAGE_NAMESPACE}/dev-builder-${BASE_IMAGE}:latest \
 	make build \
 	CARGO_EXTENSION="${CARGO_EXTENSION}" \
 	CARGO_PROFILE=${CARGO_PROFILE} \
@@ -94,7 +90,7 @@ build-by-dev-builder: ## Build greptime by dev-builder.
 build-android-bin: ## Build greptime binary for android.
 	docker run --network=host \
 	-v ${PWD}:/greptimedb -v ${CARGO_REGISTRY_CACHE}:/root/.cargo/registry \
-	-w /greptimedb ${IMAGE_REGISTRY}/${IMAGE_NAMESPACE}/dev-builder-android:${DEV_BUILDER_IMAGE_TAG} \
+	-w /greptimedb ${IMAGE_REGISTRY}/${IMAGE_NAMESPACE}/dev-builder-android:latest \
 	make build \
 	CARGO_EXTENSION="ndk --platform 23 -t aarch64-linux-android" \
 	CARGO_PROFILE=release \
@@ -108,8 +104,8 @@ build-android-bin: ## Build greptime binary for android.
 strip-android-bin: build-android-bin ## Strip greptime binary for android.
 	docker run --network=host \
 	-v ${PWD}:/greptimedb \
-	-w /greptimedb ${IMAGE_REGISTRY}/${IMAGE_NAMESPACE}/dev-builder-android:${DEV_BUILDER_IMAGE_TAG} \
-	bash -c '$${NDK_ROOT}/toolchains/llvm/prebuilt/linux-x86_64/bin/llvm-strip --strip-debug /greptimedb/target/aarch64-linux-android/release/greptime'
+	-w /greptimedb ${IMAGE_REGISTRY}/${IMAGE_NAMESPACE}/dev-builder-android:latest \
+	bash -c '$${NDK_ROOT}/toolchains/llvm/prebuilt/linux-x86_64/bin/llvm-strip /greptimedb/target/aarch64-linux-android/release/greptime'

 .PHONY: clean
 clean: ## Clean the project.
@@ -148,7 +144,7 @@ dev-builder: multi-platform-buildx ## Build dev-builder image.
 	docker buildx build --builder ${BUILDX_BUILDER_NAME} \
 	--build-arg="RUST_TOOLCHAIN=${RUST_TOOLCHAIN}" \
 	-f docker/dev-builder/${BASE_IMAGE}/Dockerfile \
-	-t ${IMAGE_REGISTRY}/${IMAGE_NAMESPACE}/dev-builder-${BASE_IMAGE}:${DEV_BUILDER_IMAGE_TAG} ${BUILDX_MULTI_PLATFORM_BUILD_OPTS} .
+	-t ${IMAGE_REGISTRY}/${IMAGE_NAMESPACE}/dev-builder-${BASE_IMAGE}:${IMAGE_TAG} ${BUILDX_MULTI_PLATFORM_BUILD_OPTS} .

 .PHONY: multi-platform-buildx
 multi-platform-buildx: ## Create buildx multi-platform builder.
@@ -165,17 +161,7 @@ nextest: ## Install nextest tools.

 .PHONY: sqlness-test
 sqlness-test: ## Run sqlness test.
-	cargo sqlness ${SQLNESS_OPTS}
-
-RUNS ?= 1
-FUZZ_TARGET ?= fuzz_alter_table
-.PHONY: fuzz
-fuzz: ## Run fuzz test ${FUZZ_TARGET}.
-	cargo fuzz run ${FUZZ_TARGET} --fuzz-dir tests-fuzz -D -s none -- -runs=${RUNS}
-
-.PHONY: fuzz-ls
-fuzz-ls: ## List all fuzz targets.
-	cargo fuzz list --fuzz-dir tests-fuzz
+	cargo sqlness

 .PHONY: check
 check: ## Cargo check all the targets.
@@ -192,7 +178,6 @@ fix-clippy: ## Fix clippy violations.
 .PHONY: fmt-check
 fmt-check: ## Check code format.
 	cargo fmt --all -- --check
-	python3 scripts/check-snafu.py

 .PHONY: start-etcd
 start-etcd: ## Start single node etcd for testing purpose.
@@ -206,23 +191,15 @@ stop-etcd: ## Stop single node etcd for testing purpose.
 run-it-in-container: start-etcd ## Run integration tests in dev-builder.
 	docker run --network=host \
 	-v ${PWD}:/greptimedb -v ${CARGO_REGISTRY_CACHE}:/root/.cargo/registry -v /tmp:/tmp \
-	-w /greptimedb ${IMAGE_REGISTRY}/${IMAGE_NAMESPACE}/dev-builder-${BASE_IMAGE}:${DEV_BUILDER_IMAGE_TAG} \
+	-w /greptimedb ${IMAGE_REGISTRY}/${IMAGE_NAMESPACE}/dev-builder-${BASE_IMAGE}:latest \
 	make test sqlness-test BUILD_JOBS=${BUILD_JOBS}

-.PHONY: start-cluster
-start-cluster: ## Start the greptimedb cluster with etcd by using docker compose.
-	 docker compose -f ./docker/docker-compose/cluster-with-etcd.yaml up
-
-.PHONY: stop-cluster
-stop-cluster: ## Stop the greptimedb cluster that created by docker compose.
-	docker compose -f ./docker/docker-compose/cluster-with-etcd.yaml stop
-
 ##@ Docs
 config-docs: ## Generate configuration documentation from toml files.
 	docker run --rm \
    -v ${PWD}:/greptimedb \
    -w /greptimedb/config \
-    toml2docs/toml2docs:v0.1.3 \
+    toml2docs/toml2docs:v0.1.1 \
    -p '##' \
    -t ./config-docs-template.md \
    -o ./config.md
--- a/README.md
+++ b/README.md
@@ -6,14 +6,14 @@
  </picture>
 </p>

-<h2 align="center">Unified & Cost-Effective Time Series Database for Metrics, Logs, and Events</h2>
+<h1 align="center">Cloud-scale, Fast and Efficient Time Series Database</h1>

 <div align="center">
 <h3 align="center">
  <a href="https://greptime.com/product/cloud">GreptimeCloud</a> |
-  <a href="https://docs.greptime.com/">User Guide</a> |
+  <a href="https://docs.greptime.com/">User guide</a> |
  <a href="https://greptimedb.rs/">API Docs</a> |
-  <a href="https://github.com/GreptimeTeam/greptimedb/issues/5446">Roadmap 2025</a>
+  <a href="https://github.com/GreptimeTeam/greptimedb/issues/3412">Roadmap 2024</a>
 </h4>

 <a href="https://github.com/GreptimeTeam/greptimedb/releases/latest">
@@ -48,51 +48,38 @@
 </a>
 </div>

- [Introduction](#introduction)
- [**Features: Why GreptimeDB**](#why-greptimedb)
- [Architecture](https://docs.greptime.com/contributor-guide/overview/#architecture)
- [Try it for free](#try-greptimedb)
- [Getting Started](#getting-started)
- [Project Status](#project-status)
- [Join the community](#community)
-  - [Contributing](#contributing)
- [Tools & Extensions](#tools--extensions)
- [License](#license)
- [Acknowledgement](#acknowledgement)
-
 ## Introduction

-**GreptimeDB** is an open-source unified & cost-effective time-series database for **Metrics**, **Logs**, and **Events** (also **Traces** in plan). You can gain real-time insights from Edge to Cloud at Any Scale.
+**GreptimeDB** is an open-source time-series database focusing on efficiency, scalability, and analytical capabilities.
+Designed to work on infrastructure of the cloud era, GreptimeDB benefits users with its elasticity and commodity storage, offering a fast and cost-effective **alternative to InfluxDB** and a **long-term storage for Prometheus**.

 ## Why GreptimeDB

-Our core developers have been building time-series data platforms for years. Based on our best practices, GreptimeDB was born to give you:
+Our core developers have been building time-series data platforms for years. Based on our best-practices, GreptimeDB is born to give you:

-* **Unified Processing of Metrics, Logs, and Events**
+* **Easy horizontal scaling**

-  GreptimeDB unifies time series data processing by treating all data - whether metrics, logs, or events - as timestamped events with context. Users can analyze this data using either [SQL](https://docs.greptime.com/user-guide/query-data/sql) or [PromQL](https://docs.greptime.com/user-guide/query-data/promql) and leverage stream processing ([Flow](https://docs.greptime.com/user-guide/flow-computation/overview)) to enable continuous aggregation. [Read more](https://docs.greptime.com/user-guide/concepts/data-model).
+  Seamless scalability from a standalone binary at edge to a robust, highly available distributed cluster in cloud, with a transparent experience for both developers and administrators.

-* **Cloud-native Distributed Database**
+* **Analyzing time-series data**

-  Built for [Kubernetes](https://docs.greptime.com/user-guide/deployments/deploy-on-kubernetes/greptimedb-operator-management). GreptimeDB achieves seamless scalability with its [cloud-native architecture](https://docs.greptime.com/user-guide/concepts/architecture) of separated compute and storage, built on object storage (AWS S3, Azure Blob Storage, etc.) while enabling cross-cloud deployment through a unified data access layer.
+  Query your time-series data with SQL and PromQL. Use Python scripts to facilitate complex analytical tasks.
+
+* **Cloud-native distributed database**
+
+  Fully open-source distributed cluster architecture that harnesses the power of cloud-native elastic computing resources.

 * **Performance and Cost-effective**

-  Written in pure Rust for superior performance and reliability. GreptimeDB features a distributed query engine with intelligent indexing to handle high cardinality data efficiently. Its optimized columnar storage achieves 50x cost efficiency on cloud object storage through advanced compression. [Benchmark reports](https://www.greptime.com/blogs/2024-09-09-report-summary).
+  Flexible indexing capabilities and distributed, parallel-processing query engine, tackling high cardinality issues down. Optimized columnar layout for handling time-series data; compacted, compressed, and stored on various storage backends, particularly cloud object storage with 50x cost efficiency.

-* **Cloud-Edge Collaboration**
+* **Compatible with InfluxDB, Prometheus and more protocols**

-  GreptimeDB seamlessly operates across cloud and edge (ARM/Android/Linux), providing consistent APIs and control plane for unified data management and efficient synchronization. [Learn how to run on Android](https://docs.greptime.com/user-guide/deployments/run-on-android/).
-
-* **Multi-protocol Ingestion, SQL & PromQL Ready**
-
-  Widely adopted database protocols and APIs, including MySQL, PostgreSQL, InfluxDB, OpenTelemetry, Loki and Prometheus, etc.  Effortless Adoption & Seamless Migration. [Supported Protocols Overview](https://docs.greptime.com/user-guide/protocols/overview).
-
-For more detailed info please read  [Why GreptimeDB](https://docs.greptime.com/user-guide/concepts/why-greptimedb).
+  Widely adopted database protocols and APIs, including MySQL, PostgreSQL, and Prometheus Remote Storage, etc. [Read more](https://docs.greptime.com/user-guide/clients/overview).

 ## Try GreptimeDB

-### 1. [Live Demo](https://greptime.com/playground)
+### 1. [GreptimePlay](https://greptime.com/playground)

 Try out the features of GreptimeDB right from your browser.

@@ -111,26 +98,17 @@ docker pull greptime/greptimedb
 Start a GreptimeDB container with:

 ```shell
-docker run -p 127.0.0.1:4000-4003:4000-4003 \
-  -v "$(pwd)/greptimedb:/tmp/greptimedb" \
-  --name greptime --rm \
-  greptime/greptimedb:latest standalone start \
-  --http-addr 0.0.0.0:4000 \
-  --rpc-bind-addr 0.0.0.0:4001 \
-  --mysql-addr 0.0.0.0:4002 \
-  --postgres-addr 0.0.0.0:4003
+docker run --rm --name greptime --net=host greptime/greptimedb standalone start
 ```

-Access the dashboard via `http://localhost:4000/dashboard`.
-
 Read more about [Installation](https://docs.greptime.com/getting-started/installation/overview) on docs.

 ## Getting Started

-* [Quickstart](https://docs.greptime.com/getting-started/quick-start)
-* [User Guide](https://docs.greptime.com/user-guide/overview)
-* [Demos](https://github.com/GreptimeTeam/demo-scene)
-* [FAQ](https://docs.greptime.com/faq-and-others/faq)
+* [Quickstart](https://docs.greptime.com/getting-started/quick-start/overview)
+* [Write Data](https://docs.greptime.com/user-guide/clients/overview)
+* [Query Data](https://docs.greptime.com/user-guide/query-data/overview)
+* [Operations](https://docs.greptime.com/user-guide/operations/overview)

 ## Build

@@ -138,8 +116,7 @@ Check the prerequisite:

 * [Rust toolchain](https://www.rust-lang.org/tools/install) (nightly)
 * [Protobuf compiler](https://grpc.io/docs/protoc-installation/) (>= 3.15)
-* C/C++ building essentials, including `gcc`/`g++`/`autoconf` and glibc library (eg. `libc6-dev` on Ubuntu and `glibc-devel` on Fedora)
-* Python toolchain (optional): Required only if using some test scripts.
+* Python toolchain (optional): Required only if built with PyO3 backend. More detail for compiling with PyO3 can be found in its [documentation](https://pyo3.rs/v0.18.1/building_and_distribution#configuring-the-python-version).

 Build GreptimeDB binary:

@@ -153,11 +130,7 @@ Run a standalone server:
 cargo run -- standalone start
 ```

-## Tools & Extensions
-
-### Kubernetes
-
- [GreptimeDB Operator](https://github.com/GrepTimeTeam/greptimedb-operator)
+## Extension

 ### Dashboard

@@ -174,19 +147,13 @@ cargo run -- standalone start

 ### Grafana Dashboard

-Our official Grafana dashboard for monitoring GreptimeDB is available at [grafana](grafana/README.md) directory.
+Our official Grafana dashboard is available at [grafana](grafana/README.md) directory.

 ## Project Status

-GreptimeDB is currently in Beta. We are targeting GA (General Availability) with v1.0 release by Early 2025.
-
-While in Beta, GreptimeDB is already:
-
-* Being used in production by early adopters
-* Actively maintained with regular releases, [about version number](https://docs.greptime.com/nightly/reference/about-greptimedb-version)
-* Suitable for testing and evaluation
-
-For production use, we recommend using the latest stable release.
+The current version has not yet reached General Availability version standards.
+In line with our Greptime 2024 Roadmap, we plan to achieve a production-level
+version with the update to v1.0 in August. [[Join Force]](https://github.com/GreptimeTeam/greptimedb/issues/3412)

 ## Community

@@ -205,13 +172,6 @@ In addition, you may:
 - Connect us with [Linkedin](https://www.linkedin.com/company/greptime/)
 - Follow us on [Twitter](https://twitter.com/greptime)

-## Commercial Support
-
-If you are running GreptimeDB OSS in your organization, we offer additional
-enterprise add-ons, installation services, training, and consulting. [Contact
-us](https://greptime.com/contactus) and we will reach out to you with more
-detail of our commercial license.
-
 ## License

 GreptimeDB uses the [Apache License 2.0](https://apache.org/licenses/LICENSE-2.0.txt) to strike a balance between
@@ -223,9 +183,8 @@ Please refer to [contribution guidelines](CONTRIBUTING.md) and [internal concept

 ## Acknowledgement

-Special thanks to all the contributors who have propelled GreptimeDB forward. For a complete list of contributors, please refer to [AUTHOR.md](AUTHOR.md).
-
 - GreptimeDB uses [Apache Arrow™](https://arrow.apache.org/) as the memory model and [Apache Parquet™](https://parquet.apache.org/) as the persistent file format.
 - GreptimeDB's query engine is powered by [Apache Arrow DataFusion™](https://arrow.apache.org/datafusion/).
 - [Apache OpenDAL™](https://opendal.apache.org) gives GreptimeDB a very general and elegant data access abstraction layer.
 - GreptimeDB's meta service is based on [etcd](https://etcd.io/).
+- GreptimeDB uses [RustPython](https://github.com/RustPython/RustPython) for experimental embedded python scripting.
--- a/benchmarks/Cargo.toml
+++ b/benchmarks/Cargo.toml
@@ -0,0 +1,38 @@
+[package]
+name = "benchmarks"
+version.workspace = true
+edition.workspace = true
+license.workspace = true
+
+[lints]
+workspace = true
+
+[dependencies]
+api.workspace = true
+arrow.workspace = true
+chrono.workspace = true
+clap.workspace = true
+client = { workspace = true, features = ["testing"] }
+common-base.workspace = true
+common-telemetry.workspace = true
+common-wal.workspace = true
+dotenv.workspace = true
+futures.workspace = true
+futures-util.workspace = true
+humantime.workspace = true
+humantime-serde.workspace = true
+indicatif = "0.17.1"
+itertools.workspace = true
+lazy_static.workspace = true
+log-store.workspace = true
+mito2.workspace = true
+num_cpus.workspace = true
+parquet.workspace = true
+prometheus.workspace = true
+rand.workspace = true
+rskafka.workspace = true
+serde.workspace = true
+store-api.workspace = true
+tokio.workspace = true
+toml.workspace = true
+uuid.workspace = true
--- a/benchmarks/README.md
+++ b/benchmarks/README.md
@@ -0,0 +1,11 @@
+Benchmarkers for GreptimeDB
+--------------------------------
+
+## Wal Benchmarker
+The wal benchmarker serves to evaluate the performance of GreptimeDB's Write-Ahead Log (WAL) component. It meticulously assesses the read/write performance of the WAL under diverse workloads generated by the benchmarker. 
+
+
+### How to use
+To compile the benchmarker, navigate to the `greptimedb/benchmarks` directory and execute `cargo build --release`. Subsequently, you'll find the compiled target located at `greptimedb/target/release/wal_bench`.
+
+The `./wal_bench -h` command reveals numerous arguments that the target accepts. Among these, a notable one is the `cfg-file` argument. By utilizing a configuration file in the TOML format, you can bypass the need to repeatedly specify cumbersome arguments.
--- a/benchmarks/config/wal_bench.example.toml
+++ b/benchmarks/config/wal_bench.example.toml
@@ -0,0 +1,21 @@
+# Refers to the documents of `Args` in benchmarks/src/wal.rs`.
+wal_provider = "kafka"
+bootstrap_brokers = ["localhost:9092"]
+num_workers = 10
+num_topics = 32
+num_regions = 1000
+num_scrapes = 1000
+num_rows = 5
+col_types = "ifs"
+max_batch_size = "512KB"
+linger = "1ms"
+backoff_init = "10ms"
+backoff_max = "1ms"
+backoff_base = 2
+backoff_deadline = "3s"
+compression = "zstd"
+rng_seed = 42
+skip_read = false
+skip_write = false
+random_topics = true
+report_metrics = false
--- a/benchmarks/src/bin/wal_bench.rs
+++ b/benchmarks/src/bin/wal_bench.rs
@@ -0,0 +1,326 @@
+// Copyright 2023 Greptime Team
+//
+// Licensed under the Apache License, Version 2.0 (the "License");
+// you may not use this file except in compliance with the License.
+// You may obtain a copy of the License at
+//
+//     http://www.apache.org/licenses/LICENSE-2.0
+//
+// Unless required by applicable law or agreed to in writing, software
+// distributed under the License is distributed on an "AS IS" BASIS,
+// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+// See the License for the specific language governing permissions and
+// limitations under the License.
+
+#![feature(int_roundings)]
+
+use std::fs;
+use std::sync::Arc;
+use std::time::Instant;
+
+use api::v1::{ColumnDataType, ColumnSchema, SemanticType};
+use benchmarks::metrics;
+use benchmarks::wal_bench::{Args, Config, Region, WalProvider};
+use clap::Parser;
+use common_telemetry::info;
+use common_wal::config::kafka::common::BackoffConfig;
+use common_wal::config::kafka::DatanodeKafkaConfig as KafkaConfig;
+use common_wal::config::raft_engine::RaftEngineConfig;
+use common_wal::options::{KafkaWalOptions, WalOptions};
+use itertools::Itertools;
+use log_store::kafka::log_store::KafkaLogStore;
+use log_store::raft_engine::log_store::RaftEngineLogStore;
+use mito2::wal::Wal;
+use prometheus::{Encoder, TextEncoder};
+use rand::distributions::{Alphanumeric, DistString};
+use rand::rngs::SmallRng;
+use rand::SeedableRng;
+use rskafka::client::partition::Compression;
+use rskafka::client::ClientBuilder;
+use store_api::logstore::LogStore;
+use store_api::storage::RegionId;
+
+async fn run_benchmarker<S: LogStore>(cfg: &Config, topics: &[String], wal: Arc<Wal<S>>) {
+    let chunk_size = cfg.num_regions.div_ceil(cfg.num_workers);
+    let region_chunks = (0..cfg.num_regions)
+        .map(|id| {
+            build_region(
+                id as u64,
+                topics,
+                &mut SmallRng::seed_from_u64(cfg.rng_seed),
+                cfg,
+            )
+        })
+        .chunks(chunk_size as usize)
+        .into_iter()
+        .map(|chunk| Arc::new(chunk.collect::<Vec<_>>()))
+        .collect::<Vec<_>>();
+
+    let mut write_elapsed = 0;
+    let mut read_elapsed = 0;
+
+    if !cfg.skip_write {
+        info!("Benchmarking write ...");
+
+        let num_scrapes = cfg.num_scrapes;
+        let timer = Instant::now();
+        futures::future::join_all((0..cfg.num_workers).map(|i| {
+            let wal = wal.clone();
+            let regions = region_chunks[i as usize].clone();
+            tokio::spawn(async move {
+                for _ in 0..num_scrapes {
+                    let mut wal_writer = wal.writer();
+                    regions
+                        .iter()
+                        .for_each(|region| region.add_wal_entry(&mut wal_writer));
+                    wal_writer.write_to_wal().await.unwrap();
+                }
+            })
+        }))
+        .await;
+        write_elapsed += timer.elapsed().as_millis();
+    }
+
+    if !cfg.skip_read {
+        info!("Benchmarking read ...");
+
+        let timer = Instant::now();
+        futures::future::join_all((0..cfg.num_workers).map(|i| {
+            let wal = wal.clone();
+            let regions = region_chunks[i as usize].clone();
+            tokio::spawn(async move {
+                for region in regions.iter() {
+                    region.replay(&wal).await;
+                }
+            })
+        }))
+        .await;
+        read_elapsed = timer.elapsed().as_millis();
+    }
+
+    dump_report(cfg, write_elapsed, read_elapsed);
+}
+
+fn build_region(id: u64, topics: &[String], rng: &mut SmallRng, cfg: &Config) -> Region {
+    let wal_options = match cfg.wal_provider {
+        WalProvider::Kafka => {
+            assert!(!topics.is_empty());
+            WalOptions::Kafka(KafkaWalOptions {
+                topic: topics.get(id as usize % topics.len()).cloned().unwrap(),
+            })
+        }
+        WalProvider::RaftEngine => WalOptions::RaftEngine,
+    };
+    Region::new(
+        RegionId::from_u64(id),
+        build_schema(&parse_col_types(&cfg.col_types), rng),
+        wal_options,
+        cfg.num_rows,
+        cfg.rng_seed,
+    )
+}
+
+fn build_schema(col_types: &[ColumnDataType], mut rng: &mut SmallRng) -> Vec<ColumnSchema> {
+    col_types
+        .iter()
+        .map(|col_type| ColumnSchema {
+            column_name: Alphanumeric.sample_string(&mut rng, 5),
+            datatype: *col_type as i32,
+            semantic_type: SemanticType::Field as i32,
+            datatype_extension: None,
+        })
+        .chain(vec![ColumnSchema {
+            column_name: "ts".to_string(),
+            datatype: ColumnDataType::TimestampMillisecond as i32,
+            semantic_type: SemanticType::Tag as i32,
+            datatype_extension: None,
+        }])
+        .collect()
+}
+
+fn dump_report(cfg: &Config, write_elapsed: u128, read_elapsed: u128) {
+    let cost_report = format!(
+        "write costs: {} ms, read costs: {} ms",
+        write_elapsed, read_elapsed,
+    );
+
+    let total_written_bytes = metrics::METRIC_WAL_WRITE_BYTES_TOTAL.get() as u128;
+    let write_throughput = if write_elapsed > 0 {
+        (total_written_bytes * 1000).div_floor(write_elapsed)
+    } else {
+        0
+    };
+    let total_read_bytes = metrics::METRIC_WAL_READ_BYTES_TOTAL.get() as u128;
+    let read_throughput = if read_elapsed > 0 {
+        (total_read_bytes * 1000).div_floor(read_elapsed)
+    } else {
+        0
+    };
+
+    let throughput_report = format!(
+        "total written bytes: {} bytes, total read bytes: {} bytes, write throuput: {} bytes/s ({} mb/s), read throughput: {} bytes/s ({} mb/s)",
+        total_written_bytes,
+        total_read_bytes,
+        write_throughput,
+        write_throughput.div_floor(1 << 20),
+        read_throughput,
+        read_throughput.div_floor(1 << 20),
+    );
+
+    let metrics_report = if cfg.report_metrics {
+        let mut buffer = Vec::new();
+        let encoder = TextEncoder::new();
+        let metrics = prometheus::gather();
+        encoder.encode(&metrics, &mut buffer).unwrap();
+        String::from_utf8(buffer).unwrap()
+    } else {
+        String::new()
+    };
+
+    info!(
+        r#"
+Benchmark config: 
+{cfg:?}
+
+Benchmark report:
+{cost_report}
+{throughput_report}
+{metrics_report}"#
+    );
+}
+
+async fn create_topics(cfg: &Config) -> Vec<String> {
+    // Creates topics.
+    let client = ClientBuilder::new(cfg.bootstrap_brokers.clone())
+        .build()
+        .await
+        .unwrap();
+    let ctrl_client = client.controller_client().unwrap();
+    let (topics, tasks): (Vec<_>, Vec<_>) = (0..cfg.num_topics)
+        .map(|i| {
+            let topic = if cfg.random_topics {
+                format!(
+                    "greptime_wal_bench_topic_{}_{}",
+                    uuid::Uuid::new_v4().as_u128(),
+                    i
+                )
+            } else {
+                format!("greptime_wal_bench_topic_{}", i)
+            };
+            let task = ctrl_client.create_topic(
+                topic.clone(),
+                1,
+                cfg.bootstrap_brokers.len() as i16,
+                2000,
+            );
+            (topic, task)
+        })
+        .unzip();
+    // Must ignore errors since we allow topics being created more than once.
+    let _ = futures::future::try_join_all(tasks).await;
+
+    topics
+}
+
+fn parse_compression(comp: &str) -> Compression {
+    match comp {
+        "no" => Compression::NoCompression,
+        "gzip" => Compression::Gzip,
+        "lz4" => Compression::Lz4,
+        "snappy" => Compression::Snappy,
+        "zstd" => Compression::Zstd,
+        other => unreachable!("Unrecognized compression {other}"),
+    }
+}
+
+fn parse_col_types(col_types: &str) -> Vec<ColumnDataType> {
+    let parts = col_types.split('x').collect::<Vec<_>>();
+    assert!(parts.len() <= 2);
+
+    let pattern = parts[0];
+    let repeat = parts
+        .get(1)
+        .map(|r| r.parse::<usize>().unwrap())
+        .unwrap_or(1);
+
+    pattern
+        .chars()
+        .map(|c| match c {
+            'i' | 'I' => ColumnDataType::Int64,
+            'f' | 'F' => ColumnDataType::Float64,
+            's' | 'S' => ColumnDataType::String,
+            other => unreachable!("Cannot parse {other} as a column data type"),
+        })
+        .cycle()
+        .take(pattern.len() * repeat)
+        .collect()
+}
+
+fn main() {
+    // Sets the global logging to INFO and suppress loggings from rskafka other than ERROR and upper ones.
+    std::env::set_var("UNITTEST_LOG_LEVEL", "info,rskafka=error");
+    common_telemetry::init_default_ut_logging();
+
+    let args = Args::parse();
+    let cfg = if !args.cfg_file.is_empty() {
+        toml::from_str(&fs::read_to_string(&args.cfg_file).unwrap()).unwrap()
+    } else {
+        Config::from(args)
+    };
+
+    // Validates arguments.
+    if cfg.num_regions < cfg.num_workers {
+        panic!("num_regions must be greater than or equal to num_workers");
+    }
+    if cfg
+        .num_workers
+        .min(cfg.num_topics)
+        .min(cfg.num_regions)
+        .min(cfg.num_scrapes)
+        .min(cfg.max_batch_size.as_bytes() as u32)
+        .min(cfg.bootstrap_brokers.len() as u32)
+        == 0
+    {
+        panic!("Invalid arguments");
+    }
+
+    tokio::runtime::Builder::new_multi_thread()
+        .enable_all()
+        .build()
+        .unwrap()
+        .block_on(async {
+            match cfg.wal_provider {
+                WalProvider::Kafka => {
+                    let topics = create_topics(&cfg).await;
+                    let kafka_cfg = KafkaConfig {
+                        broker_endpoints: cfg.bootstrap_brokers.clone(),
+                        max_batch_size: cfg.max_batch_size,
+                        linger: cfg.linger,
+                        backoff: BackoffConfig {
+                            init: cfg.backoff_init,
+                            max: cfg.backoff_max,
+                            base: cfg.backoff_base,
+                            deadline: Some(cfg.backoff_deadline),
+                        },
+                        compression: parse_compression(&cfg.compression),
+                        ..Default::default()
+                    };
+                    let store = Arc::new(KafkaLogStore::try_new(&kafka_cfg).await.unwrap());
+                    let wal = Arc::new(Wal::new(store));
+                    run_benchmarker(&cfg, &topics, wal).await;
+                }
+                WalProvider::RaftEngine => {
+                    // The benchmarker assumes the raft engine directory exists.
+                    let store = RaftEngineLogStore::try_new(
+                        "/tmp/greptimedb/raft-engine-wal".to_string(),
+                        RaftEngineConfig::default(),
+                    )
+                    .await
+                    .map(Arc::new)
+                    .unwrap();
+                    let wal = Arc::new(Wal::new(store));
+                    run_benchmarker(&cfg, &[], wal).await;
+                }
+            }
+        });
+}
--- a/src/common/options/src/lib.rs
+++ b/src/common/options/src/lib.rs
@@ -12,4 +12,5 @@
 // See the License for the specific language governing permissions and
 // limitations under the License.

-pub mod datanode;
+pub mod metrics;
+pub mod wal_bench;
--- a/benchmarks/src/metrics.rs
+++ b/benchmarks/src/metrics.rs
@@ -0,0 +1,39 @@
+// Copyright 2023 Greptime Team
+//
+// Licensed under the Apache License, Version 2.0 (the "License");
+// you may not use this file except in compliance with the License.
+// You may obtain a copy of the License at
+//
+//     http://www.apache.org/licenses/LICENSE-2.0
+//
+// Unless required by applicable law or agreed to in writing, software
+// distributed under the License is distributed on an "AS IS" BASIS,
+// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+// See the License for the specific language governing permissions and
+// limitations under the License.
+
+use lazy_static::lazy_static;
+use prometheus::*;
+
+/// Logstore label.
+pub const LOGSTORE_LABEL: &str = "logstore";
+/// Operation type label.
+pub const OPTYPE_LABEL: &str = "optype";
+
+lazy_static! {
+    /// Counters of bytes of each operation on a logstore.
+    pub static ref METRIC_WAL_OP_BYTES_TOTAL: IntCounterVec = register_int_counter_vec!(
+        "greptime_bench_wal_op_bytes_total",
+        "wal operation bytes total",
+        &[OPTYPE_LABEL],
+    )
+    .unwrap();
+    /// Counter of bytes of the append_batch operation.
+    pub static ref METRIC_WAL_WRITE_BYTES_TOTAL: IntCounter = METRIC_WAL_OP_BYTES_TOTAL.with_label_values(
+        &["write"],
+    );
+    /// Counter of bytes of the read operation.
+    pub static ref METRIC_WAL_READ_BYTES_TOTAL: IntCounter = METRIC_WAL_OP_BYTES_TOTAL.with_label_values(
+        &["read"],
+    );
+}
--- a/benchmarks/src/wal_bench.rs
+++ b/benchmarks/src/wal_bench.rs
@@ -0,0 +1,366 @@
+// Copyright 2023 Greptime Team
+//
+// Licensed under the Apache License, Version 2.0 (the "License");
+// you may not use this file except in compliance with the License.
+// You may obtain a copy of the License at
+//
+//     http://www.apache.org/licenses/LICENSE-2.0
+//
+// Unless required by applicable law or agreed to in writing, software
+// distributed under the License is distributed on an "AS IS" BASIS,
+// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+// See the License for the specific language governing permissions and
+// limitations under the License.
+
+use std::mem::size_of;
+use std::sync::atomic::{AtomicI64, AtomicU64, Ordering};
+use std::sync::{Arc, Mutex};
+use std::time::Duration;
+
+use api::v1::value::ValueData;
+use api::v1::{ColumnDataType, ColumnSchema, Mutation, OpType, Row, Rows, Value, WalEntry};
+use clap::{Parser, ValueEnum};
+use common_base::readable_size::ReadableSize;
+use common_wal::options::WalOptions;
+use futures::StreamExt;
+use mito2::wal::{Wal, WalWriter};
+use rand::distributions::{Alphanumeric, DistString, Uniform};
+use rand::rngs::SmallRng;
+use rand::{Rng, SeedableRng};
+use serde::{Deserialize, Serialize};
+use store_api::logstore::provider::Provider;
+use store_api::logstore::LogStore;
+use store_api::storage::RegionId;
+
+use crate::metrics;
+
+/// The wal provider.
+#[derive(Clone, ValueEnum, Default, Debug, PartialEq, Serialize, Deserialize)]
+#[serde(rename_all = "snake_case")]
+pub enum WalProvider {
+    #[default]
+    RaftEngine,
+    Kafka,
+}
+
+#[derive(Parser)]
+pub struct Args {
+    /// The provided configuration file.
+    /// The example configuration file can be found at `greptimedb/benchmarks/config/wal_bench.example.toml`.
+    #[clap(long, short = 'c')]
+    pub cfg_file: String,
+
+    /// The wal provider.
+    #[clap(long, value_enum, default_value_t = WalProvider::default())]
+    pub wal_provider: WalProvider,
+
+    /// The advertised addresses of the kafka brokers.
+    /// If there're multiple bootstrap brokers, their addresses should be separated by comma, for e.g. "localhost:9092,localhost:9093".
+    #[clap(long, short = 'b', default_value = "localhost:9092")]
+    pub bootstrap_brokers: String,
+
+    /// The number of workers each running in a dedicated thread.
+    #[clap(long, default_value_t = num_cpus::get() as u32)]
+    pub num_workers: u32,
+
+    /// The number of kafka topics to be created.
+    #[clap(long, default_value_t = 32)]
+    pub num_topics: u32,
+
+    /// The number of regions.
+    #[clap(long, default_value_t = 1000)]
+    pub num_regions: u32,
+
+    /// The number of times each region is scraped.
+    #[clap(long, default_value_t = 1000)]
+    pub num_scrapes: u32,
+
+    /// The number of rows in each wal entry.
+    /// Each time a region is scraped, a wal entry containing will be produced.
+    #[clap(long, default_value_t = 5)]
+    pub num_rows: u32,
+
+    /// The column types of the schema for each region.
+    /// Currently, three column types are supported:
+    /// - i = ColumnDataType::Int64
+    /// - f = ColumnDataType::Float64
+    /// - s = ColumnDataType::String  
+    /// For e.g., "ifs" will be parsed as three columns: i64, f64, and string.
+    ///
+    /// Additionally, a "x" sign can be provided to repeat the column types for a given number of times.
+    /// For e.g., "iix2" will be parsed as 4 columns: i64, i64, i64, and i64.
+    /// This feature is useful if you want to specify many columns.
+    #[clap(long, default_value = "ifs")]
+    pub col_types: String,
+
+    /// The maximum size of a batch of kafka records.
+    /// The default value is 1mb.
+    #[clap(long, default_value = "512KB")]
+    pub max_batch_size: ReadableSize,
+
+    /// The minimum latency the kafka client issues a batch of kafka records.
+    /// However, a batch of kafka records would be immediately issued if a record cannot be fit into the batch.
+    #[clap(long, default_value = "1ms")]
+    pub linger: String,
+
+    /// The initial backoff delay of the kafka consumer.
+    #[clap(long, default_value = "10ms")]
+    pub backoff_init: String,
+
+    /// The maximum backoff delay of the kafka consumer.
+    #[clap(long, default_value = "1s")]
+    pub backoff_max: String,
+
+    /// The exponential backoff rate of the kafka consumer. The next back off = base * the current backoff.
+    #[clap(long, default_value_t = 2)]
+    pub backoff_base: u32,
+
+    /// The deadline of backoff. The backoff ends if the total backoff delay reaches the deadline.
+    #[clap(long, default_value = "3s")]
+    pub backoff_deadline: String,
+
+    /// The client-side compression algorithm for kafka records.
+    #[clap(long, default_value = "zstd")]
+    pub compression: String,
+
+    /// The seed of random number generators.
+    #[clap(long, default_value_t = 42)]
+    pub rng_seed: u64,
+
+    /// Skips the read phase, aka. region replay, if set to true.
+    #[clap(long, default_value_t = false)]
+    pub skip_read: bool,
+
+    /// Skips the write phase if set to true.
+    #[clap(long, default_value_t = false)]
+    pub skip_write: bool,
+
+    /// Randomly generates topic names if set to true.
+    /// Useful when you want to run the benchmarker without worrying about the topics created before.
+    #[clap(long, default_value_t = false)]
+    pub random_topics: bool,
+
+    /// Logs out the gathered prometheus metrics when the benchmarker ends.
+    #[clap(long, default_value_t = false)]
+    pub report_metrics: bool,
+}
+
+/// Benchmarker config.
+#[derive(Debug, Clone, Serialize, Deserialize)]
+pub struct Config {
+    pub wal_provider: WalProvider,
+    pub bootstrap_brokers: Vec<String>,
+    pub num_workers: u32,
+    pub num_topics: u32,
+    pub num_regions: u32,
+    pub num_scrapes: u32,
+    pub num_rows: u32,
+    pub col_types: String,
+    pub max_batch_size: ReadableSize,
+    #[serde(with = "humantime_serde")]
+    pub linger: Duration,
+    #[serde(with = "humantime_serde")]
+    pub backoff_init: Duration,
+    #[serde(with = "humantime_serde")]
+    pub backoff_max: Duration,
+    pub backoff_base: u32,
+    #[serde(with = "humantime_serde")]
+    pub backoff_deadline: Duration,
+    pub compression: String,
+    pub rng_seed: u64,
+    pub skip_read: bool,
+    pub skip_write: bool,
+    pub random_topics: bool,
+    pub report_metrics: bool,
+}
+
+impl From<Args> for Config {
+    fn from(args: Args) -> Self {
+        let cfg = Self {
+            wal_provider: args.wal_provider,
+            bootstrap_brokers: args
+                .bootstrap_brokers
+                .split(',')
+                .map(ToString::to_string)
+                .collect::<Vec<_>>(),
+            num_workers: args.num_workers.min(num_cpus::get() as u32),
+            num_topics: args.num_topics,
+            num_regions: args.num_regions,
+            num_scrapes: args.num_scrapes,
+            num_rows: args.num_rows,
+            col_types: args.col_types,
+            max_batch_size: args.max_batch_size,
+            linger: humantime::parse_duration(&args.linger).unwrap(),
+            backoff_init: humantime::parse_duration(&args.backoff_init).unwrap(),
+            backoff_max: humantime::parse_duration(&args.backoff_max).unwrap(),
+            backoff_base: args.backoff_base,
+            backoff_deadline: humantime::parse_duration(&args.backoff_deadline).unwrap(),
+            compression: args.compression,
+            rng_seed: args.rng_seed,
+            skip_read: args.skip_read,
+            skip_write: args.skip_write,
+            random_topics: args.random_topics,
+            report_metrics: args.report_metrics,
+        };
+
+        cfg
+    }
+}
+
+/// The region used for wal benchmarker.
+pub struct Region {
+    id: RegionId,
+    schema: Vec<ColumnSchema>,
+    provider: Provider,
+    next_sequence: AtomicU64,
+    next_entry_id: AtomicU64,
+    next_timestamp: AtomicI64,
+    rng: Mutex<Option<SmallRng>>,
+    num_rows: u32,
+}
+
+impl Region {
+    /// Creates a new region.
+    pub fn new(
+        id: RegionId,
+        schema: Vec<ColumnSchema>,
+        wal_options: WalOptions,
+        num_rows: u32,
+        rng_seed: u64,
+    ) -> Self {
+        let provider = match wal_options {
+            WalOptions::RaftEngine => Provider::raft_engine_provider(id.as_u64()),
+            WalOptions::Kafka(opts) => Provider::kafka_provider(opts.topic),
+        };
+        Self {
+            id,
+            schema,
+            provider,
+            next_sequence: AtomicU64::new(1),
+            next_entry_id: AtomicU64::new(1),
+            next_timestamp: AtomicI64::new(1655276557000),
+            rng: Mutex::new(Some(SmallRng::seed_from_u64(rng_seed))),
+            num_rows,
+        }
+    }
+
+    /// Scrapes the region and adds the generated entry to wal.
+    pub fn add_wal_entry<S: LogStore>(&self, wal_writer: &mut WalWriter<S>) {
+        let mutation = Mutation {
+            op_type: OpType::Put as i32,
+            sequence: self
+                .next_sequence
+                .fetch_add(self.num_rows as u64, Ordering::Relaxed),
+            rows: Some(self.build_rows()),
+        };
+        let entry = WalEntry {
+            mutations: vec![mutation],
+        };
+        metrics::METRIC_WAL_WRITE_BYTES_TOTAL.inc_by(Self::entry_estimated_size(&entry) as u64);
+
+        wal_writer
+            .add_entry(
+                self.id,
+                self.next_entry_id.fetch_add(1, Ordering::Relaxed),
+                &entry,
+                &self.provider,
+            )
+            .unwrap();
+    }
+
+    /// Replays the region.
+    pub async fn replay<S: LogStore>(&self, wal: &Arc<Wal<S>>) {
+        let mut wal_stream = wal.scan(self.id, 0, &self.provider).unwrap();
+        while let Some(res) = wal_stream.next().await {
+            let (_, entry) = res.unwrap();
+            metrics::METRIC_WAL_READ_BYTES_TOTAL.inc_by(Self::entry_estimated_size(&entry) as u64);
+        }
+    }
+
+    /// Computes the estimated size in bytes of the entry.
+    pub fn entry_estimated_size(entry: &WalEntry) -> usize {
+        let wrapper_size = size_of::<WalEntry>()
+            + entry.mutations.capacity() * size_of::<Mutation>()
+            + size_of::<Rows>();
+
+        let rows = entry.mutations[0].rows.as_ref().unwrap();
+
+        let schema_size = rows.schema.capacity() * size_of::<ColumnSchema>()
+            + rows
+                .schema
+                .iter()
+                .map(|s| s.column_name.capacity())
+                .sum::<usize>();
+        let values_size = (rows.rows.capacity() * size_of::<Row>())
+            + rows
+                .rows
+                .iter()
+                .map(|r| r.values.capacity() * size_of::<Value>())
+                .sum::<usize>();
+
+        wrapper_size + schema_size + values_size
+    }
+
+    fn build_rows(&self) -> Rows {
+        let cols = self
+            .schema
+            .iter()
+            .map(|col_schema| {
+                let col_data_type = ColumnDataType::try_from(col_schema.datatype).unwrap();
+                self.build_col(&col_data_type, self.num_rows)
+            })
+            .collect::<Vec<_>>();
+
+        let rows = (0..self.num_rows)
+            .map(|i| {
+                let values = cols.iter().map(|col| col[i as usize].clone()).collect();
+                Row { values }
+            })
+            .collect();
+
+        Rows {
+            schema: self.schema.clone(),
+            rows,
+        }
+    }
+
+    fn build_col(&self, col_data_type: &ColumnDataType, num_rows: u32) -> Vec<Value> {
+        let mut rng_guard = self.rng.lock().unwrap();
+        let rng = rng_guard.as_mut().unwrap();
+        match col_data_type {
+            ColumnDataType::TimestampMillisecond => (0..num_rows)
+                .map(|_| {
+                    let ts = self.next_timestamp.fetch_add(1000, Ordering::Relaxed);
+                    Value {
+                        value_data: Some(ValueData::TimestampMillisecondValue(ts)),
+                    }
+                })
+                .collect(),
+            ColumnDataType::Int64 => (0..num_rows)
+                .map(|_| {
+                    let v = rng.sample(Uniform::new(0, 10_000));
+                    Value {
+                        value_data: Some(ValueData::I64Value(v)),
+                    }
+                })
+                .collect(),
+            ColumnDataType::Float64 => (0..num_rows)
+                .map(|_| {
+                    let v = rng.sample(Uniform::new(0.0, 5000.0));
+                    Value {
+                        value_data: Some(ValueData::F64Value(v)),
+                    }
+                })
+                .collect(),
+            ColumnDataType::String => (0..num_rows)
+                .map(|_| {
+                    let v = Alphanumeric.sample_string(rng, 10);
+                    Value {
+                        value_data: Some(ValueData::StringValue(v)),
+                    }
+                })
+                .collect(),
+            _ => unreachable!(),
+        }
+    }
+}
--- a/config/config-docs-template.md
+++ b/config/config-docs-template.md
@@ -1,12 +1,10 @@
 # Configurations

- [Configurations](#configurations)
-  - [Standalone Mode](#standalone-mode)
-  - [Distributed Mode](#distributed-mode)
+- [Standalone Mode](#standalone-mode)
+- [Distributed Mode](#distributed-mode)
    - [Frontend](#frontend)
    - [Metasrv](#metasrv)
    - [Datanode](#datanode)
-    - [Flownode](#flownode)

 ## Standalone Mode

@@ -25,7 +23,3 @@
 ### Datanode

 {{ toml2docs "./datanode.example.toml" }}
-
-### Flownode
-
-{{ toml2docs "./flownode.example.toml"}}
--- a/config/config.md
+++ b/config/config.md
@@ -1,129 +1,98 @@
 # Configurations

- [Configurations](#configurations)
-  - [Standalone Mode](#standalone-mode)
-  - [Distributed Mode](#distributed-mode)
+- [Standalone Mode](#standalone-mode)
+- [Distributed Mode](#distributed-mode)
    - [Frontend](#frontend)
    - [Metasrv](#metasrv)
    - [Datanode](#datanode)
-    - [Flownode](#flownode)

 ## Standalone Mode

 | Key | Type | Default | Descriptions |
 | --- | -----| ------- | ----------- |
 | `mode` | String | `standalone` | The running mode of the datanode. It can be `standalone` or `distributed`. |
-| `default_timezone` | String | Unset | The default timezone of the server. |
-| `init_regions_in_background` | Bool | `false` | Initialize all regions in the background during the startup.<br/>By default, it provides services after all regions have been initialized. |
-| `init_regions_parallelism` | Integer | `16` | Parallelism of initializing regions. |
-| `max_concurrent_queries` | Integer | `0` | The maximum current queries allowed to be executed. Zero means unlimited. |
-| `enable_telemetry` | Bool | `true` | Enable telemetry to collect anonymous usage data. Enabled by default. |
-| `max_in_flight_write_bytes` | String | Unset | The maximum in-flight write bytes. |
-| `runtime` | -- | -- | The runtime options. |
-| `runtime.global_rt_size` | Integer | `8` | The number of threads to execute the runtime for global read operations. |
-| `runtime.compact_rt_size` | Integer | `4` | The number of threads to execute the runtime for global write operations. |
+| `enable_telemetry` | Bool | `true` | Enable telemetry to collect anonymous usage data. |
+| `default_timezone` | String | `None` | The default timezone of the server. |
 | `http` | -- | -- | The HTTP server options. |
 | `http.addr` | String | `127.0.0.1:4000` | The address to bind the HTTP server. |
-| `http.timeout` | String | `30s` | HTTP request timeout. Set to 0 to disable timeout. |
-| `http.body_limit` | String | `64MB` | HTTP request body limit.<br/>The following units are supported: `B`, `KB`, `KiB`, `MB`, `MiB`, `GB`, `GiB`, `TB`, `TiB`, `PB`, `PiB`.<br/>Set to 0 to disable limit. |
-| `http.enable_cors` | Bool | `true` | HTTP CORS support, it's turned on by default<br/>This allows browser to access http APIs without CORS restrictions |
-| `http.cors_allowed_origins` | Array | Unset | Customize allowed origins for HTTP CORS. |
+| `http.timeout` | String | `30s` | HTTP request timeout. |
+| `http.body_limit` | String | `64MB` | HTTP request body limit.<br/>Support the following units are supported: `B`, `KB`, `KiB`, `MB`, `MiB`, `GB`, `GiB`, `TB`, `TiB`, `PB`, `PiB`. |
 | `grpc` | -- | -- | The gRPC server options. |
-| `grpc.bind_addr` | String | `127.0.0.1:4001` | The address to bind the gRPC server. |
+| `grpc.addr` | String | `127.0.0.1:4001` | The address to bind the gRPC server. |
 | `grpc.runtime_size` | Integer | `8` | The number of server worker threads. |
 | `grpc.tls` | -- | -- | gRPC server TLS options, see `mysql.tls` section. |
 | `grpc.tls.mode` | String | `disable` | TLS mode. |
-| `grpc.tls.cert_path` | String | Unset | Certificate file path. |
-| `grpc.tls.key_path` | String | Unset | Private key file path. |
+| `grpc.tls.cert_path` | String | `None` | Certificate file path. |
+| `grpc.tls.key_path` | String | `None` | Private key file path. |
 | `grpc.tls.watch` | Bool | `false` | Watch for Certificate and key file change and auto reload.<br/>For now, gRPC tls config does not support auto reload. |
 | `mysql` | -- | -- | MySQL server options. |
 | `mysql.enable` | Bool | `true` | Whether to enable. |
 | `mysql.addr` | String | `127.0.0.1:4002` | The addr to bind the MySQL server. |
 | `mysql.runtime_size` | Integer | `2` | The number of server worker threads. |
-| `mysql.keep_alive` | String | `0s` | Server-side keep-alive time.<br/>Set to 0 (default) to disable. |
 | `mysql.tls` | -- | -- | -- |
 | `mysql.tls.mode` | String | `disable` | TLS mode, refer to https://www.postgresql.org/docs/current/libpq-ssl.html<br/>- `disable` (default value)<br/>- `prefer`<br/>- `require`<br/>- `verify-ca`<br/>- `verify-full` |
-| `mysql.tls.cert_path` | String | Unset | Certificate file path. |
-| `mysql.tls.key_path` | String | Unset | Private key file path. |
+| `mysql.tls.cert_path` | String | `None` | Certificate file path. |
+| `mysql.tls.key_path` | String | `None` | Private key file path. |
 | `mysql.tls.watch` | Bool | `false` | Watch for Certificate and key file change and auto reload |
 | `postgres` | -- | -- | PostgresSQL server options. |
 | `postgres.enable` | Bool | `true` | Whether to enable |
 | `postgres.addr` | String | `127.0.0.1:4003` | The addr to bind the PostgresSQL server. |
 | `postgres.runtime_size` | Integer | `2` | The number of server worker threads. |
-| `postgres.keep_alive` | String | `0s` | Server-side keep-alive time.<br/>Set to 0 (default) to disable. |
 | `postgres.tls` | -- | -- | PostgresSQL server TLS options, see `mysql.tls` section. |
 | `postgres.tls.mode` | String | `disable` | TLS mode. |
-| `postgres.tls.cert_path` | String | Unset | Certificate file path. |
-| `postgres.tls.key_path` | String | Unset | Private key file path. |
+| `postgres.tls.cert_path` | String | `None` | Certificate file path. |
+| `postgres.tls.key_path` | String | `None` | Private key file path. |
 | `postgres.tls.watch` | Bool | `false` | Watch for Certificate and key file change and auto reload |
 | `opentsdb` | -- | -- | OpenTSDB protocol options. |
 | `opentsdb.enable` | Bool | `true` | Whether to enable OpenTSDB put in HTTP API. |
 | `influxdb` | -- | -- | InfluxDB protocol options. |
 | `influxdb.enable` | Bool | `true` | Whether to enable InfluxDB protocol in HTTP API. |
-| `jaeger` | -- | -- | Jaeger protocol options. |
-| `jaeger.enable` | Bool | `true` | Whether to enable Jaeger protocol in HTTP API. |
 | `prom_store` | -- | -- | Prometheus remote storage options |
 | `prom_store.enable` | Bool | `true` | Whether to enable Prometheus remote write and read in HTTP API. |
 | `prom_store.with_metric_engine` | Bool | `true` | Whether to store the data from Prometheus remote write in metric engine. |
 | `wal` | -- | -- | The WAL options. |
 | `wal.provider` | String | `raft_engine` | The provider of the WAL.<br/>- `raft_engine`: the wal is stored in the local file system by raft-engine.<br/>- `kafka`: it's remote wal that data is stored in Kafka. |
-| `wal.dir` | String | Unset | The directory to store the WAL files.<br/>**It's only used when the provider is `raft_engine`**. |
-| `wal.file_size` | String | `128MB` | The size of the WAL segment file.<br/>**It's only used when the provider is `raft_engine`**. |
-| `wal.purge_threshold` | String | `1GB` | The threshold of the WAL size to trigger a purge.<br/>**It's only used when the provider is `raft_engine`**. |
-| `wal.purge_interval` | String | `1m` | The interval to trigger a purge.<br/>**It's only used when the provider is `raft_engine`**. |
+| `wal.dir` | String | `None` | The directory to store the WAL files.<br/>**It's only used when the provider is `raft_engine`**. |
+| `wal.file_size` | String | `256MB` | The size of the WAL segment file.<br/>**It's only used when the provider is `raft_engine`**. |
+| `wal.purge_threshold` | String | `4GB` | The threshold of the WAL size to trigger a flush.<br/>**It's only used when the provider is `raft_engine`**. |
+| `wal.purge_interval` | String | `10m` | The interval to trigger a flush.<br/>**It's only used when the provider is `raft_engine`**. |
 | `wal.read_batch_size` | Integer | `128` | The read batch size.<br/>**It's only used when the provider is `raft_engine`**. |
 | `wal.sync_write` | Bool | `false` | Whether to use sync write.<br/>**It's only used when the provider is `raft_engine`**. |
 | `wal.enable_log_recycle` | Bool | `true` | Whether to reuse logically truncated log files.<br/>**It's only used when the provider is `raft_engine`**. |
 | `wal.prefill_log_files` | Bool | `false` | Whether to pre-create log files on start up.<br/>**It's only used when the provider is `raft_engine`**. |
 | `wal.sync_period` | String | `10s` | Duration for fsyncing log files.<br/>**It's only used when the provider is `raft_engine`**. |
-| `wal.recovery_parallelism` | Integer | `2` | Parallelism during WAL recovery. |
 | `wal.broker_endpoints` | Array | -- | The Kafka broker endpoints.<br/>**It's only used when the provider is `kafka`**. |
-| `wal.auto_create_topics` | Bool | `true` | Automatically create topics for WAL.<br/>Set to `true` to automatically create topics for WAL.<br/>Otherwise, use topics named `topic_name_prefix_[0..num_topics)` |
-| `wal.num_topics` | Integer | `64` | Number of topics.<br/>**It's only used when the provider is `kafka`**. |
-| `wal.selector_type` | String | `round_robin` | Topic selector type.<br/>Available selector types:<br/>- `round_robin` (default)<br/>**It's only used when the provider is `kafka`**. |
-| `wal.topic_name_prefix` | String | `greptimedb_wal_topic` | A Kafka topic is constructed by concatenating `topic_name_prefix` and `topic_id`.<br/>i.g., greptimedb_wal_topic_0, greptimedb_wal_topic_1.<br/>**It's only used when the provider is `kafka`**. |
-| `wal.replication_factor` | Integer | `1` | Expected number of replicas of each partition.<br/>**It's only used when the provider is `kafka`**. |
-| `wal.create_topic_timeout` | String | `30s` | Above which a topic creation operation will be cancelled.<br/>**It's only used when the provider is `kafka`**. |
-| `wal.max_batch_bytes` | String | `1MB` | The max size of a single producer batch.<br/>Warning: Kafka has a default limit of 1MB per message in a topic.<br/>**It's only used when the provider is `kafka`**. |
+| `wal.max_batch_size` | String | `1MB` | The max size of a single producer batch.<br/>Warning: Kafka has a default limit of 1MB per message in a topic.<br/>**It's only used when the provider is `kafka`**. |
+| `wal.linger` | String | `200ms` | The linger duration of a kafka batch producer.<br/>**It's only used when the provider is `kafka`**. |
 | `wal.consumer_wait_timeout` | String | `100ms` | The consumer wait timeout.<br/>**It's only used when the provider is `kafka`**. |
 | `wal.backoff_init` | String | `500ms` | The initial backoff delay.<br/>**It's only used when the provider is `kafka`**. |
 | `wal.backoff_max` | String | `10s` | The maximum backoff delay.<br/>**It's only used when the provider is `kafka`**. |
 | `wal.backoff_base` | Integer | `2` | The exponential backoff rate, i.e. next backoff = base * current backoff.<br/>**It's only used when the provider is `kafka`**. |
 | `wal.backoff_deadline` | String | `5mins` | The deadline of retries.<br/>**It's only used when the provider is `kafka`**. |
-| `wal.overwrite_entry_start_id` | Bool | `false` | Ignore missing entries during read WAL.<br/>**It's only used when the provider is `kafka`**.<br/><br/>This option ensures that when Kafka messages are deleted, the system<br/>can still successfully replay memtable data without throwing an<br/>out-of-range error.<br/>However, enabling this option might lead to unexpected data loss,<br/>as the system will skip over missing entries instead of treating<br/>them as critical errors. |
 | `metadata_store` | -- | -- | Metadata storage options. |
-| `metadata_store.file_size` | String | `64MB` | The size of the metadata store log file. |
-| `metadata_store.purge_threshold` | String | `256MB` | The threshold of the metadata store size to trigger a purge. |
-| `metadata_store.purge_interval` | String | `1m` | The interval of the metadata store to trigger a purge. |
+| `metadata_store.file_size` | String | `256MB` | Kv file size in bytes. |
+| `metadata_store.purge_threshold` | String | `4GB` | Kv purge threshold. |
 | `procedure` | -- | -- | Procedure storage options. |
 | `procedure.max_retry_times` | Integer | `3` | Procedure max retry time. |
 | `procedure.retry_delay` | String | `500ms` | Initial retry delay of procedures, increases exponentially |
-| `flow` | -- | -- | flow engine options. |
-| `flow.num_workers` | Integer | `0` | The number of flow worker in flownode.<br/>Not setting(or set to 0) this value will use the number of CPU cores divided by 2. |
 | `storage` | -- | -- | The data storage options. |
 | `storage.data_home` | String | `/tmp/greptimedb/` | The working home directory. |
 | `storage.type` | String | `File` | The storage type used to store the data.<br/>- `File`: the data is stored in the local file system.<br/>- `S3`: the data is stored in the S3 object storage.<br/>- `Gcs`: the data is stored in the Google Cloud Storage.<br/>- `Azblob`: the data is stored in the Azure Blob Storage.<br/>- `Oss`: the data is stored in the Aliyun OSS. |
-| `storage.cache_path` | String | Unset | Read cache configuration for object storage such as 'S3' etc, it's configured by default when using object storage. It is recommended to configure it when using object storage for better performance.<br/>A local file directory, defaults to `{data_home}`. An empty string means disabling. |
-| `storage.cache_capacity` | String | Unset | The local file cache capacity in bytes. If your disk space is sufficient, it is recommended to set it larger. |
-| `storage.bucket` | String | Unset | The S3 bucket name.<br/>**It's only used when the storage type is `S3`, `Oss` and `Gcs`**. |
-| `storage.root` | String | Unset | The S3 data will be stored in the specified prefix, for example, `s3://${bucket}/${root}`.<br/>**It's only used when the storage type is `S3`, `Oss` and `Azblob`**. |
-| `storage.access_key_id` | String | Unset | The access key id of the aws account.<br/>It's **highly recommended** to use AWS IAM roles instead of hardcoding the access key id and secret key.<br/>**It's only used when the storage type is `S3` and `Oss`**. |
-| `storage.secret_access_key` | String | Unset | The secret access key of the aws account.<br/>It's **highly recommended** to use AWS IAM roles instead of hardcoding the access key id and secret key.<br/>**It's only used when the storage type is `S3`**. |
-| `storage.access_key_secret` | String | Unset | The secret access key of the aliyun account.<br/>**It's only used when the storage type is `Oss`**. |
-| `storage.account_name` | String | Unset | The account key of the azure account.<br/>**It's only used when the storage type is `Azblob`**. |
-| `storage.account_key` | String | Unset | The account key of the azure account.<br/>**It's only used when the storage type is `Azblob`**. |
-| `storage.scope` | String | Unset | The scope of the google cloud storage.<br/>**It's only used when the storage type is `Gcs`**. |
-| `storage.credential_path` | String | Unset | The credential path of the google cloud storage.<br/>**It's only used when the storage type is `Gcs`**. |
-| `storage.credential` | String | Unset | The credential of the google cloud storage.<br/>**It's only used when the storage type is `Gcs`**. |
-| `storage.container` | String | Unset | The container of the azure account.<br/>**It's only used when the storage type is `Azblob`**. |
-| `storage.sas_token` | String | Unset | The sas token of the azure account.<br/>**It's only used when the storage type is `Azblob`**. |
-| `storage.endpoint` | String | Unset | The endpoint of the S3 service.<br/>**It's only used when the storage type is `S3`, `Oss`, `Gcs` and `Azblob`**. |
-| `storage.region` | String | Unset | The region of the S3 service.<br/>**It's only used when the storage type is `S3`, `Oss`, `Gcs` and `Azblob`**. |
-| `storage.http_client` | -- | -- | The http client options to the storage.<br/>**It's only used when the storage type is `S3`, `Oss`, `Gcs` and `Azblob`**. |
-| `storage.http_client.pool_max_idle_per_host` | Integer | `1024` | The maximum idle connection per host allowed in the pool. |
-| `storage.http_client.connect_timeout` | String | `30s` | The timeout for only the connect phase of a http client. |
-| `storage.http_client.timeout` | String | `30s` | The total request timeout, applied from when the request starts connecting until the response body has finished.<br/>Also considered a total deadline. |
-| `storage.http_client.pool_idle_timeout` | String | `90s` | The timeout for idle sockets being kept-alive. |
+| `storage.cache_path` | String | `None` | Cache configuration for object storage such as 'S3' etc.<br/>The local file cache directory. |
+| `storage.cache_capacity` | String | `None` | The local file cache capacity in bytes. |
+| `storage.bucket` | String | `None` | The S3 bucket name.<br/>**It's only used when the storage type is `S3`, `Oss` and `Gcs`**. |
+| `storage.root` | String | `None` | The S3 data will be stored in the specified prefix, for example, `s3://${bucket}/${root}`.<br/>**It's only used when the storage type is `S3`, `Oss` and `Azblob`**. |
+| `storage.access_key_id` | String | `None` | The access key id of the aws account.<br/>It's **highly recommended** to use AWS IAM roles instead of hardcoding the access key id and secret key.<br/>**It's only used when the storage type is `S3` and `Oss`**. |
+| `storage.secret_access_key` | String | `None` | The secret access key of the aws account.<br/>It's **highly recommended** to use AWS IAM roles instead of hardcoding the access key id and secret key.<br/>**It's only used when the storage type is `S3`**. |
+| `storage.access_key_secret` | String | `None` | The secret access key of the aliyun account.<br/>**It's only used when the storage type is `Oss`**. |
+| `storage.account_name` | String | `None` | The account key of the azure account.<br/>**It's only used when the storage type is `Azblob`**. |
+| `storage.account_key` | String | `None` | The account key of the azure account.<br/>**It's only used when the storage type is `Azblob`**. |
+| `storage.scope` | String | `None` | The scope of the google cloud storage.<br/>**It's only used when the storage type is `Gcs`**. |
+| `storage.credential_path` | String | `None` | The credential path of the google cloud storage.<br/>**It's only used when the storage type is `Gcs`**. |
+| `storage.container` | String | `None` | The container of the azure account.<br/>**It's only used when the storage type is `Azblob`**. |
+| `storage.sas_token` | String | `None` | The sas token of the azure account.<br/>**It's only used when the storage type is `Azblob`**. |
+| `storage.endpoint` | String | `None` | The endpoint of the S3 service.<br/>**It's only used when the storage type is `S3`, `Oss`, `Gcs` and `Azblob`**. |
+| `storage.region` | String | `None` | The region of the S3 service.<br/>**It's only used when the storage type is `S3`, `Oss`, `Gcs` and `Azblob`**. |
 | `[[region_engine]]` | -- | -- | The region engine options. You can configure multiple region engines. |
 | `region_engine.mito` | -- | -- | The Mito engine options. |
 | `region_engine.mito.num_workers` | Integer | `8` | Number of region workers. |
@@ -131,79 +100,50 @@
 | `region_engine.mito.worker_request_batch_size` | Integer | `64` | Max batch size for a worker to handle requests. |
 | `region_engine.mito.manifest_checkpoint_distance` | Integer | `10` | Number of meta action updated to trigger a new checkpoint for the manifest. |
 | `region_engine.mito.compress_manifest` | Bool | `false` | Whether to compress manifest and checkpoint file by gzip (default false). |
-| `region_engine.mito.max_background_flushes` | Integer | Auto | Max number of running background flush jobs (default: 1/2 of cpu cores). |
-| `region_engine.mito.max_background_compactions` | Integer | Auto | Max number of running background compaction jobs (default: 1/4 of cpu cores). |
-| `region_engine.mito.max_background_purges` | Integer | Auto | Max number of running background purge jobs (default: number of cpu cores). |
+| `region_engine.mito.max_background_jobs` | Integer | `4` | Max number of running background jobs |
 | `region_engine.mito.auto_flush_interval` | String | `1h` | Interval to auto flush a region if it has not flushed yet. |
-| `region_engine.mito.global_write_buffer_size` | String | Auto | Global write buffer size for all regions. If not set, it's default to 1/8 of OS memory with a max limitation of 1GB. |
-| `region_engine.mito.global_write_buffer_reject_size` | String | Auto | Global write buffer size threshold to reject write requests. If not set, it's default to 2 times of `global_write_buffer_size`. |
-| `region_engine.mito.sst_meta_cache_size` | String | Auto | Cache size for SST metadata. Setting it to 0 to disable the cache.<br/>If not set, it's default to 1/32 of OS memory with a max limitation of 128MB. |
-| `region_engine.mito.vector_cache_size` | String | Auto | Cache size for vectors and arrow arrays. Setting it to 0 to disable the cache.<br/>If not set, it's default to 1/16 of OS memory with a max limitation of 512MB. |
-| `region_engine.mito.page_cache_size` | String | Auto | Cache size for pages of SST row groups. Setting it to 0 to disable the cache.<br/>If not set, it's default to 1/8 of OS memory. |
-| `region_engine.mito.selector_result_cache_size` | String | Auto | Cache size for time series selector (e.g. `last_value()`). Setting it to 0 to disable the cache.<br/>If not set, it's default to 1/16 of OS memory with a max limitation of 512MB. |
-| `region_engine.mito.enable_write_cache` | Bool | `false` | Whether to enable the write cache, it's enabled by default when using object storage. It is recommended to enable it when using object storage for better performance. |
-| `region_engine.mito.write_cache_path` | String | `""` | File system path for write cache, defaults to `{data_home}`. |
-| `region_engine.mito.write_cache_size` | String | `5GiB` | Capacity for write cache. If your disk space is sufficient, it is recommended to set it larger. |
-| `region_engine.mito.write_cache_ttl` | String | Unset | TTL for write cache. |
+| `region_engine.mito.global_write_buffer_size` | String | `1GB` | Global write buffer size for all regions. If not set, it's default to 1/8 of OS memory with a max limitation of 1GB. |
+| `region_engine.mito.global_write_buffer_reject_size` | String | `2GB` | Global write buffer size threshold to reject write requests. If not set, it's default to 2 times of `global_write_buffer_size` |
+| `region_engine.mito.sst_meta_cache_size` | String | `128MB` | Cache size for SST metadata. Setting it to 0 to disable the cache.<br/>If not set, it's default to 1/32 of OS memory with a max limitation of 128MB. |
+| `region_engine.mito.vector_cache_size` | String | `512MB` | Cache size for vectors and arrow arrays. Setting it to 0 to disable the cache.<br/>If not set, it's default to 1/16 of OS memory with a max limitation of 512MB. |
+| `region_engine.mito.page_cache_size` | String | `512MB` | Cache size for pages of SST row groups. Setting it to 0 to disable the cache.<br/>If not set, it's default to 1/16 of OS memory with a max limitation of 512MB. |
+| `region_engine.mito.enable_experimental_write_cache` | Bool | `false` | Whether to enable the experimental write cache. |
+| `region_engine.mito.experimental_write_cache_path` | String | `""` | File system path for write cache, defaults to `{data_home}/write_cache`. |
+| `region_engine.mito.experimental_write_cache_size` | String | `512MB` | Capacity for write cache. |
+| `region_engine.mito.experimental_write_cache_ttl` | String | `1h` | TTL for write cache. |
 | `region_engine.mito.sst_write_buffer_size` | String | `8MB` | Buffer size for SST writing. |
+| `region_engine.mito.scan_parallelism` | Integer | `0` | Parallelism to scan a region (default: 1/4 of cpu cores).<br/>- `0`: using the default value (1/4 of cpu cores).<br/>- `1`: scan in current thread.<br/>- `n`: scan in parallelism n. |
 | `region_engine.mito.parallel_scan_channel_size` | Integer | `32` | Capacity of the channel to send data from parallel scan tasks to the main task. |
 | `region_engine.mito.allow_stale_entries` | Bool | `false` | Whether to allow stale WAL entries read during replay. |
-| `region_engine.mito.min_compaction_interval` | String | `0m` | Minimum time interval between two compactions.<br/>To align with the old behavior, the default value is 0 (no restrictions). |
-| `region_engine.mito.index` | -- | -- | The options for index in Mito engine. |
-| `region_engine.mito.index.aux_path` | String | `""` | Auxiliary directory path for the index in filesystem, used to store intermediate files for<br/>creating the index and staging files for searching the index, defaults to `{data_home}/index_intermediate`.<br/>The default name for this directory is `index_intermediate` for backward compatibility.<br/><br/>This path contains two subdirectories:<br/>- `__intm`: for storing intermediate files used during creating index.<br/>- `staging`: for storing staging files used during searching index. |
-| `region_engine.mito.index.staging_size` | String | `2GB` | The max capacity of the staging directory. |
-| `region_engine.mito.index.staging_ttl` | String | `7d` | The TTL of the staging directory.<br/>Defaults to 7 days.<br/>Setting it to "0s" to disable TTL. |
-| `region_engine.mito.index.metadata_cache_size` | String | `64MiB` | Cache size for inverted index metadata. |
-| `region_engine.mito.index.content_cache_size` | String | `128MiB` | Cache size for inverted index content. |
-| `region_engine.mito.index.content_cache_page_size` | String | `64KiB` | Page size for inverted index content cache. |
 | `region_engine.mito.inverted_index` | -- | -- | The options for inverted index in Mito engine. |
-| `region_engine.mito.inverted_index.create_on_flush` | String | `auto` | Whether to create the index on flush.<br/>- `auto`: automatically (default)<br/>- `disable`: never |
-| `region_engine.mito.inverted_index.create_on_compaction` | String | `auto` | Whether to create the index on compaction.<br/>- `auto`: automatically (default)<br/>- `disable`: never |
-| `region_engine.mito.inverted_index.apply_on_query` | String | `auto` | Whether to apply the index on query<br/>- `auto`: automatically (default)<br/>- `disable`: never |
-| `region_engine.mito.inverted_index.mem_threshold_on_create` | String | `auto` | Memory threshold for performing an external sort during index creation.<br/>- `auto`: automatically determine the threshold based on the system memory size (default)<br/>- `unlimited`: no memory limit<br/>- `[size]` e.g. `64MB`: fixed memory threshold |
-| `region_engine.mito.inverted_index.intermediate_path` | String | `""` | Deprecated, use `region_engine.mito.index.aux_path` instead. |
-| `region_engine.mito.fulltext_index` | -- | -- | The options for full-text index in Mito engine. |
-| `region_engine.mito.fulltext_index.create_on_flush` | String | `auto` | Whether to create the index on flush.<br/>- `auto`: automatically (default)<br/>- `disable`: never |
-| `region_engine.mito.fulltext_index.create_on_compaction` | String | `auto` | Whether to create the index on compaction.<br/>- `auto`: automatically (default)<br/>- `disable`: never |
-| `region_engine.mito.fulltext_index.apply_on_query` | String | `auto` | Whether to apply the index on query<br/>- `auto`: automatically (default)<br/>- `disable`: never |
-| `region_engine.mito.fulltext_index.mem_threshold_on_create` | String | `auto` | Memory threshold for index creation.<br/>- `auto`: automatically determine the threshold based on the system memory size (default)<br/>- `unlimited`: no memory limit<br/>- `[size]` e.g. `64MB`: fixed memory threshold |
-| `region_engine.mito.bloom_filter_index` | -- | -- | The options for bloom filter in Mito engine. |
-| `region_engine.mito.bloom_filter_index.create_on_flush` | String | `auto` | Whether to create the bloom filter on flush.<br/>- `auto`: automatically (default)<br/>- `disable`: never |
-| `region_engine.mito.bloom_filter_index.create_on_compaction` | String | `auto` | Whether to create the bloom filter on compaction.<br/>- `auto`: automatically (default)<br/>- `disable`: never |
-| `region_engine.mito.bloom_filter_index.apply_on_query` | String | `auto` | Whether to apply the bloom filter on query<br/>- `auto`: automatically (default)<br/>- `disable`: never |
-| `region_engine.mito.bloom_filter_index.mem_threshold_on_create` | String | `auto` | Memory threshold for bloom filter creation.<br/>- `auto`: automatically determine the threshold based on the system memory size (default)<br/>- `unlimited`: no memory limit<br/>- `[size]` e.g. `64MB`: fixed memory threshold |
+| `region_engine.mito.inverted_index.create_on_flush` | String | `auto` | Whether to create the index on flush.<br/>- `auto`: automatically<br/>- `disable`: never |
+| `region_engine.mito.inverted_index.create_on_compaction` | String | `auto` | Whether to create the index on compaction.<br/>- `auto`: automatically<br/>- `disable`: never |
+| `region_engine.mito.inverted_index.apply_on_query` | String | `auto` | Whether to apply the index on query<br/>- `auto`: automatically<br/>- `disable`: never |
+| `region_engine.mito.inverted_index.mem_threshold_on_create` | String | `64M` | Memory threshold for performing an external sort during index creation.<br/>Setting to empty will disable external sorting, forcing all sorting operations to happen in memory. |
+| `region_engine.mito.inverted_index.intermediate_path` | String | `""` | File system path to store intermediate files for external sorting (default `{data_home}/index_intermediate`). |
 | `region_engine.mito.memtable` | -- | -- | -- |
 | `region_engine.mito.memtable.type` | String | `time_series` | Memtable type.<br/>- `time_series`: time-series memtable<br/>- `partition_tree`: partition tree memtable (experimental) |
 | `region_engine.mito.memtable.index_max_keys_per_shard` | Integer | `8192` | The max number of keys in one shard.<br/>Only available for `partition_tree` memtable. |
 | `region_engine.mito.memtable.data_freeze_threshold` | Integer | `32768` | The max rows of data inside the actively writing buffer in one shard.<br/>Only available for `partition_tree` memtable. |
 | `region_engine.mito.memtable.fork_dictionary_bytes` | String | `1GiB` | Max dictionary bytes.<br/>Only available for `partition_tree` memtable. |
-| `region_engine.file` | -- | -- | Enable the file engine. |
-| `region_engine.metric` | -- | -- | Metric engine options. |
-| `region_engine.metric.experimental_sparse_primary_key_encoding` | Bool | `false` | Whether to enable the experimental sparse primary key encoding. |
 | `logging` | -- | -- | The logging options. |
-| `logging.dir` | String | `/tmp/greptimedb/logs` | The directory to store the log files. If set to empty, logs will not be written to files. |
-| `logging.level` | String | Unset | The log level. Can be `info`/`debug`/`warn`/`error`. |
+| `logging.dir` | String | `/tmp/greptimedb/logs` | The directory to store the log files. |
+| `logging.level` | String | `None` | The log level. Can be `info`/`debug`/`warn`/`error`. |
 | `logging.enable_otlp_tracing` | Bool | `false` | Enable OTLP tracing. |
-| `logging.otlp_endpoint` | String | `http://localhost:4317` | The OTLP tracing endpoint. |
+| `logging.otlp_endpoint` | String | `None` | The OTLP tracing endpoint. |
 | `logging.append_stdout` | Bool | `true` | Whether to append logs to stdout. |
-| `logging.log_format` | String | `text` | The log format. Can be `text`/`json`. |
-| `logging.max_log_files` | Integer | `720` | The maximum amount of log files. |
 | `logging.tracing_sample_ratio` | -- | -- | The percentage of tracing will be sampled and exported.<br/>Valid range `[0, 1]`, 1 means all traces are sampled, 0 means all traces are not sampled, the default value is 1.<br/>ratio > 1 are treated as 1. Fractions < 0 are treated as 0 |
 | `logging.tracing_sample_ratio.default_ratio` | Float | `1.0` | -- |
-| `logging.slow_query` | -- | -- | The slow query log options. |
-| `logging.slow_query.enable` | Bool | `false` | Whether to enable slow query log. |
-| `logging.slow_query.threshold` | String | Unset | The threshold of slow query. |
-| `logging.slow_query.sample_ratio` | Float | Unset | The sampling ratio of slow query log. The value should be in the range of (0, 1]. |
 | `export_metrics` | -- | -- | The datanode can export its metrics and send to Prometheus compatible service (e.g. send to `greptimedb` itself) from remote-write API.<br/>This is only used for `greptimedb` to export its own metrics internally. It's different from prometheus scrape. |
 | `export_metrics.enable` | Bool | `false` | whether enable export metrics. |
 | `export_metrics.write_interval` | String | `30s` | The interval of export metrics. |
-| `export_metrics.self_import` | -- | -- | For `standalone` mode, `self_import` is recommended to collect metrics generated by itself<br/>You must create the database before enabling it. |
-| `export_metrics.self_import.db` | String | Unset | -- |
+| `export_metrics.self_import` | -- | -- | For `standalone` mode, `self_import` is recommend to collect metrics generated by itself |
+| `export_metrics.self_import.db` | String | `None` | -- |
 | `export_metrics.remote_write` | -- | -- | -- |
-| `export_metrics.remote_write.url` | String | `""` | The url the metrics send to. The url example can be: `http://127.0.0.1:4000/v1/prometheus/write?db=greptime_metrics`. |
+| `export_metrics.remote_write.url` | String | `""` | The url the metrics send to. The url example can be: `http://127.0.0.1:4000/v1/prometheus/write?db=information_schema`. |
 | `export_metrics.remote_write.headers` | InlineTable | -- | HTTP headers of Prometheus remote-write carry. |
 | `tracing` | -- | -- | The tracing options. Only effect when compiled with `tokio-console` feature. |
-| `tracing.tokio_console_addr` | String | Unset | The tokio console address. |
+| `tracing.tokio_console_addr` | String | `None` | The tokio console address. |


 ## Distributed Mode
@@ -212,55 +152,45 @@

 | Key | Type | Default | Descriptions |
 | --- | -----| ------- | ----------- |
-| `default_timezone` | String | Unset | The default timezone of the server. |
-| `max_in_flight_write_bytes` | String | Unset | The maximum in-flight write bytes. |
-| `runtime` | -- | -- | The runtime options. |
-| `runtime.global_rt_size` | Integer | `8` | The number of threads to execute the runtime for global read operations. |
-| `runtime.compact_rt_size` | Integer | `4` | The number of threads to execute the runtime for global write operations. |
+| `mode` | String | `standalone` | The running mode of the datanode. It can be `standalone` or `distributed`. |
+| `default_timezone` | String | `None` | The default timezone of the server. |
 | `heartbeat` | -- | -- | The heartbeat options. |
 | `heartbeat.interval` | String | `18s` | Interval for sending heartbeat messages to the metasrv. |
 | `heartbeat.retry_interval` | String | `3s` | Interval for retrying to send heartbeat messages to the metasrv. |
 | `http` | -- | -- | The HTTP server options. |
 | `http.addr` | String | `127.0.0.1:4000` | The address to bind the HTTP server. |
-| `http.timeout` | String | `30s` | HTTP request timeout. Set to 0 to disable timeout. |
-| `http.body_limit` | String | `64MB` | HTTP request body limit.<br/>The following units are supported: `B`, `KB`, `KiB`, `MB`, `MiB`, `GB`, `GiB`, `TB`, `TiB`, `PB`, `PiB`.<br/>Set to 0 to disable limit. |
-| `http.enable_cors` | Bool | `true` | HTTP CORS support, it's turned on by default<br/>This allows browser to access http APIs without CORS restrictions |
-| `http.cors_allowed_origins` | Array | Unset | Customize allowed origins for HTTP CORS. |
+| `http.timeout` | String | `30s` | HTTP request timeout. |
+| `http.body_limit` | String | `64MB` | HTTP request body limit.<br/>Support the following units are supported: `B`, `KB`, `KiB`, `MB`, `MiB`, `GB`, `GiB`, `TB`, `TiB`, `PB`, `PiB`. |
 | `grpc` | -- | -- | The gRPC server options. |
-| `grpc.bind_addr` | String | `127.0.0.1:4001` | The address to bind the gRPC server. |
-| `grpc.server_addr` | String | `127.0.0.1:4001` | The address advertised to the metasrv, and used for connections from outside the host.<br/>If left empty or unset, the server will automatically use the IP address of the first network interface<br/>on the host, with the same port number as the one specified in `grpc.bind_addr`. |
+| `grpc.addr` | String | `127.0.0.1:4001` | The address to bind the gRPC server. |
 | `grpc.runtime_size` | Integer | `8` | The number of server worker threads. |
 | `grpc.tls` | -- | -- | gRPC server TLS options, see `mysql.tls` section. |
 | `grpc.tls.mode` | String | `disable` | TLS mode. |
-| `grpc.tls.cert_path` | String | Unset | Certificate file path. |
-| `grpc.tls.key_path` | String | Unset | Private key file path. |
+| `grpc.tls.cert_path` | String | `None` | Certificate file path. |
+| `grpc.tls.key_path` | String | `None` | Private key file path. |
 | `grpc.tls.watch` | Bool | `false` | Watch for Certificate and key file change and auto reload.<br/>For now, gRPC tls config does not support auto reload. |
 | `mysql` | -- | -- | MySQL server options. |
 | `mysql.enable` | Bool | `true` | Whether to enable. |
 | `mysql.addr` | String | `127.0.0.1:4002` | The addr to bind the MySQL server. |
 | `mysql.runtime_size` | Integer | `2` | The number of server worker threads. |
-| `mysql.keep_alive` | String | `0s` | Server-side keep-alive time.<br/>Set to 0 (default) to disable. |
 | `mysql.tls` | -- | -- | -- |
 | `mysql.tls.mode` | String | `disable` | TLS mode, refer to https://www.postgresql.org/docs/current/libpq-ssl.html<br/>- `disable` (default value)<br/>- `prefer`<br/>- `require`<br/>- `verify-ca`<br/>- `verify-full` |
-| `mysql.tls.cert_path` | String | Unset | Certificate file path. |
-| `mysql.tls.key_path` | String | Unset | Private key file path. |
+| `mysql.tls.cert_path` | String | `None` | Certificate file path. |
+| `mysql.tls.key_path` | String | `None` | Private key file path. |
 | `mysql.tls.watch` | Bool | `false` | Watch for Certificate and key file change and auto reload |
 | `postgres` | -- | -- | PostgresSQL server options. |
 | `postgres.enable` | Bool | `true` | Whether to enable |
 | `postgres.addr` | String | `127.0.0.1:4003` | The addr to bind the PostgresSQL server. |
 | `postgres.runtime_size` | Integer | `2` | The number of server worker threads. |
-| `postgres.keep_alive` | String | `0s` | Server-side keep-alive time.<br/>Set to 0 (default) to disable. |
 | `postgres.tls` | -- | -- | PostgresSQL server TLS options, see `mysql.tls` section. |
 | `postgres.tls.mode` | String | `disable` | TLS mode. |
-| `postgres.tls.cert_path` | String | Unset | Certificate file path. |
-| `postgres.tls.key_path` | String | Unset | Private key file path. |
+| `postgres.tls.cert_path` | String | `None` | Certificate file path. |
+| `postgres.tls.key_path` | String | `None` | Private key file path. |
 | `postgres.tls.watch` | Bool | `false` | Watch for Certificate and key file change and auto reload |
 | `opentsdb` | -- | -- | OpenTSDB protocol options. |
 | `opentsdb.enable` | Bool | `true` | Whether to enable OpenTSDB put in HTTP API. |
 | `influxdb` | -- | -- | InfluxDB protocol options. |
 | `influxdb.enable` | Bool | `true` | Whether to enable InfluxDB protocol in HTTP API. |
-| `jaeger` | -- | -- | Jaeger protocol options. |
-| `jaeger.enable` | Bool | `true` | Whether to enable Jaeger protocol in HTTP API. |
 | `prom_store` | -- | -- | Prometheus remote storage options |
 | `prom_store.enable` | Bool | `true` | Whether to enable Prometheus remote write and read in HTTP API. |
 | `prom_store.with_metric_engine` | Bool | `true` | Whether to store the data from Prometheus remote write in metric engine. |
@@ -279,29 +209,23 @@
 | `datanode.client.connect_timeout` | String | `10s` | -- |
 | `datanode.client.tcp_nodelay` | Bool | `true` | -- |
 | `logging` | -- | -- | The logging options. |
-| `logging.dir` | String | `/tmp/greptimedb/logs` | The directory to store the log files. If set to empty, logs will not be written to files. |
-| `logging.level` | String | Unset | The log level. Can be `info`/`debug`/`warn`/`error`. |
+| `logging.dir` | String | `/tmp/greptimedb/logs` | The directory to store the log files. |
+| `logging.level` | String | `None` | The log level. Can be `info`/`debug`/`warn`/`error`. |
 | `logging.enable_otlp_tracing` | Bool | `false` | Enable OTLP tracing. |
-| `logging.otlp_endpoint` | String | `http://localhost:4317` | The OTLP tracing endpoint. |
+| `logging.otlp_endpoint` | String | `None` | The OTLP tracing endpoint. |
 | `logging.append_stdout` | Bool | `true` | Whether to append logs to stdout. |
-| `logging.log_format` | String | `text` | The log format. Can be `text`/`json`. |
-| `logging.max_log_files` | Integer | `720` | The maximum amount of log files. |
 | `logging.tracing_sample_ratio` | -- | -- | The percentage of tracing will be sampled and exported.<br/>Valid range `[0, 1]`, 1 means all traces are sampled, 0 means all traces are not sampled, the default value is 1.<br/>ratio > 1 are treated as 1. Fractions < 0 are treated as 0 |
 | `logging.tracing_sample_ratio.default_ratio` | Float | `1.0` | -- |
-| `logging.slow_query` | -- | -- | The slow query log options. |
-| `logging.slow_query.enable` | Bool | `false` | Whether to enable slow query log. |
-| `logging.slow_query.threshold` | String | Unset | The threshold of slow query. |
-| `logging.slow_query.sample_ratio` | Float | Unset | The sampling ratio of slow query log. The value should be in the range of (0, 1]. |
 | `export_metrics` | -- | -- | The datanode can export its metrics and send to Prometheus compatible service (e.g. send to `greptimedb` itself) from remote-write API.<br/>This is only used for `greptimedb` to export its own metrics internally. It's different from prometheus scrape. |
 | `export_metrics.enable` | Bool | `false` | whether enable export metrics. |
 | `export_metrics.write_interval` | String | `30s` | The interval of export metrics. |
-| `export_metrics.self_import` | -- | -- | For `standalone` mode, `self_import` is recommend to collect metrics generated by itself<br/>You must create the database before enabling it. |
-| `export_metrics.self_import.db` | String | Unset | -- |
+| `export_metrics.self_import` | -- | -- | For `standalone` mode, `self_import` is recommend to collect metrics generated by itself |
+| `export_metrics.self_import.db` | String | `None` | -- |
 | `export_metrics.remote_write` | -- | -- | -- |
-| `export_metrics.remote_write.url` | String | `""` | The url the metrics send to. The url example can be: `http://127.0.0.1:4000/v1/prometheus/write?db=greptime_metrics`. |
+| `export_metrics.remote_write.url` | String | `""` | The url the metrics send to. The url example can be: `http://127.0.0.1:4000/v1/prometheus/write?db=information_schema`. |
 | `export_metrics.remote_write.headers` | InlineTable | -- | HTTP headers of Prometheus remote-write carry. |
 | `tracing` | -- | -- | The tracing options. Only effect when compiled with `tokio-console` feature. |
-| `tracing.tokio_console_addr` | String | Unset | The tokio console address. |
+| `tracing.tokio_console_addr` | String | `None` | The tokio console address. |


 ### Metasrv
@@ -310,41 +234,32 @@
 | --- | -----| ------- | ----------- |
 | `data_home` | String | `/tmp/metasrv/` | The working home directory. |
 | `bind_addr` | String | `127.0.0.1:3002` | The bind address of metasrv. |
-| `server_addr` | String | `127.0.0.1:3002` | The communication server address for the frontend and datanode to connect to metasrv.<br/>If left empty or unset, the server will automatically use the IP address of the first network interface<br/>on the host, with the same port number as the one specified in `bind_addr`. |
-| `store_addrs` | Array | -- | Store server address default to etcd store.<br/>For postgres store, the format is:<br/>"password=password dbname=postgres user=postgres host=localhost port=5432"<br/>For etcd store, the format is:<br/>"127.0.0.1:2379" |
-| `store_key_prefix` | String | `""` | If it's not empty, the metasrv will store all data with this key prefix. |
-| `backend` | String | `etcd_store` | The datastore for meta server.<br/>Available values:<br/>- `etcd_store` (default value)<br/>- `memory_store`<br/>- `postgres_store` |
-| `meta_table_name` | String | `greptime_metakv` | Table name in RDS to store metadata. Effect when using a RDS kvbackend.<br/>**Only used when backend is `postgres_store`.** |
-| `meta_election_lock_id` | Integer | `1` | Advisory lock id in PostgreSQL for election. Effect when using PostgreSQL as kvbackend<br/>Only used when backend is `postgres_store`. |
-| `selector` | String | `round_robin` | Datanode selector type.<br/>- `round_robin` (default value)<br/>- `lease_based`<br/>- `load_based`<br/>For details, please see "https://docs.greptime.com/developer-guide/metasrv/selector". |
+| `server_addr` | String | `127.0.0.1:3002` | The communication server address for frontend and datanode to connect to metasrv,  "127.0.0.1:3002" by default for localhost. |
+| `store_addr` | String | `127.0.0.1:2379` | Etcd server address. |
+| `selector` | String | `lease_based` | Datanode selector type.<br/>- `lease_based` (default value).<br/>- `load_based`<br/>For details, please see "https://docs.greptime.com/developer-guide/metasrv/selector". |
 | `use_memory_store` | Bool | `false` | Store data in memory. |
-| `enable_region_failover` | Bool | `false` | Whether to enable region failover.<br/>This feature is only available on GreptimeDB running on cluster mode and<br/>- Using Remote WAL<br/>- Using shared storage (e.g., s3). |
-| `node_max_idle_time` | String | `24hours` | Max allowed idle time before removing node info from metasrv memory. |
-| `enable_telemetry` | Bool | `true` | Whether to enable greptimedb telemetry. Enabled by default. |
-| `runtime` | -- | -- | The runtime options. |
-| `runtime.global_rt_size` | Integer | `8` | The number of threads to execute the runtime for global read operations. |
-| `runtime.compact_rt_size` | Integer | `4` | The number of threads to execute the runtime for global write operations. |
+| `enable_telemetry` | Bool | `true` | Whether to enable greptimedb telemetry. |
+| `store_key_prefix` | String | `""` | If it's not empty, the metasrv will store all data with this key prefix. |
 | `procedure` | -- | -- | Procedure storage options. |
 | `procedure.max_retry_times` | Integer | `12` | Procedure max retry time. |
 | `procedure.retry_delay` | String | `500ms` | Initial retry delay of procedures, increases exponentially |
 | `procedure.max_metadata_value_size` | String | `1500KiB` | Auto split large value<br/>GreptimeDB procedure uses etcd as the default metadata storage backend.<br/>The etcd the maximum size of any request is 1.5 MiB<br/>1500KiB = 1536KiB (1.5MiB) - 36KiB (reserved size of key)<br/>Comments out the `max_metadata_value_size`, for don't split large value (no limit). |
 | `failure_detector` | -- | -- | -- |
-| `failure_detector.threshold` | Float | `8.0` | The threshold value used by the failure detector to determine failure conditions. |
-| `failure_detector.min_std_deviation` | String | `100ms` | The minimum standard deviation of the heartbeat intervals, used to calculate acceptable variations. |
-| `failure_detector.acceptable_heartbeat_pause` | String | `10000ms` | The acceptable pause duration between heartbeats, used to determine if a heartbeat interval is acceptable. |
-| `failure_detector.first_heartbeat_estimate` | String | `1000ms` | The initial estimate of the heartbeat interval used by the failure detector. |
+| `failure_detector.threshold` | Float | `8.0` | -- |
+| `failure_detector.min_std_deviation` | String | `100ms` | -- |
+| `failure_detector.acceptable_heartbeat_pause` | String | `3000ms` | -- |
+| `failure_detector.first_heartbeat_estimate` | String | `1000ms` | -- |
 | `datanode` | -- | -- | Datanode options. |
 | `datanode.client` | -- | -- | Datanode client options. |
-| `datanode.client.timeout` | String | `10s` | Operation timeout. |
-| `datanode.client.connect_timeout` | String | `10s` | Connect server timeout. |
-| `datanode.client.tcp_nodelay` | Bool | `true` | `TCP_NODELAY` option for accepted connections. |
+| `datanode.client.timeout` | String | `10s` | -- |
+| `datanode.client.connect_timeout` | String | `10s` | -- |
+| `datanode.client.tcp_nodelay` | Bool | `true` | -- |
 | `wal` | -- | -- | -- |
 | `wal.provider` | String | `raft_engine` | -- |
 | `wal.broker_endpoints` | Array | -- | The broker endpoints of the Kafka cluster. |
-| `wal.auto_create_topics` | Bool | `true` | Automatically create topics for WAL.<br/>Set to `true` to automatically create topics for WAL.<br/>Otherwise, use topics named `topic_name_prefix_[0..num_topics)` |
-| `wal.num_topics` | Integer | `64` | Number of topics. |
+| `wal.num_topics` | Integer | `64` | Number of topics to be created upon start. |
 | `wal.selector_type` | String | `round_robin` | Topic selector type.<br/>Available selector types:<br/>- `round_robin` (default) |
-| `wal.topic_name_prefix` | String | `greptimedb_wal_topic` | A Kafka topic is constructed by concatenating `topic_name_prefix` and `topic_id`.<br/>Only accepts strings that match the following regular expression pattern:<br/>[a-zA-Z_:-][a-zA-Z0-9_:\-\.@#]*<br/>i.g., greptimedb_wal_topic_0, greptimedb_wal_topic_1. |
+| `wal.topic_name_prefix` | String | `greptimedb_wal_topic` | A Kafka topic is constructed by concatenating `topic_name_prefix` and `topic_id`. |
 | `wal.replication_factor` | Integer | `1` | Expected number of replicas of each partition. |
 | `wal.create_topic_timeout` | String | `30s` | Above which a topic creation operation will be cancelled. |
 | `wal.backoff_init` | String | `500ms` | The initial backoff for kafka clients. |
@@ -352,29 +267,23 @@
 | `wal.backoff_base` | Integer | `2` | Exponential backoff rate, i.e. next backoff = base * current backoff. |
 | `wal.backoff_deadline` | String | `5mins` | Stop reconnecting if the total wait time reaches the deadline. If this config is missing, the reconnecting won't terminate. |
 | `logging` | -- | -- | The logging options. |
-| `logging.dir` | String | `/tmp/greptimedb/logs` | The directory to store the log files. If set to empty, logs will not be written to files. |
-| `logging.level` | String | Unset | The log level. Can be `info`/`debug`/`warn`/`error`. |
+| `logging.dir` | String | `/tmp/greptimedb/logs` | The directory to store the log files. |
+| `logging.level` | String | `None` | The log level. Can be `info`/`debug`/`warn`/`error`. |
 | `logging.enable_otlp_tracing` | Bool | `false` | Enable OTLP tracing. |
-| `logging.otlp_endpoint` | String | `http://localhost:4317` | The OTLP tracing endpoint. |
+| `logging.otlp_endpoint` | String | `None` | The OTLP tracing endpoint. |
 | `logging.append_stdout` | Bool | `true` | Whether to append logs to stdout. |
-| `logging.log_format` | String | `text` | The log format. Can be `text`/`json`. |
-| `logging.max_log_files` | Integer | `720` | The maximum amount of log files. |
 | `logging.tracing_sample_ratio` | -- | -- | The percentage of tracing will be sampled and exported.<br/>Valid range `[0, 1]`, 1 means all traces are sampled, 0 means all traces are not sampled, the default value is 1.<br/>ratio > 1 are treated as 1. Fractions < 0 are treated as 0 |
 | `logging.tracing_sample_ratio.default_ratio` | Float | `1.0` | -- |
-| `logging.slow_query` | -- | -- | The slow query log options. |
-| `logging.slow_query.enable` | Bool | `false` | Whether to enable slow query log. |
-| `logging.slow_query.threshold` | String | Unset | The threshold of slow query. |
-| `logging.slow_query.sample_ratio` | Float | Unset | The sampling ratio of slow query log. The value should be in the range of (0, 1]. |
 | `export_metrics` | -- | -- | The datanode can export its metrics and send to Prometheus compatible service (e.g. send to `greptimedb` itself) from remote-write API.<br/>This is only used for `greptimedb` to export its own metrics internally. It's different from prometheus scrape. |
 | `export_metrics.enable` | Bool | `false` | whether enable export metrics. |
 | `export_metrics.write_interval` | String | `30s` | The interval of export metrics. |
-| `export_metrics.self_import` | -- | -- | For `standalone` mode, `self_import` is recommend to collect metrics generated by itself<br/>You must create the database before enabling it. |
-| `export_metrics.self_import.db` | String | Unset | -- |
+| `export_metrics.self_import` | -- | -- | For `standalone` mode, `self_import` is recommend to collect metrics generated by itself |
+| `export_metrics.self_import.db` | String | `None` | -- |
 | `export_metrics.remote_write` | -- | -- | -- |
-| `export_metrics.remote_write.url` | String | `""` | The url the metrics send to. The url example can be: `http://127.0.0.1:4000/v1/prometheus/write?db=greptime_metrics`. |
+| `export_metrics.remote_write.url` | String | `""` | The url the metrics send to. The url example can be: `http://127.0.0.1:4000/v1/prometheus/write?db=information_schema`. |
 | `export_metrics.remote_write.headers` | InlineTable | -- | HTTP headers of Prometheus remote-write carry. |
 | `tracing` | -- | -- | The tracing options. Only effect when compiled with `tokio-console` feature. |
-| `tracing.tokio_console_addr` | String | Unset | The tokio console address. |
+| `tracing.tokio_console_addr` | String | `None` | The tokio console address. |


 ### Datanode
@@ -382,30 +291,15 @@
 | Key | Type | Default | Descriptions |
 | --- | -----| ------- | ----------- |
 | `mode` | String | `standalone` | The running mode of the datanode. It can be `standalone` or `distributed`. |
-| `node_id` | Integer | Unset | The datanode identifier and should be unique in the cluster. |
+| `node_id` | Integer | `None` | The datanode identifier and should be unique in the cluster. |
 | `require_lease_before_startup` | Bool | `false` | Start services after regions have obtained leases.<br/>It will block the datanode start if it can't receive leases in the heartbeat from metasrv. |
 | `init_regions_in_background` | Bool | `false` | Initialize all regions in the background during the startup.<br/>By default, it provides services after all regions have been initialized. |
-| `init_regions_parallelism` | Integer | `16` | Parallelism of initializing regions. |
-| `max_concurrent_queries` | Integer | `0` | The maximum current queries allowed to be executed. Zero means unlimited. |
-| `enable_telemetry` | Bool | `true` | Enable telemetry to collect anonymous usage data. Enabled by default. |
-| `http` | -- | -- | The HTTP server options. |
-| `http.addr` | String | `127.0.0.1:4000` | The address to bind the HTTP server. |
-| `http.timeout` | String | `30s` | HTTP request timeout. Set to 0 to disable timeout. |
-| `http.body_limit` | String | `64MB` | HTTP request body limit.<br/>The following units are supported: `B`, `KB`, `KiB`, `MB`, `MiB`, `GB`, `GiB`, `TB`, `TiB`, `PB`, `PiB`.<br/>Set to 0 to disable limit. |
-| `grpc` | -- | -- | The gRPC server options. |
-| `grpc.bind_addr` | String | `127.0.0.1:3001` | The address to bind the gRPC server. |
-| `grpc.server_addr` | String | `127.0.0.1:3001` | The address advertised to the metasrv, and used for connections from outside the host.<br/>If left empty or unset, the server will automatically use the IP address of the first network interface<br/>on the host, with the same port number as the one specified in `grpc.bind_addr`. |
-| `grpc.runtime_size` | Integer | `8` | The number of server worker threads. |
-| `grpc.max_recv_message_size` | String | `512MB` | The maximum receive message size for gRPC server. |
-| `grpc.max_send_message_size` | String | `512MB` | The maximum send message size for gRPC server. |
-| `grpc.tls` | -- | -- | gRPC server TLS options, see `mysql.tls` section. |
-| `grpc.tls.mode` | String | `disable` | TLS mode. |
-| `grpc.tls.cert_path` | String | Unset | Certificate file path. |
-| `grpc.tls.key_path` | String | Unset | Private key file path. |
-| `grpc.tls.watch` | Bool | `false` | Watch for Certificate and key file change and auto reload.<br/>For now, gRPC tls config does not support auto reload. |
-| `runtime` | -- | -- | The runtime options. |
-| `runtime.global_rt_size` | Integer | `8` | The number of threads to execute the runtime for global read operations. |
-| `runtime.compact_rt_size` | Integer | `4` | The number of threads to execute the runtime for global write operations. |
+| `rpc_addr` | String | `127.0.0.1:3001` | The gRPC address of the datanode. |
+| `rpc_hostname` | String | `None` | The hostname of the datanode. |
+| `rpc_runtime_size` | Integer | `8` | The number of gRPC server worker threads. |
+| `rpc_max_recv_message_size` | String | `512MB` | The maximum receive message size for gRPC server. |
+| `rpc_max_send_message_size` | String | `512MB` | The maximum send message size for gRPC server. |
+| `enable_telemetry` | Bool | `true` | Enable telemetry to collect anonymous usage data. |
 | `heartbeat` | -- | -- | The heartbeat options. |
 | `heartbeat.interval` | String | `3s` | Interval for sending heartbeat messages to the metasrv. |
 | `heartbeat.retry_interval` | String | `3s` | Interval for retrying to send heartbeat messages to the metasrv. |
@@ -421,50 +315,41 @@
 | `meta_client.metadata_cache_tti` | String | `5m` | -- |
 | `wal` | -- | -- | The WAL options. |
 | `wal.provider` | String | `raft_engine` | The provider of the WAL.<br/>- `raft_engine`: the wal is stored in the local file system by raft-engine.<br/>- `kafka`: it's remote wal that data is stored in Kafka. |
-| `wal.dir` | String | Unset | The directory to store the WAL files.<br/>**It's only used when the provider is `raft_engine`**. |
-| `wal.file_size` | String | `128MB` | The size of the WAL segment file.<br/>**It's only used when the provider is `raft_engine`**. |
-| `wal.purge_threshold` | String | `1GB` | The threshold of the WAL size to trigger a flush.<br/>**It's only used when the provider is `raft_engine`**. |
-| `wal.purge_interval` | String | `1m` | The interval to trigger a flush.<br/>**It's only used when the provider is `raft_engine`**. |
+| `wal.dir` | String | `None` | The directory to store the WAL files.<br/>**It's only used when the provider is `raft_engine`**. |
+| `wal.file_size` | String | `256MB` | The size of the WAL segment file.<br/>**It's only used when the provider is `raft_engine`**. |
+| `wal.purge_threshold` | String | `4GB` | The threshold of the WAL size to trigger a flush.<br/>**It's only used when the provider is `raft_engine`**. |
+| `wal.purge_interval` | String | `10m` | The interval to trigger a flush.<br/>**It's only used when the provider is `raft_engine`**. |
 | `wal.read_batch_size` | Integer | `128` | The read batch size.<br/>**It's only used when the provider is `raft_engine`**. |
 | `wal.sync_write` | Bool | `false` | Whether to use sync write.<br/>**It's only used when the provider is `raft_engine`**. |
 | `wal.enable_log_recycle` | Bool | `true` | Whether to reuse logically truncated log files.<br/>**It's only used when the provider is `raft_engine`**. |
 | `wal.prefill_log_files` | Bool | `false` | Whether to pre-create log files on start up.<br/>**It's only used when the provider is `raft_engine`**. |
 | `wal.sync_period` | String | `10s` | Duration for fsyncing log files.<br/>**It's only used when the provider is `raft_engine`**. |
-| `wal.recovery_parallelism` | Integer | `2` | Parallelism during WAL recovery. |
 | `wal.broker_endpoints` | Array | -- | The Kafka broker endpoints.<br/>**It's only used when the provider is `kafka`**. |
-| `wal.max_batch_bytes` | String | `1MB` | The max size of a single producer batch.<br/>Warning: Kafka has a default limit of 1MB per message in a topic.<br/>**It's only used when the provider is `kafka`**. |
+| `wal.max_batch_size` | String | `1MB` | The max size of a single producer batch.<br/>Warning: Kafka has a default limit of 1MB per message in a topic.<br/>**It's only used when the provider is `kafka`**. |
+| `wal.linger` | String | `200ms` | The linger duration of a kafka batch producer.<br/>**It's only used when the provider is `kafka`**. |
 | `wal.consumer_wait_timeout` | String | `100ms` | The consumer wait timeout.<br/>**It's only used when the provider is `kafka`**. |
 | `wal.backoff_init` | String | `500ms` | The initial backoff delay.<br/>**It's only used when the provider is `kafka`**. |
 | `wal.backoff_max` | String | `10s` | The maximum backoff delay.<br/>**It's only used when the provider is `kafka`**. |
 | `wal.backoff_base` | Integer | `2` | The exponential backoff rate, i.e. next backoff = base * current backoff.<br/>**It's only used when the provider is `kafka`**. |
 | `wal.backoff_deadline` | String | `5mins` | The deadline of retries.<br/>**It's only used when the provider is `kafka`**. |
-| `wal.create_index` | Bool | `true` | Whether to enable WAL index creation.<br/>**It's only used when the provider is `kafka`**. |
-| `wal.dump_index_interval` | String | `60s` | The interval for dumping WAL indexes.<br/>**It's only used when the provider is `kafka`**. |
-| `wal.overwrite_entry_start_id` | Bool | `false` | Ignore missing entries during read WAL.<br/>**It's only used when the provider is `kafka`**.<br/><br/>This option ensures that when Kafka messages are deleted, the system<br/>can still successfully replay memtable data without throwing an<br/>out-of-range error.<br/>However, enabling this option might lead to unexpected data loss,<br/>as the system will skip over missing entries instead of treating<br/>them as critical errors. |
 | `storage` | -- | -- | The data storage options. |
 | `storage.data_home` | String | `/tmp/greptimedb/` | The working home directory. |
 | `storage.type` | String | `File` | The storage type used to store the data.<br/>- `File`: the data is stored in the local file system.<br/>- `S3`: the data is stored in the S3 object storage.<br/>- `Gcs`: the data is stored in the Google Cloud Storage.<br/>- `Azblob`: the data is stored in the Azure Blob Storage.<br/>- `Oss`: the data is stored in the Aliyun OSS. |
-| `storage.cache_path` | String | Unset | Read cache configuration for object storage such as 'S3' etc, it's configured by default when using object storage. It is recommended to configure it when using object storage for better performance.<br/>A local file directory, defaults to `{data_home}`. An empty string means disabling. |
-| `storage.cache_capacity` | String | Unset | The local file cache capacity in bytes. If your disk space is sufficient, it is recommended to set it larger. |
-| `storage.bucket` | String | Unset | The S3 bucket name.<br/>**It's only used when the storage type is `S3`, `Oss` and `Gcs`**. |
-| `storage.root` | String | Unset | The S3 data will be stored in the specified prefix, for example, `s3://${bucket}/${root}`.<br/>**It's only used when the storage type is `S3`, `Oss` and `Azblob`**. |
-| `storage.access_key_id` | String | Unset | The access key id of the aws account.<br/>It's **highly recommended** to use AWS IAM roles instead of hardcoding the access key id and secret key.<br/>**It's only used when the storage type is `S3` and `Oss`**. |
-| `storage.secret_access_key` | String | Unset | The secret access key of the aws account.<br/>It's **highly recommended** to use AWS IAM roles instead of hardcoding the access key id and secret key.<br/>**It's only used when the storage type is `S3`**. |
-| `storage.access_key_secret` | String | Unset | The secret access key of the aliyun account.<br/>**It's only used when the storage type is `Oss`**. |
-| `storage.account_name` | String | Unset | The account key of the azure account.<br/>**It's only used when the storage type is `Azblob`**. |
-| `storage.account_key` | String | Unset | The account key of the azure account.<br/>**It's only used when the storage type is `Azblob`**. |
-| `storage.scope` | String | Unset | The scope of the google cloud storage.<br/>**It's only used when the storage type is `Gcs`**. |
-| `storage.credential_path` | String | Unset | The credential path of the google cloud storage.<br/>**It's only used when the storage type is `Gcs`**. |
-| `storage.credential` | String | Unset | The credential of the google cloud storage.<br/>**It's only used when the storage type is `Gcs`**. |
-| `storage.container` | String | Unset | The container of the azure account.<br/>**It's only used when the storage type is `Azblob`**. |
-| `storage.sas_token` | String | Unset | The sas token of the azure account.<br/>**It's only used when the storage type is `Azblob`**. |
-| `storage.endpoint` | String | Unset | The endpoint of the S3 service.<br/>**It's only used when the storage type is `S3`, `Oss`, `Gcs` and `Azblob`**. |
-| `storage.region` | String | Unset | The region of the S3 service.<br/>**It's only used when the storage type is `S3`, `Oss`, `Gcs` and `Azblob`**. |
-| `storage.http_client` | -- | -- | The http client options to the storage.<br/>**It's only used when the storage type is `S3`, `Oss`, `Gcs` and `Azblob`**. |
-| `storage.http_client.pool_max_idle_per_host` | Integer | `1024` | The maximum idle connection per host allowed in the pool. |
-| `storage.http_client.connect_timeout` | String | `30s` | The timeout for only the connect phase of a http client. |
-| `storage.http_client.timeout` | String | `30s` | The total request timeout, applied from when the request starts connecting until the response body has finished.<br/>Also considered a total deadline. |
-| `storage.http_client.pool_idle_timeout` | String | `90s` | The timeout for idle sockets being kept-alive. |
+| `storage.cache_path` | String | `None` | Cache configuration for object storage such as 'S3' etc.<br/>The local file cache directory. |
+| `storage.cache_capacity` | String | `None` | The local file cache capacity in bytes. |
+| `storage.bucket` | String | `None` | The S3 bucket name.<br/>**It's only used when the storage type is `S3`, `Oss` and `Gcs`**. |
+| `storage.root` | String | `None` | The S3 data will be stored in the specified prefix, for example, `s3://${bucket}/${root}`.<br/>**It's only used when the storage type is `S3`, `Oss` and `Azblob`**. |
+| `storage.access_key_id` | String | `None` | The access key id of the aws account.<br/>It's **highly recommended** to use AWS IAM roles instead of hardcoding the access key id and secret key.<br/>**It's only used when the storage type is `S3` and `Oss`**. |
+| `storage.secret_access_key` | String | `None` | The secret access key of the aws account.<br/>It's **highly recommended** to use AWS IAM roles instead of hardcoding the access key id and secret key.<br/>**It's only used when the storage type is `S3`**. |
+| `storage.access_key_secret` | String | `None` | The secret access key of the aliyun account.<br/>**It's only used when the storage type is `Oss`**. |
+| `storage.account_name` | String | `None` | The account key of the azure account.<br/>**It's only used when the storage type is `Azblob`**. |
+| `storage.account_key` | String | `None` | The account key of the azure account.<br/>**It's only used when the storage type is `Azblob`**. |
+| `storage.scope` | String | `None` | The scope of the google cloud storage.<br/>**It's only used when the storage type is `Gcs`**. |
+| `storage.credential_path` | String | `None` | The credential path of the google cloud storage.<br/>**It's only used when the storage type is `Gcs`**. |
+| `storage.container` | String | `None` | The container of the azure account.<br/>**It's only used when the storage type is `Azblob`**. |
+| `storage.sas_token` | String | `None` | The sas token of the azure account.<br/>**It's only used when the storage type is `Azblob`**. |
+| `storage.endpoint` | String | `None` | The endpoint of the S3 service.<br/>**It's only used when the storage type is `S3`, `Oss`, `Gcs` and `Azblob`**. |
+| `storage.region` | String | `None` | The region of the S3 service.<br/>**It's only used when the storage type is `S3`, `Oss`, `Gcs` and `Azblob`**. |
 | `[[region_engine]]` | -- | -- | The region engine options. You can configure multiple region engines. |
 | `region_engine.mito` | -- | -- | The Mito engine options. |
 | `region_engine.mito.num_workers` | Integer | `8` | Number of region workers. |
@@ -472,125 +357,47 @@
 | `region_engine.mito.worker_request_batch_size` | Integer | `64` | Max batch size for a worker to handle requests. |
 | `region_engine.mito.manifest_checkpoint_distance` | Integer | `10` | Number of meta action updated to trigger a new checkpoint for the manifest. |
 | `region_engine.mito.compress_manifest` | Bool | `false` | Whether to compress manifest and checkpoint file by gzip (default false). |
-| `region_engine.mito.max_background_flushes` | Integer | Auto | Max number of running background flush jobs (default: 1/2 of cpu cores). |
-| `region_engine.mito.max_background_compactions` | Integer | Auto | Max number of running background compaction jobs (default: 1/4 of cpu cores). |
-| `region_engine.mito.max_background_purges` | Integer | Auto | Max number of running background purge jobs (default: number of cpu cores). |
+| `region_engine.mito.max_background_jobs` | Integer | `4` | Max number of running background jobs |
 | `region_engine.mito.auto_flush_interval` | String | `1h` | Interval to auto flush a region if it has not flushed yet. |
-| `region_engine.mito.global_write_buffer_size` | String | Auto | Global write buffer size for all regions. If not set, it's default to 1/8 of OS memory with a max limitation of 1GB. |
-| `region_engine.mito.global_write_buffer_reject_size` | String | Auto | Global write buffer size threshold to reject write requests. If not set, it's default to 2 times of `global_write_buffer_size` |
-| `region_engine.mito.sst_meta_cache_size` | String | Auto | Cache size for SST metadata. Setting it to 0 to disable the cache.<br/>If not set, it's default to 1/32 of OS memory with a max limitation of 128MB. |
-| `region_engine.mito.vector_cache_size` | String | Auto | Cache size for vectors and arrow arrays. Setting it to 0 to disable the cache.<br/>If not set, it's default to 1/16 of OS memory with a max limitation of 512MB. |
-| `region_engine.mito.page_cache_size` | String | Auto | Cache size for pages of SST row groups. Setting it to 0 to disable the cache.<br/>If not set, it's default to 1/8 of OS memory. |
-| `region_engine.mito.selector_result_cache_size` | String | Auto | Cache size for time series selector (e.g. `last_value()`). Setting it to 0 to disable the cache.<br/>If not set, it's default to 1/16 of OS memory with a max limitation of 512MB. |
-| `region_engine.mito.enable_write_cache` | Bool | `false` | Whether to enable the write cache, it's enabled by default when using object storage. It is recommended to enable it when using object storage for better performance. |
-| `region_engine.mito.write_cache_path` | String | `""` | File system path for write cache, defaults to `{data_home}`. |
-| `region_engine.mito.write_cache_size` | String | `5GiB` | Capacity for write cache. If your disk space is sufficient, it is recommended to set it larger. |
-| `region_engine.mito.write_cache_ttl` | String | Unset | TTL for write cache. |
+| `region_engine.mito.global_write_buffer_size` | String | `1GB` | Global write buffer size for all regions. If not set, it's default to 1/8 of OS memory with a max limitation of 1GB. |
+| `region_engine.mito.global_write_buffer_reject_size` | String | `2GB` | Global write buffer size threshold to reject write requests. If not set, it's default to 2 times of `global_write_buffer_size` |
+| `region_engine.mito.sst_meta_cache_size` | String | `128MB` | Cache size for SST metadata. Setting it to 0 to disable the cache.<br/>If not set, it's default to 1/32 of OS memory with a max limitation of 128MB. |
+| `region_engine.mito.vector_cache_size` | String | `512MB` | Cache size for vectors and arrow arrays. Setting it to 0 to disable the cache.<br/>If not set, it's default to 1/16 of OS memory with a max limitation of 512MB. |
+| `region_engine.mito.page_cache_size` | String | `512MB` | Cache size for pages of SST row groups. Setting it to 0 to disable the cache.<br/>If not set, it's default to 1/16 of OS memory with a max limitation of 512MB. |
+| `region_engine.mito.enable_experimental_write_cache` | Bool | `false` | Whether to enable the experimental write cache. |
+| `region_engine.mito.experimental_write_cache_path` | String | `""` | File system path for write cache, defaults to `{data_home}/write_cache`. |
+| `region_engine.mito.experimental_write_cache_size` | String | `512MB` | Capacity for write cache. |
+| `region_engine.mito.experimental_write_cache_ttl` | String | `1h` | TTL for write cache. |
 | `region_engine.mito.sst_write_buffer_size` | String | `8MB` | Buffer size for SST writing. |
+| `region_engine.mito.scan_parallelism` | Integer | `0` | Parallelism to scan a region (default: 1/4 of cpu cores).<br/>- `0`: using the default value (1/4 of cpu cores).<br/>- `1`: scan in current thread.<br/>- `n`: scan in parallelism n. |
 | `region_engine.mito.parallel_scan_channel_size` | Integer | `32` | Capacity of the channel to send data from parallel scan tasks to the main task. |
 | `region_engine.mito.allow_stale_entries` | Bool | `false` | Whether to allow stale WAL entries read during replay. |
-| `region_engine.mito.min_compaction_interval` | String | `0m` | Minimum time interval between two compactions.<br/>To align with the old behavior, the default value is 0 (no restrictions). |
-| `region_engine.mito.index` | -- | -- | The options for index in Mito engine. |
-| `region_engine.mito.index.aux_path` | String | `""` | Auxiliary directory path for the index in filesystem, used to store intermediate files for<br/>creating the index and staging files for searching the index, defaults to `{data_home}/index_intermediate`.<br/>The default name for this directory is `index_intermediate` for backward compatibility.<br/><br/>This path contains two subdirectories:<br/>- `__intm`: for storing intermediate files used during creating index.<br/>- `staging`: for storing staging files used during searching index. |
-| `region_engine.mito.index.staging_size` | String | `2GB` | The max capacity of the staging directory. |
-| `region_engine.mito.index.staging_ttl` | String | `7d` | The TTL of the staging directory.<br/>Defaults to 7 days.<br/>Setting it to "0s" to disable TTL. |
-| `region_engine.mito.index.metadata_cache_size` | String | `64MiB` | Cache size for inverted index metadata. |
-| `region_engine.mito.index.content_cache_size` | String | `128MiB` | Cache size for inverted index content. |
-| `region_engine.mito.index.content_cache_page_size` | String | `64KiB` | Page size for inverted index content cache. |
 | `region_engine.mito.inverted_index` | -- | -- | The options for inverted index in Mito engine. |
-| `region_engine.mito.inverted_index.create_on_flush` | String | `auto` | Whether to create the index on flush.<br/>- `auto`: automatically (default)<br/>- `disable`: never |
-| `region_engine.mito.inverted_index.create_on_compaction` | String | `auto` | Whether to create the index on compaction.<br/>- `auto`: automatically (default)<br/>- `disable`: never |
-| `region_engine.mito.inverted_index.apply_on_query` | String | `auto` | Whether to apply the index on query<br/>- `auto`: automatically (default)<br/>- `disable`: never |
-| `region_engine.mito.inverted_index.mem_threshold_on_create` | String | `auto` | Memory threshold for performing an external sort during index creation.<br/>- `auto`: automatically determine the threshold based on the system memory size (default)<br/>- `unlimited`: no memory limit<br/>- `[size]` e.g. `64MB`: fixed memory threshold |
-| `region_engine.mito.inverted_index.intermediate_path` | String | `""` | Deprecated, use `region_engine.mito.index.aux_path` instead. |
-| `region_engine.mito.fulltext_index` | -- | -- | The options for full-text index in Mito engine. |
-| `region_engine.mito.fulltext_index.create_on_flush` | String | `auto` | Whether to create the index on flush.<br/>- `auto`: automatically (default)<br/>- `disable`: never |
-| `region_engine.mito.fulltext_index.create_on_compaction` | String | `auto` | Whether to create the index on compaction.<br/>- `auto`: automatically (default)<br/>- `disable`: never |
-| `region_engine.mito.fulltext_index.apply_on_query` | String | `auto` | Whether to apply the index on query<br/>- `auto`: automatically (default)<br/>- `disable`: never |
-| `region_engine.mito.fulltext_index.mem_threshold_on_create` | String | `auto` | Memory threshold for index creation.<br/>- `auto`: automatically determine the threshold based on the system memory size (default)<br/>- `unlimited`: no memory limit<br/>- `[size]` e.g. `64MB`: fixed memory threshold |
-| `region_engine.mito.bloom_filter_index` | -- | -- | The options for bloom filter index in Mito engine. |
-| `region_engine.mito.bloom_filter_index.create_on_flush` | String | `auto` | Whether to create the index on flush.<br/>- `auto`: automatically (default)<br/>- `disable`: never |
-| `region_engine.mito.bloom_filter_index.create_on_compaction` | String | `auto` | Whether to create the index on compaction.<br/>- `auto`: automatically (default)<br/>- `disable`: never |
-| `region_engine.mito.bloom_filter_index.apply_on_query` | String | `auto` | Whether to apply the index on query<br/>- `auto`: automatically (default)<br/>- `disable`: never |
-| `region_engine.mito.bloom_filter_index.mem_threshold_on_create` | String | `auto` | Memory threshold for the index creation.<br/>- `auto`: automatically determine the threshold based on the system memory size (default)<br/>- `unlimited`: no memory limit<br/>- `[size]` e.g. `64MB`: fixed memory threshold |
+| `region_engine.mito.inverted_index.create_on_flush` | String | `auto` | Whether to create the index on flush.<br/>- `auto`: automatically<br/>- `disable`: never |
+| `region_engine.mito.inverted_index.create_on_compaction` | String | `auto` | Whether to create the index on compaction.<br/>- `auto`: automatically<br/>- `disable`: never |
+| `region_engine.mito.inverted_index.apply_on_query` | String | `auto` | Whether to apply the index on query<br/>- `auto`: automatically<br/>- `disable`: never |
+| `region_engine.mito.inverted_index.mem_threshold_on_create` | String | `64M` | Memory threshold for performing an external sort during index creation.<br/>Setting to empty will disable external sorting, forcing all sorting operations to happen in memory. |
+| `region_engine.mito.inverted_index.intermediate_path` | String | `""` | File system path to store intermediate files for external sorting (default `{data_home}/index_intermediate`). |
 | `region_engine.mito.memtable` | -- | -- | -- |
 | `region_engine.mito.memtable.type` | String | `time_series` | Memtable type.<br/>- `time_series`: time-series memtable<br/>- `partition_tree`: partition tree memtable (experimental) |
 | `region_engine.mito.memtable.index_max_keys_per_shard` | Integer | `8192` | The max number of keys in one shard.<br/>Only available for `partition_tree` memtable. |
 | `region_engine.mito.memtable.data_freeze_threshold` | Integer | `32768` | The max rows of data inside the actively writing buffer in one shard.<br/>Only available for `partition_tree` memtable. |
 | `region_engine.mito.memtable.fork_dictionary_bytes` | String | `1GiB` | Max dictionary bytes.<br/>Only available for `partition_tree` memtable. |
-| `region_engine.file` | -- | -- | Enable the file engine. |
-| `region_engine.metric` | -- | -- | Metric engine options. |
-| `region_engine.metric.experimental_sparse_primary_key_encoding` | Bool | `false` | Whether to enable the experimental sparse primary key encoding. |
 | `logging` | -- | -- | The logging options. |
-| `logging.dir` | String | `/tmp/greptimedb/logs` | The directory to store the log files. If set to empty, logs will not be written to files. |
-| `logging.level` | String | Unset | The log level. Can be `info`/`debug`/`warn`/`error`. |
+| `logging.dir` | String | `/tmp/greptimedb/logs` | The directory to store the log files. |
+| `logging.level` | String | `None` | The log level. Can be `info`/`debug`/`warn`/`error`. |
 | `logging.enable_otlp_tracing` | Bool | `false` | Enable OTLP tracing. |
-| `logging.otlp_endpoint` | String | `http://localhost:4317` | The OTLP tracing endpoint. |
+| `logging.otlp_endpoint` | String | `None` | The OTLP tracing endpoint. |
 | `logging.append_stdout` | Bool | `true` | Whether to append logs to stdout. |
-| `logging.log_format` | String | `text` | The log format. Can be `text`/`json`. |
-| `logging.max_log_files` | Integer | `720` | The maximum amount of log files. |
 | `logging.tracing_sample_ratio` | -- | -- | The percentage of tracing will be sampled and exported.<br/>Valid range `[0, 1]`, 1 means all traces are sampled, 0 means all traces are not sampled, the default value is 1.<br/>ratio > 1 are treated as 1. Fractions < 0 are treated as 0 |
 | `logging.tracing_sample_ratio.default_ratio` | Float | `1.0` | -- |
-| `logging.slow_query` | -- | -- | The slow query log options. |
-| `logging.slow_query.enable` | Bool | `false` | Whether to enable slow query log. |
-| `logging.slow_query.threshold` | String | Unset | The threshold of slow query. |
-| `logging.slow_query.sample_ratio` | Float | Unset | The sampling ratio of slow query log. The value should be in the range of (0, 1]. |
 | `export_metrics` | -- | -- | The datanode can export its metrics and send to Prometheus compatible service (e.g. send to `greptimedb` itself) from remote-write API.<br/>This is only used for `greptimedb` to export its own metrics internally. It's different from prometheus scrape. |
 | `export_metrics.enable` | Bool | `false` | whether enable export metrics. |
 | `export_metrics.write_interval` | String | `30s` | The interval of export metrics. |
-| `export_metrics.self_import` | -- | -- | For `standalone` mode, `self_import` is recommend to collect metrics generated by itself<br/>You must create the database before enabling it. |
-| `export_metrics.self_import.db` | String | Unset | -- |
+| `export_metrics.self_import` | -- | -- | For `standalone` mode, `self_import` is recommend to collect metrics generated by itself |
+| `export_metrics.self_import.db` | String | `None` | -- |
 | `export_metrics.remote_write` | -- | -- | -- |
-| `export_metrics.remote_write.url` | String | `""` | The url the metrics send to. The url example can be: `http://127.0.0.1:4000/v1/prometheus/write?db=greptime_metrics`. |
+| `export_metrics.remote_write.url` | String | `""` | The url the metrics send to. The url example can be: `http://127.0.0.1:4000/v1/prometheus/write?db=information_schema`. |
 | `export_metrics.remote_write.headers` | InlineTable | -- | HTTP headers of Prometheus remote-write carry. |
 | `tracing` | -- | -- | The tracing options. Only effect when compiled with `tokio-console` feature. |
-| `tracing.tokio_console_addr` | String | Unset | The tokio console address. |
-
-
-### Flownode
-
-| Key | Type | Default | Descriptions |
-| --- | -----| ------- | ----------- |
-| `mode` | String | `distributed` | The running mode of the flownode. It can be `standalone` or `distributed`. |
-| `node_id` | Integer | Unset | The flownode identifier and should be unique in the cluster. |
-| `flow` | -- | -- | flow engine options. |
-| `flow.num_workers` | Integer | `0` | The number of flow worker in flownode.<br/>Not setting(or set to 0) this value will use the number of CPU cores divided by 2. |
-| `grpc` | -- | -- | The gRPC server options. |
-| `grpc.bind_addr` | String | `127.0.0.1:6800` | The address to bind the gRPC server. |
-| `grpc.server_addr` | String | `127.0.0.1:6800` | The address advertised to the metasrv,<br/>and used for connections from outside the host |
-| `grpc.runtime_size` | Integer | `2` | The number of server worker threads. |
-| `grpc.max_recv_message_size` | String | `512MB` | The maximum receive message size for gRPC server. |
-| `grpc.max_send_message_size` | String | `512MB` | The maximum send message size for gRPC server. |
-| `http` | -- | -- | The HTTP server options. |
-| `http.addr` | String | `127.0.0.1:4000` | The address to bind the HTTP server. |
-| `http.timeout` | String | `30s` | HTTP request timeout. Set to 0 to disable timeout. |
-| `http.body_limit` | String | `64MB` | HTTP request body limit.<br/>The following units are supported: `B`, `KB`, `KiB`, `MB`, `MiB`, `GB`, `GiB`, `TB`, `TiB`, `PB`, `PiB`.<br/>Set to 0 to disable limit. |
-| `meta_client` | -- | -- | The metasrv client options. |
-| `meta_client.metasrv_addrs` | Array | -- | The addresses of the metasrv. |
-| `meta_client.timeout` | String | `3s` | Operation timeout. |
-| `meta_client.heartbeat_timeout` | String | `500ms` | Heartbeat timeout. |
-| `meta_client.ddl_timeout` | String | `10s` | DDL timeout. |
-| `meta_client.connect_timeout` | String | `1s` | Connect server timeout. |
-| `meta_client.tcp_nodelay` | Bool | `true` | `TCP_NODELAY` option for accepted connections. |
-| `meta_client.metadata_cache_max_capacity` | Integer | `100000` | The configuration about the cache of the metadata. |
-| `meta_client.metadata_cache_ttl` | String | `10m` | TTL of the metadata cache. |
-| `meta_client.metadata_cache_tti` | String | `5m` | -- |
-| `heartbeat` | -- | -- | The heartbeat options. |
-| `heartbeat.interval` | String | `3s` | Interval for sending heartbeat messages to the metasrv. |
-| `heartbeat.retry_interval` | String | `3s` | Interval for retrying to send heartbeat messages to the metasrv. |
-| `logging` | -- | -- | The logging options. |
-| `logging.dir` | String | `/tmp/greptimedb/logs` | The directory to store the log files. If set to empty, logs will not be written to files. |
-| `logging.level` | String | Unset | The log level. Can be `info`/`debug`/`warn`/`error`. |
-| `logging.enable_otlp_tracing` | Bool | `false` | Enable OTLP tracing. |
-| `logging.otlp_endpoint` | String | `http://localhost:4317` | The OTLP tracing endpoint. |
-| `logging.append_stdout` | Bool | `true` | Whether to append logs to stdout. |
-| `logging.log_format` | String | `text` | The log format. Can be `text`/`json`. |
-| `logging.max_log_files` | Integer | `720` | The maximum amount of log files. |
-| `logging.tracing_sample_ratio` | -- | -- | The percentage of tracing will be sampled and exported.<br/>Valid range `[0, 1]`, 1 means all traces are sampled, 0 means all traces are not sampled, the default value is 1.<br/>ratio > 1 are treated as 1. Fractions < 0 are treated as 0 |
-| `logging.tracing_sample_ratio.default_ratio` | Float | `1.0` | -- |
-| `logging.slow_query` | -- | -- | The slow query log options. |
-| `logging.slow_query.enable` | Bool | `false` | Whether to enable slow query log. |
-| `logging.slow_query.threshold` | String | Unset | The threshold of slow query. |
-| `logging.slow_query.sample_ratio` | Float | Unset | The sampling ratio of slow query log. The value should be in the range of (0, 1]. |
-| `tracing` | -- | -- | The tracing options. Only effect when compiled with `tokio-console` feature. |
-| `tracing.tokio_console_addr` | String | Unset | The tokio console address. |
+| `tracing.tokio_console_addr` | String | `None` | The tokio console address. |
--- a/config/datanode.example.toml
+++ b/config/datanode.example.toml
@@ -2,7 +2,7 @@
 mode = "standalone"

 ## The datanode identifier and should be unique in the cluster.
-## @toml2docs:none-default
+## +toml2docs:none-default
 node_id = 42

 ## Start services after regions have obtained leases.
@@ -13,64 +13,24 @@ require_lease_before_startup = false
 ## By default, it provides services after all regions have been initialized.
 init_regions_in_background = false

-## Parallelism of initializing regions.
-init_regions_parallelism = 16
+## The gRPC address of the datanode.
+rpc_addr = "127.0.0.1:3001"

-## The maximum current queries allowed to be executed. Zero means unlimited.
-max_concurrent_queries = 0
+## The hostname of the datanode.
+## +toml2docs:none-default
+rpc_hostname = "127.0.0.1"

-## Enable telemetry to collect anonymous usage data. Enabled by default.
-#+ enable_telemetry = true
+## The number of gRPC server worker threads.
+rpc_runtime_size = 8

-## The HTTP server options.
-[http]
-## The address to bind the HTTP server.
-addr = "127.0.0.1:4000"
-## HTTP request timeout. Set to 0 to disable timeout.
-timeout = "30s"
-## HTTP request body limit.
-## The following units are supported: `B`, `KB`, `KiB`, `MB`, `MiB`, `GB`, `GiB`, `TB`, `TiB`, `PB`, `PiB`.
-## Set to 0 to disable limit.
-body_limit = "64MB"
-
-## The gRPC server options.
-[grpc]
-## The address to bind the gRPC server.
-bind_addr = "127.0.0.1:3001"
-## The address advertised to the metasrv, and used for connections from outside the host.
-## If left empty or unset, the server will automatically use the IP address of the first network interface
-## on the host, with the same port number as the one specified in `grpc.bind_addr`.
-server_addr = "127.0.0.1:3001"
-## The number of server worker threads.
-runtime_size = 8
 ## The maximum receive message size for gRPC server.
-max_recv_message_size = "512MB"
+rpc_max_recv_message_size = "512MB"
+
 ## The maximum send message size for gRPC server.
-max_send_message_size = "512MB"
+rpc_max_send_message_size = "512MB"

-## gRPC server TLS options, see `mysql.tls` section.
-[grpc.tls]
-## TLS mode.
-mode = "disable"
-
-## Certificate file path.
-## @toml2docs:none-default
-cert_path = ""
-
-## Private key file path.
-## @toml2docs:none-default
-key_path = ""
-
-## Watch for Certificate and key file change and auto reload.
-## For now, gRPC tls config does not support auto reload.
-watch = false
-
-## The runtime options.
-#+ [runtime]
-## The number of threads to execute the runtime for global read operations.
-#+ global_rt_size = 8
-## The number of threads to execute the runtime for global write operations.
-#+ compact_rt_size = 4
+## Enable telemetry to collect anonymous usage data.
+enable_telemetry = true

 ## The heartbeat options.
 [heartbeat]
@@ -118,20 +78,20 @@ provider = "raft_engine"

 ## The directory to store the WAL files.
 ## **It's only used when the provider is `raft_engine`**.
-## @toml2docs:none-default
+## +toml2docs:none-default
 dir = "/tmp/greptimedb/wal"

 ## The size of the WAL segment file.
 ## **It's only used when the provider is `raft_engine`**.
-file_size = "128MB"
+file_size = "256MB"

 ## The threshold of the WAL size to trigger a flush.
 ## **It's only used when the provider is `raft_engine`**.
-purge_threshold = "1GB"
+purge_threshold = "4GB"

 ## The interval to trigger a flush.
 ## **It's only used when the provider is `raft_engine`**.
-purge_interval = "1m"
+purge_interval = "10m"

 ## The read batch size.
 ## **It's only used when the provider is `raft_engine`**.
@@ -153,9 +113,6 @@ prefill_log_files = false
 ## **It's only used when the provider is `raft_engine`**.
 sync_period = "10s"

-## Parallelism during WAL recovery.
-recovery_parallelism = 2
-
 ## The Kafka broker endpoints.
 ## **It's only used when the provider is `kafka`**.
 broker_endpoints = ["127.0.0.1:9092"]
@@ -163,7 +120,11 @@ broker_endpoints = ["127.0.0.1:9092"]
 ## The max size of a single producer batch.
 ## Warning: Kafka has a default limit of 1MB per message in a topic.
 ## **It's only used when the provider is `kafka`**.
-max_batch_bytes = "1MB"
+max_batch_size = "1MB"
+
+## The linger duration of a kafka batch producer.
+## **It's only used when the provider is `kafka`**.
+linger = "200ms"

 ## The consumer wait timeout.
 ## **It's only used when the provider is `kafka`**.
@@ -185,43 +146,6 @@ backoff_base = 2
 ## **It's only used when the provider is `kafka`**.
 backoff_deadline = "5mins"

-## Whether to enable WAL index creation.
-## **It's only used when the provider is `kafka`**.
-create_index = true
-
-## The interval for dumping WAL indexes.
-## **It's only used when the provider is `kafka`**.
-dump_index_interval = "60s"
-
-## Ignore missing entries during read WAL.
-## **It's only used when the provider is `kafka`**.
-##
-## This option ensures that when Kafka messages are deleted, the system
-## can still successfully replay memtable data without throwing an
-## out-of-range error.
-## However, enabling this option might lead to unexpected data loss,
-## as the system will skip over missing entries instead of treating
-## them as critical errors.
-overwrite_entry_start_id = false
-
-# The Kafka SASL configuration.
-# **It's only used when the provider is `kafka`**.
-# Available SASL mechanisms:
-# - `PLAIN`
-# - `SCRAM-SHA-256`
-# - `SCRAM-SHA-512`
-# [wal.sasl]
-# type = "SCRAM-SHA-512"
-# username = "user_kafka"
-# password = "secret"
-
-# The Kafka TLS configuration.
-# **It's only used when the provider is `kafka`**.
-# [wal.tls]
-# server_ca_cert_path = "/path/to/server_cert"
-# client_cert_path = "/path/to/client_cert"
-# client_key_path = "/path/to/key"
-
 # Example of using S3 as the storage.
 # [storage]
 # type = "S3"
@@ -231,7 +155,6 @@ overwrite_entry_start_id = false
 # secret_access_key = "123456"
 # endpoint = "https://s3.amazonaws.com"
 # region = "us-west-2"
-# enable_virtual_host_style = false

 # Example of using Oss as the storage.
 # [storage]
@@ -259,7 +182,6 @@ overwrite_entry_start_id = false
 # root = "data"
 # scope = "test"
 # credential_path = "123456"
-# credential = "base64-credential"
 # endpoint = "https://storage.googleapis.com"

 ## The data storage options.
@@ -275,123 +197,87 @@ data_home = "/tmp/greptimedb/"
 ## - `Oss`: the data is stored in the Aliyun OSS.
 type = "File"

-## Read cache configuration for object storage such as 'S3' etc, it's configured by default when using object storage. It is recommended to configure it when using object storage for better performance.
-## A local file directory, defaults to `{data_home}`. An empty string means disabling.
-## @toml2docs:none-default
-#+ cache_path = ""
+## Cache configuration for object storage such as 'S3' etc.
+## The local file cache directory.
+## +toml2docs:none-default
+cache_path = "/path/local_cache"

-## The local file cache capacity in bytes. If your disk space is sufficient, it is recommended to set it larger.
-## @toml2docs:none-default
-cache_capacity = "5GiB"
+## The local file cache capacity in bytes.
+## +toml2docs:none-default
+cache_capacity = "256MB"

 ## The S3 bucket name.
 ## **It's only used when the storage type is `S3`, `Oss` and `Gcs`**.
-## @toml2docs:none-default
+## +toml2docs:none-default
 bucket = "greptimedb"

 ## The S3 data will be stored in the specified prefix, for example, `s3://${bucket}/${root}`.
 ## **It's only used when the storage type is `S3`, `Oss` and `Azblob`**.
-## @toml2docs:none-default
+## +toml2docs:none-default
 root = "greptimedb"

 ## The access key id of the aws account.
 ## It's **highly recommended** to use AWS IAM roles instead of hardcoding the access key id and secret key.
 ## **It's only used when the storage type is `S3` and `Oss`**.
-## @toml2docs:none-default
+## +toml2docs:none-default
 access_key_id = "test"

 ## The secret access key of the aws account.
 ## It's **highly recommended** to use AWS IAM roles instead of hardcoding the access key id and secret key.
 ## **It's only used when the storage type is `S3`**.
-## @toml2docs:none-default
+## +toml2docs:none-default
 secret_access_key = "test"

 ## The secret access key of the aliyun account.
 ## **It's only used when the storage type is `Oss`**.
-## @toml2docs:none-default
+## +toml2docs:none-default
 access_key_secret = "test"

 ## The account key of the azure account.
 ## **It's only used when the storage type is `Azblob`**.
-## @toml2docs:none-default
+## +toml2docs:none-default
 account_name = "test"

 ## The account key of the azure account.
 ## **It's only used when the storage type is `Azblob`**.
-## @toml2docs:none-default
+## +toml2docs:none-default
 account_key = "test"

 ## The scope of the google cloud storage.
 ## **It's only used when the storage type is `Gcs`**.
-## @toml2docs:none-default
+## +toml2docs:none-default
 scope = "test"

 ## The credential path of the google cloud storage.
 ## **It's only used when the storage type is `Gcs`**.
-## @toml2docs:none-default
+## +toml2docs:none-default
 credential_path = "test"

-## The credential of the google cloud storage.
-## **It's only used when the storage type is `Gcs`**.
-## @toml2docs:none-default
-credential = "base64-credential"
-
 ## The container of the azure account.
 ## **It's only used when the storage type is `Azblob`**.
-## @toml2docs:none-default
+## +toml2docs:none-default
 container = "greptimedb"

 ## The sas token of the azure account.
 ## **It's only used when the storage type is `Azblob`**.
-## @toml2docs:none-default
+## +toml2docs:none-default
 sas_token = ""

 ## The endpoint of the S3 service.
 ## **It's only used when the storage type is `S3`, `Oss`, `Gcs` and `Azblob`**.
-## @toml2docs:none-default
+## +toml2docs:none-default
 endpoint = "https://s3.amazonaws.com"

 ## The region of the S3 service.
 ## **It's only used when the storage type is `S3`, `Oss`, `Gcs` and `Azblob`**.
-## @toml2docs:none-default
+## +toml2docs:none-default
 region = "us-west-2"

-## The http client options to the storage.
-## **It's only used when the storage type is `S3`, `Oss`, `Gcs` and `Azblob`**.
-[storage.http_client]
-
-## The maximum idle connection per host allowed in the pool.
-pool_max_idle_per_host = 1024
-
-## The timeout for only the connect phase of a http client.
-connect_timeout = "30s"
-
-## The total request timeout, applied from when the request starts connecting until the response body has finished.
-## Also considered a total deadline.
-timeout = "30s"
-
-## The timeout for idle sockets being kept-alive.
-pool_idle_timeout = "90s"
-
 # Custom storage options
 # [[storage.providers]]
-# name = "S3"
 # type = "S3"
-# bucket = "greptimedb"
-# root = "data"
-# access_key_id = "test"
-# secret_access_key = "123456"
-# endpoint = "https://s3.amazonaws.com"
-# region = "us-west-2"
 # [[storage.providers]]
-# name = "Gcs"
 # type = "Gcs"
-# bucket = "greptimedb"
-# root = "data"
-# scope = "test"
-# credential_path = "123456"
-# credential = "base64-credential"
-# endpoint = "https://storage.googleapis.com"

 ## The region engine options. You can configure multiple region engines.
 [[region_engine]]
@@ -400,7 +286,7 @@ pool_idle_timeout = "90s"
 [region_engine.mito]

 ## Number of region workers.
-#+ num_workers = 8
+num_workers = 8

 ## Request channel size of each worker.
 worker_channel_size = 128
@@ -414,179 +300,82 @@ manifest_checkpoint_distance = 10
 ## Whether to compress manifest and checkpoint file by gzip (default false).
 compress_manifest = false

-## Max number of running background flush jobs (default: 1/2 of cpu cores).
-## @toml2docs:none-default="Auto"
-#+ max_background_flushes = 4
-
-## Max number of running background compaction jobs (default: 1/4 of cpu cores).
-## @toml2docs:none-default="Auto"
-#+ max_background_compactions = 2
-
-## Max number of running background purge jobs (default: number of cpu cores).
-## @toml2docs:none-default="Auto"
-#+ max_background_purges = 8
+## Max number of running background jobs
+max_background_jobs = 4

 ## Interval to auto flush a region if it has not flushed yet.
 auto_flush_interval = "1h"

 ## Global write buffer size for all regions. If not set, it's default to 1/8 of OS memory with a max limitation of 1GB.
-## @toml2docs:none-default="Auto"
-#+ global_write_buffer_size = "1GB"
+global_write_buffer_size = "1GB"

 ## Global write buffer size threshold to reject write requests. If not set, it's default to 2 times of `global_write_buffer_size`
-## @toml2docs:none-default="Auto"
-#+ global_write_buffer_reject_size = "2GB"
+global_write_buffer_reject_size = "2GB"

 ## Cache size for SST metadata. Setting it to 0 to disable the cache.
 ## If not set, it's default to 1/32 of OS memory with a max limitation of 128MB.
-## @toml2docs:none-default="Auto"
-#+ sst_meta_cache_size = "128MB"
+sst_meta_cache_size = "128MB"

 ## Cache size for vectors and arrow arrays. Setting it to 0 to disable the cache.
 ## If not set, it's default to 1/16 of OS memory with a max limitation of 512MB.
-## @toml2docs:none-default="Auto"
-#+ vector_cache_size = "512MB"
+vector_cache_size = "512MB"

 ## Cache size for pages of SST row groups. Setting it to 0 to disable the cache.
-## If not set, it's default to 1/8 of OS memory.
-## @toml2docs:none-default="Auto"
-#+ page_cache_size = "512MB"
-
-## Cache size for time series selector (e.g. `last_value()`). Setting it to 0 to disable the cache.
 ## If not set, it's default to 1/16 of OS memory with a max limitation of 512MB.
-## @toml2docs:none-default="Auto"
-#+ selector_result_cache_size = "512MB"
+page_cache_size = "512MB"

-## Whether to enable the write cache, it's enabled by default when using object storage. It is recommended to enable it when using object storage for better performance.
-enable_write_cache = false
+## Whether to enable the experimental write cache.
+enable_experimental_write_cache = false

-## File system path for write cache, defaults to `{data_home}`.
-write_cache_path = ""
+## File system path for write cache, defaults to `{data_home}/write_cache`.
+experimental_write_cache_path = ""

-## Capacity for write cache. If your disk space is sufficient, it is recommended to set it larger.
-write_cache_size = "5GiB"
+## Capacity for write cache.
+experimental_write_cache_size = "512MB"

 ## TTL for write cache.
-## @toml2docs:none-default
-write_cache_ttl = "8h"
+experimental_write_cache_ttl = "1h"

 ## Buffer size for SST writing.
 sst_write_buffer_size = "8MB"

+## Parallelism to scan a region (default: 1/4 of cpu cores).
+## - `0`: using the default value (1/4 of cpu cores).
+## - `1`: scan in current thread.
+## - `n`: scan in parallelism n.
+scan_parallelism = 0
+
 ## Capacity of the channel to send data from parallel scan tasks to the main task.
 parallel_scan_channel_size = 32

 ## Whether to allow stale WAL entries read during replay.
 allow_stale_entries = false

-## Minimum time interval between two compactions.
-## To align with the old behavior, the default value is 0 (no restrictions).
-min_compaction_interval = "0m"
-
-## The options for index in Mito engine.
-[region_engine.mito.index]
-
-## Auxiliary directory path for the index in filesystem, used to store intermediate files for
-## creating the index and staging files for searching the index, defaults to `{data_home}/index_intermediate`.
-## The default name for this directory is `index_intermediate` for backward compatibility.
-##
-## This path contains two subdirectories:
-## - `__intm`: for storing intermediate files used during creating index.
-## - `staging`: for storing staging files used during searching index.
-aux_path = ""
-
-## The max capacity of the staging directory.
-staging_size = "2GB"
-
-## The TTL of the staging directory.
-## Defaults to 7 days.
-## Setting it to "0s" to disable TTL.
-staging_ttl = "7d"
-
-## Cache size for inverted index metadata.
-metadata_cache_size = "64MiB"
-
-## Cache size for inverted index content.
-content_cache_size = "128MiB"
-
-## Page size for inverted index content cache.
-content_cache_page_size = "64KiB"
-
 ## The options for inverted index in Mito engine.
 [region_engine.mito.inverted_index]

 ## Whether to create the index on flush.
-## - `auto`: automatically (default)
+## - `auto`: automatically
 ## - `disable`: never
 create_on_flush = "auto"

 ## Whether to create the index on compaction.
-## - `auto`: automatically (default)
+## - `auto`: automatically
 ## - `disable`: never
 create_on_compaction = "auto"

 ## Whether to apply the index on query
-## - `auto`: automatically (default)
+## - `auto`: automatically
 ## - `disable`: never
 apply_on_query = "auto"

 ## Memory threshold for performing an external sort during index creation.
-## - `auto`: automatically determine the threshold based on the system memory size (default)
-## - `unlimited`: no memory limit
-## - `[size]` e.g. `64MB`: fixed memory threshold
-mem_threshold_on_create = "auto"
+## Setting to empty will disable external sorting, forcing all sorting operations to happen in memory.
+mem_threshold_on_create = "64M"

-## Deprecated, use `region_engine.mito.index.aux_path` instead.
+## File system path to store intermediate files for external sorting (default `{data_home}/index_intermediate`).
 intermediate_path = ""

-## The options for full-text index in Mito engine.
-[region_engine.mito.fulltext_index]
-
-## Whether to create the index on flush.
-## - `auto`: automatically (default)
-## - `disable`: never
-create_on_flush = "auto"
-
-## Whether to create the index on compaction.
-## - `auto`: automatically (default)
-## - `disable`: never
-create_on_compaction = "auto"
-
-## Whether to apply the index on query
-## - `auto`: automatically (default)
-## - `disable`: never
-apply_on_query = "auto"
-
-## Memory threshold for index creation.
-## - `auto`: automatically determine the threshold based on the system memory size (default)
-## - `unlimited`: no memory limit
-## - `[size]` e.g. `64MB`: fixed memory threshold
-mem_threshold_on_create = "auto"
-
-## The options for bloom filter index in Mito engine.
-[region_engine.mito.bloom_filter_index]
-
-## Whether to create the index on flush.
-## - `auto`: automatically (default)
-## - `disable`: never
-create_on_flush = "auto"
-
-## Whether to create the index on compaction.
-## - `auto`: automatically (default)
-## - `disable`: never
-create_on_compaction = "auto"
-
-## Whether to apply the index on query
-## - `auto`: automatically (default)
-## - `disable`: never
-apply_on_query = "auto"
-
-## Memory threshold for the index creation.
-## - `auto`: automatically determine the threshold based on the system memory size (default)
-## - `unlimited`: no memory limit
-## - `[size]` e.g. `64MB`: fixed memory threshold
-mem_threshold_on_create = "auto"
-
 [region_engine.mito.memtable]
 ## Memtable type.
 ## - `time_series`: time-series memtable
@@ -605,59 +394,31 @@ data_freeze_threshold = 32768
 ## Only available for `partition_tree` memtable.
 fork_dictionary_bytes = "1GiB"

-[[region_engine]]
-## Enable the file engine.
-[region_engine.file]
-
-[[region_engine]]
-## Metric engine options.
-[region_engine.metric]
-## Whether to enable the experimental sparse primary key encoding.
-experimental_sparse_primary_key_encoding = false
-
 ## The logging options.
 [logging]
-## The directory to store the log files. If set to empty, logs will not be written to files.
+## The directory to store the log files.
 dir = "/tmp/greptimedb/logs"

 ## The log level. Can be `info`/`debug`/`warn`/`error`.
-## @toml2docs:none-default
+## +toml2docs:none-default
 level = "info"

 ## Enable OTLP tracing.
 enable_otlp_tracing = false

 ## The OTLP tracing endpoint.
-otlp_endpoint = "http://localhost:4317"
+## +toml2docs:none-default
+otlp_endpoint = ""

 ## Whether to append logs to stdout.
 append_stdout = true

-## The log format. Can be `text`/`json`.
-log_format = "text"
-
-## The maximum amount of log files.
-max_log_files = 720
-
 ## The percentage of tracing will be sampled and exported.
 ## Valid range `[0, 1]`, 1 means all traces are sampled, 0 means all traces are not sampled, the default value is 1.
 ## ratio > 1 are treated as 1. Fractions < 0 are treated as 0
 [logging.tracing_sample_ratio]
 default_ratio = 1.0

-## The slow query log options.
-[logging.slow_query]
-## Whether to enable slow query log.
-enable = false
-
-## The threshold of slow query.
-## @toml2docs:none-default
-threshold = "10s"
-
-## The sampling ratio of slow query log. The value should be in the range of (0, 1].
-## @toml2docs:none-default
-sample_ratio = 1.0
-
 ## The datanode can export its metrics and send to Prometheus compatible service (e.g. send to `greptimedb` itself) from remote-write API.
 ## This is only used for `greptimedb` to export its own metrics internally. It's different from prometheus scrape.
 [export_metrics]
@@ -669,20 +430,19 @@ enable = false
 write_interval = "30s"

 ## For `standalone` mode, `self_import` is recommend to collect metrics generated by itself
-## You must create the database before enabling it.
 [export_metrics.self_import]
-## @toml2docs:none-default
-db = "greptime_metrics"
+## +toml2docs:none-default
+db = "information_schema"

 [export_metrics.remote_write]
-## The url the metrics send to. The url example can be: `http://127.0.0.1:4000/v1/prometheus/write?db=greptime_metrics`.
+## The url the metrics send to. The url example can be: `http://127.0.0.1:4000/v1/prometheus/write?db=information_schema`.
 url = ""

 ## HTTP headers of Prometheus remote-write carry.
 headers = { }

 ## The tracing options. Only effect when compiled with `tokio-console` feature.
-#+ [tracing]
+[tracing]
 ## The tokio console address.
-## @toml2docs:none-default
-#+ tokio_console_addr = "127.0.0.1"
+## +toml2docs:none-default
+tokio_console_addr = "127.0.0.1"
--- a/config/flownode.example.toml
+++ b/config/flownode.example.toml
@@ -1,124 +0,0 @@
-## The running mode of the flownode. It can be `standalone` or `distributed`.
-mode = "distributed"
-
-## The flownode identifier and should be unique in the cluster.
-## @toml2docs:none-default
-node_id = 14
-
-## flow engine options.
-[flow]
-## The number of flow worker in flownode.
-## Not setting(or set to 0) this value will use the number of CPU cores divided by 2.
-#+num_workers=0
-
-## The gRPC server options.
-[grpc]
-## The address to bind the gRPC server.
-bind_addr = "127.0.0.1:6800"
-## The address advertised to the metasrv,
-## and used for connections from outside the host
-server_addr = "127.0.0.1:6800"
-## The number of server worker threads.
-runtime_size = 2
-## The maximum receive message size for gRPC server.
-max_recv_message_size = "512MB"
-## The maximum send message size for gRPC server.
-max_send_message_size = "512MB"
-
-## The HTTP server options.
-[http]
-## The address to bind the HTTP server.
-addr = "127.0.0.1:4000"
-## HTTP request timeout. Set to 0 to disable timeout.
-timeout = "30s"
-## HTTP request body limit.
-## The following units are supported: `B`, `KB`, `KiB`, `MB`, `MiB`, `GB`, `GiB`, `TB`, `TiB`, `PB`, `PiB`.
-## Set to 0 to disable limit.
-body_limit = "64MB"
-
-## The metasrv client options.
-[meta_client]
-## The addresses of the metasrv.
-metasrv_addrs = ["127.0.0.1:3002"]
-
-## Operation timeout.
-timeout = "3s"
-
-## Heartbeat timeout.
-heartbeat_timeout = "500ms"
-
-## DDL timeout.
-ddl_timeout = "10s"
-
-## Connect server timeout.
-connect_timeout = "1s"
-
-## `TCP_NODELAY` option for accepted connections.
-tcp_nodelay = true
-
-## The configuration about the cache of the metadata.
-metadata_cache_max_capacity = 100000
-
-## TTL of the metadata cache.
-metadata_cache_ttl = "10m"
-
-# TTI of the metadata cache.
-metadata_cache_tti = "5m"
-
-## The heartbeat options.
-[heartbeat]
-## Interval for sending heartbeat messages to the metasrv.
-interval = "3s"
-
-## Interval for retrying to send heartbeat messages to the metasrv.
-retry_interval = "3s"
-
-## The logging options.
-[logging]
-## The directory to store the log files. If set to empty, logs will not be written to files.
-dir = "/tmp/greptimedb/logs"
-
-## The log level. Can be `info`/`debug`/`warn`/`error`.
-## @toml2docs:none-default
-level = "info"
-
-## Enable OTLP tracing.
-enable_otlp_tracing = false
-
-## The OTLP tracing endpoint.
-otlp_endpoint = "http://localhost:4317"
-
-## Whether to append logs to stdout.
-append_stdout = true
-
-## The log format. Can be `text`/`json`.
-log_format = "text"
-
-## The maximum amount of log files.
-max_log_files = 720
-
-## The percentage of tracing will be sampled and exported.
-## Valid range `[0, 1]`, 1 means all traces are sampled, 0 means all traces are not sampled, the default value is 1.
-## ratio > 1 are treated as 1. Fractions < 0 are treated as 0
-[logging.tracing_sample_ratio]
-default_ratio = 1.0
-
-## The slow query log options.
-[logging.slow_query]
-## Whether to enable slow query log.
-enable = false
-
-## The threshold of slow query.
-## @toml2docs:none-default
-threshold = "10s"
-
-## The sampling ratio of slow query log. The value should be in the range of (0, 1].
-## @toml2docs:none-default
-sample_ratio = 1.0
-
-## The tracing options. Only effect when compiled with `tokio-console` feature.
-#+ [tracing]
-## The tokio console address.
-## @toml2docs:none-default
-#+ tokio_console_addr = "127.0.0.1"
-
--- a/config/frontend.example.toml
+++ b/config/frontend.example.toml
@@ -1,18 +1,10 @@
+## The running mode of the datanode. It can be `standalone` or `distributed`.
+mode = "standalone"
+
 ## The default timezone of the server.
-## @toml2docs:none-default
+## +toml2docs:none-default
 default_timezone = "UTC"

-## The maximum in-flight write bytes.
-## @toml2docs:none-default
-#+ max_in_flight_write_bytes = "500MB"
-
-## The runtime options.
-#+ [runtime]
-## The number of threads to execute the runtime for global read operations.
-#+ global_rt_size = 8
-## The number of threads to execute the runtime for global write operations.
-#+ compact_rt_size = 4
-
 ## The heartbeat options.
 [heartbeat]
 ## Interval for sending heartbeat messages to the metasrv.
@@ -25,27 +17,16 @@ retry_interval = "3s"
 [http]
 ## The address to bind the HTTP server.
 addr = "127.0.0.1:4000"
-## HTTP request timeout. Set to 0 to disable timeout.
+## HTTP request timeout.
 timeout = "30s"
 ## HTTP request body limit.
-## The following units are supported: `B`, `KB`, `KiB`, `MB`, `MiB`, `GB`, `GiB`, `TB`, `TiB`, `PB`, `PiB`.
-## Set to 0 to disable limit.
+## Support the following units are supported: `B`, `KB`, `KiB`, `MB`, `MiB`, `GB`, `GiB`, `TB`, `TiB`, `PB`, `PiB`.
 body_limit = "64MB"
-## HTTP CORS support, it's turned on by default
-## This allows browser to access http APIs without CORS restrictions
-enable_cors = true
-## Customize allowed origins for HTTP CORS.
-## @toml2docs:none-default
-cors_allowed_origins = ["https://example.com"]

 ## The gRPC server options.
 [grpc]
 ## The address to bind the gRPC server.
-bind_addr = "127.0.0.1:4001"
-## The address advertised to the metasrv, and used for connections from outside the host.
-## If left empty or unset, the server will automatically use the IP address of the first network interface
-## on the host, with the same port number as the one specified in `grpc.bind_addr`.
-server_addr = "127.0.0.1:4001"
+addr = "127.0.0.1:4001"
 ## The number of server worker threads.
 runtime_size = 8

@@ -55,11 +36,11 @@ runtime_size = 8
 mode = "disable"

 ## Certificate file path.
-## @toml2docs:none-default
+## +toml2docs:none-default
 cert_path = ""

 ## Private key file path.
-## @toml2docs:none-default
+## +toml2docs:none-default
 key_path = ""

 ## Watch for Certificate and key file change and auto reload.
@@ -74,9 +55,6 @@ enable = true
 addr = "127.0.0.1:4002"
 ## The number of server worker threads.
 runtime_size = 2
-## Server-side keep-alive time.
-## Set to 0 (default) to disable.
-keep_alive = "0s"

 # MySQL server TLS options.
 [mysql.tls]
@@ -90,11 +68,11 @@ keep_alive = "0s"
 mode = "disable"

 ## Certificate file path.
-## @toml2docs:none-default
+## +toml2docs:none-default
 cert_path = ""

 ## Private key file path.
-## @toml2docs:none-default
+## +toml2docs:none-default
 key_path = ""

 ## Watch for Certificate and key file change and auto reload
@@ -108,9 +86,6 @@ enable = true
 addr = "127.0.0.1:4003"
 ## The number of server worker threads.
 runtime_size = 2
-## Server-side keep-alive time.
-## Set to 0 (default) to disable.
-keep_alive = "0s"

 ## PostgresSQL server TLS options, see `mysql.tls` section.
 [postgres.tls]
@@ -118,11 +93,11 @@ keep_alive = "0s"
 mode = "disable"

 ## Certificate file path.
-## @toml2docs:none-default
+## +toml2docs:none-default
 cert_path = ""

 ## Private key file path.
-## @toml2docs:none-default
+## +toml2docs:none-default
 key_path = ""

 ## Watch for Certificate and key file change and auto reload
@@ -138,11 +113,6 @@ enable = true
 ## Whether to enable InfluxDB protocol in HTTP API.
 enable = true

-## Jaeger protocol options.
-[jaeger]
-## Whether to enable Jaeger protocol in HTTP API.
-enable = true
-
 ## Prometheus remote storage options
 [prom_store]
 ## Whether to enable Prometheus remote write and read in HTTP API.
@@ -188,47 +158,29 @@ tcp_nodelay = true

 ## The logging options.
 [logging]
-## The directory to store the log files. If set to empty, logs will not be written to files.
+## The directory to store the log files.
 dir = "/tmp/greptimedb/logs"

 ## The log level. Can be `info`/`debug`/`warn`/`error`.
-## @toml2docs:none-default
+## +toml2docs:none-default
 level = "info"

 ## Enable OTLP tracing.
 enable_otlp_tracing = false

 ## The OTLP tracing endpoint.
-otlp_endpoint = "http://localhost:4317"
+## +toml2docs:none-default
+otlp_endpoint = ""

 ## Whether to append logs to stdout.
 append_stdout = true

-## The log format. Can be `text`/`json`.
-log_format = "text"
-
-## The maximum amount of log files.
-max_log_files = 720
-
 ## The percentage of tracing will be sampled and exported.
 ## Valid range `[0, 1]`, 1 means all traces are sampled, 0 means all traces are not sampled, the default value is 1.
 ## ratio > 1 are treated as 1. Fractions < 0 are treated as 0
 [logging.tracing_sample_ratio]
 default_ratio = 1.0

-## The slow query log options.
-[logging.slow_query]
-## Whether to enable slow query log.
-enable = false
-
-## The threshold of slow query.
-## @toml2docs:none-default
-threshold = "10s"
-
-## The sampling ratio of slow query log. The value should be in the range of (0, 1].
-## @toml2docs:none-default
-sample_ratio = 1.0
-
 ## The datanode can export its metrics and send to Prometheus compatible service (e.g. send to `greptimedb` itself) from remote-write API.
 ## This is only used for `greptimedb` to export its own metrics internally. It's different from prometheus scrape.
 [export_metrics]
@@ -240,20 +192,19 @@ enable = false
 write_interval = "30s"

 ## For `standalone` mode, `self_import` is recommend to collect metrics generated by itself
-## You must create the database before enabling it.
 [export_metrics.self_import]
-## @toml2docs:none-default
-db = "greptime_metrics"
+## +toml2docs:none-default
+db = "information_schema"

 [export_metrics.remote_write]
-## The url the metrics send to. The url example can be: `http://127.0.0.1:4000/v1/prometheus/write?db=greptime_metrics`.
+## The url the metrics send to. The url example can be: `http://127.0.0.1:4000/v1/prometheus/write?db=information_schema`.
 url = ""

 ## HTTP headers of Prometheus remote-write carry.
 headers = { }

 ## The tracing options. Only effect when compiled with `tokio-console` feature.
-#+ [tracing]
+[tracing]
 ## The tokio console address.
-## @toml2docs:none-default
-#+ tokio_console_addr = "127.0.0.1"
+## +toml2docs:none-default
+tokio_console_addr = "127.0.0.1"
--- a/config/metasrv.example.toml
+++ b/config/metasrv.example.toml
@@ -4,64 +4,26 @@ data_home = "/tmp/metasrv/"
 ## The bind address of metasrv.
 bind_addr = "127.0.0.1:3002"

-## The communication server address for the frontend and datanode to connect to metasrv.
-## If left empty or unset, the server will automatically use the IP address of the first network interface
-## on the host, with the same port number as the one specified in `bind_addr`.
+## The communication server address for frontend and datanode to connect to metasrv,  "127.0.0.1:3002" by default for localhost.
 server_addr = "127.0.0.1:3002"

-## Store server address default to etcd store.
-## For postgres store, the format is:
-## "password=password dbname=postgres user=postgres host=localhost port=5432"
-## For etcd store, the format is:
-## "127.0.0.1:2379"
-store_addrs = ["127.0.0.1:2379"]
-
-## If it's not empty, the metasrv will store all data with this key prefix.
-store_key_prefix = ""
-
-## The datastore for meta server.
-## Available values:
-## - `etcd_store` (default value)
-## - `memory_store`
-## - `postgres_store`
-backend = "etcd_store"
-
-## Table name in RDS to store metadata. Effect when using a RDS kvbackend.
-## **Only used when backend is `postgres_store`.**
-meta_table_name = "greptime_metakv"
-
-## Advisory lock id in PostgreSQL for election. Effect when using PostgreSQL as kvbackend
-## Only used when backend is `postgres_store`.
-meta_election_lock_id = 1
+## Etcd server address.
+store_addr = "127.0.0.1:2379"

 ## Datanode selector type.
-## - `round_robin` (default value)
-## - `lease_based`
+## - `lease_based` (default value).
 ## - `load_based`
 ## For details, please see "https://docs.greptime.com/developer-guide/metasrv/selector".
-selector = "round_robin"
+selector = "lease_based"

 ## Store data in memory.
 use_memory_store = false

-## Whether to enable region failover.
-## This feature is only available on GreptimeDB running on cluster mode and
-## - Using Remote WAL
-## - Using shared storage (e.g., s3).
-enable_region_failover = false
+## Whether to enable greptimedb telemetry.
+enable_telemetry = true

-## Max allowed idle time before removing node info from metasrv memory.
-node_max_idle_time = "24hours"
-
-## Whether to enable greptimedb telemetry. Enabled by default.
-#+ enable_telemetry = true
-
-## The runtime options.
-#+ [runtime]
-## The number of threads to execute the runtime for global read operations.
-#+ global_rt_size = 8
-## The number of threads to execute the runtime for global write operations.
-#+ compact_rt_size = 4
+## If it's not empty, the metasrv will store all data with this key prefix.
+store_key_prefix = ""

 ## Procedure storage options.
 [procedure]
@@ -81,32 +43,17 @@ max_metadata_value_size = "1500KiB"

 # Failure detectors options.
 [failure_detector]
-
-## The threshold value used by the failure detector to determine failure conditions.
 threshold = 8.0
-
-## The minimum standard deviation of the heartbeat intervals, used to calculate acceptable variations.
 min_std_deviation = "100ms"
-
-## The acceptable pause duration between heartbeats, used to determine if a heartbeat interval is acceptable.
-acceptable_heartbeat_pause = "10000ms"
-
-## The initial estimate of the heartbeat interval used by the failure detector.
+acceptable_heartbeat_pause = "3000ms"
 first_heartbeat_estimate = "1000ms"

 ## Datanode options.
 [datanode]
-
 ## Datanode client options.
 [datanode.client]
-
-## Operation timeout.
 timeout = "10s"
-
-## Connect server timeout.
 connect_timeout = "10s"
-
-## `TCP_NODELAY` option for accepted connections.
 tcp_nodelay = true

 [wal]
@@ -120,12 +67,7 @@ provider = "raft_engine"
 ## The broker endpoints of the Kafka cluster.
 broker_endpoints = ["127.0.0.1:9092"]

-## Automatically create topics for WAL.
-## Set to `true` to automatically create topics for WAL.
-## Otherwise, use topics named `topic_name_prefix_[0..num_topics)`
-auto_create_topics = true
-
-## Number of topics.
+## Number of topics to be created upon start.
 num_topics = 64

 ## Topic selector type.
@@ -134,9 +76,6 @@ num_topics = 64
 selector_type = "round_robin"

 ## A Kafka topic is constructed by concatenating `topic_name_prefix` and `topic_id`.
-## Only accepts strings that match the following regular expression pattern:
-## [a-zA-Z_:-][a-zA-Z0-9_:\-\.@#]*
-## i.g., greptimedb_wal_topic_0, greptimedb_wal_topic_1.
 topic_name_prefix = "greptimedb_wal_topic"

 ## Expected number of replicas of each partition.
@@ -156,67 +95,31 @@ backoff_base = 2
 ## Stop reconnecting if the total wait time reaches the deadline. If this config is missing, the reconnecting won't terminate.
 backoff_deadline = "5mins"

-# The Kafka SASL configuration.
-# **It's only used when the provider is `kafka`**.
-# Available SASL mechanisms:
-# - `PLAIN`
-# - `SCRAM-SHA-256`
-# - `SCRAM-SHA-512`
-# [wal.sasl]
-# type = "SCRAM-SHA-512"
-# username = "user_kafka"
-# password = "secret"
-
-# The Kafka TLS configuration.
-# **It's only used when the provider is `kafka`**.
-# [wal.tls]
-# server_ca_cert_path = "/path/to/server_cert"
-# client_cert_path = "/path/to/client_cert"
-# client_key_path = "/path/to/key"
-
 ## The logging options.
 [logging]
-## The directory to store the log files. If set to empty, logs will not be written to files.
+## The directory to store the log files.
 dir = "/tmp/greptimedb/logs"

 ## The log level. Can be `info`/`debug`/`warn`/`error`.
-## @toml2docs:none-default
+## +toml2docs:none-default
 level = "info"

 ## Enable OTLP tracing.
 enable_otlp_tracing = false

 ## The OTLP tracing endpoint.
-otlp_endpoint = "http://localhost:4317"
+## +toml2docs:none-default
+otlp_endpoint = ""

 ## Whether to append logs to stdout.
 append_stdout = true

-## The log format. Can be `text`/`json`.
-log_format = "text"
-
-## The maximum amount of log files.
-max_log_files = 720
-
 ## The percentage of tracing will be sampled and exported.
 ## Valid range `[0, 1]`, 1 means all traces are sampled, 0 means all traces are not sampled, the default value is 1.
 ## ratio > 1 are treated as 1. Fractions < 0 are treated as 0
 [logging.tracing_sample_ratio]
 default_ratio = 1.0

-## The slow query log options.
-[logging.slow_query]
-## Whether to enable slow query log.
-enable = false
-
-## The threshold of slow query.
-## @toml2docs:none-default
-threshold = "10s"
-
-## The sampling ratio of slow query log. The value should be in the range of (0, 1].
-## @toml2docs:none-default
-sample_ratio = 1.0
-
 ## The datanode can export its metrics and send to Prometheus compatible service (e.g. send to `greptimedb` itself) from remote-write API.
 ## This is only used for `greptimedb` to export its own metrics internally. It's different from prometheus scrape.
 [export_metrics]
@@ -228,20 +131,19 @@ enable = false
 write_interval = "30s"

 ## For `standalone` mode, `self_import` is recommend to collect metrics generated by itself
-## You must create the database before enabling it.
 [export_metrics.self_import]
-## @toml2docs:none-default
-db = "greptime_metrics"
+## +toml2docs:none-default
+db = "information_schema"

 [export_metrics.remote_write]
-## The url the metrics send to. The url example can be: `http://127.0.0.1:4000/v1/prometheus/write?db=greptime_metrics`.
+## The url the metrics send to. The url example can be: `http://127.0.0.1:4000/v1/prometheus/write?db=information_schema`.
 url = ""

 ## HTTP headers of Prometheus remote-write carry.
 headers = { }

 ## The tracing options. Only effect when compiled with `tokio-console` feature.
-#+ [tracing]
+[tracing]
 ## The tokio console address.
-## @toml2docs:none-default
-#+ tokio_console_addr = "127.0.0.1"
+## +toml2docs:none-default
+tokio_console_addr = "127.0.0.1"
--- a/config/standalone.example.toml
+++ b/config/standalone.example.toml
@@ -1,55 +1,27 @@
 ## The running mode of the datanode. It can be `standalone` or `distributed`.
 mode = "standalone"

+## Enable telemetry to collect anonymous usage data.
+enable_telemetry = true
+
 ## The default timezone of the server.
-## @toml2docs:none-default
+## +toml2docs:none-default
 default_timezone = "UTC"

-## Initialize all regions in the background during the startup.
-## By default, it provides services after all regions have been initialized.
-init_regions_in_background = false
-
-## Parallelism of initializing regions.
-init_regions_parallelism = 16
-
-## The maximum current queries allowed to be executed. Zero means unlimited.
-max_concurrent_queries = 0
-
-## Enable telemetry to collect anonymous usage data. Enabled by default.
-#+ enable_telemetry = true
-
-## The maximum in-flight write bytes.
-## @toml2docs:none-default
-#+ max_in_flight_write_bytes = "500MB"
-
-## The runtime options.
-#+ [runtime]
-## The number of threads to execute the runtime for global read operations.
-#+ global_rt_size = 8
-## The number of threads to execute the runtime for global write operations.
-#+ compact_rt_size = 4
-
 ## The HTTP server options.
 [http]
 ## The address to bind the HTTP server.
 addr = "127.0.0.1:4000"
-## HTTP request timeout. Set to 0 to disable timeout.
+## HTTP request timeout.
 timeout = "30s"
 ## HTTP request body limit.
-## The following units are supported: `B`, `KB`, `KiB`, `MB`, `MiB`, `GB`, `GiB`, `TB`, `TiB`, `PB`, `PiB`.
-## Set to 0 to disable limit.
+## Support the following units are supported: `B`, `KB`, `KiB`, `MB`, `MiB`, `GB`, `GiB`, `TB`, `TiB`, `PB`, `PiB`.
 body_limit = "64MB"
-## HTTP CORS support, it's turned on by default
-## This allows browser to access http APIs without CORS restrictions
-enable_cors = true
-## Customize allowed origins for HTTP CORS.
-## @toml2docs:none-default
-cors_allowed_origins = ["https://example.com"]

 ## The gRPC server options.
 [grpc]
 ## The address to bind the gRPC server.
-bind_addr = "127.0.0.1:4001"
+addr = "127.0.0.1:4001"
 ## The number of server worker threads.
 runtime_size = 8

@@ -59,11 +31,11 @@ runtime_size = 8
 mode = "disable"

 ## Certificate file path.
-## @toml2docs:none-default
+## +toml2docs:none-default
 cert_path = ""

 ## Private key file path.
-## @toml2docs:none-default
+## +toml2docs:none-default
 key_path = ""

 ## Watch for Certificate and key file change and auto reload.
@@ -78,9 +50,6 @@ enable = true
 addr = "127.0.0.1:4002"
 ## The number of server worker threads.
 runtime_size = 2
-## Server-side keep-alive time.
-## Set to 0 (default) to disable.
-keep_alive = "0s"

 # MySQL server TLS options.
 [mysql.tls]
@@ -94,11 +63,11 @@ keep_alive = "0s"
 mode = "disable"

 ## Certificate file path.
-## @toml2docs:none-default
+## +toml2docs:none-default
 cert_path = ""

 ## Private key file path.
-## @toml2docs:none-default
+## +toml2docs:none-default
 key_path = ""

 ## Watch for Certificate and key file change and auto reload
@@ -112,9 +81,6 @@ enable = true
 addr = "127.0.0.1:4003"
 ## The number of server worker threads.
 runtime_size = 2
-## Server-side keep-alive time.
-## Set to 0 (default) to disable.
-keep_alive = "0s"

 ## PostgresSQL server TLS options, see `mysql.tls` section.
 [postgres.tls]
@@ -122,11 +88,11 @@ keep_alive = "0s"
 mode = "disable"

 ## Certificate file path.
-## @toml2docs:none-default
+## +toml2docs:none-default
 cert_path = ""

 ## Private key file path.
-## @toml2docs:none-default
+## +toml2docs:none-default
 key_path = ""

 ## Watch for Certificate and key file change and auto reload
@@ -142,11 +108,6 @@ enable = true
 ## Whether to enable InfluxDB protocol in HTTP API.
 enable = true

-## Jaeger protocol options.
-[jaeger]
-## Whether to enable Jaeger protocol in HTTP API.
-enable = true
-
 ## Prometheus remote storage options
 [prom_store]
 ## Whether to enable Prometheus remote write and read in HTTP API.
@@ -163,20 +124,20 @@ provider = "raft_engine"

 ## The directory to store the WAL files.
 ## **It's only used when the provider is `raft_engine`**.
-## @toml2docs:none-default
+## +toml2docs:none-default
 dir = "/tmp/greptimedb/wal"

 ## The size of the WAL segment file.
 ## **It's only used when the provider is `raft_engine`**.
-file_size = "128MB"
+file_size = "256MB"

-## The threshold of the WAL size to trigger a purge.
+## The threshold of the WAL size to trigger a flush.
 ## **It's only used when the provider is `raft_engine`**.
-purge_threshold = "1GB"
+purge_threshold = "4GB"

-## The interval to trigger a purge.
+## The interval to trigger a flush.
 ## **It's only used when the provider is `raft_engine`**.
-purge_interval = "1m"
+purge_interval = "10m"

 ## The read batch size.
 ## **It's only used when the provider is `raft_engine`**.
@@ -198,45 +159,18 @@ prefill_log_files = false
 ## **It's only used when the provider is `raft_engine`**.
 sync_period = "10s"

-## Parallelism during WAL recovery.
-recovery_parallelism = 2
-
 ## The Kafka broker endpoints.
 ## **It's only used when the provider is `kafka`**.
 broker_endpoints = ["127.0.0.1:9092"]

-## Automatically create topics for WAL.
-## Set to `true` to automatically create topics for WAL.
-## Otherwise, use topics named `topic_name_prefix_[0..num_topics)`
-auto_create_topics = true
-
-## Number of topics.
-## **It's only used when the provider is `kafka`**.
-num_topics = 64
-
-## Topic selector type.
-## Available selector types:
-## - `round_robin` (default)
-## **It's only used when the provider is `kafka`**.
-selector_type = "round_robin"
-
-## A Kafka topic is constructed by concatenating `topic_name_prefix` and `topic_id`.
-## i.g., greptimedb_wal_topic_0, greptimedb_wal_topic_1.
-## **It's only used when the provider is `kafka`**.
-topic_name_prefix = "greptimedb_wal_topic"
-
-## Expected number of replicas of each partition.
-## **It's only used when the provider is `kafka`**.
-replication_factor = 1
-
-## Above which a topic creation operation will be cancelled.
-## **It's only used when the provider is `kafka`**.
-create_topic_timeout = "30s"
-
 ## The max size of a single producer batch.
 ## Warning: Kafka has a default limit of 1MB per message in a topic.
 ## **It's only used when the provider is `kafka`**.
-max_batch_bytes = "1MB"
+max_batch_size = "1MB"
+
+## The linger duration of a kafka batch producer.
+## **It's only used when the provider is `kafka`**.
+linger = "200ms"

 ## The consumer wait timeout.
 ## **It's only used when the provider is `kafka`**.
@@ -258,43 +192,12 @@ backoff_base = 2
 ## **It's only used when the provider is `kafka`**.
 backoff_deadline = "5mins"

-## Ignore missing entries during read WAL.
-## **It's only used when the provider is `kafka`**.
-##
-## This option ensures that when Kafka messages are deleted, the system
-## can still successfully replay memtable data without throwing an
-## out-of-range error.
-## However, enabling this option might lead to unexpected data loss,
-## as the system will skip over missing entries instead of treating
-## them as critical errors.
-overwrite_entry_start_id = false
-
-# The Kafka SASL configuration.
-# **It's only used when the provider is `kafka`**.
-# Available SASL mechanisms:
-# - `PLAIN`
-# - `SCRAM-SHA-256`
-# - `SCRAM-SHA-512`
-# [wal.sasl]
-# type = "SCRAM-SHA-512"
-# username = "user_kafka"
-# password = "secret"
-
-# The Kafka TLS configuration.
-# **It's only used when the provider is `kafka`**.
-# [wal.tls]
-# server_ca_cert_path = "/path/to/server_cert"
-# client_cert_path = "/path/to/client_cert"
-# client_key_path = "/path/to/key"
-
 ## Metadata storage options.
 [metadata_store]
-## The size of the metadata store log file.
-file_size = "64MB"
-## The threshold of the metadata store size to trigger a purge.
-purge_threshold = "256MB"
-## The interval of the metadata store to trigger a purge.
-purge_interval = "1m"
+## Kv file size in bytes.
+file_size = "256MB"
+## Kv purge threshold.
+purge_threshold = "4GB"

 ## Procedure storage options.
 [procedure]
@@ -303,12 +206,6 @@ max_retry_times = 3
 ## Initial retry delay of procedures, increases exponentially
 retry_delay = "500ms"

-## flow engine options.
-[flow]
-## The number of flow worker in flownode.
-## Not setting(or set to 0) this value will use the number of CPU cores divided by 2.
-#+num_workers=0
-
 # Example of using S3 as the storage.
 # [storage]
 # type = "S3"
@@ -318,7 +215,6 @@ retry_delay = "500ms"
 # secret_access_key = "123456"
 # endpoint = "https://s3.amazonaws.com"
 # region = "us-west-2"
-# enable_virtual_host_style = false

 # Example of using Oss as the storage.
 # [storage]
@@ -346,7 +242,6 @@ retry_delay = "500ms"
 # root = "data"
 # scope = "test"
 # credential_path = "123456"
-# credential = "base64-credential"
 # endpoint = "https://storage.googleapis.com"

 ## The data storage options.
@@ -362,123 +257,87 @@ data_home = "/tmp/greptimedb/"
 ## - `Oss`: the data is stored in the Aliyun OSS.
 type = "File"

-## Read cache configuration for object storage such as 'S3' etc, it's configured by default when using object storage. It is recommended to configure it when using object storage for better performance.
-## A local file directory, defaults to `{data_home}`. An empty string means disabling.
-## @toml2docs:none-default
-#+ cache_path = ""
+## Cache configuration for object storage such as 'S3' etc.
+## The local file cache directory.
+## +toml2docs:none-default
+cache_path = "/path/local_cache"

-## The local file cache capacity in bytes. If your disk space is sufficient, it is recommended to set it larger.
-## @toml2docs:none-default
-cache_capacity = "5GiB"
+## The local file cache capacity in bytes.
+## +toml2docs:none-default
+cache_capacity = "256MB"

 ## The S3 bucket name.
 ## **It's only used when the storage type is `S3`, `Oss` and `Gcs`**.
-## @toml2docs:none-default
+## +toml2docs:none-default
 bucket = "greptimedb"

 ## The S3 data will be stored in the specified prefix, for example, `s3://${bucket}/${root}`.
 ## **It's only used when the storage type is `S3`, `Oss` and `Azblob`**.
-## @toml2docs:none-default
+## +toml2docs:none-default
 root = "greptimedb"

 ## The access key id of the aws account.
 ## It's **highly recommended** to use AWS IAM roles instead of hardcoding the access key id and secret key.
 ## **It's only used when the storage type is `S3` and `Oss`**.
-## @toml2docs:none-default
+## +toml2docs:none-default
 access_key_id = "test"

 ## The secret access key of the aws account.
 ## It's **highly recommended** to use AWS IAM roles instead of hardcoding the access key id and secret key.
 ## **It's only used when the storage type is `S3`**.
-## @toml2docs:none-default
+## +toml2docs:none-default
 secret_access_key = "test"

 ## The secret access key of the aliyun account.
 ## **It's only used when the storage type is `Oss`**.
-## @toml2docs:none-default
+## +toml2docs:none-default
 access_key_secret = "test"

 ## The account key of the azure account.
 ## **It's only used when the storage type is `Azblob`**.
-## @toml2docs:none-default
+## +toml2docs:none-default
 account_name = "test"

 ## The account key of the azure account.
 ## **It's only used when the storage type is `Azblob`**.
-## @toml2docs:none-default
+## +toml2docs:none-default
 account_key = "test"

 ## The scope of the google cloud storage.
 ## **It's only used when the storage type is `Gcs`**.
-## @toml2docs:none-default
+## +toml2docs:none-default
 scope = "test"

 ## The credential path of the google cloud storage.
 ## **It's only used when the storage type is `Gcs`**.
-## @toml2docs:none-default
+## +toml2docs:none-default
 credential_path = "test"

-## The credential of the google cloud storage.
-## **It's only used when the storage type is `Gcs`**.
-## @toml2docs:none-default
-credential = "base64-credential"
-
 ## The container of the azure account.
 ## **It's only used when the storage type is `Azblob`**.
-## @toml2docs:none-default
+## +toml2docs:none-default
 container = "greptimedb"

 ## The sas token of the azure account.
 ## **It's only used when the storage type is `Azblob`**.
-## @toml2docs:none-default
+## +toml2docs:none-default
 sas_token = ""

 ## The endpoint of the S3 service.
 ## **It's only used when the storage type is `S3`, `Oss`, `Gcs` and `Azblob`**.
-## @toml2docs:none-default
+## +toml2docs:none-default
 endpoint = "https://s3.amazonaws.com"

 ## The region of the S3 service.
 ## **It's only used when the storage type is `S3`, `Oss`, `Gcs` and `Azblob`**.
-## @toml2docs:none-default
+## +toml2docs:none-default
 region = "us-west-2"

-## The http client options to the storage.
-## **It's only used when the storage type is `S3`, `Oss`, `Gcs` and `Azblob`**.
-[storage.http_client]
-
-## The maximum idle connection per host allowed in the pool.
-pool_max_idle_per_host = 1024
-
-## The timeout for only the connect phase of a http client.
-connect_timeout = "30s"
-
-## The total request timeout, applied from when the request starts connecting until the response body has finished.
-## Also considered a total deadline.
-timeout = "30s"
-
-## The timeout for idle sockets being kept-alive.
-pool_idle_timeout = "90s"
-
 # Custom storage options
 # [[storage.providers]]
-# name = "S3"
 # type = "S3"
-# bucket = "greptimedb"
-# root = "data"
-# access_key_id = "test"
-# secret_access_key = "123456"
-# endpoint = "https://s3.amazonaws.com"
-# region = "us-west-2"
 # [[storage.providers]]
-# name = "Gcs"
 # type = "Gcs"
-# bucket = "greptimedb"
-# root = "data"
-# scope = "test"
-# credential_path = "123456"
-# credential = "base64-credential"
-# endpoint = "https://storage.googleapis.com"

 ## The region engine options. You can configure multiple region engines.
 [[region_engine]]
@@ -487,7 +346,7 @@ pool_idle_timeout = "90s"
 [region_engine.mito]

 ## Number of region workers.
-#+ num_workers = 8
+num_workers = 8

 ## Request channel size of each worker.
 worker_channel_size = 128
@@ -501,179 +360,82 @@ manifest_checkpoint_distance = 10
 ## Whether to compress manifest and checkpoint file by gzip (default false).
 compress_manifest = false

-## Max number of running background flush jobs (default: 1/2 of cpu cores).
-## @toml2docs:none-default="Auto"
-#+ max_background_flushes = 4
-
-## Max number of running background compaction jobs (default: 1/4 of cpu cores).
-## @toml2docs:none-default="Auto"
-#+ max_background_compactions = 2
-
-## Max number of running background purge jobs (default: number of cpu cores).
-## @toml2docs:none-default="Auto"
-#+ max_background_purges = 8
+## Max number of running background jobs
+max_background_jobs = 4

 ## Interval to auto flush a region if it has not flushed yet.
 auto_flush_interval = "1h"

 ## Global write buffer size for all regions. If not set, it's default to 1/8 of OS memory with a max limitation of 1GB.
-## @toml2docs:none-default="Auto"
-#+ global_write_buffer_size = "1GB"
+global_write_buffer_size = "1GB"

-## Global write buffer size threshold to reject write requests. If not set, it's default to 2 times of `global_write_buffer_size`.
-## @toml2docs:none-default="Auto"
-#+ global_write_buffer_reject_size = "2GB"
+## Global write buffer size threshold to reject write requests. If not set, it's default to 2 times of `global_write_buffer_size`
+global_write_buffer_reject_size = "2GB"

 ## Cache size for SST metadata. Setting it to 0 to disable the cache.
 ## If not set, it's default to 1/32 of OS memory with a max limitation of 128MB.
-## @toml2docs:none-default="Auto"
-#+ sst_meta_cache_size = "128MB"
+sst_meta_cache_size = "128MB"

 ## Cache size for vectors and arrow arrays. Setting it to 0 to disable the cache.
 ## If not set, it's default to 1/16 of OS memory with a max limitation of 512MB.
-## @toml2docs:none-default="Auto"
-#+ vector_cache_size = "512MB"
+vector_cache_size = "512MB"

 ## Cache size for pages of SST row groups. Setting it to 0 to disable the cache.
-## If not set, it's default to 1/8 of OS memory.
-## @toml2docs:none-default="Auto"
-#+ page_cache_size = "512MB"
-
-## Cache size for time series selector (e.g. `last_value()`). Setting it to 0 to disable the cache.
 ## If not set, it's default to 1/16 of OS memory with a max limitation of 512MB.
-## @toml2docs:none-default="Auto"
-#+ selector_result_cache_size = "512MB"
+page_cache_size = "512MB"

-## Whether to enable the write cache, it's enabled by default when using object storage. It is recommended to enable it when using object storage for better performance.
-enable_write_cache = false
+## Whether to enable the experimental write cache.
+enable_experimental_write_cache = false

-## File system path for write cache, defaults to `{data_home}`.
-write_cache_path = ""
+## File system path for write cache, defaults to `{data_home}/write_cache`.
+experimental_write_cache_path = ""

-## Capacity for write cache. If your disk space is sufficient, it is recommended to set it larger.
-write_cache_size = "5GiB"
+## Capacity for write cache.
+experimental_write_cache_size = "512MB"

 ## TTL for write cache.
-## @toml2docs:none-default
-write_cache_ttl = "8h"
+experimental_write_cache_ttl = "1h"

 ## Buffer size for SST writing.
 sst_write_buffer_size = "8MB"

+## Parallelism to scan a region (default: 1/4 of cpu cores).
+## - `0`: using the default value (1/4 of cpu cores).
+## - `1`: scan in current thread.
+## - `n`: scan in parallelism n.
+scan_parallelism = 0
+
 ## Capacity of the channel to send data from parallel scan tasks to the main task.
 parallel_scan_channel_size = 32

 ## Whether to allow stale WAL entries read during replay.
 allow_stale_entries = false

-## Minimum time interval between two compactions.
-## To align with the old behavior, the default value is 0 (no restrictions).
-min_compaction_interval = "0m"
-
-## The options for index in Mito engine.
-[region_engine.mito.index]
-
-## Auxiliary directory path for the index in filesystem, used to store intermediate files for
-## creating the index and staging files for searching the index, defaults to `{data_home}/index_intermediate`.
-## The default name for this directory is `index_intermediate` for backward compatibility.
-##
-## This path contains two subdirectories:
-## - `__intm`: for storing intermediate files used during creating index.
-## - `staging`: for storing staging files used during searching index.
-aux_path = ""
-
-## The max capacity of the staging directory.
-staging_size = "2GB"
-
-## The TTL of the staging directory.
-## Defaults to 7 days.
-## Setting it to "0s" to disable TTL.
-staging_ttl = "7d"
-
-## Cache size for inverted index metadata.
-metadata_cache_size = "64MiB"
-
-## Cache size for inverted index content.
-content_cache_size = "128MiB"
-
-## Page size for inverted index content cache.
-content_cache_page_size = "64KiB"
-
 ## The options for inverted index in Mito engine.
 [region_engine.mito.inverted_index]

 ## Whether to create the index on flush.
-## - `auto`: automatically (default)
+## - `auto`: automatically
 ## - `disable`: never
 create_on_flush = "auto"

 ## Whether to create the index on compaction.
-## - `auto`: automatically (default)
+## - `auto`: automatically
 ## - `disable`: never
 create_on_compaction = "auto"

 ## Whether to apply the index on query
-## - `auto`: automatically (default)
+## - `auto`: automatically
 ## - `disable`: never
 apply_on_query = "auto"

 ## Memory threshold for performing an external sort during index creation.
-## - `auto`: automatically determine the threshold based on the system memory size (default)
-## - `unlimited`: no memory limit
-## - `[size]` e.g. `64MB`: fixed memory threshold
-mem_threshold_on_create = "auto"
+## Setting to empty will disable external sorting, forcing all sorting operations to happen in memory.
+mem_threshold_on_create = "64M"

-## Deprecated, use `region_engine.mito.index.aux_path` instead.
+## File system path to store intermediate files for external sorting (default `{data_home}/index_intermediate`).
 intermediate_path = ""

-## The options for full-text index in Mito engine.
-[region_engine.mito.fulltext_index]
-
-## Whether to create the index on flush.
-## - `auto`: automatically (default)
-## - `disable`: never
-create_on_flush = "auto"
-
-## Whether to create the index on compaction.
-## - `auto`: automatically (default)
-## - `disable`: never
-create_on_compaction = "auto"
-
-## Whether to apply the index on query
-## - `auto`: automatically (default)
-## - `disable`: never
-apply_on_query = "auto"
-
-## Memory threshold for index creation.
-## - `auto`: automatically determine the threshold based on the system memory size (default)
-## - `unlimited`: no memory limit
-## - `[size]` e.g. `64MB`: fixed memory threshold
-mem_threshold_on_create = "auto"
-
-## The options for bloom filter in Mito engine.
-[region_engine.mito.bloom_filter_index]
-
-## Whether to create the bloom filter on flush.
-## - `auto`: automatically (default)
-## - `disable`: never
-create_on_flush = "auto"
-
-## Whether to create the bloom filter on compaction.
-## - `auto`: automatically (default)
-## - `disable`: never
-create_on_compaction = "auto"
-
-## Whether to apply the bloom filter on query
-## - `auto`: automatically (default)
-## - `disable`: never
-apply_on_query = "auto"
-
-## Memory threshold for bloom filter creation.
-## - `auto`: automatically determine the threshold based on the system memory size (default)
-## - `unlimited`: no memory limit
-## - `[size]` e.g. `64MB`: fixed memory threshold
-mem_threshold_on_create = "auto"
-
 [region_engine.mito.memtable]
 ## Memtable type.
 ## - `time_series`: time-series memtable
@@ -692,59 +454,31 @@ data_freeze_threshold = 32768
 ## Only available for `partition_tree` memtable.
 fork_dictionary_bytes = "1GiB"

-[[region_engine]]
-## Enable the file engine.
-[region_engine.file]
-
-[[region_engine]]
-## Metric engine options.
-[region_engine.metric]
-## Whether to enable the experimental sparse primary key encoding.
-experimental_sparse_primary_key_encoding = false
-
 ## The logging options.
 [logging]
-## The directory to store the log files. If set to empty, logs will not be written to files.
+## The directory to store the log files.
 dir = "/tmp/greptimedb/logs"

 ## The log level. Can be `info`/`debug`/`warn`/`error`.
-## @toml2docs:none-default
+## +toml2docs:none-default
 level = "info"

 ## Enable OTLP tracing.
 enable_otlp_tracing = false

 ## The OTLP tracing endpoint.
-otlp_endpoint = "http://localhost:4317"
+## +toml2docs:none-default
+otlp_endpoint = ""

 ## Whether to append logs to stdout.
 append_stdout = true

-## The log format. Can be `text`/`json`.
-log_format = "text"
-
-## The maximum amount of log files.
-max_log_files = 720
-
 ## The percentage of tracing will be sampled and exported.
 ## Valid range `[0, 1]`, 1 means all traces are sampled, 0 means all traces are not sampled, the default value is 1.
 ## ratio > 1 are treated as 1. Fractions < 0 are treated as 0
 [logging.tracing_sample_ratio]
 default_ratio = 1.0

-## The slow query log options.
-[logging.slow_query]
-## Whether to enable slow query log.
-enable = false
-
-## The threshold of slow query.
-## @toml2docs:none-default
-threshold = "10s"
-
-## The sampling ratio of slow query log. The value should be in the range of (0, 1].
-## @toml2docs:none-default
-sample_ratio = 1.0
-
 ## The datanode can export its metrics and send to Prometheus compatible service (e.g. send to `greptimedb` itself) from remote-write API.
 ## This is only used for `greptimedb` to export its own metrics internally. It's different from prometheus scrape.
 [export_metrics]
@@ -755,21 +489,20 @@ enable = false
 ## The interval of export metrics.
 write_interval = "30s"

-## For `standalone` mode, `self_import` is recommended to collect metrics generated by itself
-## You must create the database before enabling it.
+## For `standalone` mode, `self_import` is recommend to collect metrics generated by itself
 [export_metrics.self_import]
-## @toml2docs:none-default
-db = "greptime_metrics"
+## +toml2docs:none-default
+db = "information_schema"

 [export_metrics.remote_write]
-## The url the metrics send to. The url example can be: `http://127.0.0.1:4000/v1/prometheus/write?db=greptime_metrics`.
+## The url the metrics send to. The url example can be: `http://127.0.0.1:4000/v1/prometheus/write?db=information_schema`.
 url = ""

 ## HTTP headers of Prometheus remote-write carry.
 headers = { }

 ## The tracing options. Only effect when compiled with `tokio-console` feature.
-#+ [tracing]
+[tracing]
 ## The tokio console address.
-## @toml2docs:none-default
-#+ tokio_console_addr = "127.0.0.1"
+## +toml2docs:none-default
+tokio_console_addr = "127.0.0.1"
--- a/cyborg/bin/bump-doc-version.ts
+++ b/cyborg/bin/bump-doc-version.ts
@@ -1,75 +0,0 @@
-/*
- * Copyright 2023 Greptime Team
- *
- * Licensed under the Apache License, Version 2.0 (the "License");
- * you may not use this file except in compliance with the License.
- * You may obtain a copy of the License at
- *
- *     http://www.apache.org/licenses/LICENSE-2.0
- *
- * Unless required by applicable law or agreed to in writing, software
- * distributed under the License is distributed on an "AS IS" BASIS,
- * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
- * See the License for the specific language governing permissions and
- * limitations under the License.
- */
-
-import * as core from "@actions/core";
-import {obtainClient} from "@/common";
-
-async function triggerWorkflow(workflowId: string, version: string) {
-  const docsClient = obtainClient("DOCS_REPO_TOKEN")
-  try {
-    await docsClient.rest.actions.createWorkflowDispatch({
-      owner: "GreptimeTeam",
-      repo: "docs",
-      workflow_id: workflowId,
-      ref: "main",
-      inputs: {
-        version,
-      },
-    });
-    console.log(`Successfully triggered ${workflowId} workflow with version ${version}`);
-  } catch (error) {
-    core.setFailed(`Failed to trigger workflow: ${error.message}`);
-  }
-}
-
-function determineWorkflow(version: string): [string, string] {
-  // Check if it's a nightly version
-  if (version.includes('nightly')) {
-    return ['bump-nightly-version.yml', version];
-  }
-
-  const parts = version.split('.');
-
-  if (parts.length !== 3) {
-    throw new Error('Invalid version format');
-  }
-
-  // If patch version (last number) is 0, it's a major version
-  // Return only major.minor version
-  if (parts[2] === '0') {
-    return ['bump-version.yml', `${parts[0]}.${parts[1]}`];
-  }
-
-  // Otherwise it's a patch version, use full version
-  return ['bump-patch-version.yml', version];
-}
-
-const version = process.env.VERSION;
-if (!version) {
-  core.setFailed("VERSION environment variable is required");
-  process.exit(1);
-}
-
-// Remove 'v' prefix if exists
-const cleanVersion = version.startsWith('v') ? version.slice(1) : version;
-
-try {
-  const [workflowId, apiVersion] = determineWorkflow(cleanVersion);
-  triggerWorkflow(workflowId, apiVersion);
-} catch (error) {
-  core.setFailed(`Error processing version: ${error.message}`);
-  process.exit(1);
-}
--- a/docker/buildx/centos/Dockerfile
+++ b/docker/buildx/centos/Dockerfile
@@ -13,6 +13,8 @@ RUN yum install -y epel-release  \
    openssl \
    openssl-devel  \
    centos-release-scl  \
+    rh-python38  \
+    rh-python38-python-devel \
    which

 # Install protoc
@@ -22,7 +24,7 @@ RUN unzip protoc-3.15.8-linux-x86_64.zip -d /usr/local/
 # Install Rust
 SHELL ["/bin/bash", "-c"]
 RUN curl --proto '=https' --tlsv1.2 -sSf https://sh.rustup.rs | sh -s -- --no-modify-path --default-toolchain none -y
-ENV PATH /usr/local/bin:/root/.cargo/bin/:$PATH
+ENV PATH /opt/rh/rh-python38/root/usr/bin:/usr/local/bin:/root/.cargo/bin/:$PATH

 # Build the project in release mode.
 RUN --mount=target=.,rw \
@@ -41,6 +43,8 @@ RUN yum install -y epel-release \
    openssl \
    openssl-devel  \
    centos-release-scl  \
+    rh-python38  \
+    rh-python38-python-devel \
    which

 WORKDIR /greptime
--- a/docker/buildx/ubuntu/Dockerfile
+++ b/docker/buildx/ubuntu/Dockerfile
@@ -1,4 +1,4 @@
-FROM ubuntu:22.04 as builder
+FROM ubuntu:20.04 as builder

 ARG CARGO_PROFILE
 ARG FEATURES
@@ -7,8 +7,10 @@ ARG OUTPUT_DIR
 ENV LANG en_US.utf8
 WORKDIR /greptimedb

+# Add PPA for Python 3.10.
 RUN apt-get update && \
-    DEBIAN_FRONTEND=noninteractive apt-get install -y software-properties-common
+    DEBIAN_FRONTEND=noninteractive apt-get install -y software-properties-common && \
+    add-apt-repository ppa:deadsnakes/ppa -y

 # Install dependencies.
 RUN --mount=type=cache,target=/var/cache/apt \
@@ -18,7 +20,10 @@ RUN --mount=type=cache,target=/var/cache/apt \
    curl \
    git \
    build-essential \
-    pkg-config
+    pkg-config \
+    python3.10 \
+    python3.10-dev \
+    python3-pip

 # Install Rust.
 SHELL ["/bin/bash", "-c"]
@@ -41,8 +46,15 @@ ARG OUTPUT_DIR

 RUN apt-get update && DEBIAN_FRONTEND=noninteractive apt-get \
    -y install ca-certificates \
+    python3.10 \
+    python3.10-dev \
+    python3-pip \
    curl

+COPY ./docker/python/requirements.txt /etc/greptime/requirements.txt
+
+RUN python3 -m pip install -r /etc/greptime/requirements.txt
+
 WORKDIR /greptime
 COPY --from=builder /out/target/${OUTPUT_DIR}/greptime /greptime/bin/
 ENV PATH /greptime/bin/:$PATH
--- a/docker/ci/centos/Dockerfile
+++ b/docker/ci/centos/Dockerfile
@@ -1,13 +1,11 @@
 FROM centos:7

-# Note: CentOS 7 has reached EOL since 2024-07-01 thus `mirror.centos.org` is no longer available and we need to use `vault.centos.org` instead.
-RUN sed -i s/mirror.centos.org/vault.centos.org/g /etc/yum.repos.d/*.repo
-RUN sed -i s/^#.*baseurl=http/baseurl=http/g /etc/yum.repos.d/*.repo
-
 RUN yum install -y epel-release \
    openssl \
    openssl-devel  \
-    centos-release-scl
+    centos-release-scl  \
+    rh-python38  \
+    rh-python38-python-devel

 ARG TARGETARCH

--- a/docker/ci/ubuntu/Dockerfile
+++ b/docker/ci/ubuntu/Dockerfile
@@ -8,8 +8,15 @@ ARG TARGET_BIN=greptime

 RUN apt-get update && DEBIAN_FRONTEND=noninteractive apt-get install -y \
    ca-certificates \
+    python3.10 \
+    python3.10-dev \
+    python3-pip \
    curl

+COPY $DOCKER_BUILD_ROOT/docker/python/requirements.txt /etc/greptime/requirements.txt
+
+RUN python3 -m pip install -r /etc/greptime/requirements.txt
+
 ARG TARGETARCH

 ADD $TARGETARCH/$TARGET_BIN /greptime/bin/
--- a/docker/ci/ubuntu/Dockerfile.fuzztests
+++ b/docker/ci/ubuntu/Dockerfile.fuzztests
@@ -1,4 +1,4 @@
-FROM ubuntu:latest
+FROM ubuntu:22.04

 # The binary name of GreptimeDB executable.
 # Defaults to "greptime", but sometimes in other projects it might be different.
--- a/docker/dev-builder/android/Dockerfile
+++ b/docker/dev-builder/android/Dockerfile
@@ -9,20 +9,16 @@ RUN cp ${NDK_ROOT}/toolchains/llvm/prebuilt/linux-x86_64/lib64/clang/14.0.7/lib/
 # Install dependencies.
 RUN apt-get update && apt-get install -y \
    libssl-dev \
+    protobuf-compiler \
    curl \
    git \
-    unzip \
    build-essential \
-    pkg-config
-
-# Install protoc
-ARG PROTOBUF_VERSION=29.3
-
-RUN curl -OL https://github.com/protocolbuffers/protobuf/releases/download/v${PROTOBUF_VERSION}/protoc-${PROTOBUF_VERSION}-linux-x86_64.zip && \
-    unzip protoc-${PROTOBUF_VERSION}-linux-x86_64.zip -d protoc3;
-    
-RUN mv protoc3/bin/* /usr/local/bin/
-RUN mv protoc3/include/* /usr/local/include/
+    pkg-config \
+    python3 \
+    python3-dev \
+    python3-pip \
+    && pip3 install --upgrade pip \
+    && pip3 install pyarrow

 # Trust workdir
 RUN git config --global --add safe.directory /greptimedb
--- a/docker/dev-builder/binstall/pull_binstall.sh
+++ b/docker/dev-builder/binstall/pull_binstall.sh
@@ -1,50 +0,0 @@
-#!/bin/bash
-
-set -euxo pipefail
-
-cd "$(mktemp -d)"
-# Fix version to v1.6.6, this is different than the latest version in original install script in
-# https://raw.githubusercontent.com/cargo-bins/cargo-binstall/main/install-from-binstall-release.sh
-base_url="https://github.com/cargo-bins/cargo-binstall/releases/download/v1.6.6/cargo-binstall-"
-
-os="$(uname -s)"
-if [ "$os" == "Darwin" ]; then
-    url="${base_url}universal-apple-darwin.zip"
-    curl -LO --proto '=https' --tlsv1.2 -sSf "$url"
-    unzip cargo-binstall-universal-apple-darwin.zip
-elif [ "$os" == "Linux" ]; then
-    machine="$(uname -m)"
-    if [ "$machine" == "armv7l" ]; then
-        machine="armv7"
-    fi
-    target="${machine}-unknown-linux-musl"
-    if [ "$machine" == "armv7" ]; then
-        target="${target}eabihf"
-    fi
-
-    url="${base_url}${target}.tgz"
-    curl -L --proto '=https' --tlsv1.2 -sSf "$url" | tar -xvzf -
-elif [ "${OS-}" = "Windows_NT" ]; then
-    machine="$(uname -m)"
-    target="${machine}-pc-windows-msvc"
-    url="${base_url}${target}.zip"
-    curl -LO --proto '=https' --tlsv1.2 -sSf "$url"
-    unzip "cargo-binstall-${target}.zip"
-else
-    echo "Unsupported OS ${os}"
-    exit 1
-fi
-
-./cargo-binstall -y --force cargo-binstall
-
-CARGO_HOME="${CARGO_HOME:-$HOME/.cargo}"
-
-if ! [[ ":$PATH:" == *":$CARGO_HOME/bin:"* ]]; then
-    if [ -n "${CI:-}" ] && [ -n "${GITHUB_PATH:-}" ]; then
-        echo "$CARGO_HOME/bin" >> "$GITHUB_PATH"
-    else
-        echo
-        printf "\033[0;31mYour path is missing %s, you might want to add it.\033[0m\n" "$CARGO_HOME/bin"
-        echo
-    fi
-fi
--- a/docker/dev-builder/centos/Dockerfile
+++ b/docker/dev-builder/centos/Dockerfile
@@ -2,42 +2,29 @@ FROM centos:7 as builder

 ENV LANG en_US.utf8

-# Note: CentOS 7 has reached EOL since 2024-07-01 thus `mirror.centos.org` is no longer available and we need to use `vault.centos.org` instead.
-RUN sed -i s/mirror.centos.org/vault.centos.org/g /etc/yum.repos.d/*.repo
-RUN sed -i s/^#.*baseurl=http/baseurl=http/g /etc/yum.repos.d/*.repo
-
 # Install dependencies
 RUN ulimit -n 1024000 && yum groupinstall -y 'Development Tools'
 RUN yum install -y epel-release  \
    openssl \
    openssl-devel  \
    centos-release-scl  \
+    rh-python38  \
+    rh-python38-python-devel \
    which

 # Install protoc
-ARG PROTOBUF_VERSION=29.3
-
-RUN curl -OL https://github.com/protocolbuffers/protobuf/releases/download/v${PROTOBUF_VERSION}/protoc-${PROTOBUF_VERSION}-linux-x86_64.zip && \
-    unzip protoc-${PROTOBUF_VERSION}-linux-x86_64.zip -d protoc3;
-    
-RUN mv protoc3/bin/* /usr/local/bin/
-RUN mv protoc3/include/* /usr/local/include/
+RUN curl -LO https://github.com/protocolbuffers/protobuf/releases/download/v3.15.8/protoc-3.15.8-linux-x86_64.zip
+RUN unzip protoc-3.15.8-linux-x86_64.zip -d /usr/local/

 # Install Rust
 SHELL ["/bin/bash", "-c"]
 RUN curl --proto '=https' --tlsv1.2 -sSf https://sh.rustup.rs | sh -s -- --no-modify-path --default-toolchain none -y
-ENV PATH /usr/local/bin:/root/.cargo/bin/:$PATH
+ENV PATH /opt/rh/rh-python38/root/usr/bin:/usr/local/bin:/root/.cargo/bin/:$PATH

 # Install Rust toolchains.
 ARG RUST_TOOLCHAIN
 RUN rustup toolchain install ${RUST_TOOLCHAIN}

-
-# Install cargo-binstall with a specific version to adapt the current rust toolchain.
-# Note: if we use the latest version, we may encounter the following `use of unstable library feature 'io_error_downcast'` error.
-# compile from source take too long, so we use the precompiled binary instead
-COPY $DOCKER_BUILD_ROOT/docker/dev-builder/binstall/pull_binstall.sh /usr/local/bin/pull_binstall.sh
-RUN chmod +x /usr/local/bin/pull_binstall.sh && /usr/local/bin/pull_binstall.sh
-
 # Install nextest.
+RUN cargo install cargo-binstall --locked
 RUN cargo binstall cargo-nextest --no-confirm
--- a/docker/dev-builder/ubuntu/Dockerfile
+++ b/docker/dev-builder/ubuntu/Dockerfile
@@ -1,4 +1,4 @@
-FROM ubuntu:22.04
+FROM ubuntu:20.04

 # The root path under which contains all the dependencies to build this Dockerfile.
 ARG DOCKER_BUILD_ROOT=.
@@ -6,34 +6,29 @@ ARG DOCKER_BUILD_ROOT=.
 ENV LANG en_US.utf8
 WORKDIR /greptimedb

+# Add PPA for Python 3.10.
 RUN apt-get update && \
-    DEBIAN_FRONTEND=noninteractive apt-get install -y software-properties-common
+    DEBIAN_FRONTEND=noninteractive apt-get install -y software-properties-common && \
+    add-apt-repository ppa:deadsnakes/ppa -y
+
 # Install dependencies.
 RUN apt-get update && DEBIAN_FRONTEND=noninteractive apt-get install -y \
    libssl-dev \
    tzdata \
+    protobuf-compiler \
    curl \
-    unzip \
    ca-certificates \
    git \
    build-essential \
-    pkg-config
+    pkg-config \
+    python3.10 \
+    python3.10-dev

-ARG TARGETPLATFORM
-RUN echo "target platform: $TARGETPLATFORM"
-
-ARG PROTOBUF_VERSION=29.3
-
-# Install protobuf, because the one in the apt is too old (v3.12).
-RUN if [ "$TARGETPLATFORM" = "linux/arm64" ]; then \
-    curl -OL https://github.com/protocolbuffers/protobuf/releases/download/v${PROTOBUF_VERSION}/protoc-${PROTOBUF_VERSION}-linux-aarch_64.zip && \
-    unzip protoc-${PROTOBUF_VERSION}-linux-aarch_64.zip -d protoc3; \
-elif [ "$TARGETPLATFORM" = "linux/amd64" ]; then \
-    curl -OL https://github.com/protocolbuffers/protobuf/releases/download/v${PROTOBUF_VERSION}/protoc-${PROTOBUF_VERSION}-linux-x86_64.zip && \
-    unzip protoc-${PROTOBUF_VERSION}-linux-x86_64.zip -d protoc3; \
-fi
-RUN mv protoc3/bin/* /usr/local/bin/
-RUN mv protoc3/include/* /usr/local/include/
+# Remove Python 3.8 and install pip.
+RUN apt-get -y purge python3.8 && \
+    apt-get -y autoremove && \
+    ln -s /usr/bin/python3.10 /usr/bin/python3 && \
+    curl -sS https://bootstrap.pypa.io/get-pip.py | python3.10

 # Silence all `safe.directory` warnings, to avoid the "detect dubious repository" error when building with submodules.
 # Disabling the safe directory check here won't pose extra security issues, because in our usage for this dev build
@@ -41,11 +36,15 @@ RUN mv protoc3/include/* /usr/local/include/
 # and the repositories are pulled from trusted sources (still us, of course). Doing so does not violate the intention
 # of the Git's addition to the "safe.directory" at the first place (see the commit message here:
 # https://github.com/git/git/commit/8959555cee7ec045958f9b6dd62e541affb7e7d9).
-# There's also another solution to this, that we add the desired submodules to the safe directory, instead of using
+# There's also another solution to this, that we add the desired submodules to the safe directory, instead of using 
 # wildcard here. However, that requires the git's config files and the submodules all owned by the very same user.
 # It's troublesome to do this since the dev build runs in Docker, which is under user "root"; while outside the Docker,
 # it can be a different user that have prepared the submodules.
-RUN git config --global --add safe.directory '*'
+RUN git config --global --add safe.directory *
+
+# Install Python dependencies.
+COPY $DOCKER_BUILD_ROOT/docker/python/requirements.txt /etc/greptime/requirements.txt
+RUN python3 -m pip install -r /etc/greptime/requirements.txt

 # Install Rust.
 SHELL ["/bin/bash", "-c"]
@@ -56,11 +55,6 @@ ENV PATH /root/.cargo/bin/:$PATH
 ARG RUST_TOOLCHAIN
 RUN rustup toolchain install ${RUST_TOOLCHAIN}

-# Install cargo-binstall with a specific version to adapt the current rust toolchain.
-# Note: if we use the latest version, we may encounter the following `use of unstable library feature 'io_error_downcast'` error.
-# compile from source take too long, so we use the precompiled binary instead
-COPY $DOCKER_BUILD_ROOT/docker/dev-builder/binstall/pull_binstall.sh /usr/local/bin/pull_binstall.sh
-RUN chmod +x /usr/local/bin/pull_binstall.sh && /usr/local/bin/pull_binstall.sh
-
 # Install nextest.
+RUN cargo install cargo-binstall --locked
 RUN cargo binstall cargo-nextest --no-confirm
--- a/docker/dev-builder/ubuntu/Dockerfile-18.10
+++ b/docker/dev-builder/ubuntu/Dockerfile-18.10
@@ -0,0 +1,48 @@
+# Use the legacy glibc 2.28.
+FROM ubuntu:18.10
+
+ENV LANG en_US.utf8
+WORKDIR /greptimedb
+
+# Use old-releases.ubuntu.com to avoid 404s: https://help.ubuntu.com/community/EOLUpgrades.
+RUN echo "deb http://old-releases.ubuntu.com/ubuntu/ cosmic main restricted universe multiverse\n\
+deb http://old-releases.ubuntu.com/ubuntu/ cosmic-updates main restricted universe multiverse\n\
+deb http://old-releases.ubuntu.com/ubuntu/ cosmic-security main restricted universe multiverse" > /etc/apt/sources.list
+
+# Install dependencies.
+RUN apt-get update && DEBIAN_FRONTEND=noninteractive apt-get install -y \
+    libssl-dev \
+    tzdata \
+    curl \
+    ca-certificates \
+    git \
+    build-essential \
+    unzip \
+    pkg-config
+
+# Install protoc.
+ENV PROTOC_VERSION=25.1
+RUN if [ "$(uname -m)" = "x86_64" ]; then \
+        PROTOC_ZIP=protoc-${PROTOC_VERSION}-linux-x86_64.zip; \
+    elif [ "$(uname -m)" = "aarch64" ]; then \
+        PROTOC_ZIP=protoc-${PROTOC_VERSION}-linux-aarch_64.zip; \
+    else \
+        echo "Unsupported architecture"; exit 1; \
+    fi && \
+    curl -OL https://github.com/protocolbuffers/protobuf/releases/download/v${PROTOC_VERSION}/${PROTOC_ZIP} && \
+    unzip -o ${PROTOC_ZIP} -d /usr/local bin/protoc && \
+    unzip -o ${PROTOC_ZIP} -d /usr/local 'include/*' && \
+    rm -f ${PROTOC_ZIP}
+
+# Install Rust.
+SHELL ["/bin/bash", "-c"]
+RUN curl --proto '=https' --tlsv1.2 -sSf https://sh.rustup.rs | sh -s -- --no-modify-path --default-toolchain none -y
+ENV PATH /root/.cargo/bin/:$PATH
+
+# Install Rust toolchains.
+ARG RUST_TOOLCHAIN
+RUN rustup toolchain install ${RUST_TOOLCHAIN}
+
+# Install nextest.
+RUN cargo install cargo-binstall --locked
+RUN cargo binstall cargo-nextest --no-confirm
--- a/docker/dev-builder/ubuntu/Dockerfile-20.04
+++ b/docker/dev-builder/ubuntu/Dockerfile-20.04
@@ -1,66 +0,0 @@
-FROM ubuntu:20.04
-
-# The root path under which contains all the dependencies to build this Dockerfile.
-ARG DOCKER_BUILD_ROOT=.
-
-ENV LANG en_US.utf8
-WORKDIR /greptimedb
-
-RUN apt-get update && \
-    DEBIAN_FRONTEND=noninteractive apt-get install -y software-properties-common
-# Install dependencies.
-RUN apt-get update && DEBIAN_FRONTEND=noninteractive apt-get install -y \
-    libssl-dev \
-    tzdata \
-    curl \
-    unzip \
-    ca-certificates \
-    git \
-    build-essential \
-    pkg-config
-
-ARG TARGETPLATFORM
-RUN echo "target platform: $TARGETPLATFORM"
-
-ARG PROTOBUF_VERSION=29.3
-
-# Install protobuf, because the one in the apt is too old (v3.12).
-RUN if [ "$TARGETPLATFORM" = "linux/arm64" ]; then \
-    curl -OL https://github.com/protocolbuffers/protobuf/releases/download/v${PROTOBUF_VERSION}/protoc-${PROTOBUF_VERSION}-linux-aarch_64.zip && \
-    unzip protoc-${PROTOBUF_VERSION}-linux-aarch_64.zip -d protoc3; \
-elif [ "$TARGETPLATFORM" = "linux/amd64" ]; then \
-    curl -OL https://github.com/protocolbuffers/protobuf/releases/download/v${PROTOBUF_VERSION}/protoc-${PROTOBUF_VERSION}-linux-x86_64.zip && \
-    unzip protoc-${PROTOBUF_VERSION}-linux-x86_64.zip -d protoc3; \
-fi
-RUN mv protoc3/bin/* /usr/local/bin/
-RUN mv protoc3/include/* /usr/local/include/
-
-# Silence all `safe.directory` warnings, to avoid the "detect dubious repository" error when building with submodules.
-# Disabling the safe directory check here won't pose extra security issues, because in our usage for this dev build
-# image, we use it solely on our own environment (that github action's VM, or ECS created dynamically by ourselves),
-# and the repositories are pulled from trusted sources (still us, of course). Doing so does not violate the intention
-# of the Git's addition to the "safe.directory" at the first place (see the commit message here:
-# https://github.com/git/git/commit/8959555cee7ec045958f9b6dd62e541affb7e7d9).
-# There's also another solution to this, that we add the desired submodules to the safe directory, instead of using 
-# wildcard here. However, that requires the git's config files and the submodules all owned by the very same user.
-# It's troublesome to do this since the dev build runs in Docker, which is under user "root"; while outside the Docker,
-# it can be a different user that have prepared the submodules.
-RUN git config --global --add safe.directory '*'
-
-# Install Rust.
-SHELL ["/bin/bash", "-c"]
-RUN curl --proto '=https' --tlsv1.2 -sSf https://sh.rustup.rs | sh -s -- --no-modify-path --default-toolchain none -y
-ENV PATH /root/.cargo/bin/:$PATH
-
-# Install Rust toolchains.
-ARG RUST_TOOLCHAIN
-RUN rustup toolchain install ${RUST_TOOLCHAIN}
-
-# Install cargo-binstall with a specific version to adapt the current rust toolchain.
-# Note: if we use the latest version, we may encounter the following `use of unstable library feature 'io_error_downcast'` error.
-# compile from source take too long, so we use the precompiled binary instead
-COPY $DOCKER_BUILD_ROOT/docker/dev-builder/binstall/pull_binstall.sh /usr/local/bin/pull_binstall.sh
-RUN chmod +x /usr/local/bin/pull_binstall.sh && /usr/local/bin/pull_binstall.sh
-
-# Install nextest.
-RUN cargo binstall cargo-nextest --no-confirm
--- a/docker/docker-compose/cluster-with-etcd.yaml
+++ b/docker/docker-compose/cluster-with-etcd.yaml
@@ -1,142 +0,0 @@
-x-custom:
-  etcd_initial_cluster_token: &etcd_initial_cluster_token "--initial-cluster-token=etcd-cluster"
-  etcd_common_settings: &etcd_common_settings
-    image: "${ETCD_REGISTRY:-quay.io}/${ETCD_NAMESPACE:-coreos}/etcd:${ETCD_VERSION:-v3.5.10}"
-    entrypoint: /usr/local/bin/etcd
-  greptimedb_image: &greptimedb_image "${GREPTIMEDB_REGISTRY:-docker.io}/${GREPTIMEDB_NAMESPACE:-greptime}/greptimedb:${GREPTIMEDB_VERSION:-latest}"
-
-services:
-  etcd0:
-    <<: *etcd_common_settings
-    container_name: etcd0
-    ports:
-      - 2379:2379
-      - 2380:2380
-    command:
-      - --name=etcd0
-      - --data-dir=/var/lib/etcd
-      - --initial-advertise-peer-urls=http://etcd0:2380
-      - --listen-peer-urls=http://0.0.0.0:2380
-      - --listen-client-urls=http://0.0.0.0:2379
-      - --advertise-client-urls=http://etcd0:2379
-      - --heartbeat-interval=250
-      - --election-timeout=1250
-      - --initial-cluster=etcd0=http://etcd0:2380
-      - --initial-cluster-state=new
-      - *etcd_initial_cluster_token
-    volumes:
-      - /tmp/greptimedb-cluster-docker-compose/etcd0:/var/lib/etcd
-    healthcheck:
-      test: [ "CMD", "etcdctl", "--endpoints=http://etcd0:2379", "endpoint", "health" ]
-      interval: 5s
-      timeout: 3s
-      retries: 5
-    networks:
-      - greptimedb
-
-  metasrv:
-    image: *greptimedb_image
-    container_name: metasrv
-    ports:
-      - 3002:3002
-      - 3000:3000
-    command:
-      - metasrv
-      - start
-      - --rpc-bind-addr=0.0.0.0:3002
-      - --rpc-server-addr=metasrv:3002
-      - --store-addrs=etcd0:2379
-      - --http-addr=0.0.0.0:3000
-    healthcheck:
-      test: [ "CMD", "curl", "-f", "http://metasrv:3000/health" ]
-      interval: 5s
-      timeout: 3s
-      retries: 5
-    depends_on:
-      etcd0:
-        condition: service_healthy
-    networks:
-      - greptimedb
-
-  datanode0:
-    image: *greptimedb_image
-    container_name: datanode0
-    ports:
-      - 3001:3001
-      - 5000:5000
-    command:
-      - datanode
-      - start
-      - --node-id=0
-      - --rpc-bind-addr=0.0.0.0:3001
-      - --rpc-server-addr=datanode0:3001
-      - --metasrv-addrs=metasrv:3002
-      - --http-addr=0.0.0.0:5000
-    volumes:
-      - /tmp/greptimedb-cluster-docker-compose/datanode0:/tmp/greptimedb
-    healthcheck:
-      test: [ "CMD", "curl", "-fv", "http://datanode0:5000/health" ]
-      interval: 5s
-      timeout: 3s
-      retries: 10
-    depends_on:
-      metasrv:
-        condition: service_healthy
-    networks:
-      - greptimedb
-
-  frontend0:
-    image: *greptimedb_image
-    container_name: frontend0
-    ports:
-      - 4000:4000
-      - 4001:4001
-      - 4002:4002
-      - 4003:4003
-    command:
-      - frontend
-      - start
-      - --metasrv-addrs=metasrv:3002
-      - --http-addr=0.0.0.0:4000
-      - --rpc-bind-addr=0.0.0.0:4001
-      - --mysql-addr=0.0.0.0:4002
-      - --postgres-addr=0.0.0.0:4003
-    healthcheck:
-      test: [ "CMD", "curl", "-f", "http://frontend0:4000/health" ]
-      interval: 5s
-      timeout: 3s
-      retries: 5
-    depends_on:
-      datanode0:
-        condition: service_healthy
-    networks:
-      - greptimedb
-
-  flownode0:
-    image: *greptimedb_image
-    container_name: flownode0
-    ports:
-      - 4004:4004
-      - 4005:4005
-    command:
-      - flownode
-      - start
-      - --node-id=0
-      - --metasrv-addrs=metasrv:3002
-      - --rpc-bind-addr=0.0.0.0:4004
-      - --rpc-server-addr=flownode0:4004
-      - --http-addr=0.0.0.0:4005
-    depends_on:
-      frontend0:
-        condition: service_healthy
-    healthcheck:
-      test: [ "CMD", "curl", "-f", "http://flownode0:4005/health" ]
-      interval: 5s
-      timeout: 3s
-      retries: 5
-    networks:
-      - greptimedb
-
-networks:
-  greptimedb:
-    name: greptimedb
--- a/docker/python/requirements.txt
+++ b/docker/python/requirements.txt
@@ -0,0 +1,5 @@
+numpy>=1.24.2
+pandas>=1.5.3
+pyarrow>=11.0.0
+requests>=2.28.2
+scipy>=1.10.1
--- a/docs/benchmarks/log/README.md
+++ b/docs/benchmarks/log/README.md
@@ -1,51 +0,0 @@
-# Log benchmark configuration
-This repo holds the configuration we used to benchmark GreptimeDB, Clickhouse and Elastic Search.
-
-Here are the versions of databases we used in the benchmark
-
-| name          | version    |
-| :------------ | :--------- |
-| GreptimeDB    | v0.9.2     |
-| Clickhouse    | 24.9.1.219 |
-| Elasticsearch | 8.15.0     |
-
-## Structured model vs Unstructured model
-We divide test into two parts, using structured model and unstructured model accordingly. You can also see the difference in create table clause.
-
-__Structured model__
-
-The log data is pre-processed into columns by vector. For example an insert request looks like following
-```SQL
-INSERT INTO test_table (bytes, http_version, ip, method, path, status, user, timestamp) VALUES ()
-```
-The goal is to test string/text support for each database. In real scenarios it means the datasource(or log data producers) have separate fields defined, or have already processed the raw input.
-
-__Unstructured model__
-
-The log data is inserted as a long string, and then we build fulltext index upon these strings. For example an insert request looks like following
-```SQL
-INSERT INTO test_table (message, timestamp) VALUES ()
-```
-The goal is to test fuzzy search performance for each database. In real scenarios it means the log is produced by some kind of middleware and inserted directly into the database.
-
-## Creating tables
-See [here](./create_table.sql) for GreptimeDB and Clickhouse's create table clause.
-The mapping of Elastic search is created automatically.
-
-## Vector Configuration
-We use vector to generate random log data and send inserts to databases.
-Please refer to [structured config](./structured_vector.toml) and [unstructured config](./unstructured_vector.toml) for detailed configuration.
-
-## SQLs and payloads
-Please refer to [SQL query](./query.sql) for GreptimeDB and Clickhouse, and [query payload](./query.md) for Elastic search.
-
-## Steps to reproduce
-0. Decide whether to run structured model test or unstructured mode test.
-1. Build vector binary(see vector's config file for specific branch) and databases binaries accordingly.
-2. Create table in GreptimeDB and Clickhouse in advance.
-3. Run vector to insert data.
-4. When data insertion is finished, run queries against each database. Note: you'll need to update timerange value after data insertion.
-
-## Addition
- You can tune GreptimeDB's configuration to get better performance.
- You can setup GreptimeDB to use S3 as storage, see [here](https://docs.greptime.com/user-guide/deployments/configuration#storage-options).
--- a/docs/benchmarks/log/create_table.sql
+++ b/docs/benchmarks/log/create_table.sql
@@ -1,56 +0,0 @@
-- GreptimeDB create table clause
-- structured test, use vector to pre-process log data into fields
-CREATE TABLE IF NOT EXISTS `test_table` (
-    `bytes` Int64 NULL,
-    `http_version` STRING NULL,
-    `ip` STRING NULL,
-    `method` STRING NULL,
-    `path` STRING NULL,
-    `status` SMALLINT UNSIGNED NULL,
-    `user` STRING NULL,
-    `timestamp` TIMESTAMP(3) NOT NULL,
-    PRIMARY KEY (`user`, `path`, `status`),
-    TIME INDEX (`timestamp`)
-)
-ENGINE=mito
-WITH(
-    append_mode = 'true'
-);
-
-- unstructured test, build fulltext index on message column
-CREATE TABLE IF NOT EXISTS `test_table` (
-    `message` STRING NULL FULLTEXT WITH(analyzer = 'English', case_sensitive = 'false'),
-    `timestamp` TIMESTAMP(3) NOT NULL,
-    TIME INDEX (`timestamp`)
-)
-ENGINE=mito
-WITH(
-    append_mode = 'true'
-);
-
-- Clickhouse create table clause
-- structured test
-CREATE TABLE IF NOT EXISTS test_table
-(
-    bytes UInt64 NOT NULL,
-    http_version String NOT NULL,
-    ip String NOT NULL,
-    method String NOT NULL,
-    path String NOT NULL,
-    status UInt8 NOT NULL,
-    user String NOT NULL,
-    timestamp String NOT NULL,
-)
-ENGINE = MergeTree()
-ORDER BY (user, path, status);
-
-- unstructured test
-SET allow_experimental_full_text_index = true;
-CREATE TABLE IF NOT EXISTS test_table
-(
-    message String,
-    timestamp String,
-    INDEX inv_idx(message) TYPE full_text(0) GRANULARITY 1
-)
-ENGINE = MergeTree()
-ORDER BY tuple();
--- a/docs/benchmarks/log/query.md
+++ b/docs/benchmarks/log/query.md
@@ -1,199 +0,0 @@
-# Query URL and payload for Elastic Search
-## Count
-URL: `http://127.0.0.1:9200/_count`
-
-## Query by timerange
-URL: `http://127.0.0.1:9200/_search`
-
-You can use the following payload to get the full timerange first.
-```JSON
-{"size":0,"aggs":{"max_timestamp":{"max":{"field":"timestamp"}},"min_timestamp":{"min":{"field":"timestamp"}}}}
-```
-
-And then use this payload to query by timerange.
-```JSON
-{
-  "from": 0,
-  "size": 1000,
-  "query": {
-    "range": {
-      "timestamp": {
-        "gte": "2024-08-16T04:30:44.000Z",
-        "lte": "2024-08-16T04:51:52.000Z"
-      }
-    }
-  }
-}
-```
-
-## Query by condition
-URL: `http://127.0.0.1:9200/_search`
-### Structured payload
-```JSON
-{
-  "from": 0,
-  "size": 10000,
-  "query": {
-    "bool": {
-      "must": [
-        {
-          "term": {
-            "user.keyword": "CrucifiX"
-          }
-        },
-        {
-          "term": {
-            "method.keyword": "OPTION"
-          }
-        },
-        {
-          "term": {
-            "path.keyword": "/user/booperbot124"
-          }
-        },
-        {
-          "term": {
-            "http_version.keyword": "HTTP/1.1"
-          }
-        },
-        {
-          "term": {
-            "status": "401"
-          }
-        }
-      ]
-    }
-  }
-}
-```
-### Unstructured payload
-```JSON
-{
-  "from": 0,
-  "size": 10000,
-  "query": {
-    "bool": {
-      "must": [
-        {
-          "match_phrase": {
-            "message": "CrucifiX"
-          }
-        },
-        {
-          "match_phrase": {
-            "message": "OPTION"
-          }
-        },
-        {
-          "match_phrase": {
-            "message": "/user/booperbot124"
-          }
-        },
-        {
-          "match_phrase": {
-            "message": "HTTP/1.1"
-          }
-        },
-        {
-          "match_phrase": {
-            "message": "401"
-          }
-        }
-      ]
-    }
-  }
-}
-```
-
-## Query by condition and timerange
-URL: `http://127.0.0.1:9200/_search`
-### Structured payload
-```JSON
-{
-  "size": 10000,
-  "query": {
-    "bool": {
-      "must": [
-        {
-          "term": {
-            "user.keyword": "CrucifiX"
-          }
-        },
-        {
-          "term": {
-            "method.keyword": "OPTION"
-          }
-        },
-        {
-          "term": {
-            "path.keyword": "/user/booperbot124"
-          }
-        },
-        {
-          "term": {
-            "http_version.keyword": "HTTP/1.1"
-          }
-        },
-        {
-          "term": {
-            "status": "401"
-          }
-        },
-        {
-          "range": {
-            "timestamp": {
-              "gte": "2024-08-19T07:03:37.383Z",
-              "lte": "2024-08-19T07:24:58.883Z"
-            }
-          }
-        }
-      ]
-    }
-  }
-}
-```
-### Unstructured payload
-```JSON
-{
-  "size": 10000,
-  "query": {
-    "bool": {
-      "must": [
-        {
-          "match_phrase": {
-            "message": "CrucifiX"
-          }
-        },
-        {
-          "match_phrase": {
-            "message": "OPTION"
-          }
-        },
-        {
-          "match_phrase": {
-            "message": "/user/booperbot124"
-          }
-        },
-        {
-          "match_phrase": {
-            "message": "HTTP/1.1"
-          }
-        },
-        {
-          "match_phrase": {
-            "message": "401"
-          }
-        },
-        {
-          "range": {
-            "timestamp": {
-              "gte": "2024-08-19T05:16:17.099Z",
-              "lte": "2024-08-19T05:46:02.722Z"
-            }
-          }
-        }
-      ]
-    }
-  }
-}
-```
--- a/docs/benchmarks/log/query.sql
+++ b/docs/benchmarks/log/query.sql
@@ -1,50 +0,0 @@
-- Structured query for GreptimeDB and Clickhouse
-
-- query count
-select count(*) from test_table;
-
-- query by timerange. Note: place the timestamp range in the where clause
-- GreptimeDB
-- you can use `select max(timestamp)::bigint from test_table;` and `select min(timestamp)::bigint from test_table;`
-- to get the full timestamp range
-select * from test_table where timestamp between 1723710843619 and 1723711367588;
-- Clickhouse
-- you can use `select max(timestamp) from test_table;` and `select min(timestamp) from test_table;`
-- to get the full timestamp range
-select * from test_table where timestamp between '2024-08-16T03:58:46Z' and '2024-08-16T04:03:50Z';
-
-- query by condition
-SELECT * FROM test_table WHERE user = 'CrucifiX' and method = 'OPTION' and path = '/user/booperbot124' and http_version = 'HTTP/1.1' and status = 401;
-
-- query by condition and timerange
-- GreptimeDB
-SELECT * FROM test_table WHERE user = "CrucifiX" and method = "OPTION" and path = "/user/booperbot124" and http_version = "HTTP/1.1" and status = 401 
-and timestamp between 1723774396760 and 1723774788760;
-- Clickhouse
-SELECT * FROM test_table WHERE user = 'CrucifiX' and method = 'OPTION' and path = '/user/booperbot124' and http_version = 'HTTP/1.1' and status = 401 
-and timestamp between '2024-08-16T03:58:46Z' and '2024-08-16T04:03:50Z';
-
-- Unstructured query for GreptimeDB and Clickhouse
-
-
-- query by condition
-- GreptimeDB
-SELECT * FROM test_table WHERE MATCHES(message, "+CrucifiX +OPTION +/user/booperbot124 +HTTP/1.1 +401");
-- Clickhouse
-SELECT * FROM test_table WHERE (message LIKE '%CrucifiX%') 
-AND (message LIKE '%OPTION%') 
-AND (message LIKE '%/user/booperbot124%') 
-AND (message LIKE '%HTTP/1.1%') 
-AND (message LIKE '%401%');
-
-- query by condition and timerange
-- GreptimeDB
-SELECT * FROM test_table WHERE MATCHES(message, "+CrucifiX +OPTION +/user/booperbot124 +HTTP/1.1 +401") 
-and timestamp between 1723710843619 and 1723711367588;
-- Clickhouse
-SELECT * FROM test_table WHERE (message LIKE '%CrucifiX%') 
-AND (message LIKE '%OPTION%') 
-AND (message LIKE '%/user/booperbot124%') 
-AND (message LIKE '%HTTP/1.1%') 
-AND (message LIKE '%401%') 
-AND timestamp between '2024-08-15T10:25:26.524000000Z' AND '2024-08-15T10:31:31.746000000Z';
--- a/docs/benchmarks/log/structured_vector.toml
+++ b/docs/benchmarks/log/structured_vector.toml
@@ -1,57 +0,0 @@
-# Please note we use patched branch to build vector
-# https://github.com/shuiyisong/vector/tree/chore/greptime_log_ingester_logitem
-
-[sources.demo_logs]
-type = "demo_logs"
-format = "apache_common"
-# interval value = 1 / rps
-# say you want to insert at 20k/s, that is 1 / 20000 = 0.00005
-# set to 0 to run as fast as possible
-interval = 0
-# total rows to insert
-count = 100000000
-lines = [ "line1" ]
-
-[transforms.parse_logs]
-type = "remap"
-inputs = ["demo_logs"]
-source = '''
-. = parse_regex!(.message, r'^(?P<ip>\S+) - (?P<user>\S+) \[(?P<timestamp>[^\]]+)\] "(?P<method>\S+) (?P<path>\S+) (?P<http_version>\S+)" (?P<status>\d+) (?P<bytes>\d+)$')
-
-# Convert timestamp to a standard format
-.timestamp = parse_timestamp!(.timestamp, format: "%d/%b/%Y:%H:%M:%S %z")
-
-# Convert status and bytes to integers
-.status = to_int!(.status)
-.bytes = to_int!(.bytes)
-'''
-
-[sinks.sink_greptime_logs]
-type = "greptimedb_logs"
-# The table to insert into
-table = "test_table"
-pipeline_name = "demo_pipeline"
-compression = "none"
-inputs = [ "parse_logs" ]
-endpoint = "http://127.0.0.1:4000"
-# Batch size for each insertion
-batch.max_events = 4000
-
-[sinks.clickhouse]
-type = "clickhouse"
-inputs = [ "parse_logs" ]
-database = "default"
-endpoint = "http://127.0.0.1:8123"
-format = "json_each_row"
-# The table to insert into
-table = "test_table"
-
-[sinks.sink_elasticsearch]
-type = "elasticsearch"
-inputs = [ "parse_logs" ]
-api_version = "auto"
-compression = "none"
-doc_type = "_doc"
-endpoints = [ "http://127.0.0.1:9200" ]
-id_key = "id"
-mode = "bulk"
--- a/docs/benchmarks/log/unstructured_vector.toml
+++ b/docs/benchmarks/log/unstructured_vector.toml
@@ -1,43 +0,0 @@
-# Please note we use patched branch to build vector
-# https://github.com/shuiyisong/vector/tree/chore/greptime_log_ingester_ft
-
-[sources.demo_logs]
-type = "demo_logs"
-format = "apache_common"
-# interval value = 1 / rps
-# say you want to insert at 20k/s, that is 1 / 20000 = 0.00005
-# set to 0 to run as fast as possible
-interval = 0
-# total rows to insert
-count = 100000000
-lines = [ "line1" ]
-
-[sinks.sink_greptime_logs]
-type = "greptimedb_logs"
-# The table to insert into
-table = "test_table"
-pipeline_name = "demo_pipeline"
-compression = "none"
-inputs = [ "demo_logs" ]
-endpoint = "http://127.0.0.1:4000"
-# Batch size for each insertion
-batch.max_events = 500
-
-[sinks.clickhouse]
-type = "clickhouse"
-inputs = [ "demo_logs" ]
-database = "default"
-endpoint = "http://127.0.0.1:8123"
-format = "json_each_row"
-# The table to insert into
-table = "test_table"
-
-[sinks.sink_elasticsearch]
-type = "elasticsearch"
-inputs = [ "demo_logs" ]
-api_version = "auto"
-compression = "none"
-doc_type = "_doc"
-endpoints = [ "http://127.0.0.1:9200" ]
-id_key = "id"
-mode = "bulk"
--- a/docs/benchmarks/tsbs/README.md
+++ b/docs/benchmarks/tsbs/README.md
@@ -1,253 +0,0 @@
-# How to run TSBS Benchmark
-
-This document contains the steps to run TSBS Benchmark. Our results are listed in other files in the same directory.
-
-## Prerequires
-
-You need the following tools to run TSBS Benchmark:
- Go
- git
- make
- rust (optional, if you want to build the DB from source)
-
-## Build TSBS suite
-
-Clone our fork of TSBS:
-
-```shell
-git clone https://github.com/GreptimeTeam/tsbs.git
-```
-
-Then build it:
-
-```shell
-cd tsbs
-make
-```
-
-You can check the `bin/` directory for compiled binaries. We will only use some of them.
-
-```shell
-ls ./bin/
-```
-
-Binaries we will use later:
- `tsbs_generate_data`
- `tsbs_generate_queries`
- `tsbs_load_greptime`
- `tsbs_run_queries_influx`
-
-## Generate test data and queries
-
-The data is generated by `tsbs_generate_data`
-
-```shell
-mkdir bench-data
-./bin/tsbs_generate_data --use-case="cpu-only" --seed=123 --scale=4000 \
-    --timestamp-start="2023-06-11T00:00:00Z" \
-    --timestamp-end="2023-06-14T00:00:00Z" \
-    --log-interval="10s" --format="influx" \
-    > ./bench-data/influx-data.lp
-```
-
-Here we generates 4000 time-series in 3 days with 10s interval. We'll use influx line protocol to write so the target format is `influx`.
-
-Queries are generated by `tsbs_generate_queries`. You can change the parameters but need to make sure it matches with `tsbs_generate_data`.
-
-```shell
-./bin/tsbs_generate_queries \
-    --use-case="devops" --seed=123 --scale=4000 \
-    --timestamp-start="2023-06-11T00:00:00Z" \
-    --timestamp-end="2023-06-14T00:00:01Z" \
-    --queries=100 \
-    --query-type cpu-max-all-1 \
-    --format="greptime" \
-    > ./bench-data/greptime-queries-cpu-max-all-1.dat
-./bin/tsbs_generate_queries \
-    --use-case="devops" --seed=123 --scale=4000 \
-    --timestamp-start="2023-06-11T00:00:00Z" \
-    --timestamp-end="2023-06-14T00:00:01Z" \
-    --queries=100 \
-    --query-type cpu-max-all-8 \
-    --format="greptime" \
-    > ./bench-data/greptime-queries-cpu-max-all-8.dat
-./bin/tsbs_generate_queries \
-    --use-case="devops" --seed=123 --scale=4000 \
-    --timestamp-start="2023-06-11T00:00:00Z" \
-    --timestamp-end="2023-06-14T00:00:01Z" \
-    --queries=50 \
-    --query-type double-groupby-1 \
-    --format="greptime" \
-    > ./bench-data/greptime-queries-double-groupby-1.dat
-./bin/tsbs_generate_queries \
-    --use-case="devops" --seed=123 --scale=4000 \
-    --timestamp-start="2023-06-11T00:00:00Z" \
-    --timestamp-end="2023-06-14T00:00:01Z" \
-    --queries=50 \
-    --query-type double-groupby-5 \
-    --format="greptime" \
-    > ./bench-data/greptime-queries-double-groupby-5.dat
-./bin/tsbs_generate_queries \
-    --use-case="devops" --seed=123 --scale=4000 \
-    --timestamp-start="2023-06-11T00:00:00Z" \
-    --timestamp-end="2023-06-14T00:00:01Z" \
-    --queries=50 \
-    --query-type double-groupby-all \
-    --format="greptime" \
-    > ./bench-data/greptime-queries-double-groupby-all.dat
-./bin/tsbs_generate_queries \
-    --use-case="devops" --seed=123 --scale=4000 \
-    --timestamp-start="2023-06-11T00:00:00Z" \
-    --timestamp-end="2023-06-14T00:00:01Z" \
-    --queries=50 \
-    --query-type groupby-orderby-limit \
-    --format="greptime" \
-    > ./bench-data/greptime-queries-groupby-orderby-limit.dat
-./bin/tsbs_generate_queries \
-    --use-case="devops" --seed=123 --scale=4000 \
-    --timestamp-start="2023-06-11T00:00:00Z" \
-    --timestamp-end="2023-06-14T00:00:01Z" \
-    --queries=100 \
-    --query-type high-cpu-1 \
-    --format="greptime" \
-    > ./bench-data/greptime-queries-high-cpu-1.dat
-./bin/tsbs_generate_queries \
-    --use-case="devops" --seed=123 --scale=4000 \
-    --timestamp-start="2023-06-11T00:00:00Z" \
-    --timestamp-end="2023-06-14T00:00:01Z" \
-    --queries=50 \
-    --query-type high-cpu-all \
-    --format="greptime" \
-    > ./bench-data/greptime-queries-high-cpu-all.dat
-./bin/tsbs_generate_queries \
-    --use-case="devops" --seed=123 --scale=4000 \
-    --timestamp-start="2023-06-11T00:00:00Z" \
-    --timestamp-end="2023-06-14T00:00:01Z" \
-    --queries=10 \
-    --query-type lastpoint \
-    --format="greptime" \
-    > ./bench-data/greptime-queries-lastpoint.dat
-./bin/tsbs_generate_queries \
-    --use-case="devops" --seed=123 --scale=4000 \
-    --timestamp-start="2023-06-11T00:00:00Z" \
-    --timestamp-end="2023-06-14T00:00:01Z" \
-    --queries=100 \
-    --query-type single-groupby-1-1-1 \
-    --format="greptime" \
-    > ./bench-data/greptime-queries-single-groupby-1-1-1.dat
-./bin/tsbs_generate_queries \
-    --use-case="devops" --seed=123 --scale=4000 \
-    --timestamp-start="2023-06-11T00:00:00Z" \
-    --timestamp-end="2023-06-14T00:00:01Z" \
-    --queries=100 \
-    --query-type single-groupby-1-1-12 \
-    --format="greptime" \
-    > ./bench-data/greptime-queries-single-groupby-1-1-12.dat
-./bin/tsbs_generate_queries \
-    --use-case="devops" --seed=123 --scale=4000 \
-    --timestamp-start="2023-06-11T00:00:00Z" \
-    --timestamp-end="2023-06-14T00:00:01Z" \
-    --queries=100 \
-    --query-type single-groupby-1-8-1 \
-    --format="greptime" \
-    > ./bench-data/greptime-queries-single-groupby-1-8-1.dat
-./bin/tsbs_generate_queries \
-    --use-case="devops" --seed=123 --scale=4000 \
-    --timestamp-start="2023-06-11T00:00:00Z" \
-    --timestamp-end="2023-06-14T00:00:01Z" \
-    --queries=100 \
-    --query-type single-groupby-5-1-1 \
-    --format="greptime" \
-    > ./bench-data/greptime-queries-single-groupby-5-1-1.dat
-./bin/tsbs_generate_queries \
-    --use-case="devops" --seed=123 --scale=4000 \
-    --timestamp-start="2023-06-11T00:00:00Z" \
-    --timestamp-end="2023-06-14T00:00:01Z" \
-    --queries=100 \
-    --query-type single-groupby-5-1-12 \
-    --format="greptime" \
-    > ./bench-data/greptime-queries-single-groupby-5-1-12.dat
-./bin/tsbs_generate_queries \
-    --use-case="devops" --seed=123 --scale=4000 \
-    --timestamp-start="2023-06-11T00:00:00Z" \
-    --timestamp-end="2023-06-14T00:00:01Z" \
-    --queries=100 \
-    --query-type single-groupby-5-8-1 \
-    --format="greptime" \
-    > ./bench-data/greptime-queries-single-groupby-5-8-1.dat
-```
-
-## Start GreptimeDB
-
-Reference to our [document](https://docs.greptime.com/getting-started/installation/overview) for how to install and start a GreptimeDB. Or you can also check this [document](https://docs.greptime.com/contributor-guide/getting-started#compile-and-run) for how to build a GreptimeDB from source.
-
-## Write Data
-
-After the DB is started, we can use `tsbs_load_greptime` to test the write performance.
-
-```shell
-./bin/tsbs_load_greptime \
-    --urls=http://localhost:4000 \
-    --file=./bench-data/influx-data.lp \
-    --batch-size=3000 \
-    --gzip=false \
-    --workers=6
-```
-
-Parameters here are only provided as an example. You can choose whatever you like or adjust them to match your target scenario.
-
-Notice that if you want to rerun `tsbs_load_greptime`, please destroy and restart the DB and clear its previous data first. Existing duplicated data will impact the write and query performance.
-
-## Query Data
-
-After the data is imported, you can then run queries. The following script runs all queries. You can also choose a subset of queries to run.
-
-```shell
-./bin/tsbs_run_queries_influx --file=./bench-data/greptime-queries-cpu-max-all-1.dat \
-    --db-name=benchmark \
-    --urls="http://localhost:4000"
-./bin/tsbs_run_queries_influx --file=./bench-data/greptime-queries-cpu-max-all-8.dat \
-    --db-name=benchmark \
-    --urls="http://localhost:4000"
-./bin/tsbs_run_queries_influx --file=./bench-data/greptime-queries-double-groupby-1.dat \
-    --db-name=benchmark \
-    --urls="http://localhost:4000"
-./bin/tsbs_run_queries_influx --file=./bench-data/greptime-queries-double-groupby-5.dat \
-    --db-name=benchmark \
-    --urls="http://localhost:4000"
-./bin/tsbs_run_queries_influx --file=./bench-data/greptime-queries-double-groupby-all.dat \
-    --db-name=benchmark \
-    --urls="http://localhost:4000"
-./bin/tsbs_run_queries_influx --file=./bench-data/greptime-queries-groupby-orderby-limit.dat \
-    --db-name=benchmark \
-    --urls="http://localhost:4000"
-./bin/tsbs_run_queries_influx --file=./bench-data/greptime-queries-high-cpu-1.dat \
-    --db-name=benchmark \
-    --urls="http://localhost:4000"
-./bin/tsbs_run_queries_influx --file=./bench-data/greptime-queries-high-cpu-all.dat \
-    --db-name=benchmark \
-    --urls="http://localhost:4000"
-./bin/tsbs_run_queries_influx --file=./bench-data/greptime-queries-lastpoint.dat \
-    --db-name=benchmark \
-    --urls="http://localhost:4000"
-./bin/tsbs_run_queries_influx --file=./bench-data/greptime-queries-single-groupby-1-1-1.dat \
-    --db-name=benchmark \
-    --urls="http://localhost:4000"
-./bin/tsbs_run_queries_influx --file=./bench-data/greptime-queries-single-groupby-1-1-12.dat \
-    --db-name=benchmark \
-    --urls="http://localhost:4000"
-./bin/tsbs_run_queries_influx --file=./bench-data/greptime-queries-single-groupby-1-8-1.dat \
-    --db-name=benchmark \
-    --urls="http://localhost:4000"
-./bin/tsbs_run_queries_influx --file=./bench-data/greptime-queries-single-groupby-5-1-1.dat \
-    --db-name=benchmark \
-    --urls="http://localhost:4000"
-./bin/tsbs_run_queries_influx --file=./bench-data/greptime-queries-single-groupby-5-1-12.dat \
-    --db-name=benchmark \
-    --urls="http://localhost:4000"
-./bin/tsbs_run_queries_influx --file=./bench-data/greptime-queries-single-groupby-5-8-1.dat \
-    --db-name=benchmark \
-    --urls="http://localhost:4000"
-```
-
-Rerun queries need not to re-import data. Just execute the corresponding command again is fine.
--- a/docs/benchmarks/tsbs/v0.12.0.md
+++ b/docs/benchmarks/tsbs/v0.12.0.md
@@ -1,40 +0,0 @@
-# TSBS benchmark - v0.12.0
-
-## Environment
-
-### Amazon EC2
-
-|         |                         |
-|---------|-------------------------|
-| Machine | c5d.2xlarge             |
-| CPU     | 8 core                  |
-| Memory  | 16GB                    |
-| Disk    | 100GB (GP3)             |
-| OS      | Ubuntu Server 24.04 LTS |
-
-## Write performance
-
-| Environment     | Ingest rate (rows/s) |
-|-----------------|----------------------|
-| EC2 c5d.2xlarge | 326839.28            |
-
-## Query performance
-
-| Query type            | EC2 c5d.2xlarge (ms) |
-|-----------------------|----------------------|
-| cpu-max-all-1         | 12.46                |
-| cpu-max-all-8         | 24.20                |
-| double-groupby-1      | 673.08               |
-| double-groupby-5      | 963.99               |
-| double-groupby-all    | 1330.05              |
-| groupby-orderby-limit | 952.46               |
-| high-cpu-1            | 5.08                 |
-| high-cpu-all          | 4638.57              |
-| lastpoint             | 591.02               |
-| single-groupby-1-1-1  | 4.06                 |
-| single-groupby-1-1-12 | 4.73                 |
-| single-groupby-1-8-1  | 8.23                 |
-| single-groupby-5-1-1  | 4.61                 |
-| single-groupby-5-1-12 | 5.61                 |
-| single-groupby-5-8-1  | 9.74                 |
-
--- a/docs/benchmarks/tsbs/v0.9.1.md
+++ b/docs/benchmarks/tsbs/v0.9.1.md
@@ -1,58 +0,0 @@
-# TSBS benchmark - v0.9.1
-
-## Environment
-
-### Local
-
-|        |                                    |
-| ------ | ---------------------------------- |
-| CPU    | AMD Ryzen 7 7735HS (8 core 3.2GHz) |
-| Memory | 32GB                               |
-| Disk   | SOLIDIGM SSDPFKNU010TZ             |
-| OS     | Ubuntu 22.04.2 LTS                 |
-
-### Amazon EC2
-
-|         |                         |
-| ------- | ----------------------- |
-| Machine | c5d.2xlarge             |
-| CPU     | 8 core                  |
-| Memory  | 16GB                    |
-| Disk    | 100GB (GP3)             |
-| OS      | Ubuntu Server 24.04 LTS |
-
-## Write performance
-
-| Environment     | Ingest rate (rows/s) |
-| --------------- | -------------------- |
-| Local           | 387697.68            |
-| EC2 c5d.2xlarge | 234620.19            |
-
-## Query performance
-
-| Query type            | Local (ms) | EC2 c5d.2xlarge (ms) |
-| --------------------- | ---------- | -------------------- |
-| cpu-max-all-1         | 21.14      | 14.75                |
-| cpu-max-all-8         | 36.79      | 30.69                |
-| double-groupby-1      | 529.02     | 987.85               |
-| double-groupby-5      | 1064.53    | 1455.95              |
-| double-groupby-all    | 1625.33    | 2143.96              |
-| groupby-orderby-limit | 529.19     | 1353.49              |
-| high-cpu-1            | 12.09      | 8.24                 |
-| high-cpu-all          | 3619.47    | 5312.82              |
-| lastpoint             | 224.91     | 576.06               |
-| single-groupby-1-1-1  | 10.82      | 6.01                 |
-| single-groupby-1-1-12 | 11.16      | 7.42                 |
-| single-groupby-1-8-1  | 13.50      | 10.20                |
-| single-groupby-5-1-1  | 11.99      | 6.70                 |
-| single-groupby-5-1-12 | 13.17      | 8.72                 |
-| single-groupby-5-8-1  | 16.01      | 12.07                |
-
-`single-groupby-1-1-1` query throughput
-
-| Environment     | Client concurrency | mean time (ms) | qps (queries/sec) |
-| --------------- | ------------------ | -------------- | ----------------- |
-| Local           | 50                 | 33.04          | 1511.74           |
-| Local           | 100                | 67.70          | 1476.14           |
-| EC2 c5d.2xlarge | 50                 | 61.93          | 806.97            |
-| EC2 c5d.2xlarge | 100                | 126.31         | 791.40            |
--- a/docs/how-to/how-to-change-log-level-on-the-fly.md
+++ b/docs/how-to/how-to-change-log-level-on-the-fly.md
@@ -1,16 +0,0 @@
-# Change Log Level on the Fly
-
-## HTTP API
-
-example:
-```bash
-curl --data "trace,flow=debug" 127.0.0.1:4000/debug/log_level
-```
-And database will reply with something like:
-```bash
-Log Level changed from Some("info") to "trace,flow=debug"%
-```
-
-The data is a string in the format of `global_level,module1=level1,module2=level2,...` that follow the same rule of `RUST_LOG`. 
-
-The module is the module name of the log, and the level is the log level. The log level can be one of the following: `trace`, `debug`, `info`, `warn`, `error`, `off`(case insensitive).
--- a/docs/how-to/how-to-profile-memory.md
+++ b/docs/how-to/how-to-profile-memory.md
@@ -1,61 +0,0 @@
-# Profile memory usage of GreptimeDB
-
-This crate provides an easy approach to dump memory profiling info.
-
-## Prerequisites
-### jemalloc
-jeprof is already compiled in the target directory of GreptimeDB. You can find the binary and use it.
-```
-# find jeprof binary
-find . -name 'jeprof'
-# add executable permission
-chmod +x <path_to_jeprof>
-```
-The path is usually under `./target/${PROFILE}/build/tikv-jemalloc-sys-${HASH}/out/build/bin/jeprof`.
-The default version of jemalloc installed from the package manager may not have the `--collapsed` option.
-You may need to check the whether the `jeprof` version is >= `5.3.0` if you want to install it from the package manager.
-```bash
-# for macOS
-brew install jemalloc
-
-# for Ubuntu
-sudo apt install libjemalloc-dev
-```
-
-### [flamegraph](https://github.com/brendangregg/FlameGraph)
-
-```bash
-curl https://raw.githubusercontent.com/brendangregg/FlameGraph/master/flamegraph.pl > ./flamegraph.pl
-```
-
-## Profiling
-
-Start GreptimeDB instance with environment variables:
-
-```bash
-# for Linux
-MALLOC_CONF=prof:true ./target/debug/greptime standalone start
-
-# for macOS
-_RJEM_MALLOC_CONF=prof:true ./target/debug/greptime standalone start
-```
-
-Dump memory profiling data through HTTP API:
-
-```bash
-curl -X POST localhost:4000/debug/prof/mem > greptime.hprof
-```
-
-You can periodically dump profiling data and compare them to find the delta memory usage.
-
-## Analyze profiling data with flamegraph
-
-To create flamegraph according to dumped profiling data:
-
-```bash
-sudo apt install -y libjemalloc-dev
-
-jeprof <path_to_greptime_binary> <profile_data> --collapse | ./flamegraph.pl > mem-prof.svg
-
-jeprof <path_to_greptime_binary> --base <baseline_prof> <profile_data> --collapse | ./flamegraph.pl > output.svg
-```
--- a/docs/how-to/how-to-write-fuzz-tests.md
+++ b/docs/how-to/how-to-write-fuzz-tests.md
@@ -105,7 +105,7 @@ use tests_fuzz::utils::{init_greptime_connections, Connections};

 fuzz_target!(|input: FuzzInput| {
    common_telemetry::init_default_ut_logging();
-    common_runtime::block_on_global(async {
+    common_runtime::block_on_write(async {
        let Connections { mysql } = init_greptime_connections().await;
            let mut rng = ChaChaRng::seed_from_u64(input.seed);
            let columns = rng.gen_range(2..30);
--- a/docs/logo-text-padding.png
+++ b/docs/logo-text-padding.png
--- a/docs/rfcs/2024-08-06-json-datatype.md
+++ b/docs/rfcs/2024-08-06-json-datatype.md
@@ -1,197 +0,0 @@
---
-Feature Name: Json Datatype
-Tracking Issue: https://github.com/GreptimeTeam/greptimedb/issues/4230
-Date: 2024-8-6
-Author: "Yuhan Wang <profsyb@gmail.com>"
---
-
-# Summary
-This RFC proposes a method for storing and querying JSON data in the database.
-
-# Motivation
-JSON is widely used across various scenarios. Direct support for writing and querying JSON can significantly enhance the database's flexibility.
-
-# Details
-
-## Storage and Query
-
-GreptimeDB's type system is built on Arrow/DataFusion, where each data type in GreptimeDB corresponds to a data type in Arrow/DataFusion. The proposed JSON type will be implemented on top of the existing `Binary` type, leveraging the current `datatype::value::Value` and `datatype::vectors::BinaryVector` implementations, utilizing the JSONB format as the encoding of JSON data. JSON data is stored and processed similarly to binary data within the storage layer and query engine.
-
-This approach brings problems when dealing with insertions and queries of JSON columns.
-
-## Insertion
-
-Users commonly write JSON data as strings. Thus we need to make conversions between string and JSONB. There are 2 ways to do this:
-
-1. MySQL and PostgreSQL servers provide auto-conversions between strings and JSONB. When a string is inserted into a JSON column, the server will try to parse the string as JSON and convert it to JSONB. The non-JSON strings will be rejected.
-
-2. A function `parse_json` is provided to convert string to JSONB. If the string is not a valid JSON string, the function will return an error.
-
-For example, in MySQL client:
-```SQL
-CREATE TABLE IF NOT EXISTS test (
-    ts TIMESTAMP TIME INDEX,
-    a INT,
-    b JSON
-);
-
-INSERT INTO test VALUES(
-    0,
-    0,
-    '{
-        "name": "jHl2oDDnPc1i2OzlP5Y",
-        "timestamp": "2024-07-25T04:33:11.369386Z",
-        "attributes": { "event_attributes": 48.28667 }
-    }'
-);
-
-INSERT INTO test VALUES(
-    0,
-    0,
-    parse_json('{
-        "name": "jHl2oDDnPc1i2OzlP5Y",
-        "timestamp": "2024-07-25T04:33:11.369386Z",
-        "attributes": { "event_attributes": 48.28667 }
-    }')
-);
-```
-Are both valid.
-
-The dataflow of the insertion process is as follows:
-```
-Insert JSON strings directly through client:
-                                   Parse                       Insert
-        String(Serialized JSON)┌──────────┐Arrow Binary(JSONB)┌──────┐Arrow Binary(JSONB)
- Client ---------------------->│  Server  │------------------>│ Mito │------------------> Storage
-                               └──────────┘                   └──────┘
-        (Server identifies JSON type and performs auto-conversion)
-
-Insert JSON strings through parse_json function:
-                                                                   Parse                     Insert
-        String(Serialized JSON)┌──────────┐String(Serialized JSON)┌─────┐Arrow Binary(JSONB)┌──────┐Arrow Binary(JSONB)
- Client ---------------------->│  Server  │---------------------->│ UDF │------------------>│ Mito │------------------> Storage
-                               └──────────┘                       └─────┘                   └──────┘
-                                            (Conversion is performed by UDF inside Query Engine)
-```
-
-Servers identify JSON column through column schema and perform auto-conversions. But when using prepared statements and binding parameters, the corresponding cached plans in datafusion generated by prepared statements cannot identify JSON columns. Under this circumstance, the servers identify JSON columns through the given parameters and perform auto-conversions.
-
-The following is an example of inserting JSON data through prepared statements:
-```Rust
-sqlx::query(
-    "create table test(ts timestamp time index, j json)",
-)
-.execute(&pool)
-.await
-.unwrap();
-
-let json = serde_json::json!({
-    "code": 200,
-    "success": true,
-    "payload": {
-        "features": [
-            "serde",
-            "json"
-        ],
-        "homepage": null
-    }
-});
-
-// Valid, can identify serde_json::Value as JSON type
-sqlx::query("insert into test values($1, $2)")
-    .bind(i)
-    .bind(json)
-    .execute(&pool)
-    .await
-    .unwrap();
-
-// Invalid, cannot identify String as JSON type
-sqlx::query("insert into test values($1, $2)")
-    .bind(i)
-    .bind(json.to_string())
-    .execute(&pool)
-    .await
-    .unwrap();
-```
-
-## Query
-
-Correspondingly, users prefer to display JSON data as strings. Thus we need to make conversions between JSON data and strings before presenting JSON data. There are also 2 ways to do this: auto-conversions on MySQL and PostgreSQL servers, and function `json_to_string`.
-
-For example, in MySQL client:
-```SQL
-SELECT b FROM test;
-
-SELECT json_to_string(b) FROM test;
-```
-Will both return the JSON as human-readable strings.
-
-Specifically, to perform auto-conversions, we attach a message to JSON data in the `metadata` of `Field` in Arrow/Datafusion schema when scanning a JSON column. Frontend servers could identify JSON data and convert it to strings.
-
-The dataflow of the query process is as follows:
-```
-Query directly through client:
-                                  Decode                            Scan
-        String(Serialized JSON)┌──────────┐Arrow Binary(JSONB)┌──────────────┐Arrow Binary(JSONB)
- Client <----------------------│  Server  │<------------------│ Query Engine │<----------------- Storage
-                               └──────────┘                   └──────────────┘
-(Server identifies JSON type and performs auto-conversion based on column metadata)
-
-Query through json_to_string function:
-                                                                   Scan & Decode
-        String(Serialized JSON)┌──────────┐String(Serialized JSON)┌──────────────┐Arrow Binary(JSONB)
- Client <----------------------│  Server  │<----------------------│ Query Engine │<----------------- Storage
-                               └──────────┘                       └──────────────┘
-                                                 (Conversion is performed by UDF inside Query Engine)
-
-```
-
-However, if a function uses JSON type as its return type, the metadata method mentioned above is not applicable. Thus the functions of JSON type should specify the return type explicitly instead of returning a JSON type, such as `json_get_int` and `json_get_float` which return corresponding data of `INT` and `FLOAT` type respectively.
-
-## Functions
-Similar to the common JSON type, JSON data can be queried with functions.
-
-For example:
-```SQL
-CREATE TABLE IF NOT EXISTS test (
-    ts TIMESTAMP TIME INDEX,
-    a INT,
-    b JSON
-);
-
-INSERT INTO test VALUES(
-    0,
-    0,
-    '{
-        "name": "jHl2oDDnPc1i2OzlP5Y",
-        "timestamp": "2024-07-25T04:33:11.369386Z",
-        "attributes": { "event_attributes": 48.28667 }
-    }'
-);
-
-SELECT json_get_string(b, 'name') FROM test;
-+---------------------+
-| b.name              |
-+---------------------+
-| jHl2oDDnPc1i2OzlP5Y |
-+---------------------+
-
-SELECT json_get_float(b, 'attributes.event_attributes') FROM test;
-+--------------------------------+
-| b.attributes.event_attributes  |
-+--------------------------------+
-| 48.28667                       |
-+--------------------------------+
-
-```
-And more functions can be added in the future.
-
-# Drawbacks
-
-As a general purpose JSON data type, JSONB may not be as efficient as specialized data types for specific scenarios.
-
-The auto-conversion mechanism is not supported in all scenarios. We need to find workarounds for these scenarios.
-
-# Alternatives
-
-Extract and flatten JSON schema to store in a structured format through pipeline. For nested data, we can provide nested types like `STRUCT` or `ARRAY`.
--- a/docs/schema-structs.md
+++ b/docs/schema-structs.md
@@ -0,0 +1,527 @@
+# Schema Structs
+
+# Common Schemas
+The `datatypes` crate defines the elementary schema struct to describe the metadata.
+
+## ColumnSchema
+[ColumnSchema](https://github.com/GreptimeTeam/greptimedb/blob/9fa871a3fad07f583dc1863a509414da393747f8/src/datatypes/src/schema/column_schema.rs#L36) represents the metadata of a column. It is equivalent to arrow's [Field](https://docs.rs/arrow/latest/arrow/datatypes/struct.Field.html) with additional metadata such as default constraint and whether the column is a time index. The time index is the column with a `TIME INDEX` constraint of a table. We can convert the `ColumnSchema` into an arrow `Field` and convert the `Field` back to the `ColumnSchema` without losing metadata.
+
+```rust
+pub struct ColumnSchema {
+    pub name: String,
+    pub data_type: ConcreteDataType,
+    is_nullable: bool,
+    is_time_index: bool,
+    default_constraint: Option<ColumnDefaultConstraint>,
+    metadata: Metadata,
+}
+```
+
+## Schema
+[Schema](https://github.com/GreptimeTeam/greptimedb/blob/9fa871a3fad07f583dc1863a509414da393747f8/src/datatypes/src/schema.rs#L38) is an ordered sequence of `ColumnSchema`. It is equivalent to arrow's [Schema](https://docs.rs/arrow/latest/arrow/datatypes/struct.Schema.html) with additional metadata including the index of the time index column and the version of this schema. Same as `ColumnSchema`, we can convert our `Schema` from/to arrow's `Schema`.
+
+```rust
+use arrow::datatypes::Schema as ArrowSchema;
+
+pub struct Schema {
+    column_schemas: Vec<ColumnSchema>,
+    name_to_index: HashMap<String, usize>,
+    arrow_schema: Arc<ArrowSchema>,
+    timestamp_index: Option<usize>,
+    version: u32,
+}
+
+pub type SchemaRef = Arc<Schema>;
+```
+
+We alias `Arc<Schema>` as `SchemaRef` since it is used frequently. Mostly, we use our `ColumnSchema` and `Schema` structs instead of Arrow's `Field` and `Schema` unless we need to invoke third-party libraries (like DataFusion or ArrowFlight) that rely on Arrow.
+
+## RawSchema
+`Schema` contains fields like a map from column names to their indices in the `ColumnSchema` sequences and a cached arrow `Schema`. We can construct these fields from the `ColumnSchema` sequences thus we don't want to serialize them. This is why we don't derive `Serialize` and `Deserialize` for `Schema`. We introduce a new struct [RawSchema](https://github.com/GreptimeTeam/greptimedb/blob/9fa871a3fad07f583dc1863a509414da393747f8/src/datatypes/src/schema/raw.rs#L24) which keeps all required fields of a `Schema` and derives the serialization traits. To serialize a `Schema`, we need to convert it into a `RawSchema` first and serialize the `RawSchema`.
+
+```rust
+pub struct RawSchema {
+    pub column_schemas: Vec<ColumnSchema>,
+    pub timestamp_index: Option<usize>,
+    pub version: u32,
+}
+```
+
+We want to keep the `Schema` simple and avoid putting too much business-related metadata in it as many different structs or traits rely on it.
+
+# Schema of the Table
+A table maintains its schema in [TableMeta](https://github.com/GreptimeTeam/greptimedb/blob/9fa871a3fad07f583dc1863a509414da393747f8/src/table/src/metadata.rs#L97).
+```rust
+pub struct TableMeta {
+    pub schema: SchemaRef,
+    pub primary_key_indices: Vec<usize>,
+    pub value_indices: Vec<usize>,
+    // ...
+}
+```
+
+The order of columns in `TableMeta::schema` is the same as the order specified in the `CREATE TABLE` statement which users use to create this table.
+
+The field `primary_key_indices` stores indices of primary key columns. The field `value_indices` records the indices of value columns (non-primary key and time index, we sometimes call them field columns).
+
+Suppose we create a table with the following SQL
+```sql
+CREATE TABLE cpu (
+    ts TIMESTAMP,
+    host STRING,
+    usage_user DOUBLE,
+    usage_system DOUBLE,
+    datacenter STRING,
+    TIME INDEX (ts),
+    PRIMARY KEY(datacenter, host)) ENGINE=mito;
+```
+
+Then the table's `TableMeta` may look like this:
+```json
+{
+    "schema":{
+        "column_schemas":[
+            "ts",
+            "host",
+            "usage_user",
+            "usage_system",
+            "datacenter"
+        ],
+        "time_index":0,
+        "version":0
+    },
+    "primary_key_indices":[
+        4,
+        1
+    ],
+    "value_indices":[
+        2,
+        3
+    ]
+}
+```
+
+
+# Schemas of the storage engine
+We split a table into one or more units with the same schema and then store these units in the storage engine. Each unit is a region in the storage engine.
+
+The storage engine maintains schemas of regions in more complicated ways because it
+- adds internal columns that are invisible to users to store additional metadata for each row
+- provides a data model similar to the key-value model so it organizes columns in a different order
+- maintains additional metadata like column id or column family
+
+So the storage engine defines several schema structs:
+- RegionSchema
+- StoreSchema
+- ProjectedSchema
+
+## RegionSchema
+A [RegionSchema](https://github.com/GreptimeTeam/greptimedb/blob/9fa871a3fad07f583dc1863a509414da393747f8/src/storage/src/schema/region.rs#L37) describes the schema of a region.
+
+```rust
+pub struct RegionSchema {
+    user_schema: SchemaRef,
+    store_schema: StoreSchemaRef,
+    columns: ColumnsMetadataRef,
+}
+```
+
+Each region reserves some columns called `internal columns` for internal usage:
+- `__sequence`, sequence number of a row
+- `__op_type`, operation type of a row, such as `PUT` or `DELETE`
+- `__version`, user-specified version of a row, reserved but not used. We might remove this in the future
+
+The table engine can't see the `__sequence` and `__op_type` columns, so the `RegionSchema` itself maintains two internal schemas:
+- User schema, a `Schema` struct that doesn't have internal columns
+- Store schema, a `StoreSchema` struct that has internal columns
+
+The `ColumnsMetadata` struct keeps metadata about all columns but most time we only need to use metadata in user schema and store schema, so we just ignore it. We may remove this struct in the future.
+
+`RegionSchema` organizes columns in the following order:
+```
+key columns, timestamp, [__version,] value columns, __sequence, __op_type
+```
+
+We can ignore the `__version` column because it is disabled now:
+
+```
+key columns, timestamp, value columns, __sequence, __op_type
+```
+
+Key columns are columns of a table's primary key. Timestamp is the time index column. A region sorts all rows by key columns, timestamp, sequence, and op type.
+
+So the `RegionSchema` of our `cpu` table above looks like this:
+```json
+{
+    "user_schema":[
+        "datacenter",
+        "host",
+        "ts",
+        "usage_user",
+        "usage_system"
+    ],
+    "store_schema":[
+        "datacenter",
+        "host",
+        "ts",
+        "usage_user",
+        "usage_system",
+        "__sequence",
+        "__op_type"
+    ]
+}
+```
+
+## StoreSchema
+As described above, a [StoreSchema](https://github.com/GreptimeTeam/greptimedb/blob/9fa871a3fad07f583dc1863a509414da393747f8/src/storage/src/schema/store.rs#L36) is a schema that knows all internal columns.
+```rust
+struct StoreSchema {
+    columns: Vec<ColumnMetadata>,
+    schema: SchemaRef,
+    row_key_end: usize,
+    user_column_end: usize,
+}
+```
+
+The columns in the `columns` and `schema` fields have the same order. The `ColumnMetadata` has metadata like column id, column family id, and comment. The `StoreSchema` also stores this metadata in `StoreSchema::schema`, so we can convert the `StoreSchema` between arrow's `Schema`. We use this feature to persist the `StoreSchema` in the SST since our SST format is `Parquet`, which can take arrow's `Schema` as its schema.
+
+The `StoreSchema` of the region above is similar to this:
+```json
+{
+    "schema":{
+        "column_schemas":[
+            "datacenter",
+            "host",
+            "ts",
+            "usage_user",
+            "usage_system",
+            "__sequence",
+            "__op_type"
+        ],
+        "time_index":2,
+        "version":0
+    },
+    "row_key_end":3,
+    "user_column_end":5
+}
+```
+
+The key and timestamp columns form row keys of rows. We put them together so we can use `row_key_end` to get indices of all row key columns. Similarly, we can use the `user_column_end` to get indices of all user columns (non-internal columns).
+```rust
+impl StoreSchema {
+    #[inline]
+    pub(crate) fn row_key_indices(&self) -> impl Iterator<Item = usize> {
+        0..self.row_key_end
+    }
+
+    #[inline]
+    pub(crate) fn value_indices(&self) -> impl Iterator<Item = usize> {
+        self.row_key_end..self.user_column_end
+    }
+}
+```
+
+Another useful feature of `StoreSchema` is that we ensure it always contains key columns, a timestamp column, and internal columns because we need them to perform merge, deduplication, and delete. Projection on `StoreSchema` only projects value columns.
+
+## ProjectedSchema
+To support arbitrary projection, we introduce the [ProjectedSchema](https://github.com/GreptimeTeam/greptimedb/blob/9fa871a3fad07f583dc1863a509414da393747f8/src/storage/src/schema/projected.rs#L106).
+```rust
+pub struct ProjectedSchema {
+    projection: Option<Projection>,
+    schema_to_read: StoreSchemaRef,
+    projected_user_schema: SchemaRef,
+}
+```
+
+We need to handle many cases while doing projection:
+- The columns' order of table and region is different
+- The projection can be in arbitrary order, e.g. `select usage_user, host from cpu` and `select host, usage_user from cpu` have different projection order
+- We support `ALTER TABLE` so data files may have different schemas.
+
+### Projection
+Let's take an example to see how projection works. Suppose we want to select `ts`, `usage_system` from the `cpu` table.
+
+```sql
+CREATE TABLE cpu (
+    ts TIMESTAMP,
+    host STRING,
+    usage_user DOUBLE,
+    usage_system DOUBLE,
+    datacenter STRING,
+    TIME INDEX (ts),
+    PRIMARY KEY(datacenter, host)) ENGINE=mito;
+
+select ts, usage_system from cpu;
+```
+
+The query engine uses the projection `[0, 3]` to scan the table. However, columns in the region have a different order, so the table engine adjusts the projection to `2, 4`.
+```json
+{
+    "user_schema":[
+        "datacenter",
+        "host",
+        "ts",
+        "usage_user",
+        "usage_system"
+    ],
+}
+```
+
+As you can see, the output order is still `[ts, usage_system]`. This is the schema users can see after projection so we call it `projected user schema`.
+
+But the storage engine also needs to read key columns, a timestamp column, and internal columns. So we maintain a `StoreSchema` after projection in the `ProjectedSchema`.
+
+The `Projection` struct is a helper struct to help compute the projected user schema and store schema.
+
+So we can construct the following `ProjectedSchema`:
+```json
+{
+    "schema_to_read":{
+        "schema":{
+            "column_schemas":[
+                "datacenter",
+                "host",
+                "ts",
+                "usage_system",
+                "__sequence",
+                "__op_type"
+            ],
+            "time_index":2,
+            "version":0
+        },
+        "row_key_end":3,
+        "user_column_end":4
+    },
+    "projected_user_schema":{
+        "column_schemas":[
+            "ts",
+            "usage_system"
+        ],
+        "time_index":0
+    }
+}
+```
+
+As you can see, `schema_to_read` doesn't contain the column `usage_user` that is not intended to be read (not in projection).
+
+### ReadAdapter
+As mentioned above, we can alter a table so the underlying files (SSTs) and memtables in the storage engine may have different schemas.
+
+To simplify the logic of `ProjectedSchema`, we handle the difference between schemas before projection (constructing the `ProjectedSchema`). We introduce [ReadAdapter](https://github.com/GreptimeTeam/greptimedb/blob/9fa871a3fad07f583dc1863a509414da393747f8/src/storage/src/schema/compat.rs#L90) that adapts rows with different source schemas to the same expected schema.
+
+So we can always use the current `RegionSchema` of the region to construct the `ProjectedSchema`, and then create a `ReadAdapter` for each memtable or SST.
+```rust
+#[derive(Debug)]
+pub struct ReadAdapter {
+    source_schema: StoreSchemaRef,
+    dest_schema: ProjectedSchemaRef,
+    indices_in_result: Vec<Option<usize>>,
+    is_source_needed: Vec<bool>,
+}
+```
+
+For each column required by `dest_schema`, `indices_in_result` stores the index of that column in the row read from the source memtable or SST. If the source row doesn't contain that column, the index is `None`.
+
+The field `is_source_needed` stores whether a column in the source memtable or SST is needed.
+
+Suppose we add a new column `usage_idle` to the table `cpu`.
+```sql
+ALTER TABLE cpu ADD COLUMN usage_idle DOUBLE;
+```
+
+The new `StoreSchema` becomes:
+```json
+{
+    "schema":{
+        "column_schemas":[
+            "datacenter",
+            "host",
+            "ts",
+            "usage_user",
+            "usage_system",
+            "usage_idle",
+            "__sequence",
+            "__op_type"
+        ],
+        "time_index":2,
+        "version":1
+    },
+    "row_key_end":3,
+    "user_column_end":6
+}
+```
+
+Note that we bump the version of the schema to 1.
+
+If we want to select `ts`, `usage_system`, and `usage_idle`. While reading from the old schema, the storage engine creates a `ReadAdapter` like this:
+```json
+{
+    "source_schema":{
+        "schema":{
+            "column_schemas":[
+                "datacenter",
+                "host",
+                "ts",
+                "usage_user",
+                "usage_system",
+                "__sequence",
+                "__op_type"
+            ],
+            "time_index":2,
+            "version":0
+        },
+        "row_key_end":3,
+        "user_column_end":5
+    },
+    "dest_schema":{
+        "schema_to_read":{
+            "schema":{
+                "column_schemas":[
+                    "datacenter",
+                    "host",
+                    "ts",
+                    "usage_system",
+                    "usage_idle",
+                    "__sequence",
+                    "__op_type"
+                ],
+                "time_index":2,
+                "version":1
+            },
+            "row_key_end":3,
+            "user_column_end":5
+        },
+        "projected_user_schema":{
+            "column_schemas":[
+                "ts",
+                "usage_system",
+                "usage_idle"
+            ],
+            "time_index":0
+        }
+    },
+    "indices_in_result":[
+        0,
+        1,
+        2,
+        3,
+        null,
+        4,
+        5
+    ],
+    "is_source_needed":[
+        true,
+        true,
+        true,
+        false,
+        true,
+        true,
+        true
+    ]
+}
+```
+
+We don't need to read `usage_user` so `is_source_needed[3]` is false. The old schema doesn't have column `usage_idle` so `indices_in_result[4]` is `null` and the `ReadAdapter` needs to insert a null column to the output row so the output schema still contains `usage_idle`.
+
+The figure below shows the relationship between `RegionSchema`, `StoreSchema`, `ProjectedSchema`, and `ReadAdapter`.
+
+```text
+                   ┌──────────────────────────────┐
+                   │                              │
+                   │    ┌────────────────────┐    │
+                   │    │    store_schema    │    │
+                   │    │                    │    │
+                   │    │     StoreSchema    │    │
+                   │    │      version 1     │    │
+                   │    └────────────────────┘    │
+                   │                              │
+                   │    ┌────────────────────┐    │
+                   │    │     user_schema    │    │
+                   │    └────────────────────┘    │
+                   │                              │
+                   │         RegionSchema         │
+                   │                              │
+                   └──────────────┬───────────────┘
+                                  │
+                                  │
+                                  │
+                   ┌──────────────▼───────────────┐
+                   │                              │
+                   │ ┌──────────────────────────┐ │
+                   │ │     schema_to_read       │ │
+                   │ │                          │ │
+                   │ │  StoreSchema (projected) │ │
+                   │ │       version 1          │ │
+                   │ └──────────────────────────┘ │
+               ┌───┤                              ├───┐
+               │   │ ┌──────────────────────────┐ │   │
+               │   │ │  projected_user_schema   │ │   │
+               │   │ └──────────────────────────┘ │   │
+               │   │                              │   │
+               │   │       ProjectedSchema        │   │
+  dest schema  │   └──────────────────────────────┘   │   dest schema
+               │                                      │
+               │                                      │
+        ┌──────▼───────┐                      ┌───────▼──────┐
+        │              │                      │              │
+        │  ReadAdapter │                      │  ReadAdapter │
+        │              │                      │              │
+        └──────▲───────┘                      └───────▲──────┘
+               │                                      │
+               │                                      │
+source schema  │                                      │  source schema
+               │                                      │
+       ┌───────┴─────────┐                   ┌────────┴────────┐
+       │                 │                   │                 │
+       │ ┌─────────────┐ │                   │ ┌─────────────┐ │
+       │ │             │ │                   │ │             │ │
+       │ │ StoreSchema │ │                   │ │ StoreSchema │ │
+       │ │             │ │                   │ │             │ │
+       │ │  version 0  │ │                   │ │  version 1  │ │
+       │ │             │ │                   │ │             │ │
+       │ └─────────────┘ │                   │ └─────────────┘ │
+       │                 │                   │                 │
+       │      SST 0      │                   │      SST 1      │
+       │                 │                   │                 │
+       └─────────────────┘                   └─────────────────┘
+```
+
+# Conversion
+This figure shows the conversion between schemas:
+```text
+              ┌─────────────┐     schema                      From             ┌─────────────┐
+              │             ├──────────────────┐  ┌────────────────────────────►             │
+              │  TableMeta  │                  │  │                            │  RawSchema  │
+              │             │                  │  │  ┌─────────────────────────┤             │
+              └─────────────┘                  │  │  │        TryFrom          └─────────────┘
+                                               │  │  │
+                                               │  │  │
+                                               │  │  │
+                                               │  │  │
+                                               │  │  │
+    ┌───────────────────┐                ┌─────▼──┴──▼──┐   arrow_schema()    ┌─────────────────┐
+    │                   │                │              ├─────────────────────►                 │
+    │  ColumnsMetadata  │          ┌─────►    Schema    │                     │   ArrowSchema   ├──┐
+    │                   │          │     │              ◄─────────────────────┤                 │  │
+    └────┬───────────▲──┘          │     └───▲───▲──────┘       TryFrom       └─────────────────┘  │
+         │           │             │         │   │                                                 │
+         │           │             │         │   └────────────────────────────────────────┐        │
+         │           │             │         │                                            │        │
+         │   columns │    user_schema()      │                                            │        │
+         │           │             │         │ projected_user_schema()                 schema()    │
+         │           │             │         │                                            │        │
+         │       ┌───┴─────────────┴─┐       │                 ┌────────────────────┐     │        │
+columns  │       │                   │       └─────────────────┤                    │     │        │  TryFrom
+         │       │    RegionSchema   │                         │   ProjectedSchema  │     │        │
+         │       │                   ├─────────────────────────►                    │     │        │
+         │       └─────────────────┬─┘  ProjectedSchema::new() └──────────────────┬─┘     │        │
+         │                         │                                              │       │        │
+         │                         │                                              │       │        │
+         │                         │                                              │       │        │
+         │                         │                                              │       │        │
+    ┌────▼────────────────────┐    │               store_schema()            ┌────▼───────┴──┐     │
+    │                         │    └─────────────────────────────────────────►               │     │
+    │   Vec<ColumnMetadata>   │                                              │  StoreSchema  ◄─────┘
+    │                         ◄──────────────────────────────────────────────┤               │
+    └─────────────────────────┘                     columns                  └───────────────┘
+```
--- a/flake.lock
+++ b/flake.lock
@@ -1,100 +0,0 @@
-{
-  "nodes": {
-    "fenix": {
-      "inputs": {
-        "nixpkgs": [
-          "nixpkgs"
-        ],
-        "rust-analyzer-src": "rust-analyzer-src"
-      },
-      "locked": {
-        "lastModified": 1737613896,
-        "narHash": "sha256-ldqXIglq74C7yKMFUzrS9xMT/EVs26vZpOD68Sh7OcU=",
-        "owner": "nix-community",
-        "repo": "fenix",
-        "rev": "303a062fdd8e89f233db05868468975d17855d80",
-        "type": "github"
-      },
-      "original": {
-        "owner": "nix-community",
-        "repo": "fenix",
-        "type": "github"
-      }
-    },
-    "flake-utils": {
-      "inputs": {
-        "systems": "systems"
-      },
-      "locked": {
-        "lastModified": 1731533236,
-        "narHash": "sha256-l0KFg5HjrsfsO/JpG+r7fRrqm12kzFHyUHqHCVpMMbI=",
-        "owner": "numtide",
-        "repo": "flake-utils",
-        "rev": "11707dc2f618dd54ca8739b309ec4fc024de578b",
-        "type": "github"
-      },
-      "original": {
-        "owner": "numtide",
-        "repo": "flake-utils",
-        "type": "github"
-      }
-    },
-    "nixpkgs": {
-      "locked": {
-        "lastModified": 1737569578,
-        "narHash": "sha256-6qY0pk2QmUtBT9Mywdvif0i/CLVgpCjMUn6g9vB+f3M=",
-        "owner": "NixOS",
-        "repo": "nixpkgs",
-        "rev": "47addd76727f42d351590c905d9d1905ca895b82",
-        "type": "github"
-      },
-      "original": {
-        "owner": "NixOS",
-        "ref": "nixos-24.11",
-        "repo": "nixpkgs",
-        "type": "github"
-      }
-    },
-    "root": {
-      "inputs": {
-        "fenix": "fenix",
-        "flake-utils": "flake-utils",
-        "nixpkgs": "nixpkgs"
-      }
-    },
-    "rust-analyzer-src": {
-      "flake": false,
-      "locked": {
-        "lastModified": 1737581772,
-        "narHash": "sha256-t1P2Pe3FAX9TlJsCZbmJ3wn+C4qr6aSMypAOu8WNsN0=",
-        "owner": "rust-lang",
-        "repo": "rust-analyzer",
-        "rev": "582af7ee9c8d84f5d534272fc7de9f292bd849be",
-        "type": "github"
-      },
-      "original": {
-        "owner": "rust-lang",
-        "ref": "nightly",
-        "repo": "rust-analyzer",
-        "type": "github"
-      }
-    },
-    "systems": {
-      "locked": {
-        "lastModified": 1681028828,
-        "narHash": "sha256-Vy1rq5AaRuLzOxct8nz4T6wlgyUR7zLU309k9mBC768=",
-        "owner": "nix-systems",
-        "repo": "default",
-        "rev": "da67096a3b9bf56a91d16901293e51ba5b49a27e",
-        "type": "github"
-      },
-      "original": {
-        "owner": "nix-systems",
-        "repo": "default",
-        "type": "github"
-      }
-    }
-  },
-  "root": "root",
-  "version": 7
-}
--- a/flake.nix
+++ b/flake.nix
@@ -1,56 +0,0 @@
-{
-  description = "Development environment flake";
-
-  inputs = {
-    nixpkgs.url = "github:NixOS/nixpkgs/nixos-24.11";
-    fenix = {
-      url = "github:nix-community/fenix";
-      inputs.nixpkgs.follows = "nixpkgs";
-    };
-    flake-utils.url = "github:numtide/flake-utils";
-  };
-
-  outputs = { self, nixpkgs, fenix, flake-utils }:
-    flake-utils.lib.eachDefaultSystem (system:
-      let
-        pkgs = nixpkgs.legacyPackages.${system};
-        buildInputs = with pkgs; [
-          libgit2
-          libz
-        ];
-        lib = nixpkgs.lib;
-        rustToolchain = fenix.packages.${system}.fromToolchainName {
-          name = (lib.importTOML ./rust-toolchain.toml).toolchain.channel;
-          sha256 = "sha256-f/CVA1EC61EWbh0SjaRNhLL0Ypx2ObupbzigZp8NmL4=";
-        };
-      in
-      {
-        devShells.default = pkgs.mkShell {
-          nativeBuildInputs = with pkgs; [
-            pkg-config
-            git
-            clang
-            gcc
-            protobuf
-            gnumake
-            mold
-            (rustToolchain.withComponents [
-              "cargo"
-              "clippy"
-              "rust-src"
-              "rustc"
-              "rustfmt"
-              "rust-analyzer"
-              "llvm-tools"
-            ])
-            cargo-nextest
-            cargo-llvm-cov
-            taplo
-            curl
-            gnuplot ## for cargo bench
-          ];
-
-          LD_LIBRARY_PATH = pkgs.lib.makeLibraryPath buildInputs;
-        };
-      });
-}
--- a/grafana/README.md
+++ b/grafana/README.md
@@ -5,13 +5,6 @@ GreptimeDB's official Grafana dashboard.

 Status notify: we are still working on this config. It's expected to change frequently in the recent days. Please feel free to submit your feedback and/or contribution to this dashboard 🤗

-If you use Helm [chart](https://github.com/GreptimeTeam/helm-charts) to deploy GreptimeDB cluster, you can enable self-monitoring by setting the following values in your Helm chart:
-
- `monitoring.enabled=true`: Deploys a standalone GreptimeDB instance dedicated to monitoring the cluster;
- `grafana.enabled=true`: Deploys Grafana and automatically imports the monitoring dashboard;
-
-The standalone GreptimeDB instance will collect metrics from your cluster and the dashboard will be available in the Grafana UI. For detailed deployment instructions, please refer to our [Kubernetes deployment guide](https://docs.greptime.com/nightly/user-guide/deployments/deploy-on-kubernetes/getting-started).
-
 # How to use

 ## `greptimedb.json`
@@ -32,7 +25,7 @@ Please ensure the following configuration before importing the dashboard into Gr

 __1. Prometheus scrape config__

-Configure Prometheus to scrape the cluster.
+Assign `greptime_pod` label to each host target. We use this label to identify each node instance.

 ```yml
 # example config
@@ -41,15 +34,27 @@ Configure Prometheus to scrape the cluster.
 scrape_configs:
  - job_name: metasrv
    static_configs:
-    - targets: ['<metasrv-ip>:<port>']
+    - targets: ['<ip>:<port>']
+      labels:
+        greptime_pod: metasrv

  - job_name: datanode
    static_configs:
-    - targets: ['<datanode0-ip>:<port>', '<datanode1-ip>:<port>', '<datanode2-ip>:<port>']
+    - targets: ['<ip>:<port>']
+      labels:
+        greptime_pod: datanode1
+    - targets: ['<ip>:<port>']
+      labels:
+        greptime_pod: datanode2
+    - targets: ['<ip>:<port>']
+      labels:
+        greptime_pod: datanode3

  - job_name: frontend
    static_configs:
-    - targets: ['<frontend-ip>:<port>']
+    - targets: ['<ip>:<port>']
+      labels:
+        greptime_pod: frontend
 ```

 __2. Grafana config__
@@ -58,4 +63,4 @@ Create a Prometheus data source in Grafana before using this dashboard. We use `

 ### Usage

-Use `datasource` or `instance` on the upper-left corner to filter data from certain node.
+Use `datasource` or `greptime_pod` on the upper-left corner to filter data from certain node.
--- a/grafana/check.sh
+++ b/grafana/check.sh
@@ -1,19 +0,0 @@
-#!/usr/bin/env bash
-
-BASEDIR=$(dirname "$0")
-
-# Use jq to check for panels with empty or missing descriptions
-invalid_panels=$(cat $BASEDIR/greptimedb-cluster.json | jq -r '
-  .panels[]
-  | select((.type == "stats" or .type == "timeseries") and (.description == "" or .description == null))
-')
-
-# Check if any invalid panels were found
-if [[ -n "$invalid_panels" ]]; then
-  echo "Error: The following panels have empty or missing descriptions:"
-  echo "$invalid_panels"
-  exit 1
-else
-  echo "All panels with type 'stats' or 'timeseries' have valid descriptions."
-  exit 0
-fi
--- a/grafana/greptimedb-cluster.json
+++ b/grafana/greptimedb-cluster.json
--- a/grafana/greptimedb.json
+++ b/grafana/greptimedb.json
--- a/Show More
+++ b/Show More
Author	SHA1	Message	Date
Ruihang Xia	1bfba48755	Revert "build(deps): upgrade opendal to 0.46 (#4037 )" This reverts commit `f9db5ff0d6`.	2024-06-03 20:28:59 +08:00
Ruihang Xia	457998f0fe	Merge branch 'main' into avoid-query-meta Signed-off-by: Ruihang Xia <waynestxia@gmail.com>	2024-05-31 18:16:36 +08:00
Ruihang Xia	b02c256157	perf: use memory state to check if a logical region exists Signed-off-by: Ruihang Xia <waynestxia@gmail.com>	2024-05-31 18:16:08 +08:00