fix(log-query): panic on prometheus (#5429 )

* fix(log-query): panic on prometheus Signed-off-by: Ruihang Xia <waynestxia@gmail.com> * fix test environment setup Signed-off-by: Ruihang Xia <waynestxia@gmail.com> --------- Signed-off-by: Ruihang Xia <waynestxia@gmail.com>
fix: avoid suppress manual compaction (#5399 )
2025-12-27 16:32:54 +00:00 · 2025-01-23 23:03:20 +08:00 · 2025-01-23 19:23:06 +08:00 · 2025-01-23 19:23:06 +08:00 · 2025-01-23 19:23:06 +08:00 · 2025-01-23 19:23:06 +08:00
315 changed files with 15254 additions and 3238 deletions
--- a/.github/actions/build-greptime-binary/action.yml
+++ b/.github/actions/build-greptime-binary/action.yml
@@ -54,7 +54,7 @@ runs:
        PROFILE_TARGET: ${{ inputs.cargo-profile == 'dev' && 'debug' || inputs.cargo-profile }}
      with:
        artifacts-dir: ${{ inputs.artifacts-dir }}
-        target-file: ./target/$PROFILE_TARGET/greptime
+        target-files: ./target/$PROFILE_TARGET/greptime
        version: ${{ inputs.version }}
        working-dir: ${{ inputs.working-dir }}

@@ -72,6 +72,6 @@ runs:
      if: ${{ inputs.build-android-artifacts == 'true' }}
      with:
        artifacts-dir: ${{ inputs.artifacts-dir }}
-        target-file: ./target/aarch64-linux-android/release/greptime
+        target-files: ./target/aarch64-linux-android/release/greptime
        version: ${{ inputs.version }}
        working-dir: ${{ inputs.working-dir }}
--- a/.github/actions/build-images/action.yml
+++ b/.github/actions/build-images/action.yml
@@ -41,8 +41,8 @@ runs:
        image-name: ${{ inputs.image-name }}
        image-tag: ${{ inputs.version }}
        docker-file: docker/ci/ubuntu/Dockerfile
-        amd64-artifact-name: greptime-linux-amd64-pyo3-${{ inputs.version }}
-        arm64-artifact-name: greptime-linux-arm64-pyo3-${{ inputs.version }}
+        amd64-artifact-name: greptime-linux-amd64-${{ inputs.version }}
+        arm64-artifact-name: greptime-linux-arm64-${{ inputs.version }}
        platforms: linux/amd64,linux/arm64
        push-latest-tag: ${{ inputs.push-latest-tag }}

--- a/.github/actions/build-linux-artifacts/action.yml
+++ b/.github/actions/build-linux-artifacts/action.yml
@@ -48,19 +48,7 @@ runs:
        path: /tmp/greptime-*.log
        retention-days: 3

-    - name: Build standard greptime
-      uses: ./.github/actions/build-greptime-binary
-      with:
-        base-image: ubuntu
-        features: pyo3_backend,servers/dashboard
-        cargo-profile: ${{ inputs.cargo-profile }}
-        artifacts-dir: greptime-linux-${{ inputs.arch }}-pyo3-${{ inputs.version }}
-        version: ${{ inputs.version }}
-        working-dir: ${{ inputs.working-dir }}
-        image-registry: ${{ inputs.image-registry }}
-        image-namespace: ${{ inputs.image-namespace }}
-
-    - name: Build greptime without pyo3
+    - name: Build greptime
      if: ${{ inputs.dev-mode == 'false' }}
      uses: ./.github/actions/build-greptime-binary
      with:
--- a/.github/actions/build-macos-artifacts/action.yml
+++ b/.github/actions/build-macos-artifacts/action.yml
@@ -90,5 +90,5 @@ runs:
      uses: ./.github/actions/upload-artifacts
      with:
        artifacts-dir: ${{ inputs.artifacts-dir }}
-        target-file: target/${{ inputs.arch }}/${{ inputs.cargo-profile }}/greptime
+        target-files: target/${{ inputs.arch }}/${{ inputs.cargo-profile }}/greptime
        version: ${{ inputs.version }}
--- a/.github/actions/build-windows-artifacts/action.yml
+++ b/.github/actions/build-windows-artifacts/action.yml
@@ -33,15 +33,6 @@ runs:
    - name: Rust Cache
      uses: Swatinem/rust-cache@v2

-    - name: Install Python
-      uses: actions/setup-python@v5
-      with:
-        python-version: "3.10"
-
-    - name: Install PyArrow Package
-      shell: pwsh
-      run: pip install pyarrow numpy
-
    - name: Install WSL distribution
      uses: Vampire/setup-wsl@v2
      with:
@@ -76,5 +67,5 @@ runs:
      uses: ./.github/actions/upload-artifacts
      with:
        artifacts-dir: ${{ inputs.artifacts-dir }}
-        target-file: target/${{ inputs.arch }}/${{ inputs.cargo-profile }}/greptime
+        target-files: target/${{ inputs.arch }}/${{ inputs.cargo-profile }}/greptime,target/${{ inputs.arch }}/${{ inputs.cargo-profile }}/greptime.pdb
        version: ${{ inputs.version }}
--- a/.github/actions/setup-greptimedb-cluster/with-minio-and-cache.yaml
+++ b/.github/actions/setup-greptimedb-cluster/with-minio-and-cache.yaml
@@ -5,7 +5,7 @@ meta:

    [datanode]
    [datanode.client]
-    timeout = "60s"
+    timeout = "120s"
 datanode:
  configData: |-
    [runtime]
@@ -21,7 +21,7 @@ frontend:
    global_rt_size = 4

    [meta_client]
-    ddl_timeout = "60s"
+    ddl_timeout = "120s"
 objectStorage:
  s3:
    bucket: default
--- a/.github/actions/setup-greptimedb-cluster/with-minio.yaml
+++ b/.github/actions/setup-greptimedb-cluster/with-minio.yaml
@@ -5,7 +5,7 @@ meta:
    
    [datanode]
    [datanode.client]
-    timeout = "60s"
+    timeout = "120s"
 datanode:
  configData: |-
    [runtime]
@@ -17,7 +17,7 @@ frontend:
    global_rt_size = 4

    [meta_client]
-    ddl_timeout = "60s"
+    ddl_timeout = "120s"
 objectStorage:
  s3:
    bucket: default
--- a/.github/actions/setup-greptimedb-cluster/with-remote-wal.yaml
+++ b/.github/actions/setup-greptimedb-cluster/with-remote-wal.yaml
@@ -11,7 +11,7 @@ meta:
        
    [datanode]
    [datanode.client]
-    timeout = "60s"
+    timeout = "120s"
 datanode:
  configData: |-
    [runtime]
@@ -28,7 +28,7 @@ frontend:
    global_rt_size = 4

    [meta_client]
-    ddl_timeout = "60s"
+    ddl_timeout = "120s"
 objectStorage:
  s3:
    bucket: default
--- a/.github/actions/upload-artifacts/action.yml
+++ b/.github/actions/upload-artifacts/action.yml
@@ -4,8 +4,8 @@ inputs:
  artifacts-dir:
    description: Directory to store artifacts
    required: true
-  target-file:
-    description: The path of the target artifact
+  target-files:
+    description: The multiple target files to upload, separated by comma
    required: false
  version:
    description: Version of the artifact
@@ -18,12 +18,16 @@ runs:
  using: composite
  steps:
    - name: Create artifacts directory
-      if: ${{ inputs.target-file != '' }}
+      if: ${{ inputs.target-files != '' }}
      working-directory: ${{ inputs.working-dir }}
      shell: bash
      run: |
-        mkdir -p ${{ inputs.artifacts-dir }} && \
-        cp ${{ inputs.target-file }} ${{ inputs.artifacts-dir }}
+        set -e
+        mkdir -p ${{ inputs.artifacts-dir }}
+        IFS=',' read -ra FILES <<< "${{ inputs.target-files }}"
+        for file in "${FILES[@]}"; do
+          cp "$file" ${{ inputs.artifacts-dir }}/
+        done

    # The compressed artifacts will use the following layout:
    # greptime-linux-amd64-pyo3-v0.3.0sha256sum
--- a/.github/workflows/dependency-check.yml
+++ b/.github/workflows/dependency-check.yml
@@ -1,9 +1,6 @@
 name: Check Dependencies

 on:
-  push:
-    branches:
-      - main
  pull_request:
    branches:
      - main
--- a/.github/workflows/develop.yml
+++ b/.github/workflows/develop.yml
@@ -10,17 +10,6 @@ on:
      - 'docker/**'
      - '.gitignore'
      - 'grafana/**'
-  push:
-    branches:
-      - main
-    paths-ignore:
-      - 'docs/**'
-      - 'config/**'
-      - '**.md'
-      - '.dockerignore'
-      - 'docker/**'
-      - '.gitignore'
-      - 'grafana/**'
  workflow_dispatch:

 name: CI
@@ -54,7 +43,7 @@ jobs:
    runs-on: ${{ matrix.os }}
    strategy:
      matrix:
-        os: [ windows-2022, ubuntu-20.04 ]
+        os: [ ubuntu-20.04 ]
    timeout-minutes: 60
    steps:
      - uses: actions/checkout@v4
@@ -68,6 +57,8 @@ jobs:
          # Shares across multiple jobs
          # Shares with `Clippy` job
          shared-key: "check-lint"
+          cache-all-crates: "true"
+          save-if: ${{ github.ref == 'refs/heads/main' }}
      - name: Run cargo check
        run: cargo check --locked --workspace --all-targets

@@ -78,13 +69,8 @@ jobs:
    steps:
      - uses: actions/checkout@v4
      - uses: actions-rust-lang/setup-rust-toolchain@v1
-      - name: Rust Cache
-        uses: Swatinem/rust-cache@v2
-        with:
-          # Shares across multiple jobs
-          shared-key: "check-toml"
      - name: Install taplo
-        run: cargo +stable install taplo-cli --version ^0.9 --locked
+        run: cargo +stable install taplo-cli --version ^0.9 --locked --force
      - name: Run taplo
        run: taplo format --check

@@ -105,13 +91,15 @@ jobs:
        with:
          # Shares across multiple jobs
          shared-key: "build-binaries"
+          cache-all-crates: "true"
+          save-if: ${{ github.ref == 'refs/heads/main' }}
      - name: Install cargo-gc-bin
        shell: bash
-        run: cargo install cargo-gc-bin
+        run: cargo install cargo-gc-bin --force
      - name: Build greptime binaries
        shell: bash
        # `cargo gc` will invoke `cargo build` with specified args
-        run: cargo gc -- --bin greptime --bin sqlness-runner
+        run: cargo gc -- --bin greptime --bin sqlness-runner --features pg_kvbackend
      - name: Pack greptime binaries
        shell: bash
        run: |
@@ -153,17 +141,12 @@ jobs:
        with:
          repo-token: ${{ secrets.GITHUB_TOKEN }}
      - uses: actions-rust-lang/setup-rust-toolchain@v1
-      - name: Rust Cache
-        uses: Swatinem/rust-cache@v2
-        with:
-          # Shares across multiple jobs
-          shared-key: "fuzz-test-targets"
      - name: Set Rust Fuzz
        shell: bash
        run: |
          sudo apt-get install -y libfuzzer-14-dev
          rustup install nightly
-          cargo +nightly install cargo-fuzz cargo-gc-bin
+          cargo +nightly install cargo-fuzz cargo-gc-bin --force
      - name: Download pre-built binaries
        uses: actions/download-artifact@v4
        with:
@@ -211,16 +194,11 @@ jobs:
        with:
          repo-token: ${{ secrets.GITHUB_TOKEN }}
      - uses: actions-rust-lang/setup-rust-toolchain@v1
-      - name: Rust Cache
-        uses: Swatinem/rust-cache@v2
-        with:
-          # Shares across multiple jobs
-          shared-key: "fuzz-test-targets"
      - name: Set Rust Fuzz
        shell: bash
        run: |
          sudo apt update && sudo apt install -y libfuzzer-14-dev
-          cargo install cargo-fuzz cargo-gc-bin
+          cargo install cargo-fuzz cargo-gc-bin --force
      - name: Download pre-built binariy
        uses: actions/download-artifact@v4
        with:
@@ -266,13 +244,15 @@ jobs:
        with:
          # Shares across multiple jobs
          shared-key: "build-greptime-ci"
+          cache-all-crates: "true"
+          save-if: ${{ github.ref == 'refs/heads/main' }}
      - name: Install cargo-gc-bin
        shell: bash
-        run: cargo install cargo-gc-bin
+        run: cargo install cargo-gc-bin --force
      - name: Build greptime bianry
        shell: bash
        # `cargo gc` will invoke `cargo build` with specified args
-        run: cargo gc --profile ci -- --bin greptime
+        run: cargo gc --profile ci -- --bin greptime --features pg_kvbackend
      - name: Pack greptime binary
        shell: bash
        run: |
@@ -328,17 +308,12 @@ jobs:
        with:
          repo-token: ${{ secrets.GITHUB_TOKEN }}
      - uses: actions-rust-lang/setup-rust-toolchain@v1
-      - name: Rust Cache
-        uses: Swatinem/rust-cache@v2
-        with:
-          # Shares across multiple jobs
-          shared-key: "fuzz-test-targets"
      - name: Set Rust Fuzz
        shell: bash
        run: |
          sudo apt-get install -y libfuzzer-14-dev
          rustup install nightly
-          cargo +nightly install cargo-fuzz cargo-gc-bin
+          cargo +nightly install cargo-fuzz cargo-gc-bin --force
      # Downloads ci image
      - name: Download pre-built binariy
        uses: actions/download-artifact@v4
@@ -477,17 +452,12 @@ jobs:
        with:
          repo-token: ${{ secrets.GITHUB_TOKEN }}
      - uses: actions-rust-lang/setup-rust-toolchain@v1
-      - name: Rust Cache
-        uses: Swatinem/rust-cache@v2
-        with:
-          # Shares across multiple jobs
-          shared-key: "fuzz-test-targets"
      - name: Set Rust Fuzz
        shell: bash
        run: |
          sudo apt-get install -y libfuzzer-14-dev
          rustup install nightly
-          cargo +nightly install cargo-fuzz cargo-gc-bin
+          cargo +nightly install cargo-fuzz cargo-gc-bin --force
      # Downloads ci image
      - name: Download pre-built binariy
        uses: actions/download-artifact@v4
@@ -589,8 +559,8 @@ jobs:
      - uses: actions/checkout@v4
      - if: matrix.mode.kafka
        name: Setup kafka server
-        working-directory: tests-integration/fixtures/kafka
-        run: docker compose -f docker-compose-standalone.yml up -d --wait
+        working-directory: tests-integration/fixtures
+        run: docker compose up -d --wait kafka
      - name: Download pre-built binaries
        uses: actions/download-artifact@v4
        with:
@@ -620,11 +590,6 @@ jobs:
      - uses: actions-rust-lang/setup-rust-toolchain@v1
        with:
          components: rustfmt
-      - name: Rust Cache
-        uses: Swatinem/rust-cache@v2
-        with:
-          # Shares across multiple jobs
-          shared-key: "check-rust-fmt"
      - name: Check format
        run: make fmt-check

@@ -646,11 +611,69 @@ jobs:
          # Shares across multiple jobs
          # Shares with `Check` job
          shared-key: "check-lint"
+          cache-all-crates: "true"
+          save-if: ${{ github.ref == 'refs/heads/main' }}
      - name: Run cargo clippy
        run: make clippy

+  conflict-check:
+    name: Check for conflict
+    runs-on: ubuntu-latest
+    steps:
+      - uses: actions/checkout@v4
+      - name: Merge Conflict Finder
+        uses: olivernybroe/action-conflict-finder@v4.0
+
+  test:
+    if: github.event_name != 'merge_group'
+    runs-on: ubuntu-24.04-arm
+    timeout-minutes: 60
+    needs:  [conflict-check, clippy, fmt]
+    steps:
+      - uses: actions/checkout@v4
+      - uses: arduino/setup-protoc@v3
+        with:
+          repo-token: ${{ secrets.GITHUB_TOKEN }}
+      - uses: rui314/setup-mold@v1
+      - name: Install toolchain
+        uses: actions-rust-lang/setup-rust-toolchain@v1
+        with:
+            cache: false
+      - name: Rust Cache
+        uses: Swatinem/rust-cache@v2
+        with:
+          # Shares cross multiple jobs
+          shared-key: "coverage-test"
+          cache-all-crates: "true"
+          save-if: ${{ github.ref == 'refs/heads/main' }}
+      - name: Install latest nextest release
+        uses: taiki-e/install-action@nextest
+      - name: Setup external services
+        working-directory: tests-integration/fixtures
+        run: docker compose up -d --wait
+      - name: Run nextest cases
+        run: cargo nextest run --workspace -F dashboard -F pg_kvbackend
+        env:
+          CARGO_BUILD_RUSTFLAGS: "-C link-arg=-fuse-ld=mold"
+          RUST_BACKTRACE: 1
+          CARGO_INCREMENTAL: 0
+          GT_S3_BUCKET: ${{ vars.AWS_CI_TEST_BUCKET }}
+          GT_S3_ACCESS_KEY_ID: ${{ secrets.AWS_CI_TEST_ACCESS_KEY_ID }}
+          GT_S3_ACCESS_KEY: ${{ secrets.AWS_CI_TEST_SECRET_ACCESS_KEY }}
+          GT_S3_REGION: ${{ vars.AWS_CI_TEST_BUCKET_REGION }}
+          GT_MINIO_BUCKET: greptime
+          GT_MINIO_ACCESS_KEY_ID: superpower_ci_user
+          GT_MINIO_ACCESS_KEY: superpower_password
+          GT_MINIO_REGION: us-west-2
+          GT_MINIO_ENDPOINT_URL: http://127.0.0.1:9000
+          GT_ETCD_ENDPOINTS: http://127.0.0.1:2379
+          GT_POSTGRES_ENDPOINTS: postgres://greptimedb:admin@127.0.0.1:5432/postgres
+          GT_KAFKA_ENDPOINTS: 127.0.0.1:9092
+          GT_KAFKA_SASL_ENDPOINTS: 127.0.0.1:9093
+          UNITTEST_LOG_DIR: "__unittest_logs"
+
  coverage:
-    if: github.event.pull_request.draft == false
+    if: github.event_name == 'merge_group'
    runs-on: ubuntu-20.04-8-cores
    timeout-minutes: 60
    steps:
@@ -658,48 +681,29 @@ jobs:
      - uses: arduino/setup-protoc@v3
        with:
          repo-token: ${{ secrets.GITHUB_TOKEN }}
-      - uses: KyleMayes/install-llvm-action@v1
-        with:
-          version: "14.0"
+      - uses: rui314/setup-mold@v1
      - name: Install toolchain
        uses: actions-rust-lang/setup-rust-toolchain@v1
        with:
-          components: llvm-tools-preview
+          components: llvm-tools
+          cache: false
      - name: Rust Cache
        uses: Swatinem/rust-cache@v2
        with:
          # Shares cross multiple jobs
          shared-key: "coverage-test"
-      - name: Docker Cache
-        uses: ScribeMD/docker-cache@0.3.7
-        with:
-          key: docker-${{ runner.os }}-coverage
+          save-if: ${{ github.ref == 'refs/heads/main' }}
      - name: Install latest nextest release
        uses: taiki-e/install-action@nextest
      - name: Install cargo-llvm-cov
        uses: taiki-e/install-action@cargo-llvm-cov
-      - name: Install Python
-        uses: actions/setup-python@v5
-        with:
-          python-version: '3.10'
-      - name: Install PyArrow Package
-        run: pip install pyarrow numpy
-      - name: Setup etcd server
-        working-directory: tests-integration/fixtures/etcd
-        run: docker compose -f docker-compose-standalone.yml up -d --wait
-      - name: Setup kafka server
-        working-directory: tests-integration/fixtures/kafka
-        run: docker compose -f docker-compose-standalone.yml up -d --wait
-      - name: Setup minio
-        working-directory: tests-integration/fixtures/minio
-        run: docker compose -f docker-compose-standalone.yml up -d --wait
-      - name: Setup postgres server
-        working-directory: tests-integration/fixtures/postgres
-        run: docker compose -f docker-compose-standalone.yml up -d --wait
+      - name: Setup external services
+        working-directory: tests-integration/fixtures
+        run: docker compose up -d --wait
      - name: Run nextest cases
-        run: cargo llvm-cov nextest --workspace --lcov --output-path lcov.info -F pyo3_backend -F dashboard
+        run: cargo llvm-cov nextest --workspace --lcov --output-path lcov.info -F dashboard -F pg_kvbackend
        env:
-          CARGO_BUILD_RUSTFLAGS: "-C link-arg=-fuse-ld=lld"
+          CARGO_BUILD_RUSTFLAGS: "-C link-arg=-fuse-ld=mold"
          RUST_BACKTRACE: 1
          CARGO_INCREMENTAL: 0
          GT_S3_BUCKET: ${{ vars.AWS_CI_TEST_BUCKET }}
--- a/.github/workflows/nightly-ci.yml
+++ b/.github/workflows/nightly-ci.yml
@@ -1,6 +1,6 @@
 on:
  schedule:
-    - cron: "0 23 * * 1-5"
+    - cron: "0 23 * * 1-4"
  workflow_dispatch:

 name: Nightly CI
@@ -91,18 +91,12 @@ jobs:
        uses: Swatinem/rust-cache@v2
      - name: Install Cargo Nextest
        uses: taiki-e/install-action@nextest
-      - name: Install Python
-        uses: actions/setup-python@v5
-        with:
-          python-version: "3.10"
-      - name: Install PyArrow Package
-        run: pip install pyarrow numpy
      - name: Install WSL distribution
        uses: Vampire/setup-wsl@v2
        with:
          distribution: Ubuntu-22.04
      - name: Running tests
-        run: cargo nextest run -F pyo3_backend,dashboard
+        run: cargo nextest run -F dashboard
        env:
          CARGO_BUILD_RUSTFLAGS: "-C linker=lld-link"
          RUST_BACKTRACE: 1
@@ -114,10 +108,55 @@ jobs:
          GT_S3_REGION: ${{ vars.AWS_CI_TEST_BUCKET_REGION }}
          UNITTEST_LOG_DIR: "__unittest_logs"

+  ## this is designed for generating cache that usable for pull requests
+  test-on-linux:
+    name: Run tests on Linux
+    if: ${{ github.repository == 'GreptimeTeam/greptimedb' }}
+    runs-on: ubuntu-20.04-8-cores
+    timeout-minutes: 60
+    steps:
+      - uses: actions/checkout@v4
+      - uses: arduino/setup-protoc@v3
+        with:
+          repo-token: ${{ secrets.GITHUB_TOKEN }}
+      - uses: rui314/setup-mold@v1
+      - name: Install Rust toolchain
+        uses: actions-rust-lang/setup-rust-toolchain@v1
+      - name: Rust Cache
+        uses: Swatinem/rust-cache@v2
+        with:
+          # Shares cross multiple jobs
+          shared-key: "coverage-test"
+      - name: Install Cargo Nextest
+        uses: taiki-e/install-action@nextest
+      - name: Setup external services
+        working-directory: tests-integration/fixtures
+        run: docker compose up -d --wait
+      - name: Running tests
+        run: cargo nextest run -F dashboard -F pg_kvbackend
+        env:
+          CARGO_BUILD_RUSTFLAGS: "-C link-arg=-fuse-ld=mold"
+          RUST_BACKTRACE: 1
+          CARGO_INCREMENTAL: 0
+          GT_S3_BUCKET: ${{ vars.AWS_CI_TEST_BUCKET }}
+          GT_S3_ACCESS_KEY_ID: ${{ secrets.AWS_CI_TEST_ACCESS_KEY_ID }}
+          GT_S3_ACCESS_KEY: ${{ secrets.AWS_CI_TEST_SECRET_ACCESS_KEY }}
+          GT_S3_REGION: ${{ vars.AWS_CI_TEST_BUCKET_REGION }}
+          GT_MINIO_BUCKET: greptime
+          GT_MINIO_ACCESS_KEY_ID: superpower_ci_user
+          GT_MINIO_ACCESS_KEY: superpower_password
+          GT_MINIO_REGION: us-west-2
+          GT_MINIO_ENDPOINT_URL: http://127.0.0.1:9000
+          GT_ETCD_ENDPOINTS: http://127.0.0.1:2379
+          GT_POSTGRES_ENDPOINTS: postgres://greptimedb:admin@127.0.0.1:5432/postgres
+          GT_KAFKA_ENDPOINTS: 127.0.0.1:9092
+          GT_KAFKA_SASL_ENDPOINTS: 127.0.0.1:9093
+          UNITTEST_LOG_DIR: "__unittest_logs"
+
  cleanbuild-linux-nix:
+    name: Run clean build on Linux
    runs-on: ubuntu-latest-8-cores
    timeout-minutes: 60
-    needs: [coverage, fmt, clippy, check]
    steps:
      - uses: actions/checkout@v4
      - uses: cachix/install-nix-action@v27
--- a/.github/workflows/release.yml
+++ b/.github/workflows/release.yml
@@ -222,18 +222,10 @@ jobs:
            arch: aarch64-apple-darwin
            features: servers/dashboard
            artifacts-dir-prefix: greptime-darwin-arm64
-          - os: ${{ needs.allocate-runners.outputs.macos-runner }}
-            arch: aarch64-apple-darwin
-            features: pyo3_backend,servers/dashboard
-            artifacts-dir-prefix: greptime-darwin-arm64-pyo3
          - os: ${{ needs.allocate-runners.outputs.macos-runner }}
            features: servers/dashboard
            arch: x86_64-apple-darwin
            artifacts-dir-prefix: greptime-darwin-amd64
-          - os: ${{ needs.allocate-runners.outputs.macos-runner }}
-            features: pyo3_backend,servers/dashboard
-            arch: x86_64-apple-darwin
-            artifacts-dir-prefix: greptime-darwin-amd64-pyo3
    runs-on: ${{ matrix.os }}
    outputs:
      build-macos-result: ${{ steps.set-build-macos-result.outputs.build-macos-result }}
@@ -271,10 +263,6 @@ jobs:
            arch: x86_64-pc-windows-msvc
            features: servers/dashboard
            artifacts-dir-prefix: greptime-windows-amd64
-          - os: ${{ needs.allocate-runners.outputs.windows-runner }}
-            arch: x86_64-pc-windows-msvc
-            features: pyo3_backend,servers/dashboard
-            artifacts-dir-prefix: greptime-windows-amd64-pyo3
    runs-on: ${{ matrix.os }}
    outputs:
      build-windows-result: ${{ steps.set-build-windows-result.outputs.build-windows-result }}
@@ -448,6 +436,22 @@ jobs:
          aws-region: ${{ vars.EC2_RUNNER_REGION }}
          github-token: ${{ secrets.GH_PERSONAL_ACCESS_TOKEN }}

+  bump-doc-version:
+    name: Bump doc version
+    if: ${{ github.event_name == 'push' || github.event_name == 'schedule' }}
+    needs: [allocate-runners]
+    runs-on: ubuntu-20.04
+    steps:
+      - uses: actions/checkout@v4
+      - uses: ./.github/actions/setup-cyborg
+      - name: Bump doc version
+        working-directory: cyborg
+        run: pnpm tsx bin/bump-doc-version.ts
+        env:
+          VERSION: ${{ needs.allocate-runners.outputs.version }}
+          GITHUB_TOKEN: ${{ secrets.GITHUB_TOKEN }}
+          DOCS_REPO_TOKEN: ${{ secrets.DOCS_REPO_TOKEN }}
+
  notification:
    if: ${{ github.repository == 'GreptimeTeam/greptimedb' && (github.event_name == 'push' || github.event_name == 'schedule') && always() }}
    name: Send notification to Greptime team
--- a/Cargo.lock
+++ b/Cargo.lock
@@ -188,7 +188,7 @@ checksum = "d301b3b94cb4b2f23d7917810addbbaff90738e0ca2be692bd027e70d7e0330c"

 [[package]]
 name = "api"
-version = "0.11.1"
+version = "0.11.3"
 dependencies = [
 "common-base",
 "common-decimal",
@@ -773,7 +773,7 @@ dependencies = [

 [[package]]
 name = "auth"
-version = "0.11.1"
+version = "0.11.3"
 dependencies = [
 "api",
 "async-trait",
@@ -896,18 +896,6 @@ dependencies = [
 "rand",
 ]

-[[package]]
-name = "backon"
-version = "0.4.4"
-source = "registry+https://github.com/rust-lang/crates.io-index"
-checksum = "d67782c3f868daa71d3533538e98a8e13713231969def7536e8039606fc46bf0"
-dependencies = [
- "fastrand",
- "futures-core",
- "pin-project",
- "tokio",
-]
-
 [[package]]
 name = "backon"
 version = "1.2.0"
@@ -1326,7 +1314,7 @@ dependencies = [

 [[package]]
 name = "cache"
-version = "0.11.1"
+version = "0.11.3"
 dependencies = [
 "catalog",
 "common-error",
@@ -1360,7 +1348,7 @@ checksum = "37b2a672a2cb129a2e41c10b1224bb368f9f37a2b16b612598138befd7b37eb5"

 [[package]]
 name = "catalog"
-version = "0.11.1"
+version = "0.11.3"
 dependencies = [
 "api",
 "arrow",
@@ -1696,7 +1684,7 @@ checksum = "1462739cb27611015575c0c11df5df7601141071f07518d56fcc1be504cbec97"

 [[package]]
 name = "cli"
-version = "0.11.1"
+version = "0.11.3"
 dependencies = [
 "async-trait",
 "auth",
@@ -1739,7 +1727,7 @@ dependencies = [
 "session",
 "snafu 0.8.5",
 "store-api",
- "substrait 0.11.1",
+ "substrait 0.11.3",
 "table",
 "tempfile",
 "tokio",
@@ -1748,7 +1736,7 @@ dependencies = [

 [[package]]
 name = "client"
-version = "0.11.1"
+version = "0.11.3"
 dependencies = [
 "api",
 "arc-swap",
@@ -1775,7 +1763,7 @@ dependencies = [
 "rand",
 "serde_json",
 "snafu 0.8.5",
- "substrait 0.11.1",
+ "substrait 0.11.3",
 "substrait 0.37.3",
 "tokio",
 "tokio-stream",
@@ -1816,7 +1804,7 @@ dependencies = [

 [[package]]
 name = "cmd"
-version = "0.11.1"
+version = "0.11.3"
 dependencies = [
 "async-trait",
 "auth",
@@ -1876,7 +1864,7 @@ dependencies = [
 "similar-asserts",
 "snafu 0.8.5",
 "store-api",
- "substrait 0.11.1",
+ "substrait 0.11.3",
 "table",
 "temp-env",
 "tempfile",
@@ -1928,7 +1916,7 @@ checksum = "55b672471b4e9f9e95499ea597ff64941a309b2cdbffcc46f2cc5e2d971fd335"

 [[package]]
 name = "common-base"
-version = "0.11.1"
+version = "0.11.3"
 dependencies = [
 "anymap2",
 "async-trait",
@@ -1950,11 +1938,11 @@ dependencies = [

 [[package]]
 name = "common-catalog"
-version = "0.11.1"
+version = "0.11.3"

 [[package]]
 name = "common-config"
-version = "0.11.1"
+version = "0.11.3"
 dependencies = [
 "common-base",
 "common-error",
@@ -1977,7 +1965,7 @@ dependencies = [

 [[package]]
 name = "common-datasource"
-version = "0.11.1"
+version = "0.11.3"
 dependencies = [
 "arrow",
 "arrow-schema",
@@ -2013,7 +2001,7 @@ dependencies = [

 [[package]]
 name = "common-decimal"
-version = "0.11.1"
+version = "0.11.3"
 dependencies = [
 "bigdecimal 0.4.5",
 "common-error",
@@ -2026,8 +2014,9 @@ dependencies = [

 [[package]]
 name = "common-error"
-version = "0.11.1"
+version = "0.11.3"
 dependencies = [
+ "http 0.2.12",
 "snafu 0.8.5",
 "strum 0.25.0",
 "tonic 0.11.0",
@@ -2035,7 +2024,7 @@ dependencies = [

 [[package]]
 name = "common-frontend"
-version = "0.11.1"
+version = "0.11.3"
 dependencies = [
 "async-trait",
 "common-error",
@@ -2045,7 +2034,7 @@ dependencies = [

 [[package]]
 name = "common-function"
-version = "0.11.1"
+version = "0.11.3"
 dependencies = [
 "api",
 "approx 0.5.1",
@@ -2089,7 +2078,7 @@ dependencies = [

 [[package]]
 name = "common-greptimedb-telemetry"
-version = "0.11.1"
+version = "0.11.3"
 dependencies = [
 "async-trait",
 "common-runtime",
@@ -2106,7 +2095,7 @@ dependencies = [

 [[package]]
 name = "common-grpc"
-version = "0.11.1"
+version = "0.11.3"
 dependencies = [
 "api",
 "arrow-flight",
@@ -2132,7 +2121,7 @@ dependencies = [

 [[package]]
 name = "common-grpc-expr"
-version = "0.11.1"
+version = "0.11.3"
 dependencies = [
 "api",
 "common-base",
@@ -2151,7 +2140,7 @@ dependencies = [

 [[package]]
 name = "common-macro"
-version = "0.11.1"
+version = "0.11.3"
 dependencies = [
 "arc-swap",
 "common-query",
@@ -2165,7 +2154,7 @@ dependencies = [

 [[package]]
 name = "common-mem-prof"
-version = "0.11.1"
+version = "0.11.3"
 dependencies = [
 "common-error",
 "common-macro",
@@ -2178,7 +2167,7 @@ dependencies = [

 [[package]]
 name = "common-meta"
-version = "0.11.1"
+version = "0.11.3"
 dependencies = [
 "anymap2",
 "api",
@@ -2235,7 +2224,7 @@ dependencies = [

 [[package]]
 name = "common-options"
-version = "0.11.1"
+version = "0.11.3"
 dependencies = [
 "common-grpc",
 "humantime-serde",
@@ -2244,11 +2233,11 @@ dependencies = [

 [[package]]
 name = "common-plugins"
-version = "0.11.1"
+version = "0.11.3"

 [[package]]
 name = "common-pprof"
-version = "0.11.1"
+version = "0.11.3"
 dependencies = [
 "common-error",
 "common-macro",
@@ -2260,11 +2249,11 @@ dependencies = [

 [[package]]
 name = "common-procedure"
-version = "0.11.1"
+version = "0.11.3"
 dependencies = [
 "async-stream",
 "async-trait",
- "backon 1.2.0",
+ "backon",
 "common-base",
 "common-error",
 "common-macro",
@@ -2287,7 +2276,7 @@ dependencies = [

 [[package]]
 name = "common-procedure-test"
-version = "0.11.1"
+version = "0.11.3"
 dependencies = [
 "async-trait",
 "common-procedure",
@@ -2295,7 +2284,7 @@ dependencies = [

 [[package]]
 name = "common-query"
-version = "0.11.1"
+version = "0.11.3"
 dependencies = [
 "api",
 "async-trait",
@@ -2321,7 +2310,7 @@ dependencies = [

 [[package]]
 name = "common-recordbatch"
-version = "0.11.1"
+version = "0.11.3"
 dependencies = [
 "arc-swap",
 "common-error",
@@ -2340,7 +2329,7 @@ dependencies = [

 [[package]]
 name = "common-runtime"
-version = "0.11.1"
+version = "0.11.3"
 dependencies = [
 "async-trait",
 "clap 4.5.19",
@@ -2362,13 +2351,15 @@ dependencies = [
 "snafu 0.8.5",
 "tempfile",
 "tokio",
+ "tokio-metrics",
+ "tokio-metrics-collector",
 "tokio-test",
 "tokio-util",
 ]

 [[package]]
 name = "common-telemetry"
-version = "0.11.1"
+version = "0.11.3"
 dependencies = [
 "atty",
 "backtrace",
@@ -2396,7 +2387,7 @@ dependencies = [

 [[package]]
 name = "common-test-util"
-version = "0.11.1"
+version = "0.11.3"
 dependencies = [
 "client",
 "common-query",
@@ -2408,7 +2399,7 @@ dependencies = [

 [[package]]
 name = "common-time"
-version = "0.11.1"
+version = "0.11.3"
 dependencies = [
 "arrow",
 "chrono",
@@ -2426,7 +2417,7 @@ dependencies = [

 [[package]]
 name = "common-version"
-version = "0.11.1"
+version = "0.11.3"
 dependencies = [
 "build-data",
 "const_format",
@@ -2436,7 +2427,7 @@ dependencies = [

 [[package]]
 name = "common-wal"
-version = "0.11.1"
+version = "0.11.3"
 dependencies = [
 "common-base",
 "common-error",
@@ -3235,7 +3226,7 @@ dependencies = [

 [[package]]
 name = "datanode"
-version = "0.11.1"
+version = "0.11.3"
 dependencies = [
 "api",
 "arrow-flight",
@@ -3286,7 +3277,7 @@ dependencies = [
 "session",
 "snafu 0.8.5",
 "store-api",
- "substrait 0.11.1",
+ "substrait 0.11.3",
 "table",
 "tokio",
 "toml 0.8.19",
@@ -3295,7 +3286,7 @@ dependencies = [

 [[package]]
 name = "datatypes"
-version = "0.11.1"
+version = "0.11.3"
 dependencies = [
 "arrow",
 "arrow-array",
@@ -3919,7 +3910,7 @@ dependencies = [

 [[package]]
 name = "file-engine"
-version = "0.11.1"
+version = "0.11.3"
 dependencies = [
 "api",
 "async-trait",
@@ -4035,7 +4026,7 @@ checksum = "8bf7cc16383c4b8d58b9905a8509f02926ce3058053c056376248d958c9df1e8"

 [[package]]
 name = "flow"
-version = "0.11.1"
+version = "0.11.3"
 dependencies = [
 "api",
 "arrow",
@@ -4073,6 +4064,7 @@ dependencies = [
 "get-size-derive2",
 "get-size2",
 "greptime-proto",
+ "http 0.2.12",
 "hydroflow",
 "itertools 0.10.5",
 "lazy_static",
@@ -4093,7 +4085,7 @@ dependencies = [
 "snafu 0.8.5",
 "store-api",
 "strum 0.25.0",
- "substrait 0.11.1",
+ "substrait 0.11.3",
 "table",
 "tokio",
 "tonic 0.11.0",
@@ -4131,7 +4123,7 @@ checksum = "6c2141d6d6c8512188a7891b4b01590a45f6dac67afb4f255c4124dbb86d4eaa"

 [[package]]
 name = "frontend"
-version = "0.11.1"
+version = "0.11.3"
 dependencies = [
 "api",
 "arc-swap",
@@ -4164,6 +4156,7 @@ dependencies = [
 "futures",
 "humantime-serde",
 "lazy_static",
+ "log-query",
 "log-store",
 "meta-client",
 "opentelemetry-proto 0.5.0",
@@ -4565,7 +4558,7 @@ dependencies = [
 [[package]]
 name = "greptime-proto"
 version = "0.1.0"
-source = "git+https://github.com/GreptimeTeam/greptime-proto.git?rev=a875e976441188028353f7274a46a7e6e065c5d4#a875e976441188028353f7274a46a7e6e065c5d4"
+source = "git+https://github.com/GreptimeTeam/greptime-proto.git?rev=43ddd8dea69f4df0fe2e8b5cdc0044d2cfa35908#43ddd8dea69f4df0fe2e8b5cdc0044d2cfa35908"
 dependencies = [
 "prost 0.12.6",
 "serde",
@@ -5280,7 +5273,7 @@ dependencies = [

 [[package]]
 name = "index"
-version = "0.11.1"
+version = "0.11.3"
 dependencies = [
 "async-trait",
 "asynchronous-codec",
@@ -5297,6 +5290,7 @@ dependencies = [
 "futures",
 "greptime-proto",
 "mockall",
+ "parquet",
 "pin-project",
 "prost 0.12.6",
 "rand",
@@ -6129,18 +6123,19 @@ checksum = "a7a70ba024b9dc04c27ea2f0c0548feb474ec5c54bba33a7f72f873a39d07b24"

 [[package]]
 name = "log-query"
-version = "0.11.1"
+version = "0.11.3"
 dependencies = [
 "chrono",
 "common-error",
 "common-macro",
+ "serde",
 "snafu 0.8.5",
 "table",
 ]

 [[package]]
 name = "log-store"
-version = "0.11.1"
+version = "0.11.3"
 dependencies = [
 "async-stream",
 "async-trait",
@@ -6484,7 +6479,7 @@ dependencies = [

 [[package]]
 name = "meta-client"
-version = "0.11.1"
+version = "0.11.3"
 dependencies = [
 "api",
 "async-trait",
@@ -6511,7 +6506,7 @@ dependencies = [

 [[package]]
 name = "meta-srv"
-version = "0.11.1"
+version = "0.11.3"
 dependencies = [
 "api",
 "async-trait",
@@ -6590,7 +6585,7 @@ dependencies = [

 [[package]]
 name = "metric-engine"
-version = "0.11.1"
+version = "0.11.3"
 dependencies = [
 "api",
 "aquamarine",
@@ -6684,7 +6679,7 @@ dependencies = [

 [[package]]
 name = "mito2"
-version = "0.11.1"
+version = "0.11.3"
 dependencies = [
 "api",
 "aquamarine",
@@ -7421,7 +7416,7 @@ dependencies = [

 [[package]]
 name = "object-store"
-version = "0.11.1"
+version = "0.11.3"
 dependencies = [
 "anyhow",
 "bytes",
@@ -7481,13 +7476,12 @@ checksum = "b410bbe7e14ab526a0e86877eb47c6996a2bd7746f027ba551028c925390e4e9"

 [[package]]
 name = "opendal"
-version = "0.49.2"
-source = "registry+https://github.com/rust-lang/crates.io-index"
-checksum = "9b04d09b9822c2f75a1d2fc513a2c1279c70e91e7407936fffdf6a6976ec530a"
+version = "0.50.2"
+source = "git+https://github.com/GreptimeTeam/opendal.git?rev=c82605177f2feec83e49dcaa537c505639d94024#c82605177f2feec83e49dcaa537c505639d94024"
 dependencies = [
 "anyhow",
 "async-trait",
- "backon 0.4.4",
+ "backon",
 "base64 0.22.1",
 "bytes",
 "chrono",
@@ -7500,6 +7494,7 @@ dependencies = [
 "md-5",
 "once_cell",
 "percent-encoding",
+ "prometheus",
 "quick-xml 0.36.2",
 "reqsign",
 "reqwest",
@@ -7674,7 +7669,7 @@ dependencies = [

 [[package]]
 name = "operator"
-version = "0.11.1"
+version = "0.11.3"
 dependencies = [
 "ahash 0.8.11",
 "api",
@@ -7722,7 +7717,7 @@ dependencies = [
 "sql",
 "sqlparser 0.45.0 (git+https://github.com/GreptimeTeam/sqlparser-rs.git?rev=54a267ac89c09b11c0c88934690530807185d3e7)",
 "store-api",
- "substrait 0.11.1",
+ "substrait 0.11.3",
 "table",
 "tokio",
 "tokio-util",
@@ -7972,7 +7967,7 @@ dependencies = [

 [[package]]
 name = "partition"
-version = "0.11.1"
+version = "0.11.3"
 dependencies = [
 "api",
 "async-trait",
@@ -8171,7 +8166,7 @@ dependencies = [
 "rand",
 "ring 0.17.8",
 "rust_decimal",
- "thiserror 2.0.4",
+ "thiserror 2.0.6",
 "tokio",
 "tokio-rustls 0.26.0",
 "tokio-util",
@@ -8258,7 +8253,7 @@ checksum = "8b870d8c151b6f2fb93e84a13146138f05d02ed11c7e7c54f8826aaaf7c9f184"

 [[package]]
 name = "pipeline"
-version = "0.11.1"
+version = "0.11.3"
 dependencies = [
 "ahash 0.8.11",
 "api",
@@ -8420,7 +8415,7 @@ dependencies = [

 [[package]]
 name = "plugins"
-version = "0.11.1"
+version = "0.11.3"
 dependencies = [
 "auth",
 "clap 4.5.19",
@@ -8708,7 +8703,7 @@ dependencies = [

 [[package]]
 name = "promql"
-version = "0.11.1"
+version = "0.11.3"
 dependencies = [
 "ahash 0.8.11",
 "async-trait",
@@ -8943,7 +8938,7 @@ dependencies = [

 [[package]]
 name = "puffin"
-version = "0.11.1"
+version = "0.11.3"
 dependencies = [
 "async-compression 0.4.13",
 "async-trait",
@@ -9068,7 +9063,7 @@ dependencies = [

 [[package]]
 name = "query"
-version = "0.11.1"
+version = "0.11.3"
 dependencies = [
 "ahash 0.8.11",
 "api",
@@ -9109,8 +9104,10 @@ dependencies = [
 "humantime",
 "itertools 0.10.5",
 "lazy_static",
+ "log-query",
 "meter-core",
 "meter-macros",
+ "nalgebra 0.33.2",
 "num",
 "num-traits",
 "object-store",
@@ -9131,7 +9128,7 @@ dependencies = [
 "sqlparser 0.45.0 (git+https://github.com/GreptimeTeam/sqlparser-rs.git?rev=54a267ac89c09b11c0c88934690530807185d3e7)",
 "statrs",
 "store-api",
- "substrait 0.11.1",
+ "substrait 0.11.3",
 "table",
 "tokio",
 "tokio-stream",
@@ -9515,9 +9512,9 @@ dependencies = [

 [[package]]
 name = "reqsign"
-version = "0.16.0"
+version = "0.16.1"
 source = "registry+https://github.com/rust-lang/crates.io-index"
-checksum = "03dd4ba7c3901dd43e6b8c7446a760d45bc1ea4301002e1a6fa48f97c3a796fa"
+checksum = "eb0075a66c8bfbf4cc8b70dca166e722e1f55a3ea9250ecbb85f4d92a5f64149"
 dependencies = [
 "anyhow",
 "async-trait",
@@ -10615,7 +10612,7 @@ checksum = "94143f37725109f92c262ed2cf5e59bce7498c01bcc1502d7b9afe439a4e9f49"

 [[package]]
 name = "script"
-version = "0.11.1"
+version = "0.11.3"
 dependencies = [
 "api",
 "arc-swap",
@@ -10907,7 +10904,7 @@ dependencies = [

 [[package]]
 name = "servers"
-version = "0.11.1"
+version = "0.11.3"
 dependencies = [
 "ahash 0.8.11",
 "api",
@@ -10963,6 +10960,7 @@ dependencies = [
 "json5",
 "jsonb",
 "lazy_static",
+ "log-query",
 "loki-api",
 "mime_guess",
 "mysql_async",
@@ -11018,7 +11016,7 @@ dependencies = [

 [[package]]
 name = "session"
-version = "0.11.1"
+version = "0.11.3"
 dependencies = [
 "api",
 "arc-swap",
@@ -11372,7 +11370,7 @@ dependencies = [

 [[package]]
 name = "sql"
-version = "0.11.1"
+version = "0.11.3"
 dependencies = [
 "api",
 "chrono",
@@ -11436,7 +11434,7 @@ dependencies = [

 [[package]]
 name = "sqlness-runner"
-version = "0.11.1"
+version = "0.11.3"
 dependencies = [
 "async-trait",
 "clap 4.5.19",
@@ -11654,7 +11652,7 @@ dependencies = [

 [[package]]
 name = "store-api"
-version = "0.11.1"
+version = "0.11.3"
 dependencies = [
 "api",
 "aquamarine",
@@ -11816,7 +11814,7 @@ dependencies = [

 [[package]]
 name = "substrait"
-version = "0.11.1"
+version = "0.11.3"
 dependencies = [
 "async-trait",
 "bytes",
@@ -12015,7 +12013,7 @@ dependencies = [

 [[package]]
 name = "table"
-version = "0.11.1"
+version = "0.11.3"
 dependencies = [
 "api",
 "async-trait",
@@ -12292,7 +12290,7 @@ checksum = "3369f5ac52d5eb6ab48c6b4ffdc8efbcad6b89c765749064ba298f2c68a16a76"

 [[package]]
 name = "tests-fuzz"
-version = "0.11.1"
+version = "0.11.3"
 dependencies = [
 "arbitrary",
 "async-trait",
@@ -12335,7 +12333,7 @@ dependencies = [

 [[package]]
 name = "tests-integration"
-version = "0.11.1"
+version = "0.11.3"
 dependencies = [
 "api",
 "arrow-flight",
@@ -12375,6 +12373,7 @@ dependencies = [
 "futures-util",
 "hex",
 "itertools 0.10.5",
+ "log-query",
 "loki-api",
 "meta-client",
 "meta-srv",
@@ -12399,7 +12398,7 @@ dependencies = [
 "sql",
 "sqlx",
 "store-api",
- "substrait 0.11.1",
+ "substrait 0.11.3",
 "table",
 "tempfile",
 "time",
@@ -12445,11 +12444,11 @@ dependencies = [

 [[package]]
 name = "thiserror"
-version = "2.0.4"
+version = "2.0.6"
 source = "registry+https://github.com/rust-lang/crates.io-index"
-checksum = "2f49a1853cf82743e3b7950f77e0f4d622ca36cf4317cba00c767838bac8d490"
+checksum = "8fec2a1820ebd077e2b90c4df007bebf344cd394098a13c563957d0afc83ea47"
 dependencies = [
- "thiserror-impl 2.0.4",
+ "thiserror-impl 2.0.6",
 ]

 [[package]]
@@ -12465,9 +12464,9 @@ dependencies = [

 [[package]]
 name = "thiserror-impl"
-version = "2.0.4"
+version = "2.0.6"
 source = "registry+https://github.com/rust-lang/crates.io-index"
-checksum = "8381894bb3efe0c4acac3ded651301ceee58a15d47c2e34885ed1908ad667061"
+checksum = "d65750cab40f4ff1929fb1ba509e9914eb756131cef4210da8d5d700d26f6312"
 dependencies = [
 "proc-macro2",
 "quote",
@@ -12623,9 +12622,9 @@ checksum = "1f3ccbac311fea05f86f61904b462b55fb3df8837a366dfc601a0161d0532f20"

 [[package]]
 name = "tokio"
-version = "1.40.0"
+version = "1.42.0"
 source = "registry+https://github.com/rust-lang/crates.io-index"
-checksum = "e2b070231665d27ad9ec9b8df639893f46727666c6767db40317fbe920a5d998"
+checksum = "5cec9b21b0450273377fc97bd4c33a8acffc8c996c987a7c5b319a0083707551"
 dependencies = [
 "backtrace",
 "bytes",
@@ -12661,6 +12660,31 @@ dependencies = [
 "syn 2.0.90",
 ]

+[[package]]
+name = "tokio-metrics"
+version = "0.3.1"
+source = "registry+https://github.com/rust-lang/crates.io-index"
+checksum = "eace09241d62c98b7eeb1107d4c5c64ca3bd7da92e8c218c153ab3a78f9be112"
+dependencies = [
+ "futures-util",
+ "pin-project-lite",
+ "tokio",
+ "tokio-stream",
+]
+
+[[package]]
+name = "tokio-metrics-collector"
+version = "0.2.3"
+source = "registry+https://github.com/rust-lang/crates.io-index"
+checksum = "a8092b7a97ed5dac2f44892db190eca8f476ede0fa585bc87664de4151cd0b64"
+dependencies = [
+ "lazy_static",
+ "parking_lot 0.12.3",
+ "prometheus",
+ "tokio",
+ "tokio-metrics",
+]
+
 [[package]]
 name = "tokio-postgres"
 version = "0.7.12"
--- a/Cargo.toml
+++ b/Cargo.toml
@@ -68,7 +68,7 @@ members = [
 resolver = "2"

 [workspace.package]
-version = "0.11.1"
+version = "0.11.3"
 edition = "2021"
 license = "Apache-2.0"

@@ -124,8 +124,9 @@ etcd-client = "0.13"
 fst = "0.4.7"
 futures = "0.3"
 futures-util = "0.3"
-greptime-proto = { git = "https://github.com/GreptimeTeam/greptime-proto.git", rev = "a875e976441188028353f7274a46a7e6e065c5d4" }
+greptime-proto = { git = "https://github.com/GreptimeTeam/greptime-proto.git", rev = "43ddd8dea69f4df0fe2e8b5cdc0044d2cfa35908" }
 hex = "0.4"
+http = "0.2"
 humantime = "2.1"
 humantime-serde = "1.1"
 itertools = "0.10"
@@ -134,6 +135,7 @@ lazy_static = "1.4"
 meter-core = { git = "https://github.com/GreptimeTeam/greptime-meter.git", rev = "a10facb353b41460eeb98578868ebf19c2084fac" }
 mockall = "0.11.4"
 moka = "0.12"
+nalgebra = "0.33"
 notify = "6.1"
 num_cpus = "1.16"
 once_cell = "1.18"
@@ -238,6 +240,7 @@ file-engine = { path = "src/file-engine" }
 flow = { path = "src/flow" }
 frontend = { path = "src/frontend", default-features = false }
 index = { path = "src/index" }
+log-query = { path = "src/log-query" }
 log-store = { path = "src/log-store" }
 meta-client = { path = "src/meta-client" }
 meta-srv = { path = "src/meta-srv" }
--- a/README.md
+++ b/README.md
@@ -70,23 +70,23 @@ Our core developers have been building time-series data platforms for years. Bas

 * **Unified Processing of Metrics, Logs, and Events**

-GreptimeDB unifies time series data processing by treating all data - whether metrics, logs, or events - as timestamped events with context. Users can analyze this data using either [SQL](https://docs.greptime.com/user-guide/query-data/sql) or [PromQL](https://docs.greptime.com/user-guide/query-data/promql) and leverage stream processing ([Flow](https://docs.greptime.com/user-guide/continuous-aggregation/overview)) to enable continuous aggregation. [Read more](https://docs.greptime.com/user-guide/concepts/data-model).
+  GreptimeDB unifies time series data processing by treating all data - whether metrics, logs, or events - as timestamped events with context. Users can analyze this data using either [SQL](https://docs.greptime.com/user-guide/query-data/sql) or [PromQL](https://docs.greptime.com/user-guide/query-data/promql) and leverage stream processing ([Flow](https://docs.greptime.com/user-guide/flow-computation/overview)) to enable continuous aggregation. [Read more](https://docs.greptime.com/user-guide/concepts/data-model).

 * **Cloud-native Distributed Database**

-Built for [Kubernetes](https://docs.greptime.com/user-guide/deployments/deploy-on-kubernetes/greptimedb-operator-management). GreptimeDB achieves seamless scalability with its [cloud-native architecture](https://docs.greptime.com/user-guide/concepts/architecture) of separated compute and storage, built on object storage (AWS S3, Azure Blob Storage, etc.) while enabling cross-cloud deployment through a unified data access layer.
+  Built for [Kubernetes](https://docs.greptime.com/user-guide/deployments/deploy-on-kubernetes/greptimedb-operator-management). GreptimeDB achieves seamless scalability with its [cloud-native architecture](https://docs.greptime.com/user-guide/concepts/architecture) of separated compute and storage, built on object storage (AWS S3, Azure Blob Storage, etc.) while enabling cross-cloud deployment through a unified data access layer.

 * **Performance and Cost-effective**

-Written in pure Rust for superior performance and reliability. GreptimeDB features a distributed query engine with intelligent indexing to handle high cardinality data efficiently. Its optimized columnar storage achieves 50x cost efficiency on cloud object storage through advanced compression. [Benchmark reports](https://www.greptime.com/blogs/2024-09-09-report-summary).
+  Written in pure Rust for superior performance and reliability. GreptimeDB features a distributed query engine with intelligent indexing to handle high cardinality data efficiently. Its optimized columnar storage achieves 50x cost efficiency on cloud object storage through advanced compression. [Benchmark reports](https://www.greptime.com/blogs/2024-09-09-report-summary).

 * **Cloud-Edge Collaboration**

-GreptimeDB seamlessly operates across cloud and edge (ARM/Android/Linux), providing consistent APIs and control plane for unified data management and efficient synchronization. [Learn how to run on Android](https://docs.greptime.com/user-guide/deployments/run-on-android/).
+  GreptimeDB seamlessly operates across cloud and edge (ARM/Android/Linux), providing consistent APIs and control plane for unified data management and efficient synchronization. [Learn how to run on Android](https://docs.greptime.com/user-guide/deployments/run-on-android/).

 * **Multi-protocol Ingestion, SQL & PromQL Ready**

-Widely adopted database protocols and APIs, including MySQL, PostgreSQL, InfluxDB, OpenTelemetry, Loki and Prometheus, etc.  Effortless Adoption & Seamless Migration. [Supported Protocols Overview](https://docs.greptime.com/user-guide/protocols/overview).
+  Widely adopted database protocols and APIs, including MySQL, PostgreSQL, InfluxDB, OpenTelemetry, Loki and Prometheus, etc.  Effortless Adoption & Seamless Migration. [Supported Protocols Overview](https://docs.greptime.com/user-guide/protocols/overview).

 For more detailed info please read  [Why GreptimeDB](https://docs.greptime.com/user-guide/concepts/why-greptimedb).

@@ -138,7 +138,7 @@ Check the prerequisite:

 * [Rust toolchain](https://www.rust-lang.org/tools/install) (nightly)
 * [Protobuf compiler](https://grpc.io/docs/protoc-installation/) (>= 3.15)
-* Python toolchain (optional): Required only if built with PyO3 backend. More detail for compiling with PyO3 can be found in its [documentation](https://pyo3.rs/v0.18.1/building_and_distribution#configuring-the-python-version).
+* Python toolchain (optional): Required only if built with PyO3 backend. More details for compiling with PyO3 can be found in its [documentation](https://pyo3.rs/v0.18.1/building_and_distribution#configuring-the-python-version).

 Build GreptimeDB binary:

@@ -154,6 +154,10 @@ cargo run -- standalone start

 ## Tools & Extensions

+### Kubernetes
+
+- [GreptimeDB Operator](https://github.com/GrepTimeTeam/greptimedb-operator)
+
 ### Dashboard

 - [The dashboard UI for GreptimeDB](https://github.com/GreptimeTeam/dashboard)
@@ -173,7 +177,7 @@ Our official Grafana dashboard for monitoring GreptimeDB is available at [grafan

 ## Project Status

-GreptimeDB is currently in Beta. We are targeting GA (General Availability) with v1.0 release by Early 2025. 
+GreptimeDB is currently in Beta. We are targeting GA (General Availability) with v1.0 release by Early 2025.

 While in Beta, GreptimeDB is already:

--- a/config/config.md
+++ b/config/config.md
@@ -18,6 +18,7 @@
 | `init_regions_parallelism` | Integer | `16` | Parallelism of initializing regions. |
 | `max_concurrent_queries` | Integer | `0` | The maximum current queries allowed to be executed. Zero means unlimited. |
 | `enable_telemetry` | Bool | `true` | Enable telemetry to collect anonymous usage data. Enabled by default. |
+| `max_in_flight_write_bytes` | String | Unset | The maximum in-flight write bytes. |
 | `runtime` | -- | -- | The runtime options. |
 | `runtime.global_rt_size` | Integer | `8` | The number of threads to execute the runtime for global read operations. |
 | `runtime.compact_rt_size` | Integer | `4` | The number of threads to execute the runtime for global write operations. |
@@ -150,12 +151,17 @@
 | `region_engine.mito.inverted_index.intermediate_path` | String | `""` | Deprecated, use `region_engine.mito.index.aux_path` instead. |
 | `region_engine.mito.inverted_index.metadata_cache_size` | String | `64MiB` | Cache size for inverted index metadata. |
 | `region_engine.mito.inverted_index.content_cache_size` | String | `128MiB` | Cache size for inverted index content. |
-| `region_engine.mito.inverted_index.content_cache_page_size` | String | `8MiB` | Page size for inverted index content cache. |
+| `region_engine.mito.inverted_index.content_cache_page_size` | String | `64KiB` | Page size for inverted index content cache. |
 | `region_engine.mito.fulltext_index` | -- | -- | The options for full-text index in Mito engine. |
 | `region_engine.mito.fulltext_index.create_on_flush` | String | `auto` | Whether to create the index on flush.<br/>- `auto`: automatically (default)<br/>- `disable`: never |
 | `region_engine.mito.fulltext_index.create_on_compaction` | String | `auto` | Whether to create the index on compaction.<br/>- `auto`: automatically (default)<br/>- `disable`: never |
 | `region_engine.mito.fulltext_index.apply_on_query` | String | `auto` | Whether to apply the index on query<br/>- `auto`: automatically (default)<br/>- `disable`: never |
 | `region_engine.mito.fulltext_index.mem_threshold_on_create` | String | `auto` | Memory threshold for index creation.<br/>- `auto`: automatically determine the threshold based on the system memory size (default)<br/>- `unlimited`: no memory limit<br/>- `[size]` e.g. `64MB`: fixed memory threshold |
+| `region_engine.mito.bloom_filter_index` | -- | -- | The options for bloom filter in Mito engine. |
+| `region_engine.mito.bloom_filter_index.create_on_flush` | String | `auto` | Whether to create the bloom filter on flush.<br/>- `auto`: automatically (default)<br/>- `disable`: never |
+| `region_engine.mito.bloom_filter_index.create_on_compaction` | String | `auto` | Whether to create the bloom filter on compaction.<br/>- `auto`: automatically (default)<br/>- `disable`: never |
+| `region_engine.mito.bloom_filter_index.apply_on_query` | String | `auto` | Whether to apply the bloom filter on query<br/>- `auto`: automatically (default)<br/>- `disable`: never |
+| `region_engine.mito.bloom_filter_index.mem_threshold_on_create` | String | `auto` | Memory threshold for bloom filter creation.<br/>- `auto`: automatically determine the threshold based on the system memory size (default)<br/>- `unlimited`: no memory limit<br/>- `[size]` e.g. `64MB`: fixed memory threshold |
 | `region_engine.mito.memtable` | -- | -- | -- |
 | `region_engine.mito.memtable.type` | String | `time_series` | Memtable type.<br/>- `time_series`: time-series memtable<br/>- `partition_tree`: partition tree memtable (experimental) |
 | `region_engine.mito.memtable.index_max_keys_per_shard` | Integer | `8192` | The max number of keys in one shard.<br/>Only available for `partition_tree` memtable. |
@@ -195,6 +201,7 @@
 | Key | Type | Default | Descriptions |
 | --- | -----| ------- | ----------- |
 | `default_timezone` | String | Unset | The default timezone of the server. |
+| `max_in_flight_write_bytes` | String | Unset | The maximum in-flight write bytes. |
 | `runtime` | -- | -- | The runtime options. |
 | `runtime.global_rt_size` | Integer | `8` | The number of threads to execute the runtime for global read operations. |
 | `runtime.compact_rt_size` | Integer | `4` | The number of threads to execute the runtime for global write operations. |
@@ -421,7 +428,7 @@
 | `storage` | -- | -- | The data storage options. |
 | `storage.data_home` | String | `/tmp/greptimedb/` | The working home directory. |
 | `storage.type` | String | `File` | The storage type used to store the data.<br/>- `File`: the data is stored in the local file system.<br/>- `S3`: the data is stored in the S3 object storage.<br/>- `Gcs`: the data is stored in the Google Cloud Storage.<br/>- `Azblob`: the data is stored in the Azure Blob Storage.<br/>- `Oss`: the data is stored in the Aliyun OSS. |
-| `storage.cache_path` | String | Unset | Read cache configuration for object storage such as 'S3' etc, it's configured by default when using object storage. It is recommended to configure it when using object storage for better performance.<br/>A local file directory, defaults to `{data_home}/object_cache/read`. An empty string means disabling. |
+| `storage.cache_path` | String | Unset | Read cache configuration for object storage such as 'S3' etc, it's configured by default when using object storage. It is recommended to configure it when using object storage for better performance.<br/>A local file directory, defaults to `{data_home}`. An empty string means disabling. |
 | `storage.cache_capacity` | String | Unset | The local file cache capacity in bytes. If your disk space is sufficient, it is recommended to set it larger. |
 | `storage.bucket` | String | Unset | The S3 bucket name.<br/>**It's only used when the storage type is `S3`, `Oss` and `Gcs`**. |
 | `storage.root` | String | Unset | The S3 data will be stored in the specified prefix, for example, `s3://${bucket}/${root}`.<br/>**It's only used when the storage type is `S3`, `Oss` and `Azblob`**. |
@@ -460,7 +467,7 @@
 | `region_engine.mito.page_cache_size` | String | Auto | Cache size for pages of SST row groups. Setting it to 0 to disable the cache.<br/>If not set, it's default to 1/8 of OS memory. |
 | `region_engine.mito.selector_result_cache_size` | String | Auto | Cache size for time series selector (e.g. `last_value()`). Setting it to 0 to disable the cache.<br/>If not set, it's default to 1/16 of OS memory with a max limitation of 512MB. |
 | `region_engine.mito.enable_experimental_write_cache` | Bool | `false` | Whether to enable the experimental write cache, it's enabled by default when using object storage. It is recommended to enable it when using object storage for better performance. |
-| `region_engine.mito.experimental_write_cache_path` | String | `""` | File system path for write cache, defaults to `{data_home}/object_cache/write`. |
+| `region_engine.mito.experimental_write_cache_path` | String | `""` | File system path for write cache, defaults to `{data_home}`. |
 | `region_engine.mito.experimental_write_cache_size` | String | `5GiB` | Capacity for write cache. If your disk space is sufficient, it is recommended to set it larger. |
 | `region_engine.mito.experimental_write_cache_ttl` | String | Unset | TTL for write cache. |
 | `region_engine.mito.sst_write_buffer_size` | String | `8MB` | Buffer size for SST writing. |
@@ -478,12 +485,17 @@
 | `region_engine.mito.inverted_index.intermediate_path` | String | `""` | Deprecated, use `region_engine.mito.index.aux_path` instead. |
 | `region_engine.mito.inverted_index.metadata_cache_size` | String | `64MiB` | Cache size for inverted index metadata. |
 | `region_engine.mito.inverted_index.content_cache_size` | String | `128MiB` | Cache size for inverted index content. |
-| `region_engine.mito.inverted_index.content_cache_page_size` | String | `8MiB` | Page size for inverted index content cache. |
+| `region_engine.mito.inverted_index.content_cache_page_size` | String | `64KiB` | Page size for inverted index content cache. |
 | `region_engine.mito.fulltext_index` | -- | -- | The options for full-text index in Mito engine. |
 | `region_engine.mito.fulltext_index.create_on_flush` | String | `auto` | Whether to create the index on flush.<br/>- `auto`: automatically (default)<br/>- `disable`: never |
 | `region_engine.mito.fulltext_index.create_on_compaction` | String | `auto` | Whether to create the index on compaction.<br/>- `auto`: automatically (default)<br/>- `disable`: never |
 | `region_engine.mito.fulltext_index.apply_on_query` | String | `auto` | Whether to apply the index on query<br/>- `auto`: automatically (default)<br/>- `disable`: never |
 | `region_engine.mito.fulltext_index.mem_threshold_on_create` | String | `auto` | Memory threshold for index creation.<br/>- `auto`: automatically determine the threshold based on the system memory size (default)<br/>- `unlimited`: no memory limit<br/>- `[size]` e.g. `64MB`: fixed memory threshold |
+| `region_engine.mito.bloom_filter_index` | -- | -- | The options for bloom filter index in Mito engine. |
+| `region_engine.mito.bloom_filter_index.create_on_flush` | String | `auto` | Whether to create the index on flush.<br/>- `auto`: automatically (default)<br/>- `disable`: never |
+| `region_engine.mito.bloom_filter_index.create_on_compaction` | String | `auto` | Whether to create the index on compaction.<br/>- `auto`: automatically (default)<br/>- `disable`: never |
+| `region_engine.mito.bloom_filter_index.apply_on_query` | String | `auto` | Whether to apply the index on query<br/>- `auto`: automatically (default)<br/>- `disable`: never |
+| `region_engine.mito.bloom_filter_index.mem_threshold_on_create` | String | `auto` | Memory threshold for the index creation.<br/>- `auto`: automatically determine the threshold based on the system memory size (default)<br/>- `unlimited`: no memory limit<br/>- `[size]` e.g. `64MB`: fixed memory threshold |
 | `region_engine.mito.memtable` | -- | -- | -- |
 | `region_engine.mito.memtable.type` | String | `time_series` | Memtable type.<br/>- `time_series`: time-series memtable<br/>- `partition_tree`: partition tree memtable (experimental) |
 | `region_engine.mito.memtable.index_max_keys_per_shard` | Integer | `8192` | The max number of keys in one shard.<br/>Only available for `partition_tree` memtable. |
--- a/config/datanode.example.toml
+++ b/config/datanode.example.toml
@@ -294,7 +294,7 @@ data_home = "/tmp/greptimedb/"
 type = "File"

 ## Read cache configuration for object storage such as 'S3' etc, it's configured by default when using object storage. It is recommended to configure it when using object storage for better performance.
-## A local file directory, defaults to `{data_home}/object_cache/read`. An empty string means disabling.
+## A local file directory, defaults to `{data_home}`. An empty string means disabling.
 ## @toml2docs:none-default
 #+ cache_path = ""

@@ -478,7 +478,7 @@ auto_flush_interval = "1h"
 ## Whether to enable the experimental write cache, it's enabled by default when using object storage. It is recommended to enable it when using object storage for better performance.
 enable_experimental_write_cache = false

-## File system path for write cache, defaults to `{data_home}/object_cache/write`.
+## File system path for write cache, defaults to `{data_home}`.
 experimental_write_cache_path = ""

 ## Capacity for write cache. If your disk space is sufficient, it is recommended to set it larger.
@@ -550,7 +550,7 @@ metadata_cache_size = "64MiB"
 content_cache_size = "128MiB"

 ## Page size for inverted index content cache.
-content_cache_page_size = "8MiB"
+content_cache_page_size = "64KiB"

 ## The options for full-text index in Mito engine.
 [region_engine.mito.fulltext_index]
@@ -576,6 +576,30 @@ apply_on_query = "auto"
 ## - `[size]` e.g. `64MB`: fixed memory threshold
 mem_threshold_on_create = "auto"

+## The options for bloom filter index in Mito engine.
+[region_engine.mito.bloom_filter_index]
+
+## Whether to create the index on flush.
+## - `auto`: automatically (default)
+## - `disable`: never
+create_on_flush = "auto"
+
+## Whether to create the index on compaction.
+## - `auto`: automatically (default)
+## - `disable`: never
+create_on_compaction = "auto"
+
+## Whether to apply the index on query
+## - `auto`: automatically (default)
+## - `disable`: never
+apply_on_query = "auto"
+
+## Memory threshold for the index creation.
+## - `auto`: automatically determine the threshold based on the system memory size (default)
+## - `unlimited`: no memory limit
+## - `[size]` e.g. `64MB`: fixed memory threshold
+mem_threshold_on_create = "auto"
+
 [region_engine.mito.memtable]
 ## Memtable type.
 ## - `time_series`: time-series memtable
--- a/config/frontend.example.toml
+++ b/config/frontend.example.toml
@@ -2,6 +2,10 @@
 ## @toml2docs:none-default
 default_timezone = "UTC"

+## The maximum in-flight write bytes.
+## @toml2docs:none-default
+#+ max_in_flight_write_bytes = "500MB"
+
 ## The runtime options.
 #+ [runtime]
 ## The number of threads to execute the runtime for global read operations.
--- a/config/standalone.example.toml
+++ b/config/standalone.example.toml
@@ -18,6 +18,10 @@ max_concurrent_queries = 0
 ## Enable telemetry to collect anonymous usage data. Enabled by default.
 #+ enable_telemetry = true

+## The maximum in-flight write bytes.
+## @toml2docs:none-default
+#+ max_in_flight_write_bytes = "500MB"
+
 ## The runtime options.
 #+ [runtime]
 ## The number of threads to execute the runtime for global read operations.
@@ -589,7 +593,7 @@ metadata_cache_size = "64MiB"
 content_cache_size = "128MiB"

 ## Page size for inverted index content cache.
-content_cache_page_size = "8MiB"
+content_cache_page_size = "64KiB"

 ## The options for full-text index in Mito engine.
 [region_engine.mito.fulltext_index]
@@ -615,6 +619,30 @@ apply_on_query = "auto"
 ## - `[size]` e.g. `64MB`: fixed memory threshold
 mem_threshold_on_create = "auto"

+## The options for bloom filter in Mito engine.
+[region_engine.mito.bloom_filter_index]
+
+## Whether to create the bloom filter on flush.
+## - `auto`: automatically (default)
+## - `disable`: never
+create_on_flush = "auto"
+
+## Whether to create the bloom filter on compaction.
+## - `auto`: automatically (default)
+## - `disable`: never
+create_on_compaction = "auto"
+
+## Whether to apply the bloom filter on query
+## - `auto`: automatically (default)
+## - `disable`: never
+apply_on_query = "auto"
+
+## Memory threshold for bloom filter creation.
+## - `auto`: automatically determine the threshold based on the system memory size (default)
+## - `unlimited`: no memory limit
+## - `[size]` e.g. `64MB`: fixed memory threshold
+mem_threshold_on_create = "auto"
+
 [region_engine.mito.memtable]
 ## Memtable type.
 ## - `time_series`: time-series memtable
--- a/cyborg/bin/bump-doc-version.ts
+++ b/cyborg/bin/bump-doc-version.ts
@@ -0,0 +1,75 @@
+/*
+ * Copyright 2023 Greptime Team
+ *
+ * Licensed under the Apache License, Version 2.0 (the "License");
+ * you may not use this file except in compliance with the License.
+ * You may obtain a copy of the License at
+ *
+ *     http://www.apache.org/licenses/LICENSE-2.0
+ *
+ * Unless required by applicable law or agreed to in writing, software
+ * distributed under the License is distributed on an "AS IS" BASIS,
+ * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+ * See the License for the specific language governing permissions and
+ * limitations under the License.
+ */
+
+import * as core from "@actions/core";
+import {obtainClient} from "@/common";
+
+async function triggerWorkflow(workflowId: string, version: string) {
+  const docsClient = obtainClient("DOCS_REPO_TOKEN")
+  try {
+    await docsClient.rest.actions.createWorkflowDispatch({
+      owner: "GreptimeTeam",
+      repo: "docs",
+      workflow_id: workflowId,
+      ref: "main",
+      inputs: {
+        version,
+      },
+    });
+    console.log(`Successfully triggered ${workflowId} workflow with version ${version}`);
+  } catch (error) {
+    core.setFailed(`Failed to trigger workflow: ${error.message}`);
+  }
+}
+
+function determineWorkflow(version: string): [string, string] {
+  // Check if it's a nightly version
+  if (version.includes('nightly')) {
+    return ['bump-nightly-version.yml', version];
+  }
+
+  const parts = version.split('.');
+
+  if (parts.length !== 3) {
+    throw new Error('Invalid version format');
+  }
+
+  // If patch version (last number) is 0, it's a major version
+  // Return only major.minor version
+  if (parts[2] === '0') {
+    return ['bump-version.yml', `${parts[0]}.${parts[1]}`];
+  }
+
+  // Otherwise it's a patch version, use full version
+  return ['bump-patch-version.yml', version];
+}
+
+const version = process.env.VERSION;
+if (!version) {
+  core.setFailed("VERSION environment variable is required");
+  process.exit(1);
+}
+
+// Remove 'v' prefix if exists
+const cleanVersion = version.startsWith('v') ? version.slice(1) : version;
+
+try {
+  const [workflowId, apiVersion] = determineWorkflow(cleanVersion);
+  triggerWorkflow(workflowId, apiVersion);
+} catch (error) {
+  core.setFailed(`Error processing version: ${error.message}`);
+  process.exit(1);
+}
--- a/docker/buildx/centos/Dockerfile
+++ b/docker/buildx/centos/Dockerfile
@@ -13,8 +13,6 @@ RUN yum install -y epel-release  \
    openssl \
    openssl-devel  \
    centos-release-scl  \
-    rh-python38  \
-    rh-python38-python-devel \
    which

 # Install protoc
@@ -43,8 +41,6 @@ RUN yum install -y epel-release \
    openssl \
    openssl-devel  \
    centos-release-scl  \
-    rh-python38  \
-    rh-python38-python-devel \
    which

 WORKDIR /greptime
--- a/docker/buildx/ubuntu/Dockerfile
+++ b/docker/buildx/ubuntu/Dockerfile
@@ -20,10 +20,7 @@ RUN --mount=type=cache,target=/var/cache/apt \
    curl \
    git \
    build-essential \
-    pkg-config \
-    python3.10 \
-    python3.10-dev \
-    python3-pip
+    pkg-config

 # Install Rust.
 SHELL ["/bin/bash", "-c"]
@@ -46,15 +43,8 @@ ARG OUTPUT_DIR

 RUN apt-get update && DEBIAN_FRONTEND=noninteractive apt-get \
    -y install ca-certificates \
-    python3.10 \
-    python3.10-dev \
-    python3-pip \
    curl

-COPY ./docker/python/requirements.txt /etc/greptime/requirements.txt
-
-RUN python3 -m pip install -r /etc/greptime/requirements.txt
-
 WORKDIR /greptime
 COPY --from=builder /out/target/${OUTPUT_DIR}/greptime /greptime/bin/
 ENV PATH /greptime/bin/:$PATH
--- a/docker/ci/centos/Dockerfile
+++ b/docker/ci/centos/Dockerfile
@@ -7,9 +7,7 @@ RUN sed -i s/^#.*baseurl=http/baseurl=http/g /etc/yum.repos.d/*.repo
 RUN yum install -y epel-release \
    openssl \
    openssl-devel  \
-    centos-release-scl  \
-    rh-python38  \
-    rh-python38-python-devel
+    centos-release-scl

 ARG TARGETARCH

--- a/docker/ci/ubuntu/Dockerfile
+++ b/docker/ci/ubuntu/Dockerfile
@@ -8,15 +8,8 @@ ARG TARGET_BIN=greptime

 RUN apt-get update && DEBIAN_FRONTEND=noninteractive apt-get install -y \
    ca-certificates \
-    python3.10 \
-    python3.10-dev \
-    python3-pip \
    curl

-COPY $DOCKER_BUILD_ROOT/docker/python/requirements.txt /etc/greptime/requirements.txt
-
-RUN python3 -m pip install -r /etc/greptime/requirements.txt
-
 ARG TARGETARCH

 ADD $TARGETARCH/$TARGET_BIN /greptime/bin/
--- a/rust-toolchain.toml
+++ b/rust-toolchain.toml
@@ -1,3 +1,3 @@
 [toolchain]
 channel = "nightly-2024-10-19"
-components = ["rust-analyzer"]
+components = ["rust-analyzer", "llvm-tools"]
--- a/scripts/check-snafu.py
+++ b/scripts/check-snafu.py
@@ -14,6 +14,7 @@

 import os
 import re
+from multiprocessing import Pool


 def find_rust_files(directory):
@@ -33,13 +34,11 @@ def extract_branch_names(file_content):
    return pattern.findall(file_content)


-def check_snafu_in_files(branch_name, rust_files):
+def check_snafu_in_files(branch_name, rust_files_content):
    branch_name_snafu = f"{branch_name}Snafu"
-    for rust_file in rust_files:
-        with open(rust_file, "r") as file:
-            content = file.read()
-            if branch_name_snafu in content:
-                return True
+    for content in rust_files_content.values():
+        if branch_name_snafu in content:
+            return True
    return False


@@ -49,21 +48,24 @@ def main():

    for error_file in error_files:
        with open(error_file, "r") as file:
-            content = file.read()
-            branch_names.extend(extract_branch_names(content))
+            branch_names.extend(extract_branch_names(file.read()))

-    unused_snafu = [
-        branch_name
-        for branch_name in branch_names
-        if not check_snafu_in_files(branch_name, other_rust_files)
-    ]
+    # Read all rust files into memory once
+    rust_files_content = {}
+    for rust_file in other_rust_files:
+        with open(rust_file, "r") as file:
+            rust_files_content[rust_file] = file.read()
+
+    with Pool() as pool:
+        results = pool.starmap(
+            check_snafu_in_files, [(bn, rust_files_content) for bn in branch_names]
+        )
+    unused_snafu = [bn for bn, found in zip(branch_names, results) if not found]

    if unused_snafu:
        print("Unused error variants:")
        for name in unused_snafu:
            print(name)
-
-    if unused_snafu:
        raise SystemExit(1)


--- a/shell.nix
+++ b/shell.nix
@@ -1,5 +1,5 @@
 let
-  nixpkgs = fetchTarball "https://github.com/NixOS/nixpkgs/tarball/nixos-unstable";
+  nixpkgs = fetchTarball "https://github.com/NixOS/nixpkgs/tarball/nixos-24.11";
  fenix = import (fetchTarball "https://github.com/nix-community/fenix/archive/main.tar.gz") {};
  pkgs = import nixpkgs { config = {}; overlays = []; };
 in
@@ -11,16 +11,20 @@ pkgs.mkShell rec {
    clang
    gcc
    protobuf
+    gnumake
    mold
    (fenix.fromToolchainFile {
      dir = ./.;
    })
    cargo-nextest
+    cargo-llvm-cov
    taplo
+    curl
  ];

  buildInputs = with pkgs; [
    libgit2
+    libz
  ];

  LD_LIBRARY_PATH = pkgs.lib.makeLibraryPath buildInputs;
--- a/src/auth/src/permission.rs
+++ b/src/auth/src/permission.rs
@@ -25,6 +25,7 @@ pub enum PermissionReq<'a> {
    GrpcRequest(&'a Request),
    SqlStatement(&'a Statement),
    PromQuery,
+    LogQuery,
    Opentsdb,
    LineProtocol,
    PromStoreWrite,
--- a/src/catalog/src/kvbackend/table_cache.rs
+++ b/src/catalog/src/kvbackend/table_cache.rs
@@ -38,7 +38,7 @@ pub fn new_table_cache(
 ) -> TableCache {
    let init = init_factory(table_info_cache, table_name_cache);

-    CacheContainer::new(name, cache, Box::new(invalidator), init, Box::new(filter))
+    CacheContainer::new(name, cache, Box::new(invalidator), init, filter)
 }

 fn init_factory(
--- a/src/cli/Cargo.toml
+++ b/src/cli/Cargo.toml
@@ -15,7 +15,7 @@ cache.workspace = true
 catalog.workspace = true
 chrono.workspace = true
 clap.workspace = true
-client.workspace = true
+client = { workspace = true, features = ["testing"] }
 common-base.workspace = true
 common-catalog.workspace = true
 common-config.workspace = true
@@ -56,7 +56,6 @@ tokio.workspace = true
 tracing-appender.workspace = true

 [dev-dependencies]
-client = { workspace = true, features = ["testing"] }
 common-test-util.workspace = true
 common-version.workspace = true
 serde.workspace = true
--- a/src/cmd/src/standalone.rs
+++ b/src/cmd/src/standalone.rs
@@ -22,6 +22,7 @@ use catalog::information_schema::InformationExtension;
 use catalog::kvbackend::KvBackendCatalogManager;
 use clap::Parser;
 use client::api::v1::meta::RegionRole;
+use common_base::readable_size::ReadableSize;
 use common_base::Plugins;
 use common_catalog::consts::{MIN_USER_FLOW_ID, MIN_USER_TABLE_ID};
 use common_config::{metadata_store_dir, Configurable, KvBackendConfig};
@@ -152,6 +153,7 @@ pub struct StandaloneOptions {
    pub tracing: TracingOptions,
    pub init_regions_in_background: bool,
    pub init_regions_parallelism: usize,
+    pub max_in_flight_write_bytes: Option<ReadableSize>,
 }

 impl Default for StandaloneOptions {
@@ -181,6 +183,7 @@ impl Default for StandaloneOptions {
            tracing: TracingOptions::default(),
            init_regions_in_background: false,
            init_regions_parallelism: 16,
+            max_in_flight_write_bytes: None,
        }
    }
 }
@@ -218,6 +221,7 @@ impl StandaloneOptions {
            user_provider: cloned_opts.user_provider,
            // Handle the export metrics task run by standalone to frontend for execution
            export_metrics: cloned_opts.export_metrics,
+            max_in_flight_write_bytes: cloned_opts.max_in_flight_write_bytes,
            ..Default::default()
        }
    }
--- a/src/common/datasource/src/object_store/fs.rs
+++ b/src/common/datasource/src/object_store/fs.rs
@@ -27,7 +27,7 @@ pub fn build_fs_backend(root: &str) -> Result<ObjectStore> {
            DefaultLoggingInterceptor,
        ))
        .layer(object_store::layers::TracingLayer)
-        .layer(object_store::layers::PrometheusMetricsLayer::new(true))
+        .layer(object_store::layers::build_prometheus_metrics_layer(true))
        .finish();
    Ok(object_store)
 }
--- a/src/common/datasource/src/object_store/s3.rs
+++ b/src/common/datasource/src/object_store/s3.rs
@@ -89,7 +89,7 @@ pub fn build_s3_backend(
            DefaultLoggingInterceptor,
        ))
        .layer(object_store::layers::TracingLayer)
-        .layer(object_store::layers::PrometheusMetricsLayer::new(true))
+        .layer(object_store::layers::build_prometheus_metrics_layer(true))
        .finish())
 }

--- a/src/common/datasource/tests/orc/write.py
+++ b/src/common/datasource/tests/orc/write.py
@@ -35,10 +35,23 @@ data = {
    "bigint_other": [5, -5, 1, 5, 5],
    "utf8_increase": ["a", "bb", "ccc", "dddd", "eeeee"],
    "utf8_decrease": ["eeeee", "dddd", "ccc", "bb", "a"],
-    "timestamp_simple": [datetime.datetime(2023, 4, 1, 20, 15, 30, 2000), datetime.datetime.fromtimestamp(int('1629617204525777000')/1000000000), datetime.datetime(2023, 1, 1), datetime.datetime(2023, 2, 1), datetime.datetime(2023, 3, 1)],
-    "date_simple": [datetime.date(2023, 4, 1), datetime.date(2023, 3, 1), datetime.date(2023, 1, 1), datetime.date(2023, 2, 1), datetime.date(2023, 3, 1)]
+    "timestamp_simple": [
+        datetime.datetime(2023, 4, 1, 20, 15, 30, 2000),
+        datetime.datetime.fromtimestamp(int("1629617204525777000") / 1000000000),
+        datetime.datetime(2023, 1, 1),
+        datetime.datetime(2023, 2, 1),
+        datetime.datetime(2023, 3, 1),
+    ],
+    "date_simple": [
+        datetime.date(2023, 4, 1),
+        datetime.date(2023, 3, 1),
+        datetime.date(2023, 1, 1),
+        datetime.date(2023, 2, 1),
+        datetime.date(2023, 3, 1),
+    ],
 }

+
 def infer_schema(data):
    schema = "struct<"
    for key, value in data.items():
@@ -56,7 +69,7 @@ def infer_schema(data):
        elif key.startswith("date"):
            dt = "date"
        else:
-            print(key,value,dt)
+            print(key, value, dt)
            raise NotImplementedError
        if key.startswith("double"):
            dt = "double"
@@ -68,7 +81,6 @@ def infer_schema(data):
    return schema


-
 def _write(
    schema: str,
    data,
--- a/src/common/error/Cargo.toml
+++ b/src/common/error/Cargo.toml
@@ -8,6 +8,7 @@ license.workspace = true
 workspace = true

 [dependencies]
+http.workspace = true
 snafu.workspace = true
 strum.workspace = true
 tonic.workspace = true
--- a/src/common/error/src/lib.rs
+++ b/src/common/error/src/lib.rs
@@ -18,9 +18,30 @@ pub mod ext;
 pub mod mock;
 pub mod status_code;

+use http::{HeaderMap, HeaderValue};
 pub use snafu;

 // HACK - these headers are here for shared in gRPC services. For common HTTP headers,
 // please define in `src/servers/src/http/header.rs`.
 pub const GREPTIME_DB_HEADER_ERROR_CODE: &str = "x-greptime-err-code";
 pub const GREPTIME_DB_HEADER_ERROR_MSG: &str = "x-greptime-err-msg";
+
+/// Create a http header map from error code and message.
+/// using `GREPTIME_DB_HEADER_ERROR_CODE` and `GREPTIME_DB_HEADER_ERROR_MSG` as keys.
+pub fn from_err_code_msg_to_header(code: u32, msg: &str) -> HeaderMap {
+    let mut header = HeaderMap::new();
+
+    let msg = HeaderValue::from_str(msg).unwrap_or_else(|_| {
+        HeaderValue::from_bytes(
+            &msg.as_bytes()
+                .iter()
+                .flat_map(|b| std::ascii::escape_default(*b))
+                .collect::<Vec<u8>>(),
+        )
+        .expect("Already escaped string should be valid ascii")
+    });
+
+    header.insert(GREPTIME_DB_HEADER_ERROR_CODE, code.into());
+    header.insert(GREPTIME_DB_HEADER_ERROR_MSG, msg);
+    header
+}
--- a/src/common/function/Cargo.toml
+++ b/src/common/function/Cargo.toml
@@ -33,7 +33,7 @@ geo-types = { version = "0.7", optional = true }
 geohash = { version = "0.13", optional = true }
 h3o = { version = "0.6", optional = true }
 jsonb.workspace = true
-nalgebra = "0.33"
+nalgebra.workspace = true
 num = "0.4"
 num-traits = "0.2"
 once_cell.workspace = true
--- a/src/common/function/src/lib.rs
+++ b/src/common/function/src/lib.rs
@@ -26,3 +26,4 @@ pub mod function_registry;
 pub mod handlers;
 pub mod helper;
 pub mod state;
+pub mod utils;
--- a/src/common/function/src/scalars/aggregate.rs
+++ b/src/common/function/src/scalars/aggregate.rs
@@ -32,6 +32,7 @@ pub use scipy_stats_norm_cdf::ScipyStatsNormCdfAccumulatorCreator;
 pub use scipy_stats_norm_pdf::ScipyStatsNormPdfAccumulatorCreator;

 use crate::function_registry::FunctionRegistry;
+use crate::scalars::vector::sum::VectorSumCreator;

 /// A function creates `AggregateFunctionCreator`.
 /// "Aggregator" *is* AggregatorFunction. Since the later one is long, we named an short alias for it.
@@ -91,6 +92,7 @@ impl AggregateFunctions {
        register_aggr_func!("argmin", 1, ArgminAccumulatorCreator);
        register_aggr_func!("scipystatsnormcdf", 2, ScipyStatsNormCdfAccumulatorCreator);
        register_aggr_func!("scipystatsnormpdf", 2, ScipyStatsNormPdfAccumulatorCreator);
+        register_aggr_func!("vec_sum", 1, VectorSumCreator);

        #[cfg(feature = "geo")]
        register_aggr_func!(
--- a/src/common/function/src/scalars/matches.rs
+++ b/src/common/function/src/scalars/matches.rs
@@ -204,20 +204,10 @@ impl PatternAst {
    fn convert_literal(column: &str, pattern: &str) -> Expr {
        logical_expr::col(column).like(logical_expr::lit(format!(
            "%{}%",
-            Self::escape_pattern(pattern)
+            crate::utils::escape_like_pattern(pattern)
        )))
    }

-    fn escape_pattern(pattern: &str) -> String {
-        pattern
-            .chars()
-            .flat_map(|c| match c {
-                '\\' | '%' | '_' => vec!['\\', c],
-                _ => vec![c],
-            })
-            .collect::<String>()
-    }
-
    /// Transform this AST with preset rules to make it correct.
    fn transform_ast(self) -> Result<Self> {
        self.transform_up(Self::collapse_binary_branch_fn)
@@ -735,7 +725,8 @@ struct Tokenizer {
 impl Tokenizer {
    pub fn tokenize(mut self, pattern: &str) -> Result<Vec<Token>> {
        let mut tokens = vec![];
-        while self.cursor < pattern.len() {
+        let char_len = pattern.chars().count();
+        while self.cursor < char_len {
            // TODO: collect pattern into Vec<char> if this tokenizer is bottleneck in the future
            let c = pattern.chars().nth(self.cursor).unwrap();
            match c {
@@ -804,7 +795,8 @@ impl Tokenizer {
        let mut phase = String::new();
        let mut is_quote_present = false;

-        while self.cursor < pattern.len() {
+        let char_len = pattern.chars().count();
+        while self.cursor < char_len {
            let mut c = pattern.chars().nth(self.cursor).unwrap();

            match c {
@@ -909,6 +901,26 @@ mod test {
                    Phase("c".to_string()),
                ],
            ),
+            (
+                r#"中文 测试"#,
+                vec![Phase("中文".to_string()), Phase("测试".to_string())],
+            ),
+            (
+                r#"中文 AND 测试"#,
+                vec![Phase("中文".to_string()), And, Phase("测试".to_string())],
+            ),
+            (
+                r#"中文 +测试"#,
+                vec![Phase("中文".to_string()), Must, Phase("测试".to_string())],
+            ),
+            (
+                r#"中文 -测试"#,
+                vec![
+                    Phase("中文".to_string()),
+                    Negative,
+                    Phase("测试".to_string()),
+                ],
+            ),
        ];

        for (query, expected) in cases {
@@ -1040,6 +1052,61 @@ mod test {
                    ],
                },
            ),
+            (
+                r#"中文 测试"#,
+                PatternAst::Binary {
+                    op: BinaryOp::Or,
+                    children: vec![
+                        PatternAst::Literal {
+                            op: UnaryOp::Optional,
+                            pattern: "中文".to_string(),
+                        },
+                        PatternAst::Literal {
+                            op: UnaryOp::Optional,
+                            pattern: "测试".to_string(),
+                        },
+                    ],
+                },
+            ),
+            (
+                r#"中文 AND 测试"#,
+                PatternAst::Binary {
+                    op: BinaryOp::And,
+                    children: vec![
+                        PatternAst::Literal {
+                            op: UnaryOp::Optional,
+                            pattern: "中文".to_string(),
+                        },
+                        PatternAst::Literal {
+                            op: UnaryOp::Optional,
+                            pattern: "测试".to_string(),
+                        },
+                    ],
+                },
+            ),
+            (
+                r#"中文 +测试"#,
+                PatternAst::Literal {
+                    op: UnaryOp::Must,
+                    pattern: "测试".to_string(),
+                },
+            ),
+            (
+                r#"中文 -测试"#,
+                PatternAst::Binary {
+                    op: BinaryOp::And,
+                    children: vec![
+                        PatternAst::Literal {
+                            op: UnaryOp::Negative,
+                            pattern: "测试".to_string(),
+                        },
+                        PatternAst::Literal {
+                            op: UnaryOp::Optional,
+                            pattern: "中文".to_string(),
+                        },
+                    ],
+                },
+            ),
        ];

        for (query, expected) in cases {
--- a/src/common/function/src/scalars/vector.rs
+++ b/src/common/function/src/scalars/vector.rs
@@ -14,9 +14,14 @@

 mod convert;
 mod distance;
-pub(crate) mod impl_conv;
+mod elem_sum;
+pub mod impl_conv;
 mod scalar_add;
 mod scalar_mul;
+mod sub;
+pub(crate) mod sum;
+mod vector_div;
+mod vector_mul;

 use std::sync::Arc;

@@ -38,5 +43,11 @@ impl VectorFunction {
        // scalar calculation
        registry.register(Arc::new(scalar_add::ScalarAddFunction));
        registry.register(Arc::new(scalar_mul::ScalarMulFunction));
+
+        // vector calculation
+        registry.register(Arc::new(vector_mul::VectorMulFunction));
+        registry.register(Arc::new(vector_div::VectorDivFunction));
+        registry.register(Arc::new(sub::SubFunction));
+        registry.register(Arc::new(elem_sum::ElemSumFunction));
    }
 }
--- a/src/common/function/src/scalars/vector/elem_sum.rs
+++ b/src/common/function/src/scalars/vector/elem_sum.rs
@@ -0,0 +1,129 @@
+// Copyright 2023 Greptime Team
+//
+// Licensed under the Apache License, Version 2.0 (the "License");
+// you may not use this file except in compliance with the License.
+// You may obtain a copy of the License at
+//
+//     http://www.apache.org/licenses/LICENSE-2.0
+//
+// Unless required by applicable law or agreed to in writing, software
+// distributed under the License is distributed on an "AS IS" BASIS,
+// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+// See the License for the specific language governing permissions and
+// limitations under the License.
+
+use std::borrow::Cow;
+use std::fmt::Display;
+
+use common_query::error::InvalidFuncArgsSnafu;
+use common_query::prelude::{Signature, TypeSignature, Volatility};
+use datatypes::prelude::ConcreteDataType;
+use datatypes::scalars::ScalarVectorBuilder;
+use datatypes::vectors::{Float32VectorBuilder, MutableVector, VectorRef};
+use nalgebra::DVectorView;
+use snafu::ensure;
+
+use crate::function::{Function, FunctionContext};
+use crate::scalars::vector::impl_conv::{as_veclit, as_veclit_if_const};
+
+const NAME: &str = "vec_elem_sum";
+
+#[derive(Debug, Clone, Default)]
+pub struct ElemSumFunction;
+
+impl Function for ElemSumFunction {
+    fn name(&self) -> &str {
+        NAME
+    }
+
+    fn return_type(
+        &self,
+        _input_types: &[ConcreteDataType],
+    ) -> common_query::error::Result<ConcreteDataType> {
+        Ok(ConcreteDataType::float32_datatype())
+    }
+
+    fn signature(&self) -> Signature {
+        Signature::one_of(
+            vec![
+                TypeSignature::Exact(vec![ConcreteDataType::string_datatype()]),
+                TypeSignature::Exact(vec![ConcreteDataType::binary_datatype()]),
+            ],
+            Volatility::Immutable,
+        )
+    }
+
+    fn eval(
+        &self,
+        _func_ctx: FunctionContext,
+        columns: &[VectorRef],
+    ) -> common_query::error::Result<VectorRef> {
+        ensure!(
+            columns.len() == 1,
+            InvalidFuncArgsSnafu {
+                err_msg: format!(
+                    "The length of the args is not correct, expect exactly one, have: {}",
+                    columns.len()
+                )
+            }
+        );
+        let arg0 = &columns[0];
+
+        let len = arg0.len();
+        let mut result = Float32VectorBuilder::with_capacity(len);
+        if len == 0 {
+            return Ok(result.to_vector());
+        }
+
+        let arg0_const = as_veclit_if_const(arg0)?;
+
+        for i in 0..len {
+            let arg0 = match arg0_const.as_ref() {
+                Some(arg0) => Some(Cow::Borrowed(arg0.as_ref())),
+                None => as_veclit(arg0.get_ref(i))?,
+            };
+            let Some(arg0) = arg0 else {
+                result.push_null();
+                continue;
+            };
+            result.push(Some(DVectorView::from_slice(&arg0, arg0.len()).sum()));
+        }
+
+        Ok(result.to_vector())
+    }
+}
+
+impl Display for ElemSumFunction {
+    fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result {
+        write!(f, "{}", NAME.to_ascii_uppercase())
+    }
+}
+
+#[cfg(test)]
+mod tests {
+    use std::sync::Arc;
+
+    use datatypes::vectors::StringVector;
+
+    use super::*;
+    use crate::function::FunctionContext;
+
+    #[test]
+    fn test_elem_sum() {
+        let func = ElemSumFunction;
+
+        let input0 = Arc::new(StringVector::from(vec![
+            Some("[1.0,2.0,3.0]".to_string()),
+            Some("[4.0,5.0,6.0]".to_string()),
+            None,
+        ]));
+
+        let result = func.eval(FunctionContext::default(), &[input0]).unwrap();
+
+        let result = result.as_ref();
+        assert_eq!(result.len(), 3);
+        assert_eq!(result.get_ref(0).as_f32().unwrap(), Some(6.0));
+        assert_eq!(result.get_ref(1).as_f32().unwrap(), Some(15.0));
+        assert_eq!(result.get_ref(2).as_f32().unwrap(), None);
+    }
+}
--- a/src/common/function/src/scalars/vector/sub.rs
+++ b/src/common/function/src/scalars/vector/sub.rs
@@ -0,0 +1,223 @@
+// Copyright 2023 Greptime Team
+//
+// Licensed under the Apache License, Version 2.0 (the "License");
+// you may not use this file except in compliance with the License.
+// You may obtain a copy of the License at
+//
+//     http://www.apache.org/licenses/LICENSE-2.0
+//
+// Unless required by applicable law or agreed to in writing, software
+// distributed under the License is distributed on an "AS IS" BASIS,
+// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+// See the License for the specific language governing permissions and
+// limitations under the License.
+
+use std::borrow::Cow;
+use std::fmt::Display;
+
+use common_query::error::InvalidFuncArgsSnafu;
+use common_query::prelude::Signature;
+use datatypes::prelude::ConcreteDataType;
+use datatypes::scalars::ScalarVectorBuilder;
+use datatypes::vectors::{BinaryVectorBuilder, MutableVector, VectorRef};
+use nalgebra::DVectorView;
+use snafu::ensure;
+
+use crate::function::{Function, FunctionContext};
+use crate::helper;
+use crate::scalars::vector::impl_conv::{as_veclit, as_veclit_if_const, veclit_to_binlit};
+
+const NAME: &str = "vec_sub";
+
+/// Subtracts corresponding elements of two vectors, returns a vector.
+///
+/// # Example
+///
+/// ```sql
+/// SELECT vec_to_string(vec_sub("[1.0, 1.0]", "[1.0, 2.0]")) as result;
+///
+/// +---------------------------------------------------------------+
+/// | vec_to_string(vec_sub(Utf8("[1.0, 1.0]"),Utf8("[1.0, 2.0]"))) |
+/// +---------------------------------------------------------------+
+/// | [0,-1]                                                        |
+/// +---------------------------------------------------------------+
+///
+/// -- Negative scalar to simulate subtraction
+/// SELECT vec_to_string(vec_sub('[-1.0, -1.0]', '[1.0, 2.0]'));
+///
+/// +-----------------------------------------------------------------+
+/// | vec_to_string(vec_sub(Utf8("[-1.0, -1.0]"),Utf8("[1.0, 2.0]"))) |
+/// +-----------------------------------------------------------------+
+/// | [-2,-3]                                                         |
+/// +-----------------------------------------------------------------+
+///
+#[derive(Debug, Clone, Default)]
+pub struct SubFunction;
+
+impl Function for SubFunction {
+    fn name(&self) -> &str {
+        NAME
+    }
+
+    fn return_type(
+        &self,
+        _input_types: &[ConcreteDataType],
+    ) -> common_query::error::Result<ConcreteDataType> {
+        Ok(ConcreteDataType::binary_datatype())
+    }
+
+    fn signature(&self) -> Signature {
+        helper::one_of_sigs2(
+            vec![
+                ConcreteDataType::string_datatype(),
+                ConcreteDataType::binary_datatype(),
+            ],
+            vec![
+                ConcreteDataType::string_datatype(),
+                ConcreteDataType::binary_datatype(),
+            ],
+        )
+    }
+
+    fn eval(
+        &self,
+        _func_ctx: FunctionContext,
+        columns: &[VectorRef],
+    ) -> common_query::error::Result<VectorRef> {
+        ensure!(
+            columns.len() == 2,
+            InvalidFuncArgsSnafu {
+                err_msg: format!(
+                    "The length of the args is not correct, expect exactly two, have: {}",
+                    columns.len()
+                )
+            }
+        );
+        let arg0 = &columns[0];
+        let arg1 = &columns[1];
+
+        ensure!(
+            arg0.len() == arg1.len(),
+            InvalidFuncArgsSnafu {
+                err_msg: format!(
+                    "The lengths of the vector are not aligned, args 0: {}, args 1: {}",
+                    arg0.len(),
+                    arg1.len(),
+                )
+            }
+        );
+
+        let len = arg0.len();
+        let mut result = BinaryVectorBuilder::with_capacity(len);
+        if len == 0 {
+            return Ok(result.to_vector());
+        }
+
+        let arg0_const = as_veclit_if_const(arg0)?;
+        let arg1_const = as_veclit_if_const(arg1)?;
+
+        for i in 0..len {
+            let arg0 = match arg0_const.as_ref() {
+                Some(arg0) => Some(Cow::Borrowed(arg0.as_ref())),
+                None => as_veclit(arg0.get_ref(i))?,
+            };
+            let arg1 = match arg1_const.as_ref() {
+                Some(arg1) => Some(Cow::Borrowed(arg1.as_ref())),
+                None => as_veclit(arg1.get_ref(i))?,
+            };
+            let (Some(arg0), Some(arg1)) = (arg0, arg1) else {
+                result.push_null();
+                continue;
+            };
+            let vec0 = DVectorView::from_slice(&arg0, arg0.len());
+            let vec1 = DVectorView::from_slice(&arg1, arg1.len());
+
+            let vec_res = vec0 - vec1;
+            let veclit = vec_res.as_slice();
+            let binlit = veclit_to_binlit(veclit);
+            result.push(Some(&binlit));
+        }
+
+        Ok(result.to_vector())
+    }
+}
+
+impl Display for SubFunction {
+    fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result {
+        write!(f, "{}", NAME.to_ascii_uppercase())
+    }
+}
+
+#[cfg(test)]
+mod tests {
+    use std::sync::Arc;
+
+    use common_query::error::Error;
+    use datatypes::vectors::StringVector;
+
+    use super::*;
+
+    #[test]
+    fn test_sub() {
+        let func = SubFunction;
+
+        let input0 = Arc::new(StringVector::from(vec![
+            Some("[1.0,2.0,3.0]".to_string()),
+            Some("[4.0,5.0,6.0]".to_string()),
+            None,
+            Some("[2.0,3.0,3.0]".to_string()),
+        ]));
+        let input1 = Arc::new(StringVector::from(vec![
+            Some("[1.0,1.0,1.0]".to_string()),
+            Some("[6.0,5.0,4.0]".to_string()),
+            Some("[3.0,2.0,2.0]".to_string()),
+            None,
+        ]));
+
+        let result = func
+            .eval(FunctionContext::default(), &[input0, input1])
+            .unwrap();
+
+        let result = result.as_ref();
+        assert_eq!(result.len(), 4);
+        assert_eq!(
+            result.get_ref(0).as_binary().unwrap(),
+            Some(veclit_to_binlit(&[0.0, 1.0, 2.0]).as_slice())
+        );
+        assert_eq!(
+            result.get_ref(1).as_binary().unwrap(),
+            Some(veclit_to_binlit(&[-2.0, 0.0, 2.0]).as_slice())
+        );
+        assert!(result.get_ref(2).is_null());
+        assert!(result.get_ref(3).is_null());
+    }
+
+    #[test]
+    fn test_sub_error() {
+        let func = SubFunction;
+
+        let input0 = Arc::new(StringVector::from(vec![
+            Some("[1.0,2.0,3.0]".to_string()),
+            Some("[4.0,5.0,6.0]".to_string()),
+            None,
+            Some("[2.0,3.0,3.0]".to_string()),
+        ]));
+        let input1 = Arc::new(StringVector::from(vec![
+            Some("[1.0,1.0,1.0]".to_string()),
+            Some("[6.0,5.0,4.0]".to_string()),
+            Some("[3.0,2.0,2.0]".to_string()),
+        ]));
+
+        let result = func.eval(FunctionContext::default(), &[input0, input1]);
+
+        match result {
+            Err(Error::InvalidFuncArgs { err_msg, .. }) => {
+                assert_eq!(
+                    err_msg,
+                    "The lengths of the vector are not aligned, args 0: 4, args 1: 3"
+                )
+            }
+            _ => unreachable!(),
+        }
+    }
+}
--- a/src/common/function/src/scalars/vector/sum.rs
+++ b/src/common/function/src/scalars/vector/sum.rs
@@ -0,0 +1,202 @@
+// Copyright 2023 Greptime Team
+//
+// Licensed under the Apache License, Version 2.0 (the "License");
+// you may not use this file except in compliance with the License.
+// You may obtain a copy of the License at
+//
+//     http://www.apache.org/licenses/LICENSE-2.0
+//
+// Unless required by applicable law or agreed to in writing, software
+// distributed under the License is distributed on an "AS IS" BASIS,
+// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+// See the License for the specific language governing permissions and
+// limitations under the License.
+
+use std::sync::Arc;
+
+use common_macro::{as_aggr_func_creator, AggrFuncTypeStore};
+use common_query::error::{CreateAccumulatorSnafu, Error, InvalidFuncArgsSnafu};
+use common_query::logical_plan::{Accumulator, AggregateFunctionCreator};
+use common_query::prelude::AccumulatorCreatorFunction;
+use datatypes::prelude::{ConcreteDataType, Value, *};
+use datatypes::vectors::VectorRef;
+use nalgebra::{Const, DVectorView, Dyn, OVector};
+use snafu::ensure;
+
+use crate::scalars::vector::impl_conv::{as_veclit, as_veclit_if_const, veclit_to_binlit};
+
+#[derive(Debug, Default)]
+pub struct VectorSum {
+    sum: Option<OVector<f32, Dyn>>,
+    has_null: bool,
+}
+
+#[as_aggr_func_creator]
+#[derive(Debug, Default, AggrFuncTypeStore)]
+pub struct VectorSumCreator {}
+
+impl AggregateFunctionCreator for VectorSumCreator {
+    fn creator(&self) -> AccumulatorCreatorFunction {
+        let creator: AccumulatorCreatorFunction = Arc::new(move |types: &[ConcreteDataType]| {
+            ensure!(
+                types.len() == 1,
+                InvalidFuncArgsSnafu {
+                    err_msg: format!(
+                        "The length of the args is not correct, expect exactly one, have: {}",
+                        types.len()
+                    )
+                }
+            );
+            let input_type = &types[0];
+            match input_type {
+                ConcreteDataType::String(_) | ConcreteDataType::Binary(_) => {
+                    Ok(Box::new(VectorSum::default()))
+                }
+                _ => {
+                    let err_msg = format!(
+                        "\"VEC_SUM\" aggregate function not support data type {:?}",
+                        input_type.logical_type_id(),
+                    );
+                    CreateAccumulatorSnafu { err_msg }.fail()?
+                }
+            }
+        });
+        creator
+    }
+
+    fn output_type(&self) -> common_query::error::Result<ConcreteDataType> {
+        Ok(ConcreteDataType::binary_datatype())
+    }
+
+    fn state_types(&self) -> common_query::error::Result<Vec<ConcreteDataType>> {
+        Ok(vec![self.output_type()?])
+    }
+}
+
+impl VectorSum {
+    fn inner(&mut self, len: usize) -> &mut OVector<f32, Dyn> {
+        self.sum
+            .get_or_insert_with(|| OVector::zeros_generic(Dyn(len), Const::<1>))
+    }
+
+    fn update(&mut self, values: &[VectorRef], is_update: bool) -> Result<(), Error> {
+        if values.is_empty() || self.has_null {
+            return Ok(());
+        };
+        let column = &values[0];
+        let len = column.len();
+
+        match as_veclit_if_const(column)? {
+            Some(column) => {
+                let vec_column = DVectorView::from_slice(&column, column.len()).scale(len as f32);
+                *self.inner(vec_column.len()) += vec_column;
+            }
+            None => {
+                for i in 0..len {
+                    let Some(arg0) = as_veclit(column.get_ref(i))? else {
+                        if is_update {
+                            self.has_null = true;
+                            self.sum = None;
+                        }
+                        return Ok(());
+                    };
+                    let vec_column = DVectorView::from_slice(&arg0, arg0.len());
+                    *self.inner(vec_column.len()) += vec_column;
+                }
+            }
+        }
+        Ok(())
+    }
+}
+
+impl Accumulator for VectorSum {
+    fn state(&self) -> common_query::error::Result<Vec<Value>> {
+        self.evaluate().map(|v| vec![v])
+    }
+
+    fn update_batch(&mut self, values: &[VectorRef]) -> common_query::error::Result<()> {
+        self.update(values, true)
+    }
+
+    fn merge_batch(&mut self, states: &[VectorRef]) -> common_query::error::Result<()> {
+        self.update(states, false)
+    }
+
+    fn evaluate(&self) -> common_query::error::Result<Value> {
+        match &self.sum {
+            None => Ok(Value::Null),
+            Some(vector) => Ok(Value::from(veclit_to_binlit(vector.as_slice()))),
+        }
+    }
+}
+
+#[cfg(test)]
+mod tests {
+    use std::sync::Arc;
+
+    use datatypes::vectors::{ConstantVector, StringVector};
+
+    use super::*;
+
+    #[test]
+    fn test_update_batch() {
+        // test update empty batch, expect not updating anything
+        let mut vec_sum = VectorSum::default();
+        vec_sum.update_batch(&[]).unwrap();
+        assert!(vec_sum.sum.is_none());
+        assert!(!vec_sum.has_null);
+        assert_eq!(Value::Null, vec_sum.evaluate().unwrap());
+
+        // test update one not-null value
+        let mut vec_sum = VectorSum::default();
+        let v: Vec<VectorRef> = vec![Arc::new(StringVector::from(vec![Some(
+            "[1.0,2.0,3.0]".to_string(),
+        )]))];
+        vec_sum.update_batch(&v).unwrap();
+        assert_eq!(
+            Value::from(veclit_to_binlit(&[1.0, 2.0, 3.0])),
+            vec_sum.evaluate().unwrap()
+        );
+
+        // test update one null value
+        let mut vec_sum = VectorSum::default();
+        let v: Vec<VectorRef> = vec![Arc::new(StringVector::from(vec![Option::<String>::None]))];
+        vec_sum.update_batch(&v).unwrap();
+        assert_eq!(Value::Null, vec_sum.evaluate().unwrap());
+
+        // test update no null-value batch
+        let mut vec_sum = VectorSum::default();
+        let v: Vec<VectorRef> = vec![Arc::new(StringVector::from(vec![
+            Some("[1.0,2.0,3.0]".to_string()),
+            Some("[4.0,5.0,6.0]".to_string()),
+            Some("[7.0,8.0,9.0]".to_string()),
+        ]))];
+        vec_sum.update_batch(&v).unwrap();
+        assert_eq!(
+            Value::from(veclit_to_binlit(&[12.0, 15.0, 18.0])),
+            vec_sum.evaluate().unwrap()
+        );
+
+        // test update null-value batch
+        let mut vec_sum = VectorSum::default();
+        let v: Vec<VectorRef> = vec![Arc::new(StringVector::from(vec![
+            Some("[1.0,2.0,3.0]".to_string()),
+            None,
+            Some("[7.0,8.0,9.0]".to_string()),
+        ]))];
+        vec_sum.update_batch(&v).unwrap();
+        assert_eq!(Value::Null, vec_sum.evaluate().unwrap());
+
+        // test update with constant vector
+        let mut vec_sum = VectorSum::default();
+        let v: Vec<VectorRef> = vec![Arc::new(ConstantVector::new(
+            Arc::new(StringVector::from_vec(vec!["[1.0,2.0,3.0]".to_string()])),
+            4,
+        ))];
+        vec_sum.update_batch(&v).unwrap();
+        assert_eq!(
+            Value::from(veclit_to_binlit(&[4.0, 8.0, 12.0])),
+            vec_sum.evaluate().unwrap()
+        );
+    }
+}
--- a/src/common/function/src/scalars/vector/vector_div.rs
+++ b/src/common/function/src/scalars/vector/vector_div.rs
@@ -0,0 +1,218 @@
+// Copyright 2023 Greptime Team
+//
+// Licensed under the Apache License, Version 2.0 (the "License");
+// you may not use this file except in compliance with the License.
+// You may obtain a copy of the License at
+//
+//     http://www.apache.org/licenses/LICENSE-2.0
+//
+// Unless required by applicable law or agreed to in writing, software
+// distributed under the License is distributed on an "AS IS" BASIS,
+// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+// See the License for the specific language governing permissions and
+// limitations under the License.
+
+use std::borrow::Cow;
+use std::fmt::Display;
+
+use common_query::error::{InvalidFuncArgsSnafu, Result};
+use common_query::prelude::Signature;
+use datatypes::prelude::ConcreteDataType;
+use datatypes::scalars::ScalarVectorBuilder;
+use datatypes::vectors::{BinaryVectorBuilder, MutableVector, VectorRef};
+use nalgebra::DVectorView;
+use snafu::ensure;
+
+use crate::function::{Function, FunctionContext};
+use crate::helper;
+use crate::scalars::vector::impl_conv::{as_veclit, as_veclit_if_const, veclit_to_binlit};
+
+const NAME: &str = "vec_div";
+
+/// Divides corresponding elements of two vectors.
+///
+/// # Example
+///
+/// ```sql
+/// SELECT vec_to_string(vec_div("[2, 4, 6]", "[2, 2, 2]")) as result;
+///
+/// +---------+
+/// | result  |
+/// +---------+
+/// | [1,2,3] |
+/// +---------+
+///
+/// ```
+#[derive(Debug, Clone, Default)]
+pub struct VectorDivFunction;
+
+impl Function for VectorDivFunction {
+    fn name(&self) -> &str {
+        NAME
+    }
+
+    fn return_type(&self, _input_types: &[ConcreteDataType]) -> Result<ConcreteDataType> {
+        Ok(ConcreteDataType::binary_datatype())
+    }
+
+    fn signature(&self) -> Signature {
+        helper::one_of_sigs2(
+            vec![
+                ConcreteDataType::string_datatype(),
+                ConcreteDataType::binary_datatype(),
+            ],
+            vec![
+                ConcreteDataType::string_datatype(),
+                ConcreteDataType::binary_datatype(),
+            ],
+        )
+    }
+
+    fn eval(&self, _func_ctx: FunctionContext, columns: &[VectorRef]) -> Result<VectorRef> {
+        ensure!(
+            columns.len() == 2,
+            InvalidFuncArgsSnafu {
+                err_msg: format!(
+                    "The length of the args is not correct, expect exactly two, have: {}",
+                    columns.len()
+                ),
+            }
+        );
+
+        let arg0 = &columns[0];
+        let arg1 = &columns[1];
+
+        let len = arg0.len();
+        let mut result = BinaryVectorBuilder::with_capacity(len);
+        if len == 0 {
+            return Ok(result.to_vector());
+        }
+
+        let arg0_const = as_veclit_if_const(arg0)?;
+        let arg1_const = as_veclit_if_const(arg1)?;
+
+        for i in 0..len {
+            let arg0 = match arg0_const.as_ref() {
+                Some(arg0) => Some(Cow::Borrowed(arg0.as_ref())),
+                None => as_veclit(arg0.get_ref(i))?,
+            };
+
+            let arg1 = match arg1_const.as_ref() {
+                Some(arg1) => Some(Cow::Borrowed(arg1.as_ref())),
+                None => as_veclit(arg1.get_ref(i))?,
+            };
+
+            if let (Some(arg0), Some(arg1)) = (arg0, arg1) {
+                ensure!(
+                    arg0.len() == arg1.len(),
+                    InvalidFuncArgsSnafu {
+                        err_msg: format!(
+                            "The length of the vectors must match for division, have: {} vs {}",
+                            arg0.len(),
+                            arg1.len()
+                        ),
+                    }
+                );
+                let vec0 = DVectorView::from_slice(&arg0, arg0.len());
+                let vec1 = DVectorView::from_slice(&arg1, arg1.len());
+                let vec_res = vec0.component_div(&vec1);
+
+                let veclit = vec_res.as_slice();
+                let binlit = veclit_to_binlit(veclit);
+                result.push(Some(&binlit));
+            } else {
+                result.push_null();
+            }
+        }
+
+        Ok(result.to_vector())
+    }
+}
+
+impl Display for VectorDivFunction {
+    fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result {
+        write!(f, "{}", NAME.to_ascii_uppercase())
+    }
+}
+
+#[cfg(test)]
+mod tests {
+    use std::sync::Arc;
+
+    use common_query::error;
+    use datatypes::vectors::StringVector;
+
+    use super::*;
+
+    #[test]
+    fn test_vector_mul() {
+        let func = VectorDivFunction;
+
+        let vec0 = vec![1.0, 2.0, 3.0];
+        let vec1 = vec![1.0, 1.0];
+        let (len0, len1) = (vec0.len(), vec1.len());
+        let input0 = Arc::new(StringVector::from(vec![Some(format!("{vec0:?}"))]));
+        let input1 = Arc::new(StringVector::from(vec![Some(format!("{vec1:?}"))]));
+
+        let err = func
+            .eval(FunctionContext::default(), &[input0, input1])
+            .unwrap_err();
+
+        match err {
+            error::Error::InvalidFuncArgs { err_msg, .. } => {
+                assert_eq!(
+                    err_msg,
+                    format!(
+                        "The length of the vectors must match for division, have: {} vs {}",
+                        len0, len1
+                    )
+                )
+            }
+            _ => unreachable!(),
+        }
+
+        let input0 = Arc::new(StringVector::from(vec![
+            Some("[1.0,2.0,3.0]".to_string()),
+            Some("[8.0,10.0,12.0]".to_string()),
+            Some("[7.0,8.0,9.0]".to_string()),
+            None,
+        ]));
+
+        let input1 = Arc::new(StringVector::from(vec![
+            Some("[1.0,1.0,1.0]".to_string()),
+            Some("[2.0,2.0,2.0]".to_string()),
+            None,
+            Some("[3.0,3.0,3.0]".to_string()),
+        ]));
+
+        let result = func
+            .eval(FunctionContext::default(), &[input0, input1])
+            .unwrap();
+
+        let result = result.as_ref();
+        assert_eq!(result.len(), 4);
+        assert_eq!(
+            result.get_ref(0).as_binary().unwrap(),
+            Some(veclit_to_binlit(&[1.0, 2.0, 3.0]).as_slice())
+        );
+        assert_eq!(
+            result.get_ref(1).as_binary().unwrap(),
+            Some(veclit_to_binlit(&[4.0, 5.0, 6.0]).as_slice())
+        );
+        assert!(result.get_ref(2).is_null());
+        assert!(result.get_ref(3).is_null());
+
+        let input0 = Arc::new(StringVector::from(vec![Some("[1.0,-2.0]".to_string())]));
+        let input1 = Arc::new(StringVector::from(vec![Some("[0.0,0.0]".to_string())]));
+
+        let result = func
+            .eval(FunctionContext::default(), &[input0, input1])
+            .unwrap();
+
+        let result = result.as_ref();
+        assert_eq!(
+            result.get_ref(0).as_binary().unwrap(),
+            Some(veclit_to_binlit(&[f64::INFINITY as f32, f64::NEG_INFINITY as f32]).as_slice())
+        );
+    }
+}
--- a/src/common/function/src/scalars/vector/vector_mul.rs
+++ b/src/common/function/src/scalars/vector/vector_mul.rs
@@ -0,0 +1,205 @@
+// Copyright 2023 Greptime Team
+//
+// Licensed under the Apache License, Version 2.0 (the "License");
+// you may not use this file except in compliance with the License.
+// You may obtain a copy of the License at
+//
+//     http://www.apache.org/licenses/LICENSE-2.0
+//
+// Unless required by applicable law or agreed to in writing, software
+// distributed under the License is distributed on an "AS IS" BASIS,
+// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+// See the License for the specific language governing permissions and
+// limitations under the License.
+
+use std::borrow::Cow;
+use std::fmt::Display;
+
+use common_query::error::{InvalidFuncArgsSnafu, Result};
+use common_query::prelude::Signature;
+use datatypes::prelude::ConcreteDataType;
+use datatypes::scalars::ScalarVectorBuilder;
+use datatypes::vectors::{BinaryVectorBuilder, MutableVector, VectorRef};
+use nalgebra::DVectorView;
+use snafu::ensure;
+
+use crate::function::{Function, FunctionContext};
+use crate::helper;
+use crate::scalars::vector::impl_conv::{as_veclit, as_veclit_if_const, veclit_to_binlit};
+
+const NAME: &str = "vec_mul";
+
+/// Multiplies corresponding elements of two vectors.
+///
+/// # Example
+///
+/// ```sql
+/// SELECT vec_to_string(vec_mul("[1, 2, 3]", "[1, 2, 3]")) as result;
+///
+/// +---------+
+/// | result  |
+/// +---------+
+/// | [1,4,9] |
+/// +---------+
+///
+/// ```
+#[derive(Debug, Clone, Default)]
+pub struct VectorMulFunction;
+
+impl Function for VectorMulFunction {
+    fn name(&self) -> &str {
+        NAME
+    }
+
+    fn return_type(&self, _input_types: &[ConcreteDataType]) -> Result<ConcreteDataType> {
+        Ok(ConcreteDataType::binary_datatype())
+    }
+
+    fn signature(&self) -> Signature {
+        helper::one_of_sigs2(
+            vec![
+                ConcreteDataType::string_datatype(),
+                ConcreteDataType::binary_datatype(),
+            ],
+            vec![
+                ConcreteDataType::string_datatype(),
+                ConcreteDataType::binary_datatype(),
+            ],
+        )
+    }
+
+    fn eval(&self, _func_ctx: FunctionContext, columns: &[VectorRef]) -> Result<VectorRef> {
+        ensure!(
+            columns.len() == 2,
+            InvalidFuncArgsSnafu {
+                err_msg: format!(
+                    "The length of the args is not correct, expect exactly two, have: {}",
+                    columns.len()
+                ),
+            }
+        );
+
+        let arg0 = &columns[0];
+        let arg1 = &columns[1];
+
+        let len = arg0.len();
+        let mut result = BinaryVectorBuilder::with_capacity(len);
+        if len == 0 {
+            return Ok(result.to_vector());
+        }
+
+        let arg0_const = as_veclit_if_const(arg0)?;
+        let arg1_const = as_veclit_if_const(arg1)?;
+
+        for i in 0..len {
+            let arg0 = match arg0_const.as_ref() {
+                Some(arg0) => Some(Cow::Borrowed(arg0.as_ref())),
+                None => as_veclit(arg0.get_ref(i))?,
+            };
+
+            let arg1 = match arg1_const.as_ref() {
+                Some(arg1) => Some(Cow::Borrowed(arg1.as_ref())),
+                None => as_veclit(arg1.get_ref(i))?,
+            };
+
+            if let (Some(arg0), Some(arg1)) = (arg0, arg1) {
+                ensure!(
+                    arg0.len() == arg1.len(),
+                    InvalidFuncArgsSnafu {
+                        err_msg: format!(
+                            "The length of the vectors must match for multiplying, have: {} vs {}",
+                            arg0.len(),
+                            arg1.len()
+                        ),
+                    }
+                );
+                let vec0 = DVectorView::from_slice(&arg0, arg0.len());
+                let vec1 = DVectorView::from_slice(&arg1, arg1.len());
+                let vec_res = vec1.component_mul(&vec0);
+
+                let veclit = vec_res.as_slice();
+                let binlit = veclit_to_binlit(veclit);
+                result.push(Some(&binlit));
+            } else {
+                result.push_null();
+            }
+        }
+
+        Ok(result.to_vector())
+    }
+}
+
+impl Display for VectorMulFunction {
+    fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result {
+        write!(f, "{}", NAME.to_ascii_uppercase())
+    }
+}
+
+#[cfg(test)]
+mod tests {
+    use std::sync::Arc;
+
+    use common_query::error;
+    use datatypes::vectors::StringVector;
+
+    use super::*;
+
+    #[test]
+    fn test_vector_mul() {
+        let func = VectorMulFunction;
+
+        let vec0 = vec![1.0, 2.0, 3.0];
+        let vec1 = vec![1.0, 1.0];
+        let (len0, len1) = (vec0.len(), vec1.len());
+        let input0 = Arc::new(StringVector::from(vec![Some(format!("{vec0:?}"))]));
+        let input1 = Arc::new(StringVector::from(vec![Some(format!("{vec1:?}"))]));
+
+        let err = func
+            .eval(FunctionContext::default(), &[input0, input1])
+            .unwrap_err();
+
+        match err {
+            error::Error::InvalidFuncArgs { err_msg, .. } => {
+                assert_eq!(
+                    err_msg,
+                    format!(
+                        "The length of the vectors must match for multiplying, have: {} vs {}",
+                        len0, len1
+                    )
+                )
+            }
+            _ => unreachable!(),
+        }
+
+        let input0 = Arc::new(StringVector::from(vec![
+            Some("[1.0,2.0,3.0]".to_string()),
+            Some("[8.0,10.0,12.0]".to_string()),
+            Some("[7.0,8.0,9.0]".to_string()),
+            None,
+        ]));
+
+        let input1 = Arc::new(StringVector::from(vec![
+            Some("[1.0,1.0,1.0]".to_string()),
+            Some("[2.0,2.0,2.0]".to_string()),
+            None,
+            Some("[3.0,3.0,3.0]".to_string()),
+        ]));
+
+        let result = func
+            .eval(FunctionContext::default(), &[input0, input1])
+            .unwrap();
+
+        let result = result.as_ref();
+        assert_eq!(result.len(), 4);
+        assert_eq!(
+            result.get_ref(0).as_binary().unwrap(),
+            Some(veclit_to_binlit(&[1.0, 2.0, 3.0]).as_slice())
+        );
+        assert_eq!(
+            result.get_ref(1).as_binary().unwrap(),
+            Some(veclit_to_binlit(&[16.0, 20.0, 24.0]).as_slice())
+        );
+        assert!(result.get_ref(2).is_null());
+        assert!(result.get_ref(3).is_null());
+    }
+}
--- a/src/common/function/src/utils.rs
+++ b/src/common/function/src/utils.rs
@@ -0,0 +1,58 @@
+// Copyright 2023 Greptime Team
+//
+// Licensed under the Apache License, Version 2.0 (the "License");
+// you may not use this file except in compliance with the License.
+// You may obtain a copy of the License at
+//
+//     http://www.apache.org/licenses/LICENSE-2.0
+//
+// Unless required by applicable law or agreed to in writing, software
+// distributed under the License is distributed on an "AS IS" BASIS,
+// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+// See the License for the specific language governing permissions and
+// limitations under the License.
+
+/// Escapes special characters in the provided pattern string for `LIKE`.
+///
+/// Specifically, it prefixes the backslash (`\`), percent (`%`), and underscore (`_`)
+/// characters with an additional backslash to ensure they are treated literally.
+///
+/// # Examples
+///
+/// ```rust
+/// let escaped = escape_pattern("100%_some\\path");
+/// assert_eq!(escaped, "100\\%\\_some\\\\path");
+/// ```
+pub fn escape_like_pattern(pattern: &str) -> String {
+    pattern
+        .chars()
+        .flat_map(|c| match c {
+            '\\' | '%' | '_' => vec!['\\', c],
+            _ => vec![c],
+        })
+        .collect::<String>()
+}
+#[cfg(test)]
+mod tests {
+    use super::*;
+
+    #[test]
+    fn test_escape_like_pattern() {
+        assert_eq!(
+            escape_like_pattern("100%_some\\path"),
+            "100\\%\\_some\\\\path"
+        );
+        assert_eq!(escape_like_pattern(""), "");
+        assert_eq!(escape_like_pattern("hello"), "hello");
+        assert_eq!(escape_like_pattern("\\%_"), "\\\\\\%\\_");
+        assert_eq!(escape_like_pattern("%%__\\\\"), "\\%\\%\\_\\_\\\\\\\\");
+        assert_eq!(escape_like_pattern("abc123"), "abc123");
+        assert_eq!(escape_like_pattern("%_\\"), "\\%\\_\\\\");
+        assert_eq!(
+            escape_like_pattern("%%__\\\\another%string"),
+            "\\%\\%\\_\\_\\\\\\\\another\\%string"
+        );
+        assert_eq!(escape_like_pattern("foo%bar_"), "foo\\%bar\\_");
+        assert_eq!(escape_like_pattern("\\_\\%"), "\\\\\\_\\\\\\%");
+    }
+}
--- a/src/common/grpc-expr/src/alter.rs
+++ b/src/common/grpc-expr/src/alter.rs
@@ -60,6 +60,7 @@ pub fn alter_expr_to_request(table_id: TableId, expr: AlterTableExpr) -> Result<
                        column_schema: schema,
                        is_key: column_def.semantic_type == SemanticType::Tag as i32,
                        location: parse_location(ac.location)?,
+                        add_if_not_exists: ac.add_if_not_exists,
                    })
                })
                .collect::<Result<Vec<_>>>()?;
@@ -220,6 +221,7 @@ mod tests {
                        ..Default::default()
                    }),
                    location: None,
+                    add_if_not_exists: true,
                }],
            })),
        };
@@ -240,6 +242,7 @@ mod tests {
            add_column.column_schema.data_type
        );
        assert_eq!(None, add_column.location);
+        assert!(add_column.add_if_not_exists);
    }

    #[test]
@@ -265,6 +268,7 @@ mod tests {
                            location_type: LocationType::First.into(),
                            after_column_name: String::default(),
                        }),
+                        add_if_not_exists: false,
                    },
                    AddColumn {
                        column_def: Some(ColumnDef {
@@ -280,6 +284,7 @@ mod tests {
                            location_type: LocationType::After.into(),
                            after_column_name: "ts".to_string(),
                        }),
+                        add_if_not_exists: true,
                    },
                ],
            })),
@@ -308,6 +313,7 @@ mod tests {
            }),
            add_column.location
        );
+        assert!(add_column.add_if_not_exists);

        let add_column = add_columns.pop().unwrap();
        assert!(!add_column.is_key);
@@ -317,6 +323,7 @@ mod tests {
            add_column.column_schema.data_type
        );
        assert_eq!(Some(AddColumnLocation::First), add_column.location);
+        assert!(!add_column.add_if_not_exists);
    }

    #[test]
--- a/src/common/grpc-expr/src/insert.rs
+++ b/src/common/grpc-expr/src/insert.rs
@@ -299,6 +299,7 @@ mod tests {
                .unwrap()
            )
        );
+        assert!(host_column.add_if_not_exists);

        let memory_column = &add_columns.add_columns[1];
        assert_eq!(
@@ -311,6 +312,7 @@ mod tests {
                .unwrap()
            )
        );
+        assert!(host_column.add_if_not_exists);

        let time_column = &add_columns.add_columns[2];
        assert_eq!(
@@ -323,6 +325,7 @@ mod tests {
                .unwrap()
            )
        );
+        assert!(host_column.add_if_not_exists);

        let interval_column = &add_columns.add_columns[3];
        assert_eq!(
@@ -335,6 +338,7 @@ mod tests {
                .unwrap()
            )
        );
+        assert!(host_column.add_if_not_exists);

        let decimal_column = &add_columns.add_columns[4];
        assert_eq!(
@@ -352,6 +356,7 @@ mod tests {
                .unwrap()
            )
        );
+        assert!(host_column.add_if_not_exists);
    }

    #[test]
--- a/src/common/grpc-expr/src/util.rs
+++ b/src/common/grpc-expr/src/util.rs
@@ -192,6 +192,9 @@ pub fn build_create_table_expr(
    Ok(expr)
 }

+/// Find columns that are not present in the schema and return them as `AddColumns`
+/// for adding columns automatically.
+/// It always sets `add_if_not_exists` to `true` for now.
 pub fn extract_new_columns(
    schema: &Schema,
    column_exprs: Vec<ColumnExpr>,
@@ -213,6 +216,7 @@ pub fn extract_new_columns(
            AddColumn {
                column_def,
                location: None,
+                add_if_not_exists: true,
            }
        })
        .collect::<Vec<_>>();
--- a/src/common/meta/src/cache/container.rs
+++ b/src/common/meta/src/cache/container.rs
@@ -43,7 +43,7 @@ pub struct CacheContainer<K, V, CacheToken> {
    cache: Cache<K, V>,
    invalidator: Invalidator<K, V, CacheToken>,
    initializer: Initializer<K, V>,
-    token_filter: TokenFilter<CacheToken>,
+    token_filter: fn(&CacheToken) -> bool,
 }

 impl<K, V, CacheToken> CacheContainer<K, V, CacheToken>
@@ -58,7 +58,7 @@ where
        cache: Cache<K, V>,
        invalidator: Invalidator<K, V, CacheToken>,
        initializer: Initializer<K, V>,
-        token_filter: TokenFilter<CacheToken>,
+        token_filter: fn(&CacheToken) -> bool,
    ) -> Self {
        Self {
            name,
@@ -206,10 +206,13 @@ mod tests {
        name: &'a str,
    }

+    fn always_true_filter(_: &String) -> bool {
+        true
+    }
+
    #[tokio::test]
    async fn test_get() {
        let cache: Cache<NameKey, String> = CacheBuilder::new(128).build();
-        let filter: TokenFilter<String> = Box::new(|_| true);
        let counter = Arc::new(AtomicI32::new(0));
        let moved_counter = counter.clone();
        let init: Initializer<NameKey, String> = Arc::new(move |_| {
@@ -219,7 +222,13 @@ mod tests {
        let invalidator: Invalidator<NameKey, String, String> =
            Box::new(|_, _| Box::pin(async { Ok(()) }));

-        let adv_cache = CacheContainer::new("test".to_string(), cache, invalidator, init, filter);
+        let adv_cache = CacheContainer::new(
+            "test".to_string(),
+            cache,
+            invalidator,
+            init,
+            always_true_filter,
+        );
        let key = NameKey { name: "key" };
        let value = adv_cache.get(key).await.unwrap().unwrap();
        assert_eq!(value, "hi");
@@ -233,7 +242,6 @@ mod tests {
    #[tokio::test]
    async fn test_get_by_ref() {
        let cache: Cache<String, String> = CacheBuilder::new(128).build();
-        let filter: TokenFilter<String> = Box::new(|_| true);
        let counter = Arc::new(AtomicI32::new(0));
        let moved_counter = counter.clone();
        let init: Initializer<String, String> = Arc::new(move |_| {
@@ -243,7 +251,13 @@ mod tests {
        let invalidator: Invalidator<String, String, String> =
            Box::new(|_, _| Box::pin(async { Ok(()) }));

-        let adv_cache = CacheContainer::new("test".to_string(), cache, invalidator, init, filter);
+        let adv_cache = CacheContainer::new(
+            "test".to_string(),
+            cache,
+            invalidator,
+            init,
+            always_true_filter,
+        );
        let value = adv_cache.get_by_ref("foo").await.unwrap().unwrap();
        assert_eq!(value, "hi");
        let value = adv_cache.get_by_ref("foo").await.unwrap().unwrap();
@@ -257,13 +271,18 @@ mod tests {
    #[tokio::test]
    async fn test_get_value_not_exits() {
        let cache: Cache<String, String> = CacheBuilder::new(128).build();
-        let filter: TokenFilter<String> = Box::new(|_| true);
        let init: Initializer<String, String> =
            Arc::new(move |_| Box::pin(async { error::ValueNotExistSnafu {}.fail() }));
        let invalidator: Invalidator<String, String, String> =
            Box::new(|_, _| Box::pin(async { Ok(()) }));

-        let adv_cache = CacheContainer::new("test".to_string(), cache, invalidator, init, filter);
+        let adv_cache = CacheContainer::new(
+            "test".to_string(),
+            cache,
+            invalidator,
+            init,
+            always_true_filter,
+        );
        let value = adv_cache.get_by_ref("foo").await.unwrap();
        assert!(value.is_none());
    }
@@ -271,7 +290,6 @@ mod tests {
    #[tokio::test]
    async fn test_invalidate() {
        let cache: Cache<String, String> = CacheBuilder::new(128).build();
-        let filter: TokenFilter<String> = Box::new(|_| true);
        let counter = Arc::new(AtomicI32::new(0));
        let moved_counter = counter.clone();
        let init: Initializer<String, String> = Arc::new(move |_| {
@@ -285,7 +303,13 @@ mod tests {
            })
        });

-        let adv_cache = CacheContainer::new("test".to_string(), cache, invalidator, init, filter);
+        let adv_cache = CacheContainer::new(
+            "test".to_string(),
+            cache,
+            invalidator,
+            init,
+            always_true_filter,
+        );
        let value = adv_cache.get_by_ref("foo").await.unwrap().unwrap();
        assert_eq!(value, "hi");
        let value = adv_cache.get_by_ref("foo").await.unwrap().unwrap();
--- a/src/common/meta/src/cache/flow/table_flownode.rs
+++ b/src/common/meta/src/cache/flow/table_flownode.rs
@@ -45,7 +45,7 @@ pub fn new_table_flownode_set_cache(
    let table_flow_manager = Arc::new(TableFlowManager::new(kv_backend));
    let init = init_factory(table_flow_manager);

-    CacheContainer::new(name, cache, Box::new(invalidator), init, Box::new(filter))
+    CacheContainer::new(name, cache, Box::new(invalidator), init, filter)
 }

 fn init_factory(table_flow_manager: TableFlowManagerRef) -> Initializer<TableId, FlownodeSet> {
--- a/src/common/meta/src/cache/registry.rs
+++ b/src/common/meta/src/cache/registry.rs
@@ -151,12 +151,15 @@ mod tests {
    use crate::cache::*;
    use crate::instruction::CacheIdent;

+    fn always_true_filter(_: &CacheIdent) -> bool {
+        true
+    }
+
    fn test_cache(
        name: &str,
        invalidator: Invalidator<String, String, CacheIdent>,
    ) -> CacheContainer<String, String, CacheIdent> {
        let cache: Cache<String, String> = CacheBuilder::new(128).build();
-        let filter: TokenFilter<CacheIdent> = Box::new(|_| true);
        let counter = Arc::new(AtomicI32::new(0));
        let moved_counter = counter.clone();
        let init: Initializer<String, String> = Arc::new(move |_| {
@@ -164,7 +167,13 @@ mod tests {
            Box::pin(async { Ok(Some("hi".to_string())) })
        });

-        CacheContainer::new(name.to_string(), cache, invalidator, init, filter)
+        CacheContainer::new(
+            name.to_string(),
+            cache,
+            invalidator,
+            init,
+            always_true_filter,
+        )
    }

    fn test_i32_cache(
@@ -172,7 +181,6 @@ mod tests {
        invalidator: Invalidator<i32, String, CacheIdent>,
    ) -> CacheContainer<i32, String, CacheIdent> {
        let cache: Cache<i32, String> = CacheBuilder::new(128).build();
-        let filter: TokenFilter<CacheIdent> = Box::new(|_| true);
        let counter = Arc::new(AtomicI32::new(0));
        let moved_counter = counter.clone();
        let init: Initializer<i32, String> = Arc::new(move |_| {
@@ -180,7 +188,13 @@ mod tests {
            Box::pin(async { Ok(Some("foo".to_string())) })
        });

-        CacheContainer::new(name.to_string(), cache, invalidator, init, filter)
+        CacheContainer::new(
+            name.to_string(),
+            cache,
+            invalidator,
+            init,
+            always_true_filter,
+        )
    }

    #[tokio::test]
--- a/src/common/meta/src/cache/table/schema.rs
+++ b/src/common/meta/src/cache/table/schema.rs
@@ -36,7 +36,7 @@ pub fn new_schema_cache(
    let schema_manager = SchemaManager::new(kv_backend.clone());
    let init = init_factory(schema_manager);

-    CacheContainer::new(name, cache, Box::new(invalidator), init, Box::new(filter))
+    CacheContainer::new(name, cache, Box::new(invalidator), init, filter)
 }

 fn init_factory(schema_manager: SchemaManager) -> Initializer<SchemaName, Arc<SchemaNameValue>> {
--- a/src/common/meta/src/cache/table/table_info.rs
+++ b/src/common/meta/src/cache/table/table_info.rs
@@ -41,7 +41,7 @@ pub fn new_table_info_cache(
    let table_info_manager = Arc::new(TableInfoManager::new(kv_backend));
    let init = init_factory(table_info_manager);

-    CacheContainer::new(name, cache, Box::new(invalidator), init, Box::new(filter))
+    CacheContainer::new(name, cache, Box::new(invalidator), init, filter)
 }

 fn init_factory(table_info_manager: TableInfoManagerRef) -> Initializer<TableId, Arc<TableInfo>> {
--- a/src/common/meta/src/cache/table/table_name.rs
+++ b/src/common/meta/src/cache/table/table_name.rs
@@ -41,7 +41,7 @@ pub fn new_table_name_cache(
    let table_name_manager = Arc::new(TableNameManager::new(kv_backend));
    let init = init_factory(table_name_manager);

-    CacheContainer::new(name, cache, Box::new(invalidator), init, Box::new(filter))
+    CacheContainer::new(name, cache, Box::new(invalidator), init, filter)
 }

 fn init_factory(table_name_manager: TableNameManagerRef) -> Initializer<TableName, TableId> {
--- a/src/common/meta/src/cache/table/table_route.rs
+++ b/src/common/meta/src/cache/table/table_route.rs
@@ -65,7 +65,7 @@ pub fn new_table_route_cache(
    let table_info_manager = Arc::new(TableRouteManager::new(kv_backend));
    let init = init_factory(table_info_manager);

-    CacheContainer::new(name, cache, Box::new(invalidator), init, Box::new(filter))
+    CacheContainer::new(name, cache, Box::new(invalidator), init, filter)
 }

 fn init_factory(
--- a/src/common/meta/src/cache/table/table_schema.rs
+++ b/src/common/meta/src/cache/table/table_schema.rs
@@ -40,7 +40,7 @@ pub fn new_table_schema_cache(
    let table_info_manager = TableInfoManager::new(kv_backend);
    let init = init_factory(table_info_manager);

-    CacheContainer::new(name, cache, Box::new(invalidator), init, Box::new(filter))
+    CacheContainer::new(name, cache, Box::new(invalidator), init, filter)
 }

 fn init_factory(table_info_manager: TableInfoManager) -> Initializer<TableId, Arc<SchemaName>> {
--- a/src/common/meta/src/cache/table/view_info.rs
+++ b/src/common/meta/src/cache/table/view_info.rs
@@ -40,7 +40,7 @@ pub fn new_view_info_cache(
    let view_info_manager = Arc::new(ViewInfoManager::new(kv_backend));
    let init = init_factory(view_info_manager);

-    CacheContainer::new(name, cache, Box::new(invalidator), init, Box::new(filter))
+    CacheContainer::new(name, cache, Box::new(invalidator), init, filter)
 }

 fn init_factory(view_info_manager: ViewInfoManagerRef) -> Initializer<TableId, Arc<ViewInfoValue>> {
--- a/src/common/meta/src/ddl/alter_logical_tables/update_metadata.rs
+++ b/src/common/meta/src/ddl/alter_logical_tables/update_metadata.rs
@@ -105,7 +105,7 @@ impl AlterLogicalTablesProcedure {
                .context(ConvertAlterTableRequestSnafu)?;
        let new_meta = table_info
            .meta
-            .builder_with_alter_kind(table_ref.table, &request.alter_kind, true)
+            .builder_with_alter_kind(table_ref.table, &request.alter_kind)
            .context(error::TableSnafu)?
            .build()
            .with_context(|_| error::BuildTableMetaSnafu {
--- a/src/common/meta/src/ddl/alter_table.rs
+++ b/src/common/meta/src/ddl/alter_table.rs
@@ -28,13 +28,13 @@ use common_procedure::error::{FromJsonSnafu, Result as ProcedureResult, ToJsonSn
 use common_procedure::{
    Context as ProcedureContext, Error as ProcedureError, LockKey, Procedure, Status, StringKey,
 };
-use common_telemetry::{debug, info};
+use common_telemetry::{debug, error, info};
 use futures::future;
 use serde::{Deserialize, Serialize};
 use snafu::ResultExt;
 use store_api::storage::RegionId;
 use strum::AsRefStr;
-use table::metadata::{RawTableInfo, TableId};
+use table::metadata::{RawTableInfo, TableId, TableInfo};
 use table::table_reference::TableReference;

 use crate::cache_invalidator::Context;
@@ -51,10 +51,14 @@ use crate::{metrics, ClusterId};

 /// The alter table procedure
 pub struct AlterTableProcedure {
-    // The runtime context.
+    /// The runtime context.
    context: DdlContext,
-    // The serialized data.
+    /// The serialized data.
    data: AlterTableData,
+    /// Cached new table metadata in the prepare step.
+    /// If we recover the procedure from json, then the table info value is not cached.
+    /// But we already validated it in the prepare step.
+    new_table_info: Option<TableInfo>,
 }

 impl AlterTableProcedure {
@@ -70,18 +74,31 @@ impl AlterTableProcedure {
        Ok(Self {
            context,
            data: AlterTableData::new(task, table_id, cluster_id),
+            new_table_info: None,
        })
    }

    pub fn from_json(json: &str, context: DdlContext) -> ProcedureResult<Self> {
        let data: AlterTableData = serde_json::from_str(json).context(FromJsonSnafu)?;
-        Ok(AlterTableProcedure { context, data })
+        Ok(AlterTableProcedure {
+            context,
+            data,
+            new_table_info: None,
+        })
    }

    // Checks whether the table exists.
    pub(crate) async fn on_prepare(&mut self) -> Result<Status> {
        self.check_alter().await?;
        self.fill_table_info().await?;
+
+        // Validates the request and builds the new table info.
+        // We need to build the new table info here because we should ensure the alteration
+        // is valid in `UpdateMeta` state as we already altered the region.
+        // Safety: `fill_table_info()` already set it.
+        let table_info_value = self.data.table_info_value.as_ref().unwrap();
+        self.new_table_info = Some(self.build_new_table_info(&table_info_value.table_info)?);
+
        // Safety: Checked in `AlterTableProcedure::new`.
        let alter_kind = self.data.task.alter_table.kind.as_ref().unwrap();
        if matches!(alter_kind, Kind::RenameTable { .. }) {
@@ -106,6 +123,14 @@ impl AlterTableProcedure {

        let leaders = find_leaders(&physical_table_route.region_routes);
        let mut alter_region_tasks = Vec::with_capacity(leaders.len());
+        let alter_kind = self.make_region_alter_kind()?;
+
+        info!(
+            "Submitting alter region requests for table {}, table_id: {}, alter_kind: {:?}",
+            self.data.table_ref(),
+            table_id,
+            alter_kind,
+        );

        for datanode in leaders {
            let requester = self.context.node_manager.datanode(&datanode).await;
@@ -113,7 +138,7 @@ impl AlterTableProcedure {

            for region in regions {
                let region_id = RegionId::new(table_id, region);
-                let request = self.make_alter_region_request(region_id)?;
+                let request = self.make_alter_region_request(region_id, alter_kind.clone())?;
                debug!("Submitting {request:?} to {datanode}");

                let datanode = datanode.clone();
@@ -150,7 +175,15 @@ impl AlterTableProcedure {
        let table_ref = self.data.table_ref();
        // Safety: checked before.
        let table_info_value = self.data.table_info_value.as_ref().unwrap();
-        let new_info = self.build_new_table_info(&table_info_value.table_info)?;
+        // Gets the table info from the cache or builds it.
+        let new_info = match &self.new_table_info {
+            Some(cached) => cached.clone(),
+            None => self.build_new_table_info(&table_info_value.table_info)
+                .inspect_err(|e| {
+                    // We already check the table info in the prepare step so this should not happen.
+                    error!(e; "Unable to build info for table {} in update metadata step, table_id: {}", table_ref, table_id);
+                })?,
+        };

        debug!(
            "Starting update table: {} metadata, new table info {:?}",
@@ -174,7 +207,7 @@ impl AlterTableProcedure {
            .await?;
        }

-        info!("Updated table metadata for table {table_ref}, table_id: {table_id}");
+        info!("Updated table metadata for table {table_ref}, table_id: {table_id}, kind: {alter_kind:?}");
        self.data.state = AlterTableState::InvalidateTableCache;
        Ok(Status::executing(true))
    }
--- a/src/common/meta/src/ddl/alter_table/region_request.rs
+++ b/src/common/meta/src/ddl/alter_table/region_request.rs
@@ -12,6 +12,8 @@
 // See the License for the specific language governing permissions and
 // limitations under the License.

+use std::collections::HashSet;
+
 use api::v1::alter_table_expr::Kind;
 use api::v1::region::region_request::Body;
 use api::v1::region::{
@@ -27,13 +29,15 @@ use crate::ddl::alter_table::AlterTableProcedure;
 use crate::error::{InvalidProtoMsgSnafu, Result};

 impl AlterTableProcedure {
-    /// Makes alter region request.
-    pub(crate) fn make_alter_region_request(&self, region_id: RegionId) -> Result<RegionRequest> {
-        // Safety: Checked in `AlterTableProcedure::new`.
-        let alter_kind = self.data.task.alter_table.kind.as_ref().unwrap();
+    /// Makes alter region request from existing an alter kind.
+    /// Region alter request always add columns if not exist.
+    pub(crate) fn make_alter_region_request(
+        &self,
+        region_id: RegionId,
+        kind: Option<alter_request::Kind>,
+    ) -> Result<RegionRequest> {
        // Safety: checked
        let table_info = self.data.table_info().unwrap();
-        let kind = create_proto_alter_kind(table_info, alter_kind)?;

        Ok(RegionRequest {
            header: Some(RegionRequestHeader {
@@ -47,45 +51,66 @@ impl AlterTableProcedure {
            })),
        })
    }
+
+    /// Makes alter kind proto that all regions can reuse.
+    /// Region alter request always add columns if not exist.
+    pub(crate) fn make_region_alter_kind(&self) -> Result<Option<alter_request::Kind>> {
+        // Safety: Checked in `AlterTableProcedure::new`.
+        let alter_kind = self.data.task.alter_table.kind.as_ref().unwrap();
+        // Safety: checked
+        let table_info = self.data.table_info().unwrap();
+        let kind = create_proto_alter_kind(table_info, alter_kind)?;
+
+        Ok(kind)
+    }
 }

 /// Creates region proto alter kind from `table_info` and `alter_kind`.
 ///
-/// Returns the kind and next column id if it adds new columns.
+/// It always adds column if not exists and drops column if exists.
+/// It skips the column if it already exists in the table.
 fn create_proto_alter_kind(
    table_info: &RawTableInfo,
    alter_kind: &Kind,
 ) -> Result<Option<alter_request::Kind>> {
    match alter_kind {
        Kind::AddColumns(x) => {
+            // Construct a set of existing columns in the table.
+            let existing_columns: HashSet<_> = table_info
+                .meta
+                .schema
+                .column_schemas
+                .iter()
+                .map(|col| &col.name)
+                .collect();
            let mut next_column_id = table_info.meta.next_column_id;

-            let add_columns = x
-                .add_columns
-                .iter()
-                .map(|add_column| {
-                    let column_def =
-                        add_column
-                            .column_def
-                            .as_ref()
-                            .context(InvalidProtoMsgSnafu {
-                                err_msg: "'column_def' is absent",
-                            })?;
+            let mut add_columns = Vec::with_capacity(x.add_columns.len());
+            for add_column in &x.add_columns {
+                let column_def = add_column
+                    .column_def
+                    .as_ref()
+                    .context(InvalidProtoMsgSnafu {
+                        err_msg: "'column_def' is absent",
+                    })?;

-                    let column_id = next_column_id;
-                    next_column_id += 1;
+                // Skips existing columns.
+                if existing_columns.contains(&column_def.name) {
+                    continue;
+                }

-                    let column_def = RegionColumnDef {
-                        column_def: Some(column_def.clone()),
-                        column_id,
-                    };
+                let column_id = next_column_id;
+                next_column_id += 1;
+                let column_def = RegionColumnDef {
+                    column_def: Some(column_def.clone()),
+                    column_id,
+                };

-                    Ok(AddColumn {
-                        column_def: Some(column_def),
-                        location: add_column.location.clone(),
-                    })
-                })
-                .collect::<Result<Vec<_>>>()?;
+                add_columns.push(AddColumn {
+                    column_def: Some(column_def),
+                    location: add_column.location.clone(),
+                });
+            }

            Ok(Some(alter_request::Kind::AddColumns(AddColumns {
                add_columns,
@@ -143,6 +168,7 @@ mod tests {
    use crate::rpc::router::{Region, RegionRoute};
    use crate::test_util::{new_ddl_context, MockDatanodeManager};

+    /// Prepares a region with schema `[ts: Timestamp, host: Tag, cpu: Field]`.
    async fn prepare_ddl_context() -> (DdlContext, u64, TableId, RegionId, String) {
        let datanode_manager = Arc::new(MockDatanodeManager::new(()));
        let ddl_context = new_ddl_context(datanode_manager);
@@ -171,6 +197,7 @@ mod tests {
                    .name("cpu")
                    .data_type(ColumnDataType::Float64)
                    .semantic_type(SemanticType::Field)
+                    .is_nullable(true)
                    .build()
                    .unwrap()
                    .into(),
@@ -225,15 +252,16 @@ mod tests {
                            name: "my_tag3".to_string(),
                            data_type: ColumnDataType::String as i32,
                            is_nullable: true,
-                            default_constraint: b"hello".to_vec(),
+                            default_constraint: Vec::new(),
                            semantic_type: SemanticType::Tag as i32,
                            comment: String::new(),
                            ..Default::default()
                        }),
                        location: Some(AddColumnLocation {
                            location_type: LocationType::After as i32,
-                            after_column_name: "my_tag2".to_string(),
+                            after_column_name: "host".to_string(),
                        }),
+                        add_if_not_exists: false,
                    }],
                })),
            },
@@ -242,8 +270,11 @@ mod tests {
        let mut procedure =
            AlterTableProcedure::new(cluster_id, table_id, task, ddl_context).unwrap();
        procedure.on_prepare().await.unwrap();
-        let Some(Body::Alter(alter_region_request)) =
-            procedure.make_alter_region_request(region_id).unwrap().body
+        let alter_kind = procedure.make_region_alter_kind().unwrap();
+        let Some(Body::Alter(alter_region_request)) = procedure
+            .make_alter_region_request(region_id, alter_kind)
+            .unwrap()
+            .body
        else {
            unreachable!()
        };
@@ -259,7 +290,7 @@ mod tests {
                                name: "my_tag3".to_string(),
                                data_type: ColumnDataType::String as i32,
                                is_nullable: true,
-                                default_constraint: b"hello".to_vec(),
+                                default_constraint: Vec::new(),
                                semantic_type: SemanticType::Tag as i32,
                                comment: String::new(),
                                ..Default::default()
@@ -268,7 +299,7 @@ mod tests {
                        }),
                        location: Some(AddColumnLocation {
                            location_type: LocationType::After as i32,
-                            after_column_name: "my_tag2".to_string(),
+                            after_column_name: "host".to_string(),
                        }),
                    }]
                }
@@ -299,8 +330,11 @@ mod tests {
        let mut procedure =
            AlterTableProcedure::new(cluster_id, table_id, task, ddl_context).unwrap();
        procedure.on_prepare().await.unwrap();
-        let Some(Body::Alter(alter_region_request)) =
-            procedure.make_alter_region_request(region_id).unwrap().body
+        let alter_kind = procedure.make_region_alter_kind().unwrap();
+        let Some(Body::Alter(alter_region_request)) = procedure
+            .make_alter_region_request(region_id, alter_kind)
+            .unwrap()
+            .body
        else {
            unreachable!()
        };
--- a/src/common/meta/src/ddl/alter_table/update_metadata.rs
+++ b/src/common/meta/src/ddl/alter_table/update_metadata.rs
@@ -23,7 +23,9 @@ use crate::key::table_info::TableInfoValue;
 use crate::key::{DeserializedValueWithBytes, RegionDistribution};

 impl AlterTableProcedure {
-    /// Builds new_meta
+    /// Builds new table info after alteration.
+    /// It bumps the column id of the table by the number of the add column requests.
+    /// So there may be holes in the column id sequence.
    pub(crate) fn build_new_table_info(&self, table_info: &RawTableInfo) -> Result<TableInfo> {
        let table_info =
            TableInfo::try_from(table_info.clone()).context(error::ConvertRawTableInfoSnafu)?;
@@ -34,7 +36,7 @@ impl AlterTableProcedure {

        let new_meta = table_info
            .meta
-            .builder_with_alter_kind(table_ref.table, &request.alter_kind, false)
+            .builder_with_alter_kind(table_ref.table, &request.alter_kind)
            .context(error::TableSnafu)?
            .build()
            .with_context(|_| error::BuildTableMetaSnafu {
@@ -46,6 +48,9 @@ impl AlterTableProcedure {
        new_info.ident.version = table_info.ident.version + 1;
        match request.alter_kind {
            AlterKind::AddColumns { columns } => {
+                // Bumps the column id for the new columns.
+                // It may bump more than the actual number of columns added if there are
+                // existing columns, but it's fine.
                new_info.meta.next_column_id += columns.len() as u32;
            }
            AlterKind::RenameTable { new_table_name } => {
--- a/src/common/meta/src/ddl/test_util/alter_table.rs
+++ b/src/common/meta/src/ddl/test_util/alter_table.rs
@@ -30,6 +30,8 @@ pub struct TestAlterTableExpr {
    add_columns: Vec<ColumnDef>,
    #[builder(setter(into, strip_option))]
    new_table_name: Option<String>,
+    #[builder(setter)]
+    add_if_not_exists: bool,
 }

 impl From<TestAlterTableExpr> for AlterTableExpr {
@@ -53,6 +55,7 @@ impl From<TestAlterTableExpr> for AlterTableExpr {
                        .map(|col| AddColumn {
                            column_def: Some(col),
                            location: None,
+                            add_if_not_exists: value.add_if_not_exists,
                        })
                        .collect(),
                })),
--- a/src/common/meta/src/ddl/tests/alter_logical_tables.rs
+++ b/src/common/meta/src/ddl/tests/alter_logical_tables.rs
@@ -56,6 +56,7 @@ fn make_alter_logical_table_add_column_task(
    let alter_table = alter_table
        .table_name(table.to_string())
        .add_columns(add_columns)
+        .add_if_not_exists(true)
        .build()
        .unwrap();

--- a/src/common/meta/src/ddl/tests/alter_table.rs
+++ b/src/common/meta/src/ddl/tests/alter_table.rs
@@ -139,7 +139,7 @@ async fn test_on_submit_alter_request() {
            table_name: table_name.to_string(),
            kind: Some(Kind::DropColumns(DropColumns {
                drop_columns: vec![DropColumn {
-                    name: "my_field_column".to_string(),
+                    name: "cpu".to_string(),
                }],
            })),
        },
@@ -225,7 +225,7 @@ async fn test_on_submit_alter_request_with_outdated_request() {
            table_name: table_name.to_string(),
            kind: Some(Kind::DropColumns(DropColumns {
                drop_columns: vec![DropColumn {
-                    name: "my_field_column".to_string(),
+                    name: "cpu".to_string(),
                }],
            })),
        },
@@ -330,6 +330,7 @@ async fn test_on_update_metadata_add_columns() {
                        ..Default::default()
                    }),
                    location: None,
+                    add_if_not_exists: false,
                }],
            })),
        },
--- a/src/common/meta/src/key/catalog_name.rs
+++ b/src/common/meta/src/key/catalog_name.rs
@@ -13,7 +13,6 @@
 // limitations under the License.

 use std::fmt::Display;
-use std::sync::Arc;

 use common_catalog::consts::DEFAULT_CATALOG_NAME;
 use futures::stream::BoxStream;
@@ -146,7 +145,7 @@ impl CatalogManager {
            self.kv_backend.clone(),
            req,
            DEFAULT_PAGE_SIZE,
-            Arc::new(catalog_decoder),
+            catalog_decoder,
        )
        .into_stream();

@@ -156,6 +155,8 @@ impl CatalogManager {

 #[cfg(test)]
 mod tests {
+    use std::sync::Arc;
+
    use super::*;
    use crate::kv_backend::memory::MemoryKvBackend;

--- a/src/common/meta/src/key/datanode_table.rs
+++ b/src/common/meta/src/key/datanode_table.rs
@@ -14,7 +14,6 @@

 use std::collections::HashMap;
 use std::fmt::Display;
-use std::sync::Arc;

 use futures::stream::BoxStream;
 use serde::{Deserialize, Serialize};
@@ -166,7 +165,7 @@ impl DatanodeTableManager {
            self.kv_backend.clone(),
            req,
            DEFAULT_PAGE_SIZE,
-            Arc::new(datanode_table_value_decoder),
+            datanode_table_value_decoder,
        )
        .into_stream();

--- a/src/common/meta/src/key/flow/flow_name.rs
+++ b/src/common/meta/src/key/flow/flow_name.rs
@@ -12,8 +12,6 @@
 // See the License for the specific language governing permissions and
 // limitations under the License.

-use std::sync::Arc;
-
 use futures::stream::BoxStream;
 use lazy_static::lazy_static;
 use regex::Regex;
@@ -201,7 +199,7 @@ impl FlowNameManager {
            self.kv_backend.clone(),
            req,
            DEFAULT_PAGE_SIZE,
-            Arc::new(flow_name_decoder),
+            flow_name_decoder,
        )
        .into_stream();

--- a/src/common/meta/src/key/flow/flow_route.rs
+++ b/src/common/meta/src/key/flow/flow_route.rs
@@ -12,8 +12,6 @@
 // See the License for the specific language governing permissions and
 // limitations under the License.

-use std::sync::Arc;
-
 use futures::stream::BoxStream;
 use lazy_static::lazy_static;
 use regex::Regex;
@@ -179,7 +177,7 @@ impl FlowRouteManager {
            self.kv_backend.clone(),
            req,
            DEFAULT_PAGE_SIZE,
-            Arc::new(flow_route_decoder),
+            flow_route_decoder,
        )
        .into_stream();

--- a/src/common/meta/src/key/flow/flownode_flow.rs
+++ b/src/common/meta/src/key/flow/flownode_flow.rs
@@ -12,8 +12,6 @@
 // See the License for the specific language governing permissions and
 // limitations under the License.

-use std::sync::Arc;
-
 use futures::stream::BoxStream;
 use futures::TryStreamExt;
 use lazy_static::lazy_static;
@@ -179,7 +177,7 @@ impl FlownodeFlowManager {
            self.kv_backend.clone(),
            req,
            DEFAULT_PAGE_SIZE,
-            Arc::new(flownode_flow_key_decoder),
+            flownode_flow_key_decoder,
        )
        .into_stream();

--- a/src/common/meta/src/key/flow/table_flow.rs
+++ b/src/common/meta/src/key/flow/table_flow.rs
@@ -206,7 +206,7 @@ impl TableFlowManager {
            self.kv_backend.clone(),
            req,
            DEFAULT_PAGE_SIZE,
-            Arc::new(table_flow_decoder),
+            table_flow_decoder,
        )
        .into_stream();

--- a/src/common/meta/src/key/schema_metadata_manager.rs
+++ b/src/common/meta/src/key/schema_metadata_manager.rs
@@ -28,13 +28,10 @@ pub type SchemaMetadataManagerRef = Arc<SchemaMetadataManager>;
 pub struct SchemaMetadataManager {
    table_id_schema_cache: TableSchemaCacheRef,
    schema_cache: SchemaCacheRef,
-    #[cfg(any(test, feature = "testing"))]
-    kv_backend: crate::kv_backend::KvBackendRef,
 }

 impl SchemaMetadataManager {
    /// Creates a new database meta
-    #[cfg(not(any(test, feature = "testing")))]
    pub fn new(table_id_schema_cache: TableSchemaCacheRef, schema_cache: SchemaCacheRef) -> Self {
        Self {
            table_id_schema_cache,
@@ -42,20 +39,6 @@ impl SchemaMetadataManager {
        }
    }

-    /// Creates a new database meta
-    #[cfg(any(test, feature = "testing"))]
-    pub fn new(
-        kv_backend: crate::kv_backend::KvBackendRef,
-        table_id_schema_cache: TableSchemaCacheRef,
-        schema_cache: SchemaCacheRef,
-    ) -> Self {
-        Self {
-            table_id_schema_cache,
-            schema_cache,
-            kv_backend,
-        }
-    }
-
    /// Gets schema options by table id.
    pub async fn get_schema_options_by_table_id(
        &self,
@@ -80,6 +63,7 @@ impl SchemaMetadataManager {
        schema_name: &str,
        catalog_name: &str,
        schema_value: Option<crate::key::schema_name::SchemaNameValue>,
+        kv_backend: crate::kv_backend::KvBackendRef,
    ) {
        use table::metadata::{RawTableInfo, TableType};
        let value = crate::key::table_info::TableInfoValue::new(RawTableInfo {
@@ -91,19 +75,18 @@ impl SchemaMetadataManager {
            meta: Default::default(),
            table_type: TableType::Base,
        });
-        let table_info_manager =
-            crate::key::table_info::TableInfoManager::new(self.kv_backend.clone());
+        let table_info_manager = crate::key::table_info::TableInfoManager::new(kv_backend.clone());
        let (txn, _) = table_info_manager
            .build_create_txn(table_id, &value)
            .unwrap();
-        let resp = self.kv_backend.txn(txn).await.unwrap();
+        let resp = kv_backend.txn(txn).await.unwrap();
        assert!(resp.succeeded, "Failed to create table metadata");
        let key = crate::key::schema_name::SchemaNameKey {
            catalog: catalog_name,
            schema: schema_name,
        };

-        crate::key::schema_name::SchemaManager::new(self.kv_backend.clone())
+        crate::key::schema_name::SchemaManager::new(kv_backend.clone())
            .create(key, schema_value, false)
            .await
            .expect("Failed to create schema metadata");
--- a/src/common/meta/src/key/schema_name.rs
+++ b/src/common/meta/src/key/schema_name.rs
@@ -14,7 +14,6 @@

 use std::collections::HashMap;
 use std::fmt::Display;
-use std::sync::Arc;

 use common_catalog::consts::{DEFAULT_CATALOG_NAME, DEFAULT_SCHEMA_NAME};
 use common_time::DatabaseTimeToLive;
@@ -283,7 +282,7 @@ impl SchemaManager {
            self.kv_backend.clone(),
            req,
            DEFAULT_PAGE_SIZE,
-            Arc::new(schema_decoder),
+            schema_decoder,
        )
        .into_stream();

@@ -308,6 +307,7 @@ impl<'a> From<&'a SchemaName> for SchemaNameKey<'a> {

 #[cfg(test)]
 mod tests {
+    use std::sync::Arc;
    use std::time::Duration;

    use super::*;
--- a/src/common/meta/src/key/table_name.rs
+++ b/src/common/meta/src/key/table_name.rs
@@ -269,7 +269,7 @@ impl TableNameManager {
            self.kv_backend.clone(),
            req,
            DEFAULT_PAGE_SIZE,
-            Arc::new(table_decoder),
+            table_decoder,
        )
        .into_stream();

--- a/src/common/meta/src/kv_backend/postgres.rs
+++ b/src/common/meta/src/kv_backend/postgres.rs
@@ -16,6 +16,7 @@ use std::any::Any;
 use std::borrow::Cow;
 use std::sync::Arc;

+use common_telemetry::error;
 use snafu::ResultExt;
 use tokio_postgres::types::ToSql;
 use tokio_postgres::{Client, NoTls};
@@ -97,7 +98,11 @@ impl PgStore {
        let (client, conn) = tokio_postgres::connect(url, NoTls)
            .await
            .context(ConnectPostgresSnafu)?;
-        tokio::spawn(async move { conn.await.context(ConnectPostgresSnafu) });
+        tokio::spawn(async move {
+            if let Err(e) = conn.await {
+                error!(e; "connection error");
+            }
+        });
        Self::with_pg_client(client).await
    }

--- a/src/common/meta/src/range_stream.rs
+++ b/src/common/meta/src/range_stream.rs
@@ -12,8 +12,6 @@
 // See the License for the specific language governing permissions and
 // limitations under the License.

-use std::sync::Arc;
-
 use async_stream::try_stream;
 use common_telemetry::debug;
 use futures::Stream;
@@ -148,7 +146,7 @@ impl PaginationStreamFactory {
 }

 pub struct PaginationStream<T> {
-    decoder_fn: Arc<KeyValueDecoderFn<T>>,
+    decoder_fn: fn(KeyValue) -> Result<T>,
    factory: PaginationStreamFactory,
 }

@@ -158,7 +156,7 @@ impl<T> PaginationStream<T> {
        kv: KvBackendRef,
        req: RangeRequest,
        page_size: usize,
-        decoder_fn: Arc<KeyValueDecoderFn<T>>,
+        decoder_fn: fn(KeyValue) -> Result<T>,
    ) -> Self {
        Self {
            decoder_fn,
@@ -191,6 +189,7 @@ mod tests {

    use std::assert_matches::assert_matches;
    use std::collections::BTreeMap;
+    use std::sync::Arc;

    use futures::TryStreamExt;

@@ -250,7 +249,7 @@ mod tests {
                ..Default::default()
            },
            DEFAULT_PAGE_SIZE,
-            Arc::new(decoder),
+            decoder,
        )
        .into_stream();
        let kv = stream.try_collect::<Vec<_>>().await.unwrap();
@@ -290,7 +289,7 @@ mod tests {
                ..Default::default()
            },
            2,
-            Arc::new(decoder),
+            decoder,
        );
        let kv = stream
            .into_stream()
--- a/src/common/meta/src/state_store.rs
+++ b/src/common/meta/src/state_store.rs
@@ -12,8 +12,6 @@
 // See the License for the specific language governing permissions and
 // limitations under the License.

-use std::sync::Arc;
-
 use async_trait::async_trait;
 use common_error::ext::BoxedError;
 use common_procedure::error::{DeleteStatesSnafu, ListStateSnafu, PutStateSnafu};
@@ -171,7 +169,7 @@ impl StateStore for KvStateStore {
            self.kv_backend.clone(),
            req,
            self.max_num_per_range_request.unwrap_or_default(),
-            Arc::new(decode_kv),
+            decode_kv,
        )
        .into_stream();

--- a/src/common/procedure/src/local/runner.rs
+++ b/src/common/procedure/src/local/runner.rs
@@ -544,7 +544,7 @@ mod tests {
    use common_test_util::temp_dir::create_temp_dir;
    use futures_util::future::BoxFuture;
    use futures_util::FutureExt;
-    use object_store::ObjectStore;
+    use object_store::{EntryMode, ObjectStore};
    use tokio::sync::mpsc;

    use super::*;
@@ -578,7 +578,11 @@ mod tests {
    ) {
        let dir = proc_path!(procedure_store, "{procedure_id}/");
        let lister = object_store.list(&dir).await.unwrap();
-        let mut files_in_dir: Vec<_> = lister.into_iter().map(|de| de.name().to_string()).collect();
+        let mut files_in_dir: Vec<_> = lister
+            .into_iter()
+            .filter(|x| x.metadata().mode() == EntryMode::FILE)
+            .map(|de| de.name().to_string())
+            .collect();
        files_in_dir.sort_unstable();
        assert_eq!(files, files_in_dir);
    }
--- a/src/common/runtime/Cargo.toml
+++ b/src/common/runtime/Cargo.toml
@@ -39,3 +39,7 @@ tokio-util.workspace = true

 [dev-dependencies]
 tokio-test = "0.4"
+
+[target.'cfg(tokio_unstable)'.dependencies]
+tokio-metrics = { version = "0.3" }
+tokio-metrics-collector = { version = "0.2" }
--- a/src/datanode/src/datanode.rs
+++ b/src/datanode/src/datanode.rs
@@ -224,7 +224,6 @@ impl DatanodeBuilder {
            cache_registry.get().context(MissingCacheSnafu)?;

        let schema_metadata_manager = Arc::new(SchemaMetadataManager::new(
-            kv_backend.clone(),
            table_id_schema_cache,
            schema_cache,
        ));
--- a/src/datanode/src/error.rs
+++ b/src/datanode/src/error.rs
@@ -193,6 +193,14 @@ pub enum Error {
        location: Location,
    },

+    #[snafu(display("Failed to build http client"))]
+    BuildHttpClient {
+        #[snafu(implicit)]
+        location: Location,
+        #[snafu(source)]
+        error: reqwest::Error,
+    },
+
    #[snafu(display("Missing required field: {}", name))]
    MissingRequiredField {
        name: String,
@@ -406,9 +414,10 @@ impl ErrorExt for Error {
            | MissingKvBackend { .. }
            | TomlFormat { .. } => StatusCode::InvalidArguments,

-            PayloadNotExist { .. } | Unexpected { .. } | WatchAsyncTaskChange { .. } => {
-                StatusCode::Unexpected
-            }
+            PayloadNotExist { .. }
+            | Unexpected { .. }
+            | WatchAsyncTaskChange { .. }
+            | BuildHttpClient { .. } => StatusCode::Unexpected,

            AsyncTaskExecute { source, .. } => source.status_code(),

--- a/src/datanode/src/store.rs
+++ b/src/datanode/src/store.rs
@@ -28,11 +28,11 @@ use common_telemetry::{info, warn};
 use object_store::layers::{LruCacheLayer, RetryInterceptor, RetryLayer};
 use object_store::services::Fs;
 use object_store::util::{join_dir, normalize_dir, with_instrument_layers};
-use object_store::{Access, Error, HttpClient, ObjectStore, ObjectStoreBuilder, OBJECT_CACHE_DIR};
+use object_store::{Access, Error, HttpClient, ObjectStore, ObjectStoreBuilder};
 use snafu::prelude::*;

 use crate::config::{HttpClientConfig, ObjectStoreConfig, DEFAULT_OBJECT_STORE_CACHE_SIZE};
-use crate::error::{self, CreateDirSnafu, Result};
+use crate::error::{self, BuildHttpClientSnafu, CreateDirSnafu, Result};

 pub(crate) async fn new_raw_object_store(
    store: &ObjectStoreConfig,
@@ -147,12 +147,10 @@ async fn build_cache_layer(
    };

    // Enable object cache by default
-    // Set the cache_path to be `${data_home}/object_cache/read/{name}` by default
+    // Set the cache_path to be `${data_home}` by default
    // if it's not present
    if cache_path.is_none() {
-        let object_cache_path = join_dir(data_home, OBJECT_CACHE_DIR);
-        let read_cache_path = join_dir(&object_cache_path, "read");
-        let read_cache_path = join_dir(&read_cache_path, &name.to_lowercase());
+        let read_cache_path = data_home.to_string();
        tokio::fs::create_dir_all(Path::new(&read_cache_path))
            .await
            .context(CreateDirSnafu {
@@ -236,7 +234,8 @@ pub(crate) fn build_http_client(config: &HttpClientConfig) -> Result<HttpClient>
        builder.timeout(config.timeout)
    };

-    HttpClient::build(http_builder).context(error::InitBackendSnafu)
+    let client = http_builder.build().context(BuildHttpClientSnafu)?;
+    Ok(HttpClient::with(client))
 }
 struct PrintDetailedError;

--- a/src/datatypes/src/schema.rs
+++ b/src/datatypes/src/schema.rs
@@ -29,7 +29,7 @@ use crate::error::{self, DuplicateColumnSnafu, Error, ProjectArrowSchemaSnafu, R
 use crate::prelude::ConcreteDataType;
 pub use crate::schema::column_schema::{
    ColumnSchema, FulltextAnalyzer, FulltextOptions, Metadata, SkippingIndexOptions,
-    COLUMN_FULLTEXT_CHANGE_OPT_KEY_ENABLE, COLUMN_FULLTEXT_OPT_KEY_ANALYZER,
+    SkippingIndexType, COLUMN_FULLTEXT_CHANGE_OPT_KEY_ENABLE, COLUMN_FULLTEXT_OPT_KEY_ANALYZER,
    COLUMN_FULLTEXT_OPT_KEY_CASE_SENSITIVE, COLUMN_SKIPPING_INDEX_OPT_KEY_GRANULARITY,
    COLUMN_SKIPPING_INDEX_OPT_KEY_TYPE, COMMENT_KEY, FULLTEXT_KEY, INVERTED_INDEX_KEY,
    SKIPPING_INDEX_KEY, TIME_INDEX_KEY,
--- a/src/datatypes/src/schema/column_schema.rs
+++ b/src/datatypes/src/schema/column_schema.rs
@@ -123,6 +123,14 @@ impl ColumnSchema {
        self.default_constraint.as_ref()
    }

+    /// Check if the default constraint is a impure function.
+    pub fn is_default_impure(&self) -> bool {
+        self.default_constraint
+            .as_ref()
+            .map(|c| c.is_function())
+            .unwrap_or(false)
+    }
+
    #[inline]
    pub fn metadata(&self) -> &Metadata {
        &self.metadata
@@ -283,6 +291,15 @@ impl ColumnSchema {
        }
    }

+    /// Creates an impure default value for this column, only if it have a impure default constraint.
+    /// Otherwise, returns `Ok(None)`.
+    pub fn create_impure_default(&self) -> Result<Option<Value>> {
+        match &self.default_constraint {
+            Some(c) => c.create_impure_default(&self.data_type),
+            None => Ok(None),
+        }
+    }
+
    /// Retrieves the fulltext options for the column.
    pub fn fulltext_options(&self) -> Result<Option<FulltextOptions>> {
        match self.metadata.get(FULLTEXT_KEY) {
@@ -543,7 +560,7 @@ pub struct SkippingIndexOptions {
    pub granularity: u32,
    /// The type of the skip index.
    #[serde(default)]
-    pub index_type: SkipIndexType,
+    pub index_type: SkippingIndexType,
 }

 impl fmt::Display for SkippingIndexOptions {
@@ -556,15 +573,15 @@ impl fmt::Display for SkippingIndexOptions {

 /// Skip index types.
 #[derive(Debug, Default, Clone, PartialEq, Eq, Serialize, Deserialize, Visit, VisitMut)]
-pub enum SkipIndexType {
+pub enum SkippingIndexType {
    #[default]
    BloomFilter,
 }

-impl fmt::Display for SkipIndexType {
+impl fmt::Display for SkippingIndexType {
    fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result {
        match self {
-            SkipIndexType::BloomFilter => write!(f, "BLOOM"),
+            SkippingIndexType::BloomFilter => write!(f, "BLOOM"),
        }
    }
 }
@@ -587,7 +604,7 @@ impl TryFrom<HashMap<String, String>> for SkippingIndexOptions {
        // Parse index type with default value BloomFilter
        let index_type = match options.get(COLUMN_SKIPPING_INDEX_OPT_KEY_TYPE) {
            Some(typ) => match typ.to_ascii_uppercase().as_str() {
-                "BLOOM" => SkipIndexType::BloomFilter,
+                "BLOOM" => SkippingIndexType::BloomFilter,
                _ => {
                    return error::InvalidSkippingIndexOptionSnafu {
                        msg: format!("Invalid index type: {typ}, expected: 'BLOOM'"),
@@ -595,7 +612,7 @@ impl TryFrom<HashMap<String, String>> for SkippingIndexOptions {
                    .fail();
                }
            },
-            None => SkipIndexType::default(),
+            None => SkippingIndexType::default(),
        };

        Ok(SkippingIndexOptions {
--- a/src/datatypes/src/schema/constraint.rs
+++ b/src/datatypes/src/schema/constraint.rs
@@ -178,12 +178,63 @@ impl ColumnDefaultConstraint {
        }
    }

+    /// Only create default vector if it's impure, i.e., it's a function.
+    ///
+    /// This helps to delay creating constant default values to mito engine while also keeps impure default have consistent values
+    pub fn create_impure_default_vector(
+        &self,
+        data_type: &ConcreteDataType,
+        num_rows: usize,
+    ) -> Result<Option<VectorRef>> {
+        assert!(num_rows > 0);
+
+        match self {
+            ColumnDefaultConstraint::Function(expr) => {
+                // Functions should also ensure its return value is not null when
+                // is_nullable is true.
+                match &expr[..] {
+                    // TODO(dennis): we only supports current_timestamp right now,
+                    //   it's better to use a expression framework in future.
+                    CURRENT_TIMESTAMP | CURRENT_TIMESTAMP_FN | NOW_FN => {
+                        create_current_timestamp_vector(data_type, num_rows).map(Some)
+                    }
+                    _ => error::UnsupportedDefaultExprSnafu { expr }.fail(),
+                }
+            }
+            ColumnDefaultConstraint::Value(_) => Ok(None),
+        }
+    }
+
+    /// Only create default value if it's impure, i.e., it's a function.
+    ///
+    /// This helps to delay creating constant default values to mito engine while also keeps impure default have consistent values
+    pub fn create_impure_default(&self, data_type: &ConcreteDataType) -> Result<Option<Value>> {
+        match self {
+            ColumnDefaultConstraint::Function(expr) => {
+                // Functions should also ensure its return value is not null when
+                // is_nullable is true.
+                match &expr[..] {
+                    CURRENT_TIMESTAMP | CURRENT_TIMESTAMP_FN | NOW_FN => {
+                        create_current_timestamp(data_type).map(Some)
+                    }
+                    _ => error::UnsupportedDefaultExprSnafu { expr }.fail(),
+                }
+            }
+            ColumnDefaultConstraint::Value(_) => Ok(None),
+        }
+    }
+
    /// Returns true if this constraint might creates NULL.
    fn maybe_null(&self) -> bool {
        // Once we support more functions, we may return true if given function
        // could return null.
        matches!(self, ColumnDefaultConstraint::Value(Value::Null))
    }
+
+    /// Returns true if this constraint is a function.
+    pub fn is_function(&self) -> bool {
+        matches!(self, ColumnDefaultConstraint::Function(_))
+    }
 }

 fn create_current_timestamp(data_type: &ConcreteDataType) -> Result<Value> {
--- a/src/file-engine/src/manifest.rs
+++ b/src/file-engine/src/manifest.rs
@@ -46,7 +46,7 @@ impl FileRegionManifest {
    pub async fn store(&self, region_dir: &str, object_store: &ObjectStore) -> Result<()> {
        let path = &region_manifest_path(region_dir);
        let exist = object_store
-            .is_exist(path)
+            .exists(path)
            .await
            .context(CheckObjectSnafu { path })?;
        ensure!(!exist, ManifestExistsSnafu { path });
--- a/src/file-engine/src/region.rs
+++ b/src/file-engine/src/region.rs
@@ -130,7 +130,7 @@ mod tests {
        assert_eq!(region.metadata.primary_key, vec![1]);

        assert!(object_store
-            .is_exist("create_region_dir/manifest/_file_manifest")
+            .exists("create_region_dir/manifest/_file_manifest")
            .await
            .unwrap());

@@ -198,13 +198,13 @@ mod tests {
            .unwrap();

        assert!(object_store
-            .is_exist("drop_region_dir/manifest/_file_manifest")
+            .exists("drop_region_dir/manifest/_file_manifest")
            .await
            .unwrap());

        FileRegion::drop(&region, &object_store).await.unwrap();
        assert!(!object_store
-            .is_exist("drop_region_dir/manifest/_file_manifest")
+            .exists("drop_region_dir/manifest/_file_manifest")
            .await
            .unwrap());

--- a/src/flow/Cargo.toml
+++ b/src/flow/Cargo.toml
@@ -45,6 +45,7 @@ get-size2 = "0.1.2"
 greptime-proto.workspace = true
 # This fork of hydroflow is simply for keeping our dependency in our org, and pin the version
 # otherwise it is the same with upstream repo
+http.workspace = true
 hydroflow = { git = "https://github.com/GreptimeTeam/hydroflow.git", branch = "main" }
 itertools.workspace = true
 lazy_static.workspace = true
--- a/src/flow/src/adapter.rs
+++ b/src/flow/src/adapter.rs
@@ -30,7 +30,7 @@ use common_telemetry::{debug, info, trace};
 use datatypes::schema::ColumnSchema;
 use datatypes::value::Value;
 use greptime_proto::v1;
-use itertools::Itertools;
+use itertools::{EitherOrBoth, Itertools};
 use meta_client::MetaClientOptions;
 use query::QueryEngine;
 use serde::{Deserialize, Serialize};
@@ -45,18 +45,15 @@ use tokio::sync::broadcast::error::TryRecvError;
 use tokio::sync::{broadcast, watch, Mutex, RwLock};

 pub(crate) use crate::adapter::node_context::FlownodeContext;
-use crate::adapter::table_source::TableSource;
-use crate::adapter::util::column_schemas_to_proto;
+use crate::adapter::table_source::ManagedTableSource;
+use crate::adapter::util::relation_desc_to_column_schemas_with_fallback;
 use crate::adapter::worker::{create_worker, Worker, WorkerHandle};
 use crate::compute::ErrCollector;
 use crate::df_optimizer::sql_to_flow_plan;
-use crate::error::{
-    EvalSnafu, ExternalSnafu, FlowAlreadyExistSnafu, InternalSnafu, TableNotFoundSnafu,
-    UnexpectedSnafu,
-};
-use crate::expr::{Batch, GlobalId};
-use crate::metrics::{METRIC_FLOW_INSERT_ELAPSED, METRIC_FLOW_RUN_INTERVAL_MS};
-use crate::repr::{self, DiffRow, Row, BATCH_SIZE};
+use crate::error::{EvalSnafu, ExternalSnafu, InternalSnafu, InvalidQuerySnafu, UnexpectedSnafu};
+use crate::expr::Batch;
+use crate::metrics::{METRIC_FLOW_INSERT_ELAPSED, METRIC_FLOW_ROWS, METRIC_FLOW_RUN_INTERVAL_MS};
+use crate::repr::{self, DiffRow, RelationDesc, Row, BATCH_SIZE};

 mod flownode_impl;
 mod parse_expr;
@@ -67,7 +64,7 @@ mod util;
 mod worker;

 pub(crate) mod node_context;
-mod table_source;
+pub(crate) mod table_source;

 use crate::error::Error;
 use crate::utils::StateReportHandler;
@@ -127,7 +124,7 @@ pub struct FlowWorkerManager {
    /// The query engine that will be used to parse the query and convert it to a dataflow plan
    pub query_engine: Arc<dyn QueryEngine>,
    /// Getting table name and table schema from table info manager
-    table_info_source: TableSource,
+    table_info_source: ManagedTableSource,
    frontend_invoker: RwLock<Option<FrontendInvoker>>,
    /// contains mapping from table name to global id, and table schema
    node_context: RwLock<FlownodeContext>,
@@ -156,11 +153,11 @@ impl FlowWorkerManager {
        query_engine: Arc<dyn QueryEngine>,
        table_meta: TableMetadataManagerRef,
    ) -> Self {
-        let srv_map = TableSource::new(
+        let srv_map = ManagedTableSource::new(
            table_meta.table_info_manager().clone(),
            table_meta.table_name_manager().clone(),
        );
-        let node_context = FlownodeContext::default();
+        let node_context = FlownodeContext::new(Box::new(srv_map.clone()) as _);
        let tick_manager = FlowTickManager::new();
        let worker_handles = Vec::new();
        FlowWorkerManager {
@@ -245,16 +242,26 @@ impl FlowWorkerManager {
            let (catalog, schema) = (table_name[0].clone(), table_name[1].clone());
            let ctx = Arc::new(QueryContext::with(&catalog, &schema));

-            let (is_ts_placeholder, proto_schema) =
-                self.try_fetch_or_create_table(&table_name).await?;
+            let (is_ts_placeholder, proto_schema) = self
+                .try_fetch_existing_table(&table_name)
+                .await?
+                .context(UnexpectedSnafu {
+                    reason: format!("Table not found: {}", table_name.join(".")),
+                })?;
            let schema_len = proto_schema.len();

+            let total_rows = reqs.iter().map(|r| r.len()).sum::<usize>();
            trace!(
                "Sending {} writeback requests to table {}, reqs total rows={}",
                reqs.len(),
                table_name.join("."),
                reqs.iter().map(|r| r.len()).sum::<usize>()
            );
+
+            METRIC_FLOW_ROWS
+                .with_label_values(&["out"])
+                .inc_by(total_rows as u64);
+
            let now = self.tick_manager.tick();
            for req in reqs {
                match req {
@@ -390,16 +397,14 @@ impl FlowWorkerManager {
        Ok(output)
    }

-    /// Fetch table info or create table from flow's schema if not exist
-    async fn try_fetch_or_create_table(
+    /// Fetch table schema and primary key from table info source, if table not exist return None
+    async fn fetch_table_pk_schema(
        &self,
        table_name: &TableName,
-    ) -> Result<(bool, Vec<api::v1::ColumnSchema>), Error> {
-        // TODO(discord9): instead of auto build table from request schema, actually build table
-        // before `create flow` to be able to assign pk and ts etc.
-        let (primary_keys, schema, is_ts_placeholder) = if let Some(table_id) = self
+    ) -> Result<Option<(Vec<String>, Option<usize>, Vec<ColumnSchema>)>, Error> {
+        if let Some(table_id) = self
            .table_info_source
-            .get_table_id_from_name(table_name)
+            .get_opt_table_id_from_name(table_name)
            .await?
        {
            let table_info = self
@@ -414,97 +419,64 @@ impl FlowWorkerManager {
                .map(|i| meta.schema.column_schemas[i].name.clone())
                .collect_vec();
            let schema = meta.schema.column_schemas;
-            // check if the last column is the auto created timestamp column, hence the table is auto created from
-            // flow's plan type
-            let is_auto_create = {
-                let correct_name = schema
-                    .last()
-                    .map(|s| s.name == AUTO_CREATED_PLACEHOLDER_TS_COL)
-                    .unwrap_or(false);
-                let correct_time_index = meta.schema.timestamp_index == Some(schema.len() - 1);
-                correct_name && correct_time_index
-            };
-            (primary_keys, schema, is_auto_create)
+            let time_index = meta.schema.timestamp_index;
+            Ok(Some((primary_keys, time_index, schema)))
        } else {
-            // TODO(discord9): condiser remove buggy auto create by schema
+            Ok(None)
+        }
+    }

-            let node_ctx = self.node_context.read().await;
-            let gid: GlobalId = node_ctx
-                .table_repr
-                .get_by_name(table_name)
-                .map(|x| x.1)
-                .unwrap();
-            let schema = node_ctx
-                .schema
-                .get(&gid)
-                .with_context(|| TableNotFoundSnafu {
-                    name: format!("Table name = {:?}", table_name),
-                })?
-                .clone();
-            // TODO(discord9): use default key from schema
-            let primary_keys = schema
-                .typ()
-                .keys
-                .first()
-                .map(|v| {
-                    v.column_indices
-                        .iter()
-                        .map(|i| {
-                            schema
-                                .get_name(*i)
-                                .clone()
-                                .unwrap_or_else(|| format!("col_{i}"))
-                        })
-                        .collect_vec()
-                })
-                .unwrap_or_default();
-            let update_at = ColumnSchema::new(
-                UPDATE_AT_TS_COL,
+    /// return (primary keys, schema and if the table have a placeholder timestamp column)
+    /// schema of the table comes from flow's output plan
+    ///
+    /// adjust to add `update_at` column and ts placeholder if needed
+    async fn adjust_auto_created_table_schema(
+        &self,
+        schema: &RelationDesc,
+    ) -> Result<(Vec<String>, Vec<ColumnSchema>, bool), Error> {
+        // TODO(discord9): condiser remove buggy auto create by schema
+
+        // TODO(discord9): use default key from schema
+        let primary_keys = schema
+            .typ()
+            .keys
+            .first()
+            .map(|v| {
+                v.column_indices
+                    .iter()
+                    .map(|i| {
+                        schema
+                            .get_name(*i)
+                            .clone()
+                            .unwrap_or_else(|| format!("col_{i}"))
+                    })
+                    .collect_vec()
+            })
+            .unwrap_or_default();
+        let update_at = ColumnSchema::new(
+            UPDATE_AT_TS_COL,
+            ConcreteDataType::timestamp_millisecond_datatype(),
+            true,
+        );
+
+        let original_schema = relation_desc_to_column_schemas_with_fallback(schema);
+
+        let mut with_auto_added_col = original_schema.clone();
+        with_auto_added_col.push(update_at);
+
+        // if no time index, add one as placeholder
+        let no_time_index = schema.typ().time_index.is_none();
+        if no_time_index {
+            let ts_col = ColumnSchema::new(
+                AUTO_CREATED_PLACEHOLDER_TS_COL,
                ConcreteDataType::timestamp_millisecond_datatype(),
                true,
-            );
+            )
+            .with_time_index(true);
+            with_auto_added_col.push(ts_col);
+        }

-            let original_schema = schema
-                .typ()
-                .column_types
-                .clone()
-                .into_iter()
-                .enumerate()
-                .map(|(idx, typ)| {
-                    let name = schema
-                        .names
-                        .get(idx)
-                        .cloned()
-                        .flatten()
-                        .unwrap_or(format!("col_{}", idx));
-                    let ret = ColumnSchema::new(name, typ.scalar_type, typ.nullable);
-                    if schema.typ().time_index == Some(idx) {
-                        ret.with_time_index(true)
-                    } else {
-                        ret
-                    }
-                })
-                .collect_vec();
-
-            let mut with_auto_added_col = original_schema.clone();
-            with_auto_added_col.push(update_at);
-
-            // if no time index, add one as placeholder
-            let no_time_index = schema.typ().time_index.is_none();
-            if no_time_index {
-                let ts_col = ColumnSchema::new(
-                    AUTO_CREATED_PLACEHOLDER_TS_COL,
-                    ConcreteDataType::timestamp_millisecond_datatype(),
-                    true,
-                )
-                .with_time_index(true);
-                with_auto_added_col.push(ts_col);
-            }
-
-            (primary_keys, with_auto_added_col, no_time_index)
-        };
-        let proto_schema = column_schemas_to_proto(schema, &primary_keys)?;
-        Ok((is_ts_placeholder, proto_schema))
+        Ok((primary_keys, with_auto_added_col, no_time_index))
    }
 }

@@ -752,43 +724,6 @@ impl FlowWorkerManager {
            query_ctx,
        } = args;

-        let already_exist = {
-            let mut flag = false;
-
-            // check if the task already exists
-            for handle in self.worker_handles.iter() {
-                if handle.lock().await.contains_flow(flow_id).await? {
-                    flag = true;
-                    break;
-                }
-            }
-            flag
-        };
-        match (create_if_not_exists, or_replace, already_exist) {
-            // do replace
-            (_, true, true) => {
-                info!("Replacing flow with id={}", flow_id);
-                self.remove_flow(flow_id).await?;
-            }
-            (false, false, true) => FlowAlreadyExistSnafu { id: flow_id }.fail()?,
-            // do nothing if exists
-            (true, false, true) => {
-                info!("Flow with id={} already exists, do nothing", flow_id);
-                return Ok(None);
-            }
-            // create if not exists
-            (_, _, false) => (),
-        }
-
-        if create_if_not_exists {
-            // check if the task already exists
-            for handle in self.worker_handles.iter() {
-                if handle.lock().await.contains_flow(flow_id).await? {
-                    return Ok(None);
-                }
-            }
-        }
-
        let mut node_ctx = self.node_context.write().await;
        // assign global id to source and sink table
        for source in &source_table_ids {
@@ -807,7 +742,67 @@ impl FlowWorkerManager {
        let flow_plan = sql_to_flow_plan(&mut node_ctx, &self.query_engine, &sql).await?;

        debug!("Flow {:?}'s Plan is {:?}", flow_id, flow_plan);
-        node_ctx.assign_table_schema(&sink_table_name, flow_plan.schema.clone())?;
+
+        // check schema against actual table schema if exists
+        // if not exist create sink table immediately
+        if let Some((_, _, real_schema)) = self.fetch_table_pk_schema(&sink_table_name).await? {
+            let auto_schema = relation_desc_to_column_schemas_with_fallback(&flow_plan.schema);
+
+            // for column schema, only `data_type` need to be check for equality
+            // since one can omit flow's column name when write flow query
+            // print a user friendly error message about mismatch and how to correct them
+            for (idx, zipped) in auto_schema
+                .iter()
+                .zip_longest(real_schema.iter())
+                .enumerate()
+            {
+                match zipped {
+                    EitherOrBoth::Both(auto, real) => {
+                        if auto.data_type != real.data_type {
+                            InvalidQuerySnafu {
+                                    reason: format!(
+                                        "Column {}(name is '{}', flow inferred name is '{}')'s data type mismatch, expect {:?} got {:?}",
+                                        idx,
+                                        real.name,
+                                        auto.name,
+                                        real.data_type,
+                                        auto.data_type
+                                    ),
+                                }
+                                .fail()?;
+                        }
+                    }
+                    EitherOrBoth::Right(real) if real.data_type.is_timestamp() => {
+                        // if table is auto created, the last one or two column should be timestamp(update at and ts placeholder)
+                        continue;
+                    }
+                    _ => InvalidQuerySnafu {
+                        reason: format!(
+                            "schema length mismatched, expected {} found {}",
+                            real_schema.len(),
+                            auto_schema.len()
+                        ),
+                    }
+                    .fail()?,
+                }
+            }
+        } else {
+            // assign inferred schema to sink table
+            // create sink table
+            let did_create = self
+                .create_table_from_relation(
+                    &format!("flow-id={flow_id}"),
+                    &sink_table_name,
+                    &flow_plan.schema,
+                )
+                .await?;
+            if !did_create {
+                UnexpectedSnafu {
+                    reason: format!("Failed to create table {:?}", sink_table_name),
+                }
+                .fail()?;
+            }
+        }

        let _ = comment;
        let _ = flow_options;
@@ -842,9 +837,11 @@ impl FlowWorkerManager {
            source_ids,
            src_recvs: source_receivers,
            expire_after,
+            or_replace,
            create_if_not_exists,
            err_collector,
        };
+
        handle.create_flow(create_request).await?;
        info!("Successfully create flow with id={}", flow_id);
        Ok(Some(flow_id))
--- a/src/flow/src/adapter/flownode_impl.rs
+++ b/src/flow/src/adapter/flownode_impl.rs
@@ -24,21 +24,26 @@ use common_error::ext::BoxedError;
 use common_meta::error::{ExternalSnafu, Result, UnexpectedSnafu};
 use common_meta::node_manager::Flownode;
 use common_telemetry::{debug, trace};
+use datatypes::value::Value;
 use itertools::Itertools;
-use snafu::{OptionExt, ResultExt};
+use snafu::{IntoError, OptionExt, ResultExt};
 use store_api::storage::RegionId;

-use super::util::from_proto_to_data_type;
 use crate::adapter::{CreateFlowArgs, FlowWorkerManager};
-use crate::error::InternalSnafu;
+use crate::error::{CreateFlowSnafu, InsertIntoFlowSnafu, InternalSnafu};
 use crate::metrics::METRIC_FLOW_TASK_COUNT;
 use crate::repr::{self, DiffRow};

-fn to_meta_err(err: crate::error::Error) -> common_meta::error::Error {
-    // TODO(discord9): refactor this
-    Err::<(), _>(BoxedError::new(err))
-        .with_context(|_| ExternalSnafu)
-        .unwrap_err()
+/// return a function to convert `crate::error::Error` to `common_meta::error::Error`
+fn to_meta_err(
+    location: snafu::Location,
+) -> impl FnOnce(crate::error::Error) -> common_meta::error::Error {
+    move |err: crate::error::Error| -> common_meta::error::Error {
+        common_meta::error::Error::External {
+            location,
+            source: BoxedError::new(err),
+        }
+    }
 }

 #[async_trait::async_trait]
@@ -75,11 +80,16 @@ impl Flownode for FlowWorkerManager {
                    or_replace,
                    expire_after,
                    comment: Some(comment),
-                    sql,
+                    sql: sql.clone(),
                    flow_options,
                    query_ctx,
                };
-                let ret = self.create_flow(args).await.map_err(to_meta_err)?;
+                let ret = self
+                    .create_flow(args)
+                    .await
+                    .map_err(BoxedError::new)
+                    .with_context(|_| CreateFlowSnafu { sql: sql.clone() })
+                    .map_err(to_meta_err(snafu::location!()))?;
                METRIC_FLOW_TASK_COUNT.inc();
                Ok(FlowResponse {
                    affected_flows: ret
@@ -94,7 +104,7 @@ impl Flownode for FlowWorkerManager {
            })) => {
                self.remove_flow(flow_id.id as u64)
                    .await
-                    .map_err(to_meta_err)?;
+                    .map_err(to_meta_err(snafu::location!()))?;
                METRIC_FLOW_TASK_COUNT.dec();
                Ok(Default::default())
            }
@@ -112,9 +122,15 @@ impl Flownode for FlowWorkerManager {
                    .await
                    .flush_all_sender()
                    .await
-                    .map_err(to_meta_err)?;
-                let rows_send = self.run_available(true).await.map_err(to_meta_err)?;
-                let row = self.send_writeback_requests().await.map_err(to_meta_err)?;
+                    .map_err(to_meta_err(snafu::location!()))?;
+                let rows_send = self
+                    .run_available(true)
+                    .await
+                    .map_err(to_meta_err(snafu::location!()))?;
+                let row = self
+                    .send_writeback_requests()
+                    .await
+                    .map_err(to_meta_err(snafu::location!()))?;

                debug!(
                    "Done to flush flow_id={:?} with {} input rows flushed, {} rows sended and {} output rows flushed",
@@ -138,7 +154,7 @@ impl Flownode for FlowWorkerManager {
    }

    async fn handle_inserts(&self, request: InsertRequests) -> Result<FlowResponse> {
-        // using try_read makesure two things:
+        // using try_read to ensure two things:
        // 1. flush wouldn't happen until inserts before it is inserted
        // 2. inserts happening concurrently with flush wouldn't be block by flush
        let _flush_lock = self.flush_lock.try_read();
@@ -154,17 +170,41 @@ impl Flownode for FlowWorkerManager {
            // TODO(discord9): reconsider time assignment mechanism
            let now = self.tick_manager.tick();

-            let fetch_order = {
+            let (table_types, fetch_order) = {
                let ctx = self.node_context.read().await;
-                let table_col_names = ctx
-                    .table_repr
-                    .get_by_table_id(&table_id)
-                    .map(|r| r.1)
-                    .and_then(|id| ctx.schema.get(&id))
-                    .map(|desc| &desc.names)
-                    .context(UnexpectedSnafu {
-                        err_msg: format!("Table not found: {}", table_id),
-                    })?;
+
+                // TODO(discord9): also check schema version so that altered table can be reported
+                let table_schema = ctx
+                    .table_source
+                    .table_from_id(&table_id)
+                    .await
+                    .map_err(to_meta_err(snafu::location!()))?;
+                let default_vals = table_schema
+                    .default_values
+                    .iter()
+                    .zip(table_schema.relation_desc.typ().column_types.iter())
+                    .map(|(v, ty)| {
+                        v.as_ref().and_then(|v| {
+                            match v.create_default(ty.scalar_type(), ty.nullable()) {
+                                Ok(v) => Some(v),
+                                Err(err) => {
+                                    common_telemetry::error!(err; "Failed to create default value");
+                                    None
+                                }
+                            }
+                        })
+                    })
+                    .collect_vec();
+
+                let table_types = table_schema
+                    .relation_desc
+                    .typ()
+                    .column_types
+                    .clone()
+                    .into_iter()
+                    .map(|t| t.scalar_type)
+                    .collect_vec();
+                let table_col_names = table_schema.relation_desc.names;
                let table_col_names = table_col_names
                    .iter().enumerate()
                    .map(|(idx,name)| match name {
@@ -181,44 +221,80 @@ impl Flownode for FlowWorkerManager {
                        .enumerate()
                        .map(|(i, name)| (&name.column_name, i)),
                );
-                let fetch_order: Vec<usize> = table_col_names
+
+                let fetch_order: Vec<FetchFromRow> = table_col_names
                    .iter()
-                    .map(|names| {
-                        name_to_col.get(names).copied().context(UnexpectedSnafu {
-                            err_msg: format!("Column not found: {}", names),
-                        })
+                    .zip(default_vals.into_iter())
+                    .map(|(col_name, col_default_val)| {
+                        name_to_col
+                            .get(col_name)
+                            .copied()
+                            .map(FetchFromRow::Idx)
+                            .or_else(|| col_default_val.clone().map(FetchFromRow::Default))
+                            .with_context(|| UnexpectedSnafu {
+                                err_msg: format!(
+                                    "Column not found: {}, default_value: {:?}",
+                                    col_name, col_default_val
+                                ),
+                            })
                    })
                    .try_collect()?;
-                if !fetch_order.iter().enumerate().all(|(i, &v)| i == v) {
-                    trace!("Reordering columns: {:?}", fetch_order)
-                }
-                fetch_order
+
+                trace!("Reordering columns: {:?}", fetch_order);
+                (table_types, fetch_order)
            };

+            // TODO(discord9): use column instead of row
            let rows: Vec<DiffRow> = rows_proto
                .into_iter()
                .map(|r| {
                    let r = repr::Row::from(r);
-                    let reordered = fetch_order
-                        .iter()
-                        .map(|&i| r.inner[i].clone())
-                        .collect_vec();
+                    let reordered = fetch_order.iter().map(|i| i.fetch(&r)).collect_vec();
                    repr::Row::new(reordered)
                })
                .map(|r| (r, now, 1))
                .collect_vec();
-            let batch_datatypes = insert_schema
-                .iter()
-                .map(from_proto_to_data_type)
-                .collect::<std::result::Result<Vec<_>, _>>()
-                .map_err(to_meta_err)?;
-            self.handle_write_request(region_id.into(), rows, &batch_datatypes)
+            if let Err(err) = self
+                .handle_write_request(region_id.into(), rows, &table_types)
                .await
-                .map_err(|err| {
-                    common_telemetry::error!(err;"Failed to handle write request");
-                    to_meta_err(err)
-                })?;
+            {
+                let err = BoxedError::new(err);
+                let flow_ids = self
+                    .node_context
+                    .read()
+                    .await
+                    .get_flow_ids(table_id)
+                    .into_iter()
+                    .flatten()
+                    .cloned()
+                    .collect_vec();
+                let err = InsertIntoFlowSnafu {
+                    region_id,
+                    flow_ids,
+                }
+                .into_error(err);
+                common_telemetry::error!(err; "Failed to handle write request");
+                let err = to_meta_err(snafu::location!())(err);
+                return Err(err);
+            }
        }
        Ok(Default::default())
    }
 }
+
+/// Simple helper enum for fetching value from row with default value
+#[derive(Debug, Clone)]
+enum FetchFromRow {
+    Idx(usize),
+    Default(Value),
+}
+
+impl FetchFromRow {
+    /// Panic if idx is out of bound
+    fn fetch(&self, row: &repr::Row) -> Value {
+        match self {
+            FetchFromRow::Idx(idx) => row.get(*idx).unwrap().clone(),
+            FetchFromRow::Default(v) => v.clone(),
+        }
+    }
+}
--- a/src/flow/src/adapter/node_context.rs
+++ b/src/flow/src/adapter/node_context.rs
@@ -25,7 +25,8 @@ use snafu::{OptionExt, ResultExt};
 use table::metadata::TableId;
 use tokio::sync::{broadcast, mpsc, RwLock};

-use crate::adapter::{FlowId, TableName, TableSource};
+use crate::adapter::table_source::FlowTableSource;
+use crate::adapter::{FlowId, ManagedTableSource, TableName};
 use crate::error::{Error, EvalSnafu, TableNotFoundSnafu};
 use crate::expr::error::InternalSnafu;
 use crate::expr::{Batch, GlobalId};
@@ -33,7 +34,7 @@ use crate::metrics::METRIC_FLOW_INPUT_BUF_SIZE;
 use crate::repr::{DiffRow, RelationDesc, BATCH_SIZE, BROADCAST_CAP, SEND_BUF_CAP};

 /// A context that holds the information of the dataflow
-#[derive(Default, Debug)]
+#[derive(Debug)]
 pub struct FlownodeContext {
    /// mapping from source table to tasks, useful for schedule which task to run when a source table is updated
    pub source_to_tasks: BTreeMap<TableId, BTreeSet<FlowId>>,
@@ -50,13 +51,32 @@ pub struct FlownodeContext {
    /// note that the sink receiver should only have one, and we are using broadcast as mpsc channel here
    pub sink_receiver:
        BTreeMap<TableName, (mpsc::UnboundedSender<Batch>, mpsc::UnboundedReceiver<Batch>)>,
-    /// the schema of the table, query from metasrv or inferred from TypedPlan
-    pub schema: HashMap<GlobalId, RelationDesc>,
+    /// can query the schema of the table source, from metasrv with local cache
+    pub table_source: Box<dyn FlowTableSource>,
    /// All the tables that have been registered in the worker
    pub table_repr: IdToNameMap,
    pub query_context: Option<Arc<QueryContext>>,
 }

+impl FlownodeContext {
+    pub fn new(table_source: Box<dyn FlowTableSource>) -> Self {
+        Self {
+            source_to_tasks: Default::default(),
+            flow_to_sink: Default::default(),
+            sink_to_flow: Default::default(),
+            source_sender: Default::default(),
+            sink_receiver: Default::default(),
+            table_source,
+            table_repr: Default::default(),
+            query_context: Default::default(),
+        }
+    }
+
+    pub fn get_flow_ids(&self, table_id: TableId) -> Option<&BTreeSet<FlowId>> {
+        self.source_to_tasks.get(&table_id)
+    }
+}
+
 /// a simple broadcast sender with backpressure, bounded capacity and blocking on send when send buf is full
 /// note that it wouldn't evict old data, so it's possible to block forever if the receiver is slow
 ///
@@ -284,7 +304,7 @@ impl FlownodeContext {
    /// Retrieves a GlobalId and table schema representing a table previously registered by calling the [register_table] function.
    ///
    /// Returns an error if no table has been registered with the provided names
-    pub fn table(&self, name: &TableName) -> Result<(GlobalId, RelationDesc), Error> {
+    pub async fn table(&self, name: &TableName) -> Result<(GlobalId, RelationDesc), Error> {
        let id = self
            .table_repr
            .get_by_name(name)
@@ -292,14 +312,8 @@ impl FlownodeContext {
            .with_context(|| TableNotFoundSnafu {
                name: name.join("."),
            })?;
-        let schema = self
-            .schema
-            .get(&id)
-            .cloned()
-            .with_context(|| TableNotFoundSnafu {
-                name: name.join("."),
-            })?;
-        Ok((id, schema))
+        let schema = self.table_source.table(name).await?;
+        Ok((id, schema.relation_desc))
    }

    /// Assign a global id to a table, if already assigned, return the existing global id
@@ -312,7 +326,7 @@ impl FlownodeContext {
    /// merely creating a mapping from table id to global id
    pub async fn assign_global_id_to_table(
        &mut self,
-        srv_map: &TableSource,
+        srv_map: &ManagedTableSource,
        mut table_name: Option<TableName>,
        table_id: Option<TableId>,
    ) -> Result<GlobalId, Error> {
@@ -331,36 +345,18 @@ impl FlownodeContext {
        } else {
            let global_id = self.new_global_id();

+            // table id is Some meaning db must have created the table
            if let Some(table_id) = table_id {
-                let (known_table_name, schema) = srv_map.get_table_name_schema(&table_id).await?;
+                let known_table_name = srv_map.get_table_name(&table_id).await?;
                table_name = table_name.or(Some(known_table_name));
-                self.schema.insert(global_id, schema);
-            } // if we don't have table id, it means database havn't assign one yet or we don't need it
+            } // if we don't have table id, it means database haven't assign one yet or we don't need it

+            // still update the mapping with new global id
            self.table_repr.insert(table_name, table_id, global_id);
            Ok(global_id)
        }
    }

-    /// Assign a schema to a table
-    ///
-    pub fn assign_table_schema(
-        &mut self,
-        table_name: &TableName,
-        schema: RelationDesc,
-    ) -> Result<(), Error> {
-        let gid = self
-            .table_repr
-            .get_by_name(table_name)
-            .map(|(_, gid)| gid)
-            .context(TableNotFoundSnafu {
-                name: format!("Table not found: {:?} in flownode cache", table_name),
-            })?;
-
-        self.schema.insert(gid, schema);
-        Ok(())
-    }
-
    /// Get a new global id
    pub fn new_global_id(&self) -> GlobalId {
        GlobalId::User(self.table_repr.global_id_to_name_id.len() as u64)
--- a/src/flow/src/adapter/table_source.rs
+++ b/src/flow/src/adapter/table_source.rs
@@ -17,25 +17,94 @@
 use common_error::ext::BoxedError;
 use common_meta::key::table_info::{TableInfoManager, TableInfoValue};
 use common_meta::key::table_name::{TableNameKey, TableNameManager};
+use datatypes::schema::ColumnDefaultConstraint;
+use serde::{Deserialize, Serialize};
 use snafu::{OptionExt, ResultExt};
 use table::metadata::TableId;

+use crate::adapter::util::table_info_value_to_relation_desc;
 use crate::adapter::TableName;
 use crate::error::{
    Error, ExternalSnafu, TableNotFoundMetaSnafu, TableNotFoundSnafu, UnexpectedSnafu,
 };
-use crate::repr::{self, ColumnType, RelationDesc, RelationType};
+use crate::repr::RelationDesc;

-/// mapping of table name <-> table id should be query from tableinfo manager
-pub struct TableSource {
+/// Table description, include relation desc and default values, which is the minimal information flow needed for table
+#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)]
+pub struct TableDesc {
+    pub relation_desc: RelationDesc,
+    pub default_values: Vec<Option<ColumnDefaultConstraint>>,
+}
+
+impl TableDesc {
+    pub fn new(
+        relation_desc: RelationDesc,
+        default_values: Vec<Option<ColumnDefaultConstraint>>,
+    ) -> Self {
+        Self {
+            relation_desc,
+            default_values,
+        }
+    }
+
+    pub fn new_no_default(relation_desc: RelationDesc) -> Self {
+        Self {
+            relation_desc,
+            default_values: vec![],
+        }
+    }
+}
+
+/// Table source but for flow, provide table schema by table name/id
+#[async_trait::async_trait]
+pub trait FlowTableSource: Send + Sync + std::fmt::Debug {
+    async fn table_name_from_id(&self, table_id: &TableId) -> Result<TableName, Error>;
+    async fn table_id_from_name(&self, name: &TableName) -> Result<TableId, Error>;
+
+    /// Get the table schema by table name
+    async fn table(&self, name: &TableName) -> Result<TableDesc, Error> {
+        let id = self.table_id_from_name(name).await?;
+        self.table_from_id(&id).await
+    }
+    async fn table_from_id(&self, table_id: &TableId) -> Result<TableDesc, Error>;
+}
+
+/// managed table source information, query from table info manager and table name manager
+#[derive(Clone)]
+pub struct ManagedTableSource {
    /// for query `TableId -> TableName` mapping
    table_info_manager: TableInfoManager,
    table_name_manager: TableNameManager,
 }

-impl TableSource {
+#[async_trait::async_trait]
+impl FlowTableSource for ManagedTableSource {
+    async fn table_from_id(&self, table_id: &TableId) -> Result<TableDesc, Error> {
+        let table_info_value = self
+            .get_table_info_value(table_id)
+            .await?
+            .with_context(|| TableNotFoundSnafu {
+                name: format!("TableId = {:?}, Can't found table info", table_id),
+            })?;
+        let desc = table_info_value_to_relation_desc(table_info_value)?;
+
+        Ok(desc)
+    }
+    async fn table_name_from_id(&self, table_id: &TableId) -> Result<TableName, Error> {
+        self.get_table_name(table_id).await
+    }
+    async fn table_id_from_name(&self, name: &TableName) -> Result<TableId, Error> {
+        self.get_opt_table_id_from_name(name)
+            .await?
+            .with_context(|| TableNotFoundSnafu {
+                name: name.join("."),
+            })
+    }
+}
+
+impl ManagedTableSource {
    pub fn new(table_info_manager: TableInfoManager, table_name_manager: TableNameManager) -> Self {
-        TableSource {
+        ManagedTableSource {
            table_info_manager,
            table_name_manager,
        }
@@ -61,8 +130,11 @@ impl TableSource {
            .map(|id| id.table_id())
    }

-    /// If the table havn't been created in database, the tableId returned would be null
-    pub async fn get_table_id_from_name(&self, name: &TableName) -> Result<Option<TableId>, Error> {
+    /// If the table haven't been created in database, the tableId returned would be null
+    pub async fn get_opt_table_id_from_name(
+        &self,
+        name: &TableName,
+    ) -> Result<Option<TableId>, Error> {
        let ret = self
            .table_name_manager
            .get(TableNameKey::new(&name[0], &name[1], &name[2]))
@@ -106,7 +178,7 @@ impl TableSource {
    pub async fn get_table_name_schema(
        &self,
        table_id: &TableId,
-    ) -> Result<(TableName, RelationDesc), Error> {
+    ) -> Result<(TableName, TableDesc), Error> {
        let table_info_value = self
            .get_table_info_value(table_id)
            .await?
@@ -121,38 +193,125 @@ impl TableSource {
            table_name.table_name,
        ];

-        let raw_schema = table_info_value.table_info.meta.schema;
-        let (column_types, col_names): (Vec<_>, Vec<_>) = raw_schema
-            .column_schemas
-            .clone()
-            .into_iter()
-            .map(|col| {
-                (
-                    ColumnType {
-                        nullable: col.is_nullable(),
-                        scalar_type: col.data_type,
-                    },
-                    Some(col.name),
-                )
-            })
-            .unzip();
-
-        let key = table_info_value.table_info.meta.primary_key_indices;
-        let keys = vec![repr::Key::from(key)];
-
-        let time_index = raw_schema.timestamp_index;
-        Ok((
-            table_name,
-            RelationDesc {
-                typ: RelationType {
-                    column_types,
-                    keys,
-                    time_index,
-                    // by default table schema's column are all non-auto
-                    auto_columns: vec![],
-                },
-                names: col_names,
-            },
-        ))
+        let desc = table_info_value_to_relation_desc(table_info_value)?;
+        Ok((table_name, desc))
+    }
+}
+
+impl std::fmt::Debug for ManagedTableSource {
+    fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result {
+        f.debug_struct("KvBackendTableSource").finish()
+    }
+}
+
+#[cfg(test)]
+pub(crate) mod test {
+    use std::collections::HashMap;
+
+    use datatypes::data_type::ConcreteDataType as CDT;
+
+    use super::*;
+    use crate::repr::{ColumnType, RelationType};
+
+    pub struct FlowDummyTableSource {
+        pub id_names_to_desc: Vec<(TableId, TableName, TableDesc)>,
+        id_to_idx: HashMap<TableId, usize>,
+        name_to_idx: HashMap<TableName, usize>,
+    }
+
+    impl Default for FlowDummyTableSource {
+        fn default() -> Self {
+            let id_names_to_desc = vec![
+                (
+                    1024,
+                    [
+                        "greptime".to_string(),
+                        "public".to_string(),
+                        "numbers".to_string(),
+                    ],
+                    TableDesc::new_no_default(
+                        RelationType::new(vec![ColumnType::new(CDT::uint32_datatype(), false)])
+                            .into_named(vec![Some("number".to_string())]),
+                    ),
+                ),
+                (
+                    1025,
+                    [
+                        "greptime".to_string(),
+                        "public".to_string(),
+                        "numbers_with_ts".to_string(),
+                    ],
+                    TableDesc::new_no_default(
+                        RelationType::new(vec![
+                            ColumnType::new(CDT::uint32_datatype(), false),
+                            ColumnType::new(CDT::timestamp_millisecond_datatype(), false),
+                        ])
+                        .into_named(vec![Some("number".to_string()), Some("ts".to_string())]),
+                    ),
+                ),
+            ];
+            let id_to_idx = id_names_to_desc
+                .iter()
+                .enumerate()
+                .map(|(idx, (id, _name, _desc))| (*id, idx))
+                .collect();
+            let name_to_idx = id_names_to_desc
+                .iter()
+                .enumerate()
+                .map(|(idx, (_id, name, _desc))| (name.clone(), idx))
+                .collect();
+            Self {
+                id_names_to_desc,
+                id_to_idx,
+                name_to_idx,
+            }
+        }
+    }
+
+    #[async_trait::async_trait]
+    impl FlowTableSource for FlowDummyTableSource {
+        async fn table_from_id(&self, table_id: &TableId) -> Result<TableDesc, Error> {
+            let idx = self.id_to_idx.get(table_id).context(TableNotFoundSnafu {
+                name: format!("Table id = {:?}, couldn't found table desc", table_id),
+            })?;
+            let desc = self
+                .id_names_to_desc
+                .get(*idx)
+                .map(|x| x.2.clone())
+                .context(TableNotFoundSnafu {
+                    name: format!("Table id = {:?}, couldn't found table desc", table_id),
+                })?;
+            Ok(desc)
+        }
+
+        async fn table_name_from_id(&self, table_id: &TableId) -> Result<TableName, Error> {
+            let idx = self.id_to_idx.get(table_id).context(TableNotFoundSnafu {
+                name: format!("Table id = {:?}, couldn't found table desc", table_id),
+            })?;
+            self.id_names_to_desc
+                .get(*idx)
+                .map(|x| x.1.clone())
+                .context(TableNotFoundSnafu {
+                    name: format!("Table id = {:?}, couldn't found table desc", table_id),
+                })
+        }
+
+        async fn table_id_from_name(&self, name: &TableName) -> Result<TableId, Error> {
+            for (id, table_name, _desc) in &self.id_names_to_desc {
+                if name == table_name {
+                    return Ok(*id);
+                }
+            }
+            TableNotFoundSnafu {
+                name: format!("Table name = {:?}, couldn't found table id", name),
+            }
+            .fail()?
+        }
+    }
+
+    impl std::fmt::Debug for FlowDummyTableSource {
+        fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result {
+            f.debug_struct("DummyTableSource").finish()
+        }
    }
 }
--- a/src/flow/src/adapter/util.rs
+++ b/src/flow/src/adapter/util.rs
@@ -12,16 +12,160 @@
 // See the License for the specific language governing permissions and
 // limitations under the License.

+use std::sync::Arc;
+
 use api::helper::ColumnDataTypeWrapper;
 use api::v1::column_def::options_from_column_schema;
-use api::v1::{ColumnDataType, ColumnDataTypeExtension, SemanticType};
+use api::v1::{ColumnDataType, ColumnDataTypeExtension, CreateTableExpr, SemanticType};
 use common_error::ext::BoxedError;
+use common_meta::key::table_info::TableInfoValue;
 use datatypes::prelude::ConcreteDataType;
 use datatypes::schema::ColumnSchema;
 use itertools::Itertools;
-use snafu::ResultExt;
+use operator::expr_factory::CreateExprFactory;
+use session::context::QueryContextBuilder;
+use snafu::{OptionExt, ResultExt};
+use table::table_reference::TableReference;

-use crate::error::{Error, ExternalSnafu};
+use crate::adapter::table_source::TableDesc;
+use crate::adapter::{TableName, AUTO_CREATED_PLACEHOLDER_TS_COL};
+use crate::error::{Error, ExternalSnafu, UnexpectedSnafu};
+use crate::repr::{ColumnType, RelationDesc, RelationType};
+use crate::FlowWorkerManager;
+
+impl FlowWorkerManager {
+    /// Create table from given schema(will adjust to add auto column if needed), return true if table is created
+    pub(crate) async fn create_table_from_relation(
+        &self,
+        flow_name: &str,
+        table_name: &TableName,
+        relation_desc: &RelationDesc,
+    ) -> Result<bool, Error> {
+        if self.fetch_table_pk_schema(table_name).await?.is_some() {
+            return Ok(false);
+        }
+        let (pks, tys, _) = self.adjust_auto_created_table_schema(relation_desc).await?;
+
+        //create sink table using pks, column types and is_ts_auto
+
+        let proto_schema = column_schemas_to_proto(tys.clone(), &pks)?;
+
+        // create sink table
+        let create_expr = CreateExprFactory {}
+            .create_table_expr_by_column_schemas(
+                &TableReference {
+                    catalog: &table_name[0],
+                    schema: &table_name[1],
+                    table: &table_name[2],
+                },
+                &proto_schema,
+                "mito",
+                Some(&format!("Sink table for flow {}", flow_name)),
+            )
+            .map_err(BoxedError::new)
+            .context(ExternalSnafu)?;
+
+        self.submit_create_sink_table_ddl(create_expr).await?;
+        Ok(true)
+    }
+
+    /// Try fetch table with adjusted schema(added auto column if needed)
+    pub(crate) async fn try_fetch_existing_table(
+        &self,
+        table_name: &TableName,
+    ) -> Result<Option<(bool, Vec<api::v1::ColumnSchema>)>, Error> {
+        if let Some((primary_keys, time_index, schema)) =
+            self.fetch_table_pk_schema(table_name).await?
+        {
+            // check if the last column is the auto created timestamp column, hence the table is auto created from
+            // flow's plan type
+            let is_auto_create = {
+                let correct_name = schema
+                    .last()
+                    .map(|s| s.name == AUTO_CREATED_PLACEHOLDER_TS_COL)
+                    .unwrap_or(false);
+                let correct_time_index = time_index == Some(schema.len() - 1);
+                correct_name && correct_time_index
+            };
+            let proto_schema = column_schemas_to_proto(schema, &primary_keys)?;
+            Ok(Some((is_auto_create, proto_schema)))
+        } else {
+            Ok(None)
+        }
+    }
+
+    /// submit a create table ddl
+    pub(crate) async fn submit_create_sink_table_ddl(
+        &self,
+        mut create_table: CreateTableExpr,
+    ) -> Result<(), Error> {
+        let stmt_exec = {
+            self.frontend_invoker
+                .read()
+                .await
+                .as_ref()
+                .map(|f| f.statement_executor())
+        }
+        .context(UnexpectedSnafu {
+            reason: "Failed to get statement executor",
+        })?;
+        let ctx = Arc::new(
+            QueryContextBuilder::default()
+                .current_catalog(create_table.catalog_name.clone())
+                .current_schema(create_table.schema_name.clone())
+                .build(),
+        );
+        stmt_exec
+            .create_table_inner(&mut create_table, None, ctx)
+            .await
+            .map_err(BoxedError::new)
+            .context(ExternalSnafu)?;
+
+        Ok(())
+    }
+}
+
+pub fn table_info_value_to_relation_desc(
+    table_info_value: TableInfoValue,
+) -> Result<TableDesc, Error> {
+    let raw_schema = table_info_value.table_info.meta.schema;
+    let (column_types, col_names): (Vec<_>, Vec<_>) = raw_schema
+        .column_schemas
+        .clone()
+        .into_iter()
+        .map(|col| {
+            (
+                ColumnType {
+                    nullable: col.is_nullable(),
+                    scalar_type: col.data_type,
+                },
+                Some(col.name),
+            )
+        })
+        .unzip();
+
+    let key = table_info_value.table_info.meta.primary_key_indices;
+    let keys = vec![crate::repr::Key::from(key)];
+
+    let time_index = raw_schema.timestamp_index;
+    let relation_desc = RelationDesc {
+        typ: RelationType {
+            column_types,
+            keys,
+            time_index,
+            // by default table schema's column are all non-auto
+            auto_columns: vec![],
+        },
+        names: col_names,
+    };
+    let default_values = raw_schema
+        .column_schemas
+        .iter()
+        .map(|c| c.default_constraint().cloned())
+        .collect_vec();
+
+    Ok(TableDesc::new(relation_desc, default_values))
+}

 pub fn from_proto_to_data_type(
    column_schema: &api::v1::ColumnSchema,
@@ -75,3 +219,29 @@ pub fn column_schemas_to_proto(
        .collect();
    Ok(ret)
 }
+
+/// Convert `RelationDesc` to `ColumnSchema` list,
+/// if the column name is not present, use `col_{idx}` as the column name
+pub fn relation_desc_to_column_schemas_with_fallback(schema: &RelationDesc) -> Vec<ColumnSchema> {
+    schema
+        .typ()
+        .column_types
+        .clone()
+        .into_iter()
+        .enumerate()
+        .map(|(idx, typ)| {
+            let name = schema
+                .names
+                .get(idx)
+                .cloned()
+                .flatten()
+                .unwrap_or(format!("col_{}", idx));
+            let ret = ColumnSchema::new(name, typ.scalar_type, typ.nullable);
+            if schema.typ().time_index == Some(idx) {
+                ret.with_time_index(true)
+            } else {
+                ret
+            }
+        })
+        .collect_vec()
+}
--- a/src/flow/src/adapter/worker.rs
+++ b/src/flow/src/adapter/worker.rs
@@ -247,15 +247,25 @@ impl<'s> Worker<'s> {
        src_recvs: Vec<broadcast::Receiver<Batch>>,
        // TODO(discord9): set expire duration for all arrangement and compare to sys timestamp instead
        expire_after: Option<repr::Duration>,
+        or_replace: bool,
        create_if_not_exists: bool,
        err_collector: ErrCollector,
    ) -> Result<Option<FlowId>, Error> {
-        let already_exists = self.task_states.contains_key(&flow_id);
-        match (already_exists, create_if_not_exists) {
-            (true, true) => return Ok(None),
-            (true, false) => FlowAlreadyExistSnafu { id: flow_id }.fail()?,
-            (false, _) => (),
-        };
+        let already_exist = self.task_states.contains_key(&flow_id);
+        match (create_if_not_exists, or_replace, already_exist) {
+            // if replace, ignore that old flow exists
+            (_, true, true) => {
+                info!("Replacing flow with id={}", flow_id);
+            }
+            (false, false, true) => FlowAlreadyExistSnafu { id: flow_id }.fail()?,
+            // already exists, and not replace, return None
+            (true, false, true) => {
+                info!("Flow with id={} already exists, do nothing", flow_id);
+                return Ok(None);
+            }
+            // continue as normal
+            (_, _, false) => (),
+        }

        let mut cur_task_state = ActiveDataflowState::<'s> {
            err_collector,
@@ -341,6 +351,7 @@ impl<'s> Worker<'s> {
                source_ids,
                src_recvs,
                expire_after,
+                or_replace,
                create_if_not_exists,
                err_collector,
            } => {
@@ -352,6 +363,7 @@ impl<'s> Worker<'s> {
                    &source_ids,
                    src_recvs,
                    expire_after,
+                    or_replace,
                    create_if_not_exists,
                    err_collector,
                );
@@ -398,6 +410,7 @@ pub enum Request {
        source_ids: Vec<GlobalId>,
        src_recvs: Vec<broadcast::Receiver<Batch>>,
        expire_after: Option<repr::Duration>,
+        or_replace: bool,
        create_if_not_exists: bool,
        err_collector: ErrCollector,
    },
@@ -547,6 +560,7 @@ mod test {
            source_ids: src_ids,
            src_recvs: vec![rx],
            expire_after: None,
+            or_replace: false,
            create_if_not_exists: true,
            err_collector: ErrCollector::default(),
        };
--- a/src/flow/src/compute/render/reduce.rs
+++ b/src/flow/src/compute/render/reduce.rs
@@ -16,6 +16,7 @@ use std::collections::{BTreeMap, BTreeSet};
 use std::ops::Range;
 use std::sync::Arc;

+use arrow::array::new_null_array;
 use common_telemetry::trace;
 use datatypes::data_type::ConcreteDataType;
 use datatypes::prelude::DataType;
@@ -398,20 +399,54 @@ fn reduce_batch_subgraph(
                }
            }

-            // TODO: here reduce numbers of eq to minimal by keeping slicing key/val batch
+            let key_data_types = output_type
+                .column_types
+                .iter()
+                .map(|t| t.scalar_type.clone())
+                .collect_vec();
+
+            // TODO(discord9): here reduce numbers of eq to minimal by keeping slicing key/val batch
            for key_row in distinct_keys {
                let key_scalar_value = {
                    let mut key_scalar_value = Vec::with_capacity(key_row.len());
-                    for key in key_row.iter() {
+                    for (key_idx, key) in key_row.iter().enumerate() {
                        let v =
                            key.try_to_scalar_value(&key.data_type())
                                .context(DataTypeSnafu {
                                    msg: "can't convert key values to datafusion value",
                                })?;
-                        let arrow_value =
+
+                        let key_data_type = key_data_types.get(key_idx).context(InternalSnafu {
+                            reason: format!(
+                                "Key index out of bound, expected at most {} but got {}",
+                                output_type.column_types.len(),
+                                key_idx
+                            ),
+                        })?;
+
+                        // if incoming value's datatype is null, it need to be handled specially, see below
+                        if key_data_type.as_arrow_type() != v.data_type()
+                            && !v.data_type().is_null()
+                        {
+                            crate::expr::error::InternalSnafu {
+                                reason: format!(
+                                    "Key data type mismatch, expected {:?} but got {:?}",
+                                    key_data_type.as_arrow_type(),
+                                    v.data_type()
+                                ),
+                            }
+                            .fail()?
+                        }
+
+                        // handle single null key
+                        let arrow_value = if v.data_type().is_null() {
+                            let ret = new_null_array(&arrow::datatypes::DataType::Null, 1);
+                            arrow::array::Scalar::new(ret)
+                        } else {
                            v.to_scalar().context(crate::expr::error::DatafusionSnafu {
                                context: "can't convert key values to arrow value",
-                            })?;
+                            })?
+                        };
                        key_scalar_value.push(arrow_value);
                    }
                    key_scalar_value
@@ -423,7 +458,19 @@ fn reduce_batch_subgraph(
                    .zip(key_batch.batch().iter())
                    .map(|(key, col)| {
                        // TODO(discord9): this takes half of the cpu! And this is redundant amount of `eq`!
-                        arrow::compute::kernels::cmp::eq(&key, &col.to_arrow_array().as_ref() as _)
+
+                        // note that if lhs is a null, we still need to get all rows that are null! But can't use `eq` since
+                        // it will return null if input have null, so we need to use `is_null` instead
+                        if arrow::array::Datum::get(&key).0.data_type().is_null() {
+                            arrow::compute::kernels::boolean::is_null(
+                                col.to_arrow_array().as_ref() as _
+                            )
+                        } else {
+                            arrow::compute::kernels::cmp::eq(
+                                &key,
+                                &col.to_arrow_array().as_ref() as _,
+                            )
+                        }
                    })
                    .try_collect::<_, Vec<_>, _>()
                    .context(ArrowSnafu {
--- a/src/flow/src/compute/types.rs
+++ b/src/flow/src/compute/types.rs
@@ -17,6 +17,7 @@ use std::collections::{BTreeMap, VecDeque};
 use std::rc::Rc;
 use std::sync::Arc;

+use common_error::ext::ErrorExt;
 use hydroflow::scheduled::graph::Hydroflow;
 use hydroflow::scheduled::handoff::TeeingHandoff;
 use hydroflow::scheduled::port::RecvPort;
@@ -25,6 +26,7 @@ use itertools::Itertools;
 use tokio::sync::Mutex;

 use crate::expr::{Batch, EvalError, ScalarExpr};
+use crate::metrics::METRIC_FLOW_ERRORS;
 use crate::repr::DiffRow;
 use crate::utils::ArrangeHandler;

@@ -185,6 +187,9 @@ impl ErrCollector {
    }

    pub fn push_err(&self, err: EvalError) {
+        METRIC_FLOW_ERRORS
+            .with_label_values(&[err.status_code().as_ref()])
+            .inc();
        self.inner.blocking_lock().push_back(err)
    }

--- a/src/flow/src/df_optimizer.rs
+++ b/src/flow/src/df_optimizer.rs
@@ -492,7 +492,7 @@ impl ScalarUDFImpl for TumbleExpand {
                if let Some(start_time) = opt{
                    if !matches!(start_time,  Utf8 | Date32 | Date64 | Timestamp(_, _)){
                        return Err(DataFusionError::Plan(
-                            format!("Expect start_time to either be date, timestampe or string, found {:?}", start_time)
+                            format!("Expect start_time to either be date, timestamp or string, found {:?}", start_time)
                        ));
                    }
                }
--- a/src/flow/src/error.rs
+++ b/src/flow/src/error.rs
@@ -16,12 +16,13 @@

 use std::any::Any;

-use common_error::define_into_tonic_status;
 use common_error::ext::BoxedError;
+use common_error::{define_into_tonic_status, from_err_code_msg_to_header};
 use common_macro::stack_trace_debug;
 use common_telemetry::common_error::ext::ErrorExt;
 use common_telemetry::common_error::status_code::StatusCode;
 use snafu::{Location, Snafu};
+use tonic::metadata::MetadataMap;

 use crate::adapter::FlowId;
 use crate::expr::EvalError;
@@ -31,6 +32,27 @@ use crate::expr::EvalError;
 #[snafu(visibility(pub))]
 #[stack_trace_debug]
 pub enum Error {
+    #[snafu(display(
+        "Failed to insert into flow: region_id={}, flow_ids={:?}",
+        region_id,
+        flow_ids
+    ))]
+    InsertIntoFlow {
+        region_id: u64,
+        flow_ids: Vec<u64>,
+        source: BoxedError,
+        #[snafu(implicit)]
+        location: Location,
+    },
+
+    #[snafu(display("Error encountered while creating flow: {sql}"))]
+    CreateFlow {
+        sql: String,
+        source: BoxedError,
+        #[snafu(implicit)]
+        location: Location,
+    },
+
    #[snafu(display("External error"))]
    External {
        source: BoxedError,
@@ -186,23 +208,37 @@ pub enum Error {
    },
 }

+/// the outer message is the full error stack, and inner message in header is the last error message that can be show directly to user
+pub fn to_status_with_last_err(err: impl ErrorExt) -> tonic::Status {
+    let msg = err.to_string();
+    let last_err_msg = common_error::ext::StackError::last(&err).to_string();
+    let code = err.status_code() as u32;
+    let header = from_err_code_msg_to_header(code, &last_err_msg);
+
+    tonic::Status::with_metadata(
+        tonic::Code::InvalidArgument,
+        msg,
+        MetadataMap::from_headers(header),
+    )
+}
+
 /// Result type for flow module
 pub type Result<T> = std::result::Result<T, Error>;

 impl ErrorExt for Error {
    fn status_code(&self) -> StatusCode {
        match self {
-            Self::Eval { .. } | Self::JoinTask { .. } | Self::Datafusion { .. } => {
-                StatusCode::Internal
-            }
+            Self::Eval { .. }
+            | Self::JoinTask { .. }
+            | Self::Datafusion { .. }
+            | Self::InsertIntoFlow { .. } => StatusCode::Internal,
            Self::FlowAlreadyExist { .. } => StatusCode::TableAlreadyExists,
            Self::TableNotFound { .. }
            | Self::TableNotFoundMeta { .. }
            | Self::FlowNotFound { .. }
            | Self::ListFlows { .. } => StatusCode::TableNotFound,
-            Self::InvalidQuery { .. } | Self::Plan { .. } | Self::Datatypes { .. } => {
-                StatusCode::PlanQuery
-            }
+            Self::Plan { .. } | Self::Datatypes { .. } => StatusCode::PlanQuery,
+            Self::InvalidQuery { .. } | Self::CreateFlow { .. } => StatusCode::EngineExecuteQuery,
            Self::Unexpected { .. } => StatusCode::Unexpected,
            Self::NotImplemented { .. } | Self::UnsupportedTemporalFilter { .. } => {
                StatusCode::Unsupported
--- a/Show More
+++ b/Show More