mirror of
https://github.com/neondatabase/neon.git
synced 2026-08-17 11:38:22 +00:00
Compare commits
373 Commits
| Author | SHA1 | Date | |
|---|---|---|---|
| a5491463e1 | |||
| a58827f952 | |||
| 36b790f282 | |||
| 3ef7748e6b | |||
| f3310143e4 | |||
| 05b4169644 | |||
| d1495755e7 | |||
| c8dd78c6c8 | |||
| b44ee3950a | |||
| 64334f497d | |||
| 5ffcb688cc | |||
| 32fc2dd683 | |||
| d35ddfbab7 | |||
| 3ee82a9895 | |||
| e770aeee92 | |||
| 32828cddd6 | |||
| bd2046e1ab | |||
| 7e2a3d2728 | |||
| 0e4832308d | |||
| 0a63bc4818 | |||
| 2897dcc9aa | |||
| 1d0ec50ddb | |||
| a86b43fcd7 | |||
| b917868ada | |||
| 7b7d16f52e | |||
| fee4169b6b | |||
| 47e06a2cc6 | |||
| c4423c0623 | |||
| a11cf03123 | |||
| 08b33adfee | |||
| 4fb50144dd | |||
| c500137ca9 | |||
| 252c4acec9 | |||
| db70c175e6 | |||
| ed3b4a58b4 | |||
| 2863d1df63 | |||
| 320b24eab3 | |||
| 13a8a5b09b | |||
| 64ccdf65e0 | |||
| 1ae6aa09dd | |||
| aeb68e51df | |||
| c3e5223a5d | |||
| daaa3211a4 | |||
| 7ff9989dd5 | |||
| ed3b97604c | |||
| 47c50ec460 | |||
| 8c0ec2f681 | |||
| 588bda98e7 | |||
| 504ca7720f | |||
| cf4ea92aad | |||
| 325294bced | |||
| 86c8ba2563 | |||
| feeb2dc6fa | |||
| 57f476ff5a | |||
| 7ee2bebdb7 | |||
| be598f1bf4 | |||
| 30027d94a2 | |||
| bc704917a3 | |||
| b8bbaafc03 | |||
| e1a06b40b7 | |||
| babbe125da | |||
| ca2f7d06b2 | |||
| c22c6a6c9e | |||
| deec3bc578 | |||
| 063553a51b | |||
| 939b5954a5 | |||
| 371020fe6a | |||
| f45818abed | |||
| 0384267d58 | |||
| 62b3bd968a | |||
| e3e3bc3542 | |||
| be014a2222 | |||
| 2e1fe71cc0 | |||
| 068c158ca5 | |||
| b16e4f689f | |||
| dbff725a0c | |||
| 7fa4628434 | |||
| fc538a38b9 | |||
| c2e7cb324f | |||
| 101043122e | |||
| c4d7d59825 | |||
| 0de1e1d664 | |||
| 271598b77f | |||
| 459bc479dc | |||
| c213373a59 | |||
| e0addc100d | |||
| 0519138b04 | |||
| 5da39b469c | |||
| 82027e22dd | |||
| c431e2f1c5 | |||
| 4e5724d9c3 | |||
| 0d3e499059 | |||
| 7b860b837c | |||
| 41fc96e20f | |||
| fb2b1ce57b | |||
| 464717451b | |||
| c6ed86d3d0 | |||
| f0a9017008 | |||
| bb7949ba00 | |||
| 1df0f69664 | |||
| 970066a914 | |||
| 1ebd3897c0 | |||
| 6460beffcd | |||
| 6f7f8958db | |||
| 936a00e077 | |||
| 96a4e8de66 | |||
| 01180666b0 | |||
| 6c94269c32 | |||
| edc691647d | |||
| 855d7b4781 | |||
| c49c9707ce | |||
| 2227540a0d | |||
| f1347f2417 | |||
| 30b295b017 | |||
| 1cef395266 | |||
| 78d160f76d | |||
| b9238059d6 | |||
| d0cb4b88c8 | |||
| 1ec3e39d4e | |||
| a1a74eef2c | |||
| 90e689adda | |||
| f0b2d4b053 | |||
| 299d9474c9 | |||
| 7234208b36 | |||
| 93450f11f5 | |||
| 2f0f9edf33 | |||
| d424f2b7c8 | |||
| 21315e80bc | |||
| 483b66d383 | |||
| aa72a22661 | |||
| 5c0264b591 | |||
| 9f13277729 | |||
| 54aa319805 | |||
| 4a227484bf | |||
| 2f83f85291 | |||
| d6cfcb0d93 | |||
| 392843ad2a | |||
| bd4dae8f4a | |||
| b05fe53cfd | |||
| c13a2f0df1 | |||
| 39be366fc5 | |||
| 6eda0a3158 | |||
| 306c7a1813 | |||
| 80be423a58 | |||
| 5dcfef82f2 | |||
| e67b8f69c0 | |||
| e546872ab4 | |||
| 322ea1cf7c | |||
| 3633742de9 | |||
| 079d3a37ba | |||
| a46e77b476 | |||
| a92702b01e | |||
| 8ff3253f20 | |||
| 04b82c92a7 | |||
| e5bf423e68 | |||
| 60af392e45 | |||
| 661fc41e71 | |||
| 702c488f32 | |||
| 45c5122754 | |||
| 558394f710 | |||
| 73b0898608 | |||
| e65be4c2dc | |||
| 40087b8164 | |||
| c762b59483 | |||
| 5d71601ca9 | |||
| a113c3e433 | |||
| e81fc598f4 | |||
| 48b845fa76 | |||
| 27096858dc | |||
| 4430d0ae7d | |||
| 6e183aa0de | |||
| fd6d0b7635 | |||
| 3710c32aae | |||
| be83bee49d | |||
| cf28e5922a | |||
| 7d384d6953 | |||
| 4b3b37b912 | |||
| 1d8d200f4d | |||
| 0d80d6ce18 | |||
| f653ee039f | |||
| e614a95853 | |||
| 850db4cc13 | |||
| 8a316b1277 | |||
| 4d13bae449 | |||
| 49377abd98 | |||
| a6b2f4e54e | |||
| face60d50b | |||
| 9768aa27f2 | |||
| 96b2e575e1 | |||
| 7222777784 | |||
| 5469fdede0 | |||
| 72aa6b9fdd | |||
| ae0634b7be | |||
| 70711f32fa | |||
| 52a88af0aa | |||
| b7a43bf817 | |||
| dce91b33a4 | |||
| 23ee4f3050 | |||
| 46857e8282 | |||
| 368ab0ce54 | |||
| a5987eebfd | |||
| 6686ede30f | |||
| 373c7057cc | |||
| 7d6ec16166 | |||
| 0e6fdc8a58 | |||
| 521438a5c6 | |||
| 07d7874bc8 | |||
| 1804111a02 | |||
| cd0178efed | |||
| 333574be57 | |||
| 79a799a143 | |||
| 9da06af6c9 | |||
| ce1753d036 | |||
| 67db8432b4 | |||
| 4e2e44e524 | |||
| ed786104f3 | |||
| 84b74f2bd1 | |||
| fec2ad6283 | |||
| 98eebd4682 | |||
| 2f74287c9b | |||
| aee1bf95e3 | |||
| b9de9d75ff | |||
| 7943b709e6 | |||
| d7d066d493 | |||
| e78ac22107 | |||
| 76a8f2bb44 | |||
| 8d59a8581f | |||
| b1ddd01289 | |||
| 6eae4fc9aa | |||
| 765455bca2 | |||
| 4204960942 | |||
| 67345d66ea | |||
| 2266ee5971 | |||
| b58445d855 | |||
| 36050e7f3d | |||
| 33360ed96d | |||
| 39a28d1108 | |||
| efa6aa134f | |||
| 2c724e56e2 | |||
| feff887c6f | |||
| 353d915fcf | |||
| 2e38098cbc | |||
| a6fe5ea1ac | |||
| 05b0aed0c1 | |||
| cd1705357d | |||
| 6bc7561290 | |||
| fbd3ac14b5 | |||
| e437787c8f | |||
| 3460dbf90b | |||
| 6b89d99677 | |||
| 6cc8ea86e4 | |||
| e62a492d6f | |||
| a475cdf642 | |||
| 7002c79a47 | |||
| ee6cf357b4 | |||
| e5c2086b5f | |||
| 5f1208296a | |||
| 88e8e473cd | |||
| b0a77844f6 | |||
| 1baf464307 | |||
| e9b8e81cea | |||
| 85d6194aa4 | |||
| 333a7a68ef | |||
| 6aa4e41bee | |||
| 840183e51f | |||
| cbccc94b03 | |||
| fce227df22 | |||
| bd787e800f | |||
| 4a7704b4a3 | |||
| ff1119da66 | |||
| 4c3ba1627b | |||
| 1407174fb2 | |||
| ec9dcb1889 | |||
| d11d781afc | |||
| 4e44565b71 | |||
| 4ed51ad33b | |||
| 1c1ebe5537 | |||
| c19cb7f386 | |||
| 4b97d31b16 | |||
| 923ade3dd7 | |||
| b04e711975 | |||
| afd0a6b39a | |||
| 99752286d8 | |||
| 15df93363c | |||
| bc0ab741af | |||
| 51d9dfeaa3 | |||
| f63cb18155 | |||
| 0de603d88e | |||
| 240913912a | |||
| 91a4ea0de2 | |||
| 8608704f49 | |||
| efef68ce99 | |||
| 8daefd24da | |||
| 46cc8b7982 | |||
| 38cd90dd0c | |||
| a51b269f15 | |||
| 43bf6d0a0f | |||
| 15273a9b66 | |||
| 78aca668d0 | |||
| acbf4148ea | |||
| 6508540561 | |||
| a41b5244a8 | |||
| 2b3189be95 | |||
| 248563c595 | |||
| 14cd6ca933 | |||
| eb36403e71 | |||
| 3c6f779698 | |||
| f67f0c1c11 | |||
| edb02d3299 | |||
| 664a69e65b | |||
| 478322ebf9 | |||
| 802f174072 | |||
| 47f9890bae | |||
| 262265daad | |||
| 300da5b872 | |||
| 7b22b5c433 | |||
| ffca97bc1e | |||
| cb356f3259 | |||
| c85374295f | |||
| 4992160677 | |||
| bd535b3371 | |||
| d90c5a03af | |||
| 2d02cc9079 | |||
| 49ad94b99f | |||
| 948a217398 | |||
| 125381eae7 | |||
| cd01bbc715 | |||
| d8b5e3b88d | |||
| 06d25f2186 | |||
| f759b561f3 | |||
| ece0555600 | |||
| 73ea0a0b01 | |||
| d8f6d6fd6f | |||
| d24de169a7 | |||
| 0816168296 | |||
| 277b44d57a | |||
| 68c2c3880e | |||
| 49da498f65 | |||
| 2c76ba3dd7 | |||
| dbe3dc69ad | |||
| 8e5bb3ed49 | |||
| ab0be7b8da | |||
| b4c55f5d24 | |||
| ede70d833c | |||
| 70c3d18bb0 | |||
| 7a491f52c4 | |||
| 323c4ecb4f | |||
| 3d2466607e | |||
| ed478b39f4 | |||
| 91585a558d | |||
| 93467eae1f | |||
| f3aac81d19 | |||
| 979ad60c19 | |||
| 9316cb1b1f | |||
| e7939a527a | |||
| 36d26665e1 | |||
| 873347f977 | |||
| e814ac16f9 | |||
| ad3055d386 | |||
| 94e03eb452 | |||
| 380f26ef79 | |||
| 3c5b7f59d7 | |||
| fee89f80b5 | |||
| 41cce8eaf1 | |||
| f88fe0218d | |||
| cc856eca85 | |||
| cf350c6002 | |||
| 0ce6b6a0a3 | |||
| 73f247d537 | |||
| 960be82183 | |||
| 806e5a6c19 | |||
| 8d5df07cce | |||
| df7a9d1407 |
@@ -114,6 +114,7 @@ runs:
|
|||||||
export PLATFORM=${PLATFORM:-github-actions-selfhosted}
|
export PLATFORM=${PLATFORM:-github-actions-selfhosted}
|
||||||
export POSTGRES_DISTRIB_DIR=${POSTGRES_DISTRIB_DIR:-/tmp/neon/pg_install}
|
export POSTGRES_DISTRIB_DIR=${POSTGRES_DISTRIB_DIR:-/tmp/neon/pg_install}
|
||||||
export DEFAULT_PG_VERSION=${PG_VERSION#v}
|
export DEFAULT_PG_VERSION=${PG_VERSION#v}
|
||||||
|
export LD_LIBRARY_PATH=${POSTGRES_DISTRIB_DIR}/v${DEFAULT_PG_VERSION}/lib
|
||||||
|
|
||||||
if [ "${BUILD_TYPE}" = "remote" ]; then
|
if [ "${BUILD_TYPE}" = "remote" ]; then
|
||||||
export REMOTE_ENV=1
|
export REMOTE_ENV=1
|
||||||
@@ -178,7 +179,15 @@ runs:
|
|||||||
|
|
||||||
# Wake up the cluster if we use remote neon instance
|
# Wake up the cluster if we use remote neon instance
|
||||||
if [ "${{ inputs.build_type }}" = "remote" ] && [ -n "${BENCHMARK_CONNSTR}" ]; then
|
if [ "${{ inputs.build_type }}" = "remote" ] && [ -n "${BENCHMARK_CONNSTR}" ]; then
|
||||||
${POSTGRES_DISTRIB_DIR}/v${DEFAULT_PG_VERSION}/bin/psql ${BENCHMARK_CONNSTR} -c "SELECT version();"
|
QUERIES=("SELECT version()")
|
||||||
|
if [[ "${PLATFORM}" = "neon"* ]]; then
|
||||||
|
QUERIES+=("SHOW neon.tenant_id")
|
||||||
|
QUERIES+=("SHOW neon.timeline_id")
|
||||||
|
fi
|
||||||
|
|
||||||
|
for q in "${QUERIES[@]}"; do
|
||||||
|
${POSTGRES_DISTRIB_DIR}/v${DEFAULT_PG_VERSION}/bin/psql ${BENCHMARK_CONNSTR} -c "${q}"
|
||||||
|
done
|
||||||
fi
|
fi
|
||||||
|
|
||||||
# Run the tests.
|
# Run the tests.
|
||||||
|
|||||||
@@ -239,11 +239,6 @@ jobs:
|
|||||||
path: /tmp/neon/
|
path: /tmp/neon/
|
||||||
prefix: latest
|
prefix: latest
|
||||||
|
|
||||||
- name: Add Postgres binaries to PATH
|
|
||||||
run: |
|
|
||||||
${POSTGRES_DISTRIB_DIR}/v${DEFAULT_PG_VERSION}/bin/pgbench --version
|
|
||||||
echo "${POSTGRES_DISTRIB_DIR}/v${DEFAULT_PG_VERSION}/bin" >> $GITHUB_PATH
|
|
||||||
|
|
||||||
- name: Create Neon Project
|
- name: Create Neon Project
|
||||||
if: contains(fromJson('["neon-captest-new", "neon-captest-freetier", "neonvm-captest-new", "neonvm-captest-freetier"]'), matrix.platform)
|
if: contains(fromJson('["neon-captest-new", "neon-captest-freetier", "neonvm-captest-new", "neonvm-captest-freetier"]'), matrix.platform)
|
||||||
id: create-neon-project
|
id: create-neon-project
|
||||||
@@ -282,16 +277,6 @@ jobs:
|
|||||||
|
|
||||||
echo "connstr=${CONNSTR}" >> $GITHUB_OUTPUT
|
echo "connstr=${CONNSTR}" >> $GITHUB_OUTPUT
|
||||||
|
|
||||||
QUERIES=("SELECT version()")
|
|
||||||
if [[ "${PLATFORM}" = "neon"* ]]; then
|
|
||||||
QUERIES+=("SHOW neon.tenant_id")
|
|
||||||
QUERIES+=("SHOW neon.timeline_id")
|
|
||||||
fi
|
|
||||||
|
|
||||||
for q in "${QUERIES[@]}"; do
|
|
||||||
psql ${CONNSTR} -c "${q}"
|
|
||||||
done
|
|
||||||
|
|
||||||
- name: Benchmark init
|
- name: Benchmark init
|
||||||
uses: ./.github/actions/run-python-test-set
|
uses: ./.github/actions/run-python-test-set
|
||||||
with:
|
with:
|
||||||
@@ -377,25 +362,12 @@ jobs:
|
|||||||
path: /tmp/neon/
|
path: /tmp/neon/
|
||||||
prefix: latest
|
prefix: latest
|
||||||
|
|
||||||
- name: Add Postgres binaries to PATH
|
|
||||||
run: |
|
|
||||||
${POSTGRES_DISTRIB_DIR}/v${DEFAULT_PG_VERSION}/bin/pgbench --version
|
|
||||||
echo "${POSTGRES_DISTRIB_DIR}/v${DEFAULT_PG_VERSION}/bin" >> $GITHUB_PATH
|
|
||||||
|
|
||||||
- name: Set up Connection String
|
- name: Set up Connection String
|
||||||
id: set-up-connstr
|
id: set-up-connstr
|
||||||
run: |
|
run: |
|
||||||
CONNSTR=${{ secrets.BENCHMARK_PGVECTOR_CONNSTR }}
|
CONNSTR=${{ secrets.BENCHMARK_PGVECTOR_CONNSTR }}
|
||||||
|
|
||||||
echo "connstr=${CONNSTR}" >> $GITHUB_OUTPUT
|
|
||||||
|
|
||||||
QUERIES=("SELECT version()")
|
echo "connstr=${CONNSTR}" >> $GITHUB_OUTPUT
|
||||||
QUERIES+=("SHOW neon.tenant_id")
|
|
||||||
QUERIES+=("SHOW neon.timeline_id")
|
|
||||||
|
|
||||||
for q in "${QUERIES[@]}"; do
|
|
||||||
psql ${CONNSTR} -c "${q}"
|
|
||||||
done
|
|
||||||
|
|
||||||
- name: Benchmark pgvector hnsw indexing
|
- name: Benchmark pgvector hnsw indexing
|
||||||
uses: ./.github/actions/run-python-test-set
|
uses: ./.github/actions/run-python-test-set
|
||||||
@@ -417,12 +389,12 @@ jobs:
|
|||||||
test_selection: performance/test_perf_pgvector_queries.py
|
test_selection: performance/test_perf_pgvector_queries.py
|
||||||
run_in_parallel: false
|
run_in_parallel: false
|
||||||
save_perf_report: ${{ env.SAVE_PERF_REPORT }}
|
save_perf_report: ${{ env.SAVE_PERF_REPORT }}
|
||||||
extra_params: -m remote_cluster --timeout 21600
|
extra_params: -m remote_cluster --timeout 21600
|
||||||
env:
|
env:
|
||||||
BENCHMARK_CONNSTR: ${{ steps.set-up-connstr.outputs.connstr }}
|
BENCHMARK_CONNSTR: ${{ steps.set-up-connstr.outputs.connstr }}
|
||||||
VIP_VAP_ACCESS_TOKEN: "${{ secrets.VIP_VAP_ACCESS_TOKEN }}"
|
VIP_VAP_ACCESS_TOKEN: "${{ secrets.VIP_VAP_ACCESS_TOKEN }}"
|
||||||
PERF_TEST_RESULT_CONNSTR: "${{ secrets.PERF_TEST_RESULT_CONNSTR }}"
|
PERF_TEST_RESULT_CONNSTR: "${{ secrets.PERF_TEST_RESULT_CONNSTR }}"
|
||||||
|
|
||||||
- name: Create Allure report
|
- name: Create Allure report
|
||||||
if: ${{ !cancelled() }}
|
if: ${{ !cancelled() }}
|
||||||
uses: ./.github/actions/allure-report-generate
|
uses: ./.github/actions/allure-report-generate
|
||||||
@@ -477,11 +449,6 @@ jobs:
|
|||||||
path: /tmp/neon/
|
path: /tmp/neon/
|
||||||
prefix: latest
|
prefix: latest
|
||||||
|
|
||||||
- name: Add Postgres binaries to PATH
|
|
||||||
run: |
|
|
||||||
${POSTGRES_DISTRIB_DIR}/v${DEFAULT_PG_VERSION}/bin/pgbench --version
|
|
||||||
echo "${POSTGRES_DISTRIB_DIR}/v${DEFAULT_PG_VERSION}/bin" >> $GITHUB_PATH
|
|
||||||
|
|
||||||
- name: Set up Connection String
|
- name: Set up Connection String
|
||||||
id: set-up-connstr
|
id: set-up-connstr
|
||||||
run: |
|
run: |
|
||||||
@@ -503,16 +470,6 @@ jobs:
|
|||||||
|
|
||||||
echo "connstr=${CONNSTR}" >> $GITHUB_OUTPUT
|
echo "connstr=${CONNSTR}" >> $GITHUB_OUTPUT
|
||||||
|
|
||||||
QUERIES=("SELECT version()")
|
|
||||||
if [[ "${PLATFORM}" = "neon"* ]]; then
|
|
||||||
QUERIES+=("SHOW neon.tenant_id")
|
|
||||||
QUERIES+=("SHOW neon.timeline_id")
|
|
||||||
fi
|
|
||||||
|
|
||||||
for q in "${QUERIES[@]}"; do
|
|
||||||
psql ${CONNSTR} -c "${q}"
|
|
||||||
done
|
|
||||||
|
|
||||||
- name: ClickBench benchmark
|
- name: ClickBench benchmark
|
||||||
uses: ./.github/actions/run-python-test-set
|
uses: ./.github/actions/run-python-test-set
|
||||||
with:
|
with:
|
||||||
@@ -580,11 +537,6 @@ jobs:
|
|||||||
path: /tmp/neon/
|
path: /tmp/neon/
|
||||||
prefix: latest
|
prefix: latest
|
||||||
|
|
||||||
- name: Add Postgres binaries to PATH
|
|
||||||
run: |
|
|
||||||
${POSTGRES_DISTRIB_DIR}/v${DEFAULT_PG_VERSION}/bin/pgbench --version
|
|
||||||
echo "${POSTGRES_DISTRIB_DIR}/v${DEFAULT_PG_VERSION}/bin" >> $GITHUB_PATH
|
|
||||||
|
|
||||||
- name: Get Connstring Secret Name
|
- name: Get Connstring Secret Name
|
||||||
run: |
|
run: |
|
||||||
case "${PLATFORM}" in
|
case "${PLATFORM}" in
|
||||||
@@ -613,16 +565,6 @@ jobs:
|
|||||||
|
|
||||||
echo "connstr=${CONNSTR}" >> $GITHUB_OUTPUT
|
echo "connstr=${CONNSTR}" >> $GITHUB_OUTPUT
|
||||||
|
|
||||||
QUERIES=("SELECT version()")
|
|
||||||
if [[ "${PLATFORM}" = "neon"* ]]; then
|
|
||||||
QUERIES+=("SHOW neon.tenant_id")
|
|
||||||
QUERIES+=("SHOW neon.timeline_id")
|
|
||||||
fi
|
|
||||||
|
|
||||||
for q in "${QUERIES[@]}"; do
|
|
||||||
psql ${CONNSTR} -c "${q}"
|
|
||||||
done
|
|
||||||
|
|
||||||
- name: Run TPC-H benchmark
|
- name: Run TPC-H benchmark
|
||||||
uses: ./.github/actions/run-python-test-set
|
uses: ./.github/actions/run-python-test-set
|
||||||
with:
|
with:
|
||||||
@@ -681,11 +623,6 @@ jobs:
|
|||||||
path: /tmp/neon/
|
path: /tmp/neon/
|
||||||
prefix: latest
|
prefix: latest
|
||||||
|
|
||||||
- name: Add Postgres binaries to PATH
|
|
||||||
run: |
|
|
||||||
${POSTGRES_DISTRIB_DIR}/v${DEFAULT_PG_VERSION}/bin/pgbench --version
|
|
||||||
echo "${POSTGRES_DISTRIB_DIR}/v${DEFAULT_PG_VERSION}/bin" >> $GITHUB_PATH
|
|
||||||
|
|
||||||
- name: Set up Connection String
|
- name: Set up Connection String
|
||||||
id: set-up-connstr
|
id: set-up-connstr
|
||||||
run: |
|
run: |
|
||||||
@@ -707,16 +644,6 @@ jobs:
|
|||||||
|
|
||||||
echo "connstr=${CONNSTR}" >> $GITHUB_OUTPUT
|
echo "connstr=${CONNSTR}" >> $GITHUB_OUTPUT
|
||||||
|
|
||||||
QUERIES=("SELECT version()")
|
|
||||||
if [[ "${PLATFORM}" = "neon"* ]]; then
|
|
||||||
QUERIES+=("SHOW neon.tenant_id")
|
|
||||||
QUERIES+=("SHOW neon.timeline_id")
|
|
||||||
fi
|
|
||||||
|
|
||||||
for q in "${QUERIES[@]}"; do
|
|
||||||
psql ${CONNSTR} -c "${q}"
|
|
||||||
done
|
|
||||||
|
|
||||||
- name: Run user examples
|
- name: Run user examples
|
||||||
uses: ./.github/actions/run-python-test-set
|
uses: ./.github/actions/run-python-test-set
|
||||||
with:
|
with:
|
||||||
|
|||||||
@@ -63,14 +63,16 @@ jobs:
|
|||||||
mkdir -p /tmp/.docker-custom
|
mkdir -p /tmp/.docker-custom
|
||||||
echo DOCKER_CONFIG=/tmp/.docker-custom >> $GITHUB_ENV
|
echo DOCKER_CONFIG=/tmp/.docker-custom >> $GITHUB_ENV
|
||||||
|
|
||||||
- uses: docker/setup-buildx-action@v2
|
- uses: docker/setup-buildx-action@v3
|
||||||
|
with:
|
||||||
|
cache-binary: false
|
||||||
|
|
||||||
- uses: docker/login-action@v2
|
- uses: docker/login-action@v3
|
||||||
with:
|
with:
|
||||||
username: ${{ secrets.NEON_DOCKERHUB_USERNAME }}
|
username: ${{ secrets.NEON_DOCKERHUB_USERNAME }}
|
||||||
password: ${{ secrets.NEON_DOCKERHUB_PASSWORD }}
|
password: ${{ secrets.NEON_DOCKERHUB_PASSWORD }}
|
||||||
|
|
||||||
- uses: docker/build-push-action@v4
|
- uses: docker/build-push-action@v6
|
||||||
with:
|
with:
|
||||||
context: .
|
context: .
|
||||||
provenance: false
|
provenance: false
|
||||||
@@ -82,6 +84,7 @@ jobs:
|
|||||||
tags: neondatabase/build-tools:${{ inputs.image-tag }}-${{ matrix.arch }}
|
tags: neondatabase/build-tools:${{ inputs.image-tag }}-${{ matrix.arch }}
|
||||||
|
|
||||||
- name: Remove custom docker config directory
|
- name: Remove custom docker config directory
|
||||||
|
if: always()
|
||||||
run: |
|
run: |
|
||||||
rm -rf /tmp/.docker-custom
|
rm -rf /tmp/.docker-custom
|
||||||
|
|
||||||
|
|||||||
@@ -30,7 +30,7 @@ jobs:
|
|||||||
if: ${{ !contains(github.event.pull_request.labels.*.name, 'run-no-ci') }}
|
if: ${{ !contains(github.event.pull_request.labels.*.name, 'run-no-ci') }}
|
||||||
uses: ./.github/workflows/check-permissions.yml
|
uses: ./.github/workflows/check-permissions.yml
|
||||||
with:
|
with:
|
||||||
github-event-name: ${{ github.event_name}}
|
github-event-name: ${{ github.event_name }}
|
||||||
|
|
||||||
cancel-previous-e2e-tests:
|
cancel-previous-e2e-tests:
|
||||||
needs: [ check-permissions ]
|
needs: [ check-permissions ]
|
||||||
@@ -335,6 +335,8 @@ jobs:
|
|||||||
|
|
||||||
- name: Run cargo build
|
- name: Run cargo build
|
||||||
run: |
|
run: |
|
||||||
|
PQ_LIB_DIR=$(pwd)/pg_install/v16/lib
|
||||||
|
export PQ_LIB_DIR
|
||||||
${cov_prefix} mold -run cargo build $CARGO_FLAGS $CARGO_FEATURES --bins --tests
|
${cov_prefix} mold -run cargo build $CARGO_FLAGS $CARGO_FEATURES --bins --tests
|
||||||
|
|
||||||
# Do install *before* running rust tests because they might recompile the
|
# Do install *before* running rust tests because they might recompile the
|
||||||
@@ -383,6 +385,11 @@ jobs:
|
|||||||
env:
|
env:
|
||||||
NEXTEST_RETRIES: 3
|
NEXTEST_RETRIES: 3
|
||||||
run: |
|
run: |
|
||||||
|
PQ_LIB_DIR=$(pwd)/pg_install/v16/lib
|
||||||
|
export PQ_LIB_DIR
|
||||||
|
LD_LIBRARY_PATH=$(pwd)/pg_install/v16/lib
|
||||||
|
export LD_LIBRARY_PATH
|
||||||
|
|
||||||
#nextest does not yet support running doctests
|
#nextest does not yet support running doctests
|
||||||
cargo test --doc $CARGO_FLAGS $CARGO_FEATURES
|
cargo test --doc $CARGO_FLAGS $CARGO_FEATURES
|
||||||
|
|
||||||
@@ -744,14 +751,16 @@ jobs:
|
|||||||
run: |
|
run: |
|
||||||
mkdir -p .docker-custom
|
mkdir -p .docker-custom
|
||||||
echo DOCKER_CONFIG=$(pwd)/.docker-custom >> $GITHUB_ENV
|
echo DOCKER_CONFIG=$(pwd)/.docker-custom >> $GITHUB_ENV
|
||||||
- uses: docker/setup-buildx-action@v2
|
- uses: docker/setup-buildx-action@v3
|
||||||
|
with:
|
||||||
|
cache-binary: false
|
||||||
|
|
||||||
- uses: docker/login-action@v3
|
- uses: docker/login-action@v3
|
||||||
with:
|
with:
|
||||||
username: ${{ secrets.NEON_DOCKERHUB_USERNAME }}
|
username: ${{ secrets.NEON_DOCKERHUB_USERNAME }}
|
||||||
password: ${{ secrets.NEON_DOCKERHUB_PASSWORD }}
|
password: ${{ secrets.NEON_DOCKERHUB_PASSWORD }}
|
||||||
|
|
||||||
- uses: docker/build-push-action@v5
|
- uses: docker/build-push-action@v6
|
||||||
with:
|
with:
|
||||||
context: .
|
context: .
|
||||||
build-args: |
|
build-args: |
|
||||||
@@ -822,11 +831,12 @@ jobs:
|
|||||||
run: |
|
run: |
|
||||||
mkdir -p .docker-custom
|
mkdir -p .docker-custom
|
||||||
echo DOCKER_CONFIG=$(pwd)/.docker-custom >> $GITHUB_ENV
|
echo DOCKER_CONFIG=$(pwd)/.docker-custom >> $GITHUB_ENV
|
||||||
- uses: docker/setup-buildx-action@v2
|
- uses: docker/setup-buildx-action@v3
|
||||||
with:
|
with:
|
||||||
|
cache-binary: false
|
||||||
# Disable parallelism for docker buildkit.
|
# Disable parallelism for docker buildkit.
|
||||||
# As we already build everything with `make -j$(nproc)`, running it in additional level of parallelisam blows up the Runner.
|
# As we already build everything with `make -j$(nproc)`, running it in additional level of parallelisam blows up the Runner.
|
||||||
config-inline: |
|
buildkitd-config-inline: |
|
||||||
[worker.oci]
|
[worker.oci]
|
||||||
max-parallelism = 1
|
max-parallelism = 1
|
||||||
|
|
||||||
@@ -842,7 +852,7 @@ jobs:
|
|||||||
password: ${{ secrets.AWS_SECRET_KEY_DEV }}
|
password: ${{ secrets.AWS_SECRET_KEY_DEV }}
|
||||||
|
|
||||||
- name: Build compute-node image
|
- name: Build compute-node image
|
||||||
uses: docker/build-push-action@v5
|
uses: docker/build-push-action@v6
|
||||||
with:
|
with:
|
||||||
context: .
|
context: .
|
||||||
build-args: |
|
build-args: |
|
||||||
@@ -861,7 +871,7 @@ jobs:
|
|||||||
|
|
||||||
- name: Build neon extensions test image
|
- name: Build neon extensions test image
|
||||||
if: matrix.version == 'v16'
|
if: matrix.version == 'v16'
|
||||||
uses: docker/build-push-action@v5
|
uses: docker/build-push-action@v6
|
||||||
with:
|
with:
|
||||||
context: .
|
context: .
|
||||||
build-args: |
|
build-args: |
|
||||||
@@ -882,7 +892,7 @@ jobs:
|
|||||||
- name: Build compute-tools image
|
- name: Build compute-tools image
|
||||||
# compute-tools are Postgres independent, so build it only once
|
# compute-tools are Postgres independent, so build it only once
|
||||||
if: matrix.version == 'v16'
|
if: matrix.version == 'v16'
|
||||||
uses: docker/build-push-action@v5
|
uses: docker/build-push-action@v6
|
||||||
with:
|
with:
|
||||||
target: compute-tools-image
|
target: compute-tools-image
|
||||||
context: .
|
context: .
|
||||||
@@ -1358,3 +1368,31 @@ jobs:
|
|||||||
with:
|
with:
|
||||||
from-tag: ${{ needs.build-build-tools-image.outputs.image-tag }}
|
from-tag: ${{ needs.build-build-tools-image.outputs.image-tag }}
|
||||||
secrets: inherit
|
secrets: inherit
|
||||||
|
|
||||||
|
# This job simplifies setting branch protection rules (in GitHub UI)
|
||||||
|
# by allowing to set only this job instead of listing many others.
|
||||||
|
# It also makes it easier to rename or parametrise jobs (using matrix)
|
||||||
|
# which requires changes in branch protection rules
|
||||||
|
#
|
||||||
|
# Note, that we can't add external check (like `neon-cloud-e2e`) we still need to use GitHub UI for that.
|
||||||
|
#
|
||||||
|
# https://github.com/neondatabase/neon/settings/branch_protection_rules
|
||||||
|
conclusion:
|
||||||
|
if: always()
|
||||||
|
# Format `needs` differently to make the list more readable.
|
||||||
|
# Usually we do `needs: [...]`
|
||||||
|
needs:
|
||||||
|
- check-codestyle-python
|
||||||
|
- check-codestyle-rust
|
||||||
|
- regress-tests
|
||||||
|
- test-images
|
||||||
|
runs-on: ubuntu-22.04
|
||||||
|
steps:
|
||||||
|
# The list of possible results:
|
||||||
|
# https://docs.github.com/en/actions/learn-github-actions/contexts#needs-context
|
||||||
|
- name: Fail the job if any of the dependencies do not succeed
|
||||||
|
run: exit 1
|
||||||
|
if: |
|
||||||
|
contains(needs.*.result, 'failure')
|
||||||
|
|| contains(needs.*.result, 'cancelled')
|
||||||
|
|| contains(needs.*.result, 'skipped')
|
||||||
|
|||||||
@@ -232,12 +232,19 @@ jobs:
|
|||||||
|
|
||||||
- name: Run cargo build
|
- name: Run cargo build
|
||||||
run: |
|
run: |
|
||||||
|
PQ_LIB_DIR=$(pwd)/pg_install/v16/lib
|
||||||
|
export PQ_LIB_DIR
|
||||||
mold -run cargo build --locked $CARGO_FLAGS $CARGO_FEATURES --bins --tests -j$(nproc)
|
mold -run cargo build --locked $CARGO_FLAGS $CARGO_FEATURES --bins --tests -j$(nproc)
|
||||||
|
|
||||||
- name: Run cargo test
|
- name: Run cargo test
|
||||||
env:
|
env:
|
||||||
NEXTEST_RETRIES: 3
|
NEXTEST_RETRIES: 3
|
||||||
run: |
|
run: |
|
||||||
|
PQ_LIB_DIR=$(pwd)/pg_install/v16/lib
|
||||||
|
export PQ_LIB_DIR
|
||||||
|
LD_LIBRARY_PATH=$(pwd)/pg_install/v16/lib
|
||||||
|
export LD_LIBRARY_PATH
|
||||||
|
|
||||||
cargo nextest run $CARGO_FEATURES -j$(nproc)
|
cargo nextest run $CARGO_FEATURES -j$(nproc)
|
||||||
|
|
||||||
# Run separate tests for real S3
|
# Run separate tests for real S3
|
||||||
@@ -378,7 +385,7 @@ jobs:
|
|||||||
run: make walproposer-lib -j$(nproc)
|
run: make walproposer-lib -j$(nproc)
|
||||||
|
|
||||||
- name: Produce the build stats
|
- name: Produce the build stats
|
||||||
run: cargo build --all --release --timings -j$(nproc)
|
run: PQ_LIB_DIR=$(pwd)/pg_install/v16/lib cargo build --all --release --timings -j$(nproc)
|
||||||
|
|
||||||
- name: Upload the build stats
|
- name: Upload the build stats
|
||||||
id: upload-stats
|
id: upload-stats
|
||||||
|
|||||||
@@ -0,0 +1,155 @@
|
|||||||
|
name: Periodic pagebench performance test on dedicated EC2 machine in eu-central-1 region
|
||||||
|
|
||||||
|
on:
|
||||||
|
schedule:
|
||||||
|
# * is a special character in YAML so you have to quote this string
|
||||||
|
# ┌───────────── minute (0 - 59)
|
||||||
|
# │ ┌───────────── hour (0 - 23)
|
||||||
|
# │ │ ┌───────────── day of the month (1 - 31)
|
||||||
|
# │ │ │ ┌───────────── month (1 - 12 or JAN-DEC)
|
||||||
|
# │ │ │ │ ┌───────────── day of the week (0 - 6 or SUN-SAT)
|
||||||
|
- cron: '0 18 * * *' # Runs at 6 PM UTC every day
|
||||||
|
workflow_dispatch: # Allows manual triggering of the workflow
|
||||||
|
inputs:
|
||||||
|
commit_hash:
|
||||||
|
type: string
|
||||||
|
description: 'The long neon repo commit hash for the system under test (pageserver) to be tested.'
|
||||||
|
required: false
|
||||||
|
default: ''
|
||||||
|
|
||||||
|
defaults:
|
||||||
|
run:
|
||||||
|
shell: bash -euo pipefail {0}
|
||||||
|
|
||||||
|
concurrency:
|
||||||
|
group: ${{ github.workflow }}
|
||||||
|
cancel-in-progress: false
|
||||||
|
|
||||||
|
jobs:
|
||||||
|
trigger_bench_on_ec2_machine_in_eu_central_1:
|
||||||
|
runs-on: [ self-hosted, gen3, small ]
|
||||||
|
container:
|
||||||
|
image: neondatabase/build-tools:pinned
|
||||||
|
credentials:
|
||||||
|
username: ${{ secrets.NEON_DOCKERHUB_USERNAME }}
|
||||||
|
password: ${{ secrets.NEON_DOCKERHUB_PASSWORD }}
|
||||||
|
options: --init
|
||||||
|
timeout-minutes: 360 # Set the timeout to 6 hours
|
||||||
|
env:
|
||||||
|
API_KEY: ${{ secrets.PERIODIC_PAGEBENCH_EC2_RUNNER_API_KEY }}
|
||||||
|
RUN_ID: ${{ github.run_id }}
|
||||||
|
AWS_ACCESS_KEY_ID: ${{ secrets.AWS_EC2_US_TEST_RUNNER_ACCESS_KEY_ID }}
|
||||||
|
AWS_SECRET_ACCESS_KEY : ${{ secrets.AWS_EC2_US_TEST_RUNNER_ACCESS_KEY_SECRET }}
|
||||||
|
AWS_DEFAULT_REGION : "eu-central-1"
|
||||||
|
AWS_INSTANCE_ID : "i-02a59a3bf86bc7e74"
|
||||||
|
steps:
|
||||||
|
# we don't need the neon source code because we run everything remotely
|
||||||
|
# however we still need the local github actions to run the allure step below
|
||||||
|
- uses: actions/checkout@v4
|
||||||
|
|
||||||
|
- name: Show my own (github runner) external IP address - usefull for IP allowlisting
|
||||||
|
run: curl https://ifconfig.me
|
||||||
|
|
||||||
|
- name: Start EC2 instance and wait for the instance to boot up
|
||||||
|
run: |
|
||||||
|
aws ec2 start-instances --instance-ids $AWS_INSTANCE_ID
|
||||||
|
aws ec2 wait instance-running --instance-ids $AWS_INSTANCE_ID
|
||||||
|
sleep 60 # sleep some time to allow cloudinit and our API server to start up
|
||||||
|
|
||||||
|
- name: Determine public IP of the EC2 instance and set env variable EC2_MACHINE_URL_US
|
||||||
|
run: |
|
||||||
|
public_ip=$(aws ec2 describe-instances --instance-ids $AWS_INSTANCE_ID --query 'Reservations[*].Instances[*].PublicIpAddress' --output text)
|
||||||
|
echo "Public IP of the EC2 instance: $public_ip"
|
||||||
|
echo "EC2_MACHINE_URL_US=https://${public_ip}:8443" >> $GITHUB_ENV
|
||||||
|
|
||||||
|
- name: Determine commit hash
|
||||||
|
env:
|
||||||
|
INPUT_COMMIT_HASH: ${{ github.event.inputs.commit_hash }}
|
||||||
|
run: |
|
||||||
|
if [ -z "$INPUT_COMMIT_HASH" ]; then
|
||||||
|
echo "COMMIT_HASH=$(curl -s https://api.github.com/repos/neondatabase/neon/commits/main | jq -r '.sha')" >> $GITHUB_ENV
|
||||||
|
else
|
||||||
|
echo "COMMIT_HASH=$INPUT_COMMIT_HASH" >> $GITHUB_ENV
|
||||||
|
fi
|
||||||
|
|
||||||
|
- name: Start Bench with run_id
|
||||||
|
run: |
|
||||||
|
curl -k -X 'POST' \
|
||||||
|
"${EC2_MACHINE_URL_US}/start_test/${GITHUB_RUN_ID}" \
|
||||||
|
-H 'accept: application/json' \
|
||||||
|
-H 'Content-Type: application/json' \
|
||||||
|
-H "Authorization: Bearer $API_KEY" \
|
||||||
|
-d "{\"neonRepoCommitHash\": \"${COMMIT_HASH}\"}"
|
||||||
|
|
||||||
|
- name: Poll Test Status
|
||||||
|
id: poll_step
|
||||||
|
run: |
|
||||||
|
status=""
|
||||||
|
while [[ "$status" != "failure" && "$status" != "success" ]]; do
|
||||||
|
response=$(curl -k -X 'GET' \
|
||||||
|
"${EC2_MACHINE_URL_US}/test_status/${GITHUB_RUN_ID}" \
|
||||||
|
-H 'accept: application/json' \
|
||||||
|
-H "Authorization: Bearer $API_KEY")
|
||||||
|
echo "Response: $response"
|
||||||
|
set +x
|
||||||
|
status=$(echo $response | jq -r '.status')
|
||||||
|
echo "Test status: $status"
|
||||||
|
if [[ "$status" == "failure" ]]; then
|
||||||
|
echo "Test failed"
|
||||||
|
exit 1 # Fail the job step if status is failure
|
||||||
|
elif [[ "$status" == "success" || "$status" == "null" ]]; then
|
||||||
|
break
|
||||||
|
elif [[ "$status" == "too_many_runs" ]]; then
|
||||||
|
echo "Too many runs already running"
|
||||||
|
echo "too_many_runs=true" >> "$GITHUB_OUTPUT"
|
||||||
|
exit 1
|
||||||
|
fi
|
||||||
|
|
||||||
|
sleep 60 # Poll every 60 seconds
|
||||||
|
done
|
||||||
|
|
||||||
|
- name: Retrieve Test Logs
|
||||||
|
if: always() && steps.poll_step.outputs.too_many_runs != 'true'
|
||||||
|
run: |
|
||||||
|
curl -k -X 'GET' \
|
||||||
|
"${EC2_MACHINE_URL_US}/test_log/${GITHUB_RUN_ID}" \
|
||||||
|
-H 'accept: application/gzip' \
|
||||||
|
-H "Authorization: Bearer $API_KEY" \
|
||||||
|
--output "test_log_${GITHUB_RUN_ID}.gz"
|
||||||
|
|
||||||
|
- name: Unzip Test Log and Print it into this job's log
|
||||||
|
if: always() && steps.poll_step.outputs.too_many_runs != 'true'
|
||||||
|
run: |
|
||||||
|
gzip -d "test_log_${GITHUB_RUN_ID}.gz"
|
||||||
|
cat "test_log_${GITHUB_RUN_ID}"
|
||||||
|
|
||||||
|
- name: Create Allure report
|
||||||
|
env:
|
||||||
|
AWS_ACCESS_KEY_ID: ${{ secrets.AWS_ACCESS_KEY_DEV }}
|
||||||
|
AWS_SECRET_ACCESS_KEY: ${{ secrets.AWS_SECRET_KEY_DEV }}
|
||||||
|
if: ${{ !cancelled() }}
|
||||||
|
uses: ./.github/actions/allure-report-generate
|
||||||
|
|
||||||
|
- name: Post to a Slack channel
|
||||||
|
if: ${{ github.event.schedule && failure() }}
|
||||||
|
uses: slackapi/slack-github-action@v1
|
||||||
|
with:
|
||||||
|
channel-id: "C033QLM5P7D" # dev-staging-stream
|
||||||
|
slack-message: "Periodic pagebench testing on dedicated hardware: ${{ job.status }}\n${{ github.server_url }}/${{ github.repository }}/actions/runs/${{ github.run_id }}"
|
||||||
|
env:
|
||||||
|
SLACK_BOT_TOKEN: ${{ secrets.SLACK_BOT_TOKEN }}
|
||||||
|
|
||||||
|
- name: Cleanup Test Resources
|
||||||
|
if: always()
|
||||||
|
run: |
|
||||||
|
curl -k -X 'POST' \
|
||||||
|
"${EC2_MACHINE_URL_US}/cleanup_test/${GITHUB_RUN_ID}" \
|
||||||
|
-H 'accept: application/json' \
|
||||||
|
-H "Authorization: Bearer $API_KEY" \
|
||||||
|
-d ''
|
||||||
|
|
||||||
|
- name: Stop EC2 instance and wait for the instance to be stopped
|
||||||
|
if: always() && steps.poll_step.outputs.too_many_runs != 'true'
|
||||||
|
run: |
|
||||||
|
aws ec2 stop-instances --instance-ids $AWS_INSTANCE_ID
|
||||||
|
aws ec2 wait instance-stopped --instance-ids $AWS_INSTANCE_ID
|
||||||
@@ -0,0 +1,115 @@
|
|||||||
|
name: Test Postgres client libraries
|
||||||
|
|
||||||
|
on:
|
||||||
|
schedule:
|
||||||
|
# * is a special character in YAML so you have to quote this string
|
||||||
|
# ┌───────────── minute (0 - 59)
|
||||||
|
# │ ┌───────────── hour (0 - 23)
|
||||||
|
# │ │ ┌───────────── day of the month (1 - 31)
|
||||||
|
# │ │ │ ┌───────────── month (1 - 12 or JAN-DEC)
|
||||||
|
# │ │ │ │ ┌───────────── day of the week (0 - 6 or SUN-SAT)
|
||||||
|
- cron: '23 02 * * *' # run once a day, timezone is utc
|
||||||
|
pull_request:
|
||||||
|
paths:
|
||||||
|
- '.github/workflows/pg-clients.yml'
|
||||||
|
- 'test_runner/pg_clients/**'
|
||||||
|
- 'poetry.lock'
|
||||||
|
workflow_dispatch:
|
||||||
|
|
||||||
|
concurrency:
|
||||||
|
group: ${{ github.workflow }}-${{ github.ref_name }}
|
||||||
|
cancel-in-progress: ${{ github.event_name == 'pull_request' }}
|
||||||
|
|
||||||
|
defaults:
|
||||||
|
run:
|
||||||
|
shell: bash -euxo pipefail {0}
|
||||||
|
|
||||||
|
env:
|
||||||
|
DEFAULT_PG_VERSION: 16
|
||||||
|
PLATFORM: neon-captest-new
|
||||||
|
AWS_ACCESS_KEY_ID: ${{ secrets.AWS_ACCESS_KEY_DEV }}
|
||||||
|
AWS_SECRET_ACCESS_KEY: ${{ secrets.AWS_SECRET_KEY_DEV }}
|
||||||
|
AWS_DEFAULT_REGION: eu-central-1
|
||||||
|
|
||||||
|
jobs:
|
||||||
|
check-permissions:
|
||||||
|
if: ${{ !contains(github.event.pull_request.labels.*.name, 'run-no-ci') }}
|
||||||
|
uses: ./.github/workflows/check-permissions.yml
|
||||||
|
with:
|
||||||
|
github-event-name: ${{ github.event_name }}
|
||||||
|
|
||||||
|
check-build-tools-image:
|
||||||
|
needs: [ check-permissions ]
|
||||||
|
uses: ./.github/workflows/check-build-tools-image.yml
|
||||||
|
|
||||||
|
build-build-tools-image:
|
||||||
|
needs: [ check-build-tools-image ]
|
||||||
|
uses: ./.github/workflows/build-build-tools-image.yml
|
||||||
|
with:
|
||||||
|
image-tag: ${{ needs.check-build-tools-image.outputs.image-tag }}
|
||||||
|
secrets: inherit
|
||||||
|
|
||||||
|
test-postgres-client-libs:
|
||||||
|
needs: [ build-build-tools-image ]
|
||||||
|
runs-on: ubuntu-22.04
|
||||||
|
|
||||||
|
container:
|
||||||
|
image: ${{ needs.build-build-tools-image.outputs.image }}
|
||||||
|
credentials:
|
||||||
|
username: ${{ secrets.NEON_DOCKERHUB_USERNAME }}
|
||||||
|
password: ${{ secrets.NEON_DOCKERHUB_PASSWORD }}
|
||||||
|
options: --init --user root
|
||||||
|
|
||||||
|
steps:
|
||||||
|
- uses: actions/checkout@v4
|
||||||
|
|
||||||
|
- name: Download Neon artifact
|
||||||
|
uses: ./.github/actions/download
|
||||||
|
with:
|
||||||
|
name: neon-${{ runner.os }}-${{ runner.arch }}-release-artifact
|
||||||
|
path: /tmp/neon/
|
||||||
|
prefix: latest
|
||||||
|
|
||||||
|
- name: Create Neon Project
|
||||||
|
id: create-neon-project
|
||||||
|
uses: ./.github/actions/neon-project-create
|
||||||
|
with:
|
||||||
|
api_key: ${{ secrets.NEON_STAGING_API_KEY }}
|
||||||
|
postgres_version: ${{ env.DEFAULT_PG_VERSION }}
|
||||||
|
|
||||||
|
- name: Run tests
|
||||||
|
uses: ./.github/actions/run-python-test-set
|
||||||
|
with:
|
||||||
|
build_type: remote
|
||||||
|
test_selection: pg_clients
|
||||||
|
run_in_parallel: false
|
||||||
|
extra_params: -m remote_cluster
|
||||||
|
pg_version: ${{ env.DEFAULT_PG_VERSION }}
|
||||||
|
env:
|
||||||
|
BENCHMARK_CONNSTR: ${{ steps.create-neon-project.outputs.dsn }}
|
||||||
|
|
||||||
|
- name: Delete Neon Project
|
||||||
|
if: always()
|
||||||
|
uses: ./.github/actions/neon-project-delete
|
||||||
|
with:
|
||||||
|
project_id: ${{ steps.create-neon-project.outputs.project_id }}
|
||||||
|
api_key: ${{ secrets.NEON_STAGING_API_KEY }}
|
||||||
|
|
||||||
|
- name: Create Allure report
|
||||||
|
if: ${{ !cancelled() }}
|
||||||
|
id: create-allure-report
|
||||||
|
uses: ./.github/actions/allure-report-generate
|
||||||
|
with:
|
||||||
|
store-test-results-into-db: true
|
||||||
|
env:
|
||||||
|
REGRESS_TEST_RESULT_CONNSTR_NEW: ${{ secrets.REGRESS_TEST_RESULT_CONNSTR_NEW }}
|
||||||
|
|
||||||
|
- name: Post to a Slack channel
|
||||||
|
if: github.event.schedule && failure()
|
||||||
|
uses: slackapi/slack-github-action@v1
|
||||||
|
with:
|
||||||
|
channel-id: "C06KHQVQ7U3" # on-call-qa-staging-stream
|
||||||
|
slack-message: |
|
||||||
|
Testing Postgres clients: <${{ github.server_url }}/${{ github.repository }}/actions/runs/${{ github.run_id }}|${{ job.status }}> (<${{ steps.create-allure-report.outputs.report-url }}|test report>)
|
||||||
|
env:
|
||||||
|
SLACK_BOT_TOKEN: ${{ secrets.SLACK_BOT_TOKEN }}
|
||||||
@@ -1,98 +0,0 @@
|
|||||||
name: Test Postgres client libraries
|
|
||||||
|
|
||||||
on:
|
|
||||||
schedule:
|
|
||||||
# * is a special character in YAML so you have to quote this string
|
|
||||||
# ┌───────────── minute (0 - 59)
|
|
||||||
# │ ┌───────────── hour (0 - 23)
|
|
||||||
# │ │ ┌───────────── day of the month (1 - 31)
|
|
||||||
# │ │ │ ┌───────────── month (1 - 12 or JAN-DEC)
|
|
||||||
# │ │ │ │ ┌───────────── day of the week (0 - 6 or SUN-SAT)
|
|
||||||
- cron: '23 02 * * *' # run once a day, timezone is utc
|
|
||||||
|
|
||||||
workflow_dispatch:
|
|
||||||
|
|
||||||
concurrency:
|
|
||||||
# Allow only one workflow per any non-`main` branch.
|
|
||||||
group: ${{ github.workflow }}-${{ github.ref_name }}-${{ github.ref_name == 'main' && github.sha || 'anysha' }}
|
|
||||||
cancel-in-progress: true
|
|
||||||
|
|
||||||
jobs:
|
|
||||||
test-postgres-client-libs:
|
|
||||||
# TODO: switch to gen2 runner, requires docker
|
|
||||||
runs-on: ubuntu-22.04
|
|
||||||
|
|
||||||
env:
|
|
||||||
DEFAULT_PG_VERSION: 14
|
|
||||||
TEST_OUTPUT: /tmp/test_output
|
|
||||||
|
|
||||||
steps:
|
|
||||||
- name: Checkout
|
|
||||||
uses: actions/checkout@v4
|
|
||||||
|
|
||||||
- uses: actions/setup-python@v4
|
|
||||||
with:
|
|
||||||
python-version: 3.9
|
|
||||||
|
|
||||||
- name: Install Poetry
|
|
||||||
uses: snok/install-poetry@v1
|
|
||||||
|
|
||||||
- name: Cache poetry deps
|
|
||||||
uses: actions/cache@v4
|
|
||||||
with:
|
|
||||||
path: ~/.cache/pypoetry/virtualenvs
|
|
||||||
key: v2-${{ runner.os }}-${{ runner.arch }}-python-deps-ubunutu-latest-${{ hashFiles('poetry.lock') }}
|
|
||||||
|
|
||||||
- name: Install Python deps
|
|
||||||
shell: bash -euxo pipefail {0}
|
|
||||||
run: ./scripts/pysync
|
|
||||||
|
|
||||||
- name: Create Neon Project
|
|
||||||
id: create-neon-project
|
|
||||||
uses: ./.github/actions/neon-project-create
|
|
||||||
with:
|
|
||||||
api_key: ${{ secrets.NEON_STAGING_API_KEY }}
|
|
||||||
postgres_version: ${{ env.DEFAULT_PG_VERSION }}
|
|
||||||
|
|
||||||
- name: Run pytest
|
|
||||||
env:
|
|
||||||
REMOTE_ENV: 1
|
|
||||||
BENCHMARK_CONNSTR: ${{ steps.create-neon-project.outputs.dsn }}
|
|
||||||
POSTGRES_DISTRIB_DIR: /tmp/neon/pg_install
|
|
||||||
shell: bash -euxo pipefail {0}
|
|
||||||
run: |
|
|
||||||
# Test framework expects we have psql binary;
|
|
||||||
# but since we don't really need it in this test, let's mock it
|
|
||||||
mkdir -p "$POSTGRES_DISTRIB_DIR/v${DEFAULT_PG_VERSION}/bin" && touch "$POSTGRES_DISTRIB_DIR/v${DEFAULT_PG_VERSION}/bin/psql";
|
|
||||||
./scripts/pytest \
|
|
||||||
--junitxml=$TEST_OUTPUT/junit.xml \
|
|
||||||
--tb=short \
|
|
||||||
--verbose \
|
|
||||||
-m "remote_cluster" \
|
|
||||||
-rA "test_runner/pg_clients"
|
|
||||||
|
|
||||||
- name: Delete Neon Project
|
|
||||||
if: ${{ always() }}
|
|
||||||
uses: ./.github/actions/neon-project-delete
|
|
||||||
with:
|
|
||||||
project_id: ${{ steps.create-neon-project.outputs.project_id }}
|
|
||||||
api_key: ${{ secrets.NEON_STAGING_API_KEY }}
|
|
||||||
|
|
||||||
# We use GitHub's action upload-artifact because `ubuntu-latest` doesn't have configured AWS CLI.
|
|
||||||
# It will be fixed after switching to gen2 runner
|
|
||||||
- name: Upload python test logs
|
|
||||||
if: always()
|
|
||||||
uses: actions/upload-artifact@v4
|
|
||||||
with:
|
|
||||||
retention-days: 7
|
|
||||||
name: python-test-pg_clients-${{ runner.os }}-${{ runner.arch }}-stage-logs
|
|
||||||
path: ${{ env.TEST_OUTPUT }}
|
|
||||||
|
|
||||||
- name: Post to a Slack channel
|
|
||||||
if: ${{ github.event.schedule && failure() }}
|
|
||||||
uses: slackapi/slack-github-action@v1
|
|
||||||
with:
|
|
||||||
channel-id: "C033QLM5P7D" # dev-staging-stream
|
|
||||||
slack-message: "Testing Postgres clients: ${{ job.status }}\n${{ github.server_url }}/${{ github.repository }}/actions/runs/${{ github.run_id }}"
|
|
||||||
env:
|
|
||||||
SLACK_BOT_TOKEN: ${{ secrets.SLACK_BOT_TOKEN }}
|
|
||||||
Generated
+1
@@ -6811,6 +6811,7 @@ dependencies = [
|
|||||||
"tokio-stream",
|
"tokio-stream",
|
||||||
"tokio-tar",
|
"tokio-tar",
|
||||||
"tokio-util",
|
"tokio-util",
|
||||||
|
"toml_edit 0.19.10",
|
||||||
"tracing",
|
"tracing",
|
||||||
"tracing-error",
|
"tracing-error",
|
||||||
"tracing-subscriber",
|
"tracing-subscriber",
|
||||||
|
|||||||
+4
-1
@@ -42,12 +42,13 @@ ARG CACHEPOT_BUCKET=neon-github-dev
|
|||||||
COPY --from=pg-build /home/nonroot/pg_install/v14/include/postgresql/server pg_install/v14/include/postgresql/server
|
COPY --from=pg-build /home/nonroot/pg_install/v14/include/postgresql/server pg_install/v14/include/postgresql/server
|
||||||
COPY --from=pg-build /home/nonroot/pg_install/v15/include/postgresql/server pg_install/v15/include/postgresql/server
|
COPY --from=pg-build /home/nonroot/pg_install/v15/include/postgresql/server pg_install/v15/include/postgresql/server
|
||||||
COPY --from=pg-build /home/nonroot/pg_install/v16/include/postgresql/server pg_install/v16/include/postgresql/server
|
COPY --from=pg-build /home/nonroot/pg_install/v16/include/postgresql/server pg_install/v16/include/postgresql/server
|
||||||
|
COPY --from=pg-build /home/nonroot/pg_install/v16/lib pg_install/v16/lib
|
||||||
COPY --chown=nonroot . .
|
COPY --chown=nonroot . .
|
||||||
|
|
||||||
# Show build caching stats to check if it was used in the end.
|
# Show build caching stats to check if it was used in the end.
|
||||||
# Has to be the part of the same RUN since cachepot daemon is killed in the end of this RUN, losing the compilation stats.
|
# Has to be the part of the same RUN since cachepot daemon is killed in the end of this RUN, losing the compilation stats.
|
||||||
RUN set -e \
|
RUN set -e \
|
||||||
&& RUSTFLAGS="-Clinker=clang -Clink-arg=-fuse-ld=mold -Clink-arg=-Wl,--no-rosegment" cargo build \
|
&& PQ_LIB_DIR=$(pwd)/pg_install/v16/lib RUSTFLAGS="-Clinker=clang -Clink-arg=-fuse-ld=mold -Clink-arg=-Wl,--no-rosegment" cargo build \
|
||||||
--bin pg_sni_router \
|
--bin pg_sni_router \
|
||||||
--bin pageserver \
|
--bin pageserver \
|
||||||
--bin pagectl \
|
--bin pagectl \
|
||||||
@@ -56,6 +57,7 @@ RUN set -e \
|
|||||||
--bin storage_controller \
|
--bin storage_controller \
|
||||||
--bin proxy \
|
--bin proxy \
|
||||||
--bin neon_local \
|
--bin neon_local \
|
||||||
|
--bin storage_scrubber \
|
||||||
--locked --release \
|
--locked --release \
|
||||||
&& cachepot -s
|
&& cachepot -s
|
||||||
|
|
||||||
@@ -82,6 +84,7 @@ COPY --from=build --chown=neon:neon /home/nonroot/target/release/storage_broker
|
|||||||
COPY --from=build --chown=neon:neon /home/nonroot/target/release/storage_controller /usr/local/bin
|
COPY --from=build --chown=neon:neon /home/nonroot/target/release/storage_controller /usr/local/bin
|
||||||
COPY --from=build --chown=neon:neon /home/nonroot/target/release/proxy /usr/local/bin
|
COPY --from=build --chown=neon:neon /home/nonroot/target/release/proxy /usr/local/bin
|
||||||
COPY --from=build --chown=neon:neon /home/nonroot/target/release/neon_local /usr/local/bin
|
COPY --from=build --chown=neon:neon /home/nonroot/target/release/neon_local /usr/local/bin
|
||||||
|
COPY --from=build --chown=neon:neon /home/nonroot/target/release/storage_scrubber /usr/local/bin
|
||||||
|
|
||||||
COPY --from=pg-build /home/nonroot/pg_install/v14 /usr/local/v14/
|
COPY --from=pg-build /home/nonroot/pg_install/v14 /usr/local/v14/
|
||||||
COPY --from=pg-build /home/nonroot/pg_install/v15 /usr/local/v15/
|
COPY --from=pg-build /home/nonroot/pg_install/v15 /usr/local/v15/
|
||||||
|
|||||||
+21
-2
@@ -1,5 +1,13 @@
|
|||||||
FROM debian:bullseye-slim
|
FROM debian:bullseye-slim
|
||||||
|
|
||||||
|
# Use ARG as a build-time environment variable here to allow.
|
||||||
|
# It's not supposed to be set outside.
|
||||||
|
# Alternatively it can be obtained using the following command
|
||||||
|
# ```
|
||||||
|
# . /etc/os-release && echo "${VERSION_CODENAME}"
|
||||||
|
# ```
|
||||||
|
ARG DEBIAN_VERSION_CODENAME=bullseye
|
||||||
|
|
||||||
# Add nonroot user
|
# Add nonroot user
|
||||||
RUN useradd -ms /bin/bash nonroot -b /home
|
RUN useradd -ms /bin/bash nonroot -b /home
|
||||||
SHELL ["/bin/bash", "-c"]
|
SHELL ["/bin/bash", "-c"]
|
||||||
@@ -26,7 +34,6 @@ RUN set -e \
|
|||||||
liblzma-dev \
|
liblzma-dev \
|
||||||
libncurses5-dev \
|
libncurses5-dev \
|
||||||
libncursesw5-dev \
|
libncursesw5-dev \
|
||||||
libpq-dev \
|
|
||||||
libreadline-dev \
|
libreadline-dev \
|
||||||
libseccomp-dev \
|
libseccomp-dev \
|
||||||
libsqlite3-dev \
|
libsqlite3-dev \
|
||||||
@@ -67,12 +74,24 @@ RUN curl -sL "https://github.com/peak/s5cmd/releases/download/v${S5CMD_VERSION}/
|
|||||||
# LLVM
|
# LLVM
|
||||||
ENV LLVM_VERSION=18
|
ENV LLVM_VERSION=18
|
||||||
RUN curl -fsSL 'https://apt.llvm.org/llvm-snapshot.gpg.key' | apt-key add - \
|
RUN curl -fsSL 'https://apt.llvm.org/llvm-snapshot.gpg.key' | apt-key add - \
|
||||||
&& echo "deb http://apt.llvm.org/bullseye/ llvm-toolchain-bullseye-${LLVM_VERSION} main" > /etc/apt/sources.list.d/llvm.stable.list \
|
&& echo "deb http://apt.llvm.org/${DEBIAN_VERSION_CODENAME}/ llvm-toolchain-${DEBIAN_VERSION_CODENAME}-${LLVM_VERSION} main" > /etc/apt/sources.list.d/llvm.stable.list \
|
||||||
&& apt update \
|
&& apt update \
|
||||||
&& apt install -y clang-${LLVM_VERSION} llvm-${LLVM_VERSION} \
|
&& apt install -y clang-${LLVM_VERSION} llvm-${LLVM_VERSION} \
|
||||||
&& bash -c 'for f in /usr/bin/clang*-${LLVM_VERSION} /usr/bin/llvm*-${LLVM_VERSION}; do ln -s "${f}" "${f%-${LLVM_VERSION}}"; done' \
|
&& bash -c 'for f in /usr/bin/clang*-${LLVM_VERSION} /usr/bin/llvm*-${LLVM_VERSION}; do ln -s "${f}" "${f%-${LLVM_VERSION}}"; done' \
|
||||||
&& rm -rf /var/lib/apt/lists/* /tmp/* /var/tmp/*
|
&& rm -rf /var/lib/apt/lists/* /tmp/* /var/tmp/*
|
||||||
|
|
||||||
|
# Install docker
|
||||||
|
RUN curl -fsSL https://download.docker.com/linux/ubuntu/gpg | gpg --dearmor -o /usr/share/keyrings/docker-archive-keyring.gpg \
|
||||||
|
&& echo "deb [arch=$(dpkg --print-architecture) signed-by=/usr/share/keyrings/docker-archive-keyring.gpg] https://download.docker.com/linux/debian ${DEBIAN_VERSION_CODENAME} stable" > /etc/apt/sources.list.d/docker.list \
|
||||||
|
&& apt update \
|
||||||
|
&& apt install -y docker-ce docker-ce-cli \
|
||||||
|
&& rm -rf /var/lib/apt/lists/* /tmp/* /var/tmp/*
|
||||||
|
|
||||||
|
# Configure sudo & docker
|
||||||
|
RUN usermod -aG sudo nonroot && \
|
||||||
|
echo '%sudo ALL=(ALL) NOPASSWD:ALL' >> /etc/sudoers && \
|
||||||
|
usermod -aG docker nonroot
|
||||||
|
|
||||||
# AWS CLI
|
# AWS CLI
|
||||||
RUN curl "https://awscli.amazonaws.com/awscli-exe-linux-$(uname -m).zip" -o "awscliv2.zip" \
|
RUN curl "https://awscli.amazonaws.com/awscli-exe-linux-$(uname -m).zip" -o "awscliv2.zip" \
|
||||||
&& unzip -q awscliv2.zip \
|
&& unzip -q awscliv2.zip \
|
||||||
|
|||||||
@@ -873,9 +873,8 @@ impl ComputeNode {
|
|||||||
Ok(())
|
Ok(())
|
||||||
}
|
}
|
||||||
|
|
||||||
// We could've wrapped this around `pg_ctl reload`, but right now we don't use
|
// Wrapped this around `pg_ctl reload`, but right now we don't use
|
||||||
// `pg_ctl` for start / stop, so this just seems much easier to do as we already
|
// `pg_ctl` for start / stop.
|
||||||
// have opened connection to Postgres and superuser access.
|
|
||||||
#[instrument(skip_all)]
|
#[instrument(skip_all)]
|
||||||
fn pg_reload_conf(&self) -> Result<()> {
|
fn pg_reload_conf(&self) -> Result<()> {
|
||||||
let pgctl_bin = Path::new(&self.pgbin).parent().unwrap().join("pg_ctl");
|
let pgctl_bin = Path::new(&self.pgbin).parent().unwrap().join("pg_ctl");
|
||||||
|
|||||||
@@ -489,7 +489,7 @@ pub fn handle_postgres_logs(stderr: std::process::ChildStderr) -> JoinHandle<()>
|
|||||||
/// Read Postgres logs from `stderr` until EOF. Buffer is flushed on one of the following conditions:
|
/// Read Postgres logs from `stderr` until EOF. Buffer is flushed on one of the following conditions:
|
||||||
/// - next line starts with timestamp
|
/// - next line starts with timestamp
|
||||||
/// - EOF
|
/// - EOF
|
||||||
/// - no new lines were written for the last second
|
/// - no new lines were written for the last 100 milliseconds
|
||||||
async fn handle_postgres_logs_async(stderr: tokio::process::ChildStderr) -> Result<()> {
|
async fn handle_postgres_logs_async(stderr: tokio::process::ChildStderr) -> Result<()> {
|
||||||
let mut lines = tokio::io::BufReader::new(stderr).lines();
|
let mut lines = tokio::io::BufReader::new(stderr).lines();
|
||||||
let timeout_duration = Duration::from_millis(100);
|
let timeout_duration = Duration::from_millis(100);
|
||||||
|
|||||||
@@ -21,10 +21,8 @@ use pageserver_api::config::{
|
|||||||
DEFAULT_HTTP_LISTEN_PORT as DEFAULT_PAGESERVER_HTTP_PORT,
|
DEFAULT_HTTP_LISTEN_PORT as DEFAULT_PAGESERVER_HTTP_PORT,
|
||||||
DEFAULT_PG_LISTEN_PORT as DEFAULT_PAGESERVER_PG_PORT,
|
DEFAULT_PG_LISTEN_PORT as DEFAULT_PAGESERVER_PG_PORT,
|
||||||
};
|
};
|
||||||
use pageserver_api::controller_api::PlacementPolicy;
|
use pageserver_api::controller_api::{PlacementPolicy, TenantCreateRequest};
|
||||||
use pageserver_api::models::{
|
use pageserver_api::models::{ShardParameters, TimelineCreateRequest, TimelineInfo};
|
||||||
ShardParameters, TenantCreateRequest, TimelineCreateRequest, TimelineInfo,
|
|
||||||
};
|
|
||||||
use pageserver_api::shard::{ShardCount, ShardStripeSize, TenantShardId};
|
use pageserver_api::shard::{ShardCount, ShardStripeSize, TenantShardId};
|
||||||
use postgres_backend::AuthType;
|
use postgres_backend::AuthType;
|
||||||
use postgres_connection::parse_host_port;
|
use postgres_connection::parse_host_port;
|
||||||
|
|||||||
@@ -325,11 +325,16 @@ impl LocalEnv {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
pub fn pg_bin_dir(&self, pg_version: u32) -> anyhow::Result<PathBuf> {
|
pub fn pg_dir(&self, pg_version: u32, dir_name: &str) -> anyhow::Result<PathBuf> {
|
||||||
Ok(self.pg_distrib_dir(pg_version)?.join("bin"))
|
Ok(self.pg_distrib_dir(pg_version)?.join(dir_name))
|
||||||
}
|
}
|
||||||
|
|
||||||
|
pub fn pg_bin_dir(&self, pg_version: u32) -> anyhow::Result<PathBuf> {
|
||||||
|
self.pg_dir(pg_version, "bin")
|
||||||
|
}
|
||||||
|
|
||||||
pub fn pg_lib_dir(&self, pg_version: u32) -> anyhow::Result<PathBuf> {
|
pub fn pg_lib_dir(&self, pg_version: u32) -> anyhow::Result<PathBuf> {
|
||||||
Ok(self.pg_distrib_dir(pg_version)?.join("lib"))
|
self.pg_dir(pg_version, "lib")
|
||||||
}
|
}
|
||||||
|
|
||||||
pub fn pageserver_bin(&self) -> PathBuf {
|
pub fn pageserver_bin(&self) -> PathBuf {
|
||||||
|
|||||||
@@ -17,8 +17,7 @@ use anyhow::{bail, Context};
|
|||||||
use camino::Utf8PathBuf;
|
use camino::Utf8PathBuf;
|
||||||
use futures::SinkExt;
|
use futures::SinkExt;
|
||||||
use pageserver_api::models::{
|
use pageserver_api::models::{
|
||||||
self, AuxFilePolicy, LocationConfig, ShardParameters, TenantHistorySize, TenantInfo,
|
self, AuxFilePolicy, LocationConfig, TenantHistorySize, TenantInfo, TimelineInfo,
|
||||||
TimelineInfo,
|
|
||||||
};
|
};
|
||||||
use pageserver_api::shard::TenantShardId;
|
use pageserver_api::shard::TenantShardId;
|
||||||
use pageserver_client::mgmt_api;
|
use pageserver_client::mgmt_api;
|
||||||
@@ -397,28 +396,6 @@ impl PageServerNode {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
pub async fn tenant_create(
|
|
||||||
&self,
|
|
||||||
new_tenant_id: TenantId,
|
|
||||||
generation: Option<u32>,
|
|
||||||
settings: HashMap<&str, &str>,
|
|
||||||
) -> anyhow::Result<TenantId> {
|
|
||||||
let config = Self::parse_config(settings.clone())?;
|
|
||||||
|
|
||||||
let request = models::TenantCreateRequest {
|
|
||||||
new_tenant_id: TenantShardId::unsharded(new_tenant_id),
|
|
||||||
generation,
|
|
||||||
config,
|
|
||||||
shard_parameters: ShardParameters::default(),
|
|
||||||
// Placement policy is not meaningful for creations not done via storage controller
|
|
||||||
placement_policy: None,
|
|
||||||
};
|
|
||||||
if !settings.is_empty() {
|
|
||||||
bail!("Unrecognized tenant settings: {settings:?}")
|
|
||||||
}
|
|
||||||
Ok(self.http_client.tenant_create(&request).await?)
|
|
||||||
}
|
|
||||||
|
|
||||||
pub async fn tenant_config(
|
pub async fn tenant_config(
|
||||||
&self,
|
&self,
|
||||||
tenant_id: TenantId,
|
tenant_id: TenantId,
|
||||||
|
|||||||
@@ -5,12 +5,11 @@ use crate::{
|
|||||||
use camino::{Utf8Path, Utf8PathBuf};
|
use camino::{Utf8Path, Utf8PathBuf};
|
||||||
use pageserver_api::{
|
use pageserver_api::{
|
||||||
controller_api::{
|
controller_api::{
|
||||||
NodeConfigureRequest, NodeRegisterRequest, TenantCreateResponse, TenantLocateResponse,
|
NodeConfigureRequest, NodeRegisterRequest, TenantCreateRequest, TenantCreateResponse,
|
||||||
TenantShardMigrateRequest, TenantShardMigrateResponse,
|
TenantLocateResponse, TenantShardMigrateRequest, TenantShardMigrateResponse,
|
||||||
},
|
},
|
||||||
models::{
|
models::{
|
||||||
TenantCreateRequest, TenantShardSplitRequest, TenantShardSplitResponse,
|
TenantShardSplitRequest, TenantShardSplitResponse, TimelineCreateRequest, TimelineInfo,
|
||||||
TimelineCreateRequest, TimelineInfo,
|
|
||||||
},
|
},
|
||||||
shard::{ShardStripeSize, TenantShardId},
|
shard::{ShardStripeSize, TenantShardId},
|
||||||
};
|
};
|
||||||
@@ -156,16 +155,16 @@ impl StorageController {
|
|||||||
.expect("non-Unicode path")
|
.expect("non-Unicode path")
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Find the directory containing postgres binaries, such as `initdb` and `pg_ctl`
|
/// Find the directory containing postgres subdirectories, such `bin` and `lib`
|
||||||
///
|
///
|
||||||
/// This usually uses STORAGE_CONTROLLER_POSTGRES_VERSION of postgres, but will fall back
|
/// This usually uses STORAGE_CONTROLLER_POSTGRES_VERSION of postgres, but will fall back
|
||||||
/// to other versions if that one isn't found. Some automated tests create circumstances
|
/// to other versions if that one isn't found. Some automated tests create circumstances
|
||||||
/// where only one version is available in pg_distrib_dir, such as `test_remote_extensions`.
|
/// where only one version is available in pg_distrib_dir, such as `test_remote_extensions`.
|
||||||
pub async fn get_pg_bin_dir(&self) -> anyhow::Result<Utf8PathBuf> {
|
async fn get_pg_dir(&self, dir_name: &str) -> anyhow::Result<Utf8PathBuf> {
|
||||||
let prefer_versions = [STORAGE_CONTROLLER_POSTGRES_VERSION, 15, 14];
|
let prefer_versions = [STORAGE_CONTROLLER_POSTGRES_VERSION, 15, 14];
|
||||||
|
|
||||||
for v in prefer_versions {
|
for v in prefer_versions {
|
||||||
let path = Utf8PathBuf::from_path_buf(self.env.pg_bin_dir(v)?).unwrap();
|
let path = Utf8PathBuf::from_path_buf(self.env.pg_dir(v, dir_name)?).unwrap();
|
||||||
if tokio::fs::try_exists(&path).await? {
|
if tokio::fs::try_exists(&path).await? {
|
||||||
return Ok(path);
|
return Ok(path);
|
||||||
}
|
}
|
||||||
@@ -173,11 +172,20 @@ impl StorageController {
|
|||||||
|
|
||||||
// Fall through
|
// Fall through
|
||||||
anyhow::bail!(
|
anyhow::bail!(
|
||||||
"Postgres binaries not found in {}",
|
"Postgres directory '{}' not found in {}",
|
||||||
self.env.pg_distrib_dir.display()
|
dir_name,
|
||||||
|
self.env.pg_distrib_dir.display(),
|
||||||
);
|
);
|
||||||
}
|
}
|
||||||
|
|
||||||
|
pub async fn get_pg_bin_dir(&self) -> anyhow::Result<Utf8PathBuf> {
|
||||||
|
self.get_pg_dir("bin").await
|
||||||
|
}
|
||||||
|
|
||||||
|
pub async fn get_pg_lib_dir(&self) -> anyhow::Result<Utf8PathBuf> {
|
||||||
|
self.get_pg_dir("lib").await
|
||||||
|
}
|
||||||
|
|
||||||
/// Readiness check for our postgres process
|
/// Readiness check for our postgres process
|
||||||
async fn pg_isready(&self, pg_bin_dir: &Utf8Path) -> anyhow::Result<bool> {
|
async fn pg_isready(&self, pg_bin_dir: &Utf8Path) -> anyhow::Result<bool> {
|
||||||
let bin_path = pg_bin_dir.join("pg_isready");
|
let bin_path = pg_bin_dir.join("pg_isready");
|
||||||
@@ -230,12 +238,17 @@ impl StorageController {
|
|||||||
.unwrap()
|
.unwrap()
|
||||||
.join("storage_controller_db");
|
.join("storage_controller_db");
|
||||||
let pg_bin_dir = self.get_pg_bin_dir().await?;
|
let pg_bin_dir = self.get_pg_bin_dir().await?;
|
||||||
|
let pg_lib_dir = self.get_pg_lib_dir().await?;
|
||||||
let pg_log_path = pg_data_path.join("postgres.log");
|
let pg_log_path = pg_data_path.join("postgres.log");
|
||||||
|
|
||||||
if !tokio::fs::try_exists(&pg_data_path).await? {
|
if !tokio::fs::try_exists(&pg_data_path).await? {
|
||||||
// Initialize empty database
|
// Initialize empty database
|
||||||
let initdb_path = pg_bin_dir.join("initdb");
|
let initdb_path = pg_bin_dir.join("initdb");
|
||||||
let mut child = Command::new(&initdb_path)
|
let mut child = Command::new(&initdb_path)
|
||||||
|
.envs(vec![
|
||||||
|
("LD_LIBRARY_PATH".to_owned(), pg_lib_dir.to_string()),
|
||||||
|
("DYLD_LIBRARY_PATH".to_owned(), pg_lib_dir.to_string()),
|
||||||
|
])
|
||||||
.args(["-D", pg_data_path.as_ref()])
|
.args(["-D", pg_data_path.as_ref()])
|
||||||
.spawn()
|
.spawn()
|
||||||
.expect("Failed to spawn initdb");
|
.expect("Failed to spawn initdb");
|
||||||
@@ -270,7 +283,10 @@ impl StorageController {
|
|||||||
&self.env.base_data_dir,
|
&self.env.base_data_dir,
|
||||||
pg_bin_dir.join("pg_ctl").as_std_path(),
|
pg_bin_dir.join("pg_ctl").as_std_path(),
|
||||||
db_start_args,
|
db_start_args,
|
||||||
[],
|
vec![
|
||||||
|
("LD_LIBRARY_PATH".to_owned(), pg_lib_dir.to_string()),
|
||||||
|
("DYLD_LIBRARY_PATH".to_owned(), pg_lib_dir.to_string()),
|
||||||
|
],
|
||||||
background_process::InitialPidFile::Create(self.postgres_pid_file()),
|
background_process::InitialPidFile::Create(self.postgres_pid_file()),
|
||||||
retry_timeout,
|
retry_timeout,
|
||||||
|| self.pg_isready(&pg_bin_dir),
|
|| self.pg_isready(&pg_bin_dir),
|
||||||
@@ -325,7 +341,10 @@ impl StorageController {
|
|||||||
&self.env.base_data_dir,
|
&self.env.base_data_dir,
|
||||||
&self.env.storage_controller_bin(),
|
&self.env.storage_controller_bin(),
|
||||||
args,
|
args,
|
||||||
[],
|
vec![
|
||||||
|
("LD_LIBRARY_PATH".to_owned(), pg_lib_dir.to_string()),
|
||||||
|
("DYLD_LIBRARY_PATH".to_owned(), pg_lib_dir.to_string()),
|
||||||
|
],
|
||||||
background_process::InitialPidFile::Create(self.pid_file()),
|
background_process::InitialPidFile::Create(self.pid_file()),
|
||||||
retry_timeout,
|
retry_timeout,
|
||||||
|| async {
|
|| async {
|
||||||
|
|||||||
@@ -4,13 +4,13 @@ use std::{str::FromStr, time::Duration};
|
|||||||
use clap::{Parser, Subcommand};
|
use clap::{Parser, Subcommand};
|
||||||
use pageserver_api::{
|
use pageserver_api::{
|
||||||
controller_api::{
|
controller_api::{
|
||||||
NodeAvailabilityWrapper, NodeDescribeResponse, ShardSchedulingPolicy,
|
NodeAvailabilityWrapper, NodeDescribeResponse, ShardSchedulingPolicy, TenantCreateRequest,
|
||||||
TenantDescribeResponse, TenantPolicyRequest,
|
TenantDescribeResponse, TenantPolicyRequest,
|
||||||
},
|
},
|
||||||
models::{
|
models::{
|
||||||
EvictionPolicy, EvictionPolicyLayerAccessThreshold, LocationConfigSecondary,
|
EvictionPolicy, EvictionPolicyLayerAccessThreshold, LocationConfigSecondary,
|
||||||
ShardParameters, TenantConfig, TenantConfigRequest, TenantCreateRequest,
|
ShardParameters, TenantConfig, TenantConfigRequest, TenantShardSplitRequest,
|
||||||
TenantShardSplitRequest, TenantShardSplitResponse,
|
TenantShardSplitResponse,
|
||||||
},
|
},
|
||||||
shard::{ShardStripeSize, TenantShardId},
|
shard::{ShardStripeSize, TenantShardId},
|
||||||
};
|
};
|
||||||
@@ -336,14 +336,18 @@ async fn main() -> anyhow::Result<()> {
|
|||||||
.await?;
|
.await?;
|
||||||
}
|
}
|
||||||
Command::TenantCreate { tenant_id } => {
|
Command::TenantCreate { tenant_id } => {
|
||||||
vps_client
|
storcon_client
|
||||||
.tenant_create(&TenantCreateRequest {
|
.dispatch(
|
||||||
new_tenant_id: TenantShardId::unsharded(tenant_id),
|
Method::POST,
|
||||||
generation: None,
|
"v1/tenant".to_string(),
|
||||||
shard_parameters: ShardParameters::default(),
|
Some(TenantCreateRequest {
|
||||||
placement_policy: Some(PlacementPolicy::Attached(1)),
|
new_tenant_id: TenantShardId::unsharded(tenant_id),
|
||||||
config: TenantConfig::default(),
|
generation: None,
|
||||||
})
|
shard_parameters: ShardParameters::default(),
|
||||||
|
placement_policy: Some(PlacementPolicy::Attached(1)),
|
||||||
|
config: TenantConfig::default(),
|
||||||
|
}),
|
||||||
|
)
|
||||||
.await?;
|
.await?;
|
||||||
}
|
}
|
||||||
Command::TenantDelete { tenant_id } => {
|
Command::TenantDelete { tenant_id } => {
|
||||||
|
|||||||
@@ -0,0 +1,345 @@
|
|||||||
|
# Graceful Restarts of Storage Controller Managed Clusters
|
||||||
|
|
||||||
|
## Summary
|
||||||
|
This RFC describes new storage controller APIs for draining and filling tenant shards from/on pageserver nodes.
|
||||||
|
It also covers how these new APIs should be used by an orchestrator (e.g. Ansible) in order to implement
|
||||||
|
graceful cluster restarts.
|
||||||
|
|
||||||
|
## Motivation
|
||||||
|
|
||||||
|
Pageserver restarts cause read availablity downtime for tenants.
|
||||||
|
|
||||||
|
For example pageserver-3 @ us-east-1 was unavailable for a randomly
|
||||||
|
picked tenant (which requested on-demand activation) for around 30 seconds
|
||||||
|
during the restart at 2024-04-03 16:37 UTC.
|
||||||
|
|
||||||
|
Note that lots of shutdowns on loaded pageservers do not finish within the
|
||||||
|
[10 second systemd enforced timeout](https://github.com/neondatabase/aws/blob/0a5280b383e43c063d43cbf87fa026543f6d6ad4/.github/ansible/systemd/pageserver.service#L16). This means we are shutting down without flushing ephemeral layers
|
||||||
|
and have to reingest data in order to serve requests after restarting, potentially making first request latencies worse.
|
||||||
|
|
||||||
|
This problem is not yet very acutely felt in storage controller managed pageservers since
|
||||||
|
tenant density is much lower there. However, we are planning on eventually migrating all
|
||||||
|
pageservers to storage controller management, so it makes sense to solve the issue proactively.
|
||||||
|
|
||||||
|
## Requirements
|
||||||
|
|
||||||
|
- Pageserver re-deployments cause minimal downtime for tenants
|
||||||
|
- The storage controller exposes HTTP API hooks for draining and filling tenant shards
|
||||||
|
from a given pageserver. Said hooks can be used by an orchestrator proces or a human operator.
|
||||||
|
- The storage controller exposes some HTTP API to cancel draining and filling background operations.
|
||||||
|
- Failures to drain or fill the node should not be fatal. In such cases, cluster restarts should proceed
|
||||||
|
as usual (with downtime).
|
||||||
|
- Progress of draining/filling is visible through metrics
|
||||||
|
|
||||||
|
## Non Goals
|
||||||
|
|
||||||
|
- Integration with the control plane
|
||||||
|
- Graceful restarts for large non-HA tenants.
|
||||||
|
|
||||||
|
## Impacted Components
|
||||||
|
|
||||||
|
- storage controller
|
||||||
|
- deployment orchestrator (i.e. Ansible)
|
||||||
|
- pageserver (indirectly)
|
||||||
|
|
||||||
|
## Terminology
|
||||||
|
|
||||||
|
** Draining ** is the process through which all tenant shards that can be migrated from a given pageserver
|
||||||
|
are distributed across the rest of the cluster.
|
||||||
|
|
||||||
|
** Filling ** is the symmetric opposite of draining. In this process tenant shards are migrated onto a given
|
||||||
|
pageserver until the cluster reaches a resonable, quiescent distribution of tenant shards across pageservers.
|
||||||
|
|
||||||
|
** Node scheduling policies ** act as constraints to the scheduler. For instance, when a
|
||||||
|
node is set in the `Paused` policy, no further shards will be scheduled on it.
|
||||||
|
|
||||||
|
** Node ** is a pageserver. Term is used interchangeably in this RFC.
|
||||||
|
|
||||||
|
** Deployment orchestrator ** is a generic term for whatever drives our deployments.
|
||||||
|
Currently, it's an Ansible playbook.
|
||||||
|
|
||||||
|
## Background
|
||||||
|
|
||||||
|
### Storage Controller Basics (skip if already familiar)
|
||||||
|
|
||||||
|
Fundamentally, the storage controller is a reconciler which aims to move from the observed mapping between pageservers and tenant shards to an intended mapping. Pageserver nodes and tenant shards metadata is durably persisted in a database, but note that the mapping between the two entities is not durably persisted. Instead, this mapping (*observed state*) is constructed at startup by sending `GET location_config` requests to registered pageservers.
|
||||||
|
|
||||||
|
An internal scheduler maps tenant shards to pageservers while respecting certain constraints. The result of scheduling is the *intent state*. When the intent state changes, a *reconciliation* will inform pageservers about the new assigment via `PUT location_config` requests and will notify the compute via the configured hook.
|
||||||
|
|
||||||
|
### Background Optimizations
|
||||||
|
|
||||||
|
The storage controller performs scheduling optimizations in the background. It will
|
||||||
|
migrate attachments to warm secondaries and replace secondaries in order to balance
|
||||||
|
the cluster out.
|
||||||
|
|
||||||
|
### Reconciliations Concurrency Limiting
|
||||||
|
|
||||||
|
There's a hard limit on the number of reconciles that the storage controller
|
||||||
|
can have in flight at any given time. To get an idea of scales, the limit is
|
||||||
|
128 at the time of writing.
|
||||||
|
|
||||||
|
## Implementation
|
||||||
|
|
||||||
|
Note: this section focuses on the core functionality of the graceful restart process.
|
||||||
|
It doesn't neccesarily describe the most efficient approach. Optimizations are described
|
||||||
|
separately in a later section.
|
||||||
|
|
||||||
|
### Overall Flow
|
||||||
|
|
||||||
|
This section describes how to implement graceful restarts from the perspective
|
||||||
|
of Ansible, the deployment orchestrator. Pageservers are already restarted sequentially.
|
||||||
|
The orchestrator shall implement the following epilogue and prologue steps for each
|
||||||
|
pageserver restart:
|
||||||
|
|
||||||
|
#### Prologue
|
||||||
|
|
||||||
|
The orchestrator shall first fetch the pageserver node id from the control plane or
|
||||||
|
the pageserver it aims to restart directly. Next, it issues an HTTP request
|
||||||
|
to the storage controller in order to start the drain of said pageserver node.
|
||||||
|
All error responses are retried with a short back-off. When a 202 (Accepted)
|
||||||
|
HTTP code is returned, the drain has started. Now the orchestrator polls the
|
||||||
|
node status endpoint exposed by the storage controller in order to await the
|
||||||
|
end of the drain process. When the `policy` field of the node status response
|
||||||
|
becomes `PauseForRestart`, the drain has completed and the orchestrator can
|
||||||
|
proceed with restarting the pageserver.
|
||||||
|
|
||||||
|
The prologue is subject to an overall timeout. It will have a value in the ballpark
|
||||||
|
of minutes. As storage controller managed pageservers become more loaded this timeout
|
||||||
|
will likely have to increase.
|
||||||
|
|
||||||
|
#### Epilogue
|
||||||
|
|
||||||
|
After restarting the pageserver, the orchestrator issues an HTTP request
|
||||||
|
to the storage controller to kick off the filling process. This API call
|
||||||
|
may be retried for all error codes with a short backoff. This also serves
|
||||||
|
as a synchronization primitive as the fill will be refused if the pageserver
|
||||||
|
has not yet re-attached to the storage controller. When a 202(Accepted) HTTP
|
||||||
|
code is returned, the fill has started. Now the orchestrator polls the node
|
||||||
|
status endpoint exposed by the storage controller in order to await the end of
|
||||||
|
the filling process. When the `policy` field of the node status response becomes
|
||||||
|
`Active`, the fill has completed and the orchestrator may proceed to the next pageserver.
|
||||||
|
|
||||||
|
Again, the epilogue is subject to an overall timeout. We can start off with
|
||||||
|
using the same timeout as for the prologue, but can also consider relying on
|
||||||
|
the storage controller's background optimizations with a shorter timeout.
|
||||||
|
|
||||||
|
In the case that the deployment orchestrator times out, it attempts to cancel
|
||||||
|
the fill. This operation shall be retried with a short back-off. If it ultimately
|
||||||
|
fails it will require manual intervention to set the nodes scheduling policy to
|
||||||
|
`NodeSchedulingPolicy::Active`. Not doing that is not immediately problematic,
|
||||||
|
but it constrains the scheduler as mentioned previously.
|
||||||
|
|
||||||
|
### Node Scheduling Policy State Machine
|
||||||
|
|
||||||
|
The state machine below encodes the behaviours discussed above and
|
||||||
|
the various failover situations described in a later section.
|
||||||
|
|
||||||
|
Assuming no failures and/or timeouts the flow should be:
|
||||||
|
`Active -> Draining -> PauseForRestart -> Active -> Filling -> Active`
|
||||||
|
|
||||||
|
```
|
||||||
|
Operator requested drain
|
||||||
|
+-----------------------------------------+
|
||||||
|
| |
|
||||||
|
+-------+-------+ +-------v-------+
|
||||||
|
| | | |
|
||||||
|
| Pause | +-----------> Draining +----------+
|
||||||
|
| | | | | |
|
||||||
|
+---------------+ | +-------+-------+ |
|
||||||
|
| | |
|
||||||
|
| | |
|
||||||
|
Drain requested| | |
|
||||||
|
| |Drain complete | Drain failed
|
||||||
|
| | | Cancelled/PS reattach/Storcon restart
|
||||||
|
| | |
|
||||||
|
+-------+-------+ | |
|
||||||
|
| | | |
|
||||||
|
+-------------+ Active <-----------+------------------+
|
||||||
|
| | | |
|
||||||
|
Fill requested | +---^---^-------+ |
|
||||||
|
| | | |
|
||||||
|
| | | |
|
||||||
|
| | | |
|
||||||
|
| Fill completed| | |
|
||||||
|
| | |PS reattach |
|
||||||
|
| | |after restart |
|
||||||
|
+-------v-------+ | | +-------v-------+
|
||||||
|
| | | | | |
|
||||||
|
| Filling +---------+ +-----------+PauseForRestart|
|
||||||
|
| | | |
|
||||||
|
+---------------+ +---------------+
|
||||||
|
```
|
||||||
|
|
||||||
|
### Draining/Filling APIs
|
||||||
|
|
||||||
|
The storage controller API to trigger the draining of a given node is:
|
||||||
|
`PUT /v1/control/node/:node_id/{drain,fill}`.
|
||||||
|
|
||||||
|
The following HTTP non-success return codes are used.
|
||||||
|
All of them are safely retriable from the perspective of the storage controller.
|
||||||
|
- 404: Requested node was not found
|
||||||
|
- 503: Requested node is known to the storage controller, but unavailable
|
||||||
|
- 412: Drain precondition failed: there is no other node to drain to or the node's schedulling policy forbids draining
|
||||||
|
- 409: A {drain, fill} is already in progress. Only one such background operation
|
||||||
|
is allowed per node.
|
||||||
|
|
||||||
|
When the drain is accepted and commenced a 202 HTTP code is returned.
|
||||||
|
|
||||||
|
Drains and fills shall be cancellable by the deployment orchestrator or a
|
||||||
|
human operator via: `DELETE /v1/control/node/:node_id/{drain,fill}`. A 200
|
||||||
|
response is returned when the cancelation is successful. Errors are retriable.
|
||||||
|
|
||||||
|
### Drain Process
|
||||||
|
|
||||||
|
Before accpeting a drain request the following validations is applied:
|
||||||
|
* Ensure that the node is known the storage controller
|
||||||
|
* Ensure that the schedulling policy is `NodeSchedulingPolicy::Active` or `NodeSchedulingPolicy::Pause`
|
||||||
|
* Ensure that another drain or fill is not already running on the node
|
||||||
|
* Ensure that a drain is possible (i.e. check that there is at least one
|
||||||
|
schedulable node to drain to)
|
||||||
|
|
||||||
|
After accepting the drain, the scheduling policy of the node is set to
|
||||||
|
`NodeSchedulingPolicy::Draining` and persisted in both memory and the database.
|
||||||
|
This disallows the optimizer from adding or removing shards from the node which
|
||||||
|
is desirable to avoid them racing.
|
||||||
|
|
||||||
|
Next, a separate Tokio task is spawned to manage the draining. For each tenant
|
||||||
|
shard attached to the node being drained, demote the node to a secondary and
|
||||||
|
attempt to schedule the node away. Scheduling might fail due to unsatisfiable
|
||||||
|
constraints, but that is fine. Draining is a best effort process since it might
|
||||||
|
not always be possible to cut over all shards.
|
||||||
|
|
||||||
|
Importantly, this task manages the concurrency of issued reconciles in order to
|
||||||
|
avoid drowning out the target pageservers and to allow other important reconciles
|
||||||
|
to proceed.
|
||||||
|
|
||||||
|
Once the triggered reconciles have finished or timed out, set the node's scheduling
|
||||||
|
policy to `NodeSchedulingPolicy::PauseForRestart` to signal the end of the drain.
|
||||||
|
|
||||||
|
A note on non HA tenants: These tenants do not have secondaries, so by the description
|
||||||
|
above, they would not be migrated. It makes sense to skip them (especially the large ones)
|
||||||
|
since, depending on tenant size, this might be more disruptive than the restart since the
|
||||||
|
pageserver we've moved to do will need to on-demand download the entire working set for the tenant.
|
||||||
|
We can consider expanding to small non-HA tenants in the future.
|
||||||
|
|
||||||
|
### Fill Process
|
||||||
|
|
||||||
|
Before accpeting a fill request the following validations is applied:
|
||||||
|
* Ensure that the node is known the storage controller
|
||||||
|
* Ensure that the schedulling policy is `NodeSchedulingPolicy::Active`.
|
||||||
|
This is the only acceptable policy for the fill starting state. When a node re-attaches,
|
||||||
|
it set the scheduling policy to `NodeSchedulingPolicy::Active` if it was equal to
|
||||||
|
`NodeSchedulingPolicy::PauseForRestart` or `NodeSchedulingPolicy::Draining` (possible end states for a node drain).
|
||||||
|
* Ensure that another drain or fill is not already running on the node
|
||||||
|
|
||||||
|
After accepting the drain, the scheduling policy of the node is set to
|
||||||
|
`NodeSchedulingPolicy::Filling` and persisted in both memory and the database.
|
||||||
|
This disallows the optimizer from adding or removing shards from the node which
|
||||||
|
is desirable to avoid them racing.
|
||||||
|
|
||||||
|
Next, a separate Tokio task is spawned to manage the draining. For each tenant
|
||||||
|
shard where the filled node is a secondary, promote the secondary. This is done
|
||||||
|
until we run out of shards or the counts of attached shards become balanced across
|
||||||
|
the cluster.
|
||||||
|
|
||||||
|
Like for draining, the concurrency of spawned reconciles is limited.
|
||||||
|
|
||||||
|
### Failure Modes & Handling
|
||||||
|
|
||||||
|
Failures are generally handled by transition back into the `Active`
|
||||||
|
(neutral) state. This simplifies the implementation greatly at the
|
||||||
|
cost of adding transitions to the state machine. For example, we
|
||||||
|
could detect the `Draining` state upon restart and proceed with a drain,
|
||||||
|
but how should the storage controller know that's what the orchestrator
|
||||||
|
needs still?
|
||||||
|
|
||||||
|
#### Storage Controller Crash
|
||||||
|
|
||||||
|
When the storage controller starts up reset the node scheduling policy
|
||||||
|
of all nodes in states `Draining`, `Filling` or `PauseForRestart` to
|
||||||
|
`Active`. The rationale is that when the storage controller restarts,
|
||||||
|
we have lost context of what the deployment orchestrator wants. It also
|
||||||
|
has the benefit of making things easier to reason about.
|
||||||
|
|
||||||
|
#### Pageserver Crash During Drain
|
||||||
|
|
||||||
|
The pageserver will attempt to re-attach during restart at which
|
||||||
|
point the node scheduling policy will be set back to `Active`, thus
|
||||||
|
reenabling the scheduler to use the node.
|
||||||
|
|
||||||
|
#### Non-drained Pageserver Crash During Drain
|
||||||
|
|
||||||
|
What should happen when a pageserver we are draining to crashes during the
|
||||||
|
process. Two reasonable options are: cancel the drain and focus on the failover
|
||||||
|
*or* do both, but prioritise failover. Since the number of concurrent reconciles
|
||||||
|
produced by drains/fills are limited, we get the later behaviour for free.
|
||||||
|
My suggestion is we take this approach, but the cancellation option is trivial
|
||||||
|
to implement as well.
|
||||||
|
|
||||||
|
#### Pageserver Crash During Fill
|
||||||
|
|
||||||
|
The pageserver will attempt to re-attach during restart at which
|
||||||
|
point the node scheduling policy will be set back to `Active`, thus
|
||||||
|
reenabling the scheduler to use the node.
|
||||||
|
|
||||||
|
#### Pageserver Goes unavailable During Drain/Fill
|
||||||
|
|
||||||
|
The drain and fill jobs handle this by stopping early. When the pageserver
|
||||||
|
is detected as online by storage controller heartbeats, reset its scheduling
|
||||||
|
policy to `Active`. If a restart happens instead, see the pageserver crash
|
||||||
|
failure mode.
|
||||||
|
|
||||||
|
#### Orchestrator Drain Times Out
|
||||||
|
|
||||||
|
Orchestrator will still proceed with the restart.
|
||||||
|
When the pageserver re-attaches, the scheduling policy is set back to
|
||||||
|
`Active`.
|
||||||
|
|
||||||
|
#### Orchestrator Fill Times Out
|
||||||
|
|
||||||
|
Orchestrator will attempt to cancel the fill operation. If that fails,
|
||||||
|
the fill will continue until it quiesces and the node will be left
|
||||||
|
in the `Filling` scheduling policy. This hinders the scheduler, but is
|
||||||
|
otherwise harmless. A human operator can handle this by setting the scheduling
|
||||||
|
policy to `Active`, or we can bake in a fill timeout into the storage controller.
|
||||||
|
|
||||||
|
## Optimizations
|
||||||
|
|
||||||
|
### Location Warmth
|
||||||
|
|
||||||
|
When cutting over to a secondary, the storage controller will wait for it to
|
||||||
|
become "warm" (i.e. download enough of the tenants data). This means that some
|
||||||
|
reconciliations can take significantly longer than others and hold up precious
|
||||||
|
reconciliations units. As an optimization, the drain stage can only cut over
|
||||||
|
tenants that are already "warm". Similarly, the fill stage can prioritise the
|
||||||
|
"warmest" tenants in the fill.
|
||||||
|
|
||||||
|
Given that the number of tenants by the storage controller will be fairly low
|
||||||
|
for the foreseable future, the first implementation could simply query the tenants
|
||||||
|
for secondary status. This doesn't scale well with increasing tenant counts, so
|
||||||
|
eventually we will need new pageserver API endpoints to report the sets of
|
||||||
|
"warm" and "cold" nodes.
|
||||||
|
|
||||||
|
## Alternatives Considered
|
||||||
|
|
||||||
|
### Draining and Filling Purely as Scheduling Constraints
|
||||||
|
|
||||||
|
At its core, the storage controller is a big background loop that detects changes
|
||||||
|
in the environment and reacts on them. One could express draining and filling
|
||||||
|
of nodes purely in terms of constraining the scheduler (as opposed to having
|
||||||
|
such background tasks).
|
||||||
|
|
||||||
|
While theoretically nice, I think that's harder to implement and more importantly operate and reason about.
|
||||||
|
Consider cancellation of a drain/fill operation. We would have to update the scheduler state, create
|
||||||
|
an entirely new schedule (intent state) and start work on applying that. It gets trickier if we wish
|
||||||
|
to cancel the reconciliation tasks spawned by drain/fill nodes. How would we know which ones belong
|
||||||
|
to the conceptual drain/fill? One could add labels to reconciliations, but it gets messy in my opinion.
|
||||||
|
|
||||||
|
It would also mean that reconciliations themselves have side effects that persist in the database
|
||||||
|
(persist something to the databse when the drain is done), which I'm not conceptually fond of.
|
||||||
|
|
||||||
|
## Proof of Concept
|
||||||
|
|
||||||
|
This RFC is accompanied by a POC which implements nearly everything mentioned here
|
||||||
|
apart from the optimizations and some of the failure handling:
|
||||||
|
https://github.com/neondatabase/neon/pull/7682
|
||||||
@@ -11,6 +11,27 @@ use crate::{
|
|||||||
shard::{ShardStripeSize, TenantShardId},
|
shard::{ShardStripeSize, TenantShardId},
|
||||||
};
|
};
|
||||||
|
|
||||||
|
#[derive(Serialize, Deserialize, Debug)]
|
||||||
|
#[serde(deny_unknown_fields)]
|
||||||
|
pub struct TenantCreateRequest {
|
||||||
|
pub new_tenant_id: TenantShardId,
|
||||||
|
#[serde(default)]
|
||||||
|
#[serde(skip_serializing_if = "Option::is_none")]
|
||||||
|
pub generation: Option<u32>,
|
||||||
|
|
||||||
|
// If omitted, create a single shard with TenantShardId::unsharded()
|
||||||
|
#[serde(default)]
|
||||||
|
#[serde(skip_serializing_if = "ShardParameters::is_unsharded")]
|
||||||
|
pub shard_parameters: ShardParameters,
|
||||||
|
|
||||||
|
#[serde(default)]
|
||||||
|
#[serde(skip_serializing_if = "Option::is_none")]
|
||||||
|
pub placement_policy: Option<PlacementPolicy>,
|
||||||
|
|
||||||
|
#[serde(flatten)]
|
||||||
|
pub config: TenantConfig, // as we have a flattened field, we should reject all unknown fields in it
|
||||||
|
}
|
||||||
|
|
||||||
#[derive(Serialize, Deserialize)]
|
#[derive(Serialize, Deserialize)]
|
||||||
pub struct TenantCreateResponseShard {
|
pub struct TenantCreateResponseShard {
|
||||||
pub shard_id: TenantShardId,
|
pub shard_id: TenantShardId,
|
||||||
@@ -280,4 +301,19 @@ mod test {
|
|||||||
assert_eq!(serde_json::from_str::<PlacementPolicy>(&encoded)?, v);
|
assert_eq!(serde_json::from_str::<PlacementPolicy>(&encoded)?, v);
|
||||||
Ok(())
|
Ok(())
|
||||||
}
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn test_reject_unknown_field() {
|
||||||
|
let id = TenantId::generate();
|
||||||
|
let create_request = serde_json::json!({
|
||||||
|
"new_tenant_id": id.to_string(),
|
||||||
|
"unknown_field": "unknown_value".to_string(),
|
||||||
|
});
|
||||||
|
let err = serde_json::from_value::<TenantCreateRequest>(create_request).unwrap_err();
|
||||||
|
assert!(
|
||||||
|
err.to_string().contains("unknown field `unknown_field`"),
|
||||||
|
"expect unknown field `unknown_field` error, got: {}",
|
||||||
|
err
|
||||||
|
);
|
||||||
|
}
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -29,7 +29,7 @@ pub const KEY_SIZE: usize = 18;
|
|||||||
/// See [`Key::to_i128`] for more information on the encoding.
|
/// See [`Key::to_i128`] for more information on the encoding.
|
||||||
pub const METADATA_KEY_SIZE: usize = 16;
|
pub const METADATA_KEY_SIZE: usize = 16;
|
||||||
|
|
||||||
/// The key prefix start range for the metadata keys. All keys with the first byte >= 0x40 is a metadata key.
|
/// The key prefix start range for the metadata keys. All keys with the first byte >= 0x60 is a metadata key.
|
||||||
pub const METADATA_KEY_BEGIN_PREFIX: u8 = 0x60;
|
pub const METADATA_KEY_BEGIN_PREFIX: u8 = 0x60;
|
||||||
pub const METADATA_KEY_END_PREFIX: u8 = 0x7F;
|
pub const METADATA_KEY_END_PREFIX: u8 = 0x7F;
|
||||||
|
|
||||||
|
|||||||
@@ -17,6 +17,16 @@ pub struct KeySpace {
|
|||||||
pub ranges: Vec<Range<Key>>,
|
pub ranges: Vec<Range<Key>>,
|
||||||
}
|
}
|
||||||
|
|
||||||
|
impl std::fmt::Display for KeySpace {
|
||||||
|
fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result {
|
||||||
|
write!(f, "[")?;
|
||||||
|
for range in &self.ranges {
|
||||||
|
write!(f, "{}..{},", range.start, range.end)?;
|
||||||
|
}
|
||||||
|
write!(f, "]")
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
/// A wrapper type for sparse keyspaces.
|
/// A wrapper type for sparse keyspaces.
|
||||||
#[derive(Clone, Debug, Default, PartialEq, Eq)]
|
#[derive(Clone, Debug, Default, PartialEq, Eq)]
|
||||||
pub struct SparseKeySpace(pub KeySpace);
|
pub struct SparseKeySpace(pub KeySpace);
|
||||||
|
|||||||
@@ -9,6 +9,7 @@ use std::{
|
|||||||
collections::HashMap,
|
collections::HashMap,
|
||||||
io::{BufRead, Read},
|
io::{BufRead, Read},
|
||||||
num::{NonZeroU64, NonZeroUsize},
|
num::{NonZeroU64, NonZeroUsize},
|
||||||
|
str::FromStr,
|
||||||
sync::atomic::AtomicUsize,
|
sync::atomic::AtomicUsize,
|
||||||
time::{Duration, SystemTime},
|
time::{Duration, SystemTime},
|
||||||
};
|
};
|
||||||
@@ -25,7 +26,6 @@ use utils::{
|
|||||||
serde_system_time,
|
serde_system_time,
|
||||||
};
|
};
|
||||||
|
|
||||||
use crate::controller_api::PlacementPolicy;
|
|
||||||
use crate::{
|
use crate::{
|
||||||
reltag::RelTag,
|
reltag::RelTag,
|
||||||
shard::{ShardCount, ShardStripeSize, TenantShardId},
|
shard::{ShardCount, ShardStripeSize, TenantShardId},
|
||||||
@@ -229,6 +229,11 @@ pub struct TimelineCreateRequest {
|
|||||||
pub pg_version: Option<u32>,
|
pub pg_version: Option<u32>,
|
||||||
}
|
}
|
||||||
|
|
||||||
|
#[derive(Serialize, Deserialize, Clone)]
|
||||||
|
pub struct LsnLeaseRequest {
|
||||||
|
pub lsn: Lsn,
|
||||||
|
}
|
||||||
|
|
||||||
#[derive(Serialize, Deserialize)]
|
#[derive(Serialize, Deserialize)]
|
||||||
pub struct TenantShardSplitRequest {
|
pub struct TenantShardSplitRequest {
|
||||||
pub new_shard_count: u8,
|
pub new_shard_count: u8,
|
||||||
@@ -271,28 +276,6 @@ impl Default for ShardParameters {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
#[derive(Serialize, Deserialize, Debug)]
|
|
||||||
#[serde(deny_unknown_fields)]
|
|
||||||
pub struct TenantCreateRequest {
|
|
||||||
pub new_tenant_id: TenantShardId,
|
|
||||||
#[serde(default)]
|
|
||||||
#[serde(skip_serializing_if = "Option::is_none")]
|
|
||||||
pub generation: Option<u32>,
|
|
||||||
|
|
||||||
// If omitted, create a single shard with TenantShardId::unsharded()
|
|
||||||
#[serde(default)]
|
|
||||||
#[serde(skip_serializing_if = "ShardParameters::is_unsharded")]
|
|
||||||
pub shard_parameters: ShardParameters,
|
|
||||||
|
|
||||||
// This parameter is only meaningful in requests sent to the storage controller
|
|
||||||
#[serde(default)]
|
|
||||||
#[serde(skip_serializing_if = "Option::is_none")]
|
|
||||||
pub placement_policy: Option<PlacementPolicy>,
|
|
||||||
|
|
||||||
#[serde(flatten)]
|
|
||||||
pub config: TenantConfig, // as we have a flattened field, we should reject all unknown fields in it
|
|
||||||
}
|
|
||||||
|
|
||||||
/// An alternative representation of `pageserver::tenant::TenantConf` with
|
/// An alternative representation of `pageserver::tenant::TenantConf` with
|
||||||
/// simpler types.
|
/// simpler types.
|
||||||
#[derive(Serialize, Deserialize, Debug, Default, Clone, Eq, PartialEq)]
|
#[derive(Serialize, Deserialize, Debug, Default, Clone, Eq, PartialEq)]
|
||||||
@@ -455,6 +438,51 @@ pub enum CompactionAlgorithm {
|
|||||||
Tiered,
|
Tiered,
|
||||||
}
|
}
|
||||||
|
|
||||||
|
#[derive(Debug, Clone, Copy, PartialEq, Eq, Serialize, Deserialize)]
|
||||||
|
pub enum ImageCompressionAlgorithm {
|
||||||
|
/// Disabled for writes, and never decompress during reading.
|
||||||
|
/// Never set this after you've enabled compression once!
|
||||||
|
DisabledNoDecompress,
|
||||||
|
// Disabled for writes, support decompressing during read path
|
||||||
|
Disabled,
|
||||||
|
/// Zstandard compression. Level 0 means and None mean the same (default level). Levels can be negative as well.
|
||||||
|
/// For details, see the [manual](http://facebook.github.io/zstd/zstd_manual.html).
|
||||||
|
Zstd {
|
||||||
|
level: Option<i8>,
|
||||||
|
},
|
||||||
|
}
|
||||||
|
|
||||||
|
impl ImageCompressionAlgorithm {
|
||||||
|
pub fn allow_decompression(&self) -> bool {
|
||||||
|
!matches!(self, ImageCompressionAlgorithm::DisabledNoDecompress)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
impl FromStr for ImageCompressionAlgorithm {
|
||||||
|
type Err = anyhow::Error;
|
||||||
|
fn from_str(s: &str) -> Result<Self, Self::Err> {
|
||||||
|
let mut components = s.split(['(', ')']);
|
||||||
|
let first = components
|
||||||
|
.next()
|
||||||
|
.ok_or_else(|| anyhow::anyhow!("empty string"))?;
|
||||||
|
match first {
|
||||||
|
"disabled-no-decompress" => Ok(ImageCompressionAlgorithm::DisabledNoDecompress),
|
||||||
|
"disabled" => Ok(ImageCompressionAlgorithm::Disabled),
|
||||||
|
"zstd" => {
|
||||||
|
let level = if let Some(v) = components.next() {
|
||||||
|
let v: i8 = v.parse()?;
|
||||||
|
Some(v)
|
||||||
|
} else {
|
||||||
|
None
|
||||||
|
};
|
||||||
|
|
||||||
|
Ok(ImageCompressionAlgorithm::Zstd { level })
|
||||||
|
}
|
||||||
|
_ => anyhow::bail!("invalid specifier '{first}'"),
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
#[derive(Eq, PartialEq, Debug, Clone, Serialize, Deserialize)]
|
#[derive(Eq, PartialEq, Debug, Clone, Serialize, Deserialize)]
|
||||||
pub struct CompactionAlgorithmSettings {
|
pub struct CompactionAlgorithmSettings {
|
||||||
pub kind: CompactionAlgorithm,
|
pub kind: CompactionAlgorithm,
|
||||||
@@ -547,10 +575,6 @@ pub struct LocationConfigListResponse {
|
|||||||
pub tenant_shards: Vec<(TenantShardId, Option<LocationConfig>)>,
|
pub tenant_shards: Vec<(TenantShardId, Option<LocationConfig>)>,
|
||||||
}
|
}
|
||||||
|
|
||||||
#[derive(Serialize, Deserialize)]
|
|
||||||
#[serde(transparent)]
|
|
||||||
pub struct TenantCreateResponse(pub TenantId);
|
|
||||||
|
|
||||||
#[derive(Serialize)]
|
#[derive(Serialize)]
|
||||||
pub struct StatusResponse {
|
pub struct StatusResponse {
|
||||||
pub id: NodeId,
|
pub id: NodeId,
|
||||||
@@ -670,6 +694,16 @@ pub struct TimelineInfo {
|
|||||||
pub current_physical_size: Option<u64>, // is None when timeline is Unloaded
|
pub current_physical_size: Option<u64>, // is None when timeline is Unloaded
|
||||||
pub current_logical_size_non_incremental: Option<u64>,
|
pub current_logical_size_non_incremental: Option<u64>,
|
||||||
|
|
||||||
|
/// How many bytes of WAL are within this branch's pitr_interval. If the pitr_interval goes
|
||||||
|
/// beyond the branch's branch point, we only count up to the branch point.
|
||||||
|
pub pitr_history_size: u64,
|
||||||
|
|
||||||
|
/// Whether this branch's branch point is within its ancestor's PITR interval (i.e. any
|
||||||
|
/// ancestor data used by this branch would have been retained anyway). If this is false, then
|
||||||
|
/// this branch may be imposing a cost on the ancestor by causing it to retain layers that it would
|
||||||
|
/// otherwise be able to GC.
|
||||||
|
pub within_ancestor_pitr: bool,
|
||||||
|
|
||||||
pub timeline_dir_layer_file_size_sum: Option<u64>,
|
pub timeline_dir_layer_file_size_sum: Option<u64>,
|
||||||
|
|
||||||
pub wal_source_connstr: Option<String>,
|
pub wal_source_connstr: Option<String>,
|
||||||
@@ -1507,18 +1541,6 @@ mod tests {
|
|||||||
|
|
||||||
#[test]
|
#[test]
|
||||||
fn test_reject_unknown_field() {
|
fn test_reject_unknown_field() {
|
||||||
let id = TenantId::generate();
|
|
||||||
let create_request = json!({
|
|
||||||
"new_tenant_id": id.to_string(),
|
|
||||||
"unknown_field": "unknown_value".to_string(),
|
|
||||||
});
|
|
||||||
let err = serde_json::from_value::<TenantCreateRequest>(create_request).unwrap_err();
|
|
||||||
assert!(
|
|
||||||
err.to_string().contains("unknown field `unknown_field`"),
|
|
||||||
"expect unknown field `unknown_field` error, got: {}",
|
|
||||||
err
|
|
||||||
);
|
|
||||||
|
|
||||||
let id = TenantId::generate();
|
let id = TenantId::generate();
|
||||||
let config_request = json!({
|
let config_request = json!({
|
||||||
"tenant_id": id.to_string(),
|
"tenant_id": id.to_string(),
|
||||||
@@ -1653,4 +1675,29 @@ mod tests {
|
|||||||
AuxFilePolicy::CrossValidation
|
AuxFilePolicy::CrossValidation
|
||||||
);
|
);
|
||||||
}
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn test_image_compression_algorithm_parsing() {
|
||||||
|
use ImageCompressionAlgorithm::*;
|
||||||
|
assert_eq!(
|
||||||
|
ImageCompressionAlgorithm::from_str("disabled").unwrap(),
|
||||||
|
Disabled
|
||||||
|
);
|
||||||
|
assert_eq!(
|
||||||
|
ImageCompressionAlgorithm::from_str("disabled-no-decompress").unwrap(),
|
||||||
|
DisabledNoDecompress
|
||||||
|
);
|
||||||
|
assert_eq!(
|
||||||
|
ImageCompressionAlgorithm::from_str("zstd").unwrap(),
|
||||||
|
Zstd { level: None }
|
||||||
|
);
|
||||||
|
assert_eq!(
|
||||||
|
ImageCompressionAlgorithm::from_str("zstd(18)").unwrap(),
|
||||||
|
Zstd { level: Some(18) }
|
||||||
|
);
|
||||||
|
assert_eq!(
|
||||||
|
ImageCompressionAlgorithm::from_str("zstd(-3)").unwrap(),
|
||||||
|
Zstd { level: Some(-3) }
|
||||||
|
);
|
||||||
|
}
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -356,6 +356,28 @@ impl CheckPoint {
|
|||||||
}
|
}
|
||||||
false
|
false
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// Advance next multi-XID/offset to those given in arguments.
|
||||||
|
///
|
||||||
|
/// It's important that this handles wraparound correctly. This should match the
|
||||||
|
/// MultiXactAdvanceNextMXact() logic in PostgreSQL's xlog_redo() function.
|
||||||
|
///
|
||||||
|
/// Returns 'true' if the Checkpoint was updated.
|
||||||
|
pub fn update_next_multixid(&mut self, multi_xid: u32, multi_offset: u32) -> bool {
|
||||||
|
let mut modified = false;
|
||||||
|
|
||||||
|
if multi_xid.wrapping_sub(self.nextMulti) as i32 > 0 {
|
||||||
|
self.nextMulti = multi_xid;
|
||||||
|
modified = true;
|
||||||
|
}
|
||||||
|
|
||||||
|
if multi_offset.wrapping_sub(self.nextMultiOffset) as i32 > 0 {
|
||||||
|
self.nextMultiOffset = multi_offset;
|
||||||
|
modified = true;
|
||||||
|
}
|
||||||
|
|
||||||
|
modified
|
||||||
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Generate new, empty WAL segment, with correct block headers at the first
|
/// Generate new, empty WAL segment, with correct block headers at the first
|
||||||
|
|||||||
@@ -202,6 +202,53 @@ pub fn test_update_next_xid() {
|
|||||||
assert_eq!(checkpoint.nextXid.value, 2048);
|
assert_eq!(checkpoint.nextXid.value, 2048);
|
||||||
}
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
pub fn test_update_next_multixid() {
|
||||||
|
let checkpoint_buf = [0u8; std::mem::size_of::<CheckPoint>()];
|
||||||
|
let mut checkpoint = CheckPoint::decode(&checkpoint_buf).unwrap();
|
||||||
|
|
||||||
|
// simple case
|
||||||
|
checkpoint.nextMulti = 20;
|
||||||
|
checkpoint.nextMultiOffset = 20;
|
||||||
|
checkpoint.update_next_multixid(1000, 2000);
|
||||||
|
assert_eq!(checkpoint.nextMulti, 1000);
|
||||||
|
assert_eq!(checkpoint.nextMultiOffset, 2000);
|
||||||
|
|
||||||
|
// No change
|
||||||
|
checkpoint.update_next_multixid(500, 900);
|
||||||
|
assert_eq!(checkpoint.nextMulti, 1000);
|
||||||
|
assert_eq!(checkpoint.nextMultiOffset, 2000);
|
||||||
|
|
||||||
|
// Close to wraparound, but not wrapped around yet
|
||||||
|
checkpoint.nextMulti = 0xffff0000;
|
||||||
|
checkpoint.nextMultiOffset = 0xfffe0000;
|
||||||
|
checkpoint.update_next_multixid(0xffff00ff, 0xfffe00ff);
|
||||||
|
assert_eq!(checkpoint.nextMulti, 0xffff00ff);
|
||||||
|
assert_eq!(checkpoint.nextMultiOffset, 0xfffe00ff);
|
||||||
|
|
||||||
|
// Wraparound
|
||||||
|
checkpoint.update_next_multixid(1, 900);
|
||||||
|
assert_eq!(checkpoint.nextMulti, 1);
|
||||||
|
assert_eq!(checkpoint.nextMultiOffset, 900);
|
||||||
|
|
||||||
|
// Wraparound nextMulti to 0.
|
||||||
|
//
|
||||||
|
// It's a bit surprising that nextMulti can be 0, because that's a special value
|
||||||
|
// (InvalidMultiXactId). However, that's how Postgres does it at multi-xid wraparound:
|
||||||
|
// nextMulti wraps around to 0, but then when the next multi-xid is assigned, it skips
|
||||||
|
// the 0 and the next multi-xid actually assigned is 1.
|
||||||
|
checkpoint.nextMulti = 0xffff0000;
|
||||||
|
checkpoint.nextMultiOffset = 0xfffe0000;
|
||||||
|
checkpoint.update_next_multixid(0, 0xfffe00ff);
|
||||||
|
assert_eq!(checkpoint.nextMulti, 0);
|
||||||
|
assert_eq!(checkpoint.nextMultiOffset, 0xfffe00ff);
|
||||||
|
|
||||||
|
// Wraparound nextMultiOffset to 0
|
||||||
|
checkpoint.update_next_multixid(0, 0);
|
||||||
|
assert_eq!(checkpoint.nextMulti, 0);
|
||||||
|
assert_eq!(checkpoint.nextMultiOffset, 0);
|
||||||
|
}
|
||||||
|
|
||||||
#[test]
|
#[test]
|
||||||
pub fn test_encode_logical_message() {
|
pub fn test_encode_logical_message() {
|
||||||
let expected = [
|
let expected = [
|
||||||
|
|||||||
@@ -1,6 +1,5 @@
|
|||||||
use std::{fmt::Debug, num::NonZeroUsize, str::FromStr, time::Duration};
|
use std::{fmt::Debug, num::NonZeroUsize, str::FromStr, time::Duration};
|
||||||
|
|
||||||
use anyhow::bail;
|
|
||||||
use aws_sdk_s3::types::StorageClass;
|
use aws_sdk_s3::types::StorageClass;
|
||||||
use camino::Utf8PathBuf;
|
use camino::Utf8PathBuf;
|
||||||
|
|
||||||
@@ -176,20 +175,8 @@ fn serialize_storage_class<S: serde::Serializer>(
|
|||||||
impl RemoteStorageConfig {
|
impl RemoteStorageConfig {
|
||||||
pub const DEFAULT_TIMEOUT: Duration = std::time::Duration::from_secs(120);
|
pub const DEFAULT_TIMEOUT: Duration = std::time::Duration::from_secs(120);
|
||||||
|
|
||||||
pub fn from_toml(toml: &toml_edit::Item) -> anyhow::Result<Option<RemoteStorageConfig>> {
|
pub fn from_toml(toml: &toml_edit::Item) -> anyhow::Result<RemoteStorageConfig> {
|
||||||
let document: toml_edit::Document = match toml {
|
Ok(utils::toml_edit_ext::deserialize_item(toml)?)
|
||||||
toml_edit::Item::Table(toml) => toml.clone().into(),
|
|
||||||
toml_edit::Item::Value(toml_edit::Value::InlineTable(toml)) => {
|
|
||||||
toml.clone().into_table().into()
|
|
||||||
}
|
|
||||||
_ => bail!("toml not a table or inline table"),
|
|
||||||
};
|
|
||||||
|
|
||||||
if document.is_empty() {
|
|
||||||
return Ok(None);
|
|
||||||
}
|
|
||||||
|
|
||||||
Ok(Some(toml_edit::de::from_document(document)?))
|
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -197,7 +184,7 @@ impl RemoteStorageConfig {
|
|||||||
mod tests {
|
mod tests {
|
||||||
use super::*;
|
use super::*;
|
||||||
|
|
||||||
fn parse(input: &str) -> anyhow::Result<Option<RemoteStorageConfig>> {
|
fn parse(input: &str) -> anyhow::Result<RemoteStorageConfig> {
|
||||||
let toml = input.parse::<toml_edit::Document>().unwrap();
|
let toml = input.parse::<toml_edit::Document>().unwrap();
|
||||||
RemoteStorageConfig::from_toml(toml.as_item())
|
RemoteStorageConfig::from_toml(toml.as_item())
|
||||||
}
|
}
|
||||||
@@ -207,7 +194,7 @@ mod tests {
|
|||||||
let input = "local_path = '.'
|
let input = "local_path = '.'
|
||||||
timeout = '5s'";
|
timeout = '5s'";
|
||||||
|
|
||||||
let config = parse(input).unwrap().expect("it exists");
|
let config = parse(input).unwrap();
|
||||||
|
|
||||||
assert_eq!(
|
assert_eq!(
|
||||||
config,
|
config,
|
||||||
@@ -229,7 +216,7 @@ timeout = '5s'";
|
|||||||
timeout = '7s'
|
timeout = '7s'
|
||||||
";
|
";
|
||||||
|
|
||||||
let config = parse(toml).unwrap().expect("it exists");
|
let config = parse(toml).unwrap();
|
||||||
|
|
||||||
assert_eq!(
|
assert_eq!(
|
||||||
config,
|
config,
|
||||||
@@ -257,7 +244,7 @@ timeout = '5s'";
|
|||||||
timeout = '7s'
|
timeout = '7s'
|
||||||
";
|
";
|
||||||
|
|
||||||
let config = parse(toml).unwrap().expect("it exists");
|
let config = parse(toml).unwrap();
|
||||||
|
|
||||||
assert_eq!(
|
assert_eq!(
|
||||||
config,
|
config,
|
||||||
|
|||||||
@@ -34,10 +34,10 @@ struct SegmentSize {
|
|||||||
}
|
}
|
||||||
|
|
||||||
struct SizeAlternatives {
|
struct SizeAlternatives {
|
||||||
// cheapest alternative if parent is available.
|
/// cheapest alternative if parent is available.
|
||||||
incremental: SegmentSize,
|
incremental: SegmentSize,
|
||||||
|
|
||||||
// cheapest alternative if parent node is not available
|
/// cheapest alternative if parent node is not available
|
||||||
non_incremental: Option<SegmentSize>,
|
non_incremental: Option<SegmentSize>,
|
||||||
}
|
}
|
||||||
|
|
||||||
|
|||||||
@@ -3,10 +3,17 @@ use std::fmt::Write;
|
|||||||
|
|
||||||
const SVG_WIDTH: f32 = 500.0;
|
const SVG_WIDTH: f32 = 500.0;
|
||||||
|
|
||||||
|
/// Different branch kind for SVG drawing.
|
||||||
|
#[derive(PartialEq)]
|
||||||
|
pub enum SvgBranchKind {
|
||||||
|
Timeline,
|
||||||
|
Lease,
|
||||||
|
}
|
||||||
|
|
||||||
struct SvgDraw<'a> {
|
struct SvgDraw<'a> {
|
||||||
storage: &'a StorageModel,
|
storage: &'a StorageModel,
|
||||||
branches: &'a [String],
|
branches: &'a [String],
|
||||||
seg_to_branch: &'a [usize],
|
seg_to_branch: &'a [(usize, SvgBranchKind)],
|
||||||
sizes: &'a [SegmentSizeResult],
|
sizes: &'a [SegmentSizeResult],
|
||||||
|
|
||||||
// layout
|
// layout
|
||||||
@@ -42,13 +49,18 @@ fn draw_legend(result: &mut String) -> anyhow::Result<()> {
|
|||||||
"<line x1=\"5\" y1=\"70\" x2=\"15\" y2=\"70\" stroke-width=\"1\" stroke=\"gray\" />"
|
"<line x1=\"5\" y1=\"70\" x2=\"15\" y2=\"70\" stroke-width=\"1\" stroke=\"gray\" />"
|
||||||
)?;
|
)?;
|
||||||
writeln!(result, "<text x=\"20\" y=\"75\">WAL not retained</text>")?;
|
writeln!(result, "<text x=\"20\" y=\"75\">WAL not retained</text>")?;
|
||||||
|
writeln!(
|
||||||
|
result,
|
||||||
|
"<line x1=\"10\" y1=\"85\" x2=\"10\" y2=\"95\" stroke-width=\"3\" stroke=\"blue\" />"
|
||||||
|
)?;
|
||||||
|
writeln!(result, "<text x=\"20\" y=\"95\">LSN lease</text>")?;
|
||||||
Ok(())
|
Ok(())
|
||||||
}
|
}
|
||||||
|
|
||||||
pub fn draw_svg(
|
pub fn draw_svg(
|
||||||
storage: &StorageModel,
|
storage: &StorageModel,
|
||||||
branches: &[String],
|
branches: &[String],
|
||||||
seg_to_branch: &[usize],
|
seg_to_branch: &[(usize, SvgBranchKind)],
|
||||||
sizes: &SizeResult,
|
sizes: &SizeResult,
|
||||||
) -> anyhow::Result<String> {
|
) -> anyhow::Result<String> {
|
||||||
let mut draw = SvgDraw {
|
let mut draw = SvgDraw {
|
||||||
@@ -100,7 +112,7 @@ impl<'a> SvgDraw<'a> {
|
|||||||
|
|
||||||
// Layout the timelines on Y dimension.
|
// Layout the timelines on Y dimension.
|
||||||
// TODO
|
// TODO
|
||||||
let mut y = 100.0;
|
let mut y = 120.0;
|
||||||
let mut branch_y_coordinates = Vec::new();
|
let mut branch_y_coordinates = Vec::new();
|
||||||
for _branch in self.branches {
|
for _branch in self.branches {
|
||||||
branch_y_coordinates.push(y);
|
branch_y_coordinates.push(y);
|
||||||
@@ -109,7 +121,7 @@ impl<'a> SvgDraw<'a> {
|
|||||||
|
|
||||||
// Calculate coordinates for each point
|
// Calculate coordinates for each point
|
||||||
let seg_coordinates = std::iter::zip(segments, self.seg_to_branch)
|
let seg_coordinates = std::iter::zip(segments, self.seg_to_branch)
|
||||||
.map(|(seg, branch_id)| {
|
.map(|(seg, (branch_id, _))| {
|
||||||
let x = (seg.lsn - min_lsn) as f32 / xscale;
|
let x = (seg.lsn - min_lsn) as f32 / xscale;
|
||||||
let y = branch_y_coordinates[*branch_id];
|
let y = branch_y_coordinates[*branch_id];
|
||||||
(x, y)
|
(x, y)
|
||||||
@@ -175,6 +187,22 @@ impl<'a> SvgDraw<'a> {
|
|||||||
|
|
||||||
// draw a snapshot point if it's needed
|
// draw a snapshot point if it's needed
|
||||||
let (coord_x, coord_y) = self.seg_coordinates[seg_id];
|
let (coord_x, coord_y) = self.seg_coordinates[seg_id];
|
||||||
|
|
||||||
|
let (_, kind) = &self.seg_to_branch[seg_id];
|
||||||
|
if kind == &SvgBranchKind::Lease {
|
||||||
|
let (x1, y1) = (coord_x, coord_y - 10.0);
|
||||||
|
let (x2, y2) = (coord_x, coord_y + 10.0);
|
||||||
|
|
||||||
|
let style = "stroke-width=\"3\" stroke=\"blue\"";
|
||||||
|
|
||||||
|
writeln!(
|
||||||
|
result,
|
||||||
|
"<line x1=\"{x1}\" y1=\"{y1}\" x2=\"{x2}\" y2=\"{y2}\" {style}>",
|
||||||
|
)?;
|
||||||
|
writeln!(result, " <title>leased lsn at {}</title>", seg.lsn)?;
|
||||||
|
writeln!(result, "</line>")?;
|
||||||
|
}
|
||||||
|
|
||||||
if self.sizes[seg_id].method == SegmentMethod::SnapshotHere {
|
if self.sizes[seg_id].method == SegmentMethod::SnapshotHere {
|
||||||
writeln!(
|
writeln!(
|
||||||
result,
|
result,
|
||||||
|
|||||||
@@ -40,6 +40,7 @@ thiserror.workspace = true
|
|||||||
tokio.workspace = true
|
tokio.workspace = true
|
||||||
tokio-tar.workspace = true
|
tokio-tar.workspace = true
|
||||||
tokio-util.workspace = true
|
tokio-util.workspace = true
|
||||||
|
toml_edit.workspace = true
|
||||||
tracing.workspace = true
|
tracing.workspace = true
|
||||||
tracing-error.workspace = true
|
tracing-error.workspace = true
|
||||||
tracing-subscriber = { workspace = true, features = ["json", "registry"] }
|
tracing-subscriber = { workspace = true, features = ["json", "registry"] }
|
||||||
|
|||||||
@@ -94,6 +94,8 @@ pub mod env;
|
|||||||
|
|
||||||
pub mod poison;
|
pub mod poison;
|
||||||
|
|
||||||
|
pub mod toml_edit_ext;
|
||||||
|
|
||||||
/// This is a shortcut to embed git sha into binaries and avoid copying the same build script to all packages
|
/// This is a shortcut to embed git sha into binaries and avoid copying the same build script to all packages
|
||||||
///
|
///
|
||||||
/// we have several cases:
|
/// we have several cases:
|
||||||
|
|||||||
@@ -0,0 +1,22 @@
|
|||||||
|
#[derive(Debug, thiserror::Error)]
|
||||||
|
pub enum Error {
|
||||||
|
#[error("item is not a document")]
|
||||||
|
ItemIsNotADocument,
|
||||||
|
#[error(transparent)]
|
||||||
|
Serde(toml_edit::de::Error),
|
||||||
|
}
|
||||||
|
|
||||||
|
pub fn deserialize_item<T>(item: &toml_edit::Item) -> Result<T, Error>
|
||||||
|
where
|
||||||
|
T: serde::de::DeserializeOwned,
|
||||||
|
{
|
||||||
|
let document: toml_edit::Document = match item {
|
||||||
|
toml_edit::Item::Table(toml) => toml.clone().into(),
|
||||||
|
toml_edit::Item::Value(toml_edit::Value::InlineTable(toml)) => {
|
||||||
|
toml.clone().into_table().into()
|
||||||
|
}
|
||||||
|
_ => return Err(Error::ItemIsNotADocument),
|
||||||
|
};
|
||||||
|
|
||||||
|
toml_edit::de::from_document(document).map_err(Error::Serde)
|
||||||
|
}
|
||||||
@@ -205,15 +205,6 @@ impl Client {
|
|||||||
Ok(())
|
Ok(())
|
||||||
}
|
}
|
||||||
|
|
||||||
pub async fn tenant_create(&self, req: &TenantCreateRequest) -> Result<TenantId> {
|
|
||||||
let uri = format!("{}/v1/tenant", self.mgmt_api_endpoint);
|
|
||||||
self.request(Method::POST, &uri, req)
|
|
||||||
.await?
|
|
||||||
.json()
|
|
||||||
.await
|
|
||||||
.map_err(Error::ReceiveBody)
|
|
||||||
}
|
|
||||||
|
|
||||||
/// The tenant deletion API can return 202 if deletion is incomplete, or
|
/// The tenant deletion API can return 202 if deletion is incomplete, or
|
||||||
/// 404 if it is complete. Callers are responsible for checking the status
|
/// 404 if it is complete. Callers are responsible for checking the status
|
||||||
/// code and retrying. Error codes other than 404 will return Err().
|
/// code and retrying. Error codes other than 404 will return Err().
|
||||||
|
|||||||
@@ -178,7 +178,7 @@ async fn main() -> anyhow::Result<()> {
|
|||||||
let toml_item = toml_document
|
let toml_item = toml_document
|
||||||
.get("remote_storage")
|
.get("remote_storage")
|
||||||
.expect("need remote_storage");
|
.expect("need remote_storage");
|
||||||
let config = RemoteStorageConfig::from_toml(toml_item)?.expect("incomplete config");
|
let config = RemoteStorageConfig::from_toml(toml_item)?;
|
||||||
let storage = remote_storage::GenericRemoteStorage::from_config(&config);
|
let storage = remote_storage::GenericRemoteStorage::from_config(&config);
|
||||||
let cancel = CancellationToken::new();
|
let cancel = CancellationToken::new();
|
||||||
storage
|
storage
|
||||||
|
|||||||
@@ -348,35 +348,36 @@ where
|
|||||||
self.add_rel(rel, rel).await?;
|
self.add_rel(rel, rel).await?;
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
for (path, content) in self
|
|
||||||
.timeline
|
|
||||||
.list_aux_files(self.lsn, self.ctx)
|
|
||||||
.await
|
|
||||||
.map_err(|e| BasebackupError::Server(e.into()))?
|
|
||||||
{
|
|
||||||
if path.starts_with("pg_replslot") {
|
|
||||||
let offs = pg_constants::REPL_SLOT_ON_DISK_OFFSETOF_RESTART_LSN;
|
|
||||||
let restart_lsn = Lsn(u64::from_le_bytes(
|
|
||||||
content[offs..offs + 8].try_into().unwrap(),
|
|
||||||
));
|
|
||||||
info!("Replication slot {} restart LSN={}", path, restart_lsn);
|
|
||||||
min_restart_lsn = Lsn::min(min_restart_lsn, restart_lsn);
|
|
||||||
} else if path == "pg_logical/replorigin_checkpoint" {
|
|
||||||
// replorigin_checkoint is written only on compute shutdown, so it contains
|
|
||||||
// deteriorated values. So we generate our own version of this file for the particular LSN
|
|
||||||
// based on information about replorigins extracted from transaction commit records.
|
|
||||||
// In future we will not generate AUX record for "pg_logical/replorigin_checkpoint" at all,
|
|
||||||
// but now we should handle (skip) it for backward compatibility.
|
|
||||||
continue;
|
|
||||||
}
|
|
||||||
let header = new_tar_header(&path, content.len() as u64)?;
|
|
||||||
self.ar
|
|
||||||
.append(&header, &*content)
|
|
||||||
.await
|
|
||||||
.context("could not add aux file to basebackup tarball")?;
|
|
||||||
}
|
|
||||||
}
|
}
|
||||||
|
|
||||||
|
for (path, content) in self
|
||||||
|
.timeline
|
||||||
|
.list_aux_files(self.lsn, self.ctx)
|
||||||
|
.await
|
||||||
|
.map_err(|e| BasebackupError::Server(e.into()))?
|
||||||
|
{
|
||||||
|
if path.starts_with("pg_replslot") {
|
||||||
|
let offs = pg_constants::REPL_SLOT_ON_DISK_OFFSETOF_RESTART_LSN;
|
||||||
|
let restart_lsn = Lsn(u64::from_le_bytes(
|
||||||
|
content[offs..offs + 8].try_into().unwrap(),
|
||||||
|
));
|
||||||
|
info!("Replication slot {} restart LSN={}", path, restart_lsn);
|
||||||
|
min_restart_lsn = Lsn::min(min_restart_lsn, restart_lsn);
|
||||||
|
} else if path == "pg_logical/replorigin_checkpoint" {
|
||||||
|
// replorigin_checkoint is written only on compute shutdown, so it contains
|
||||||
|
// deteriorated values. So we generate our own version of this file for the particular LSN
|
||||||
|
// based on information about replorigins extracted from transaction commit records.
|
||||||
|
// In future we will not generate AUX record for "pg_logical/replorigin_checkpoint" at all,
|
||||||
|
// but now we should handle (skip) it for backward compatibility.
|
||||||
|
continue;
|
||||||
|
}
|
||||||
|
let header = new_tar_header(&path, content.len() as u64)?;
|
||||||
|
self.ar
|
||||||
|
.append(&header, &*content)
|
||||||
|
.await
|
||||||
|
.context("could not add aux file to basebackup tarball")?;
|
||||||
|
}
|
||||||
|
|
||||||
if min_restart_lsn != Lsn::MAX {
|
if min_restart_lsn != Lsn::MAX {
|
||||||
info!(
|
info!(
|
||||||
"Min restart LSN for logical replication is {}",
|
"Min restart LSN for logical replication is {}",
|
||||||
|
|||||||
@@ -421,6 +421,10 @@ fn start_pageserver(
|
|||||||
background_jobs_can_start: background_jobs_barrier.clone(),
|
background_jobs_can_start: background_jobs_barrier.clone(),
|
||||||
};
|
};
|
||||||
|
|
||||||
|
info!(config=?conf.l0_flush, "using l0_flush config");
|
||||||
|
let l0_flush_global_state =
|
||||||
|
pageserver::l0_flush::L0FlushGlobalState::new(conf.l0_flush.clone());
|
||||||
|
|
||||||
// Scan the local 'tenants/' directory and start loading the tenants
|
// Scan the local 'tenants/' directory and start loading the tenants
|
||||||
let deletion_queue_client = deletion_queue.new_client();
|
let deletion_queue_client = deletion_queue.new_client();
|
||||||
let tenant_manager = BACKGROUND_RUNTIME.block_on(mgr::init_tenant_mgr(
|
let tenant_manager = BACKGROUND_RUNTIME.block_on(mgr::init_tenant_mgr(
|
||||||
@@ -429,6 +433,7 @@ fn start_pageserver(
|
|||||||
broker_client: broker_client.clone(),
|
broker_client: broker_client.clone(),
|
||||||
remote_storage: remote_storage.clone(),
|
remote_storage: remote_storage.clone(),
|
||||||
deletion_queue_client,
|
deletion_queue_client,
|
||||||
|
l0_flush_global_state,
|
||||||
},
|
},
|
||||||
order,
|
order,
|
||||||
shutdown_pageserver.clone(),
|
shutdown_pageserver.clone(),
|
||||||
|
|||||||
@@ -5,7 +5,7 @@
|
|||||||
//! See also `settings.md` for better description on every parameter.
|
//! See also `settings.md` for better description on every parameter.
|
||||||
|
|
||||||
use anyhow::{anyhow, bail, ensure, Context, Result};
|
use anyhow::{anyhow, bail, ensure, Context, Result};
|
||||||
use pageserver_api::shard::TenantShardId;
|
use pageserver_api::{models::ImageCompressionAlgorithm, shard::TenantShardId};
|
||||||
use remote_storage::{RemotePath, RemoteStorageConfig};
|
use remote_storage::{RemotePath, RemoteStorageConfig};
|
||||||
use serde;
|
use serde;
|
||||||
use serde::de::IntoDeserializer;
|
use serde::de::IntoDeserializer;
|
||||||
@@ -30,11 +30,11 @@ use utils::{
|
|||||||
logging::LogFormat,
|
logging::LogFormat,
|
||||||
};
|
};
|
||||||
|
|
||||||
use crate::tenant::timeline::GetVectoredImpl;
|
|
||||||
use crate::tenant::vectored_blob_io::MaxVectoredReadBytes;
|
use crate::tenant::vectored_blob_io::MaxVectoredReadBytes;
|
||||||
use crate::tenant::{config::TenantConfOpt, timeline::GetImpl};
|
use crate::tenant::{config::TenantConfOpt, timeline::GetImpl};
|
||||||
use crate::tenant::{TENANTS_SEGMENT_NAME, TIMELINES_SEGMENT_NAME};
|
use crate::tenant::{TENANTS_SEGMENT_NAME, TIMELINES_SEGMENT_NAME};
|
||||||
use crate::{disk_usage_eviction_task::DiskUsageEvictionTaskConfig, virtual_file::io_engine};
|
use crate::{disk_usage_eviction_task::DiskUsageEvictionTaskConfig, virtual_file::io_engine};
|
||||||
|
use crate::{l0_flush::L0FlushConfig, tenant::timeline::GetVectoredImpl};
|
||||||
use crate::{tenant::config::TenantConf, virtual_file};
|
use crate::{tenant::config::TenantConf, virtual_file};
|
||||||
use crate::{TENANT_HEATMAP_BASENAME, TENANT_LOCATION_CONFIG_NAME, TIMELINE_DELETE_MARK_SUFFIX};
|
use crate::{TENANT_HEATMAP_BASENAME, TENANT_LOCATION_CONFIG_NAME, TIMELINE_DELETE_MARK_SUFFIX};
|
||||||
|
|
||||||
@@ -50,6 +50,7 @@ pub mod defaults {
|
|||||||
DEFAULT_HTTP_LISTEN_ADDR, DEFAULT_HTTP_LISTEN_PORT, DEFAULT_PG_LISTEN_ADDR,
|
DEFAULT_HTTP_LISTEN_ADDR, DEFAULT_HTTP_LISTEN_PORT, DEFAULT_PG_LISTEN_ADDR,
|
||||||
DEFAULT_PG_LISTEN_PORT,
|
DEFAULT_PG_LISTEN_PORT,
|
||||||
};
|
};
|
||||||
|
use pageserver_api::models::ImageCompressionAlgorithm;
|
||||||
pub use storage_broker::DEFAULT_ENDPOINT as BROKER_DEFAULT_ENDPOINT;
|
pub use storage_broker::DEFAULT_ENDPOINT as BROKER_DEFAULT_ENDPOINT;
|
||||||
|
|
||||||
pub const DEFAULT_WAIT_LSN_TIMEOUT: &str = "60 s";
|
pub const DEFAULT_WAIT_LSN_TIMEOUT: &str = "60 s";
|
||||||
@@ -90,6 +91,9 @@ pub mod defaults {
|
|||||||
|
|
||||||
pub const DEFAULT_MAX_VECTORED_READ_BYTES: usize = 128 * 1024; // 128 KiB
|
pub const DEFAULT_MAX_VECTORED_READ_BYTES: usize = 128 * 1024; // 128 KiB
|
||||||
|
|
||||||
|
pub const DEFAULT_IMAGE_COMPRESSION: ImageCompressionAlgorithm =
|
||||||
|
ImageCompressionAlgorithm::DisabledNoDecompress;
|
||||||
|
|
||||||
pub const DEFAULT_VALIDATE_VECTORED_GET: bool = true;
|
pub const DEFAULT_VALIDATE_VECTORED_GET: bool = true;
|
||||||
|
|
||||||
pub const DEFAULT_EPHEMERAL_BYTES_PER_MEMORY_KB: usize = 0;
|
pub const DEFAULT_EPHEMERAL_BYTES_PER_MEMORY_KB: usize = 0;
|
||||||
@@ -159,7 +163,7 @@ pub mod defaults {
|
|||||||
|
|
||||||
#ephemeral_bytes_per_memory_kb = {DEFAULT_EPHEMERAL_BYTES_PER_MEMORY_KB}
|
#ephemeral_bytes_per_memory_kb = {DEFAULT_EPHEMERAL_BYTES_PER_MEMORY_KB}
|
||||||
|
|
||||||
[remote_storage]
|
#[remote_storage]
|
||||||
|
|
||||||
"#
|
"#
|
||||||
);
|
);
|
||||||
@@ -285,12 +289,16 @@ pub struct PageServerConf {
|
|||||||
|
|
||||||
pub validate_vectored_get: bool,
|
pub validate_vectored_get: bool,
|
||||||
|
|
||||||
|
pub image_compression: ImageCompressionAlgorithm,
|
||||||
|
|
||||||
/// How many bytes of ephemeral layer content will we allow per kilobyte of RAM. When this
|
/// How many bytes of ephemeral layer content will we allow per kilobyte of RAM. When this
|
||||||
/// is exceeded, we start proactively closing ephemeral layers to limit the total amount
|
/// is exceeded, we start proactively closing ephemeral layers to limit the total amount
|
||||||
/// of ephemeral data.
|
/// of ephemeral data.
|
||||||
///
|
///
|
||||||
/// Setting this to zero disables limits on total ephemeral layer size.
|
/// Setting this to zero disables limits on total ephemeral layer size.
|
||||||
pub ephemeral_bytes_per_memory_kb: usize,
|
pub ephemeral_bytes_per_memory_kb: usize,
|
||||||
|
|
||||||
|
pub l0_flush: L0FlushConfig,
|
||||||
}
|
}
|
||||||
|
|
||||||
/// We do not want to store this in a PageServerConf because the latter may be logged
|
/// We do not want to store this in a PageServerConf because the latter may be logged
|
||||||
@@ -395,7 +403,11 @@ struct PageServerConfigBuilder {
|
|||||||
|
|
||||||
validate_vectored_get: BuilderValue<bool>,
|
validate_vectored_get: BuilderValue<bool>,
|
||||||
|
|
||||||
|
image_compression: BuilderValue<ImageCompressionAlgorithm>,
|
||||||
|
|
||||||
ephemeral_bytes_per_memory_kb: BuilderValue<usize>,
|
ephemeral_bytes_per_memory_kb: BuilderValue<usize>,
|
||||||
|
|
||||||
|
l0_flush: BuilderValue<L0FlushConfig>,
|
||||||
}
|
}
|
||||||
|
|
||||||
impl PageServerConfigBuilder {
|
impl PageServerConfigBuilder {
|
||||||
@@ -482,8 +494,10 @@ impl PageServerConfigBuilder {
|
|||||||
max_vectored_read_bytes: Set(MaxVectoredReadBytes(
|
max_vectored_read_bytes: Set(MaxVectoredReadBytes(
|
||||||
NonZeroUsize::new(DEFAULT_MAX_VECTORED_READ_BYTES).unwrap(),
|
NonZeroUsize::new(DEFAULT_MAX_VECTORED_READ_BYTES).unwrap(),
|
||||||
)),
|
)),
|
||||||
|
image_compression: Set(DEFAULT_IMAGE_COMPRESSION),
|
||||||
validate_vectored_get: Set(DEFAULT_VALIDATE_VECTORED_GET),
|
validate_vectored_get: Set(DEFAULT_VALIDATE_VECTORED_GET),
|
||||||
ephemeral_bytes_per_memory_kb: Set(DEFAULT_EPHEMERAL_BYTES_PER_MEMORY_KB),
|
ephemeral_bytes_per_memory_kb: Set(DEFAULT_EPHEMERAL_BYTES_PER_MEMORY_KB),
|
||||||
|
l0_flush: Set(L0FlushConfig::default()),
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
@@ -667,10 +681,18 @@ impl PageServerConfigBuilder {
|
|||||||
self.validate_vectored_get = BuilderValue::Set(value);
|
self.validate_vectored_get = BuilderValue::Set(value);
|
||||||
}
|
}
|
||||||
|
|
||||||
|
pub fn get_image_compression(&mut self, value: ImageCompressionAlgorithm) {
|
||||||
|
self.image_compression = BuilderValue::Set(value);
|
||||||
|
}
|
||||||
|
|
||||||
pub fn get_ephemeral_bytes_per_memory_kb(&mut self, value: usize) {
|
pub fn get_ephemeral_bytes_per_memory_kb(&mut self, value: usize) {
|
||||||
self.ephemeral_bytes_per_memory_kb = BuilderValue::Set(value);
|
self.ephemeral_bytes_per_memory_kb = BuilderValue::Set(value);
|
||||||
}
|
}
|
||||||
|
|
||||||
|
pub fn l0_flush(&mut self, value: L0FlushConfig) {
|
||||||
|
self.l0_flush = BuilderValue::Set(value);
|
||||||
|
}
|
||||||
|
|
||||||
pub fn build(self) -> anyhow::Result<PageServerConf> {
|
pub fn build(self) -> anyhow::Result<PageServerConf> {
|
||||||
let default = Self::default_values();
|
let default = Self::default_values();
|
||||||
|
|
||||||
@@ -727,7 +749,9 @@ impl PageServerConfigBuilder {
|
|||||||
get_impl,
|
get_impl,
|
||||||
max_vectored_read_bytes,
|
max_vectored_read_bytes,
|
||||||
validate_vectored_get,
|
validate_vectored_get,
|
||||||
|
image_compression,
|
||||||
ephemeral_bytes_per_memory_kb,
|
ephemeral_bytes_per_memory_kb,
|
||||||
|
l0_flush,
|
||||||
}
|
}
|
||||||
CUSTOM LOGIC
|
CUSTOM LOGIC
|
||||||
{
|
{
|
||||||
@@ -918,7 +942,7 @@ impl PageServerConf {
|
|||||||
"http_auth_type" => builder.http_auth_type(parse_toml_from_str(key, item)?),
|
"http_auth_type" => builder.http_auth_type(parse_toml_from_str(key, item)?),
|
||||||
"pg_auth_type" => builder.pg_auth_type(parse_toml_from_str(key, item)?),
|
"pg_auth_type" => builder.pg_auth_type(parse_toml_from_str(key, item)?),
|
||||||
"remote_storage" => {
|
"remote_storage" => {
|
||||||
builder.remote_storage_config(RemoteStorageConfig::from_toml(item)?)
|
builder.remote_storage_config(Some(RemoteStorageConfig::from_toml(item).context("remote_storage")?))
|
||||||
}
|
}
|
||||||
"tenant_config" => {
|
"tenant_config" => {
|
||||||
t_conf = TenantConfOpt::try_from(item.to_owned()).context(format!("failed to parse: '{key}'"))?;
|
t_conf = TenantConfOpt::try_from(item.to_owned()).context(format!("failed to parse: '{key}'"))?;
|
||||||
@@ -946,7 +970,7 @@ impl PageServerConf {
|
|||||||
builder.metric_collection_endpoint(Some(endpoint));
|
builder.metric_collection_endpoint(Some(endpoint));
|
||||||
},
|
},
|
||||||
"metric_collection_bucket" => {
|
"metric_collection_bucket" => {
|
||||||
builder.metric_collection_bucket(RemoteStorageConfig::from_toml(item)?)
|
builder.metric_collection_bucket(Some(RemoteStorageConfig::from_toml(item)?))
|
||||||
}
|
}
|
||||||
"synthetic_size_calculation_interval" =>
|
"synthetic_size_calculation_interval" =>
|
||||||
builder.synthetic_size_calculation_interval(parse_toml_duration(key, item)?),
|
builder.synthetic_size_calculation_interval(parse_toml_duration(key, item)?),
|
||||||
@@ -1004,9 +1028,15 @@ impl PageServerConf {
|
|||||||
"validate_vectored_get" => {
|
"validate_vectored_get" => {
|
||||||
builder.get_validate_vectored_get(parse_toml_bool("validate_vectored_get", item)?)
|
builder.get_validate_vectored_get(parse_toml_bool("validate_vectored_get", item)?)
|
||||||
}
|
}
|
||||||
|
"image_compression" => {
|
||||||
|
builder.get_image_compression(parse_toml_from_str("image_compression", item)?)
|
||||||
|
}
|
||||||
"ephemeral_bytes_per_memory_kb" => {
|
"ephemeral_bytes_per_memory_kb" => {
|
||||||
builder.get_ephemeral_bytes_per_memory_kb(parse_toml_u64("ephemeral_bytes_per_memory_kb", item)? as usize)
|
builder.get_ephemeral_bytes_per_memory_kb(parse_toml_u64("ephemeral_bytes_per_memory_kb", item)? as usize)
|
||||||
}
|
}
|
||||||
|
"l0_flush" => {
|
||||||
|
builder.l0_flush(utils::toml_edit_ext::deserialize_item(item).context("l0_flush")?)
|
||||||
|
}
|
||||||
_ => bail!("unrecognized pageserver option '{key}'"),
|
_ => bail!("unrecognized pageserver option '{key}'"),
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
@@ -1088,8 +1118,10 @@ impl PageServerConf {
|
|||||||
NonZeroUsize::new(defaults::DEFAULT_MAX_VECTORED_READ_BYTES)
|
NonZeroUsize::new(defaults::DEFAULT_MAX_VECTORED_READ_BYTES)
|
||||||
.expect("Invalid default constant"),
|
.expect("Invalid default constant"),
|
||||||
),
|
),
|
||||||
|
image_compression: defaults::DEFAULT_IMAGE_COMPRESSION,
|
||||||
validate_vectored_get: defaults::DEFAULT_VALIDATE_VECTORED_GET,
|
validate_vectored_get: defaults::DEFAULT_VALIDATE_VECTORED_GET,
|
||||||
ephemeral_bytes_per_memory_kb: defaults::DEFAULT_EPHEMERAL_BYTES_PER_MEMORY_KB,
|
ephemeral_bytes_per_memory_kb: defaults::DEFAULT_EPHEMERAL_BYTES_PER_MEMORY_KB,
|
||||||
|
l0_flush: L0FlushConfig::default(),
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
@@ -1328,7 +1360,9 @@ background_task_maximum_delay = '334 s'
|
|||||||
.expect("Invalid default constant")
|
.expect("Invalid default constant")
|
||||||
),
|
),
|
||||||
validate_vectored_get: defaults::DEFAULT_VALIDATE_VECTORED_GET,
|
validate_vectored_get: defaults::DEFAULT_VALIDATE_VECTORED_GET,
|
||||||
|
image_compression: defaults::DEFAULT_IMAGE_COMPRESSION,
|
||||||
ephemeral_bytes_per_memory_kb: defaults::DEFAULT_EPHEMERAL_BYTES_PER_MEMORY_KB,
|
ephemeral_bytes_per_memory_kb: defaults::DEFAULT_EPHEMERAL_BYTES_PER_MEMORY_KB,
|
||||||
|
l0_flush: L0FlushConfig::default(),
|
||||||
},
|
},
|
||||||
"Correct defaults should be used when no config values are provided"
|
"Correct defaults should be used when no config values are provided"
|
||||||
);
|
);
|
||||||
@@ -1401,7 +1435,9 @@ background_task_maximum_delay = '334 s'
|
|||||||
.expect("Invalid default constant")
|
.expect("Invalid default constant")
|
||||||
),
|
),
|
||||||
validate_vectored_get: defaults::DEFAULT_VALIDATE_VECTORED_GET,
|
validate_vectored_get: defaults::DEFAULT_VALIDATE_VECTORED_GET,
|
||||||
|
image_compression: defaults::DEFAULT_IMAGE_COMPRESSION,
|
||||||
ephemeral_bytes_per_memory_kb: defaults::DEFAULT_EPHEMERAL_BYTES_PER_MEMORY_KB,
|
ephemeral_bytes_per_memory_kb: defaults::DEFAULT_EPHEMERAL_BYTES_PER_MEMORY_KB,
|
||||||
|
l0_flush: L0FlushConfig::default(),
|
||||||
},
|
},
|
||||||
"Should be able to parse all basic config values correctly"
|
"Should be able to parse all basic config values correctly"
|
||||||
);
|
);
|
||||||
@@ -1681,6 +1717,19 @@ threshold = "20m"
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn empty_remote_storage_is_error() {
|
||||||
|
let tempdir = tempdir().unwrap();
|
||||||
|
let (workdir, _) = prepare_fs(&tempdir).unwrap();
|
||||||
|
let input = r#"
|
||||||
|
remote_storage = {}
|
||||||
|
"#;
|
||||||
|
let doc = toml_edit::Document::from_str(input).unwrap();
|
||||||
|
let err = PageServerConf::parse_and_validate(&doc, &workdir)
|
||||||
|
.expect_err("empty remote_storage field should fail, don't specify it if you want no remote_storage");
|
||||||
|
assert!(format!("{err}").contains("remote_storage"), "{err}");
|
||||||
|
}
|
||||||
|
|
||||||
fn prepare_fs(tempdir: &Utf8TempDir) -> anyhow::Result<(Utf8PathBuf, Utf8PathBuf)> {
|
fn prepare_fs(tempdir: &Utf8TempDir) -> anyhow::Result<(Utf8PathBuf, Utf8PathBuf)> {
|
||||||
let tempdir_path = tempdir.path();
|
let tempdir_path = tempdir.path();
|
||||||
|
|
||||||
|
|||||||
@@ -190,7 +190,7 @@ where
|
|||||||
}
|
}
|
||||||
} else {
|
} else {
|
||||||
// If we failed validation, then do not apply any of the projected updates
|
// If we failed validation, then do not apply any of the projected updates
|
||||||
warn!("Dropped remote consistent LSN updates for tenant {tenant_id} in stale generation {:?}", tenant_lsn_state.generation);
|
info!("Dropped remote consistent LSN updates for tenant {tenant_id} in stale generation {:?}", tenant_lsn_state.generation);
|
||||||
metrics::DELETION_QUEUE.dropped_lsn_updates.inc();
|
metrics::DELETION_QUEUE.dropped_lsn_updates.inc();
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
@@ -225,7 +225,7 @@ where
|
|||||||
&& (tenant.generation == *validated_generation);
|
&& (tenant.generation == *validated_generation);
|
||||||
|
|
||||||
if !this_list_valid {
|
if !this_list_valid {
|
||||||
warn!("Dropping stale deletions for tenant {tenant_id} in generation {:?}, objects may be leaked", tenant.generation);
|
info!("Dropping stale deletions for tenant {tenant_id} in generation {:?}, objects may be leaked", tenant.generation);
|
||||||
metrics::DELETION_QUEUE.keys_dropped.inc_by(tenant.len() as u64);
|
metrics::DELETION_QUEUE.keys_dropped.inc_by(tenant.len() as u64);
|
||||||
mutated = true;
|
mutated = true;
|
||||||
} else {
|
} else {
|
||||||
|
|||||||
@@ -265,15 +265,19 @@ paths:
|
|||||||
type: string
|
type: string
|
||||||
format: hex
|
format: hex
|
||||||
post:
|
post:
|
||||||
description: Obtain lease for the given LSN
|
description: Obtains a lease for the given LSN.
|
||||||
parameters:
|
requestBody:
|
||||||
- name: lsn
|
content:
|
||||||
in: query
|
application/json:
|
||||||
required: true
|
schema:
|
||||||
schema:
|
type: object
|
||||||
type: string
|
required:
|
||||||
format: hex
|
- lsn
|
||||||
description: A LSN to obtain the lease for
|
properties:
|
||||||
|
lsn:
|
||||||
|
description: A LSN to obtain the lease for.
|
||||||
|
type: string
|
||||||
|
format: hex
|
||||||
responses:
|
responses:
|
||||||
"200":
|
"200":
|
||||||
description: OK
|
description: OK
|
||||||
|
|||||||
@@ -22,6 +22,7 @@ use pageserver_api::models::ListAuxFilesRequest;
|
|||||||
use pageserver_api::models::LocationConfig;
|
use pageserver_api::models::LocationConfig;
|
||||||
use pageserver_api::models::LocationConfigListResponse;
|
use pageserver_api::models::LocationConfigListResponse;
|
||||||
use pageserver_api::models::LsnLease;
|
use pageserver_api::models::LsnLease;
|
||||||
|
use pageserver_api::models::LsnLeaseRequest;
|
||||||
use pageserver_api::models::ShardParameters;
|
use pageserver_api::models::ShardParameters;
|
||||||
use pageserver_api::models::TenantDetails;
|
use pageserver_api::models::TenantDetails;
|
||||||
use pageserver_api::models::TenantLocationConfigResponse;
|
use pageserver_api::models::TenantLocationConfigResponse;
|
||||||
@@ -42,7 +43,7 @@ use pageserver_api::shard::TenantShardId;
|
|||||||
use remote_storage::DownloadError;
|
use remote_storage::DownloadError;
|
||||||
use remote_storage::GenericRemoteStorage;
|
use remote_storage::GenericRemoteStorage;
|
||||||
use remote_storage::TimeTravelError;
|
use remote_storage::TimeTravelError;
|
||||||
use tenant_size_model::{SizeResult, StorageModel};
|
use tenant_size_model::{svg::SvgBranchKind, SizeResult, StorageModel};
|
||||||
use tokio_util::sync::CancellationToken;
|
use tokio_util::sync::CancellationToken;
|
||||||
use tracing::*;
|
use tracing::*;
|
||||||
use utils::auth::JwtAuth;
|
use utils::auth::JwtAuth;
|
||||||
@@ -53,7 +54,6 @@ use utils::http::request::{get_request_param, must_get_query_param, parse_query_
|
|||||||
|
|
||||||
use crate::context::{DownloadBehavior, RequestContext};
|
use crate::context::{DownloadBehavior, RequestContext};
|
||||||
use crate::deletion_queue::DeletionQueueClient;
|
use crate::deletion_queue::DeletionQueueClient;
|
||||||
use crate::metrics::{StorageTimeOperation, STORAGE_TIME_GLOBAL};
|
|
||||||
use crate::pgdatadir_mapping::LsnForTimestamp;
|
use crate::pgdatadir_mapping::LsnForTimestamp;
|
||||||
use crate::task_mgr::TaskKind;
|
use crate::task_mgr::TaskKind;
|
||||||
use crate::tenant::config::{LocationConf, TenantConfOpt};
|
use crate::tenant::config::{LocationConf, TenantConfOpt};
|
||||||
@@ -75,13 +75,12 @@ use crate::tenant::timeline::CompactFlags;
|
|||||||
use crate::tenant::timeline::CompactionError;
|
use crate::tenant::timeline::CompactionError;
|
||||||
use crate::tenant::timeline::Timeline;
|
use crate::tenant::timeline::Timeline;
|
||||||
use crate::tenant::GetTimelineError;
|
use crate::tenant::GetTimelineError;
|
||||||
use crate::tenant::SpawnMode;
|
|
||||||
use crate::tenant::{LogicalSizeCalculationCause, PageReconstructError};
|
use crate::tenant::{LogicalSizeCalculationCause, PageReconstructError};
|
||||||
use crate::{config::PageServerConf, tenant::mgr};
|
use crate::{config::PageServerConf, tenant::mgr};
|
||||||
use crate::{disk_usage_eviction_task, tenant};
|
use crate::{disk_usage_eviction_task, tenant};
|
||||||
use pageserver_api::models::{
|
use pageserver_api::models::{
|
||||||
StatusResponse, TenantConfigRequest, TenantCreateRequest, TenantCreateResponse, TenantInfo,
|
StatusResponse, TenantConfigRequest, TenantInfo, TimelineCreateRequest, TimelineGcRequest,
|
||||||
TimelineCreateRequest, TimelineGcRequest, TimelineInfo,
|
TimelineInfo,
|
||||||
};
|
};
|
||||||
use utils::{
|
use utils::{
|
||||||
auth::SwappableJwtAuth,
|
auth::SwappableJwtAuth,
|
||||||
@@ -229,7 +228,7 @@ impl From<UpsertLocationError> for ApiError {
|
|||||||
BadRequest(e) => ApiError::BadRequest(e),
|
BadRequest(e) => ApiError::BadRequest(e),
|
||||||
Unavailable(_) => ApiError::ShuttingDown,
|
Unavailable(_) => ApiError::ShuttingDown,
|
||||||
e @ InProgress => ApiError::Conflict(format!("{e}")),
|
e @ InProgress => ApiError::Conflict(format!("{e}")),
|
||||||
Flush(e) | Other(e) => ApiError::InternalServerError(e),
|
Flush(e) | InternalError(e) => ApiError::InternalServerError(e),
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
@@ -408,6 +407,8 @@ async fn build_timeline_info_common(
|
|||||||
|
|
||||||
let walreceiver_status = timeline.walreceiver_status();
|
let walreceiver_status = timeline.walreceiver_status();
|
||||||
|
|
||||||
|
let (pitr_history_size, within_ancestor_pitr) = timeline.get_pitr_history_stats();
|
||||||
|
|
||||||
let info = TimelineInfo {
|
let info = TimelineInfo {
|
||||||
tenant_id: timeline.tenant_shard_id,
|
tenant_id: timeline.tenant_shard_id,
|
||||||
timeline_id: timeline.timeline_id,
|
timeline_id: timeline.timeline_id,
|
||||||
@@ -428,6 +429,8 @@ async fn build_timeline_info_common(
|
|||||||
directory_entries_counts: timeline.get_directory_metrics().to_vec(),
|
directory_entries_counts: timeline.get_directory_metrics().to_vec(),
|
||||||
current_physical_size,
|
current_physical_size,
|
||||||
current_logical_size_non_incremental: None,
|
current_logical_size_non_incremental: None,
|
||||||
|
pitr_history_size,
|
||||||
|
within_ancestor_pitr,
|
||||||
timeline_dir_layer_file_size_sum: None,
|
timeline_dir_layer_file_size_sum: None,
|
||||||
wal_source_connstr,
|
wal_source_connstr,
|
||||||
last_received_msg_lsn,
|
last_received_msg_lsn,
|
||||||
@@ -1193,10 +1196,15 @@ fn synthetic_size_html_response(
|
|||||||
timeline_map.insert(ti.timeline_id, index);
|
timeline_map.insert(ti.timeline_id, index);
|
||||||
timeline_ids.push(ti.timeline_id.to_string());
|
timeline_ids.push(ti.timeline_id.to_string());
|
||||||
}
|
}
|
||||||
let seg_to_branch: Vec<usize> = inputs
|
let seg_to_branch: Vec<(usize, SvgBranchKind)> = inputs
|
||||||
.segments
|
.segments
|
||||||
.iter()
|
.iter()
|
||||||
.map(|seg| *timeline_map.get(&seg.timeline_id).unwrap())
|
.map(|seg| {
|
||||||
|
(
|
||||||
|
*timeline_map.get(&seg.timeline_id).unwrap(),
|
||||||
|
seg.kind.into(),
|
||||||
|
)
|
||||||
|
})
|
||||||
.collect();
|
.collect();
|
||||||
|
|
||||||
let svg =
|
let svg =
|
||||||
@@ -1237,75 +1245,6 @@ pub fn html_response(status: StatusCode, data: String) -> Result<Response<Body>,
|
|||||||
Ok(response)
|
Ok(response)
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Helper for requests that may take a generation, which is mandatory
|
|
||||||
/// when control_plane_api is set, but otherwise defaults to Generation::none()
|
|
||||||
fn get_request_generation(state: &State, req_gen: Option<u32>) -> Result<Generation, ApiError> {
|
|
||||||
if state.conf.control_plane_api.is_some() {
|
|
||||||
req_gen
|
|
||||||
.map(Generation::new)
|
|
||||||
.ok_or(ApiError::BadRequest(anyhow!(
|
|
||||||
"generation attribute missing"
|
|
||||||
)))
|
|
||||||
} else {
|
|
||||||
// Legacy mode: all tenants operate with no generation
|
|
||||||
Ok(Generation::none())
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
async fn tenant_create_handler(
|
|
||||||
mut request: Request<Body>,
|
|
||||||
_cancel: CancellationToken,
|
|
||||||
) -> Result<Response<Body>, ApiError> {
|
|
||||||
let request_data: TenantCreateRequest = json_request(&mut request).await?;
|
|
||||||
let target_tenant_id = request_data.new_tenant_id;
|
|
||||||
check_permission(&request, None)?;
|
|
||||||
|
|
||||||
let _timer = STORAGE_TIME_GLOBAL
|
|
||||||
.get_metric_with_label_values(&[StorageTimeOperation::CreateTenant.into()])
|
|
||||||
.expect("bug")
|
|
||||||
.start_timer();
|
|
||||||
|
|
||||||
let tenant_conf =
|
|
||||||
TenantConfOpt::try_from(&request_data.config).map_err(ApiError::BadRequest)?;
|
|
||||||
|
|
||||||
let state = get_state(&request);
|
|
||||||
|
|
||||||
let generation = get_request_generation(state, request_data.generation)?;
|
|
||||||
|
|
||||||
let ctx = RequestContext::new(TaskKind::MgmtRequest, DownloadBehavior::Warn);
|
|
||||||
|
|
||||||
let location_conf =
|
|
||||||
LocationConf::attached_single(tenant_conf, generation, &request_data.shard_parameters);
|
|
||||||
|
|
||||||
let new_tenant = state
|
|
||||||
.tenant_manager
|
|
||||||
.upsert_location(
|
|
||||||
target_tenant_id,
|
|
||||||
location_conf,
|
|
||||||
None,
|
|
||||||
SpawnMode::Create,
|
|
||||||
&ctx,
|
|
||||||
)
|
|
||||||
.await?;
|
|
||||||
|
|
||||||
let Some(new_tenant) = new_tenant else {
|
|
||||||
// This should never happen: indicates a bug in upsert_location
|
|
||||||
return Err(ApiError::InternalServerError(anyhow::anyhow!(
|
|
||||||
"Upsert succeeded but didn't return tenant!"
|
|
||||||
)));
|
|
||||||
};
|
|
||||||
// We created the tenant. Existing API semantics are that the tenant
|
|
||||||
// is Active when this function returns.
|
|
||||||
new_tenant
|
|
||||||
.wait_to_become_active(ACTIVE_TENANT_TIMEOUT)
|
|
||||||
.await?;
|
|
||||||
|
|
||||||
json_response(
|
|
||||||
StatusCode::CREATED,
|
|
||||||
TenantCreateResponse(new_tenant.tenant_shard_id().tenant_id),
|
|
||||||
)
|
|
||||||
}
|
|
||||||
|
|
||||||
async fn get_tenant_config_handler(
|
async fn get_tenant_config_handler(
|
||||||
request: Request<Body>,
|
request: Request<Body>,
|
||||||
_cancel: CancellationToken,
|
_cancel: CancellationToken,
|
||||||
@@ -1367,7 +1306,7 @@ async fn update_tenant_config_handler(
|
|||||||
|
|
||||||
crate::tenant::Tenant::persist_tenant_config(state.conf, &tenant_shard_id, &location_conf)
|
crate::tenant::Tenant::persist_tenant_config(state.conf, &tenant_shard_id, &location_conf)
|
||||||
.await
|
.await
|
||||||
.map_err(ApiError::InternalServerError)?;
|
.map_err(|e| ApiError::InternalServerError(anyhow::anyhow!(e)))?;
|
||||||
tenant.set_new_tenant_config(new_tenant_conf);
|
tenant.set_new_tenant_config(new_tenant_conf);
|
||||||
|
|
||||||
json_response(StatusCode::OK, ())
|
json_response(StatusCode::OK, ())
|
||||||
@@ -1598,15 +1537,13 @@ async fn handle_tenant_break(
|
|||||||
|
|
||||||
// Obtains an lsn lease on the given timeline.
|
// Obtains an lsn lease on the given timeline.
|
||||||
async fn lsn_lease_handler(
|
async fn lsn_lease_handler(
|
||||||
request: Request<Body>,
|
mut request: Request<Body>,
|
||||||
_cancel: CancellationToken,
|
_cancel: CancellationToken,
|
||||||
) -> Result<Response<Body>, ApiError> {
|
) -> Result<Response<Body>, ApiError> {
|
||||||
let tenant_shard_id: TenantShardId = parse_request_param(&request, "tenant_shard_id")?;
|
let tenant_shard_id: TenantShardId = parse_request_param(&request, "tenant_shard_id")?;
|
||||||
let timeline_id: TimelineId = parse_request_param(&request, "timeline_id")?;
|
let timeline_id: TimelineId = parse_request_param(&request, "timeline_id")?;
|
||||||
check_permission(&request, Some(tenant_shard_id.tenant_id))?;
|
check_permission(&request, Some(tenant_shard_id.tenant_id))?;
|
||||||
|
let lsn = json_request::<LsnLeaseRequest>(&mut request).await?.lsn;
|
||||||
let lsn: Lsn = parse_query_param(&request, "lsn")?
|
|
||||||
.ok_or_else(|| ApiError::BadRequest(anyhow!("missing 'lsn' query parameter")))?;
|
|
||||||
|
|
||||||
let ctx = RequestContext::new(TaskKind::MgmtRequest, DownloadBehavior::Download);
|
let ctx = RequestContext::new(TaskKind::MgmtRequest, DownloadBehavior::Download);
|
||||||
|
|
||||||
@@ -2611,7 +2548,6 @@ pub fn make_router(
|
|||||||
api_handler(r, reload_auth_validation_keys_handler)
|
api_handler(r, reload_auth_validation_keys_handler)
|
||||||
})
|
})
|
||||||
.get("/v1/tenant", |r| api_handler(r, tenant_list_handler))
|
.get("/v1/tenant", |r| api_handler(r, tenant_list_handler))
|
||||||
.post("/v1/tenant", |r| api_handler(r, tenant_create_handler))
|
|
||||||
.get("/v1/tenant/:tenant_shard_id", |r| {
|
.get("/v1/tenant/:tenant_shard_id", |r| {
|
||||||
api_handler(r, tenant_status)
|
api_handler(r, tenant_status)
|
||||||
})
|
})
|
||||||
|
|||||||
@@ -0,0 +1,46 @@
|
|||||||
|
use std::{num::NonZeroUsize, sync::Arc};
|
||||||
|
|
||||||
|
use crate::tenant::ephemeral_file;
|
||||||
|
|
||||||
|
#[derive(Default, Debug, PartialEq, Eq, Clone, serde::Deserialize)]
|
||||||
|
#[serde(tag = "mode", rename_all = "kebab-case", deny_unknown_fields)]
|
||||||
|
pub enum L0FlushConfig {
|
||||||
|
#[default]
|
||||||
|
PageCached,
|
||||||
|
#[serde(rename_all = "snake_case")]
|
||||||
|
Direct { max_concurrency: NonZeroUsize },
|
||||||
|
}
|
||||||
|
|
||||||
|
#[derive(Clone)]
|
||||||
|
pub struct L0FlushGlobalState(Arc<Inner>);
|
||||||
|
|
||||||
|
pub(crate) enum Inner {
|
||||||
|
PageCached,
|
||||||
|
Direct { semaphore: tokio::sync::Semaphore },
|
||||||
|
}
|
||||||
|
|
||||||
|
impl L0FlushGlobalState {
|
||||||
|
pub fn new(config: L0FlushConfig) -> Self {
|
||||||
|
match config {
|
||||||
|
L0FlushConfig::PageCached => Self(Arc::new(Inner::PageCached)),
|
||||||
|
L0FlushConfig::Direct { max_concurrency } => {
|
||||||
|
let semaphore = tokio::sync::Semaphore::new(max_concurrency.get());
|
||||||
|
Self(Arc::new(Inner::Direct { semaphore }))
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
pub(crate) fn inner(&self) -> &Arc<Inner> {
|
||||||
|
&self.0
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
impl L0FlushConfig {
|
||||||
|
pub(crate) fn prewarm_on_write(&self) -> ephemeral_file::PrewarmPageCacheOnWrite {
|
||||||
|
use L0FlushConfig::*;
|
||||||
|
match self {
|
||||||
|
PageCached => ephemeral_file::PrewarmPageCacheOnWrite::Yes,
|
||||||
|
Direct { .. } => ephemeral_file::PrewarmPageCacheOnWrite::No,
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
@@ -11,6 +11,7 @@ pub mod deletion_queue;
|
|||||||
pub mod disk_usage_eviction_task;
|
pub mod disk_usage_eviction_task;
|
||||||
pub mod http;
|
pub mod http;
|
||||||
pub mod import_datadir;
|
pub mod import_datadir;
|
||||||
|
pub mod l0_flush;
|
||||||
pub use pageserver_api::keyspace;
|
pub use pageserver_api::keyspace;
|
||||||
pub mod aux_file;
|
pub mod aux_file;
|
||||||
pub mod metrics;
|
pub mod metrics;
|
||||||
|
|||||||
+106
-61
@@ -8,7 +8,7 @@ use metrics::{
|
|||||||
};
|
};
|
||||||
use once_cell::sync::Lazy;
|
use once_cell::sync::Lazy;
|
||||||
use pageserver_api::shard::TenantShardId;
|
use pageserver_api::shard::TenantShardId;
|
||||||
use strum::{EnumCount, IntoEnumIterator, VariantNames};
|
use strum::{EnumCount, VariantNames};
|
||||||
use strum_macros::{EnumVariantNames, IntoStaticStr};
|
use strum_macros::{EnumVariantNames, IntoStaticStr};
|
||||||
use tracing::warn;
|
use tracing::warn;
|
||||||
use utils::id::TimelineId;
|
use utils::id::TimelineId;
|
||||||
@@ -53,9 +53,6 @@ pub(crate) enum StorageTimeOperation {
|
|||||||
|
|
||||||
#[strum(serialize = "find gc cutoffs")]
|
#[strum(serialize = "find gc cutoffs")]
|
||||||
FindGcCutoffs,
|
FindGcCutoffs,
|
||||||
|
|
||||||
#[strum(serialize = "create tenant")]
|
|
||||||
CreateTenant,
|
|
||||||
}
|
}
|
||||||
|
|
||||||
pub(crate) static STORAGE_TIME_SUM_PER_TIMELINE: Lazy<CounterVec> = Lazy::new(|| {
|
pub(crate) static STORAGE_TIME_SUM_PER_TIMELINE: Lazy<CounterVec> = Lazy::new(|| {
|
||||||
@@ -467,6 +464,24 @@ static LAST_RECORD_LSN: Lazy<IntGaugeVec> = Lazy::new(|| {
|
|||||||
.expect("failed to define a metric")
|
.expect("failed to define a metric")
|
||||||
});
|
});
|
||||||
|
|
||||||
|
static PITR_HISTORY_SIZE: Lazy<UIntGaugeVec> = Lazy::new(|| {
|
||||||
|
register_uint_gauge_vec!(
|
||||||
|
"pageserver_pitr_history_size",
|
||||||
|
"Data written since PITR cutoff on this timeline",
|
||||||
|
&["tenant_id", "shard_id", "timeline_id"]
|
||||||
|
)
|
||||||
|
.expect("failed to define a metric")
|
||||||
|
});
|
||||||
|
|
||||||
|
static TIMELINE_ARCHIVE_SIZE: Lazy<UIntGaugeVec> = Lazy::new(|| {
|
||||||
|
register_uint_gauge_vec!(
|
||||||
|
"pageserver_archive_size",
|
||||||
|
"Timeline's logical size if it is considered eligible for archival (outside PITR window), else zero",
|
||||||
|
&["tenant_id", "shard_id", "timeline_id"]
|
||||||
|
)
|
||||||
|
.expect("failed to define a metric")
|
||||||
|
});
|
||||||
|
|
||||||
static STANDBY_HORIZON: Lazy<IntGaugeVec> = Lazy::new(|| {
|
static STANDBY_HORIZON: Lazy<IntGaugeVec> = Lazy::new(|| {
|
||||||
register_int_gauge_vec!(
|
register_int_gauge_vec!(
|
||||||
"pageserver_standby_horizon",
|
"pageserver_standby_horizon",
|
||||||
@@ -479,7 +494,7 @@ static STANDBY_HORIZON: Lazy<IntGaugeVec> = Lazy::new(|| {
|
|||||||
static RESIDENT_PHYSICAL_SIZE: Lazy<UIntGaugeVec> = Lazy::new(|| {
|
static RESIDENT_PHYSICAL_SIZE: Lazy<UIntGaugeVec> = Lazy::new(|| {
|
||||||
register_uint_gauge_vec!(
|
register_uint_gauge_vec!(
|
||||||
"pageserver_resident_physical_size",
|
"pageserver_resident_physical_size",
|
||||||
"The size of the layer files present in the pageserver's filesystem.",
|
"The size of the layer files present in the pageserver's filesystem, for attached locations.",
|
||||||
&["tenant_id", "shard_id", "timeline_id"]
|
&["tenant_id", "shard_id", "timeline_id"]
|
||||||
)
|
)
|
||||||
.expect("failed to define a metric")
|
.expect("failed to define a metric")
|
||||||
@@ -1079,21 +1094,12 @@ pub(crate) mod virtual_file_io_engine {
|
|||||||
});
|
});
|
||||||
}
|
}
|
||||||
|
|
||||||
#[derive(Debug)]
|
|
||||||
struct GlobalAndPerTimelineHistogram {
|
|
||||||
global: Histogram,
|
|
||||||
per_tenant_timeline: Histogram,
|
|
||||||
}
|
|
||||||
|
|
||||||
impl GlobalAndPerTimelineHistogram {
|
|
||||||
fn observe(&self, value: f64) {
|
|
||||||
self.global.observe(value);
|
|
||||||
self.per_tenant_timeline.observe(value);
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
struct GlobalAndPerTimelineHistogramTimer<'a, 'c> {
|
struct GlobalAndPerTimelineHistogramTimer<'a, 'c> {
|
||||||
h: &'a GlobalAndPerTimelineHistogram,
|
global_metric: &'a Histogram,
|
||||||
|
|
||||||
|
// Optional because not all op types are tracked per-timeline
|
||||||
|
timeline_metric: Option<&'a Histogram>,
|
||||||
|
|
||||||
ctx: &'c RequestContext,
|
ctx: &'c RequestContext,
|
||||||
start: std::time::Instant,
|
start: std::time::Instant,
|
||||||
op: SmgrQueryType,
|
op: SmgrQueryType,
|
||||||
@@ -1124,7 +1130,10 @@ impl<'a, 'c> Drop for GlobalAndPerTimelineHistogramTimer<'a, 'c> {
|
|||||||
elapsed
|
elapsed
|
||||||
}
|
}
|
||||||
};
|
};
|
||||||
self.h.observe(ex_throttled.as_secs_f64());
|
self.global_metric.observe(ex_throttled.as_secs_f64());
|
||||||
|
if let Some(timeline_metric) = self.timeline_metric {
|
||||||
|
timeline_metric.observe(ex_throttled.as_secs_f64());
|
||||||
|
}
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -1149,7 +1158,8 @@ pub enum SmgrQueryType {
|
|||||||
|
|
||||||
#[derive(Debug)]
|
#[derive(Debug)]
|
||||||
pub(crate) struct SmgrQueryTimePerTimeline {
|
pub(crate) struct SmgrQueryTimePerTimeline {
|
||||||
metrics: [GlobalAndPerTimelineHistogram; SmgrQueryType::COUNT],
|
global_metrics: [Histogram; SmgrQueryType::COUNT],
|
||||||
|
per_timeline_getpage: Histogram,
|
||||||
}
|
}
|
||||||
|
|
||||||
static SMGR_QUERY_TIME_PER_TENANT_TIMELINE: Lazy<HistogramVec> = Lazy::new(|| {
|
static SMGR_QUERY_TIME_PER_TENANT_TIMELINE: Lazy<HistogramVec> = Lazy::new(|| {
|
||||||
@@ -1227,27 +1237,32 @@ impl SmgrQueryTimePerTimeline {
|
|||||||
let tenant_id = tenant_shard_id.tenant_id.to_string();
|
let tenant_id = tenant_shard_id.tenant_id.to_string();
|
||||||
let shard_slug = format!("{}", tenant_shard_id.shard_slug());
|
let shard_slug = format!("{}", tenant_shard_id.shard_slug());
|
||||||
let timeline_id = timeline_id.to_string();
|
let timeline_id = timeline_id.to_string();
|
||||||
let metrics = std::array::from_fn(|i| {
|
let global_metrics = std::array::from_fn(|i| {
|
||||||
let op = SmgrQueryType::from_repr(i).unwrap();
|
let op = SmgrQueryType::from_repr(i).unwrap();
|
||||||
let global = SMGR_QUERY_TIME_GLOBAL
|
SMGR_QUERY_TIME_GLOBAL
|
||||||
.get_metric_with_label_values(&[op.into()])
|
.get_metric_with_label_values(&[op.into()])
|
||||||
.unwrap();
|
.unwrap()
|
||||||
let per_tenant_timeline = SMGR_QUERY_TIME_PER_TENANT_TIMELINE
|
|
||||||
.get_metric_with_label_values(&[op.into(), &tenant_id, &shard_slug, &timeline_id])
|
|
||||||
.unwrap();
|
|
||||||
GlobalAndPerTimelineHistogram {
|
|
||||||
global,
|
|
||||||
per_tenant_timeline,
|
|
||||||
}
|
|
||||||
});
|
});
|
||||||
Self { metrics }
|
|
||||||
|
let per_timeline_getpage = SMGR_QUERY_TIME_PER_TENANT_TIMELINE
|
||||||
|
.get_metric_with_label_values(&[
|
||||||
|
SmgrQueryType::GetPageAtLsn.into(),
|
||||||
|
&tenant_id,
|
||||||
|
&shard_slug,
|
||||||
|
&timeline_id,
|
||||||
|
])
|
||||||
|
.unwrap();
|
||||||
|
Self {
|
||||||
|
global_metrics,
|
||||||
|
per_timeline_getpage,
|
||||||
|
}
|
||||||
}
|
}
|
||||||
pub(crate) fn start_timer<'c: 'a, 'a>(
|
pub(crate) fn start_timer<'c: 'a, 'a>(
|
||||||
&'a self,
|
&'a self,
|
||||||
op: SmgrQueryType,
|
op: SmgrQueryType,
|
||||||
ctx: &'c RequestContext,
|
ctx: &'c RequestContext,
|
||||||
) -> impl Drop + '_ {
|
) -> Option<impl Drop + '_> {
|
||||||
let metric = &self.metrics[op as usize];
|
let global_metric = &self.global_metrics[op as usize];
|
||||||
let start = Instant::now();
|
let start = Instant::now();
|
||||||
match ctx.micros_spent_throttled.open() {
|
match ctx.micros_spent_throttled.open() {
|
||||||
Ok(()) => (),
|
Ok(()) => (),
|
||||||
@@ -1266,12 +1281,20 @@ impl SmgrQueryTimePerTimeline {
|
|||||||
});
|
});
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
GlobalAndPerTimelineHistogramTimer {
|
|
||||||
h: metric,
|
let timeline_metric = if matches!(op, SmgrQueryType::GetPageAtLsn) {
|
||||||
|
Some(&self.per_timeline_getpage)
|
||||||
|
} else {
|
||||||
|
None
|
||||||
|
};
|
||||||
|
|
||||||
|
Some(GlobalAndPerTimelineHistogramTimer {
|
||||||
|
global_metric,
|
||||||
|
timeline_metric,
|
||||||
ctx,
|
ctx,
|
||||||
start,
|
start,
|
||||||
op,
|
op,
|
||||||
}
|
})
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -1318,17 +1341,9 @@ mod smgr_query_time_tests {
|
|||||||
let get_counts = || {
|
let get_counts = || {
|
||||||
let global: u64 = ops
|
let global: u64 = ops
|
||||||
.iter()
|
.iter()
|
||||||
.map(|op| metrics.metrics[*op as usize].global.get_sample_count())
|
.map(|op| metrics.global_metrics[*op as usize].get_sample_count())
|
||||||
.sum();
|
.sum();
|
||||||
let per_tenant_timeline: u64 = ops
|
(global, metrics.per_timeline_getpage.get_sample_count())
|
||||||
.iter()
|
|
||||||
.map(|op| {
|
|
||||||
metrics.metrics[*op as usize]
|
|
||||||
.per_tenant_timeline
|
|
||||||
.get_sample_count()
|
|
||||||
})
|
|
||||||
.sum();
|
|
||||||
(global, per_tenant_timeline)
|
|
||||||
};
|
};
|
||||||
|
|
||||||
let (pre_global, pre_per_tenant_timeline) = get_counts();
|
let (pre_global, pre_per_tenant_timeline) = get_counts();
|
||||||
@@ -1339,7 +1354,12 @@ mod smgr_query_time_tests {
|
|||||||
drop(timer);
|
drop(timer);
|
||||||
|
|
||||||
let (post_global, post_per_tenant_timeline) = get_counts();
|
let (post_global, post_per_tenant_timeline) = get_counts();
|
||||||
assert_eq!(post_per_tenant_timeline, 1);
|
if matches!(op, super::SmgrQueryType::GetPageAtLsn) {
|
||||||
|
// getpage ops are tracked per-timeline, others aren't
|
||||||
|
assert_eq!(post_per_tenant_timeline, 1);
|
||||||
|
} else {
|
||||||
|
assert_eq!(post_per_tenant_timeline, 0);
|
||||||
|
}
|
||||||
assert!(post_global > pre_global);
|
assert!(post_global > pre_global);
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
@@ -1436,10 +1456,12 @@ impl<'a, 'c> BasebackupQueryTimeOngoingRecording<'a, 'c> {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
pub(crate) static LIVE_CONNECTIONS_COUNT: Lazy<IntGaugeVec> = Lazy::new(|| {
|
pub(crate) static LIVE_CONNECTIONS: Lazy<IntCounterPairVec> = Lazy::new(|| {
|
||||||
register_int_gauge_vec!(
|
register_int_counter_pair_vec!(
|
||||||
"pageserver_live_connections",
|
"pageserver_live_connections_started",
|
||||||
"Number of live network connections",
|
"Number of network connections that we started handling",
|
||||||
|
"pageserver_live_connections_finished",
|
||||||
|
"Number of network connections that we finished handling",
|
||||||
&["pageserver_connection_kind"]
|
&["pageserver_connection_kind"]
|
||||||
)
|
)
|
||||||
.expect("failed to define a metric")
|
.expect("failed to define a metric")
|
||||||
@@ -1450,7 +1472,6 @@ pub(crate) enum ComputeCommandKind {
|
|||||||
PageStreamV2,
|
PageStreamV2,
|
||||||
PageStream,
|
PageStream,
|
||||||
Basebackup,
|
Basebackup,
|
||||||
GetLastRecordRlsn,
|
|
||||||
Fullbackup,
|
Fullbackup,
|
||||||
ImportBasebackup,
|
ImportBasebackup,
|
||||||
ImportWal,
|
ImportWal,
|
||||||
@@ -1694,6 +1715,15 @@ pub(crate) static SECONDARY_MODE: Lazy<SecondaryModeMetrics> = Lazy::new(|| {
|
|||||||
}
|
}
|
||||||
});
|
});
|
||||||
|
|
||||||
|
pub(crate) static SECONDARY_RESIDENT_PHYSICAL_SIZE: Lazy<UIntGaugeVec> = Lazy::new(|| {
|
||||||
|
register_uint_gauge_vec!(
|
||||||
|
"pageserver_secondary_resident_physical_size",
|
||||||
|
"The size of the layer files present in the pageserver's filesystem, for secondary locations.",
|
||||||
|
&["tenant_id", "shard_id"]
|
||||||
|
)
|
||||||
|
.expect("failed to define a metric")
|
||||||
|
});
|
||||||
|
|
||||||
#[derive(Debug, Clone, Copy, PartialEq, Eq, Hash)]
|
#[derive(Debug, Clone, Copy, PartialEq, Eq, Hash)]
|
||||||
pub enum RemoteOpKind {
|
pub enum RemoteOpKind {
|
||||||
Upload,
|
Upload,
|
||||||
@@ -2096,6 +2126,8 @@ pub(crate) struct TimelineMetrics {
|
|||||||
pub garbage_collect_histo: StorageTimeMetrics,
|
pub garbage_collect_histo: StorageTimeMetrics,
|
||||||
pub find_gc_cutoffs_histo: StorageTimeMetrics,
|
pub find_gc_cutoffs_histo: StorageTimeMetrics,
|
||||||
pub last_record_gauge: IntGauge,
|
pub last_record_gauge: IntGauge,
|
||||||
|
pub pitr_history_size: UIntGauge,
|
||||||
|
pub archival_size: UIntGauge,
|
||||||
pub standby_horizon_gauge: IntGauge,
|
pub standby_horizon_gauge: IntGauge,
|
||||||
pub resident_physical_size_gauge: UIntGauge,
|
pub resident_physical_size_gauge: UIntGauge,
|
||||||
/// copy of LayeredTimeline.current_logical_size
|
/// copy of LayeredTimeline.current_logical_size
|
||||||
@@ -2169,6 +2201,15 @@ impl TimelineMetrics {
|
|||||||
let last_record_gauge = LAST_RECORD_LSN
|
let last_record_gauge = LAST_RECORD_LSN
|
||||||
.get_metric_with_label_values(&[&tenant_id, &shard_id, &timeline_id])
|
.get_metric_with_label_values(&[&tenant_id, &shard_id, &timeline_id])
|
||||||
.unwrap();
|
.unwrap();
|
||||||
|
|
||||||
|
let pitr_history_size = PITR_HISTORY_SIZE
|
||||||
|
.get_metric_with_label_values(&[&tenant_id, &shard_id, &timeline_id])
|
||||||
|
.unwrap();
|
||||||
|
|
||||||
|
let archival_size = TIMELINE_ARCHIVE_SIZE
|
||||||
|
.get_metric_with_label_values(&[&tenant_id, &shard_id, &timeline_id])
|
||||||
|
.unwrap();
|
||||||
|
|
||||||
let standby_horizon_gauge = STANDBY_HORIZON
|
let standby_horizon_gauge = STANDBY_HORIZON
|
||||||
.get_metric_with_label_values(&[&tenant_id, &shard_id, &timeline_id])
|
.get_metric_with_label_values(&[&tenant_id, &shard_id, &timeline_id])
|
||||||
.unwrap();
|
.unwrap();
|
||||||
@@ -2221,6 +2262,8 @@ impl TimelineMetrics {
|
|||||||
find_gc_cutoffs_histo,
|
find_gc_cutoffs_histo,
|
||||||
load_layer_map_histo,
|
load_layer_map_histo,
|
||||||
last_record_gauge,
|
last_record_gauge,
|
||||||
|
pitr_history_size,
|
||||||
|
archival_size,
|
||||||
standby_horizon_gauge,
|
standby_horizon_gauge,
|
||||||
resident_physical_size_gauge,
|
resident_physical_size_gauge,
|
||||||
current_logical_size_gauge,
|
current_logical_size_gauge,
|
||||||
@@ -2278,6 +2321,10 @@ impl TimelineMetrics {
|
|||||||
if let Some(metric) = Lazy::get(&DIRECTORY_ENTRIES_COUNT) {
|
if let Some(metric) = Lazy::get(&DIRECTORY_ENTRIES_COUNT) {
|
||||||
let _ = metric.remove_label_values(&[tenant_id, shard_id, timeline_id]);
|
let _ = metric.remove_label_values(&[tenant_id, shard_id, timeline_id]);
|
||||||
}
|
}
|
||||||
|
|
||||||
|
let _ = TIMELINE_ARCHIVE_SIZE.remove_label_values(&[tenant_id, shard_id, timeline_id]);
|
||||||
|
let _ = PITR_HISTORY_SIZE.remove_label_values(&[tenant_id, shard_id, timeline_id]);
|
||||||
|
|
||||||
let _ = EVICTIONS.remove_label_values(&[tenant_id, shard_id, timeline_id]);
|
let _ = EVICTIONS.remove_label_values(&[tenant_id, shard_id, timeline_id]);
|
||||||
let _ = AUX_FILE_SIZE.remove_label_values(&[tenant_id, shard_id, timeline_id]);
|
let _ = AUX_FILE_SIZE.remove_label_values(&[tenant_id, shard_id, timeline_id]);
|
||||||
let _ = VALID_LSN_LEASE_COUNT.remove_label_values(&[tenant_id, shard_id, timeline_id]);
|
let _ = VALID_LSN_LEASE_COUNT.remove_label_values(&[tenant_id, shard_id, timeline_id]);
|
||||||
@@ -2311,14 +2358,12 @@ impl TimelineMetrics {
|
|||||||
let _ = STORAGE_IO_SIZE.remove_label_values(&[op, tenant_id, shard_id, timeline_id]);
|
let _ = STORAGE_IO_SIZE.remove_label_values(&[op, tenant_id, shard_id, timeline_id]);
|
||||||
}
|
}
|
||||||
|
|
||||||
for op in SmgrQueryType::iter() {
|
let _ = SMGR_QUERY_TIME_PER_TENANT_TIMELINE.remove_label_values(&[
|
||||||
let _ = SMGR_QUERY_TIME_PER_TENANT_TIMELINE.remove_label_values(&[
|
SmgrQueryType::GetPageAtLsn.into(),
|
||||||
op.into(),
|
tenant_id,
|
||||||
tenant_id,
|
shard_id,
|
||||||
shard_id,
|
timeline_id,
|
||||||
timeline_id,
|
]);
|
||||||
]);
|
|
||||||
}
|
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
|||||||
@@ -55,7 +55,7 @@ use crate::basebackup::BasebackupError;
|
|||||||
use crate::context::{DownloadBehavior, RequestContext};
|
use crate::context::{DownloadBehavior, RequestContext};
|
||||||
use crate::import_datadir::import_wal_from_tar;
|
use crate::import_datadir::import_wal_from_tar;
|
||||||
use crate::metrics;
|
use crate::metrics;
|
||||||
use crate::metrics::{ComputeCommandKind, COMPUTE_COMMANDS_COUNTERS, LIVE_CONNECTIONS_COUNT};
|
use crate::metrics::{ComputeCommandKind, COMPUTE_COMMANDS_COUNTERS, LIVE_CONNECTIONS};
|
||||||
use crate::pgdatadir_mapping::Version;
|
use crate::pgdatadir_mapping::Version;
|
||||||
use crate::span::debug_assert_current_span_has_tenant_and_timeline_id;
|
use crate::span::debug_assert_current_span_has_tenant_and_timeline_id;
|
||||||
use crate::span::debug_assert_current_span_has_tenant_and_timeline_id_no_shard_id;
|
use crate::span::debug_assert_current_span_has_tenant_and_timeline_id_no_shard_id;
|
||||||
@@ -215,14 +215,9 @@ async fn page_service_conn_main(
|
|||||||
auth_type: AuthType,
|
auth_type: AuthType,
|
||||||
connection_ctx: RequestContext,
|
connection_ctx: RequestContext,
|
||||||
) -> anyhow::Result<()> {
|
) -> anyhow::Result<()> {
|
||||||
// Immediately increment the gauge, then create a job to decrement it on task exit.
|
let _guard = LIVE_CONNECTIONS
|
||||||
// One of the pros of `defer!` is that this will *most probably*
|
.with_label_values(&["page_service"])
|
||||||
// get called, even in presence of panics.
|
.guard();
|
||||||
let gauge = LIVE_CONNECTIONS_COUNT.with_label_values(&["page_service"]);
|
|
||||||
gauge.inc();
|
|
||||||
scopeguard::defer! {
|
|
||||||
gauge.dec();
|
|
||||||
}
|
|
||||||
|
|
||||||
socket
|
socket
|
||||||
.set_nodelay(true)
|
.set_nodelay(true)
|
||||||
@@ -1656,53 +1651,6 @@ where
|
|||||||
metric_recording.observe(&res);
|
metric_recording.observe(&res);
|
||||||
res?;
|
res?;
|
||||||
}
|
}
|
||||||
// return pair of prev_lsn and last_lsn
|
|
||||||
else if let Some(params) = parts.strip_prefix(&["get_last_record_rlsn"]) {
|
|
||||||
if params.len() != 2 {
|
|
||||||
return Err(QueryError::Other(anyhow::anyhow!(
|
|
||||||
"invalid param number for get_last_record_rlsn command"
|
|
||||||
)));
|
|
||||||
}
|
|
||||||
|
|
||||||
let tenant_id = TenantId::from_str(params[0])
|
|
||||||
.with_context(|| format!("Failed to parse tenant id from {}", params[0]))?;
|
|
||||||
let timeline_id = TimelineId::from_str(params[1])
|
|
||||||
.with_context(|| format!("Failed to parse timeline id from {}", params[1]))?;
|
|
||||||
|
|
||||||
tracing::Span::current()
|
|
||||||
.record("tenant_id", field::display(tenant_id))
|
|
||||||
.record("timeline_id", field::display(timeline_id));
|
|
||||||
|
|
||||||
self.check_permission(Some(tenant_id))?;
|
|
||||||
|
|
||||||
COMPUTE_COMMANDS_COUNTERS
|
|
||||||
.for_command(ComputeCommandKind::GetLastRecordRlsn)
|
|
||||||
.inc();
|
|
||||||
|
|
||||||
async {
|
|
||||||
let timeline = self
|
|
||||||
.get_active_tenant_timeline(tenant_id, timeline_id, ShardSelector::Zero)
|
|
||||||
.await?;
|
|
||||||
|
|
||||||
let end_of_timeline = timeline.get_last_record_rlsn();
|
|
||||||
|
|
||||||
pgb.write_message_noflush(&BeMessage::RowDescription(&[
|
|
||||||
RowDescriptor::text_col(b"prev_lsn"),
|
|
||||||
RowDescriptor::text_col(b"last_lsn"),
|
|
||||||
]))?
|
|
||||||
.write_message_noflush(&BeMessage::DataRow(&[
|
|
||||||
Some(end_of_timeline.prev.to_string().as_bytes()),
|
|
||||||
Some(end_of_timeline.last.to_string().as_bytes()),
|
|
||||||
]))?
|
|
||||||
.write_message_noflush(&BeMessage::CommandComplete(b"SELECT 1"))?;
|
|
||||||
anyhow::Ok(())
|
|
||||||
}
|
|
||||||
.instrument(info_span!(
|
|
||||||
"handle_get_last_record_lsn",
|
|
||||||
shard_id = tracing::field::Empty
|
|
||||||
))
|
|
||||||
.await?;
|
|
||||||
}
|
|
||||||
// same as basebackup, but result includes relational data as well
|
// same as basebackup, but result includes relational data as well
|
||||||
else if let Some(params) = parts.strip_prefix(&["fullbackup"]) {
|
else if let Some(params) = parts.strip_prefix(&["fullbackup"]) {
|
||||||
if params.len() < 2 {
|
if params.len() < 2 {
|
||||||
|
|||||||
+190
-142
@@ -73,6 +73,7 @@ use crate::deletion_queue::DeletionQueueClient;
|
|||||||
use crate::deletion_queue::DeletionQueueError;
|
use crate::deletion_queue::DeletionQueueError;
|
||||||
use crate::import_datadir;
|
use crate::import_datadir;
|
||||||
use crate::is_uninit_mark;
|
use crate::is_uninit_mark;
|
||||||
|
use crate::l0_flush::L0FlushGlobalState;
|
||||||
use crate::metrics::TENANT;
|
use crate::metrics::TENANT;
|
||||||
use crate::metrics::{
|
use crate::metrics::{
|
||||||
remove_tenant_metrics, BROKEN_TENANTS_SET, TENANT_STATE_METRIC, TENANT_SYNTHETIC_SIZE_METRIC,
|
remove_tenant_metrics, BROKEN_TENANTS_SET, TENANT_STATE_METRIC, TENANT_SYNTHETIC_SIZE_METRIC,
|
||||||
@@ -166,6 +167,7 @@ pub struct TenantSharedResources {
|
|||||||
pub broker_client: storage_broker::BrokerClientChannel,
|
pub broker_client: storage_broker::BrokerClientChannel,
|
||||||
pub remote_storage: GenericRemoteStorage,
|
pub remote_storage: GenericRemoteStorage,
|
||||||
pub deletion_queue_client: DeletionQueueClient,
|
pub deletion_queue_client: DeletionQueueClient,
|
||||||
|
pub l0_flush_global_state: L0FlushGlobalState,
|
||||||
}
|
}
|
||||||
|
|
||||||
/// A [`Tenant`] is really an _attached_ tenant. The configuration
|
/// A [`Tenant`] is really an _attached_ tenant. The configuration
|
||||||
@@ -213,8 +215,6 @@ pub(crate) enum SpawnMode {
|
|||||||
Eager,
|
Eager,
|
||||||
/// Lazy activation in the background, with the option to skip the queue if the need comes up
|
/// Lazy activation in the background, with the option to skip the queue if the need comes up
|
||||||
Lazy,
|
Lazy,
|
||||||
/// Tenant has been created during the lifetime of this process
|
|
||||||
Create,
|
|
||||||
}
|
}
|
||||||
|
|
||||||
///
|
///
|
||||||
@@ -296,6 +296,8 @@ pub struct Tenant {
|
|||||||
|
|
||||||
/// An ongoing timeline detach must be checked during attempts to GC or compact a timeline.
|
/// An ongoing timeline detach must be checked during attempts to GC or compact a timeline.
|
||||||
ongoing_timeline_detach: std::sync::Mutex<Option<(TimelineId, utils::completion::Barrier)>>,
|
ongoing_timeline_detach: std::sync::Mutex<Option<(TimelineId, utils::completion::Barrier)>>,
|
||||||
|
|
||||||
|
l0_flush_global_state: L0FlushGlobalState,
|
||||||
}
|
}
|
||||||
|
|
||||||
impl std::fmt::Debug for Tenant {
|
impl std::fmt::Debug for Tenant {
|
||||||
@@ -531,6 +533,15 @@ impl From<PageReconstructError> for GcError {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
#[derive(thiserror::Error, Debug)]
|
||||||
|
pub(crate) enum LoadConfigError {
|
||||||
|
#[error("TOML deserialization error: '{0}'")]
|
||||||
|
DeserializeToml(#[from] toml_edit::de::Error),
|
||||||
|
|
||||||
|
#[error("Config not found at {0}")]
|
||||||
|
NotFound(Utf8PathBuf),
|
||||||
|
}
|
||||||
|
|
||||||
impl Tenant {
|
impl Tenant {
|
||||||
/// Yet another helper for timeline initialization.
|
/// Yet another helper for timeline initialization.
|
||||||
///
|
///
|
||||||
@@ -669,6 +680,7 @@ impl Tenant {
|
|||||||
broker_client,
|
broker_client,
|
||||||
remote_storage,
|
remote_storage,
|
||||||
deletion_queue_client,
|
deletion_queue_client,
|
||||||
|
l0_flush_global_state,
|
||||||
} = resources;
|
} = resources;
|
||||||
|
|
||||||
let attach_mode = attached_conf.location.attach_mode;
|
let attach_mode = attached_conf.location.attach_mode;
|
||||||
@@ -683,6 +695,7 @@ impl Tenant {
|
|||||||
tenant_shard_id,
|
tenant_shard_id,
|
||||||
remote_storage.clone(),
|
remote_storage.clone(),
|
||||||
deletion_queue_client,
|
deletion_queue_client,
|
||||||
|
l0_flush_global_state,
|
||||||
));
|
));
|
||||||
|
|
||||||
// The attach task will carry a GateGuard, so that shutdown() reliably waits for it to drop out if
|
// The attach task will carry a GateGuard, so that shutdown() reliably waits for it to drop out if
|
||||||
@@ -808,9 +821,6 @@ impl Tenant {
|
|||||||
};
|
};
|
||||||
|
|
||||||
let preload = match &mode {
|
let preload = match &mode {
|
||||||
SpawnMode::Create => {
|
|
||||||
None
|
|
||||||
},
|
|
||||||
SpawnMode::Eager | SpawnMode::Lazy => {
|
SpawnMode::Eager | SpawnMode::Lazy => {
|
||||||
let _preload_timer = TENANT.preload.start_timer();
|
let _preload_timer = TENANT.preload.start_timer();
|
||||||
let res = tenant_clone
|
let res = tenant_clone
|
||||||
@@ -832,11 +842,8 @@ impl Tenant {
|
|||||||
|
|
||||||
// We will time the duration of the attach phase unless this is a creation (attach will do no work)
|
// We will time the duration of the attach phase unless this is a creation (attach will do no work)
|
||||||
let attached = {
|
let attached = {
|
||||||
let _attach_timer = match mode {
|
let _attach_timer = Some(TENANT.attach.start_timer());
|
||||||
SpawnMode::Create => None,
|
tenant_clone.attach(preload, &ctx).await
|
||||||
SpawnMode::Eager | SpawnMode::Lazy => Some(TENANT.attach.start_timer()),
|
|
||||||
};
|
|
||||||
tenant_clone.attach(preload, mode, &ctx).await
|
|
||||||
};
|
};
|
||||||
|
|
||||||
match attached {
|
match attached {
|
||||||
@@ -912,21 +919,14 @@ impl Tenant {
|
|||||||
async fn attach(
|
async fn attach(
|
||||||
self: &Arc<Tenant>,
|
self: &Arc<Tenant>,
|
||||||
preload: Option<TenantPreload>,
|
preload: Option<TenantPreload>,
|
||||||
mode: SpawnMode,
|
|
||||||
ctx: &RequestContext,
|
ctx: &RequestContext,
|
||||||
) -> anyhow::Result<()> {
|
) -> anyhow::Result<()> {
|
||||||
span::debug_assert_current_span_has_tenant_id();
|
span::debug_assert_current_span_has_tenant_id();
|
||||||
|
|
||||||
failpoint_support::sleep_millis_async!("before-attaching-tenant");
|
failpoint_support::sleep_millis_async!("before-attaching-tenant");
|
||||||
|
|
||||||
let preload = match (preload, mode) {
|
let Some(preload) = preload else {
|
||||||
(Some(p), _) => p,
|
anyhow::bail!("local-only deployment is no longer supported, https://github.com/neondatabase/neon/issues/5624");
|
||||||
(None, SpawnMode::Create) => TenantPreload {
|
|
||||||
timelines: HashMap::new(),
|
|
||||||
},
|
|
||||||
(None, _) => {
|
|
||||||
anyhow::bail!("local-only deployment is no longer supported, https://github.com/neondatabase/neon/issues/5624");
|
|
||||||
}
|
|
||||||
};
|
};
|
||||||
|
|
||||||
let mut timelines_to_resume_deletions = vec![];
|
let mut timelines_to_resume_deletions = vec![];
|
||||||
@@ -995,6 +995,7 @@ impl Tenant {
|
|||||||
TimelineResources {
|
TimelineResources {
|
||||||
remote_client,
|
remote_client,
|
||||||
timeline_get_throttle: self.timeline_get_throttle.clone(),
|
timeline_get_throttle: self.timeline_get_throttle.clone(),
|
||||||
|
l0_flush_global_state: self.l0_flush_global_state.clone(),
|
||||||
},
|
},
|
||||||
ctx,
|
ctx,
|
||||||
)
|
)
|
||||||
@@ -1364,7 +1365,7 @@ impl Tenant {
|
|||||||
initdb_lsn: Lsn,
|
initdb_lsn: Lsn,
|
||||||
pg_version: u32,
|
pg_version: u32,
|
||||||
ctx: &RequestContext,
|
ctx: &RequestContext,
|
||||||
delta_layer_desc: Vec<Vec<(pageserver_api::key::Key, Lsn, crate::repository::Value)>>,
|
delta_layer_desc: Vec<timeline::DeltaLayerTestDesc>,
|
||||||
image_layer_desc: Vec<(Lsn, Vec<(pageserver_api::key::Key, bytes::Bytes)>)>,
|
image_layer_desc: Vec<(Lsn, Vec<(pageserver_api::key::Key, bytes::Bytes)>)>,
|
||||||
end_lsn: Lsn,
|
end_lsn: Lsn,
|
||||||
) -> anyhow::Result<Arc<Timeline>> {
|
) -> anyhow::Result<Arc<Timeline>> {
|
||||||
@@ -1815,9 +1816,15 @@ impl Tenant {
|
|||||||
// If we're still attaching, fire the cancellation token early to drop out: this
|
// If we're still attaching, fire the cancellation token early to drop out: this
|
||||||
// will prevent us flushing, but ensures timely shutdown if some I/O during attach
|
// will prevent us flushing, but ensures timely shutdown if some I/O during attach
|
||||||
// is very slow.
|
// is very slow.
|
||||||
if matches!(self.current_state(), TenantState::Attaching) {
|
let shutdown_mode = if matches!(self.current_state(), TenantState::Attaching) {
|
||||||
self.cancel.cancel();
|
self.cancel.cancel();
|
||||||
}
|
|
||||||
|
// Having fired our cancellation token, do not try and flush timelines: their cancellation tokens
|
||||||
|
// are children of ours, so their flush loops will have shut down already
|
||||||
|
timeline::ShutdownMode::Hard
|
||||||
|
} else {
|
||||||
|
shutdown_mode
|
||||||
|
};
|
||||||
|
|
||||||
match self.set_stopping(shutdown_progress, false, false).await {
|
match self.set_stopping(shutdown_progress, false, false).await {
|
||||||
Ok(()) => {}
|
Ok(()) => {}
|
||||||
@@ -2484,6 +2491,7 @@ impl Tenant {
|
|||||||
tenant_shard_id: TenantShardId,
|
tenant_shard_id: TenantShardId,
|
||||||
remote_storage: GenericRemoteStorage,
|
remote_storage: GenericRemoteStorage,
|
||||||
deletion_queue_client: DeletionQueueClient,
|
deletion_queue_client: DeletionQueueClient,
|
||||||
|
l0_flush_global_state: L0FlushGlobalState,
|
||||||
) -> Tenant {
|
) -> Tenant {
|
||||||
debug_assert!(
|
debug_assert!(
|
||||||
!attached_conf.location.generation.is_none() || conf.control_plane_api.is_none()
|
!attached_conf.location.generation.is_none() || conf.control_plane_api.is_none()
|
||||||
@@ -2571,6 +2579,7 @@ impl Tenant {
|
|||||||
)),
|
)),
|
||||||
tenant_conf: Arc::new(ArcSwap::from_pointee(attached_conf)),
|
tenant_conf: Arc::new(ArcSwap::from_pointee(attached_conf)),
|
||||||
ongoing_timeline_detach: std::sync::Mutex::default(),
|
ongoing_timeline_detach: std::sync::Mutex::default(),
|
||||||
|
l0_flush_global_state,
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -2578,36 +2587,35 @@ impl Tenant {
|
|||||||
pub(super) fn load_tenant_config(
|
pub(super) fn load_tenant_config(
|
||||||
conf: &'static PageServerConf,
|
conf: &'static PageServerConf,
|
||||||
tenant_shard_id: &TenantShardId,
|
tenant_shard_id: &TenantShardId,
|
||||||
) -> anyhow::Result<LocationConf> {
|
) -> Result<LocationConf, LoadConfigError> {
|
||||||
let config_path = conf.tenant_location_config_path(tenant_shard_id);
|
let config_path = conf.tenant_location_config_path(tenant_shard_id);
|
||||||
|
|
||||||
if config_path.exists() {
|
info!("loading tenant configuration from {config_path}");
|
||||||
// New-style config takes precedence
|
|
||||||
let deserialized = Self::read_config(&config_path)?;
|
|
||||||
Ok(toml_edit::de::from_document::<LocationConf>(deserialized)?)
|
|
||||||
} else {
|
|
||||||
// The config should almost always exist for a tenant directory:
|
|
||||||
// - When attaching a tenant, the config is the first thing we write
|
|
||||||
// - When detaching a tenant, we atomically move the directory to a tmp location
|
|
||||||
// before deleting contents.
|
|
||||||
//
|
|
||||||
// The very rare edge case that can result in a missing config is if we crash during attach
|
|
||||||
// between creating directory and writing config. Callers should handle that as if the
|
|
||||||
// directory didn't exist.
|
|
||||||
anyhow::bail!("tenant config not found in {}", config_path);
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
fn read_config(path: &Utf8Path) -> anyhow::Result<toml_edit::Document> {
|
|
||||||
info!("loading tenant configuration from {path}");
|
|
||||||
|
|
||||||
// load and parse file
|
// load and parse file
|
||||||
let config = fs::read_to_string(path)
|
let config = fs::read_to_string(&config_path).map_err(|e| {
|
||||||
.with_context(|| format!("Failed to load config from path '{path}'"))?;
|
match e.kind() {
|
||||||
|
std::io::ErrorKind::NotFound => {
|
||||||
|
// The config should almost always exist for a tenant directory:
|
||||||
|
// - When attaching a tenant, the config is the first thing we write
|
||||||
|
// - When detaching a tenant, we atomically move the directory to a tmp location
|
||||||
|
// before deleting contents.
|
||||||
|
//
|
||||||
|
// The very rare edge case that can result in a missing config is if we crash during attach
|
||||||
|
// between creating directory and writing config. Callers should handle that as if the
|
||||||
|
// directory didn't exist.
|
||||||
|
|
||||||
config
|
LoadConfigError::NotFound(config_path)
|
||||||
.parse::<toml_edit::Document>()
|
}
|
||||||
.with_context(|| format!("Failed to parse config from file '{path}' as toml file"))
|
_ => {
|
||||||
|
// No IO errors except NotFound are acceptable here: other kinds of error indicate local storage or permissions issues
|
||||||
|
// that we cannot cleanly recover
|
||||||
|
crate::virtual_file::on_fatal_io_error(&e, "Reading tenant config file")
|
||||||
|
}
|
||||||
|
}
|
||||||
|
})?;
|
||||||
|
|
||||||
|
Ok(toml_edit::de::from_str::<LocationConf>(&config)?)
|
||||||
}
|
}
|
||||||
|
|
||||||
#[tracing::instrument(skip_all, fields(tenant_id=%tenant_shard_id.tenant_id, shard_id=%tenant_shard_id.shard_slug()))]
|
#[tracing::instrument(skip_all, fields(tenant_id=%tenant_shard_id.tenant_id, shard_id=%tenant_shard_id.shard_slug()))]
|
||||||
@@ -2615,7 +2623,7 @@ impl Tenant {
|
|||||||
conf: &'static PageServerConf,
|
conf: &'static PageServerConf,
|
||||||
tenant_shard_id: &TenantShardId,
|
tenant_shard_id: &TenantShardId,
|
||||||
location_conf: &LocationConf,
|
location_conf: &LocationConf,
|
||||||
) -> anyhow::Result<()> {
|
) -> std::io::Result<()> {
|
||||||
let config_path = conf.tenant_location_config_path(tenant_shard_id);
|
let config_path = conf.tenant_location_config_path(tenant_shard_id);
|
||||||
|
|
||||||
Self::persist_tenant_config_at(tenant_shard_id, &config_path, location_conf).await
|
Self::persist_tenant_config_at(tenant_shard_id, &config_path, location_conf).await
|
||||||
@@ -2626,7 +2634,7 @@ impl Tenant {
|
|||||||
tenant_shard_id: &TenantShardId,
|
tenant_shard_id: &TenantShardId,
|
||||||
config_path: &Utf8Path,
|
config_path: &Utf8Path,
|
||||||
location_conf: &LocationConf,
|
location_conf: &LocationConf,
|
||||||
) -> anyhow::Result<()> {
|
) -> std::io::Result<()> {
|
||||||
debug!("persisting tenantconf to {config_path}");
|
debug!("persisting tenantconf to {config_path}");
|
||||||
|
|
||||||
let mut conf_content = r#"# This file contains a specific per-tenant's config.
|
let mut conf_content = r#"# This file contains a specific per-tenant's config.
|
||||||
@@ -2635,22 +2643,20 @@ impl Tenant {
|
|||||||
.to_string();
|
.to_string();
|
||||||
|
|
||||||
fail::fail_point!("tenant-config-before-write", |_| {
|
fail::fail_point!("tenant-config-before-write", |_| {
|
||||||
anyhow::bail!("tenant-config-before-write");
|
Err(std::io::Error::new(
|
||||||
|
std::io::ErrorKind::Other,
|
||||||
|
"tenant-config-before-write",
|
||||||
|
))
|
||||||
});
|
});
|
||||||
|
|
||||||
// Convert the config to a toml file.
|
// Convert the config to a toml file.
|
||||||
conf_content += &toml_edit::ser::to_string_pretty(&location_conf)?;
|
conf_content +=
|
||||||
|
&toml_edit::ser::to_string_pretty(&location_conf).expect("Config serialization failed");
|
||||||
|
|
||||||
let temp_path = path_with_suffix_extension(config_path, TEMP_FILE_SUFFIX);
|
let temp_path = path_with_suffix_extension(config_path, TEMP_FILE_SUFFIX);
|
||||||
|
|
||||||
let tenant_shard_id = *tenant_shard_id;
|
|
||||||
let config_path = config_path.to_owned();
|
|
||||||
let conf_content = conf_content.into_bytes();
|
let conf_content = conf_content.into_bytes();
|
||||||
VirtualFile::crashsafe_overwrite(config_path.clone(), temp_path, conf_content)
|
VirtualFile::crashsafe_overwrite(config_path.to_owned(), temp_path, conf_content).await
|
||||||
.await
|
|
||||||
.with_context(|| format!("write tenant {tenant_shard_id} config to {config_path}"))?;
|
|
||||||
|
|
||||||
Ok(())
|
|
||||||
}
|
}
|
||||||
|
|
||||||
//
|
//
|
||||||
@@ -2868,6 +2874,7 @@ impl Tenant {
|
|||||||
{
|
{
|
||||||
let mut target = timeline.gc_info.write().unwrap();
|
let mut target = timeline.gc_info.write().unwrap();
|
||||||
|
|
||||||
|
// Cull any expired leases
|
||||||
let now = SystemTime::now();
|
let now = SystemTime::now();
|
||||||
target.leases.retain(|_, lease| !lease.is_expired(&now));
|
target.leases.retain(|_, lease| !lease.is_expired(&now));
|
||||||
|
|
||||||
@@ -2876,6 +2883,31 @@ impl Tenant {
|
|||||||
.valid_lsn_lease_count_gauge
|
.valid_lsn_lease_count_gauge
|
||||||
.set(target.leases.len() as u64);
|
.set(target.leases.len() as u64);
|
||||||
|
|
||||||
|
// Look up parent's PITR cutoff to update the child's knowledge of whether it is within parent's PITR
|
||||||
|
if let Some(ancestor_id) = timeline.get_ancestor_timeline_id() {
|
||||||
|
if let Some(ancestor_gc_cutoffs) = gc_cutoffs.get(&ancestor_id) {
|
||||||
|
target.within_ancestor_pitr =
|
||||||
|
timeline.get_ancestor_lsn() >= ancestor_gc_cutoffs.pitr;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// Update metrics that depend on GC state
|
||||||
|
timeline
|
||||||
|
.metrics
|
||||||
|
.archival_size
|
||||||
|
.set(if target.within_ancestor_pitr {
|
||||||
|
timeline.metrics.current_logical_size_gauge.get()
|
||||||
|
} else {
|
||||||
|
0
|
||||||
|
});
|
||||||
|
timeline.metrics.pitr_history_size.set(
|
||||||
|
timeline
|
||||||
|
.get_last_record_lsn()
|
||||||
|
.checked_sub(target.cutoffs.pitr)
|
||||||
|
.unwrap_or(Lsn(0))
|
||||||
|
.0,
|
||||||
|
);
|
||||||
|
|
||||||
match gc_cutoffs.remove(&timeline.timeline_id) {
|
match gc_cutoffs.remove(&timeline.timeline_id) {
|
||||||
Some(cutoffs) => {
|
Some(cutoffs) => {
|
||||||
target.retain_lsns = branchpoints;
|
target.retain_lsns = branchpoints;
|
||||||
@@ -2927,7 +2959,7 @@ impl Tenant {
|
|||||||
dst_id: TimelineId,
|
dst_id: TimelineId,
|
||||||
ancestor_lsn: Option<Lsn>,
|
ancestor_lsn: Option<Lsn>,
|
||||||
ctx: &RequestContext,
|
ctx: &RequestContext,
|
||||||
delta_layer_desc: Vec<Vec<(pageserver_api::key::Key, Lsn, crate::repository::Value)>>,
|
delta_layer_desc: Vec<timeline::DeltaLayerTestDesc>,
|
||||||
image_layer_desc: Vec<(Lsn, Vec<(pageserver_api::key::Key, bytes::Bytes)>)>,
|
image_layer_desc: Vec<(Lsn, Vec<(pageserver_api::key::Key, bytes::Bytes)>)>,
|
||||||
end_lsn: Lsn,
|
end_lsn: Lsn,
|
||||||
) -> anyhow::Result<Arc<Timeline>> {
|
) -> anyhow::Result<Arc<Timeline>> {
|
||||||
@@ -3311,6 +3343,7 @@ impl Tenant {
|
|||||||
TimelineResources {
|
TimelineResources {
|
||||||
remote_client,
|
remote_client,
|
||||||
timeline_get_throttle: self.timeline_get_throttle.clone(),
|
timeline_get_throttle: self.timeline_get_throttle.clone(),
|
||||||
|
l0_flush_global_state: self.l0_flush_global_state.clone(),
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -3647,6 +3680,7 @@ pub(crate) mod harness {
|
|||||||
use utils::logging;
|
use utils::logging;
|
||||||
|
|
||||||
use crate::deletion_queue::mock::MockDeletionQueue;
|
use crate::deletion_queue::mock::MockDeletionQueue;
|
||||||
|
use crate::l0_flush::L0FlushConfig;
|
||||||
use crate::walredo::apply_neon;
|
use crate::walredo::apply_neon;
|
||||||
use crate::{repository::Key, walrecord::NeonWalRecord};
|
use crate::{repository::Key, walrecord::NeonWalRecord};
|
||||||
|
|
||||||
@@ -3836,12 +3870,14 @@ pub(crate) mod harness {
|
|||||||
self.tenant_shard_id,
|
self.tenant_shard_id,
|
||||||
self.remote_storage.clone(),
|
self.remote_storage.clone(),
|
||||||
self.deletion_queue.new_client(),
|
self.deletion_queue.new_client(),
|
||||||
|
// TODO: ideally we should run all unit tests with both configs
|
||||||
|
L0FlushGlobalState::new(L0FlushConfig::default()),
|
||||||
));
|
));
|
||||||
|
|
||||||
let preload = tenant
|
let preload = tenant
|
||||||
.preload(&self.remote_storage, CancellationToken::new())
|
.preload(&self.remote_storage, CancellationToken::new())
|
||||||
.await?;
|
.await?;
|
||||||
tenant.attach(Some(preload), SpawnMode::Eager, ctx).await?;
|
tenant.attach(Some(preload), ctx).await?;
|
||||||
|
|
||||||
tenant.state.send_replace(TenantState::Active);
|
tenant.state.send_replace(TenantState::Active);
|
||||||
for timeline in tenant.timelines.lock().unwrap().values() {
|
for timeline in tenant.timelines.lock().unwrap().values() {
|
||||||
@@ -3923,7 +3959,7 @@ mod tests {
|
|||||||
use storage_layer::PersistentLayerKey;
|
use storage_layer::PersistentLayerKey;
|
||||||
use tests::storage_layer::ValuesReconstructState;
|
use tests::storage_layer::ValuesReconstructState;
|
||||||
use tests::timeline::{GetVectoredError, ShutdownMode};
|
use tests::timeline::{GetVectoredError, ShutdownMode};
|
||||||
use timeline::GcInfo;
|
use timeline::{DeltaLayerTestDesc, GcInfo};
|
||||||
use utils::bin_ser::BeSer;
|
use utils::bin_ser::BeSer;
|
||||||
use utils::id::TenantId;
|
use utils::id::TenantId;
|
||||||
|
|
||||||
@@ -6219,27 +6255,6 @@ mod tests {
|
|||||||
.await
|
.await
|
||||||
.unwrap();
|
.unwrap();
|
||||||
|
|
||||||
async fn get_vectored_impl_wrapper(
|
|
||||||
tline: &Arc<Timeline>,
|
|
||||||
key: Key,
|
|
||||||
lsn: Lsn,
|
|
||||||
ctx: &RequestContext,
|
|
||||||
) -> Result<Option<Bytes>, GetVectoredError> {
|
|
||||||
let mut reconstruct_state = ValuesReconstructState::new();
|
|
||||||
let mut res = tline
|
|
||||||
.get_vectored_impl(
|
|
||||||
KeySpace::single(key..key.next()),
|
|
||||||
lsn,
|
|
||||||
&mut reconstruct_state,
|
|
||||||
ctx,
|
|
||||||
)
|
|
||||||
.await?;
|
|
||||||
Ok(res.pop_last().map(|(k, v)| {
|
|
||||||
assert_eq!(k, key);
|
|
||||||
v.unwrap()
|
|
||||||
}))
|
|
||||||
}
|
|
||||||
|
|
||||||
let lsn = Lsn(0x30);
|
let lsn = Lsn(0x30);
|
||||||
|
|
||||||
// test vectored get on parent timeline
|
// test vectored get on parent timeline
|
||||||
@@ -6279,7 +6294,7 @@ mod tests {
|
|||||||
|
|
||||||
#[tokio::test]
|
#[tokio::test]
|
||||||
async fn test_vectored_missing_metadata_key_reads() -> anyhow::Result<()> {
|
async fn test_vectored_missing_metadata_key_reads() -> anyhow::Result<()> {
|
||||||
let harness = TenantHarness::create("test_vectored_missing_data_key_reads")?;
|
let harness = TenantHarness::create("test_vectored_missing_metadata_key_reads")?;
|
||||||
let (tenant, ctx) = harness.load().await;
|
let (tenant, ctx) = harness.load().await;
|
||||||
|
|
||||||
let base_key = Key::from_hex("620000000033333333444444445500000000").unwrap();
|
let base_key = Key::from_hex("620000000033333333444444445500000000").unwrap();
|
||||||
@@ -6315,27 +6330,6 @@ mod tests {
|
|||||||
.await
|
.await
|
||||||
.unwrap();
|
.unwrap();
|
||||||
|
|
||||||
async fn get_vectored_impl_wrapper(
|
|
||||||
tline: &Arc<Timeline>,
|
|
||||||
key: Key,
|
|
||||||
lsn: Lsn,
|
|
||||||
ctx: &RequestContext,
|
|
||||||
) -> Result<Option<Bytes>, GetVectoredError> {
|
|
||||||
let mut reconstruct_state = ValuesReconstructState::new();
|
|
||||||
let mut res = tline
|
|
||||||
.get_vectored_impl(
|
|
||||||
KeySpace::single(key..key.next()),
|
|
||||||
lsn,
|
|
||||||
&mut reconstruct_state,
|
|
||||||
ctx,
|
|
||||||
)
|
|
||||||
.await?;
|
|
||||||
Ok(res.pop_last().map(|(k, v)| {
|
|
||||||
assert_eq!(k, key);
|
|
||||||
v.unwrap()
|
|
||||||
}))
|
|
||||||
}
|
|
||||||
|
|
||||||
let lsn = Lsn(0x30);
|
let lsn = Lsn(0x30);
|
||||||
|
|
||||||
// test vectored get on parent timeline
|
// test vectored get on parent timeline
|
||||||
@@ -6411,9 +6405,18 @@ mod tests {
|
|||||||
&ctx,
|
&ctx,
|
||||||
// delta layers
|
// delta layers
|
||||||
vec![
|
vec![
|
||||||
vec![(key2, Lsn(0x10), Value::Image(test_img("metadata key 2")))],
|
DeltaLayerTestDesc::new_with_inferred_key_range(
|
||||||
vec![(key1, Lsn(0x20), Value::Image(Bytes::new()))],
|
Lsn(0x10)..Lsn(0x20),
|
||||||
vec![(key2, Lsn(0x20), Value::Image(Bytes::new()))],
|
vec![(key2, Lsn(0x10), Value::Image(test_img("metadata key 2")))],
|
||||||
|
),
|
||||||
|
DeltaLayerTestDesc::new_with_inferred_key_range(
|
||||||
|
Lsn(0x20)..Lsn(0x30),
|
||||||
|
vec![(key1, Lsn(0x20), Value::Image(Bytes::new()))],
|
||||||
|
),
|
||||||
|
DeltaLayerTestDesc::new_with_inferred_key_range(
|
||||||
|
Lsn(0x20)..Lsn(0x30),
|
||||||
|
vec![(key2, Lsn(0x20), Value::Image(Bytes::new()))],
|
||||||
|
),
|
||||||
],
|
],
|
||||||
// image layers
|
// image layers
|
||||||
vec![
|
vec![
|
||||||
@@ -6479,17 +6482,29 @@ mod tests {
|
|||||||
&ctx,
|
&ctx,
|
||||||
// delta layers
|
// delta layers
|
||||||
vec![
|
vec![
|
||||||
vec![(key2, Lsn(0x10), Value::Image(test_img("metadata key 2")))],
|
DeltaLayerTestDesc::new_with_inferred_key_range(
|
||||||
vec![(key1, Lsn(0x20), Value::Image(Bytes::new()))],
|
Lsn(0x10)..Lsn(0x20),
|
||||||
vec![(key2, Lsn(0x20), Value::Image(Bytes::new()))],
|
vec![(key2, Lsn(0x10), Value::Image(test_img("metadata key 2")))],
|
||||||
vec![
|
),
|
||||||
(key0, Lsn(0x30), Value::Image(test_img("metadata key 0"))),
|
DeltaLayerTestDesc::new_with_inferred_key_range(
|
||||||
(key3, Lsn(0x30), Value::Image(test_img("metadata key 3"))),
|
Lsn(0x20)..Lsn(0x30),
|
||||||
],
|
vec![(key1, Lsn(0x20), Value::Image(Bytes::new()))],
|
||||||
|
),
|
||||||
|
DeltaLayerTestDesc::new_with_inferred_key_range(
|
||||||
|
Lsn(0x20)..Lsn(0x30),
|
||||||
|
vec![(key2, Lsn(0x20), Value::Image(Bytes::new()))],
|
||||||
|
),
|
||||||
|
DeltaLayerTestDesc::new_with_inferred_key_range(
|
||||||
|
Lsn(0x30)..Lsn(0x40),
|
||||||
|
vec![
|
||||||
|
(key0, Lsn(0x30), Value::Image(test_img("metadata key 0"))),
|
||||||
|
(key3, Lsn(0x30), Value::Image(test_img("metadata key 3"))),
|
||||||
|
],
|
||||||
|
),
|
||||||
],
|
],
|
||||||
// image layers
|
// image layers
|
||||||
vec![(Lsn(0x10), vec![(key1, test_img("metadata key 1"))])],
|
vec![(Lsn(0x10), vec![(key1, test_img("metadata key 1"))])],
|
||||||
Lsn(0x30),
|
Lsn(0x40),
|
||||||
)
|
)
|
||||||
.await
|
.await
|
||||||
.unwrap();
|
.unwrap();
|
||||||
@@ -6512,7 +6527,7 @@ mod tests {
|
|||||||
|
|
||||||
// Image layers are created at last_record_lsn
|
// Image layers are created at last_record_lsn
|
||||||
let images = tline
|
let images = tline
|
||||||
.inspect_image_layers(Lsn(0x30), &ctx)
|
.inspect_image_layers(Lsn(0x40), &ctx)
|
||||||
.await
|
.await
|
||||||
.unwrap()
|
.unwrap()
|
||||||
.into_iter()
|
.into_iter()
|
||||||
@@ -6538,9 +6553,18 @@ mod tests {
|
|||||||
&ctx,
|
&ctx,
|
||||||
// delta layers
|
// delta layers
|
||||||
vec![
|
vec![
|
||||||
vec![(key2, Lsn(0x10), Value::Image(test_img("metadata key 2")))],
|
DeltaLayerTestDesc::new_with_inferred_key_range(
|
||||||
vec![(key1, Lsn(0x20), Value::Image(Bytes::new()))],
|
Lsn(0x10)..Lsn(0x20),
|
||||||
vec![(key2, Lsn(0x20), Value::Image(Bytes::new()))],
|
vec![(key2, Lsn(0x10), Value::Image(test_img("metadata key 2")))],
|
||||||
|
),
|
||||||
|
DeltaLayerTestDesc::new_with_inferred_key_range(
|
||||||
|
Lsn(0x20)..Lsn(0x30),
|
||||||
|
vec![(key1, Lsn(0x20), Value::Image(Bytes::new()))],
|
||||||
|
),
|
||||||
|
DeltaLayerTestDesc::new_with_inferred_key_range(
|
||||||
|
Lsn(0x20)..Lsn(0x30),
|
||||||
|
vec![(key2, Lsn(0x20), Value::Image(Bytes::new()))],
|
||||||
|
),
|
||||||
],
|
],
|
||||||
// image layers
|
// image layers
|
||||||
vec![(Lsn(0x10), vec![(key1, test_img("metadata key 1"))])],
|
vec![(Lsn(0x10), vec![(key1, test_img("metadata key 1"))])],
|
||||||
@@ -6588,15 +6612,21 @@ mod tests {
|
|||||||
key
|
key
|
||||||
}
|
}
|
||||||
|
|
||||||
// We create one bottom-most image layer, a delta layer D1 crossing the GC horizon, D2 below the horizon, and D3 above the horizon.
|
// We create
|
||||||
|
// - one bottom-most image layer,
|
||||||
|
// - a delta layer D1 crossing the GC horizon with data below and above the horizon,
|
||||||
|
// - a delta layer D2 crossing the GC horizon with data only below the horizon,
|
||||||
|
// - a delta layer D3 above the horizon.
|
||||||
//
|
//
|
||||||
// | D1 | | D3 |
|
// | D3 |
|
||||||
|
// | D1 |
|
||||||
// -| |-- gc horizon -----------------
|
// -| |-- gc horizon -----------------
|
||||||
// | | | D2 |
|
// | | | D2 |
|
||||||
// --------- img layer ------------------
|
// --------- img layer ------------------
|
||||||
//
|
//
|
||||||
// What we should expact from this compaction is:
|
// What we should expact from this compaction is:
|
||||||
// | Part of D1 | | D3 |
|
// | D3 |
|
||||||
|
// | Part of D1 |
|
||||||
// --------- img layer with D1+D2 at GC horizon------------------
|
// --------- img layer with D1+D2 at GC horizon------------------
|
||||||
|
|
||||||
// img layer at 0x10
|
// img layer at 0x10
|
||||||
@@ -6636,13 +6666,13 @@ mod tests {
|
|||||||
let delta3 = vec![
|
let delta3 = vec![
|
||||||
(
|
(
|
||||||
get_key(8),
|
get_key(8),
|
||||||
Lsn(0x40),
|
Lsn(0x48),
|
||||||
Value::Image(Bytes::from("value 8@0x40")),
|
Value::Image(Bytes::from("value 8@0x48")),
|
||||||
),
|
),
|
||||||
(
|
(
|
||||||
get_key(9),
|
get_key(9),
|
||||||
Lsn(0x40),
|
Lsn(0x48),
|
||||||
Value::Image(Bytes::from("value 9@0x40")),
|
Value::Image(Bytes::from("value 9@0x48")),
|
||||||
),
|
),
|
||||||
];
|
];
|
||||||
|
|
||||||
@@ -6652,7 +6682,11 @@ mod tests {
|
|||||||
Lsn(0x10),
|
Lsn(0x10),
|
||||||
DEFAULT_PG_VERSION,
|
DEFAULT_PG_VERSION,
|
||||||
&ctx,
|
&ctx,
|
||||||
vec![delta1, delta2, delta3], // delta layers
|
vec![
|
||||||
|
DeltaLayerTestDesc::new_with_inferred_key_range(Lsn(0x20)..Lsn(0x48), delta1),
|
||||||
|
DeltaLayerTestDesc::new_with_inferred_key_range(Lsn(0x20)..Lsn(0x48), delta2),
|
||||||
|
DeltaLayerTestDesc::new_with_inferred_key_range(Lsn(0x48)..Lsn(0x50), delta3),
|
||||||
|
], // delta layers
|
||||||
vec![(Lsn(0x10), img_layer)], // image layers
|
vec![(Lsn(0x10), img_layer)], // image layers
|
||||||
Lsn(0x50),
|
Lsn(0x50),
|
||||||
)
|
)
|
||||||
@@ -6673,8 +6707,8 @@ mod tests {
|
|||||||
Bytes::from_static(b"value 5@0x20"),
|
Bytes::from_static(b"value 5@0x20"),
|
||||||
Bytes::from_static(b"value 6@0x20"),
|
Bytes::from_static(b"value 6@0x20"),
|
||||||
Bytes::from_static(b"value 7@0x10"),
|
Bytes::from_static(b"value 7@0x10"),
|
||||||
Bytes::from_static(b"value 8@0x40"),
|
Bytes::from_static(b"value 8@0x48"),
|
||||||
Bytes::from_static(b"value 9@0x40"),
|
Bytes::from_static(b"value 9@0x48"),
|
||||||
];
|
];
|
||||||
|
|
||||||
for (idx, expected) in expected_result.iter().enumerate() {
|
for (idx, expected) in expected_result.iter().enumerate() {
|
||||||
@@ -6762,10 +6796,10 @@ mod tests {
|
|||||||
lsn_range: Lsn(0x30)..Lsn(0x41),
|
lsn_range: Lsn(0x30)..Lsn(0x41),
|
||||||
is_delta: true
|
is_delta: true
|
||||||
},
|
},
|
||||||
// The delta layer we created and should not be picked for the compaction
|
// The delta3 layer that should not be picked for the compaction
|
||||||
PersistentLayerKey {
|
PersistentLayerKey {
|
||||||
key_range: get_key(8)..get_key(10),
|
key_range: get_key(8)..get_key(10),
|
||||||
lsn_range: Lsn(0x40)..Lsn(0x41),
|
lsn_range: Lsn(0x48)..Lsn(0x50),
|
||||||
is_delta: true
|
is_delta: true
|
||||||
}
|
}
|
||||||
]
|
]
|
||||||
@@ -6829,7 +6863,10 @@ mod tests {
|
|||||||
Lsn(0x10),
|
Lsn(0x10),
|
||||||
DEFAULT_PG_VERSION,
|
DEFAULT_PG_VERSION,
|
||||||
&ctx,
|
&ctx,
|
||||||
vec![delta1], // delta layers
|
vec![DeltaLayerTestDesc::new_with_inferred_key_range(
|
||||||
|
Lsn(0x10)..Lsn(0x40),
|
||||||
|
delta1,
|
||||||
|
)], // delta layers
|
||||||
vec![(Lsn(0x10), image1)], // image layers
|
vec![(Lsn(0x10), image1)], // image layers
|
||||||
Lsn(0x50),
|
Lsn(0x50),
|
||||||
)
|
)
|
||||||
@@ -6953,15 +6990,21 @@ mod tests {
|
|||||||
key
|
key
|
||||||
}
|
}
|
||||||
|
|
||||||
// We create one bottom-most image layer, a delta layer D1 crossing the GC horizon, D2 below the horizon, and D3 above the horizon.
|
// We create
|
||||||
|
// - one bottom-most image layer,
|
||||||
|
// - a delta layer D1 crossing the GC horizon with data below and above the horizon,
|
||||||
|
// - a delta layer D2 crossing the GC horizon with data only below the horizon,
|
||||||
|
// - a delta layer D3 above the horizon.
|
||||||
//
|
//
|
||||||
// | D1 | | D3 |
|
// | D3 |
|
||||||
|
// | D1 |
|
||||||
// -| |-- gc horizon -----------------
|
// -| |-- gc horizon -----------------
|
||||||
// | | | D2 |
|
// | | | D2 |
|
||||||
// --------- img layer ------------------
|
// --------- img layer ------------------
|
||||||
//
|
//
|
||||||
// What we should expact from this compaction is:
|
// What we should expact from this compaction is:
|
||||||
// | Part of D1 | | D3 |
|
// | D3 |
|
||||||
|
// | Part of D1 |
|
||||||
// --------- img layer with D1+D2 at GC horizon------------------
|
// --------- img layer with D1+D2 at GC horizon------------------
|
||||||
|
|
||||||
// img layer at 0x10
|
// img layer at 0x10
|
||||||
@@ -7011,13 +7054,13 @@ mod tests {
|
|||||||
let delta3 = vec![
|
let delta3 = vec![
|
||||||
(
|
(
|
||||||
get_key(8),
|
get_key(8),
|
||||||
Lsn(0x40),
|
Lsn(0x48),
|
||||||
Value::WalRecord(NeonWalRecord::wal_append("@0x40")),
|
Value::WalRecord(NeonWalRecord::wal_append("@0x48")),
|
||||||
),
|
),
|
||||||
(
|
(
|
||||||
get_key(9),
|
get_key(9),
|
||||||
Lsn(0x40),
|
Lsn(0x48),
|
||||||
Value::WalRecord(NeonWalRecord::wal_append("@0x40")),
|
Value::WalRecord(NeonWalRecord::wal_append("@0x48")),
|
||||||
),
|
),
|
||||||
];
|
];
|
||||||
|
|
||||||
@@ -7027,7 +7070,11 @@ mod tests {
|
|||||||
Lsn(0x10),
|
Lsn(0x10),
|
||||||
DEFAULT_PG_VERSION,
|
DEFAULT_PG_VERSION,
|
||||||
&ctx,
|
&ctx,
|
||||||
vec![delta1, delta2, delta3], // delta layers
|
vec![
|
||||||
|
DeltaLayerTestDesc::new_with_inferred_key_range(Lsn(0x10)..Lsn(0x48), delta1),
|
||||||
|
DeltaLayerTestDesc::new_with_inferred_key_range(Lsn(0x10)..Lsn(0x48), delta2),
|
||||||
|
DeltaLayerTestDesc::new_with_inferred_key_range(Lsn(0x48)..Lsn(0x50), delta3),
|
||||||
|
], // delta layers
|
||||||
vec![(Lsn(0x10), img_layer)], // image layers
|
vec![(Lsn(0x10), img_layer)], // image layers
|
||||||
Lsn(0x50),
|
Lsn(0x50),
|
||||||
)
|
)
|
||||||
@@ -7042,6 +7089,7 @@ mod tests {
|
|||||||
horizon: Lsn(0x30),
|
horizon: Lsn(0x30),
|
||||||
},
|
},
|
||||||
leases: Default::default(),
|
leases: Default::default(),
|
||||||
|
within_ancestor_pitr: false,
|
||||||
};
|
};
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -7054,8 +7102,8 @@ mod tests {
|
|||||||
Bytes::from_static(b"value 5@0x10@0x20"),
|
Bytes::from_static(b"value 5@0x10@0x20"),
|
||||||
Bytes::from_static(b"value 6@0x10@0x20"),
|
Bytes::from_static(b"value 6@0x10@0x20"),
|
||||||
Bytes::from_static(b"value 7@0x10"),
|
Bytes::from_static(b"value 7@0x10"),
|
||||||
Bytes::from_static(b"value 8@0x10@0x40"),
|
Bytes::from_static(b"value 8@0x10@0x48"),
|
||||||
Bytes::from_static(b"value 9@0x10@0x40"),
|
Bytes::from_static(b"value 9@0x10@0x48"),
|
||||||
];
|
];
|
||||||
|
|
||||||
let expected_result_at_gc_horizon = [
|
let expected_result_at_gc_horizon = [
|
||||||
|
|||||||
@@ -6,13 +6,20 @@
|
|||||||
//! is written as a one byte. If it's larger than that, the length
|
//! is written as a one byte. If it's larger than that, the length
|
||||||
//! is written as a four-byte integer, in big-endian, with the high
|
//! is written as a four-byte integer, in big-endian, with the high
|
||||||
//! bit set. This way, we can detect whether it's 1- or 4-byte header
|
//! bit set. This way, we can detect whether it's 1- or 4-byte header
|
||||||
//! by peeking at the first byte.
|
//! by peeking at the first byte. For blobs larger than 128 bits,
|
||||||
|
//! we also specify three reserved bits, only one of the three bit
|
||||||
|
//! patterns is currently in use (0b011) and signifies compression
|
||||||
|
//! with zstd.
|
||||||
//!
|
//!
|
||||||
//! len < 128: 0XXXXXXX
|
//! len < 128: 0XXXXXXX
|
||||||
//! len >= 128: 1XXXXXXX XXXXXXXX XXXXXXXX XXXXXXXX
|
//! len >= 128: 1CCCXXXX XXXXXXXX XXXXXXXX XXXXXXXX
|
||||||
//!
|
//!
|
||||||
|
use async_compression::Level;
|
||||||
use bytes::{BufMut, BytesMut};
|
use bytes::{BufMut, BytesMut};
|
||||||
|
use pageserver_api::models::ImageCompressionAlgorithm;
|
||||||
|
use tokio::io::AsyncWriteExt;
|
||||||
use tokio_epoll_uring::{BoundedBuf, IoBuf, Slice};
|
use tokio_epoll_uring::{BoundedBuf, IoBuf, Slice};
|
||||||
|
use tracing::warn;
|
||||||
|
|
||||||
use crate::context::RequestContext;
|
use crate::context::RequestContext;
|
||||||
use crate::page_cache::PAGE_SZ;
|
use crate::page_cache::PAGE_SZ;
|
||||||
@@ -66,12 +73,37 @@ impl<'a> BlockCursor<'a> {
|
|||||||
len_buf.copy_from_slice(&buf[off..off + 4]);
|
len_buf.copy_from_slice(&buf[off..off + 4]);
|
||||||
off += 4;
|
off += 4;
|
||||||
}
|
}
|
||||||
len_buf[0] &= 0x7f;
|
let bit_mask = if self.read_compressed {
|
||||||
|
!LEN_COMPRESSION_BIT_MASK
|
||||||
|
} else {
|
||||||
|
0x7f
|
||||||
|
};
|
||||||
|
len_buf[0] &= bit_mask;
|
||||||
u32::from_be_bytes(len_buf) as usize
|
u32::from_be_bytes(len_buf) as usize
|
||||||
};
|
};
|
||||||
|
let compression_bits = first_len_byte & LEN_COMPRESSION_BIT_MASK;
|
||||||
|
|
||||||
dstbuf.clear();
|
let mut tmp_buf = Vec::new();
|
||||||
dstbuf.reserve(len);
|
let buf_to_write;
|
||||||
|
let compression = if compression_bits <= BYTE_UNCOMPRESSED || !self.read_compressed {
|
||||||
|
if compression_bits > BYTE_UNCOMPRESSED {
|
||||||
|
warn!("reading key above future limit ({len} bytes)");
|
||||||
|
}
|
||||||
|
buf_to_write = dstbuf;
|
||||||
|
None
|
||||||
|
} else if compression_bits == BYTE_ZSTD {
|
||||||
|
buf_to_write = &mut tmp_buf;
|
||||||
|
Some(dstbuf)
|
||||||
|
} else {
|
||||||
|
let error = std::io::Error::new(
|
||||||
|
std::io::ErrorKind::InvalidData,
|
||||||
|
format!("invalid compression byte {compression_bits:x}"),
|
||||||
|
);
|
||||||
|
return Err(error);
|
||||||
|
};
|
||||||
|
|
||||||
|
buf_to_write.clear();
|
||||||
|
buf_to_write.reserve(len);
|
||||||
|
|
||||||
// Read the payload
|
// Read the payload
|
||||||
let mut remain = len;
|
let mut remain = len;
|
||||||
@@ -85,14 +117,35 @@ impl<'a> BlockCursor<'a> {
|
|||||||
page_remain = PAGE_SZ;
|
page_remain = PAGE_SZ;
|
||||||
}
|
}
|
||||||
let this_blk_len = min(remain, page_remain);
|
let this_blk_len = min(remain, page_remain);
|
||||||
dstbuf.extend_from_slice(&buf[off..off + this_blk_len]);
|
buf_to_write.extend_from_slice(&buf[off..off + this_blk_len]);
|
||||||
remain -= this_blk_len;
|
remain -= this_blk_len;
|
||||||
off += this_blk_len;
|
off += this_blk_len;
|
||||||
}
|
}
|
||||||
|
|
||||||
|
if let Some(dstbuf) = compression {
|
||||||
|
if compression_bits == BYTE_ZSTD {
|
||||||
|
let mut decoder = async_compression::tokio::write::ZstdDecoder::new(dstbuf);
|
||||||
|
decoder.write_all(buf_to_write).await?;
|
||||||
|
decoder.flush().await?;
|
||||||
|
} else {
|
||||||
|
unreachable!("already checked above")
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
Ok(())
|
Ok(())
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// Reserved bits for length and compression
|
||||||
|
const LEN_COMPRESSION_BIT_MASK: u8 = 0xf0;
|
||||||
|
|
||||||
|
/// The maximum size of blobs we support. The highest few bits
|
||||||
|
/// are reserved for compression and other further uses.
|
||||||
|
const MAX_SUPPORTED_LEN: usize = 0x0fff_ffff;
|
||||||
|
|
||||||
|
const BYTE_UNCOMPRESSED: u8 = 0x80;
|
||||||
|
const BYTE_ZSTD: u8 = BYTE_UNCOMPRESSED | 0x10;
|
||||||
|
|
||||||
/// A wrapper of `VirtualFile` that allows users to write blobs.
|
/// A wrapper of `VirtualFile` that allows users to write blobs.
|
||||||
///
|
///
|
||||||
/// If a `BlobWriter` is dropped, the internal buffer will be
|
/// If a `BlobWriter` is dropped, the internal buffer will be
|
||||||
@@ -219,6 +272,22 @@ impl<const BUFFERED: bool> BlobWriter<BUFFERED> {
|
|||||||
&mut self,
|
&mut self,
|
||||||
srcbuf: B,
|
srcbuf: B,
|
||||||
ctx: &RequestContext,
|
ctx: &RequestContext,
|
||||||
|
) -> (B::Buf, Result<u64, Error>) {
|
||||||
|
self.write_blob_maybe_compressed(
|
||||||
|
srcbuf,
|
||||||
|
ctx,
|
||||||
|
ImageCompressionAlgorithm::DisabledNoDecompress,
|
||||||
|
)
|
||||||
|
.await
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Write a blob of data. Returns the offset that it was written to,
|
||||||
|
/// which can be used to retrieve the data later.
|
||||||
|
pub async fn write_blob_maybe_compressed<B: BoundedBuf<Buf = Buf>, Buf: IoBuf + Send>(
|
||||||
|
&mut self,
|
||||||
|
srcbuf: B,
|
||||||
|
ctx: &RequestContext,
|
||||||
|
algorithm: ImageCompressionAlgorithm,
|
||||||
) -> (B::Buf, Result<u64, Error>) {
|
) -> (B::Buf, Result<u64, Error>) {
|
||||||
let offset = self.offset;
|
let offset = self.offset;
|
||||||
|
|
||||||
@@ -226,29 +295,61 @@ impl<const BUFFERED: bool> BlobWriter<BUFFERED> {
|
|||||||
|
|
||||||
let mut io_buf = self.io_buf.take().expect("we always put it back below");
|
let mut io_buf = self.io_buf.take().expect("we always put it back below");
|
||||||
io_buf.clear();
|
io_buf.clear();
|
||||||
let (io_buf, hdr_res) = async {
|
let mut compressed_buf = None;
|
||||||
|
let ((io_buf, hdr_res), srcbuf) = async {
|
||||||
if len < 128 {
|
if len < 128 {
|
||||||
// Short blob. Write a 1-byte length header
|
// Short blob. Write a 1-byte length header
|
||||||
io_buf.put_u8(len as u8);
|
io_buf.put_u8(len as u8);
|
||||||
self.write_all(io_buf, ctx).await
|
(
|
||||||
|
self.write_all(io_buf, ctx).await,
|
||||||
|
srcbuf.slice_full().into_inner(),
|
||||||
|
)
|
||||||
} else {
|
} else {
|
||||||
// Write a 4-byte length header
|
// Write a 4-byte length header
|
||||||
if len > 0x7fff_ffff {
|
if len > MAX_SUPPORTED_LEN {
|
||||||
return (
|
return (
|
||||||
io_buf,
|
(
|
||||||
Err(Error::new(
|
io_buf,
|
||||||
ErrorKind::Other,
|
Err(Error::new(
|
||||||
format!("blob too large ({len} bytes)"),
|
ErrorKind::Other,
|
||||||
)),
|
format!("blob too large ({len} bytes)"),
|
||||||
|
)),
|
||||||
|
),
|
||||||
|
srcbuf.slice_full().into_inner(),
|
||||||
);
|
);
|
||||||
}
|
}
|
||||||
if len > 0x0fff_ffff {
|
let (high_bit_mask, len_written, srcbuf) = match algorithm {
|
||||||
tracing::warn!("writing blob above future limit ({len} bytes)");
|
ImageCompressionAlgorithm::Zstd { level } => {
|
||||||
}
|
let mut encoder = if let Some(level) = level {
|
||||||
let mut len_buf = (len as u32).to_be_bytes();
|
async_compression::tokio::write::ZstdEncoder::with_quality(
|
||||||
len_buf[0] |= 0x80;
|
Vec::new(),
|
||||||
|
Level::Precise(level.into()),
|
||||||
|
)
|
||||||
|
} else {
|
||||||
|
async_compression::tokio::write::ZstdEncoder::new(Vec::new())
|
||||||
|
};
|
||||||
|
let slice = srcbuf.slice_full();
|
||||||
|
encoder.write_all(&slice[..]).await.unwrap();
|
||||||
|
encoder.shutdown().await.unwrap();
|
||||||
|
let compressed = encoder.into_inner();
|
||||||
|
if compressed.len() < len {
|
||||||
|
let compressed_len = compressed.len();
|
||||||
|
compressed_buf = Some(compressed);
|
||||||
|
(BYTE_ZSTD, compressed_len, slice.into_inner())
|
||||||
|
} else {
|
||||||
|
(BYTE_UNCOMPRESSED, len, slice.into_inner())
|
||||||
|
}
|
||||||
|
}
|
||||||
|
ImageCompressionAlgorithm::Disabled
|
||||||
|
| ImageCompressionAlgorithm::DisabledNoDecompress => {
|
||||||
|
(BYTE_UNCOMPRESSED, len, srcbuf.slice_full().into_inner())
|
||||||
|
}
|
||||||
|
};
|
||||||
|
let mut len_buf = (len_written as u32).to_be_bytes();
|
||||||
|
assert_eq!(len_buf[0] & 0xf0, 0);
|
||||||
|
len_buf[0] |= high_bit_mask;
|
||||||
io_buf.extend_from_slice(&len_buf[..]);
|
io_buf.extend_from_slice(&len_buf[..]);
|
||||||
self.write_all(io_buf, ctx).await
|
(self.write_all(io_buf, ctx).await, srcbuf)
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
.await;
|
.await;
|
||||||
@@ -257,7 +358,12 @@ impl<const BUFFERED: bool> BlobWriter<BUFFERED> {
|
|||||||
Ok(_) => (),
|
Ok(_) => (),
|
||||||
Err(e) => return (Slice::into_inner(srcbuf.slice(..)), Err(e)),
|
Err(e) => return (Slice::into_inner(srcbuf.slice(..)), Err(e)),
|
||||||
}
|
}
|
||||||
let (srcbuf, res) = self.write_all(srcbuf, ctx).await;
|
let (srcbuf, res) = if let Some(compressed_buf) = compressed_buf {
|
||||||
|
let (_buf, res) = self.write_all(compressed_buf, ctx).await;
|
||||||
|
(Slice::into_inner(srcbuf.slice(..)), res)
|
||||||
|
} else {
|
||||||
|
self.write_all(srcbuf, ctx).await
|
||||||
|
};
|
||||||
(srcbuf, res.map(|_| offset))
|
(srcbuf, res.map(|_| offset))
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
@@ -295,6 +401,13 @@ mod tests {
|
|||||||
use rand::{Rng, SeedableRng};
|
use rand::{Rng, SeedableRng};
|
||||||
|
|
||||||
async fn round_trip_test<const BUFFERED: bool>(blobs: &[Vec<u8>]) -> Result<(), Error> {
|
async fn round_trip_test<const BUFFERED: bool>(blobs: &[Vec<u8>]) -> Result<(), Error> {
|
||||||
|
round_trip_test_compressed::<BUFFERED>(blobs, false).await
|
||||||
|
}
|
||||||
|
|
||||||
|
async fn round_trip_test_compressed<const BUFFERED: bool>(
|
||||||
|
blobs: &[Vec<u8>],
|
||||||
|
compression: bool,
|
||||||
|
) -> Result<(), Error> {
|
||||||
let temp_dir = camino_tempfile::tempdir()?;
|
let temp_dir = camino_tempfile::tempdir()?;
|
||||||
let pathbuf = temp_dir.path().join("file");
|
let pathbuf = temp_dir.path().join("file");
|
||||||
let ctx = RequestContext::new(TaskKind::UnitTest, DownloadBehavior::Error);
|
let ctx = RequestContext::new(TaskKind::UnitTest, DownloadBehavior::Error);
|
||||||
@@ -305,7 +418,16 @@ mod tests {
|
|||||||
let file = VirtualFile::create(pathbuf.as_path(), &ctx).await?;
|
let file = VirtualFile::create(pathbuf.as_path(), &ctx).await?;
|
||||||
let mut wtr = BlobWriter::<BUFFERED>::new(file, 0);
|
let mut wtr = BlobWriter::<BUFFERED>::new(file, 0);
|
||||||
for blob in blobs.iter() {
|
for blob in blobs.iter() {
|
||||||
let (_, res) = wtr.write_blob(blob.clone(), &ctx).await;
|
let (_, res) = if compression {
|
||||||
|
wtr.write_blob_maybe_compressed(
|
||||||
|
blob.clone(),
|
||||||
|
&ctx,
|
||||||
|
ImageCompressionAlgorithm::Zstd { level: Some(1) },
|
||||||
|
)
|
||||||
|
.await
|
||||||
|
} else {
|
||||||
|
wtr.write_blob(blob.clone(), &ctx).await
|
||||||
|
};
|
||||||
let offs = res?;
|
let offs = res?;
|
||||||
offsets.push(offs);
|
offsets.push(offs);
|
||||||
}
|
}
|
||||||
@@ -319,7 +441,7 @@ mod tests {
|
|||||||
|
|
||||||
let file = VirtualFile::open(pathbuf.as_path(), &ctx).await?;
|
let file = VirtualFile::open(pathbuf.as_path(), &ctx).await?;
|
||||||
let rdr = BlockReaderRef::VirtualFile(&file);
|
let rdr = BlockReaderRef::VirtualFile(&file);
|
||||||
let rdr = BlockCursor::new(rdr);
|
let rdr = BlockCursor::new_with_compression(rdr, compression);
|
||||||
for (idx, (blob, offset)) in blobs.iter().zip(offsets.iter()).enumerate() {
|
for (idx, (blob, offset)) in blobs.iter().zip(offsets.iter()).enumerate() {
|
||||||
let blob_read = rdr.read_blob(*offset, &ctx).await?;
|
let blob_read = rdr.read_blob(*offset, &ctx).await?;
|
||||||
assert_eq!(
|
assert_eq!(
|
||||||
@@ -353,6 +475,8 @@ mod tests {
|
|||||||
];
|
];
|
||||||
round_trip_test::<false>(blobs).await?;
|
round_trip_test::<false>(blobs).await?;
|
||||||
round_trip_test::<true>(blobs).await?;
|
round_trip_test::<true>(blobs).await?;
|
||||||
|
round_trip_test_compressed::<false>(blobs, true).await?;
|
||||||
|
round_trip_test_compressed::<true>(blobs, true).await?;
|
||||||
Ok(())
|
Ok(())
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -361,10 +485,15 @@ mod tests {
|
|||||||
let blobs = &[
|
let blobs = &[
|
||||||
b"test".to_vec(),
|
b"test".to_vec(),
|
||||||
random_array(10 * PAGE_SZ),
|
random_array(10 * PAGE_SZ),
|
||||||
|
b"hello".to_vec(),
|
||||||
|
random_array(66 * PAGE_SZ),
|
||||||
|
vec![0xf3; 24 * PAGE_SZ],
|
||||||
b"foobar".to_vec(),
|
b"foobar".to_vec(),
|
||||||
];
|
];
|
||||||
round_trip_test::<false>(blobs).await?;
|
round_trip_test::<false>(blobs).await?;
|
||||||
round_trip_test::<true>(blobs).await?;
|
round_trip_test::<true>(blobs).await?;
|
||||||
|
round_trip_test_compressed::<false>(blobs, true).await?;
|
||||||
|
round_trip_test_compressed::<true>(blobs, true).await?;
|
||||||
Ok(())
|
Ok(())
|
||||||
}
|
}
|
||||||
|
|
||||||
|
|||||||
@@ -37,6 +37,7 @@ where
|
|||||||
pub enum BlockLease<'a> {
|
pub enum BlockLease<'a> {
|
||||||
PageReadGuard(PageReadGuard<'static>),
|
PageReadGuard(PageReadGuard<'static>),
|
||||||
EphemeralFileMutableTail(&'a [u8; PAGE_SZ]),
|
EphemeralFileMutableTail(&'a [u8; PAGE_SZ]),
|
||||||
|
Slice(&'a [u8; PAGE_SZ]),
|
||||||
#[cfg(test)]
|
#[cfg(test)]
|
||||||
Arc(std::sync::Arc<[u8; PAGE_SZ]>),
|
Arc(std::sync::Arc<[u8; PAGE_SZ]>),
|
||||||
#[cfg(test)]
|
#[cfg(test)]
|
||||||
@@ -63,6 +64,7 @@ impl<'a> Deref for BlockLease<'a> {
|
|||||||
match self {
|
match self {
|
||||||
BlockLease::PageReadGuard(v) => v.deref(),
|
BlockLease::PageReadGuard(v) => v.deref(),
|
||||||
BlockLease::EphemeralFileMutableTail(v) => v,
|
BlockLease::EphemeralFileMutableTail(v) => v,
|
||||||
|
BlockLease::Slice(v) => v,
|
||||||
#[cfg(test)]
|
#[cfg(test)]
|
||||||
BlockLease::Arc(v) => v.deref(),
|
BlockLease::Arc(v) => v.deref(),
|
||||||
#[cfg(test)]
|
#[cfg(test)]
|
||||||
@@ -81,6 +83,7 @@ pub(crate) enum BlockReaderRef<'a> {
|
|||||||
FileBlockReader(&'a FileBlockReader<'a>),
|
FileBlockReader(&'a FileBlockReader<'a>),
|
||||||
EphemeralFile(&'a EphemeralFile),
|
EphemeralFile(&'a EphemeralFile),
|
||||||
Adapter(Adapter<&'a DeltaLayerInner>),
|
Adapter(Adapter<&'a DeltaLayerInner>),
|
||||||
|
Slice(&'a [u8]),
|
||||||
#[cfg(test)]
|
#[cfg(test)]
|
||||||
TestDisk(&'a super::disk_btree::tests::TestDisk),
|
TestDisk(&'a super::disk_btree::tests::TestDisk),
|
||||||
#[cfg(test)]
|
#[cfg(test)]
|
||||||
@@ -99,6 +102,7 @@ impl<'a> BlockReaderRef<'a> {
|
|||||||
FileBlockReader(r) => r.read_blk(blknum, ctx).await,
|
FileBlockReader(r) => r.read_blk(blknum, ctx).await,
|
||||||
EphemeralFile(r) => r.read_blk(blknum, ctx).await,
|
EphemeralFile(r) => r.read_blk(blknum, ctx).await,
|
||||||
Adapter(r) => r.read_blk(blknum, ctx).await,
|
Adapter(r) => r.read_blk(blknum, ctx).await,
|
||||||
|
Slice(s) => Self::read_blk_slice(s, blknum),
|
||||||
#[cfg(test)]
|
#[cfg(test)]
|
||||||
TestDisk(r) => r.read_blk(blknum),
|
TestDisk(r) => r.read_blk(blknum),
|
||||||
#[cfg(test)]
|
#[cfg(test)]
|
||||||
@@ -107,6 +111,24 @@ impl<'a> BlockReaderRef<'a> {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
impl<'a> BlockReaderRef<'a> {
|
||||||
|
fn read_blk_slice(slice: &[u8], blknum: u32) -> std::io::Result<BlockLease> {
|
||||||
|
let start = (blknum as usize).checked_mul(PAGE_SZ).unwrap();
|
||||||
|
let end = start.checked_add(PAGE_SZ).unwrap();
|
||||||
|
if end > slice.len() {
|
||||||
|
return Err(std::io::Error::new(
|
||||||
|
std::io::ErrorKind::UnexpectedEof,
|
||||||
|
format!("slice too short, len={} end={}", slice.len(), end),
|
||||||
|
));
|
||||||
|
}
|
||||||
|
let slice = &slice[start..end];
|
||||||
|
let page_sized: &[u8; PAGE_SZ] = slice
|
||||||
|
.try_into()
|
||||||
|
.expect("we add PAGE_SZ to start, so the slice must have PAGE_SZ");
|
||||||
|
Ok(BlockLease::Slice(page_sized))
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
///
|
///
|
||||||
/// A "cursor" for efficiently reading multiple pages from a BlockReader
|
/// A "cursor" for efficiently reading multiple pages from a BlockReader
|
||||||
///
|
///
|
||||||
@@ -127,16 +149,24 @@ impl<'a> BlockReaderRef<'a> {
|
|||||||
/// ```
|
/// ```
|
||||||
///
|
///
|
||||||
pub struct BlockCursor<'a> {
|
pub struct BlockCursor<'a> {
|
||||||
|
pub(super) read_compressed: bool,
|
||||||
reader: BlockReaderRef<'a>,
|
reader: BlockReaderRef<'a>,
|
||||||
}
|
}
|
||||||
|
|
||||||
impl<'a> BlockCursor<'a> {
|
impl<'a> BlockCursor<'a> {
|
||||||
pub(crate) fn new(reader: BlockReaderRef<'a>) -> Self {
|
pub(crate) fn new(reader: BlockReaderRef<'a>) -> Self {
|
||||||
BlockCursor { reader }
|
Self::new_with_compression(reader, false)
|
||||||
|
}
|
||||||
|
pub(crate) fn new_with_compression(reader: BlockReaderRef<'a>, read_compressed: bool) -> Self {
|
||||||
|
BlockCursor {
|
||||||
|
read_compressed,
|
||||||
|
reader,
|
||||||
|
}
|
||||||
}
|
}
|
||||||
// Needed by cli
|
// Needed by cli
|
||||||
pub fn new_fileblockreader(reader: &'a FileBlockReader) -> Self {
|
pub fn new_fileblockreader(reader: &'a FileBlockReader) -> Self {
|
||||||
BlockCursor {
|
BlockCursor {
|
||||||
|
read_compressed: false,
|
||||||
reader: BlockReaderRef::FileBlockReader(reader),
|
reader: BlockReaderRef::FileBlockReader(reader),
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
@@ -166,11 +196,25 @@ pub struct FileBlockReader<'a> {
|
|||||||
|
|
||||||
/// Unique ID of this file, used as key in the page cache.
|
/// Unique ID of this file, used as key in the page cache.
|
||||||
file_id: page_cache::FileId,
|
file_id: page_cache::FileId,
|
||||||
|
|
||||||
|
compressed_reads: bool,
|
||||||
}
|
}
|
||||||
|
|
||||||
impl<'a> FileBlockReader<'a> {
|
impl<'a> FileBlockReader<'a> {
|
||||||
pub fn new(file: &'a VirtualFile, file_id: FileId) -> Self {
|
pub fn new(file: &'a VirtualFile, file_id: FileId) -> Self {
|
||||||
FileBlockReader { file_id, file }
|
Self::new_with_compression(file, file_id, false)
|
||||||
|
}
|
||||||
|
|
||||||
|
pub fn new_with_compression(
|
||||||
|
file: &'a VirtualFile,
|
||||||
|
file_id: FileId,
|
||||||
|
compressed_reads: bool,
|
||||||
|
) -> Self {
|
||||||
|
FileBlockReader {
|
||||||
|
file_id,
|
||||||
|
file,
|
||||||
|
compressed_reads,
|
||||||
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Read a page from the underlying file into given buffer.
|
/// Read a page from the underlying file into given buffer.
|
||||||
@@ -217,7 +261,10 @@ impl<'a> FileBlockReader<'a> {
|
|||||||
|
|
||||||
impl BlockReader for FileBlockReader<'_> {
|
impl BlockReader for FileBlockReader<'_> {
|
||||||
fn block_cursor(&self) -> BlockCursor<'_> {
|
fn block_cursor(&self) -> BlockCursor<'_> {
|
||||||
BlockCursor::new(BlockReaderRef::FileBlockReader(self))
|
BlockCursor::new_with_compression(
|
||||||
|
BlockReaderRef::FileBlockReader(self),
|
||||||
|
self.compressed_reads,
|
||||||
|
)
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
|||||||
@@ -21,6 +21,7 @@ pub struct EphemeralFile {
|
|||||||
}
|
}
|
||||||
|
|
||||||
mod page_caching;
|
mod page_caching;
|
||||||
|
pub(crate) use page_caching::PrewarmOnWrite as PrewarmPageCacheOnWrite;
|
||||||
mod zero_padded_read_write;
|
mod zero_padded_read_write;
|
||||||
|
|
||||||
impl EphemeralFile {
|
impl EphemeralFile {
|
||||||
@@ -53,7 +54,7 @@ impl EphemeralFile {
|
|||||||
Ok(EphemeralFile {
|
Ok(EphemeralFile {
|
||||||
_tenant_shard_id: tenant_shard_id,
|
_tenant_shard_id: tenant_shard_id,
|
||||||
_timeline_id: timeline_id,
|
_timeline_id: timeline_id,
|
||||||
rw: page_caching::RW::new(file),
|
rw: page_caching::RW::new(file, conf.l0_flush.prewarm_on_write()),
|
||||||
})
|
})
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -65,6 +66,11 @@ impl EphemeralFile {
|
|||||||
self.rw.page_cache_file_id()
|
self.rw.page_cache_file_id()
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// See [`self::page_caching::RW::load_to_vec`].
|
||||||
|
pub(crate) async fn load_to_vec(&self, ctx: &RequestContext) -> Result<Vec<u8>, io::Error> {
|
||||||
|
self.rw.load_to_vec(ctx).await
|
||||||
|
}
|
||||||
|
|
||||||
pub(crate) async fn read_blk(
|
pub(crate) async fn read_blk(
|
||||||
&self,
|
&self,
|
||||||
blknum: u32,
|
blknum: u32,
|
||||||
|
|||||||
@@ -8,6 +8,7 @@ use crate::virtual_file::VirtualFile;
|
|||||||
|
|
||||||
use once_cell::sync::Lazy;
|
use once_cell::sync::Lazy;
|
||||||
use std::io::{self, ErrorKind};
|
use std::io::{self, ErrorKind};
|
||||||
|
use std::ops::{Deref, Range};
|
||||||
use tokio_epoll_uring::BoundedBuf;
|
use tokio_epoll_uring::BoundedBuf;
|
||||||
use tracing::*;
|
use tracing::*;
|
||||||
|
|
||||||
@@ -19,14 +20,23 @@ pub struct RW {
|
|||||||
rw: super::zero_padded_read_write::RW<PreWarmingWriter>,
|
rw: super::zero_padded_read_write::RW<PreWarmingWriter>,
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// When we flush a block to the underlying [`crate::virtual_file::VirtualFile`],
|
||||||
|
/// should we pre-warm the [`crate::page_cache`] with the contents?
|
||||||
|
#[derive(Clone, Copy)]
|
||||||
|
pub enum PrewarmOnWrite {
|
||||||
|
Yes,
|
||||||
|
No,
|
||||||
|
}
|
||||||
|
|
||||||
impl RW {
|
impl RW {
|
||||||
pub fn new(file: VirtualFile) -> Self {
|
pub fn new(file: VirtualFile, prewarm_on_write: PrewarmOnWrite) -> Self {
|
||||||
let page_cache_file_id = page_cache::next_file_id();
|
let page_cache_file_id = page_cache::next_file_id();
|
||||||
Self {
|
Self {
|
||||||
page_cache_file_id,
|
page_cache_file_id,
|
||||||
rw: super::zero_padded_read_write::RW::new(PreWarmingWriter::new(
|
rw: super::zero_padded_read_write::RW::new(PreWarmingWriter::new(
|
||||||
page_cache_file_id,
|
page_cache_file_id,
|
||||||
file,
|
file,
|
||||||
|
prewarm_on_write,
|
||||||
)),
|
)),
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
@@ -49,6 +59,43 @@ impl RW {
|
|||||||
self.rw.bytes_written()
|
self.rw.bytes_written()
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// Load all blocks that can be read via [`Self::read_blk`] into a contiguous memory buffer.
|
||||||
|
///
|
||||||
|
/// This includes the blocks that aren't yet flushed to disk by the internal buffered writer.
|
||||||
|
/// The last block is zero-padded to [`PAGE_SZ`], so, the returned buffer is always a multiple of [`PAGE_SZ`].
|
||||||
|
pub(super) async fn load_to_vec(&self, ctx: &RequestContext) -> Result<Vec<u8>, io::Error> {
|
||||||
|
// round up to the next PAGE_SZ multiple, required by blob_io
|
||||||
|
let size = {
|
||||||
|
let s = usize::try_from(self.bytes_written()).unwrap();
|
||||||
|
if s % PAGE_SZ == 0 {
|
||||||
|
s
|
||||||
|
} else {
|
||||||
|
s.checked_add(PAGE_SZ - (s % PAGE_SZ)).unwrap()
|
||||||
|
}
|
||||||
|
};
|
||||||
|
let vec = Vec::with_capacity(size);
|
||||||
|
|
||||||
|
// read from disk what we've already flushed
|
||||||
|
let writer = self.rw.as_writer();
|
||||||
|
let flushed_range = writer.written_range();
|
||||||
|
let mut vec = writer
|
||||||
|
.file
|
||||||
|
.read_exact_at(
|
||||||
|
vec.slice(0..(flushed_range.end - flushed_range.start)),
|
||||||
|
u64::try_from(flushed_range.start).unwrap(),
|
||||||
|
ctx,
|
||||||
|
)
|
||||||
|
.await?
|
||||||
|
.into_inner();
|
||||||
|
|
||||||
|
// copy from in-memory buffer what we haven't flushed yet but would return when accessed via read_blk
|
||||||
|
let buffered = self.rw.get_tail_zero_padded();
|
||||||
|
vec.extend_from_slice(buffered);
|
||||||
|
assert_eq!(vec.len(), size);
|
||||||
|
assert_eq!(vec.len() % PAGE_SZ, 0);
|
||||||
|
Ok(vec)
|
||||||
|
}
|
||||||
|
|
||||||
pub(crate) async fn read_blk(
|
pub(crate) async fn read_blk(
|
||||||
&self,
|
&self,
|
||||||
blknum: u32,
|
blknum: u32,
|
||||||
@@ -116,19 +163,40 @@ impl Drop for RW {
|
|||||||
}
|
}
|
||||||
|
|
||||||
struct PreWarmingWriter {
|
struct PreWarmingWriter {
|
||||||
|
prewarm_on_write: PrewarmOnWrite,
|
||||||
nwritten_blocks: u32,
|
nwritten_blocks: u32,
|
||||||
page_cache_file_id: page_cache::FileId,
|
page_cache_file_id: page_cache::FileId,
|
||||||
file: VirtualFile,
|
file: VirtualFile,
|
||||||
}
|
}
|
||||||
|
|
||||||
impl PreWarmingWriter {
|
impl PreWarmingWriter {
|
||||||
fn new(page_cache_file_id: page_cache::FileId, file: VirtualFile) -> Self {
|
fn new(
|
||||||
|
page_cache_file_id: page_cache::FileId,
|
||||||
|
file: VirtualFile,
|
||||||
|
prewarm_on_write: PrewarmOnWrite,
|
||||||
|
) -> Self {
|
||||||
Self {
|
Self {
|
||||||
|
prewarm_on_write,
|
||||||
nwritten_blocks: 0,
|
nwritten_blocks: 0,
|
||||||
page_cache_file_id,
|
page_cache_file_id,
|
||||||
file,
|
file,
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// Return the byte range within `file` that has been written though `write_all`.
|
||||||
|
///
|
||||||
|
/// The returned range would be invalidated by another `write_all`. To prevent that, we capture `&_`.
|
||||||
|
fn written_range(&self) -> (impl Deref<Target = Range<usize>> + '_) {
|
||||||
|
let nwritten_blocks = usize::try_from(self.nwritten_blocks).unwrap();
|
||||||
|
struct Wrapper(Range<usize>);
|
||||||
|
impl Deref for Wrapper {
|
||||||
|
type Target = Range<usize>;
|
||||||
|
fn deref(&self) -> &Range<usize> {
|
||||||
|
&self.0
|
||||||
|
}
|
||||||
|
}
|
||||||
|
Wrapper(0..nwritten_blocks * PAGE_SZ)
|
||||||
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
impl crate::virtual_file::owned_buffers_io::write::OwnedAsyncWriter for PreWarmingWriter {
|
impl crate::virtual_file::owned_buffers_io::write::OwnedAsyncWriter for PreWarmingWriter {
|
||||||
@@ -178,45 +246,51 @@ impl crate::virtual_file::owned_buffers_io::write::OwnedAsyncWriter for PreWarmi
|
|||||||
assert_eq!(&check_bounds_stuff_works, &*buf);
|
assert_eq!(&check_bounds_stuff_works, &*buf);
|
||||||
}
|
}
|
||||||
|
|
||||||
// Pre-warm page cache with the contents.
|
|
||||||
// At least in isolated bulk ingest benchmarks (test_bulk_insert.py), the pre-warming
|
|
||||||
// benefits the code that writes InMemoryLayer=>L0 layers.
|
|
||||||
let nblocks = buflen / PAGE_SZ;
|
let nblocks = buflen / PAGE_SZ;
|
||||||
let nblocks32 = u32::try_from(nblocks).unwrap();
|
let nblocks32 = u32::try_from(nblocks).unwrap();
|
||||||
let cache = page_cache::get();
|
|
||||||
static CTX: Lazy<RequestContext> = Lazy::new(|| {
|
if matches!(self.prewarm_on_write, PrewarmOnWrite::Yes) {
|
||||||
RequestContext::new(
|
// Pre-warm page cache with the contents.
|
||||||
crate::task_mgr::TaskKind::EphemeralFilePreWarmPageCache,
|
// At least in isolated bulk ingest benchmarks (test_bulk_insert.py), the pre-warming
|
||||||
crate::context::DownloadBehavior::Error,
|
// benefits the code that writes InMemoryLayer=>L0 layers.
|
||||||
)
|
|
||||||
});
|
let cache = page_cache::get();
|
||||||
for blknum_in_buffer in 0..nblocks {
|
static CTX: Lazy<RequestContext> = Lazy::new(|| {
|
||||||
let blk_in_buffer = &buf[blknum_in_buffer * PAGE_SZ..(blknum_in_buffer + 1) * PAGE_SZ];
|
RequestContext::new(
|
||||||
let blknum = self
|
crate::task_mgr::TaskKind::EphemeralFilePreWarmPageCache,
|
||||||
.nwritten_blocks
|
crate::context::DownloadBehavior::Error,
|
||||||
.checked_add(blknum_in_buffer as u32)
|
)
|
||||||
.unwrap();
|
});
|
||||||
match cache
|
for blknum_in_buffer in 0..nblocks {
|
||||||
.read_immutable_buf(self.page_cache_file_id, blknum, &CTX)
|
let blk_in_buffer =
|
||||||
.await
|
&buf[blknum_in_buffer * PAGE_SZ..(blknum_in_buffer + 1) * PAGE_SZ];
|
||||||
{
|
let blknum = self
|
||||||
Err(e) => {
|
.nwritten_blocks
|
||||||
error!("ephemeral_file write_blob failed to get immutable buf to pre-warm page cache: {e:?}");
|
.checked_add(blknum_in_buffer as u32)
|
||||||
// fail gracefully, it's not the end of the world if we can't pre-warm the cache here
|
.unwrap();
|
||||||
}
|
match cache
|
||||||
Ok(v) => match v {
|
.read_immutable_buf(self.page_cache_file_id, blknum, &CTX)
|
||||||
page_cache::ReadBufResult::Found(_guard) => {
|
.await
|
||||||
// This function takes &mut self, so, it shouldn't be possible to reach this point.
|
{
|
||||||
unreachable!("we just wrote block {blknum} to the VirtualFile, which is owned by Self, \
|
Err(e) => {
|
||||||
|
error!("ephemeral_file write_blob failed to get immutable buf to pre-warm page cache: {e:?}");
|
||||||
|
// fail gracefully, it's not the end of the world if we can't pre-warm the cache here
|
||||||
|
}
|
||||||
|
Ok(v) => match v {
|
||||||
|
page_cache::ReadBufResult::Found(_guard) => {
|
||||||
|
// This function takes &mut self, so, it shouldn't be possible to reach this point.
|
||||||
|
unreachable!("we just wrote block {blknum} to the VirtualFile, which is owned by Self, \
|
||||||
and this function takes &mut self, so, no concurrent read_blk is possible");
|
and this function takes &mut self, so, no concurrent read_blk is possible");
|
||||||
}
|
}
|
||||||
page_cache::ReadBufResult::NotFound(mut write_guard) => {
|
page_cache::ReadBufResult::NotFound(mut write_guard) => {
|
||||||
write_guard.copy_from_slice(blk_in_buffer);
|
write_guard.copy_from_slice(blk_in_buffer);
|
||||||
let _ = write_guard.mark_valid();
|
let _ = write_guard.mark_valid();
|
||||||
}
|
}
|
||||||
},
|
},
|
||||||
|
}
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
self.nwritten_blocks = self.nwritten_blocks.checked_add(nblocks32).unwrap();
|
self.nwritten_blocks = self.nwritten_blocks.checked_add(nblocks32).unwrap();
|
||||||
Ok((buflen, buf.into_inner()))
|
Ok((buflen, buf.into_inner()))
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -75,6 +75,21 @@ where
|
|||||||
flushed_offset + u64::try_from(buffer.pending()).unwrap()
|
flushed_offset + u64::try_from(buffer.pending()).unwrap()
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// Get a slice of all blocks that [`Self::read_blk`] would return as [`ReadResult::ServedFromZeroPaddedMutableTail`].
|
||||||
|
pub fn get_tail_zero_padded(&self) -> &[u8] {
|
||||||
|
let buffer: &zero_padded::Buffer<TAIL_SZ> = self.buffered_writer.inspect_buffer();
|
||||||
|
let buffer_written_up_to = buffer.pending();
|
||||||
|
// pad to next page boundary
|
||||||
|
let read_up_to = if buffer_written_up_to % PAGE_SZ == 0 {
|
||||||
|
buffer_written_up_to
|
||||||
|
} else {
|
||||||
|
buffer_written_up_to
|
||||||
|
.checked_add(PAGE_SZ - (buffer_written_up_to % PAGE_SZ))
|
||||||
|
.unwrap()
|
||||||
|
};
|
||||||
|
&buffer.as_zero_padded_slice()[0..read_up_to]
|
||||||
|
}
|
||||||
|
|
||||||
pub(crate) async fn read_blk(&self, blknum: u32) -> Result<ReadResult<'_, W>, std::io::Error> {
|
pub(crate) async fn read_blk(&self, blknum: u32) -> Result<ReadResult<'_, W>, std::io::Error> {
|
||||||
let flushed_offset = self.buffered_writer.as_inner().bytes_written();
|
let flushed_offset = self.buffered_writer.as_inner().bytes_written();
|
||||||
let buffer: &zero_padded::Buffer<TAIL_SZ> = self.buffered_writer.inspect_buffer();
|
let buffer: &zero_padded::Buffer<TAIL_SZ> = self.buffered_writer.inspect_buffer();
|
||||||
|
|||||||
+89
-102
@@ -43,7 +43,8 @@ use crate::tenant::config::{
|
|||||||
use crate::tenant::span::debug_assert_current_span_has_tenant_id;
|
use crate::tenant::span::debug_assert_current_span_has_tenant_id;
|
||||||
use crate::tenant::storage_layer::inmemory_layer;
|
use crate::tenant::storage_layer::inmemory_layer;
|
||||||
use crate::tenant::timeline::ShutdownMode;
|
use crate::tenant::timeline::ShutdownMode;
|
||||||
use crate::tenant::{AttachedTenantConf, GcError, SpawnMode, Tenant, TenantState};
|
use crate::tenant::{AttachedTenantConf, GcError, LoadConfigError, SpawnMode, Tenant, TenantState};
|
||||||
|
use crate::virtual_file::MaybeFatalIo;
|
||||||
use crate::{InitializationOrder, TEMP_FILE_SUFFIX};
|
use crate::{InitializationOrder, TEMP_FILE_SUFFIX};
|
||||||
|
|
||||||
use utils::crashsafe::path_with_suffix_extension;
|
use utils::crashsafe::path_with_suffix_extension;
|
||||||
@@ -272,7 +273,7 @@ pub struct TenantManager {
|
|||||||
}
|
}
|
||||||
|
|
||||||
fn emergency_generations(
|
fn emergency_generations(
|
||||||
tenant_confs: &HashMap<TenantShardId, anyhow::Result<LocationConf>>,
|
tenant_confs: &HashMap<TenantShardId, Result<LocationConf, LoadConfigError>>,
|
||||||
) -> HashMap<TenantShardId, TenantStartupMode> {
|
) -> HashMap<TenantShardId, TenantStartupMode> {
|
||||||
tenant_confs
|
tenant_confs
|
||||||
.iter()
|
.iter()
|
||||||
@@ -296,7 +297,7 @@ fn emergency_generations(
|
|||||||
|
|
||||||
async fn init_load_generations(
|
async fn init_load_generations(
|
||||||
conf: &'static PageServerConf,
|
conf: &'static PageServerConf,
|
||||||
tenant_confs: &HashMap<TenantShardId, anyhow::Result<LocationConf>>,
|
tenant_confs: &HashMap<TenantShardId, Result<LocationConf, LoadConfigError>>,
|
||||||
resources: &TenantSharedResources,
|
resources: &TenantSharedResources,
|
||||||
cancel: &CancellationToken,
|
cancel: &CancellationToken,
|
||||||
) -> anyhow::Result<Option<HashMap<TenantShardId, TenantStartupMode>>> {
|
) -> anyhow::Result<Option<HashMap<TenantShardId, TenantStartupMode>>> {
|
||||||
@@ -346,56 +347,32 @@ async fn init_load_generations(
|
|||||||
/// Given a directory discovered in the pageserver's tenants/ directory, attempt
|
/// Given a directory discovered in the pageserver's tenants/ directory, attempt
|
||||||
/// to load a tenant config from it.
|
/// to load a tenant config from it.
|
||||||
///
|
///
|
||||||
/// If file is missing, return Ok(None)
|
/// If we cleaned up something expected (like an empty dir or a temp dir), return None.
|
||||||
fn load_tenant_config(
|
fn load_tenant_config(
|
||||||
conf: &'static PageServerConf,
|
conf: &'static PageServerConf,
|
||||||
|
tenant_shard_id: TenantShardId,
|
||||||
dentry: Utf8DirEntry,
|
dentry: Utf8DirEntry,
|
||||||
) -> anyhow::Result<Option<(TenantShardId, anyhow::Result<LocationConf>)>> {
|
) -> Option<Result<LocationConf, LoadConfigError>> {
|
||||||
let tenant_dir_path = dentry.path().to_path_buf();
|
let tenant_dir_path = dentry.path().to_path_buf();
|
||||||
if crate::is_temporary(&tenant_dir_path) {
|
if crate::is_temporary(&tenant_dir_path) {
|
||||||
info!("Found temporary tenant directory, removing: {tenant_dir_path}");
|
info!("Found temporary tenant directory, removing: {tenant_dir_path}");
|
||||||
// No need to use safe_remove_tenant_dir_all because this is already
|
// No need to use safe_remove_tenant_dir_all because this is already
|
||||||
// a temporary path
|
// a temporary path
|
||||||
if let Err(e) = std::fs::remove_dir_all(&tenant_dir_path) {
|
std::fs::remove_dir_all(&tenant_dir_path).fatal_err("delete temporary tenant dir");
|
||||||
error!(
|
return None;
|
||||||
"Failed to remove temporary directory '{}': {:?}",
|
|
||||||
tenant_dir_path, e
|
|
||||||
);
|
|
||||||
}
|
|
||||||
return Ok(None);
|
|
||||||
}
|
}
|
||||||
|
|
||||||
// This case happens if we crash during attachment before writing a config into the dir
|
// This case happens if we crash during attachment before writing a config into the dir
|
||||||
let is_empty = tenant_dir_path
|
let is_empty = tenant_dir_path
|
||||||
.is_empty_dir()
|
.is_empty_dir()
|
||||||
.with_context(|| format!("Failed to check whether {tenant_dir_path:?} is an empty dir"))?;
|
.fatal_err("Checking for empty tenant dir");
|
||||||
if is_empty {
|
if is_empty {
|
||||||
info!("removing empty tenant directory {tenant_dir_path:?}");
|
info!("removing empty tenant directory {tenant_dir_path:?}");
|
||||||
if let Err(e) = std::fs::remove_dir(&tenant_dir_path) {
|
std::fs::remove_dir(&tenant_dir_path).fatal_err("delete empty tenant dir");
|
||||||
error!(
|
return None;
|
||||||
"Failed to remove empty tenant directory '{}': {e:#}",
|
|
||||||
tenant_dir_path
|
|
||||||
)
|
|
||||||
}
|
|
||||||
return Ok(None);
|
|
||||||
}
|
}
|
||||||
|
|
||||||
let tenant_shard_id = match tenant_dir_path
|
Some(Tenant::load_tenant_config(conf, &tenant_shard_id))
|
||||||
.file_name()
|
|
||||||
.unwrap_or_default()
|
|
||||||
.parse::<TenantShardId>()
|
|
||||||
{
|
|
||||||
Ok(id) => id,
|
|
||||||
Err(_) => {
|
|
||||||
warn!("Invalid tenant path (garbage in our repo directory?): {tenant_dir_path}",);
|
|
||||||
return Ok(None);
|
|
||||||
}
|
|
||||||
};
|
|
||||||
|
|
||||||
Ok(Some((
|
|
||||||
tenant_shard_id,
|
|
||||||
Tenant::load_tenant_config(conf, &tenant_shard_id),
|
|
||||||
)))
|
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Initial stage of load: walk the local tenants directory, clean up any temp files,
|
/// Initial stage of load: walk the local tenants directory, clean up any temp files,
|
||||||
@@ -405,32 +382,51 @@ fn load_tenant_config(
|
|||||||
/// seconds even on reasonably fast drives.
|
/// seconds even on reasonably fast drives.
|
||||||
async fn init_load_tenant_configs(
|
async fn init_load_tenant_configs(
|
||||||
conf: &'static PageServerConf,
|
conf: &'static PageServerConf,
|
||||||
) -> anyhow::Result<HashMap<TenantShardId, anyhow::Result<LocationConf>>> {
|
) -> HashMap<TenantShardId, Result<LocationConf, LoadConfigError>> {
|
||||||
let tenants_dir = conf.tenants_path();
|
let tenants_dir = conf.tenants_path();
|
||||||
|
|
||||||
let dentries = tokio::task::spawn_blocking(move || -> anyhow::Result<Vec<Utf8DirEntry>> {
|
let dentries = tokio::task::spawn_blocking(move || -> Vec<Utf8DirEntry> {
|
||||||
let dir_entries = tenants_dir
|
let context = format!("read tenants dir {tenants_dir}");
|
||||||
.read_dir_utf8()
|
let dir_entries = tenants_dir.read_dir_utf8().fatal_err(&context);
|
||||||
.with_context(|| format!("Failed to list tenants dir {tenants_dir:?}"))?;
|
|
||||||
|
|
||||||
Ok(dir_entries.collect::<Result<Vec<_>, std::io::Error>>()?)
|
dir_entries
|
||||||
|
.collect::<Result<Vec<_>, std::io::Error>>()
|
||||||
|
.fatal_err(&context)
|
||||||
})
|
})
|
||||||
.await??;
|
.await
|
||||||
|
.expect("Config load task panicked");
|
||||||
|
|
||||||
let mut configs = HashMap::new();
|
let mut configs = HashMap::new();
|
||||||
|
|
||||||
let mut join_set = JoinSet::new();
|
let mut join_set = JoinSet::new();
|
||||||
for dentry in dentries {
|
for dentry in dentries {
|
||||||
join_set.spawn_blocking(move || load_tenant_config(conf, dentry));
|
let tenant_shard_id = match dentry.file_name().parse::<TenantShardId>() {
|
||||||
|
Ok(id) => id,
|
||||||
|
Err(_) => {
|
||||||
|
warn!(
|
||||||
|
"Invalid tenant path (garbage in our repo directory?): '{}'",
|
||||||
|
dentry.file_name()
|
||||||
|
);
|
||||||
|
continue;
|
||||||
|
}
|
||||||
|
};
|
||||||
|
|
||||||
|
join_set.spawn_blocking(move || {
|
||||||
|
(
|
||||||
|
tenant_shard_id,
|
||||||
|
load_tenant_config(conf, tenant_shard_id, dentry),
|
||||||
|
)
|
||||||
|
});
|
||||||
}
|
}
|
||||||
|
|
||||||
while let Some(r) = join_set.join_next().await {
|
while let Some(r) = join_set.join_next().await {
|
||||||
if let Some((tenant_id, tenant_config)) = r?? {
|
let (tenant_shard_id, tenant_config) = r.expect("Panic in config load task");
|
||||||
configs.insert(tenant_id, tenant_config);
|
if let Some(tenant_config) = tenant_config {
|
||||||
|
configs.insert(tenant_shard_id, tenant_config);
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
Ok(configs)
|
configs
|
||||||
}
|
}
|
||||||
|
|
||||||
#[derive(Debug, thiserror::Error)]
|
#[derive(Debug, thiserror::Error)]
|
||||||
@@ -472,7 +468,7 @@ pub async fn init_tenant_mgr(
|
|||||||
);
|
);
|
||||||
|
|
||||||
// Scan local filesystem for attached tenants
|
// Scan local filesystem for attached tenants
|
||||||
let tenant_configs = init_load_tenant_configs(conf).await?;
|
let tenant_configs = init_load_tenant_configs(conf).await;
|
||||||
|
|
||||||
// Determine which tenants are to be secondary or attached, and in which generation
|
// Determine which tenants are to be secondary or attached, and in which generation
|
||||||
let tenant_modes = init_load_generations(conf, &tenant_configs, &resources, &cancel).await?;
|
let tenant_modes = init_load_generations(conf, &tenant_configs, &resources, &cancel).await?;
|
||||||
@@ -590,31 +586,23 @@ pub async fn init_tenant_mgr(
|
|||||||
);
|
);
|
||||||
// For those shards that have live configurations, construct `Tenant` or `SecondaryTenant` objects and start them running
|
// For those shards that have live configurations, construct `Tenant` or `SecondaryTenant` objects and start them running
|
||||||
for (tenant_shard_id, location_conf, config_write_result) in config_write_results {
|
for (tenant_shard_id, location_conf, config_write_result) in config_write_results {
|
||||||
// Errors writing configs are fatal
|
// Writing a config to local disk is foundational to startup up tenants: panic if we can't.
|
||||||
config_write_result?;
|
config_write_result.fatal_err("write tenant shard config file");
|
||||||
|
|
||||||
let tenant_dir_path = conf.tenant_path(&tenant_shard_id);
|
let tenant_dir_path = conf.tenant_path(&tenant_shard_id);
|
||||||
let shard_identity = location_conf.shard;
|
let shard_identity = location_conf.shard;
|
||||||
let slot = match location_conf.mode {
|
let slot = match location_conf.mode {
|
||||||
LocationMode::Attached(attached_conf) => {
|
LocationMode::Attached(attached_conf) => TenantSlot::Attached(tenant_spawn(
|
||||||
match tenant_spawn(
|
conf,
|
||||||
conf,
|
tenant_shard_id,
|
||||||
tenant_shard_id,
|
&tenant_dir_path,
|
||||||
&tenant_dir_path,
|
resources.clone(),
|
||||||
resources.clone(),
|
AttachedTenantConf::new(location_conf.tenant_conf, attached_conf),
|
||||||
AttachedTenantConf::new(location_conf.tenant_conf, attached_conf),
|
shard_identity,
|
||||||
shard_identity,
|
Some(init_order.clone()),
|
||||||
Some(init_order.clone()),
|
SpawnMode::Lazy,
|
||||||
SpawnMode::Lazy,
|
&ctx,
|
||||||
&ctx,
|
)),
|
||||||
) {
|
|
||||||
Ok(tenant) => TenantSlot::Attached(tenant),
|
|
||||||
Err(e) => {
|
|
||||||
error!(tenant_id=%tenant_shard_id.tenant_id, shard_id=%tenant_shard_id.shard_slug(), "Failed to start tenant: {e:#}");
|
|
||||||
continue;
|
|
||||||
}
|
|
||||||
}
|
|
||||||
}
|
|
||||||
LocationMode::Secondary(secondary_conf) => {
|
LocationMode::Secondary(secondary_conf) => {
|
||||||
info!(
|
info!(
|
||||||
tenant_id = %tenant_shard_id.tenant_id,
|
tenant_id = %tenant_shard_id.tenant_id,
|
||||||
@@ -649,8 +637,7 @@ pub async fn init_tenant_mgr(
|
|||||||
})
|
})
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Wrapper for Tenant::spawn that checks invariants before running, and inserts
|
/// Wrapper for Tenant::spawn that checks invariants before running
|
||||||
/// a broken tenant in the map if Tenant::spawn fails.
|
|
||||||
#[allow(clippy::too_many_arguments)]
|
#[allow(clippy::too_many_arguments)]
|
||||||
fn tenant_spawn(
|
fn tenant_spawn(
|
||||||
conf: &'static PageServerConf,
|
conf: &'static PageServerConf,
|
||||||
@@ -662,23 +649,18 @@ fn tenant_spawn(
|
|||||||
init_order: Option<InitializationOrder>,
|
init_order: Option<InitializationOrder>,
|
||||||
mode: SpawnMode,
|
mode: SpawnMode,
|
||||||
ctx: &RequestContext,
|
ctx: &RequestContext,
|
||||||
) -> anyhow::Result<Arc<Tenant>> {
|
) -> Arc<Tenant> {
|
||||||
anyhow::ensure!(
|
// All these conditions should have been satisfied by our caller: the tenant dir exists, is a well formed
|
||||||
tenant_path.is_dir(),
|
// path, and contains a configuration file. Assertions that do synchronous I/O are limited to debug mode
|
||||||
"Cannot load tenant from path {tenant_path:?}, it either does not exist or not a directory"
|
// to avoid impacting prod runtime performance.
|
||||||
);
|
assert!(!crate::is_temporary(tenant_path));
|
||||||
anyhow::ensure!(
|
debug_assert!(tenant_path.is_dir());
|
||||||
!crate::is_temporary(tenant_path),
|
debug_assert!(conf
|
||||||
"Cannot load tenant from temporary path {tenant_path:?}"
|
.tenant_location_config_path(&tenant_shard_id)
|
||||||
);
|
.try_exists()
|
||||||
anyhow::ensure!(
|
.unwrap());
|
||||||
!tenant_path.is_empty_dir().with_context(|| {
|
|
||||||
format!("Failed to check whether {tenant_path:?} is an empty dir")
|
|
||||||
})?,
|
|
||||||
"Cannot load tenant from empty directory {tenant_path:?}"
|
|
||||||
);
|
|
||||||
|
|
||||||
let tenant = Tenant::spawn(
|
Tenant::spawn(
|
||||||
conf,
|
conf,
|
||||||
tenant_shard_id,
|
tenant_shard_id,
|
||||||
resources,
|
resources,
|
||||||
@@ -687,9 +669,7 @@ fn tenant_spawn(
|
|||||||
init_order,
|
init_order,
|
||||||
mode,
|
mode,
|
||||||
ctx,
|
ctx,
|
||||||
);
|
)
|
||||||
|
|
||||||
Ok(tenant)
|
|
||||||
}
|
}
|
||||||
|
|
||||||
async fn shutdown_all_tenants0(tenants: &std::sync::RwLock<TenantsMap>) {
|
async fn shutdown_all_tenants0(tenants: &std::sync::RwLock<TenantsMap>) {
|
||||||
@@ -840,8 +820,9 @@ pub(crate) enum UpsertLocationError {
|
|||||||
#[error("Failed to flush: {0}")]
|
#[error("Failed to flush: {0}")]
|
||||||
Flush(anyhow::Error),
|
Flush(anyhow::Error),
|
||||||
|
|
||||||
|
/// This error variant is for unexpected situations (soft assertions) where the system is in an unexpected state.
|
||||||
#[error("Internal error: {0}")]
|
#[error("Internal error: {0}")]
|
||||||
Other(#[from] anyhow::Error),
|
InternalError(anyhow::Error),
|
||||||
}
|
}
|
||||||
|
|
||||||
impl TenantManager {
|
impl TenantManager {
|
||||||
@@ -971,7 +952,8 @@ impl TenantManager {
|
|||||||
match fast_path_taken {
|
match fast_path_taken {
|
||||||
Some(FastPathModified::Attached(tenant)) => {
|
Some(FastPathModified::Attached(tenant)) => {
|
||||||
Tenant::persist_tenant_config(self.conf, &tenant_shard_id, &new_location_config)
|
Tenant::persist_tenant_config(self.conf, &tenant_shard_id, &new_location_config)
|
||||||
.await?;
|
.await
|
||||||
|
.fatal_err("write tenant shard config");
|
||||||
|
|
||||||
// Transition to AttachedStale means we may well hold a valid generation
|
// Transition to AttachedStale means we may well hold a valid generation
|
||||||
// still, and have been requested to go stale as part of a migration. If
|
// still, and have been requested to go stale as part of a migration. If
|
||||||
@@ -1001,7 +983,8 @@ impl TenantManager {
|
|||||||
}
|
}
|
||||||
Some(FastPathModified::Secondary(_secondary_tenant)) => {
|
Some(FastPathModified::Secondary(_secondary_tenant)) => {
|
||||||
Tenant::persist_tenant_config(self.conf, &tenant_shard_id, &new_location_config)
|
Tenant::persist_tenant_config(self.conf, &tenant_shard_id, &new_location_config)
|
||||||
.await?;
|
.await
|
||||||
|
.fatal_err("write tenant shard config");
|
||||||
|
|
||||||
return Ok(None);
|
return Ok(None);
|
||||||
}
|
}
|
||||||
@@ -1067,7 +1050,7 @@ impl TenantManager {
|
|||||||
Some(TenantSlot::InProgress(_)) => {
|
Some(TenantSlot::InProgress(_)) => {
|
||||||
// This should never happen: acquire_slot should error out
|
// This should never happen: acquire_slot should error out
|
||||||
// if the contents of a slot were InProgress.
|
// if the contents of a slot were InProgress.
|
||||||
return Err(UpsertLocationError::Other(anyhow::anyhow!(
|
return Err(UpsertLocationError::InternalError(anyhow::anyhow!(
|
||||||
"Acquired an InProgress slot, this is a bug."
|
"Acquired an InProgress slot, this is a bug."
|
||||||
)));
|
)));
|
||||||
}
|
}
|
||||||
@@ -1086,12 +1069,14 @@ impl TenantManager {
|
|||||||
// Does not need to be fsync'd because local storage is just a cache.
|
// Does not need to be fsync'd because local storage is just a cache.
|
||||||
tokio::fs::create_dir_all(&timelines_path)
|
tokio::fs::create_dir_all(&timelines_path)
|
||||||
.await
|
.await
|
||||||
.with_context(|| format!("Creating {timelines_path}"))?;
|
.fatal_err("create timelines/ dir");
|
||||||
|
|
||||||
// Before activating either secondary or attached mode, persist the
|
// Before activating either secondary or attached mode, persist the
|
||||||
// configuration, so that on restart we will re-attach (or re-start
|
// configuration, so that on restart we will re-attach (or re-start
|
||||||
// secondary) on the tenant.
|
// secondary) on the tenant.
|
||||||
Tenant::persist_tenant_config(self.conf, &tenant_shard_id, &new_location_config).await?;
|
Tenant::persist_tenant_config(self.conf, &tenant_shard_id, &new_location_config)
|
||||||
|
.await
|
||||||
|
.fatal_err("write tenant shard config");
|
||||||
|
|
||||||
let new_slot = match &new_location_config.mode {
|
let new_slot = match &new_location_config.mode {
|
||||||
LocationMode::Secondary(secondary_config) => {
|
LocationMode::Secondary(secondary_config) => {
|
||||||
@@ -1110,13 +1095,15 @@ impl TenantManager {
|
|||||||
// from upserts. This enables creating generation-less tenants even though neon_local
|
// from upserts. This enables creating generation-less tenants even though neon_local
|
||||||
// always uses generations when calling the location conf API.
|
// always uses generations when calling the location conf API.
|
||||||
let attached_conf = if cfg!(feature = "testing") {
|
let attached_conf = if cfg!(feature = "testing") {
|
||||||
let mut conf = AttachedTenantConf::try_from(new_location_config)?;
|
let mut conf = AttachedTenantConf::try_from(new_location_config)
|
||||||
|
.map_err(UpsertLocationError::BadRequest)?;
|
||||||
if self.conf.control_plane_api.is_none() {
|
if self.conf.control_plane_api.is_none() {
|
||||||
conf.location.generation = Generation::none();
|
conf.location.generation = Generation::none();
|
||||||
}
|
}
|
||||||
conf
|
conf
|
||||||
} else {
|
} else {
|
||||||
AttachedTenantConf::try_from(new_location_config)?
|
AttachedTenantConf::try_from(new_location_config)
|
||||||
|
.map_err(UpsertLocationError::BadRequest)?
|
||||||
};
|
};
|
||||||
|
|
||||||
let tenant = tenant_spawn(
|
let tenant = tenant_spawn(
|
||||||
@@ -1129,7 +1116,7 @@ impl TenantManager {
|
|||||||
None,
|
None,
|
||||||
spawn_mode,
|
spawn_mode,
|
||||||
ctx,
|
ctx,
|
||||||
)?;
|
);
|
||||||
|
|
||||||
TenantSlot::Attached(tenant)
|
TenantSlot::Attached(tenant)
|
||||||
}
|
}
|
||||||
@@ -1143,7 +1130,7 @@ impl TenantManager {
|
|||||||
|
|
||||||
match slot_guard.upsert(new_slot) {
|
match slot_guard.upsert(new_slot) {
|
||||||
Err(TenantSlotUpsertError::InternalError(e)) => {
|
Err(TenantSlotUpsertError::InternalError(e)) => {
|
||||||
Err(UpsertLocationError::Other(anyhow::anyhow!(e)))
|
Err(UpsertLocationError::InternalError(anyhow::anyhow!(e)))
|
||||||
}
|
}
|
||||||
Err(TenantSlotUpsertError::MapState(e)) => Err(UpsertLocationError::Unavailable(e)),
|
Err(TenantSlotUpsertError::MapState(e)) => Err(UpsertLocationError::Unavailable(e)),
|
||||||
Err(TenantSlotUpsertError::ShuttingDown((new_slot, _completion))) => {
|
Err(TenantSlotUpsertError::ShuttingDown((new_slot, _completion))) => {
|
||||||
@@ -1250,7 +1237,7 @@ impl TenantManager {
|
|||||||
None,
|
None,
|
||||||
SpawnMode::Eager,
|
SpawnMode::Eager,
|
||||||
ctx,
|
ctx,
|
||||||
)?;
|
);
|
||||||
|
|
||||||
slot_guard.upsert(TenantSlot::Attached(tenant))?;
|
slot_guard.upsert(TenantSlot::Attached(tenant))?;
|
||||||
|
|
||||||
@@ -1984,7 +1971,7 @@ impl TenantManager {
|
|||||||
None,
|
None,
|
||||||
SpawnMode::Eager,
|
SpawnMode::Eager,
|
||||||
ctx,
|
ctx,
|
||||||
)?;
|
);
|
||||||
|
|
||||||
slot_guard.upsert(TenantSlot::Attached(tenant))?;
|
slot_guard.upsert(TenantSlot::Attached(tenant))?;
|
||||||
|
|
||||||
|
|||||||
@@ -519,7 +519,7 @@ impl RemoteTimelineClient {
|
|||||||
local_path: &Utf8Path,
|
local_path: &Utf8Path,
|
||||||
cancel: &CancellationToken,
|
cancel: &CancellationToken,
|
||||||
ctx: &RequestContext,
|
ctx: &RequestContext,
|
||||||
) -> anyhow::Result<u64> {
|
) -> Result<u64, DownloadError> {
|
||||||
let downloaded_size = {
|
let downloaded_size = {
|
||||||
let _unfinished_gauge_guard = self.metrics.call_begin(
|
let _unfinished_gauge_guard = self.metrics.call_begin(
|
||||||
&RemoteOpFileKind::Layer,
|
&RemoteOpFileKind::Layer,
|
||||||
|
|||||||
@@ -23,6 +23,8 @@ use super::{
|
|||||||
storage_layer::LayerName,
|
storage_layer::LayerName,
|
||||||
};
|
};
|
||||||
|
|
||||||
|
use crate::metrics::SECONDARY_RESIDENT_PHYSICAL_SIZE;
|
||||||
|
use metrics::UIntGauge;
|
||||||
use pageserver_api::{
|
use pageserver_api::{
|
||||||
models,
|
models,
|
||||||
shard::{ShardIdentity, TenantShardId},
|
shard::{ShardIdentity, TenantShardId},
|
||||||
@@ -99,6 +101,17 @@ pub(crate) struct SecondaryTenant {
|
|||||||
|
|
||||||
// Public state indicating overall progress of downloads relative to the last heatmap seen
|
// Public state indicating overall progress of downloads relative to the last heatmap seen
|
||||||
pub(crate) progress: std::sync::Mutex<models::SecondaryProgress>,
|
pub(crate) progress: std::sync::Mutex<models::SecondaryProgress>,
|
||||||
|
|
||||||
|
// Sum of layer sizes on local disk
|
||||||
|
pub(super) resident_size_metric: UIntGauge,
|
||||||
|
}
|
||||||
|
|
||||||
|
impl Drop for SecondaryTenant {
|
||||||
|
fn drop(&mut self) {
|
||||||
|
let tenant_id = self.tenant_shard_id.tenant_id.to_string();
|
||||||
|
let shard_id = format!("{}", self.tenant_shard_id.shard_slug());
|
||||||
|
let _ = SECONDARY_RESIDENT_PHYSICAL_SIZE.remove_label_values(&[&tenant_id, &shard_id]);
|
||||||
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
impl SecondaryTenant {
|
impl SecondaryTenant {
|
||||||
@@ -108,6 +121,12 @@ impl SecondaryTenant {
|
|||||||
tenant_conf: TenantConfOpt,
|
tenant_conf: TenantConfOpt,
|
||||||
config: &SecondaryLocationConfig,
|
config: &SecondaryLocationConfig,
|
||||||
) -> Arc<Self> {
|
) -> Arc<Self> {
|
||||||
|
let tenant_id = tenant_shard_id.tenant_id.to_string();
|
||||||
|
let shard_id = format!("{}", tenant_shard_id.shard_slug());
|
||||||
|
let resident_size_metric = SECONDARY_RESIDENT_PHYSICAL_SIZE
|
||||||
|
.get_metric_with_label_values(&[&tenant_id, &shard_id])
|
||||||
|
.unwrap();
|
||||||
|
|
||||||
Arc::new(Self {
|
Arc::new(Self {
|
||||||
tenant_shard_id,
|
tenant_shard_id,
|
||||||
// todo: shall we make this a descendent of the
|
// todo: shall we make this a descendent of the
|
||||||
@@ -123,6 +142,8 @@ impl SecondaryTenant {
|
|||||||
detail: std::sync::Mutex::new(SecondaryDetail::new(config.clone())),
|
detail: std::sync::Mutex::new(SecondaryDetail::new(config.clone())),
|
||||||
|
|
||||||
progress: std::sync::Mutex::default(),
|
progress: std::sync::Mutex::default(),
|
||||||
|
|
||||||
|
resident_size_metric,
|
||||||
})
|
})
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -211,16 +232,12 @@ impl SecondaryTenant {
|
|||||||
// have to 100% match what is on disk, because it's a best-effort warming
|
// have to 100% match what is on disk, because it's a best-effort warming
|
||||||
// of the cache.
|
// of the cache.
|
||||||
let mut detail = this.detail.lock().unwrap();
|
let mut detail = this.detail.lock().unwrap();
|
||||||
if let Some(timeline_detail) = detail.timelines.get_mut(&timeline_id) {
|
if let Some(removed) =
|
||||||
let removed = timeline_detail.on_disk_layers.remove(&name);
|
detail.evict_layer(name, &timeline_id, now, &this.resident_size_metric)
|
||||||
|
{
|
||||||
// We might race with removal of the same layer during downloads, if it was removed
|
// We might race with removal of the same layer during downloads, so finding the layer we
|
||||||
// from the heatmap. If we see that the OnDiskState is gone, then no need to
|
// were trying to remove is optional. Only issue the disk I/O to remove it if we found it.
|
||||||
// do a physical deletion or store in evicted_at.
|
removed.remove_blocking();
|
||||||
if let Some(removed) = removed {
|
|
||||||
removed.remove_blocking();
|
|
||||||
timeline_detail.evicted_at.insert(name, now);
|
|
||||||
}
|
|
||||||
}
|
}
|
||||||
})
|
})
|
||||||
.await
|
.await
|
||||||
|
|||||||
@@ -46,6 +46,7 @@ use crate::tenant::{
|
|||||||
use camino::Utf8PathBuf;
|
use camino::Utf8PathBuf;
|
||||||
use chrono::format::{DelayedFormat, StrftimeItems};
|
use chrono::format::{DelayedFormat, StrftimeItems};
|
||||||
use futures::Future;
|
use futures::Future;
|
||||||
|
use metrics::UIntGauge;
|
||||||
use pageserver_api::models::SecondaryProgress;
|
use pageserver_api::models::SecondaryProgress;
|
||||||
use pageserver_api::shard::TenantShardId;
|
use pageserver_api::shard::TenantShardId;
|
||||||
use remote_storage::{DownloadError, Etag, GenericRemoteStorage};
|
use remote_storage::{DownloadError, Etag, GenericRemoteStorage};
|
||||||
@@ -131,16 +132,66 @@ impl OnDiskState {
|
|||||||
.or_else(fs_ext::ignore_not_found)
|
.or_else(fs_ext::ignore_not_found)
|
||||||
.fatal_err("Deleting secondary layer")
|
.fatal_err("Deleting secondary layer")
|
||||||
}
|
}
|
||||||
|
|
||||||
|
pub(crate) fn file_size(&self) -> u64 {
|
||||||
|
self.metadata.file_size
|
||||||
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
#[derive(Debug, Clone, Default)]
|
#[derive(Debug, Clone, Default)]
|
||||||
pub(super) struct SecondaryDetailTimeline {
|
pub(super) struct SecondaryDetailTimeline {
|
||||||
pub(super) on_disk_layers: HashMap<LayerName, OnDiskState>,
|
on_disk_layers: HashMap<LayerName, OnDiskState>,
|
||||||
|
|
||||||
/// We remember when layers were evicted, to prevent re-downloading them.
|
/// We remember when layers were evicted, to prevent re-downloading them.
|
||||||
pub(super) evicted_at: HashMap<LayerName, SystemTime>,
|
pub(super) evicted_at: HashMap<LayerName, SystemTime>,
|
||||||
}
|
}
|
||||||
|
|
||||||
|
impl SecondaryDetailTimeline {
|
||||||
|
pub(super) fn remove_layer(
|
||||||
|
&mut self,
|
||||||
|
name: &LayerName,
|
||||||
|
resident_metric: &UIntGauge,
|
||||||
|
) -> Option<OnDiskState> {
|
||||||
|
let removed = self.on_disk_layers.remove(name);
|
||||||
|
if let Some(removed) = &removed {
|
||||||
|
resident_metric.sub(removed.file_size());
|
||||||
|
}
|
||||||
|
removed
|
||||||
|
}
|
||||||
|
|
||||||
|
/// `local_path`
|
||||||
|
fn touch_layer<F>(
|
||||||
|
&mut self,
|
||||||
|
conf: &'static PageServerConf,
|
||||||
|
tenant_shard_id: &TenantShardId,
|
||||||
|
timeline_id: &TimelineId,
|
||||||
|
touched: &HeatMapLayer,
|
||||||
|
resident_metric: &UIntGauge,
|
||||||
|
local_path: F,
|
||||||
|
) where
|
||||||
|
F: FnOnce() -> Utf8PathBuf,
|
||||||
|
{
|
||||||
|
use std::collections::hash_map::Entry;
|
||||||
|
match self.on_disk_layers.entry(touched.name.clone()) {
|
||||||
|
Entry::Occupied(mut v) => {
|
||||||
|
v.get_mut().access_time = touched.access_time;
|
||||||
|
}
|
||||||
|
Entry::Vacant(e) => {
|
||||||
|
e.insert(OnDiskState::new(
|
||||||
|
conf,
|
||||||
|
tenant_shard_id,
|
||||||
|
timeline_id,
|
||||||
|
touched.name.clone(),
|
||||||
|
touched.metadata.clone(),
|
||||||
|
touched.access_time,
|
||||||
|
local_path(),
|
||||||
|
));
|
||||||
|
resident_metric.add(touched.metadata.file_size);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
// Aspects of a heatmap that we remember after downloading it
|
// Aspects of a heatmap that we remember after downloading it
|
||||||
#[derive(Clone, Debug)]
|
#[derive(Clone, Debug)]
|
||||||
struct DownloadSummary {
|
struct DownloadSummary {
|
||||||
@@ -158,7 +209,7 @@ pub(super) struct SecondaryDetail {
|
|||||||
|
|
||||||
last_download: Option<DownloadSummary>,
|
last_download: Option<DownloadSummary>,
|
||||||
next_download: Option<Instant>,
|
next_download: Option<Instant>,
|
||||||
pub(super) timelines: HashMap<TimelineId, SecondaryDetailTimeline>,
|
timelines: HashMap<TimelineId, SecondaryDetailTimeline>,
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Helper for logging SystemTime
|
/// Helper for logging SystemTime
|
||||||
@@ -191,6 +242,38 @@ impl SecondaryDetail {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
pub(super) fn evict_layer(
|
||||||
|
&mut self,
|
||||||
|
name: LayerName,
|
||||||
|
timeline_id: &TimelineId,
|
||||||
|
now: SystemTime,
|
||||||
|
resident_metric: &UIntGauge,
|
||||||
|
) -> Option<OnDiskState> {
|
||||||
|
let timeline = self.timelines.get_mut(timeline_id)?;
|
||||||
|
let removed = timeline.remove_layer(&name, resident_metric);
|
||||||
|
if removed.is_some() {
|
||||||
|
timeline.evicted_at.insert(name, now);
|
||||||
|
}
|
||||||
|
removed
|
||||||
|
}
|
||||||
|
|
||||||
|
pub(super) fn remove_timeline(
|
||||||
|
&mut self,
|
||||||
|
timeline_id: &TimelineId,
|
||||||
|
resident_metric: &UIntGauge,
|
||||||
|
) {
|
||||||
|
let removed = self.timelines.remove(timeline_id);
|
||||||
|
if let Some(removed) = removed {
|
||||||
|
resident_metric.sub(
|
||||||
|
removed
|
||||||
|
.on_disk_layers
|
||||||
|
.values()
|
||||||
|
.map(|l| l.metadata.file_size)
|
||||||
|
.sum(),
|
||||||
|
);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
/// Additionally returns the total number of layers, used for more stable relative access time
|
/// Additionally returns the total number of layers, used for more stable relative access time
|
||||||
/// based eviction.
|
/// based eviction.
|
||||||
pub(super) fn get_layers_for_eviction(
|
pub(super) fn get_layers_for_eviction(
|
||||||
@@ -262,6 +345,7 @@ impl scheduler::RunningJob for RunningDownload {
|
|||||||
struct CompleteDownload {
|
struct CompleteDownload {
|
||||||
secondary_state: Arc<SecondaryTenant>,
|
secondary_state: Arc<SecondaryTenant>,
|
||||||
completed_at: Instant,
|
completed_at: Instant,
|
||||||
|
result: Result<(), UpdateError>,
|
||||||
}
|
}
|
||||||
|
|
||||||
impl scheduler::Completion for CompleteDownload {
|
impl scheduler::Completion for CompleteDownload {
|
||||||
@@ -286,21 +370,33 @@ impl JobGenerator<PendingDownload, RunningDownload, CompleteDownload, DownloadCo
|
|||||||
let CompleteDownload {
|
let CompleteDownload {
|
||||||
secondary_state,
|
secondary_state,
|
||||||
completed_at: _completed_at,
|
completed_at: _completed_at,
|
||||||
|
result,
|
||||||
} = completion;
|
} = completion;
|
||||||
|
|
||||||
tracing::debug!("Secondary tenant download completed");
|
tracing::debug!("Secondary tenant download completed");
|
||||||
|
|
||||||
let mut detail = secondary_state.detail.lock().unwrap();
|
let mut detail = secondary_state.detail.lock().unwrap();
|
||||||
|
|
||||||
let period = detail
|
match result {
|
||||||
.last_download
|
Err(UpdateError::Restart) => {
|
||||||
.as_ref()
|
// Start downloading again as soon as we can. This will involve waiting for the scheduler's
|
||||||
.map(|d| d.upload_period)
|
// scheduling interval. This slightly reduces the peak download speed of tenants that hit their
|
||||||
.unwrap_or(DEFAULT_DOWNLOAD_INTERVAL);
|
// deadline and keep restarting, but that also helps give other tenants a chance to execute rather
|
||||||
|
// that letting one big tenant dominate for a long time.
|
||||||
|
detail.next_download = Some(Instant::now());
|
||||||
|
}
|
||||||
|
_ => {
|
||||||
|
let period = detail
|
||||||
|
.last_download
|
||||||
|
.as_ref()
|
||||||
|
.map(|d| d.upload_period)
|
||||||
|
.unwrap_or(DEFAULT_DOWNLOAD_INTERVAL);
|
||||||
|
|
||||||
// We advance next_download irrespective of errors: we don't want error cases to result in
|
// We advance next_download irrespective of errors: we don't want error cases to result in
|
||||||
// expensive busy-polling.
|
// expensive busy-polling.
|
||||||
detail.next_download = Some(Instant::now() + period_jitter(period, 5));
|
detail.next_download = Some(Instant::now() + period_jitter(period, 5));
|
||||||
|
}
|
||||||
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
async fn schedule(&mut self) -> SchedulingResult<PendingDownload> {
|
async fn schedule(&mut self) -> SchedulingResult<PendingDownload> {
|
||||||
@@ -396,9 +492,10 @@ impl JobGenerator<PendingDownload, RunningDownload, CompleteDownload, DownloadCo
|
|||||||
(RunningDownload { barrier }, Box::pin(async move {
|
(RunningDownload { barrier }, Box::pin(async move {
|
||||||
let _completion = completion;
|
let _completion = completion;
|
||||||
|
|
||||||
match TenantDownloader::new(conf, &remote_storage, &secondary_state)
|
let result = TenantDownloader::new(conf, &remote_storage, &secondary_state)
|
||||||
.download(&download_ctx)
|
.download(&download_ctx)
|
||||||
.await
|
.await;
|
||||||
|
match &result
|
||||||
{
|
{
|
||||||
Err(UpdateError::NoData) => {
|
Err(UpdateError::NoData) => {
|
||||||
tracing::info!("No heatmap found for tenant. This is fine if it is new.");
|
tracing::info!("No heatmap found for tenant. This is fine if it is new.");
|
||||||
@@ -415,6 +512,9 @@ impl JobGenerator<PendingDownload, RunningDownload, CompleteDownload, DownloadCo
|
|||||||
Err(e @ (UpdateError::DownloadError(_) | UpdateError::Other(_))) => {
|
Err(e @ (UpdateError::DownloadError(_) | UpdateError::Other(_))) => {
|
||||||
tracing::error!("Error while downloading tenant: {e}");
|
tracing::error!("Error while downloading tenant: {e}");
|
||||||
},
|
},
|
||||||
|
Err(UpdateError::Restart) => {
|
||||||
|
tracing::info!("Download reached deadline & will restart to update heatmap")
|
||||||
|
}
|
||||||
Ok(()) => {}
|
Ok(()) => {}
|
||||||
};
|
};
|
||||||
|
|
||||||
@@ -436,6 +536,7 @@ impl JobGenerator<PendingDownload, RunningDownload, CompleteDownload, DownloadCo
|
|||||||
CompleteDownload {
|
CompleteDownload {
|
||||||
secondary_state,
|
secondary_state,
|
||||||
completed_at: Instant::now(),
|
completed_at: Instant::now(),
|
||||||
|
result
|
||||||
}
|
}
|
||||||
}.instrument(info_span!(parent: None, "secondary_download", tenant_id=%tenant_shard_id.tenant_id, shard_id=%tenant_shard_id.shard_slug()))))
|
}.instrument(info_span!(parent: None, "secondary_download", tenant_id=%tenant_shard_id.tenant_id, shard_id=%tenant_shard_id.shard_slug()))))
|
||||||
}
|
}
|
||||||
@@ -452,6 +553,11 @@ struct TenantDownloader<'a> {
|
|||||||
/// Errors that may be encountered while updating a tenant
|
/// Errors that may be encountered while updating a tenant
|
||||||
#[derive(thiserror::Error, Debug)]
|
#[derive(thiserror::Error, Debug)]
|
||||||
enum UpdateError {
|
enum UpdateError {
|
||||||
|
/// This is not a true failure, but it's how a download indicates that it would like to be restarted by
|
||||||
|
/// the scheduler, to pick up the latest heatmap
|
||||||
|
#[error("Reached deadline, restarting downloads")]
|
||||||
|
Restart,
|
||||||
|
|
||||||
#[error("No remote data found")]
|
#[error("No remote data found")]
|
||||||
NoData,
|
NoData,
|
||||||
#[error("Insufficient local storage space")]
|
#[error("Insufficient local storage space")]
|
||||||
@@ -578,8 +684,13 @@ impl<'a> TenantDownloader<'a> {
|
|||||||
Some(t) => t,
|
Some(t) => t,
|
||||||
None => {
|
None => {
|
||||||
// We have no existing state: need to scan local disk for layers first.
|
// We have no existing state: need to scan local disk for layers first.
|
||||||
let timeline_state =
|
let timeline_state = init_timeline_state(
|
||||||
init_timeline_state(self.conf, tenant_shard_id, timeline).await;
|
self.conf,
|
||||||
|
tenant_shard_id,
|
||||||
|
timeline,
|
||||||
|
&self.secondary_state.resident_size_metric,
|
||||||
|
)
|
||||||
|
.await;
|
||||||
|
|
||||||
// Re-acquire detail lock now that we're done with async load from local FS
|
// Re-acquire detail lock now that we're done with async load from local FS
|
||||||
self.secondary_state
|
self.secondary_state
|
||||||
@@ -603,6 +714,26 @@ impl<'a> TenantDownloader<'a> {
|
|||||||
self.prepare_timelines(&heatmap, heatmap_mtime).await?;
|
self.prepare_timelines(&heatmap, heatmap_mtime).await?;
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// Calculate a deadline for downloads: if downloading takes longer than this, it is useful to drop out and start again,
|
||||||
|
// so that we are always using reasonably a fresh heatmap. Otherwise, if we had really huge content to download, we might
|
||||||
|
// spend 10s of minutes downloading layers we don't need.
|
||||||
|
// (see https://github.com/neondatabase/neon/issues/8182)
|
||||||
|
let deadline = {
|
||||||
|
let period = self
|
||||||
|
.secondary_state
|
||||||
|
.detail
|
||||||
|
.lock()
|
||||||
|
.unwrap()
|
||||||
|
.last_download
|
||||||
|
.as_ref()
|
||||||
|
.map(|d| d.upload_period)
|
||||||
|
.unwrap_or(DEFAULT_DOWNLOAD_INTERVAL);
|
||||||
|
|
||||||
|
// Use double the period: we are not promising to complete within the period, this is just a heuristic
|
||||||
|
// to keep using a "reasonably fresh" heatmap.
|
||||||
|
Instant::now() + period * 2
|
||||||
|
};
|
||||||
|
|
||||||
// Download the layers in the heatmap
|
// Download the layers in the heatmap
|
||||||
for timeline in heatmap.timelines {
|
for timeline in heatmap.timelines {
|
||||||
let timeline_state = timeline_states
|
let timeline_state = timeline_states
|
||||||
@@ -618,7 +749,7 @@ impl<'a> TenantDownloader<'a> {
|
|||||||
}
|
}
|
||||||
|
|
||||||
let timeline_id = timeline.timeline_id;
|
let timeline_id = timeline.timeline_id;
|
||||||
self.download_timeline(timeline, timeline_state, ctx)
|
self.download_timeline(timeline, timeline_state, deadline, ctx)
|
||||||
.instrument(tracing::info_span!(
|
.instrument(tracing::info_span!(
|
||||||
"secondary_download_timeline",
|
"secondary_download_timeline",
|
||||||
tenant_id=%tenant_shard_id.tenant_id,
|
tenant_id=%tenant_shard_id.tenant_id,
|
||||||
@@ -628,6 +759,25 @@ impl<'a> TenantDownloader<'a> {
|
|||||||
.await?;
|
.await?;
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// Metrics consistency check in testing builds
|
||||||
|
if cfg!(feature = "testing") {
|
||||||
|
let detail = self.secondary_state.detail.lock().unwrap();
|
||||||
|
let resident_size = detail
|
||||||
|
.timelines
|
||||||
|
.values()
|
||||||
|
.map(|tl| {
|
||||||
|
tl.on_disk_layers
|
||||||
|
.values()
|
||||||
|
.map(|v| v.metadata.file_size)
|
||||||
|
.sum::<u64>()
|
||||||
|
})
|
||||||
|
.sum::<u64>();
|
||||||
|
assert_eq!(
|
||||||
|
resident_size,
|
||||||
|
self.secondary_state.resident_size_metric.get()
|
||||||
|
);
|
||||||
|
}
|
||||||
|
|
||||||
// Only update last_etag after a full successful download: this way will not skip
|
// Only update last_etag after a full successful download: this way will not skip
|
||||||
// the next download, even if the heatmap's actual etag is unchanged.
|
// the next download, even if the heatmap's actual etag is unchanged.
|
||||||
self.secondary_state.detail.lock().unwrap().last_download = Some(DownloadSummary {
|
self.secondary_state.detail.lock().unwrap().last_download = Some(DownloadSummary {
|
||||||
@@ -740,7 +890,7 @@ impl<'a> TenantDownloader<'a> {
|
|||||||
for delete_timeline in &delete_timelines {
|
for delete_timeline in &delete_timelines {
|
||||||
// We haven't removed from disk yet, but optimistically remove from in-memory state: if removal
|
// We haven't removed from disk yet, but optimistically remove from in-memory state: if removal
|
||||||
// from disk fails that will be a fatal error.
|
// from disk fails that will be a fatal error.
|
||||||
detail.timelines.remove(delete_timeline);
|
detail.remove_timeline(delete_timeline, &self.secondary_state.resident_size_metric);
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -758,7 +908,7 @@ impl<'a> TenantDownloader<'a> {
|
|||||||
let Some(timeline_state) = detail.timelines.get_mut(&timeline_id) else {
|
let Some(timeline_state) = detail.timelines.get_mut(&timeline_id) else {
|
||||||
continue;
|
continue;
|
||||||
};
|
};
|
||||||
timeline_state.on_disk_layers.remove(&layer_name);
|
timeline_state.remove_layer(&layer_name, &self.secondary_state.resident_size_metric);
|
||||||
}
|
}
|
||||||
|
|
||||||
for timeline_id in delete_timelines {
|
for timeline_id in delete_timelines {
|
||||||
@@ -827,26 +977,28 @@ impl<'a> TenantDownloader<'a> {
|
|||||||
.and_then(|x| x)
|
.and_then(|x| x)
|
||||||
}
|
}
|
||||||
|
|
||||||
async fn download_timeline(
|
/// Download heatmap layers that are not present on local disk, or update their
|
||||||
|
/// access time if they are already present.
|
||||||
|
async fn download_timeline_layers(
|
||||||
&self,
|
&self,
|
||||||
|
tenant_shard_id: &TenantShardId,
|
||||||
timeline: HeatMapTimeline,
|
timeline: HeatMapTimeline,
|
||||||
timeline_state: SecondaryDetailTimeline,
|
timeline_state: SecondaryDetailTimeline,
|
||||||
|
deadline: Instant,
|
||||||
ctx: &RequestContext,
|
ctx: &RequestContext,
|
||||||
) -> Result<(), UpdateError> {
|
) -> (Result<(), UpdateError>, Vec<HeatMapLayer>) {
|
||||||
debug_assert_current_span_has_tenant_and_timeline_id();
|
|
||||||
let tenant_shard_id = self.secondary_state.get_tenant_shard_id();
|
|
||||||
|
|
||||||
// Accumulate updates to the state
|
// Accumulate updates to the state
|
||||||
let mut touched = Vec::new();
|
let mut touched = Vec::new();
|
||||||
|
|
||||||
tracing::debug!(timeline_id=%timeline.timeline_id, "Downloading layers, {} in heatmap", timeline.layers.len());
|
|
||||||
|
|
||||||
// Download heatmap layers that are not present on local disk, or update their
|
|
||||||
// access time if they are already present.
|
|
||||||
for layer in timeline.layers {
|
for layer in timeline.layers {
|
||||||
if self.secondary_state.cancel.is_cancelled() {
|
if self.secondary_state.cancel.is_cancelled() {
|
||||||
tracing::debug!("Cancelled -- dropping out of layer loop");
|
tracing::debug!("Cancelled -- dropping out of layer loop");
|
||||||
return Err(UpdateError::Cancelled);
|
return (Err(UpdateError::Cancelled), touched);
|
||||||
|
}
|
||||||
|
|
||||||
|
if Instant::now() > deadline {
|
||||||
|
// We've been running downloads for a while, restart to download latest heatmap.
|
||||||
|
return (Err(UpdateError::Restart), touched);
|
||||||
}
|
}
|
||||||
|
|
||||||
// Existing on-disk layers: just update their access time.
|
// Existing on-disk layers: just update their access time.
|
||||||
@@ -916,52 +1068,66 @@ impl<'a> TenantDownloader<'a> {
|
|||||||
|
|
||||||
match self
|
match self
|
||||||
.download_layer(tenant_shard_id, &timeline.timeline_id, layer, ctx)
|
.download_layer(tenant_shard_id, &timeline.timeline_id, layer, ctx)
|
||||||
.await?
|
.await
|
||||||
{
|
{
|
||||||
Some(layer) => touched.push(layer),
|
Ok(Some(layer)) => touched.push(layer),
|
||||||
None => {
|
Ok(None) => {
|
||||||
// Not an error but we didn't download it: remote layer is missing. Don't add it to the list of
|
// Not an error but we didn't download it: remote layer is missing. Don't add it to the list of
|
||||||
// things to consider touched.
|
// things to consider touched.
|
||||||
}
|
}
|
||||||
}
|
Err(e) => {
|
||||||
}
|
return (Err(e), touched);
|
||||||
|
|
||||||
// Write updates to state to record layers we just downloaded or touched.
|
|
||||||
{
|
|
||||||
let mut detail = self.secondary_state.detail.lock().unwrap();
|
|
||||||
let timeline_detail = detail.timelines.entry(timeline.timeline_id).or_default();
|
|
||||||
|
|
||||||
tracing::info!("Wrote timeline_detail for {} touched layers", touched.len());
|
|
||||||
|
|
||||||
for t in touched {
|
|
||||||
use std::collections::hash_map::Entry;
|
|
||||||
match timeline_detail.on_disk_layers.entry(t.name.clone()) {
|
|
||||||
Entry::Occupied(mut v) => {
|
|
||||||
v.get_mut().access_time = t.access_time;
|
|
||||||
}
|
|
||||||
Entry::Vacant(e) => {
|
|
||||||
let local_path = local_layer_path(
|
|
||||||
self.conf,
|
|
||||||
tenant_shard_id,
|
|
||||||
&timeline.timeline_id,
|
|
||||||
&t.name,
|
|
||||||
&t.metadata.generation,
|
|
||||||
);
|
|
||||||
e.insert(OnDiskState::new(
|
|
||||||
self.conf,
|
|
||||||
tenant_shard_id,
|
|
||||||
&timeline.timeline_id,
|
|
||||||
t.name,
|
|
||||||
t.metadata.clone(),
|
|
||||||
t.access_time,
|
|
||||||
local_path,
|
|
||||||
));
|
|
||||||
}
|
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
Ok(())
|
(Ok(()), touched)
|
||||||
|
}
|
||||||
|
|
||||||
|
async fn download_timeline(
|
||||||
|
&self,
|
||||||
|
timeline: HeatMapTimeline,
|
||||||
|
timeline_state: SecondaryDetailTimeline,
|
||||||
|
deadline: Instant,
|
||||||
|
ctx: &RequestContext,
|
||||||
|
) -> Result<(), UpdateError> {
|
||||||
|
debug_assert_current_span_has_tenant_and_timeline_id();
|
||||||
|
let tenant_shard_id = self.secondary_state.get_tenant_shard_id();
|
||||||
|
let timeline_id = timeline.timeline_id;
|
||||||
|
|
||||||
|
tracing::debug!(timeline_id=%timeline_id, "Downloading layers, {} in heatmap", timeline.layers.len());
|
||||||
|
|
||||||
|
let (result, touched) = self
|
||||||
|
.download_timeline_layers(tenant_shard_id, timeline, timeline_state, deadline, ctx)
|
||||||
|
.await;
|
||||||
|
|
||||||
|
// Write updates to state to record layers we just downloaded or touched, irrespective of whether the overall result was successful
|
||||||
|
{
|
||||||
|
let mut detail = self.secondary_state.detail.lock().unwrap();
|
||||||
|
let timeline_detail = detail.timelines.entry(timeline_id).or_default();
|
||||||
|
|
||||||
|
tracing::info!("Wrote timeline_detail for {} touched layers", touched.len());
|
||||||
|
touched.into_iter().for_each(|t| {
|
||||||
|
timeline_detail.touch_layer(
|
||||||
|
self.conf,
|
||||||
|
tenant_shard_id,
|
||||||
|
&timeline_id,
|
||||||
|
&t,
|
||||||
|
&self.secondary_state.resident_size_metric,
|
||||||
|
|| {
|
||||||
|
local_layer_path(
|
||||||
|
self.conf,
|
||||||
|
tenant_shard_id,
|
||||||
|
&timeline_id,
|
||||||
|
&t.name,
|
||||||
|
&t.metadata.generation,
|
||||||
|
)
|
||||||
|
},
|
||||||
|
)
|
||||||
|
});
|
||||||
|
}
|
||||||
|
|
||||||
|
result
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Call this during timeline download if a layer will _not_ be downloaded, to update progress statistics
|
/// Call this during timeline download if a layer will _not_ be downloaded, to update progress statistics
|
||||||
@@ -1067,6 +1233,7 @@ async fn init_timeline_state(
|
|||||||
conf: &'static PageServerConf,
|
conf: &'static PageServerConf,
|
||||||
tenant_shard_id: &TenantShardId,
|
tenant_shard_id: &TenantShardId,
|
||||||
heatmap: &HeatMapTimeline,
|
heatmap: &HeatMapTimeline,
|
||||||
|
resident_metric: &UIntGauge,
|
||||||
) -> SecondaryDetailTimeline {
|
) -> SecondaryDetailTimeline {
|
||||||
let timeline_path = conf.timeline_path(tenant_shard_id, &heatmap.timeline_id);
|
let timeline_path = conf.timeline_path(tenant_shard_id, &heatmap.timeline_id);
|
||||||
let mut detail = SecondaryDetailTimeline::default();
|
let mut detail = SecondaryDetailTimeline::default();
|
||||||
@@ -1142,17 +1309,13 @@ async fn init_timeline_state(
|
|||||||
} else {
|
} else {
|
||||||
// We expect the access time to be initialized immediately afterwards, when
|
// We expect the access time to be initialized immediately afterwards, when
|
||||||
// the latest heatmap is applied to the state.
|
// the latest heatmap is applied to the state.
|
||||||
detail.on_disk_layers.insert(
|
detail.touch_layer(
|
||||||
name.clone(),
|
conf,
|
||||||
OnDiskState::new(
|
tenant_shard_id,
|
||||||
conf,
|
&heatmap.timeline_id,
|
||||||
tenant_shard_id,
|
remote_meta,
|
||||||
&heatmap.timeline_id,
|
resident_metric,
|
||||||
name,
|
|| file_path,
|
||||||
remote_meta.metadata.clone(),
|
|
||||||
remote_meta.access_time,
|
|
||||||
file_path,
|
|
||||||
),
|
|
||||||
);
|
);
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -3,6 +3,7 @@ use std::collections::hash_map::Entry;
|
|||||||
use std::collections::{HashMap, HashSet};
|
use std::collections::{HashMap, HashSet};
|
||||||
use std::sync::Arc;
|
use std::sync::Arc;
|
||||||
|
|
||||||
|
use tenant_size_model::svg::SvgBranchKind;
|
||||||
use tokio::sync::oneshot::error::RecvError;
|
use tokio::sync::oneshot::error::RecvError;
|
||||||
use tokio::sync::Semaphore;
|
use tokio::sync::Semaphore;
|
||||||
use tokio_util::sync::CancellationToken;
|
use tokio_util::sync::CancellationToken;
|
||||||
@@ -87,6 +88,9 @@ impl SegmentMeta {
|
|||||||
LsnKind::BranchPoint => true,
|
LsnKind::BranchPoint => true,
|
||||||
LsnKind::GcCutOff => true,
|
LsnKind::GcCutOff => true,
|
||||||
LsnKind::BranchEnd => false,
|
LsnKind::BranchEnd => false,
|
||||||
|
LsnKind::LeasePoint => true,
|
||||||
|
LsnKind::LeaseStart => false,
|
||||||
|
LsnKind::LeaseEnd => false,
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
@@ -103,6 +107,21 @@ pub enum LsnKind {
|
|||||||
GcCutOff,
|
GcCutOff,
|
||||||
/// Last record LSN
|
/// Last record LSN
|
||||||
BranchEnd,
|
BranchEnd,
|
||||||
|
/// A LSN lease is granted here.
|
||||||
|
LeasePoint,
|
||||||
|
/// A lease starts from here.
|
||||||
|
LeaseStart,
|
||||||
|
/// Last record LSN for the lease (should have the same LSN as the previous [`LsnKind::LeaseStart`]).
|
||||||
|
LeaseEnd,
|
||||||
|
}
|
||||||
|
|
||||||
|
impl From<LsnKind> for SvgBranchKind {
|
||||||
|
fn from(kind: LsnKind) -> Self {
|
||||||
|
match kind {
|
||||||
|
LsnKind::LeasePoint | LsnKind::LeaseStart | LsnKind::LeaseEnd => SvgBranchKind::Lease,
|
||||||
|
_ => SvgBranchKind::Timeline,
|
||||||
|
}
|
||||||
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Collect all relevant LSNs to the inputs. These will only be helpful in the serialized form as
|
/// Collect all relevant LSNs to the inputs. These will only be helpful in the serialized form as
|
||||||
@@ -124,6 +143,9 @@ pub struct TimelineInputs {
|
|||||||
|
|
||||||
/// Cutoff point calculated from the user-supplied 'max_retention_period'
|
/// Cutoff point calculated from the user-supplied 'max_retention_period'
|
||||||
retention_param_cutoff: Option<Lsn>,
|
retention_param_cutoff: Option<Lsn>,
|
||||||
|
|
||||||
|
/// Lease points on the timeline
|
||||||
|
lease_points: Vec<Lsn>,
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Gathers the inputs for the tenant sizing model.
|
/// Gathers the inputs for the tenant sizing model.
|
||||||
@@ -234,6 +256,13 @@ pub(super) async fn gather_inputs(
|
|||||||
None
|
None
|
||||||
};
|
};
|
||||||
|
|
||||||
|
let lease_points = gc_info
|
||||||
|
.leases
|
||||||
|
.keys()
|
||||||
|
.filter(|&&lsn| lsn > ancestor_lsn)
|
||||||
|
.copied()
|
||||||
|
.collect::<Vec<_>>();
|
||||||
|
|
||||||
// next_gc_cutoff in parent branch are not of interest (right now at least), nor do we
|
// next_gc_cutoff in parent branch are not of interest (right now at least), nor do we
|
||||||
// want to query any logical size before initdb_lsn.
|
// want to query any logical size before initdb_lsn.
|
||||||
let branch_start_lsn = cmp::max(ancestor_lsn, timeline.initdb_lsn);
|
let branch_start_lsn = cmp::max(ancestor_lsn, timeline.initdb_lsn);
|
||||||
@@ -248,6 +277,8 @@ pub(super) async fn gather_inputs(
|
|||||||
.map(|lsn| (lsn, LsnKind::BranchPoint))
|
.map(|lsn| (lsn, LsnKind::BranchPoint))
|
||||||
.collect::<Vec<_>>();
|
.collect::<Vec<_>>();
|
||||||
|
|
||||||
|
lsns.extend(lease_points.iter().map(|&lsn| (lsn, LsnKind::LeasePoint)));
|
||||||
|
|
||||||
drop(gc_info);
|
drop(gc_info);
|
||||||
|
|
||||||
// Add branch points we collected earlier, just in case there were any that were
|
// Add branch points we collected earlier, just in case there were any that were
|
||||||
@@ -296,6 +327,7 @@ pub(super) async fn gather_inputs(
|
|||||||
if kind == LsnKind::BranchPoint {
|
if kind == LsnKind::BranchPoint {
|
||||||
branchpoint_segments.insert((timeline_id, lsn), segments.len());
|
branchpoint_segments.insert((timeline_id, lsn), segments.len());
|
||||||
}
|
}
|
||||||
|
|
||||||
segments.push(SegmentMeta {
|
segments.push(SegmentMeta {
|
||||||
segment: Segment {
|
segment: Segment {
|
||||||
parent: Some(parent),
|
parent: Some(parent),
|
||||||
@@ -306,7 +338,45 @@ pub(super) async fn gather_inputs(
|
|||||||
timeline_id: timeline.timeline_id,
|
timeline_id: timeline.timeline_id,
|
||||||
kind,
|
kind,
|
||||||
});
|
});
|
||||||
parent += 1;
|
|
||||||
|
parent = segments.len() - 1;
|
||||||
|
|
||||||
|
if kind == LsnKind::LeasePoint {
|
||||||
|
// Needs `LeaseStart` and `LeaseEnd` as well to model lease as a read-only branch that never writes data
|
||||||
|
// (i.e. it's lsn has not advanced from ancestor_lsn), and therefore the three segments have the same LSN
|
||||||
|
// value. Without the other two segments, the calculation code would not count the leased LSN as a point
|
||||||
|
// to be retained.
|
||||||
|
// Did not use `BranchStart` or `BranchEnd` so we can differentiate branches and leases during debug.
|
||||||
|
//
|
||||||
|
// Alt Design: rewrite the entire calculation code to be independent of timeline id. Both leases and
|
||||||
|
// branch points can be given a synthetic id so we can unite them.
|
||||||
|
let mut lease_parent = parent;
|
||||||
|
|
||||||
|
// Start of a lease.
|
||||||
|
segments.push(SegmentMeta {
|
||||||
|
segment: Segment {
|
||||||
|
parent: Some(lease_parent),
|
||||||
|
lsn: lsn.0,
|
||||||
|
size: None, // Filled in later, if necessary
|
||||||
|
needed: lsn > next_gc_cutoff, // only needed if the point is within rentention.
|
||||||
|
},
|
||||||
|
timeline_id: timeline.timeline_id,
|
||||||
|
kind: LsnKind::LeaseStart,
|
||||||
|
});
|
||||||
|
lease_parent += 1;
|
||||||
|
|
||||||
|
// End of the lease.
|
||||||
|
segments.push(SegmentMeta {
|
||||||
|
segment: Segment {
|
||||||
|
parent: Some(lease_parent),
|
||||||
|
lsn: lsn.0,
|
||||||
|
size: None, // Filled in later, if necessary
|
||||||
|
needed: true, // everything at the lease LSN must be readable => is needed
|
||||||
|
},
|
||||||
|
timeline_id: timeline.timeline_id,
|
||||||
|
kind: LsnKind::LeaseEnd,
|
||||||
|
});
|
||||||
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
// Current end of the timeline
|
// Current end of the timeline
|
||||||
@@ -332,6 +402,7 @@ pub(super) async fn gather_inputs(
|
|||||||
pitr_cutoff,
|
pitr_cutoff,
|
||||||
next_gc_cutoff,
|
next_gc_cutoff,
|
||||||
retention_param_cutoff,
|
retention_param_cutoff,
|
||||||
|
lease_points,
|
||||||
});
|
});
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -674,7 +745,8 @@ fn verify_size_for_multiple_branches() {
|
|||||||
"horizon_cutoff": "0/2210CD0",
|
"horizon_cutoff": "0/2210CD0",
|
||||||
"pitr_cutoff": "0/2210CD0",
|
"pitr_cutoff": "0/2210CD0",
|
||||||
"next_gc_cutoff": "0/2210CD0",
|
"next_gc_cutoff": "0/2210CD0",
|
||||||
"retention_param_cutoff": null
|
"retention_param_cutoff": null,
|
||||||
|
"lease_points": []
|
||||||
},
|
},
|
||||||
{
|
{
|
||||||
"timeline_id": "454626700469f0a9914949b9d018e876",
|
"timeline_id": "454626700469f0a9914949b9d018e876",
|
||||||
@@ -684,7 +756,8 @@ fn verify_size_for_multiple_branches() {
|
|||||||
"horizon_cutoff": "0/1817770",
|
"horizon_cutoff": "0/1817770",
|
||||||
"pitr_cutoff": "0/1817770",
|
"pitr_cutoff": "0/1817770",
|
||||||
"next_gc_cutoff": "0/1817770",
|
"next_gc_cutoff": "0/1817770",
|
||||||
"retention_param_cutoff": null
|
"retention_param_cutoff": null,
|
||||||
|
"lease_points": []
|
||||||
},
|
},
|
||||||
{
|
{
|
||||||
"timeline_id": "cb5e3cbe60a4afc00d01880e1a37047f",
|
"timeline_id": "cb5e3cbe60a4afc00d01880e1a37047f",
|
||||||
@@ -694,7 +767,8 @@ fn verify_size_for_multiple_branches() {
|
|||||||
"horizon_cutoff": "0/18B3D98",
|
"horizon_cutoff": "0/18B3D98",
|
||||||
"pitr_cutoff": "0/18B3D98",
|
"pitr_cutoff": "0/18B3D98",
|
||||||
"next_gc_cutoff": "0/18B3D98",
|
"next_gc_cutoff": "0/18B3D98",
|
||||||
"retention_param_cutoff": null
|
"retention_param_cutoff": null,
|
||||||
|
"lease_points": []
|
||||||
}
|
}
|
||||||
]
|
]
|
||||||
}
|
}
|
||||||
@@ -749,7 +823,8 @@ fn verify_size_for_one_branch() {
|
|||||||
"horizon_cutoff": "47/240A5860",
|
"horizon_cutoff": "47/240A5860",
|
||||||
"pitr_cutoff": "47/240A5860",
|
"pitr_cutoff": "47/240A5860",
|
||||||
"next_gc_cutoff": "47/240A5860",
|
"next_gc_cutoff": "47/240A5860",
|
||||||
"retention_param_cutoff": "0/0"
|
"retention_param_cutoff": "0/0",
|
||||||
|
"lease_points": []
|
||||||
}
|
}
|
||||||
]
|
]
|
||||||
}"#;
|
}"#;
|
||||||
|
|||||||
@@ -49,7 +49,7 @@ use camino::{Utf8Path, Utf8PathBuf};
|
|||||||
use futures::StreamExt;
|
use futures::StreamExt;
|
||||||
use itertools::Itertools;
|
use itertools::Itertools;
|
||||||
use pageserver_api::keyspace::KeySpace;
|
use pageserver_api::keyspace::KeySpace;
|
||||||
use pageserver_api::models::LayerAccessKind;
|
use pageserver_api::models::{ImageCompressionAlgorithm, LayerAccessKind};
|
||||||
use pageserver_api::shard::TenantShardId;
|
use pageserver_api::shard::TenantShardId;
|
||||||
use rand::{distributions::Alphanumeric, Rng};
|
use rand::{distributions::Alphanumeric, Rng};
|
||||||
use serde::{Deserialize, Serialize};
|
use serde::{Deserialize, Serialize};
|
||||||
@@ -452,7 +452,12 @@ impl DeltaLayerWriterInner {
|
|||||||
ctx: &RequestContext,
|
ctx: &RequestContext,
|
||||||
) -> (Vec<u8>, anyhow::Result<()>) {
|
) -> (Vec<u8>, anyhow::Result<()>) {
|
||||||
assert!(self.lsn_range.start <= lsn);
|
assert!(self.lsn_range.start <= lsn);
|
||||||
let (val, res) = self.blob_writer.write_blob(val, ctx).await;
|
// We don't want to use compression in delta layer creation
|
||||||
|
let compression = ImageCompressionAlgorithm::DisabledNoDecompress;
|
||||||
|
let (val, res) = self
|
||||||
|
.blob_writer
|
||||||
|
.write_blob_maybe_compressed(val, ctx, compression)
|
||||||
|
.await;
|
||||||
let off = match res {
|
let off = match res {
|
||||||
Ok(off) => off,
|
Ok(off) => off,
|
||||||
Err(e) => return (val, Err(anyhow::anyhow!(e))),
|
Err(e) => return (val, Err(anyhow::anyhow!(e))),
|
||||||
|
|||||||
@@ -165,6 +165,7 @@ pub struct ImageLayerInner {
|
|||||||
file_id: FileId,
|
file_id: FileId,
|
||||||
|
|
||||||
max_vectored_read_bytes: Option<MaxVectoredReadBytes>,
|
max_vectored_read_bytes: Option<MaxVectoredReadBytes>,
|
||||||
|
compressed_reads: bool,
|
||||||
}
|
}
|
||||||
|
|
||||||
impl std::fmt::Debug for ImageLayerInner {
|
impl std::fmt::Debug for ImageLayerInner {
|
||||||
@@ -178,7 +179,8 @@ impl std::fmt::Debug for ImageLayerInner {
|
|||||||
|
|
||||||
impl ImageLayerInner {
|
impl ImageLayerInner {
|
||||||
pub(super) async fn dump(&self, ctx: &RequestContext) -> anyhow::Result<()> {
|
pub(super) async fn dump(&self, ctx: &RequestContext) -> anyhow::Result<()> {
|
||||||
let block_reader = FileBlockReader::new(&self.file, self.file_id);
|
let block_reader =
|
||||||
|
FileBlockReader::new_with_compression(&self.file, self.file_id, self.compressed_reads);
|
||||||
let tree_reader = DiskBtreeReader::<_, KEY_SIZE>::new(
|
let tree_reader = DiskBtreeReader::<_, KEY_SIZE>::new(
|
||||||
self.index_start_blk,
|
self.index_start_blk,
|
||||||
self.index_root_blk,
|
self.index_root_blk,
|
||||||
@@ -266,9 +268,10 @@ impl ImageLayer {
|
|||||||
async fn load_inner(&self, ctx: &RequestContext) -> Result<ImageLayerInner> {
|
async fn load_inner(&self, ctx: &RequestContext) -> Result<ImageLayerInner> {
|
||||||
let path = self.path();
|
let path = self.path();
|
||||||
|
|
||||||
let loaded = ImageLayerInner::load(&path, self.desc.image_layer_lsn(), None, None, ctx)
|
let loaded =
|
||||||
.await
|
ImageLayerInner::load(&path, self.desc.image_layer_lsn(), None, None, false, ctx)
|
||||||
.and_then(|res| res)?;
|
.await
|
||||||
|
.and_then(|res| res)?;
|
||||||
|
|
||||||
// not production code
|
// not production code
|
||||||
let actual_layer_name = LayerName::from_str(path.file_name().unwrap()).unwrap();
|
let actual_layer_name = LayerName::from_str(path.file_name().unwrap()).unwrap();
|
||||||
@@ -377,6 +380,7 @@ impl ImageLayerInner {
|
|||||||
lsn: Lsn,
|
lsn: Lsn,
|
||||||
summary: Option<Summary>,
|
summary: Option<Summary>,
|
||||||
max_vectored_read_bytes: Option<MaxVectoredReadBytes>,
|
max_vectored_read_bytes: Option<MaxVectoredReadBytes>,
|
||||||
|
support_compressed_reads: bool,
|
||||||
ctx: &RequestContext,
|
ctx: &RequestContext,
|
||||||
) -> Result<Result<Self, anyhow::Error>, anyhow::Error> {
|
) -> Result<Result<Self, anyhow::Error>, anyhow::Error> {
|
||||||
let file = match VirtualFile::open(path, ctx).await {
|
let file = match VirtualFile::open(path, ctx).await {
|
||||||
@@ -420,6 +424,7 @@ impl ImageLayerInner {
|
|||||||
file,
|
file,
|
||||||
file_id,
|
file_id,
|
||||||
max_vectored_read_bytes,
|
max_vectored_read_bytes,
|
||||||
|
compressed_reads: support_compressed_reads,
|
||||||
key_range: actual_summary.key_range,
|
key_range: actual_summary.key_range,
|
||||||
}))
|
}))
|
||||||
}
|
}
|
||||||
@@ -430,7 +435,8 @@ impl ImageLayerInner {
|
|||||||
reconstruct_state: &mut ValueReconstructState,
|
reconstruct_state: &mut ValueReconstructState,
|
||||||
ctx: &RequestContext,
|
ctx: &RequestContext,
|
||||||
) -> anyhow::Result<ValueReconstructResult> {
|
) -> anyhow::Result<ValueReconstructResult> {
|
||||||
let block_reader = FileBlockReader::new(&self.file, self.file_id);
|
let block_reader =
|
||||||
|
FileBlockReader::new_with_compression(&self.file, self.file_id, self.compressed_reads);
|
||||||
let tree_reader =
|
let tree_reader =
|
||||||
DiskBtreeReader::new(self.index_start_blk, self.index_root_blk, &block_reader);
|
DiskBtreeReader::new(self.index_start_blk, self.index_root_blk, &block_reader);
|
||||||
|
|
||||||
@@ -490,12 +496,14 @@ impl ImageLayerInner {
|
|||||||
&self,
|
&self,
|
||||||
ctx: &RequestContext,
|
ctx: &RequestContext,
|
||||||
) -> anyhow::Result<Vec<(Key, Lsn, Value)>> {
|
) -> anyhow::Result<Vec<(Key, Lsn, Value)>> {
|
||||||
let block_reader = FileBlockReader::new(&self.file, self.file_id);
|
let block_reader =
|
||||||
|
FileBlockReader::new_with_compression(&self.file, self.file_id, self.compressed_reads);
|
||||||
let tree_reader =
|
let tree_reader =
|
||||||
DiskBtreeReader::new(self.index_start_blk, self.index_root_blk, &block_reader);
|
DiskBtreeReader::new(self.index_start_blk, self.index_root_blk, &block_reader);
|
||||||
let mut result = Vec::new();
|
let mut result = Vec::new();
|
||||||
let mut stream = Box::pin(tree_reader.into_stream(&[0; KEY_SIZE], ctx));
|
let mut stream = Box::pin(tree_reader.into_stream(&[0; KEY_SIZE], ctx));
|
||||||
let block_reader = FileBlockReader::new(&self.file, self.file_id);
|
let block_reader =
|
||||||
|
FileBlockReader::new_with_compression(&self.file, self.file_id, self.compressed_reads);
|
||||||
let cursor = block_reader.block_cursor();
|
let cursor = block_reader.block_cursor();
|
||||||
while let Some(item) = stream.next().await {
|
while let Some(item) = stream.next().await {
|
||||||
// TODO: dedup code with get_reconstruct_value
|
// TODO: dedup code with get_reconstruct_value
|
||||||
@@ -530,7 +538,8 @@ impl ImageLayerInner {
|
|||||||
.into(),
|
.into(),
|
||||||
);
|
);
|
||||||
|
|
||||||
let block_reader = FileBlockReader::new(&self.file, self.file_id);
|
let block_reader =
|
||||||
|
FileBlockReader::new_with_compression(&self.file, self.file_id, self.compressed_reads);
|
||||||
let tree_reader =
|
let tree_reader =
|
||||||
DiskBtreeReader::new(self.index_start_blk, self.index_root_blk, block_reader);
|
DiskBtreeReader::new(self.index_start_blk, self.index_root_blk, block_reader);
|
||||||
|
|
||||||
@@ -691,7 +700,8 @@ impl ImageLayerInner {
|
|||||||
|
|
||||||
#[cfg(test)]
|
#[cfg(test)]
|
||||||
pub(crate) fn iter<'a>(&'a self, ctx: &'a RequestContext) -> ImageLayerIterator<'a> {
|
pub(crate) fn iter<'a>(&'a self, ctx: &'a RequestContext) -> ImageLayerIterator<'a> {
|
||||||
let block_reader = FileBlockReader::new(&self.file, self.file_id);
|
let block_reader =
|
||||||
|
FileBlockReader::new_with_compression(&self.file, self.file_id, self.compressed_reads);
|
||||||
let tree_reader =
|
let tree_reader =
|
||||||
DiskBtreeReader::new(self.index_start_blk, self.index_root_blk, block_reader);
|
DiskBtreeReader::new(self.index_start_blk, self.index_root_blk, block_reader);
|
||||||
ImageLayerIterator {
|
ImageLayerIterator {
|
||||||
|
|||||||
@@ -6,13 +6,14 @@
|
|||||||
//!
|
//!
|
||||||
use crate::config::PageServerConf;
|
use crate::config::PageServerConf;
|
||||||
use crate::context::{PageContentKind, RequestContext, RequestContextBuilder};
|
use crate::context::{PageContentKind, RequestContext, RequestContextBuilder};
|
||||||
|
use crate::page_cache::PAGE_SZ;
|
||||||
use crate::repository::{Key, Value};
|
use crate::repository::{Key, Value};
|
||||||
use crate::tenant::block_io::BlockReader;
|
use crate::tenant::block_io::{BlockCursor, BlockReader, BlockReaderRef};
|
||||||
use crate::tenant::ephemeral_file::EphemeralFile;
|
use crate::tenant::ephemeral_file::EphemeralFile;
|
||||||
use crate::tenant::storage_layer::ValueReconstructResult;
|
use crate::tenant::storage_layer::ValueReconstructResult;
|
||||||
use crate::tenant::timeline::GetVectoredError;
|
use crate::tenant::timeline::GetVectoredError;
|
||||||
use crate::tenant::{PageReconstructError, Timeline};
|
use crate::tenant::{PageReconstructError, Timeline};
|
||||||
use crate::{page_cache, walrecord};
|
use crate::{l0_flush, page_cache, walrecord};
|
||||||
use anyhow::{anyhow, ensure, Result};
|
use anyhow::{anyhow, ensure, Result};
|
||||||
use pageserver_api::keyspace::KeySpace;
|
use pageserver_api::keyspace::KeySpace;
|
||||||
use pageserver_api::models::InMemoryLayerInfo;
|
use pageserver_api::models::InMemoryLayerInfo;
|
||||||
@@ -410,6 +411,7 @@ impl InMemoryLayer {
|
|||||||
continue;
|
continue;
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// TODO: this uses the page cache => https://github.com/neondatabase/neon/issues/8183
|
||||||
let buf = reader.read_blob(block_read.block_offset, &ctx).await;
|
let buf = reader.read_blob(block_read.block_offset, &ctx).await;
|
||||||
if let Err(e) = buf {
|
if let Err(e) = buf {
|
||||||
reconstruct_state
|
reconstruct_state
|
||||||
@@ -620,6 +622,13 @@ impl InMemoryLayer {
|
|||||||
// rare though, so we just accept the potential latency hit for now.
|
// rare though, so we just accept the potential latency hit for now.
|
||||||
let inner = self.inner.read().await;
|
let inner = self.inner.read().await;
|
||||||
|
|
||||||
|
let l0_flush_global_state = timeline.l0_flush_global_state.inner().clone();
|
||||||
|
use l0_flush::Inner;
|
||||||
|
let _concurrency_permit = match &*l0_flush_global_state {
|
||||||
|
Inner::PageCached => None,
|
||||||
|
Inner::Direct { semaphore, .. } => Some(semaphore.acquire().await),
|
||||||
|
};
|
||||||
|
|
||||||
let end_lsn = *self.end_lsn.get().unwrap();
|
let end_lsn = *self.end_lsn.get().unwrap();
|
||||||
|
|
||||||
let key_count = if let Some(key_range) = key_range {
|
let key_count = if let Some(key_range) = key_range {
|
||||||
@@ -645,28 +654,77 @@ impl InMemoryLayer {
|
|||||||
)
|
)
|
||||||
.await?;
|
.await?;
|
||||||
|
|
||||||
let mut buf = Vec::new();
|
match &*l0_flush_global_state {
|
||||||
|
l0_flush::Inner::PageCached => {
|
||||||
|
let ctx = RequestContextBuilder::extend(ctx)
|
||||||
|
.page_content_kind(PageContentKind::InMemoryLayer)
|
||||||
|
.build();
|
||||||
|
|
||||||
let cursor = inner.file.block_cursor();
|
let mut buf = Vec::new();
|
||||||
|
|
||||||
let ctx = RequestContextBuilder::extend(ctx)
|
let cursor = inner.file.block_cursor();
|
||||||
.page_content_kind(PageContentKind::InMemoryLayer)
|
|
||||||
.build();
|
for (key, vec_map) in inner.index.iter() {
|
||||||
for (key, vec_map) in inner.index.iter() {
|
// Write all page versions
|
||||||
// Write all page versions
|
for (lsn, pos) in vec_map.as_slice() {
|
||||||
for (lsn, pos) in vec_map.as_slice() {
|
cursor.read_blob_into_buf(*pos, &mut buf, &ctx).await?;
|
||||||
cursor.read_blob_into_buf(*pos, &mut buf, &ctx).await?;
|
let will_init = Value::des(&buf)?.will_init();
|
||||||
let will_init = Value::des(&buf)?.will_init();
|
let res;
|
||||||
let res;
|
(buf, res) = delta_layer_writer
|
||||||
(buf, res) = delta_layer_writer
|
.put_value_bytes(*key, *lsn, buf, will_init, &ctx)
|
||||||
.put_value_bytes(*key, *lsn, buf, will_init, &ctx)
|
.await;
|
||||||
.await;
|
res?;
|
||||||
res?;
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
l0_flush::Inner::Direct { .. } => {
|
||||||
|
let file_contents: Vec<u8> = inner.file.load_to_vec(ctx).await?;
|
||||||
|
assert_eq!(
|
||||||
|
file_contents.len() % PAGE_SZ,
|
||||||
|
0,
|
||||||
|
"needed by BlockReaderRef::Slice"
|
||||||
|
);
|
||||||
|
assert_eq!(file_contents.len(), {
|
||||||
|
let written = usize::try_from(inner.file.len()).unwrap();
|
||||||
|
if written % PAGE_SZ == 0 {
|
||||||
|
written
|
||||||
|
} else {
|
||||||
|
written.checked_add(PAGE_SZ - (written % PAGE_SZ)).unwrap()
|
||||||
|
}
|
||||||
|
});
|
||||||
|
|
||||||
|
let cursor = BlockCursor::new(BlockReaderRef::Slice(&file_contents));
|
||||||
|
|
||||||
|
let mut buf = Vec::new();
|
||||||
|
|
||||||
|
for (key, vec_map) in inner.index.iter() {
|
||||||
|
// Write all page versions
|
||||||
|
for (lsn, pos) in vec_map.as_slice() {
|
||||||
|
// TODO: once we have blob lengths in the in-memory index, we can
|
||||||
|
// 1. get rid of the blob_io / BlockReaderRef::Slice business and
|
||||||
|
// 2. load the file contents into a Bytes and
|
||||||
|
// 3. the use `Bytes::slice` to get the `buf` that is our blob
|
||||||
|
// 4. pass that `buf` into `put_value_bytes`
|
||||||
|
// => https://github.com/neondatabase/neon/issues/8183
|
||||||
|
cursor.read_blob_into_buf(*pos, &mut buf, ctx).await?;
|
||||||
|
let will_init = Value::des(&buf)?.will_init();
|
||||||
|
let res;
|
||||||
|
(buf, res) = delta_layer_writer
|
||||||
|
.put_value_bytes(*key, *lsn, buf, will_init, ctx)
|
||||||
|
.await;
|
||||||
|
res?;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// Hold the permit until the IO is done; if we didn't, one could drop this future,
|
||||||
|
// thereby releasing the permit, but the Vec<u8> remains allocated until the IO completes.
|
||||||
|
// => we'd have more concurrenct Vec<u8> than allowed as per the semaphore.
|
||||||
|
drop(_concurrency_permit);
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
// MAX is used here because we identify L0 layers by full key range
|
// MAX is used here because we identify L0 layers by full key range
|
||||||
let delta_layer = delta_layer_writer.finish(Key::MAX, timeline, &ctx).await?;
|
let delta_layer = delta_layer_writer.finish(Key::MAX, timeline, ctx).await?;
|
||||||
Ok(Some(delta_layer))
|
Ok(Some(delta_layer))
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -1096,19 +1096,10 @@ impl LayerInner {
|
|||||||
|
|
||||||
match rx.await {
|
match rx.await {
|
||||||
Ok(Ok(res)) => Ok(res),
|
Ok(Ok(res)) => Ok(res),
|
||||||
Ok(Err(e)) => {
|
Ok(Err(remote_storage::DownloadError::Cancelled)) => {
|
||||||
// sleep already happened in the spawned task, if it was not cancelled
|
Err(DownloadError::DownloadCancelled)
|
||||||
match e.downcast_ref::<remote_storage::DownloadError>() {
|
|
||||||
// If the download failed due to its cancellation token,
|
|
||||||
// propagate the cancellation error upstream.
|
|
||||||
Some(remote_storage::DownloadError::Cancelled) => {
|
|
||||||
Err(DownloadError::DownloadCancelled)
|
|
||||||
}
|
|
||||||
// FIXME: this is not embedding the error because historically it would had
|
|
||||||
// been output to compute, however that is no longer the case.
|
|
||||||
_ => Err(DownloadError::DownloadFailed),
|
|
||||||
}
|
|
||||||
}
|
}
|
||||||
|
Ok(Err(_)) => Err(DownloadError::DownloadFailed),
|
||||||
Err(_gone) => Err(DownloadError::DownloadCancelled),
|
Err(_gone) => Err(DownloadError::DownloadCancelled),
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
@@ -1118,7 +1109,7 @@ impl LayerInner {
|
|||||||
timeline: Arc<Timeline>,
|
timeline: Arc<Timeline>,
|
||||||
permit: heavier_once_cell::InitPermit,
|
permit: heavier_once_cell::InitPermit,
|
||||||
ctx: &RequestContext,
|
ctx: &RequestContext,
|
||||||
) -> anyhow::Result<Arc<DownloadedLayer>> {
|
) -> Result<Arc<DownloadedLayer>, remote_storage::DownloadError> {
|
||||||
let result = timeline
|
let result = timeline
|
||||||
.remote_client
|
.remote_client
|
||||||
.download_layer_file(
|
.download_layer_file(
|
||||||
@@ -1694,6 +1685,7 @@ impl DownloadedLayer {
|
|||||||
lsn,
|
lsn,
|
||||||
summary,
|
summary,
|
||||||
Some(owner.conf.max_vectored_read_bytes),
|
Some(owner.conf.max_vectored_read_bytes),
|
||||||
|
owner.conf.image_compression.allow_decompression(),
|
||||||
ctx,
|
ctx,
|
||||||
)
|
)
|
||||||
.await
|
.await
|
||||||
|
|||||||
@@ -14,6 +14,7 @@ use anyhow::{anyhow, bail, ensure, Context, Result};
|
|||||||
use arc_swap::ArcSwap;
|
use arc_swap::ArcSwap;
|
||||||
use bytes::Bytes;
|
use bytes::Bytes;
|
||||||
use camino::Utf8Path;
|
use camino::Utf8Path;
|
||||||
|
use chrono::{DateTime, Utc};
|
||||||
use enumset::EnumSet;
|
use enumset::EnumSet;
|
||||||
use fail::fail_point;
|
use fail::fail_point;
|
||||||
use once_cell::sync::Lazy;
|
use once_cell::sync::Lazy;
|
||||||
@@ -65,7 +66,6 @@ use std::{
|
|||||||
ops::{Deref, Range},
|
ops::{Deref, Range},
|
||||||
};
|
};
|
||||||
|
|
||||||
use crate::metrics::GetKind;
|
|
||||||
use crate::pgdatadir_mapping::MAX_AUX_FILE_V2_DELTAS;
|
use crate::pgdatadir_mapping::MAX_AUX_FILE_V2_DELTAS;
|
||||||
use crate::{
|
use crate::{
|
||||||
aux_file::AuxFileSizeEstimator,
|
aux_file::AuxFileSizeEstimator,
|
||||||
@@ -90,6 +90,10 @@ use crate::{
|
|||||||
use crate::{
|
use crate::{
|
||||||
disk_usage_eviction_task::EvictionCandidate, tenant::storage_layer::delta_layer::DeltaEntry,
|
disk_usage_eviction_task::EvictionCandidate, tenant::storage_layer::delta_layer::DeltaEntry,
|
||||||
};
|
};
|
||||||
|
use crate::{
|
||||||
|
l0_flush::{self, L0FlushGlobalState},
|
||||||
|
metrics::GetKind,
|
||||||
|
};
|
||||||
use crate::{
|
use crate::{
|
||||||
metrics::ScanLatencyOngoingRecording, tenant::timeline::logical_size::CurrentLogicalSize,
|
metrics::ScanLatencyOngoingRecording, tenant::timeline::logical_size::CurrentLogicalSize,
|
||||||
};
|
};
|
||||||
@@ -208,6 +212,7 @@ pub struct TimelineResources {
|
|||||||
pub timeline_get_throttle: Arc<
|
pub timeline_get_throttle: Arc<
|
||||||
crate::tenant::throttle::Throttle<&'static crate::metrics::tenant_throttling::TimelineGet>,
|
crate::tenant::throttle::Throttle<&'static crate::metrics::tenant_throttling::TimelineGet>,
|
||||||
>,
|
>,
|
||||||
|
pub l0_flush_global_state: l0_flush::L0FlushGlobalState,
|
||||||
}
|
}
|
||||||
|
|
||||||
pub(crate) struct AuxFilesState {
|
pub(crate) struct AuxFilesState {
|
||||||
@@ -360,6 +365,7 @@ pub struct Timeline {
|
|||||||
repartition_threshold: u64,
|
repartition_threshold: u64,
|
||||||
|
|
||||||
last_image_layer_creation_check_at: AtomicLsn,
|
last_image_layer_creation_check_at: AtomicLsn,
|
||||||
|
last_image_layer_creation_check_instant: std::sync::Mutex<Option<Instant>>,
|
||||||
|
|
||||||
/// Current logical size of the "datadir", at the last LSN.
|
/// Current logical size of the "datadir", at the last LSN.
|
||||||
current_logical_size: LogicalSize,
|
current_logical_size: LogicalSize,
|
||||||
@@ -433,6 +439,8 @@ pub struct Timeline {
|
|||||||
/// in the future, add `extra_test_sparse_keyspace` if necessary.
|
/// in the future, add `extra_test_sparse_keyspace` if necessary.
|
||||||
#[cfg(test)]
|
#[cfg(test)]
|
||||||
pub(crate) extra_test_dense_keyspace: ArcSwap<KeySpace>,
|
pub(crate) extra_test_dense_keyspace: ArcSwap<KeySpace>,
|
||||||
|
|
||||||
|
pub(crate) l0_flush_global_state: L0FlushGlobalState,
|
||||||
}
|
}
|
||||||
|
|
||||||
pub struct WalReceiverInfo {
|
pub struct WalReceiverInfo {
|
||||||
@@ -457,6 +465,9 @@ pub(crate) struct GcInfo {
|
|||||||
|
|
||||||
/// Leases granted to particular LSNs.
|
/// Leases granted to particular LSNs.
|
||||||
pub(crate) leases: BTreeMap<Lsn, LsnLease>,
|
pub(crate) leases: BTreeMap<Lsn, LsnLease>,
|
||||||
|
|
||||||
|
/// Whether our branch point is within our ancestor's PITR interval (for cost estimation)
|
||||||
|
pub(crate) within_ancestor_pitr: bool,
|
||||||
}
|
}
|
||||||
|
|
||||||
impl GcInfo {
|
impl GcInfo {
|
||||||
@@ -845,6 +856,18 @@ impl Timeline {
|
|||||||
.map(|ancestor| ancestor.timeline_id)
|
.map(|ancestor| ancestor.timeline_id)
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// Get the bytes written since the PITR cutoff on this branch, and
|
||||||
|
/// whether this branch's ancestor_lsn is within its parent's PITR.
|
||||||
|
pub(crate) fn get_pitr_history_stats(&self) -> (u64, bool) {
|
||||||
|
let gc_info = self.gc_info.read().unwrap();
|
||||||
|
let history = self
|
||||||
|
.get_last_record_lsn()
|
||||||
|
.checked_sub(gc_info.cutoffs.pitr)
|
||||||
|
.unwrap_or(Lsn(0))
|
||||||
|
.0;
|
||||||
|
(history, gc_info.within_ancestor_pitr)
|
||||||
|
}
|
||||||
|
|
||||||
/// Lock and get timeline's GC cutoff
|
/// Lock and get timeline's GC cutoff
|
||||||
pub(crate) fn get_latest_gc_cutoff_lsn(&self) -> RcuReadGuard<Lsn> {
|
pub(crate) fn get_latest_gc_cutoff_lsn(&self) -> RcuReadGuard<Lsn> {
|
||||||
self.latest_gc_cutoff_lsn.read()
|
self.latest_gc_cutoff_lsn.read()
|
||||||
@@ -996,6 +1019,7 @@ impl Timeline {
|
|||||||
}
|
}
|
||||||
|
|
||||||
pub(crate) const MAX_GET_VECTORED_KEYS: u64 = 32;
|
pub(crate) const MAX_GET_VECTORED_KEYS: u64 = 32;
|
||||||
|
pub(crate) const VEC_GET_LAYERS_VISITED_WARN_THRESH: f64 = 512.0;
|
||||||
|
|
||||||
/// Look up multiple page versions at a given LSN
|
/// Look up multiple page versions at a given LSN
|
||||||
///
|
///
|
||||||
@@ -1228,7 +1252,7 @@ impl Timeline {
|
|||||||
let get_data_timer = crate::metrics::GET_RECONSTRUCT_DATA_TIME
|
let get_data_timer = crate::metrics::GET_RECONSTRUCT_DATA_TIME
|
||||||
.for_get_kind(get_kind)
|
.for_get_kind(get_kind)
|
||||||
.start_timer();
|
.start_timer();
|
||||||
self.get_vectored_reconstruct_data(keyspace, lsn, reconstruct_state, ctx)
|
self.get_vectored_reconstruct_data(keyspace.clone(), lsn, reconstruct_state, ctx)
|
||||||
.await?;
|
.await?;
|
||||||
get_data_timer.stop_and_record();
|
get_data_timer.stop_and_record();
|
||||||
|
|
||||||
@@ -1258,11 +1282,25 @@ impl Timeline {
|
|||||||
// (this is a requirement, not a bug). Skip updating the metric in these cases
|
// (this is a requirement, not a bug). Skip updating the metric in these cases
|
||||||
// to avoid infinite results.
|
// to avoid infinite results.
|
||||||
if !results.is_empty() {
|
if !results.is_empty() {
|
||||||
|
let avg = layers_visited as f64 / results.len() as f64;
|
||||||
|
if avg >= Self::VEC_GET_LAYERS_VISITED_WARN_THRESH {
|
||||||
|
use utils::rate_limit::RateLimit;
|
||||||
|
static LOGGED: Lazy<Mutex<RateLimit>> =
|
||||||
|
Lazy::new(|| Mutex::new(RateLimit::new(Duration::from_secs(60))));
|
||||||
|
let mut rate_limit = LOGGED.lock().unwrap();
|
||||||
|
rate_limit.call(|| {
|
||||||
|
tracing::info!(
|
||||||
|
shard_id = %self.tenant_shard_id.shard_slug(),
|
||||||
|
lsn = %lsn,
|
||||||
|
"Vectored read for {} visited {} layers on average per key and {} in total. {}/{} pages were returned",
|
||||||
|
keyspace, avg, layers_visited, results.len(), keyspace.total_raw_size());
|
||||||
|
});
|
||||||
|
}
|
||||||
|
|
||||||
// Note that this is an approximation. Tracking the exact number of layers visited
|
// Note that this is an approximation. Tracking the exact number of layers visited
|
||||||
// per key requires virtually unbounded memory usage and is inefficient
|
// per key requires virtually unbounded memory usage and is inefficient
|
||||||
// (i.e. segment tree tracking each range queried from a layer)
|
// (i.e. segment tree tracking each range queried from a layer)
|
||||||
crate::metrics::VEC_READ_NUM_LAYERS_VISITED
|
crate::metrics::VEC_READ_NUM_LAYERS_VISITED.observe(avg);
|
||||||
.observe(layers_visited as f64 / results.len() as f64);
|
|
||||||
}
|
}
|
||||||
|
|
||||||
Ok(results)
|
Ok(results)
|
||||||
@@ -1554,7 +1592,13 @@ impl Timeline {
|
|||||||
let existing_lease = occupied.get_mut();
|
let existing_lease = occupied.get_mut();
|
||||||
if valid_until > existing_lease.valid_until {
|
if valid_until > existing_lease.valid_until {
|
||||||
existing_lease.valid_until = valid_until;
|
existing_lease.valid_until = valid_until;
|
||||||
|
let dt: DateTime<Utc> = valid_until.into();
|
||||||
|
info!("lease extended to {}", dt);
|
||||||
|
} else {
|
||||||
|
let dt: DateTime<Utc> = existing_lease.valid_until.into();
|
||||||
|
info!("existing lease covers greater length, valid until {}", dt);
|
||||||
}
|
}
|
||||||
|
|
||||||
existing_lease.clone()
|
existing_lease.clone()
|
||||||
} else {
|
} else {
|
||||||
// Reject already GC-ed LSN (lsn < latest_gc_cutoff)
|
// Reject already GC-ed LSN (lsn < latest_gc_cutoff)
|
||||||
@@ -1563,6 +1607,8 @@ impl Timeline {
|
|||||||
bail!("tried to request a page version that was garbage collected. requested at {} gc cutoff {}", lsn, *latest_gc_cutoff_lsn);
|
bail!("tried to request a page version that was garbage collected. requested at {} gc cutoff {}", lsn, *latest_gc_cutoff_lsn);
|
||||||
}
|
}
|
||||||
|
|
||||||
|
let dt: DateTime<Utc> = valid_until.into();
|
||||||
|
info!("lease created, valid until {}", dt);
|
||||||
entry.or_insert(LsnLease { valid_until }).clone()
|
entry.or_insert(LsnLease { valid_until }).clone()
|
||||||
}
|
}
|
||||||
};
|
};
|
||||||
@@ -2339,6 +2385,7 @@ impl Timeline {
|
|||||||
)),
|
)),
|
||||||
repartition_threshold: 0,
|
repartition_threshold: 0,
|
||||||
last_image_layer_creation_check_at: AtomicLsn::new(0),
|
last_image_layer_creation_check_at: AtomicLsn::new(0),
|
||||||
|
last_image_layer_creation_check_instant: Mutex::new(None),
|
||||||
|
|
||||||
last_received_wal: Mutex::new(None),
|
last_received_wal: Mutex::new(None),
|
||||||
rel_size_cache: RwLock::new(RelSizeCache {
|
rel_size_cache: RwLock::new(RelSizeCache {
|
||||||
@@ -2376,6 +2423,8 @@ impl Timeline {
|
|||||||
|
|
||||||
#[cfg(test)]
|
#[cfg(test)]
|
||||||
extra_test_dense_keyspace: ArcSwap::new(Arc::new(KeySpace::default())),
|
extra_test_dense_keyspace: ArcSwap::new(Arc::new(KeySpace::default())),
|
||||||
|
|
||||||
|
l0_flush_global_state: resources.l0_flush_global_state,
|
||||||
};
|
};
|
||||||
result.repartition_threshold =
|
result.repartition_threshold =
|
||||||
result.get_checkpoint_distance() / REPARTITION_FREQ_IN_CHECKPOINT_DISTANCE;
|
result.get_checkpoint_distance() / REPARTITION_FREQ_IN_CHECKPOINT_DISTANCE;
|
||||||
@@ -4417,6 +4466,58 @@ impl Timeline {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// Predicate function which indicates whether we should check if new image layers
|
||||||
|
/// are required. Since checking if new image layers are required is expensive in
|
||||||
|
/// terms of CPU, we only do it in the following cases:
|
||||||
|
/// 1. If the timeline has ingested sufficient WAL to justify the cost
|
||||||
|
/// 2. If enough time has passed since the last check
|
||||||
|
/// 2.1. For large tenants, we wish to perform the check more often since they
|
||||||
|
/// suffer from the lack of image layers
|
||||||
|
/// 2.2. For small tenants (that can mostly fit in RAM), we use a much longer interval
|
||||||
|
fn should_check_if_image_layers_required(self: &Arc<Timeline>, lsn: Lsn) -> bool {
|
||||||
|
const LARGE_TENANT_THRESHOLD: u64 = 2 * 1024 * 1024 * 1024;
|
||||||
|
|
||||||
|
let last_checks_at = self.last_image_layer_creation_check_at.load();
|
||||||
|
let distance = lsn
|
||||||
|
.checked_sub(last_checks_at)
|
||||||
|
.expect("Attempt to compact with LSN going backwards");
|
||||||
|
let min_distance =
|
||||||
|
self.get_image_layer_creation_check_threshold() as u64 * self.get_checkpoint_distance();
|
||||||
|
|
||||||
|
let distance_based_decision = distance.0 >= min_distance;
|
||||||
|
|
||||||
|
let mut time_based_decision = false;
|
||||||
|
let mut last_check_instant = self.last_image_layer_creation_check_instant.lock().unwrap();
|
||||||
|
if let CurrentLogicalSize::Exact(logical_size) = self.current_logical_size.current_size() {
|
||||||
|
let check_required_after = if Into::<u64>::into(&logical_size) >= LARGE_TENANT_THRESHOLD
|
||||||
|
{
|
||||||
|
self.get_checkpoint_timeout()
|
||||||
|
} else {
|
||||||
|
Duration::from_secs(3600 * 48)
|
||||||
|
};
|
||||||
|
|
||||||
|
time_based_decision = match *last_check_instant {
|
||||||
|
Some(last_check) => {
|
||||||
|
let elapsed = last_check.elapsed();
|
||||||
|
elapsed >= check_required_after
|
||||||
|
}
|
||||||
|
None => true,
|
||||||
|
};
|
||||||
|
}
|
||||||
|
|
||||||
|
// Do the expensive delta layer counting only if this timeline has ingested sufficient
|
||||||
|
// WAL since the last check or a checkpoint timeout interval has elapsed since the last
|
||||||
|
// check.
|
||||||
|
let decision = distance_based_decision || time_based_decision;
|
||||||
|
|
||||||
|
if decision {
|
||||||
|
self.last_image_layer_creation_check_at.store(lsn);
|
||||||
|
*last_check_instant = Some(Instant::now());
|
||||||
|
}
|
||||||
|
|
||||||
|
decision
|
||||||
|
}
|
||||||
|
|
||||||
#[tracing::instrument(skip_all, fields(%lsn, %mode))]
|
#[tracing::instrument(skip_all, fields(%lsn, %mode))]
|
||||||
async fn create_image_layers(
|
async fn create_image_layers(
|
||||||
self: &Arc<Timeline>,
|
self: &Arc<Timeline>,
|
||||||
@@ -4439,22 +4540,7 @@ impl Timeline {
|
|||||||
// image layers <100000000..100000099> and <200000000..200000199> are not completely covering it.
|
// image layers <100000000..100000099> and <200000000..200000199> are not completely covering it.
|
||||||
let mut start = Key::MIN;
|
let mut start = Key::MIN;
|
||||||
|
|
||||||
let check_for_image_layers = {
|
let check_for_image_layers = self.should_check_if_image_layers_required(lsn);
|
||||||
let last_checks_at = self.last_image_layer_creation_check_at.load();
|
|
||||||
let distance = lsn
|
|
||||||
.checked_sub(last_checks_at)
|
|
||||||
.expect("Attempt to compact with LSN going backwards");
|
|
||||||
let min_distance = self.get_image_layer_creation_check_threshold() as u64
|
|
||||||
* self.get_checkpoint_distance();
|
|
||||||
|
|
||||||
// Skip the expensive delta layer counting if this timeline has not ingested sufficient
|
|
||||||
// WAL since the last check.
|
|
||||||
distance.0 >= min_distance
|
|
||||||
};
|
|
||||||
|
|
||||||
if check_for_image_layers {
|
|
||||||
self.last_image_layer_creation_check_at.store(lsn);
|
|
||||||
}
|
|
||||||
|
|
||||||
for partition in partitioning.parts.iter() {
|
for partition in partitioning.parts.iter() {
|
||||||
let img_range = start..partition.ranges.last().unwrap().end;
|
let img_range = start..partition.ranges.last().unwrap().end;
|
||||||
@@ -4711,6 +4797,42 @@ impl DurationRecorder {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// Descriptor for a delta layer used in testing infra. The start/end key/lsn range of the
|
||||||
|
/// delta layer might be different from the min/max key/lsn in the delta layer. Therefore,
|
||||||
|
/// the layer descriptor requires the user to provide the ranges, which should cover all
|
||||||
|
/// keys specified in the `data` field.
|
||||||
|
#[cfg(test)]
|
||||||
|
pub struct DeltaLayerTestDesc {
|
||||||
|
pub lsn_range: Range<Lsn>,
|
||||||
|
pub key_range: Range<Key>,
|
||||||
|
pub data: Vec<(Key, Lsn, Value)>,
|
||||||
|
}
|
||||||
|
|
||||||
|
#[cfg(test)]
|
||||||
|
impl DeltaLayerTestDesc {
|
||||||
|
#[allow(dead_code)]
|
||||||
|
pub fn new(lsn_range: Range<Lsn>, key_range: Range<Key>, data: Vec<(Key, Lsn, Value)>) -> Self {
|
||||||
|
Self {
|
||||||
|
lsn_range,
|
||||||
|
key_range,
|
||||||
|
data,
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
pub fn new_with_inferred_key_range(
|
||||||
|
lsn_range: Range<Lsn>,
|
||||||
|
data: Vec<(Key, Lsn, Value)>,
|
||||||
|
) -> Self {
|
||||||
|
let key_min = data.iter().map(|(key, _, _)| key).min().unwrap();
|
||||||
|
let key_max = data.iter().map(|(key, _, _)| key).max().unwrap();
|
||||||
|
Self {
|
||||||
|
key_range: (*key_min)..(key_max.next()),
|
||||||
|
lsn_range,
|
||||||
|
data,
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
impl Timeline {
|
impl Timeline {
|
||||||
async fn finish_compact_batch(
|
async fn finish_compact_batch(
|
||||||
self: &Arc<Self>,
|
self: &Arc<Self>,
|
||||||
@@ -5511,37 +5633,65 @@ impl Timeline {
|
|||||||
#[cfg(test)]
|
#[cfg(test)]
|
||||||
pub(super) async fn force_create_delta_layer(
|
pub(super) async fn force_create_delta_layer(
|
||||||
self: &Arc<Timeline>,
|
self: &Arc<Timeline>,
|
||||||
mut deltas: Vec<(Key, Lsn, Value)>,
|
mut deltas: DeltaLayerTestDesc,
|
||||||
check_start_lsn: Option<Lsn>,
|
check_start_lsn: Option<Lsn>,
|
||||||
ctx: &RequestContext,
|
ctx: &RequestContext,
|
||||||
) -> anyhow::Result<()> {
|
) -> anyhow::Result<()> {
|
||||||
let last_record_lsn = self.get_last_record_lsn();
|
let last_record_lsn = self.get_last_record_lsn();
|
||||||
deltas.sort_unstable_by(|(ka, la, _), (kb, lb, _)| (ka, la).cmp(&(kb, lb)));
|
deltas
|
||||||
let min_key = *deltas.first().map(|(k, _, _)| k).unwrap();
|
.data
|
||||||
let end_key = deltas.last().map(|(k, _, _)| k).unwrap().next();
|
.sort_unstable_by(|(ka, la, _), (kb, lb, _)| (ka, la).cmp(&(kb, lb)));
|
||||||
let min_lsn = *deltas.iter().map(|(_, lsn, _)| lsn).min().unwrap();
|
assert!(deltas.data.first().unwrap().0 >= deltas.key_range.start);
|
||||||
let max_lsn = *deltas.iter().map(|(_, lsn, _)| lsn).max().unwrap();
|
assert!(deltas.data.last().unwrap().0 < deltas.key_range.end);
|
||||||
|
for (_, lsn, _) in &deltas.data {
|
||||||
|
assert!(deltas.lsn_range.start <= *lsn && *lsn < deltas.lsn_range.end);
|
||||||
|
}
|
||||||
assert!(
|
assert!(
|
||||||
max_lsn <= last_record_lsn,
|
deltas.lsn_range.end <= last_record_lsn,
|
||||||
"advance last record lsn before inserting a layer, max_lsn={max_lsn}, last_record_lsn={last_record_lsn}"
|
"advance last record lsn before inserting a layer, end_lsn={}, last_record_lsn={}",
|
||||||
|
deltas.lsn_range.end,
|
||||||
|
last_record_lsn
|
||||||
);
|
);
|
||||||
let end_lsn = Lsn(max_lsn.0 + 1);
|
|
||||||
if let Some(check_start_lsn) = check_start_lsn {
|
if let Some(check_start_lsn) = check_start_lsn {
|
||||||
assert!(min_lsn >= check_start_lsn);
|
assert!(deltas.lsn_range.start >= check_start_lsn);
|
||||||
|
}
|
||||||
|
// check if the delta layer does not violate the LSN invariant, the legacy compaction should always produce a batch of
|
||||||
|
// layers of the same start/end LSN, and so should the force inserted layer
|
||||||
|
{
|
||||||
|
/// Checks if a overlaps with b, assume a/b = [start, end).
|
||||||
|
pub fn overlaps_with<T: Ord>(a: &Range<T>, b: &Range<T>) -> bool {
|
||||||
|
!(a.end <= b.start || b.end <= a.start)
|
||||||
|
}
|
||||||
|
|
||||||
|
let guard = self.layers.read().await;
|
||||||
|
for layer in guard.layer_map().iter_historic_layers() {
|
||||||
|
if layer.is_delta()
|
||||||
|
&& overlaps_with(&layer.lsn_range, &deltas.lsn_range)
|
||||||
|
&& layer.lsn_range != deltas.lsn_range
|
||||||
|
{
|
||||||
|
// If a delta layer overlaps with another delta layer AND their LSN range is not the same, panic
|
||||||
|
panic!(
|
||||||
|
"inserted layer violates delta layer LSN invariant: current_lsn_range={}..{}, conflict_lsn_range={}..{}",
|
||||||
|
deltas.lsn_range.start, deltas.lsn_range.end, layer.lsn_range.start, layer.lsn_range.end
|
||||||
|
);
|
||||||
|
}
|
||||||
|
}
|
||||||
}
|
}
|
||||||
let mut delta_layer_writer = DeltaLayerWriter::new(
|
let mut delta_layer_writer = DeltaLayerWriter::new(
|
||||||
self.conf,
|
self.conf,
|
||||||
self.timeline_id,
|
self.timeline_id,
|
||||||
self.tenant_shard_id,
|
self.tenant_shard_id,
|
||||||
min_key,
|
deltas.key_range.start,
|
||||||
min_lsn..end_lsn,
|
deltas.lsn_range,
|
||||||
ctx,
|
ctx,
|
||||||
)
|
)
|
||||||
.await?;
|
.await?;
|
||||||
for (key, lsn, val) in deltas {
|
for (key, lsn, val) in deltas.data {
|
||||||
delta_layer_writer.put_value(key, lsn, val, ctx).await?;
|
delta_layer_writer.put_value(key, lsn, val, ctx).await?;
|
||||||
}
|
}
|
||||||
let delta_layer = delta_layer_writer.finish(end_key, self, ctx).await?;
|
let delta_layer = delta_layer_writer
|
||||||
|
.finish(deltas.key_range.end, self, ctx)
|
||||||
|
.await?;
|
||||||
|
|
||||||
{
|
{
|
||||||
let mut guard = self.layers.write().await;
|
let mut guard = self.layers.write().await;
|
||||||
|
|||||||
@@ -272,6 +272,7 @@ impl DeleteTimelineFlow {
|
|||||||
TimelineResources {
|
TimelineResources {
|
||||||
remote_client,
|
remote_client,
|
||||||
timeline_get_throttle: tenant.timeline_get_throttle.clone(),
|
timeline_get_throttle: tenant.timeline_get_throttle.clone(),
|
||||||
|
l0_flush_global_state: tenant.l0_flush_global_state.clone(),
|
||||||
},
|
},
|
||||||
// Important. We dont pass ancestor above because it can be missing.
|
// Important. We dont pass ancestor above because it can be missing.
|
||||||
// Thus we need to skip the validation here.
|
// Thus we need to skip the validation here.
|
||||||
|
|||||||
@@ -26,7 +26,7 @@ use tracing::{debug, error, info, trace, warn, Instrument};
|
|||||||
use super::TaskStateUpdate;
|
use super::TaskStateUpdate;
|
||||||
use crate::{
|
use crate::{
|
||||||
context::RequestContext,
|
context::RequestContext,
|
||||||
metrics::{LIVE_CONNECTIONS_COUNT, WALRECEIVER_STARTED_CONNECTIONS, WAL_INGEST},
|
metrics::{LIVE_CONNECTIONS, WALRECEIVER_STARTED_CONNECTIONS, WAL_INGEST},
|
||||||
task_mgr::TaskKind,
|
task_mgr::TaskKind,
|
||||||
task_mgr::WALRECEIVER_RUNTIME,
|
task_mgr::WALRECEIVER_RUNTIME,
|
||||||
tenant::{debug_assert_current_span_has_tenant_and_timeline_id, Timeline, WalReceiverInfo},
|
tenant::{debug_assert_current_span_has_tenant_and_timeline_id, Timeline, WalReceiverInfo},
|
||||||
@@ -208,14 +208,9 @@ pub(super) async fn handle_walreceiver_connection(
|
|||||||
.instrument(tracing::info_span!("poller")),
|
.instrument(tracing::info_span!("poller")),
|
||||||
);
|
);
|
||||||
|
|
||||||
// Immediately increment the gauge, then create a job to decrement it on task exit.
|
let _guard = LIVE_CONNECTIONS
|
||||||
// One of the pros of `defer!` is that this will *most probably*
|
.with_label_values(&["wal_receiver"])
|
||||||
// get called, even in presence of panics.
|
.guard();
|
||||||
let gauge = LIVE_CONNECTIONS_COUNT.with_label_values(&["wal_receiver"]);
|
|
||||||
gauge.inc();
|
|
||||||
scopeguard::defer! {
|
|
||||||
gauge.dec();
|
|
||||||
}
|
|
||||||
|
|
||||||
let identify = identify_system(&replication_client).await?;
|
let identify = identify_system(&replication_client).await?;
|
||||||
info!("{identify:?}");
|
info!("{identify:?}");
|
||||||
|
|||||||
@@ -20,6 +20,7 @@ use std::num::NonZeroUsize;
|
|||||||
|
|
||||||
use bytes::BytesMut;
|
use bytes::BytesMut;
|
||||||
use pageserver_api::key::Key;
|
use pageserver_api::key::Key;
|
||||||
|
use tokio_epoll_uring::BoundedBuf;
|
||||||
use utils::lsn::Lsn;
|
use utils::lsn::Lsn;
|
||||||
use utils::vec_map::VecMap;
|
use utils::vec_map::VecMap;
|
||||||
|
|
||||||
@@ -316,8 +317,9 @@ impl<'a> VectoredBlobReader<'a> {
|
|||||||
);
|
);
|
||||||
let buf = self
|
let buf = self
|
||||||
.file
|
.file
|
||||||
.read_exact_at_n(buf, read.start, read.size(), ctx)
|
.read_exact_at(buf.slice(0..read.size()), read.start, ctx)
|
||||||
.await?;
|
.await?
|
||||||
|
.into_inner();
|
||||||
|
|
||||||
let blobs_at = read.blobs_at.as_slice();
|
let blobs_at = read.blobs_at.as_slice();
|
||||||
let start_offset = blobs_at.first().expect("VectoredRead is never empty").0;
|
let start_offset = blobs_at.first().expect("VectoredRead is never empty").0;
|
||||||
|
|||||||
@@ -13,7 +13,7 @@
|
|||||||
use crate::context::RequestContext;
|
use crate::context::RequestContext;
|
||||||
use crate::metrics::{StorageIoOperation, STORAGE_IO_SIZE, STORAGE_IO_TIME_METRIC};
|
use crate::metrics::{StorageIoOperation, STORAGE_IO_SIZE, STORAGE_IO_TIME_METRIC};
|
||||||
|
|
||||||
use crate::page_cache::PageWriteGuard;
|
use crate::page_cache::{PageWriteGuard, PAGE_SZ};
|
||||||
use crate::tenant::TENANTS_SEGMENT_NAME;
|
use crate::tenant::TENANTS_SEGMENT_NAME;
|
||||||
use camino::{Utf8Path, Utf8PathBuf};
|
use camino::{Utf8Path, Utf8PathBuf};
|
||||||
use once_cell::sync::OnceCell;
|
use once_cell::sync::OnceCell;
|
||||||
@@ -48,6 +48,7 @@ pub(crate) mod owned_buffers_io {
|
|||||||
//! but for the time being we're proving out the primitives in the neon.git repo
|
//! but for the time being we're proving out the primitives in the neon.git repo
|
||||||
//! for faster iteration.
|
//! for faster iteration.
|
||||||
|
|
||||||
|
pub(crate) mod slice;
|
||||||
pub(crate) mod write;
|
pub(crate) mod write;
|
||||||
pub(crate) mod util {
|
pub(crate) mod util {
|
||||||
pub(crate) mod size_tracking_writer;
|
pub(crate) mod size_tracking_writer;
|
||||||
@@ -143,16 +144,17 @@ struct SlotInner {
|
|||||||
/// Impl of [`tokio_epoll_uring::IoBuf`] and [`tokio_epoll_uring::IoBufMut`] for [`PageWriteGuard`].
|
/// Impl of [`tokio_epoll_uring::IoBuf`] and [`tokio_epoll_uring::IoBufMut`] for [`PageWriteGuard`].
|
||||||
struct PageWriteGuardBuf {
|
struct PageWriteGuardBuf {
|
||||||
page: PageWriteGuard<'static>,
|
page: PageWriteGuard<'static>,
|
||||||
init_up_to: usize,
|
|
||||||
}
|
}
|
||||||
// Safety: the [`PageWriteGuard`] gives us exclusive ownership of the page cache slot,
|
// Safety: the [`PageWriteGuard`] gives us exclusive ownership of the page cache slot,
|
||||||
// and the location remains stable even if [`Self`] or the [`PageWriteGuard`] is moved.
|
// and the location remains stable even if [`Self`] or the [`PageWriteGuard`] is moved.
|
||||||
|
// Page cache pages are zero-initialized, so, wrt uninitialized memory we're good.
|
||||||
|
// (Page cache tracks separately whether the contents are valid, see `PageWriteGuard::mark_valid`.)
|
||||||
unsafe impl tokio_epoll_uring::IoBuf for PageWriteGuardBuf {
|
unsafe impl tokio_epoll_uring::IoBuf for PageWriteGuardBuf {
|
||||||
fn stable_ptr(&self) -> *const u8 {
|
fn stable_ptr(&self) -> *const u8 {
|
||||||
self.page.as_ptr()
|
self.page.as_ptr()
|
||||||
}
|
}
|
||||||
fn bytes_init(&self) -> usize {
|
fn bytes_init(&self) -> usize {
|
||||||
self.init_up_to
|
self.page.len()
|
||||||
}
|
}
|
||||||
fn bytes_total(&self) -> usize {
|
fn bytes_total(&self) -> usize {
|
||||||
self.page.len()
|
self.page.len()
|
||||||
@@ -166,8 +168,8 @@ unsafe impl tokio_epoll_uring::IoBufMut for PageWriteGuardBuf {
|
|||||||
}
|
}
|
||||||
|
|
||||||
unsafe fn set_init(&mut self, pos: usize) {
|
unsafe fn set_init(&mut self, pos: usize) {
|
||||||
|
// There shouldn't really be any reason to call this API since bytes_init() == bytes_total().
|
||||||
assert!(pos <= self.page.len());
|
assert!(pos <= self.page.len());
|
||||||
self.init_up_to = pos;
|
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -585,37 +587,37 @@ impl VirtualFile {
|
|||||||
Ok(self.pos)
|
Ok(self.pos)
|
||||||
}
|
}
|
||||||
|
|
||||||
pub async fn read_exact_at<B>(
|
/// Read the file contents in range `offset..(offset + slice.bytes_total())` into `slice[0..slice.bytes_total()]`.
|
||||||
|
///
|
||||||
|
/// The returned `Slice<Buf>` is equivalent to the input `slice`, i.e., it's the same view into the same buffer.
|
||||||
|
pub async fn read_exact_at<Buf>(
|
||||||
&self,
|
&self,
|
||||||
buf: B,
|
slice: Slice<Buf>,
|
||||||
offset: u64,
|
offset: u64,
|
||||||
ctx: &RequestContext,
|
ctx: &RequestContext,
|
||||||
) -> Result<B, Error>
|
) -> Result<Slice<Buf>, Error>
|
||||||
where
|
where
|
||||||
B: IoBufMut + Send,
|
Buf: IoBufMut + Send,
|
||||||
{
|
{
|
||||||
let (buf, res) = read_exact_at_impl(buf, offset, None, |buf, offset| {
|
let assert_we_return_original_bounds = if cfg!(debug_assertions) {
|
||||||
self.read_at(buf, offset, ctx)
|
Some((slice.stable_ptr() as usize, slice.bytes_total()))
|
||||||
})
|
} else {
|
||||||
.await;
|
None
|
||||||
res.map(|()| buf)
|
};
|
||||||
}
|
|
||||||
|
|
||||||
pub async fn read_exact_at_n<B>(
|
let original_bounds = slice.bounds();
|
||||||
&self,
|
let (buf, res) =
|
||||||
buf: B,
|
read_exact_at_impl(slice, offset, |buf, offset| self.read_at(buf, offset, ctx)).await;
|
||||||
offset: u64,
|
let res = res.map(|_| buf.slice(original_bounds));
|
||||||
count: usize,
|
|
||||||
ctx: &RequestContext,
|
if let Some(original_bounds) = assert_we_return_original_bounds {
|
||||||
) -> Result<B, Error>
|
if let Ok(slice) = &res {
|
||||||
where
|
let returned_bounds = (slice.stable_ptr() as usize, slice.bytes_total());
|
||||||
B: IoBufMut + Send,
|
assert_eq!(original_bounds, returned_bounds);
|
||||||
{
|
}
|
||||||
let (buf, res) = read_exact_at_impl(buf, offset, Some(count), |buf, offset| {
|
}
|
||||||
self.read_at(buf, offset, ctx)
|
|
||||||
})
|
res
|
||||||
.await;
|
|
||||||
res.map(|()| buf)
|
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Like [`Self::read_exact_at`] but for [`PageWriteGuard`].
|
/// Like [`Self::read_exact_at`] but for [`PageWriteGuard`].
|
||||||
@@ -625,13 +627,11 @@ impl VirtualFile {
|
|||||||
offset: u64,
|
offset: u64,
|
||||||
ctx: &RequestContext,
|
ctx: &RequestContext,
|
||||||
) -> Result<PageWriteGuard<'static>, Error> {
|
) -> Result<PageWriteGuard<'static>, Error> {
|
||||||
let buf = PageWriteGuardBuf {
|
let buf = PageWriteGuardBuf { page }.slice_full();
|
||||||
page,
|
debug_assert_eq!(buf.bytes_total(), PAGE_SZ);
|
||||||
init_up_to: 0,
|
self.read_exact_at(buf, offset, ctx)
|
||||||
};
|
.await
|
||||||
let res = self.read_exact_at(buf, offset, ctx).await;
|
.map(|slice| slice.into_inner().page)
|
||||||
res.map(|PageWriteGuardBuf { page, .. }| page)
|
|
||||||
.map_err(|e| Error::new(ErrorKind::Other, e))
|
|
||||||
}
|
}
|
||||||
|
|
||||||
// Copied from https://doc.rust-lang.org/1.72.0/src/std/os/unix/fs.rs.html#219-235
|
// Copied from https://doc.rust-lang.org/1.72.0/src/std/os/unix/fs.rs.html#219-235
|
||||||
@@ -722,14 +722,14 @@ impl VirtualFile {
|
|||||||
(buf, Ok(n))
|
(buf, Ok(n))
|
||||||
}
|
}
|
||||||
|
|
||||||
pub(crate) async fn read_at<B>(
|
pub(crate) async fn read_at<Buf>(
|
||||||
&self,
|
&self,
|
||||||
buf: B,
|
buf: tokio_epoll_uring::Slice<Buf>,
|
||||||
offset: u64,
|
offset: u64,
|
||||||
_ctx: &RequestContext, /* TODO: use for metrics: https://github.com/neondatabase/neon/issues/6107 */
|
_ctx: &RequestContext, /* TODO: use for metrics: https://github.com/neondatabase/neon/issues/6107 */
|
||||||
) -> (B, Result<usize, Error>)
|
) -> (tokio_epoll_uring::Slice<Buf>, Result<usize, Error>)
|
||||||
where
|
where
|
||||||
B: tokio_epoll_uring::BoundedBufMut + Send,
|
Buf: tokio_epoll_uring::IoBufMut + Send,
|
||||||
{
|
{
|
||||||
let file_guard = match self.lock_file().await {
|
let file_guard = match self.lock_file().await {
|
||||||
Ok(file_guard) => file_guard,
|
Ok(file_guard) => file_guard,
|
||||||
@@ -781,26 +781,16 @@ impl VirtualFile {
|
|||||||
}
|
}
|
||||||
|
|
||||||
// Adapted from https://doc.rust-lang.org/1.72.0/src/std/os/unix/fs.rs.html#117-135
|
// Adapted from https://doc.rust-lang.org/1.72.0/src/std/os/unix/fs.rs.html#117-135
|
||||||
pub async fn read_exact_at_impl<B, F, Fut>(
|
pub async fn read_exact_at_impl<Buf, F, Fut>(
|
||||||
buf: B,
|
mut buf: tokio_epoll_uring::Slice<Buf>,
|
||||||
mut offset: u64,
|
mut offset: u64,
|
||||||
count: Option<usize>,
|
|
||||||
mut read_at: F,
|
mut read_at: F,
|
||||||
) -> (B, std::io::Result<()>)
|
) -> (Buf, std::io::Result<()>)
|
||||||
where
|
where
|
||||||
B: IoBufMut + Send,
|
Buf: IoBufMut + Send,
|
||||||
F: FnMut(tokio_epoll_uring::Slice<B>, u64) -> Fut,
|
F: FnMut(tokio_epoll_uring::Slice<Buf>, u64) -> Fut,
|
||||||
Fut: std::future::Future<Output = (tokio_epoll_uring::Slice<B>, std::io::Result<usize>)>,
|
Fut: std::future::Future<Output = (tokio_epoll_uring::Slice<Buf>, std::io::Result<usize>)>,
|
||||||
{
|
{
|
||||||
let mut buf: tokio_epoll_uring::Slice<B> = match count {
|
|
||||||
Some(count) => {
|
|
||||||
assert!(count <= buf.bytes_total());
|
|
||||||
assert!(count > 0);
|
|
||||||
buf.slice(..count) // may include uninitialized memory
|
|
||||||
}
|
|
||||||
None => buf.slice_full(), // includes all the uninitialized memory
|
|
||||||
};
|
|
||||||
|
|
||||||
while buf.bytes_total() != 0 {
|
while buf.bytes_total() != 0 {
|
||||||
let res;
|
let res;
|
||||||
(buf, res) = read_at(buf, offset).await;
|
(buf, res) = read_at(buf, offset).await;
|
||||||
@@ -882,7 +872,7 @@ mod test_read_exact_at_impl {
|
|||||||
|
|
||||||
#[tokio::test]
|
#[tokio::test]
|
||||||
async fn test_basic() {
|
async fn test_basic() {
|
||||||
let buf = Vec::with_capacity(5);
|
let buf = Vec::with_capacity(5).slice_full();
|
||||||
let mock_read_at = Arc::new(tokio::sync::Mutex::new(MockReadAt {
|
let mock_read_at = Arc::new(tokio::sync::Mutex::new(MockReadAt {
|
||||||
expectations: VecDeque::from(vec![Expectation {
|
expectations: VecDeque::from(vec![Expectation {
|
||||||
offset: 0,
|
offset: 0,
|
||||||
@@ -890,7 +880,7 @@ mod test_read_exact_at_impl {
|
|||||||
result: Ok(vec![b'a', b'b', b'c', b'd', b'e']),
|
result: Ok(vec![b'a', b'b', b'c', b'd', b'e']),
|
||||||
}]),
|
}]),
|
||||||
}));
|
}));
|
||||||
let (buf, res) = read_exact_at_impl(buf, 0, None, |buf, offset| {
|
let (buf, res) = read_exact_at_impl(buf, 0, |buf, offset| {
|
||||||
let mock_read_at = Arc::clone(&mock_read_at);
|
let mock_read_at = Arc::clone(&mock_read_at);
|
||||||
async move { mock_read_at.lock().await.read_at(buf, offset).await }
|
async move { mock_read_at.lock().await.read_at(buf, offset).await }
|
||||||
})
|
})
|
||||||
@@ -899,33 +889,13 @@ mod test_read_exact_at_impl {
|
|||||||
assert_eq!(buf, vec![b'a', b'b', b'c', b'd', b'e']);
|
assert_eq!(buf, vec![b'a', b'b', b'c', b'd', b'e']);
|
||||||
}
|
}
|
||||||
|
|
||||||
#[tokio::test]
|
|
||||||
async fn test_with_count() {
|
|
||||||
let buf = Vec::with_capacity(5);
|
|
||||||
let mock_read_at = Arc::new(tokio::sync::Mutex::new(MockReadAt {
|
|
||||||
expectations: VecDeque::from(vec![Expectation {
|
|
||||||
offset: 0,
|
|
||||||
bytes_total: 3,
|
|
||||||
result: Ok(vec![b'a', b'b', b'c']),
|
|
||||||
}]),
|
|
||||||
}));
|
|
||||||
|
|
||||||
let (buf, res) = read_exact_at_impl(buf, 0, Some(3), |buf, offset| {
|
|
||||||
let mock_read_at = Arc::clone(&mock_read_at);
|
|
||||||
async move { mock_read_at.lock().await.read_at(buf, offset).await }
|
|
||||||
})
|
|
||||||
.await;
|
|
||||||
assert!(res.is_ok());
|
|
||||||
assert_eq!(buf, vec![b'a', b'b', b'c']);
|
|
||||||
}
|
|
||||||
|
|
||||||
#[tokio::test]
|
#[tokio::test]
|
||||||
async fn test_empty_buf_issues_no_syscall() {
|
async fn test_empty_buf_issues_no_syscall() {
|
||||||
let buf = Vec::new();
|
let buf = Vec::new().slice_full();
|
||||||
let mock_read_at = Arc::new(tokio::sync::Mutex::new(MockReadAt {
|
let mock_read_at = Arc::new(tokio::sync::Mutex::new(MockReadAt {
|
||||||
expectations: VecDeque::new(),
|
expectations: VecDeque::new(),
|
||||||
}));
|
}));
|
||||||
let (_buf, res) = read_exact_at_impl(buf, 0, None, |buf, offset| {
|
let (_buf, res) = read_exact_at_impl(buf, 0, |buf, offset| {
|
||||||
let mock_read_at = Arc::clone(&mock_read_at);
|
let mock_read_at = Arc::clone(&mock_read_at);
|
||||||
async move { mock_read_at.lock().await.read_at(buf, offset).await }
|
async move { mock_read_at.lock().await.read_at(buf, offset).await }
|
||||||
})
|
})
|
||||||
@@ -935,7 +905,7 @@ mod test_read_exact_at_impl {
|
|||||||
|
|
||||||
#[tokio::test]
|
#[tokio::test]
|
||||||
async fn test_two_read_at_calls_needed_until_buf_filled() {
|
async fn test_two_read_at_calls_needed_until_buf_filled() {
|
||||||
let buf = Vec::with_capacity(4);
|
let buf = Vec::with_capacity(4).slice_full();
|
||||||
let mock_read_at = Arc::new(tokio::sync::Mutex::new(MockReadAt {
|
let mock_read_at = Arc::new(tokio::sync::Mutex::new(MockReadAt {
|
||||||
expectations: VecDeque::from(vec![
|
expectations: VecDeque::from(vec![
|
||||||
Expectation {
|
Expectation {
|
||||||
@@ -950,7 +920,7 @@ mod test_read_exact_at_impl {
|
|||||||
},
|
},
|
||||||
]),
|
]),
|
||||||
}));
|
}));
|
||||||
let (buf, res) = read_exact_at_impl(buf, 0, None, |buf, offset| {
|
let (buf, res) = read_exact_at_impl(buf, 0, |buf, offset| {
|
||||||
let mock_read_at = Arc::clone(&mock_read_at);
|
let mock_read_at = Arc::clone(&mock_read_at);
|
||||||
async move { mock_read_at.lock().await.read_at(buf, offset).await }
|
async move { mock_read_at.lock().await.read_at(buf, offset).await }
|
||||||
})
|
})
|
||||||
@@ -961,7 +931,7 @@ mod test_read_exact_at_impl {
|
|||||||
|
|
||||||
#[tokio::test]
|
#[tokio::test]
|
||||||
async fn test_eof_before_buffer_full() {
|
async fn test_eof_before_buffer_full() {
|
||||||
let buf = Vec::with_capacity(3);
|
let buf = Vec::with_capacity(3).slice_full();
|
||||||
let mock_read_at = Arc::new(tokio::sync::Mutex::new(MockReadAt {
|
let mock_read_at = Arc::new(tokio::sync::Mutex::new(MockReadAt {
|
||||||
expectations: VecDeque::from(vec![
|
expectations: VecDeque::from(vec![
|
||||||
Expectation {
|
Expectation {
|
||||||
@@ -981,7 +951,7 @@ mod test_read_exact_at_impl {
|
|||||||
},
|
},
|
||||||
]),
|
]),
|
||||||
}));
|
}));
|
||||||
let (_buf, res) = read_exact_at_impl(buf, 0, None, |buf, offset| {
|
let (_buf, res) = read_exact_at_impl(buf, 0, |buf, offset| {
|
||||||
let mock_read_at = Arc::clone(&mock_read_at);
|
let mock_read_at = Arc::clone(&mock_read_at);
|
||||||
async move { mock_read_at.lock().await.read_at(buf, offset).await }
|
async move { mock_read_at.lock().await.read_at(buf, offset).await }
|
||||||
})
|
})
|
||||||
@@ -1051,27 +1021,29 @@ impl VirtualFile {
|
|||||||
ctx: &RequestContext,
|
ctx: &RequestContext,
|
||||||
) -> Result<crate::tenant::block_io::BlockLease<'_>, std::io::Error> {
|
) -> Result<crate::tenant::block_io::BlockLease<'_>, std::io::Error> {
|
||||||
use crate::page_cache::PAGE_SZ;
|
use crate::page_cache::PAGE_SZ;
|
||||||
let buf = vec![0; PAGE_SZ];
|
let slice = Vec::with_capacity(PAGE_SZ).slice_full();
|
||||||
let buf = self
|
assert_eq!(slice.bytes_total(), PAGE_SZ);
|
||||||
.read_exact_at(buf, blknum as u64 * (PAGE_SZ as u64), ctx)
|
let slice = self
|
||||||
|
.read_exact_at(slice, blknum as u64 * (PAGE_SZ as u64), ctx)
|
||||||
.await?;
|
.await?;
|
||||||
Ok(crate::tenant::block_io::BlockLease::Vec(buf))
|
Ok(crate::tenant::block_io::BlockLease::Vec(slice.into_inner()))
|
||||||
}
|
}
|
||||||
|
|
||||||
async fn read_to_end(&mut self, buf: &mut Vec<u8>, ctx: &RequestContext) -> Result<(), Error> {
|
async fn read_to_end(&mut self, buf: &mut Vec<u8>, ctx: &RequestContext) -> Result<(), Error> {
|
||||||
let mut tmp = vec![0; 128];
|
let mut tmp = vec![0; 128];
|
||||||
loop {
|
loop {
|
||||||
let res;
|
let slice = tmp.slice(..128);
|
||||||
(tmp, res) = self.read_at(tmp, self.pos, ctx).await;
|
let (slice, res) = self.read_at(slice, self.pos, ctx).await;
|
||||||
match res {
|
match res {
|
||||||
Ok(0) => return Ok(()),
|
Ok(0) => return Ok(()),
|
||||||
Ok(n) => {
|
Ok(n) => {
|
||||||
self.pos += n as u64;
|
self.pos += n as u64;
|
||||||
buf.extend_from_slice(&tmp[..n]);
|
buf.extend_from_slice(&slice[..n]);
|
||||||
}
|
}
|
||||||
Err(ref e) if e.kind() == std::io::ErrorKind::Interrupted => {}
|
Err(ref e) if e.kind() == std::io::ErrorKind::Interrupted => {}
|
||||||
Err(e) => return Err(e),
|
Err(e) => return Err(e),
|
||||||
}
|
}
|
||||||
|
tmp = slice.into_inner();
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
@@ -1185,6 +1157,7 @@ mod tests {
|
|||||||
use crate::task_mgr::TaskKind;
|
use crate::task_mgr::TaskKind;
|
||||||
|
|
||||||
use super::*;
|
use super::*;
|
||||||
|
use owned_buffers_io::slice::SliceExt;
|
||||||
use rand::seq::SliceRandom;
|
use rand::seq::SliceRandom;
|
||||||
use rand::thread_rng;
|
use rand::thread_rng;
|
||||||
use rand::Rng;
|
use rand::Rng;
|
||||||
@@ -1206,13 +1179,16 @@ mod tests {
|
|||||||
impl MaybeVirtualFile {
|
impl MaybeVirtualFile {
|
||||||
async fn read_exact_at(
|
async fn read_exact_at(
|
||||||
&self,
|
&self,
|
||||||
mut buf: Vec<u8>,
|
mut slice: tokio_epoll_uring::Slice<Vec<u8>>,
|
||||||
offset: u64,
|
offset: u64,
|
||||||
ctx: &RequestContext,
|
ctx: &RequestContext,
|
||||||
) -> Result<Vec<u8>, Error> {
|
) -> Result<tokio_epoll_uring::Slice<Vec<u8>>, Error> {
|
||||||
match self {
|
match self {
|
||||||
MaybeVirtualFile::VirtualFile(file) => file.read_exact_at(buf, offset, ctx).await,
|
MaybeVirtualFile::VirtualFile(file) => file.read_exact_at(slice, offset, ctx).await,
|
||||||
MaybeVirtualFile::File(file) => file.read_exact_at(&mut buf, offset).map(|()| buf),
|
MaybeVirtualFile::File(file) => {
|
||||||
|
let rust_slice: &mut [u8] = slice.as_mut_rust_slice_full_zeroed();
|
||||||
|
file.read_exact_at(rust_slice, offset).map(|()| slice)
|
||||||
|
}
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
async fn write_all_at<B: BoundedBuf<Buf = Buf>, Buf: IoBuf + Send>(
|
async fn write_all_at<B: BoundedBuf<Buf = Buf>, Buf: IoBuf + Send>(
|
||||||
@@ -1286,9 +1262,12 @@ mod tests {
|
|||||||
len: usize,
|
len: usize,
|
||||||
ctx: &RequestContext,
|
ctx: &RequestContext,
|
||||||
) -> Result<String, Error> {
|
) -> Result<String, Error> {
|
||||||
let buf = vec![0; len];
|
let slice = Vec::with_capacity(len).slice_full();
|
||||||
let buf = self.read_exact_at(buf, pos, ctx).await?;
|
assert_eq!(slice.bytes_total(), len);
|
||||||
Ok(String::from_utf8(buf).unwrap())
|
let slice = self.read_exact_at(slice, pos, ctx).await?;
|
||||||
|
let vec = slice.into_inner();
|
||||||
|
assert_eq!(vec.len(), len);
|
||||||
|
Ok(String::from_utf8(vec).unwrap())
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -1507,7 +1486,11 @@ mod tests {
|
|||||||
let mut rng = rand::rngs::OsRng;
|
let mut rng = rand::rngs::OsRng;
|
||||||
for _ in 1..1000 {
|
for _ in 1..1000 {
|
||||||
let f = &files[rng.gen_range(0..files.len())];
|
let f = &files[rng.gen_range(0..files.len())];
|
||||||
buf = f.read_exact_at(buf, 0, &ctx).await.unwrap();
|
buf = f
|
||||||
|
.read_exact_at(buf.slice_full(), 0, &ctx)
|
||||||
|
.await
|
||||||
|
.unwrap()
|
||||||
|
.into_inner();
|
||||||
assert!(buf == SAMPLE);
|
assert!(buf == SAMPLE);
|
||||||
}
|
}
|
||||||
});
|
});
|
||||||
|
|||||||
@@ -107,7 +107,7 @@ use std::{
|
|||||||
sync::atomic::{AtomicU8, Ordering},
|
sync::atomic::{AtomicU8, Ordering},
|
||||||
};
|
};
|
||||||
|
|
||||||
use super::{FileGuard, Metadata};
|
use super::{owned_buffers_io::slice::SliceExt, FileGuard, Metadata};
|
||||||
|
|
||||||
#[cfg(target_os = "linux")]
|
#[cfg(target_os = "linux")]
|
||||||
fn epoll_uring_error_to_std(e: tokio_epoll_uring::Error<std::io::Error>) -> std::io::Error {
|
fn epoll_uring_error_to_std(e: tokio_epoll_uring::Error<std::io::Error>) -> std::io::Error {
|
||||||
@@ -120,38 +120,29 @@ fn epoll_uring_error_to_std(e: tokio_epoll_uring::Error<std::io::Error>) -> std:
|
|||||||
}
|
}
|
||||||
|
|
||||||
impl IoEngine {
|
impl IoEngine {
|
||||||
pub(super) async fn read_at<B>(
|
pub(super) async fn read_at<Buf>(
|
||||||
&self,
|
&self,
|
||||||
file_guard: FileGuard,
|
file_guard: FileGuard,
|
||||||
offset: u64,
|
offset: u64,
|
||||||
mut buf: B,
|
mut slice: tokio_epoll_uring::Slice<Buf>,
|
||||||
) -> ((FileGuard, B), std::io::Result<usize>)
|
) -> (
|
||||||
|
(FileGuard, tokio_epoll_uring::Slice<Buf>),
|
||||||
|
std::io::Result<usize>,
|
||||||
|
)
|
||||||
where
|
where
|
||||||
B: tokio_epoll_uring::BoundedBufMut + Send,
|
Buf: tokio_epoll_uring::IoBufMut + Send,
|
||||||
{
|
{
|
||||||
match self {
|
match self {
|
||||||
IoEngine::NotSet => panic!("not initialized"),
|
IoEngine::NotSet => panic!("not initialized"),
|
||||||
IoEngine::StdFs => {
|
IoEngine::StdFs => {
|
||||||
// SAFETY: `dst` only lives at most as long as this match arm, during which buf remains valid memory.
|
let rust_slice = slice.as_mut_rust_slice_full_zeroed();
|
||||||
let dst = unsafe {
|
let res = file_guard.with_std_file(|std_file| std_file.read_at(rust_slice, offset));
|
||||||
std::slice::from_raw_parts_mut(buf.stable_mut_ptr(), buf.bytes_total())
|
((file_guard, slice), res)
|
||||||
};
|
|
||||||
let res = file_guard.with_std_file(|std_file| std_file.read_at(dst, offset));
|
|
||||||
if let Ok(nbytes) = &res {
|
|
||||||
assert!(*nbytes <= buf.bytes_total());
|
|
||||||
// SAFETY: see above assertion
|
|
||||||
unsafe {
|
|
||||||
buf.set_init(*nbytes);
|
|
||||||
}
|
|
||||||
}
|
|
||||||
#[allow(dropping_references)]
|
|
||||||
drop(dst);
|
|
||||||
((file_guard, buf), res)
|
|
||||||
}
|
}
|
||||||
#[cfg(target_os = "linux")]
|
#[cfg(target_os = "linux")]
|
||||||
IoEngine::TokioEpollUring => {
|
IoEngine::TokioEpollUring => {
|
||||||
let system = tokio_epoll_uring_ext::thread_local_system().await;
|
let system = tokio_epoll_uring_ext::thread_local_system().await;
|
||||||
let (resources, res) = system.read(file_guard, offset, buf).await;
|
let (resources, res) = system.read(file_guard, offset, slice).await;
|
||||||
(resources, res.map_err(epoll_uring_error_to_std))
|
(resources, res.map_err(epoll_uring_error_to_std))
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -0,0 +1,121 @@
|
|||||||
|
use tokio_epoll_uring::BoundedBuf;
|
||||||
|
use tokio_epoll_uring::BoundedBufMut;
|
||||||
|
use tokio_epoll_uring::IoBufMut;
|
||||||
|
use tokio_epoll_uring::Slice;
|
||||||
|
|
||||||
|
pub(crate) trait SliceExt {
|
||||||
|
/// Get a `&mut[0..self.bytes_total()`] slice, for when you need to do borrow-based IO.
|
||||||
|
///
|
||||||
|
/// See the test case `test_slice_full_zeroed` for the difference to just doing `&slice[..]`
|
||||||
|
fn as_mut_rust_slice_full_zeroed(&mut self) -> &mut [u8];
|
||||||
|
}
|
||||||
|
|
||||||
|
impl<B> SliceExt for Slice<B>
|
||||||
|
where
|
||||||
|
B: IoBufMut,
|
||||||
|
{
|
||||||
|
#[inline(always)]
|
||||||
|
fn as_mut_rust_slice_full_zeroed(&mut self) -> &mut [u8] {
|
||||||
|
// zero-initialize the uninitialized parts of the buffer so we can create a Rust slice
|
||||||
|
//
|
||||||
|
// SAFETY: we own `slice`, don't write outside the bounds
|
||||||
|
unsafe {
|
||||||
|
let to_init = self.bytes_total() - self.bytes_init();
|
||||||
|
self.stable_mut_ptr()
|
||||||
|
.add(self.bytes_init())
|
||||||
|
.write_bytes(0, to_init);
|
||||||
|
self.set_init(self.bytes_total());
|
||||||
|
};
|
||||||
|
let bytes_total = self.bytes_total();
|
||||||
|
&mut self[0..bytes_total]
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
#[cfg(test)]
|
||||||
|
mod tests {
|
||||||
|
use std::io::Read;
|
||||||
|
|
||||||
|
use super::*;
|
||||||
|
use bytes::Buf;
|
||||||
|
use tokio_epoll_uring::Slice;
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn test_slice_full_zeroed() {
|
||||||
|
let make_fake_file = || bytes::BytesMut::from(&b"12345"[..]).reader();
|
||||||
|
|
||||||
|
// before we start the test, let's make sure we have a shared understanding of what slice_full does
|
||||||
|
{
|
||||||
|
let buf = Vec::with_capacity(3);
|
||||||
|
let slice: Slice<_> = buf.slice_full();
|
||||||
|
assert_eq!(slice.bytes_init(), 0);
|
||||||
|
assert_eq!(slice.bytes_total(), 3);
|
||||||
|
let rust_slice = &slice[..];
|
||||||
|
assert_eq!(
|
||||||
|
rust_slice.len(),
|
||||||
|
0,
|
||||||
|
"Slice only derefs to a &[u8] of the initialized part"
|
||||||
|
);
|
||||||
|
}
|
||||||
|
|
||||||
|
// and also let's establish a shared understanding of .slice()
|
||||||
|
{
|
||||||
|
let buf = Vec::with_capacity(3);
|
||||||
|
let slice: Slice<_> = buf.slice(0..2);
|
||||||
|
assert_eq!(slice.bytes_init(), 0);
|
||||||
|
assert_eq!(slice.bytes_total(), 2);
|
||||||
|
let rust_slice = &slice[..];
|
||||||
|
assert_eq!(
|
||||||
|
rust_slice.len(),
|
||||||
|
0,
|
||||||
|
"Slice only derefs to a &[u8] of the initialized part"
|
||||||
|
);
|
||||||
|
}
|
||||||
|
|
||||||
|
// the above leads to the easy mistake of using slice[..] for borrow-based IO like so:
|
||||||
|
{
|
||||||
|
let buf = Vec::with_capacity(3);
|
||||||
|
let mut slice: Slice<_> = buf.slice_full();
|
||||||
|
assert_eq!(slice[..].len(), 0);
|
||||||
|
let mut file = make_fake_file();
|
||||||
|
file.read_exact(&mut slice[..]).unwrap(); // one might think this reads 3 bytes but it reads 0
|
||||||
|
assert_eq!(&slice[..] as &[u8], &[][..] as &[u8]);
|
||||||
|
}
|
||||||
|
|
||||||
|
// With owned buffers IO like with VirtualFilem, you could totally
|
||||||
|
// pass in a `Slice` with bytes_init()=0 but bytes_total()=5
|
||||||
|
// and it will read 5 bytes into the slice, and return a slice that has bytes_init()=5.
|
||||||
|
{
|
||||||
|
// TODO: demo
|
||||||
|
}
|
||||||
|
|
||||||
|
//
|
||||||
|
// Ok, now that we have a shared understanding let's demo how to use the extension trait.
|
||||||
|
//
|
||||||
|
|
||||||
|
// slice_full()
|
||||||
|
{
|
||||||
|
let buf = Vec::with_capacity(3);
|
||||||
|
let mut slice: Slice<_> = buf.slice_full();
|
||||||
|
let rust_slice = slice.as_mut_rust_slice_full_zeroed();
|
||||||
|
assert_eq!(rust_slice.len(), 3);
|
||||||
|
assert_eq!(rust_slice, &[0, 0, 0]);
|
||||||
|
let mut file = make_fake_file();
|
||||||
|
file.read_exact(rust_slice).unwrap();
|
||||||
|
assert_eq!(rust_slice, b"123");
|
||||||
|
assert_eq!(&slice[..], b"123");
|
||||||
|
}
|
||||||
|
|
||||||
|
// .slice(..)
|
||||||
|
{
|
||||||
|
let buf = Vec::with_capacity(3);
|
||||||
|
let mut slice: Slice<_> = buf.slice(0..2);
|
||||||
|
let rust_slice = slice.as_mut_rust_slice_full_zeroed();
|
||||||
|
assert_eq!(rust_slice.len(), 2);
|
||||||
|
assert_eq!(rust_slice, &[0, 0]);
|
||||||
|
let mut file = make_fake_file();
|
||||||
|
file.read_exact(rust_slice).unwrap();
|
||||||
|
assert_eq!(rust_slice, b"12");
|
||||||
|
assert_eq!(&slice[..], b"12");
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
+55
-14
@@ -343,7 +343,33 @@ impl WalIngest {
|
|||||||
xlog_checkpoint.oldestActiveXid,
|
xlog_checkpoint.oldestActiveXid,
|
||||||
self.checkpoint.oldestActiveXid
|
self.checkpoint.oldestActiveXid
|
||||||
);
|
);
|
||||||
self.checkpoint.oldestActiveXid = xlog_checkpoint.oldestActiveXid;
|
|
||||||
|
// A shutdown checkpoint has `oldestActiveXid == InvalidTransactionid`,
|
||||||
|
// because at shutdown, all in-progress transactions will implicitly
|
||||||
|
// end. Postgres startup code knows that, and allows hot standby to start
|
||||||
|
// immediately from a shutdown checkpoint.
|
||||||
|
//
|
||||||
|
// In Neon, Postgres hot standby startup always behaves as if starting from
|
||||||
|
// an online checkpoint. It needs a valid `oldestActiveXid` value, so
|
||||||
|
// instead of overwriting self.checkpoint.oldestActiveXid with
|
||||||
|
// InvalidTransactionid from the checkpoint WAL record, update it to a
|
||||||
|
// proper value, knowing that there are no in-progress transactions at this
|
||||||
|
// point, except for prepared transactions.
|
||||||
|
//
|
||||||
|
// See also the neon code changes in the InitWalRecovery() function.
|
||||||
|
if xlog_checkpoint.oldestActiveXid == pg_constants::INVALID_TRANSACTION_ID
|
||||||
|
&& info == pg_constants::XLOG_CHECKPOINT_SHUTDOWN
|
||||||
|
{
|
||||||
|
let mut oldest_active_xid = self.checkpoint.nextXid.value as u32;
|
||||||
|
for xid in modification.tline.list_twophase_files(lsn, ctx).await? {
|
||||||
|
if (xid.wrapping_sub(oldest_active_xid) as i32) < 0 {
|
||||||
|
oldest_active_xid = xid;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
self.checkpoint.oldestActiveXid = oldest_active_xid;
|
||||||
|
} else {
|
||||||
|
self.checkpoint.oldestActiveXid = xlog_checkpoint.oldestActiveXid;
|
||||||
|
}
|
||||||
|
|
||||||
// Write a new checkpoint key-value pair on every checkpoint record, even
|
// Write a new checkpoint key-value pair on every checkpoint record, even
|
||||||
// if nothing really changed. Not strictly required, but it seems nice to
|
// if nothing really changed. Not strictly required, but it seems nice to
|
||||||
@@ -375,6 +401,7 @@ impl WalIngest {
|
|||||||
if info == pg_constants::XLOG_RUNNING_XACTS {
|
if info == pg_constants::XLOG_RUNNING_XACTS {
|
||||||
let xlrec = crate::walrecord::XlRunningXacts::decode(&mut buf);
|
let xlrec = crate::walrecord::XlRunningXacts::decode(&mut buf);
|
||||||
self.checkpoint.oldestActiveXid = xlrec.oldest_running_xid;
|
self.checkpoint.oldestActiveXid = xlrec.oldest_running_xid;
|
||||||
|
self.checkpoint_modified = true;
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
pg_constants::RM_REPLORIGIN_ID => {
|
pg_constants::RM_REPLORIGIN_ID => {
|
||||||
@@ -1277,13 +1304,10 @@ impl WalIngest {
|
|||||||
xlrec.pageno, xlrec.oldest_xid, xlrec.oldest_xid_db
|
xlrec.pageno, xlrec.oldest_xid, xlrec.oldest_xid_db
|
||||||
);
|
);
|
||||||
|
|
||||||
// Here we treat oldestXid and oldestXidDB
|
// In Postgres, oldestXid and oldestXidDB are updated in memory when the CLOG is
|
||||||
// differently from postgres redo routines.
|
// truncated, but a checkpoint record with the updated values isn't written until
|
||||||
// In postgres checkpoint.oldestXid lags behind xlrec.oldest_xid
|
// later. In Neon, a server can start at any LSN, not just on a checkpoint record,
|
||||||
// until checkpoint happens and updates the value.
|
// so we keep the oldestXid and oldestXidDB up-to-date.
|
||||||
// Here we can use the most recent value.
|
|
||||||
// It's just an optimization, though and can be deleted.
|
|
||||||
// TODO Figure out if there will be any issues with replica.
|
|
||||||
self.checkpoint.oldestXid = xlrec.oldest_xid;
|
self.checkpoint.oldestXid = xlrec.oldest_xid;
|
||||||
self.checkpoint.oldestXidDB = xlrec.oldest_xid_db;
|
self.checkpoint.oldestXidDB = xlrec.oldest_xid_db;
|
||||||
self.checkpoint_modified = true;
|
self.checkpoint_modified = true;
|
||||||
@@ -1384,14 +1408,31 @@ impl WalIngest {
|
|||||||
// Note: The multixact members can wrap around, even within one WAL record.
|
// Note: The multixact members can wrap around, even within one WAL record.
|
||||||
offset = offset.wrapping_add(n_this_page as u32);
|
offset = offset.wrapping_add(n_this_page as u32);
|
||||||
}
|
}
|
||||||
if xlrec.mid >= self.checkpoint.nextMulti {
|
let next_offset = offset;
|
||||||
self.checkpoint.nextMulti = xlrec.mid + 1;
|
assert!(xlrec.moff.wrapping_add(xlrec.nmembers) == next_offset);
|
||||||
self.checkpoint_modified = true;
|
|
||||||
}
|
// Update next-multi-xid and next-offset
|
||||||
if xlrec.moff + xlrec.nmembers > self.checkpoint.nextMultiOffset {
|
//
|
||||||
self.checkpoint.nextMultiOffset = xlrec.moff + xlrec.nmembers;
|
// NB: In PostgreSQL, the next-multi-xid stored in the control file is allowed to
|
||||||
|
// go to 0, and it's fixed up by skipping to FirstMultiXactId in functions that
|
||||||
|
// read it, like GetNewMultiXactId(). This is different from how nextXid is
|
||||||
|
// incremented! nextXid skips over < FirstNormalTransactionId when the the value
|
||||||
|
// is stored, so it's never 0 in a checkpoint.
|
||||||
|
//
|
||||||
|
// I don't know why it's done that way, it seems less error-prone to skip over 0
|
||||||
|
// when the value is stored rather than when it's read. But let's do it the same
|
||||||
|
// way here.
|
||||||
|
let next_multi_xid = xlrec.mid.wrapping_add(1);
|
||||||
|
|
||||||
|
if self
|
||||||
|
.checkpoint
|
||||||
|
.update_next_multixid(next_multi_xid, next_offset)
|
||||||
|
{
|
||||||
self.checkpoint_modified = true;
|
self.checkpoint_modified = true;
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// Also update the next-xid with the highest member. According to the comments in
|
||||||
|
// multixact_redo(), this shouldn't be necessary, but let's do the same here.
|
||||||
let max_mbr_xid = xlrec.members.iter().fold(None, |acc, mbr| {
|
let max_mbr_xid = xlrec.members.iter().fold(None, |acc, mbr| {
|
||||||
if let Some(max_xid) = acc {
|
if let Some(max_xid) = acc {
|
||||||
if mbr.xid.wrapping_sub(max_xid) as i32 > 0 {
|
if mbr.xid.wrapping_sub(max_xid) as i32 > 0 {
|
||||||
|
|||||||
+2
-1
@@ -6,6 +6,7 @@ OBJS = \
|
|||||||
$(WIN32RES) \
|
$(WIN32RES) \
|
||||||
extension_server.o \
|
extension_server.o \
|
||||||
file_cache.o \
|
file_cache.o \
|
||||||
|
hll.o \
|
||||||
libpagestore.o \
|
libpagestore.o \
|
||||||
neon.o \
|
neon.o \
|
||||||
neon_utils.o \
|
neon_utils.o \
|
||||||
@@ -22,7 +23,7 @@ SHLIB_LINK_INTERNAL = $(libpq)
|
|||||||
SHLIB_LINK = -lcurl
|
SHLIB_LINK = -lcurl
|
||||||
|
|
||||||
EXTENSION = neon
|
EXTENSION = neon
|
||||||
DATA = neon--1.0.sql neon--1.0--1.1.sql neon--1.1--1.2.sql neon--1.2--1.3.sql neon--1.3--1.2.sql neon--1.2--1.1.sql neon--1.1--1.0.sql
|
DATA = neon--1.0.sql neon--1.0--1.1.sql neon--1.1--1.2.sql neon--1.2--1.3.sql neon--1.3--1.2.sql neon--1.2--1.1.sql neon--1.1--1.0.sql neon--1.3--1.4.sql neon--1.4--1.3.sql
|
||||||
PGFILEDESC = "neon - cloud storage for PostgreSQL"
|
PGFILEDESC = "neon - cloud storage for PostgreSQL"
|
||||||
|
|
||||||
EXTRA_CLEAN = \
|
EXTRA_CLEAN = \
|
||||||
|
|||||||
+27
-15
@@ -26,7 +26,6 @@
|
|||||||
#include "miscadmin.h"
|
#include "miscadmin.h"
|
||||||
#include "pagestore_client.h"
|
#include "pagestore_client.h"
|
||||||
#include "common/hashfn.h"
|
#include "common/hashfn.h"
|
||||||
#include "lib/hyperloglog.h"
|
|
||||||
#include "pgstat.h"
|
#include "pgstat.h"
|
||||||
#include "postmaster/bgworker.h"
|
#include "postmaster/bgworker.h"
|
||||||
#include RELFILEINFO_HDR
|
#include RELFILEINFO_HDR
|
||||||
@@ -40,6 +39,8 @@
|
|||||||
#include "utils/dynahash.h"
|
#include "utils/dynahash.h"
|
||||||
#include "utils/guc.h"
|
#include "utils/guc.h"
|
||||||
|
|
||||||
|
#include "hll.h"
|
||||||
|
|
||||||
/*
|
/*
|
||||||
* Local file cache is used to temporary store relations pages in local file system.
|
* Local file cache is used to temporary store relations pages in local file system.
|
||||||
* All blocks of all relations are stored inside one file and addressed using shared hash map.
|
* All blocks of all relations are stored inside one file and addressed using shared hash map.
|
||||||
@@ -62,7 +63,6 @@
|
|||||||
#define BLOCKS_PER_CHUNK 128 /* 1Mb chunk */
|
#define BLOCKS_PER_CHUNK 128 /* 1Mb chunk */
|
||||||
#define MB ((uint64)1024*1024)
|
#define MB ((uint64)1024*1024)
|
||||||
|
|
||||||
#define HYPER_LOG_LOG_BIT_WIDTH 10
|
|
||||||
#define SIZE_MB_TO_CHUNKS(size) ((uint32)((size) * MB / BLCKSZ / BLOCKS_PER_CHUNK))
|
#define SIZE_MB_TO_CHUNKS(size) ((uint32)((size) * MB / BLCKSZ / BLOCKS_PER_CHUNK))
|
||||||
|
|
||||||
typedef struct FileCacheEntry
|
typedef struct FileCacheEntry
|
||||||
@@ -87,8 +87,7 @@ typedef struct FileCacheControl
|
|||||||
uint64 writes;
|
uint64 writes;
|
||||||
dlist_head lru; /* double linked list for LRU replacement
|
dlist_head lru; /* double linked list for LRU replacement
|
||||||
* algorithm */
|
* algorithm */
|
||||||
hyperLogLogState wss_estimation; /* estimation of wroking set size */
|
HyperLogLogState wss_estimation; /* estimation of working set size */
|
||||||
uint8_t hyperloglog_hashes[(1 << HYPER_LOG_LOG_BIT_WIDTH) + 1];
|
|
||||||
} FileCacheControl;
|
} FileCacheControl;
|
||||||
|
|
||||||
static HTAB *lfc_hash;
|
static HTAB *lfc_hash;
|
||||||
@@ -238,12 +237,7 @@ lfc_shmem_startup(void)
|
|||||||
dlist_init(&lfc_ctl->lru);
|
dlist_init(&lfc_ctl->lru);
|
||||||
|
|
||||||
/* Initialize hyper-log-log structure for estimating working set size */
|
/* Initialize hyper-log-log structure for estimating working set size */
|
||||||
initHyperLogLog(&lfc_ctl->wss_estimation, HYPER_LOG_LOG_BIT_WIDTH);
|
initSHLL(&lfc_ctl->wss_estimation);
|
||||||
|
|
||||||
/* We need hashes in shared memory */
|
|
||||||
pfree(lfc_ctl->wss_estimation.hashesArr);
|
|
||||||
memset(lfc_ctl->hyperloglog_hashes, 0, sizeof lfc_ctl->hyperloglog_hashes);
|
|
||||||
lfc_ctl->wss_estimation.hashesArr = lfc_ctl->hyperloglog_hashes;
|
|
||||||
|
|
||||||
/* Recreate file cache on restart */
|
/* Recreate file cache on restart */
|
||||||
fd = BasicOpenFile(lfc_path, O_RDWR | O_CREAT | O_TRUNC);
|
fd = BasicOpenFile(lfc_path, O_RDWR | O_CREAT | O_TRUNC);
|
||||||
@@ -545,7 +539,7 @@ lfc_read(NRelFileInfo rinfo, ForkNumber forkNum, BlockNumber blkno,
|
|||||||
|
|
||||||
/* Approximate working set */
|
/* Approximate working set */
|
||||||
tag.blockNum = blkno;
|
tag.blockNum = blkno;
|
||||||
addHyperLogLog(&lfc_ctl->wss_estimation, hash_bytes((uint8_t const*)&tag, sizeof(tag)));
|
addSHLL(&lfc_ctl->wss_estimation, hash_bytes((uint8_t const*)&tag, sizeof(tag)));
|
||||||
|
|
||||||
if (entry == NULL || (entry->bitmap[chunk_offs >> 5] & (1 << (chunk_offs & 31))) == 0)
|
if (entry == NULL || (entry->bitmap[chunk_offs >> 5] & (1 << (chunk_offs & 31))) == 0)
|
||||||
{
|
{
|
||||||
@@ -986,20 +980,38 @@ local_cache_pages(PG_FUNCTION_ARGS)
|
|||||||
SRF_RETURN_DONE(funcctx);
|
SRF_RETURN_DONE(funcctx);
|
||||||
}
|
}
|
||||||
|
|
||||||
|
PG_FUNCTION_INFO_V1(approximate_working_set_size_seconds);
|
||||||
|
|
||||||
|
Datum
|
||||||
|
approximate_working_set_size_seconds(PG_FUNCTION_ARGS)
|
||||||
|
{
|
||||||
|
if (lfc_size_limit != 0)
|
||||||
|
{
|
||||||
|
int32 dc;
|
||||||
|
time_t duration = PG_ARGISNULL(0) ? (time_t)-1 : PG_GETARG_INT32(0);
|
||||||
|
LWLockAcquire(lfc_lock, LW_SHARED);
|
||||||
|
dc = (int32) estimateSHLL(&lfc_ctl->wss_estimation, duration);
|
||||||
|
LWLockRelease(lfc_lock);
|
||||||
|
PG_RETURN_INT32(dc);
|
||||||
|
}
|
||||||
|
PG_RETURN_NULL();
|
||||||
|
}
|
||||||
|
|
||||||
PG_FUNCTION_INFO_V1(approximate_working_set_size);
|
PG_FUNCTION_INFO_V1(approximate_working_set_size);
|
||||||
|
|
||||||
Datum
|
Datum
|
||||||
approximate_working_set_size(PG_FUNCTION_ARGS)
|
approximate_working_set_size(PG_FUNCTION_ARGS)
|
||||||
{
|
{
|
||||||
int32 dc = -1;
|
|
||||||
if (lfc_size_limit != 0)
|
if (lfc_size_limit != 0)
|
||||||
{
|
{
|
||||||
|
int32 dc;
|
||||||
bool reset = PG_GETARG_BOOL(0);
|
bool reset = PG_GETARG_BOOL(0);
|
||||||
LWLockAcquire(lfc_lock, reset ? LW_EXCLUSIVE : LW_SHARED);
|
LWLockAcquire(lfc_lock, reset ? LW_EXCLUSIVE : LW_SHARED);
|
||||||
dc = (int32) estimateHyperLogLog(&lfc_ctl->wss_estimation);
|
dc = (int32) estimateSHLL(&lfc_ctl->wss_estimation, (time_t)-1);
|
||||||
if (reset)
|
if (reset)
|
||||||
memset(lfc_ctl->hyperloglog_hashes, 0, sizeof lfc_ctl->hyperloglog_hashes);
|
memset(lfc_ctl->wss_estimation.regs, 0, sizeof lfc_ctl->wss_estimation.regs);
|
||||||
LWLockRelease(lfc_lock);
|
LWLockRelease(lfc_lock);
|
||||||
|
PG_RETURN_INT32(dc);
|
||||||
}
|
}
|
||||||
PG_RETURN_INT32(dc);
|
PG_RETURN_NULL();
|
||||||
}
|
}
|
||||||
|
|||||||
+193
@@ -0,0 +1,193 @@
|
|||||||
|
/*-------------------------------------------------------------------------
|
||||||
|
*
|
||||||
|
* hll.c
|
||||||
|
* Sliding HyperLogLog cardinality estimator
|
||||||
|
*
|
||||||
|
* Portions Copyright (c) 2014-2023, PostgreSQL Global Development Group
|
||||||
|
*
|
||||||
|
* Implements https://hal.science/hal-00465313/document
|
||||||
|
*
|
||||||
|
* Based on Hideaki Ohno's C++ implementation. This is probably not ideally
|
||||||
|
* suited to estimating the cardinality of very large sets; in particular, we
|
||||||
|
* have not attempted to further optimize the implementation as described in
|
||||||
|
* the Heule, Nunkesser and Hall paper "HyperLogLog in Practice: Algorithmic
|
||||||
|
* Engineering of a State of The Art Cardinality Estimation Algorithm".
|
||||||
|
*
|
||||||
|
* A sparse representation of HyperLogLog state is used, with fixed space
|
||||||
|
* overhead.
|
||||||
|
*
|
||||||
|
* The copyright terms of Ohno's original version (the MIT license) follow.
|
||||||
|
*
|
||||||
|
* IDENTIFICATION
|
||||||
|
* src/backend/lib/hyperloglog.c
|
||||||
|
*
|
||||||
|
*-------------------------------------------------------------------------
|
||||||
|
*/
|
||||||
|
|
||||||
|
/*
|
||||||
|
* Copyright (c) 2013 Hideaki Ohno <hide.o.j55{at}gmail.com>
|
||||||
|
*
|
||||||
|
* Permission is hereby granted, free of charge, to any person obtaining a copy
|
||||||
|
* of this software and associated documentation files (the 'Software'), to
|
||||||
|
* deal in the Software without restriction, including without limitation the
|
||||||
|
* rights to use, copy, modify, merge, publish, distribute, sublicense, and/or
|
||||||
|
* sell copies of the Software, and to permit persons to whom the Software is
|
||||||
|
* furnished to do so, subject to the following conditions:
|
||||||
|
*
|
||||||
|
* The above copyright notice and this permission notice shall be included in
|
||||||
|
* all copies or substantial portions of the Software.
|
||||||
|
*
|
||||||
|
* THE SOFTWARE IS PROVIDED 'AS IS', WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
||||||
|
* IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
||||||
|
* FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
||||||
|
* AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
||||||
|
* LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING
|
||||||
|
* FROM, OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS
|
||||||
|
* IN THE SOFTWARE.
|
||||||
|
*/
|
||||||
|
|
||||||
|
#include <math.h>
|
||||||
|
|
||||||
|
#include "postgres.h"
|
||||||
|
#include "funcapi.h"
|
||||||
|
#include "port/pg_bitutils.h"
|
||||||
|
#include "utils/timestamp.h"
|
||||||
|
#include "hll.h"
|
||||||
|
|
||||||
|
|
||||||
|
#define POW_2_32 (4294967296.0)
|
||||||
|
#define NEG_POW_2_32 (-4294967296.0)
|
||||||
|
|
||||||
|
#define ALPHA_MM ((0.7213 / (1.0 + 1.079 / HLL_N_REGISTERS)) * HLL_N_REGISTERS * HLL_N_REGISTERS)
|
||||||
|
|
||||||
|
/*
|
||||||
|
* Worker for addHyperLogLog().
|
||||||
|
*
|
||||||
|
* Calculates the position of the first set bit in first b bits of x argument
|
||||||
|
* starting from the first, reading from most significant to least significant
|
||||||
|
* bits.
|
||||||
|
*
|
||||||
|
* Example (when considering fist 10 bits of x):
|
||||||
|
*
|
||||||
|
* rho(x = 0b1000000000) returns 1
|
||||||
|
* rho(x = 0b0010000000) returns 3
|
||||||
|
* rho(x = 0b0000000000) returns b + 1
|
||||||
|
*
|
||||||
|
* "The binary address determined by the first b bits of x"
|
||||||
|
*
|
||||||
|
* Return value "j" used to index bit pattern to watch.
|
||||||
|
*/
|
||||||
|
static inline uint8
|
||||||
|
rho(uint32 x, uint8 b)
|
||||||
|
{
|
||||||
|
uint8 j = 1;
|
||||||
|
|
||||||
|
if (x == 0)
|
||||||
|
return b + 1;
|
||||||
|
|
||||||
|
j = 32 - pg_leftmost_one_pos32(x);
|
||||||
|
|
||||||
|
if (j > b)
|
||||||
|
return b + 1;
|
||||||
|
|
||||||
|
return j;
|
||||||
|
}
|
||||||
|
|
||||||
|
/*
|
||||||
|
* Initialize HyperLogLog track state
|
||||||
|
*/
|
||||||
|
void
|
||||||
|
initSHLL(HyperLogLogState *cState)
|
||||||
|
{
|
||||||
|
memset(cState->regs, 0, sizeof(cState->regs));
|
||||||
|
}
|
||||||
|
|
||||||
|
/*
|
||||||
|
* Adds element to the estimator, from caller-supplied hash.
|
||||||
|
*
|
||||||
|
* It is critical that the hash value passed be an actual hash value, typically
|
||||||
|
* generated using hash_any(). The algorithm relies on a specific bit-pattern
|
||||||
|
* observable in conjunction with stochastic averaging. There must be a
|
||||||
|
* uniform distribution of bits in hash values for each distinct original value
|
||||||
|
* observed.
|
||||||
|
*/
|
||||||
|
void
|
||||||
|
addSHLL(HyperLogLogState *cState, uint32 hash)
|
||||||
|
{
|
||||||
|
uint8 count;
|
||||||
|
uint32 index;
|
||||||
|
size_t i;
|
||||||
|
size_t j;
|
||||||
|
|
||||||
|
TimestampTz now = GetCurrentTimestamp();
|
||||||
|
/* Use the first "k" (registerWidth) bits as a zero based index */
|
||||||
|
index = hash >> HLL_C_BITS;
|
||||||
|
|
||||||
|
/* Compute the rank of the remaining 32 - "k" (registerWidth) bits */
|
||||||
|
count = rho(hash << HLL_BIT_WIDTH, HLL_C_BITS);
|
||||||
|
|
||||||
|
cState->regs[index][count] = now;
|
||||||
|
}
|
||||||
|
|
||||||
|
static uint8
|
||||||
|
getMaximum(const TimestampTz* reg, TimestampTz since)
|
||||||
|
{
|
||||||
|
uint8 max = 0;
|
||||||
|
|
||||||
|
for (size_t i = 0; i < HLL_C_BITS + 1; i++)
|
||||||
|
{
|
||||||
|
if (reg[i] >= since)
|
||||||
|
{
|
||||||
|
max = i;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
return max;
|
||||||
|
}
|
||||||
|
|
||||||
|
|
||||||
|
/*
|
||||||
|
* Estimates cardinality, based on elements added so far
|
||||||
|
*/
|
||||||
|
double
|
||||||
|
estimateSHLL(HyperLogLogState *cState, time_t duration)
|
||||||
|
{
|
||||||
|
double result;
|
||||||
|
double sum = 0.0;
|
||||||
|
size_t i;
|
||||||
|
uint8 R[HLL_N_REGISTERS];
|
||||||
|
/* 0 indicates uninitialized timestamp, so if we need to cover the whole range than starts with 1 */
|
||||||
|
TimestampTz since = duration == (time_t)-1 ? 1 : GetCurrentTimestamp() - duration * USECS_PER_SEC;
|
||||||
|
|
||||||
|
for (i = 0; i < HLL_N_REGISTERS; i++)
|
||||||
|
{
|
||||||
|
R[i] = getMaximum(cState->regs[i], since);
|
||||||
|
sum += 1.0 / pow(2.0, R[i]);
|
||||||
|
}
|
||||||
|
|
||||||
|
/* result set to "raw" HyperLogLog estimate (E in the HyperLogLog paper) */
|
||||||
|
result = ALPHA_MM / sum;
|
||||||
|
|
||||||
|
if (result <= (5.0 / 2.0) * HLL_N_REGISTERS)
|
||||||
|
{
|
||||||
|
/* Small range correction */
|
||||||
|
int zero_count = 0;
|
||||||
|
|
||||||
|
for (i = 0; i < HLL_N_REGISTERS; i++)
|
||||||
|
{
|
||||||
|
zero_count += R[i] == 0;
|
||||||
|
}
|
||||||
|
|
||||||
|
if (zero_count != 0)
|
||||||
|
result = HLL_N_REGISTERS * log((double) HLL_N_REGISTERS /
|
||||||
|
zero_count);
|
||||||
|
}
|
||||||
|
else if (result > (1.0 / 30.0) * POW_2_32)
|
||||||
|
{
|
||||||
|
/* Large range correction */
|
||||||
|
result = NEG_POW_2_32 * log(1.0 - (result / POW_2_32));
|
||||||
|
}
|
||||||
|
|
||||||
|
return result;
|
||||||
|
}
|
||||||
|
|
||||||
@@ -0,0 +1,86 @@
|
|||||||
|
/*-------------------------------------------------------------------------
|
||||||
|
*
|
||||||
|
* hll.h
|
||||||
|
* Sliding HyperLogLog cardinality estimator
|
||||||
|
*
|
||||||
|
* Portions Copyright (c) 2014-2023, PostgreSQL Global Development Group
|
||||||
|
*
|
||||||
|
* Implements https://hal.science/hal-00465313/document
|
||||||
|
*
|
||||||
|
* Based on Hideaki Ohno's C++ implementation. This is probably not ideally
|
||||||
|
* suited to estimating the cardinality of very large sets; in particular, we
|
||||||
|
* have not attempted to further optimize the implementation as described in
|
||||||
|
* the Heule, Nunkesser and Hall paper "HyperLogLog in Practice: Algorithmic
|
||||||
|
* Engineering of a State of The Art Cardinality Estimation Algorithm".
|
||||||
|
*
|
||||||
|
* A sparse representation of HyperLogLog state is used, with fixed space
|
||||||
|
* overhead.
|
||||||
|
*
|
||||||
|
* The copyright terms of Ohno's original version (the MIT license) follow.
|
||||||
|
*
|
||||||
|
* IDENTIFICATION
|
||||||
|
* src/backend/lib/hyperloglog.c
|
||||||
|
*
|
||||||
|
*-------------------------------------------------------------------------
|
||||||
|
*/
|
||||||
|
|
||||||
|
/*
|
||||||
|
* Copyright (c) 2013 Hideaki Ohno <hide.o.j55{at}gmail.com>
|
||||||
|
*
|
||||||
|
* Permission is hereby granted, free of charge, to any person obtaining a copy
|
||||||
|
* of this software and associated documentation files (the 'Software'), to
|
||||||
|
* deal in the Software without restriction, including without limitation the
|
||||||
|
* rights to use, copy, modify, merge, publish, distribute, sublicense, and/or
|
||||||
|
* sell copies of the Software, and to permit persons to whom the Software is
|
||||||
|
* furnished to do so, subject to the following conditions:
|
||||||
|
*
|
||||||
|
* The above copyright notice and this permission notice shall be included in
|
||||||
|
* all copies or substantial portions of the Software.
|
||||||
|
*
|
||||||
|
* THE SOFTWARE IS PROVIDED 'AS IS', WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
||||||
|
* IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
||||||
|
* FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
||||||
|
* AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
||||||
|
* LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING
|
||||||
|
* FROM, OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS
|
||||||
|
* IN THE SOFTWARE.
|
||||||
|
*/
|
||||||
|
|
||||||
|
#ifndef HLL_H
|
||||||
|
#define HLL_H
|
||||||
|
|
||||||
|
#define HLL_BIT_WIDTH 10
|
||||||
|
#define HLL_C_BITS (32 - HLL_BIT_WIDTH)
|
||||||
|
#define HLL_N_REGISTERS (1 << HLL_BIT_WIDTH)
|
||||||
|
|
||||||
|
/*
|
||||||
|
* HyperLogLog is an approximate technique for computing the number of distinct
|
||||||
|
* entries in a set. Importantly, it does this by using a fixed amount of
|
||||||
|
* memory. See the 2007 paper "HyperLogLog: the analysis of a near-optimal
|
||||||
|
* cardinality estimation algorithm" for more.
|
||||||
|
*
|
||||||
|
* Instead of a single counter for every bits register, we have a timestamp
|
||||||
|
* for every valid number of bits we can encounter. Every time we encounter
|
||||||
|
* a certain number of bits, we update the timestamp in those registers to
|
||||||
|
* the current timestamp.
|
||||||
|
*
|
||||||
|
* We can query the sketch's stored cardinality for the range of some timestamp
|
||||||
|
* up to now: For each register, we return the highest bits bucket that has a
|
||||||
|
* modified timestamp >= the query timestamp. This value is the number of bits
|
||||||
|
* for this register in the normal HLL calculation.
|
||||||
|
*
|
||||||
|
* The memory usage is 2^B * (C + 1) * sizeof(TimetampTz), or 184kiB.
|
||||||
|
* Usage could be halved if we decide to reduce the required time dimension
|
||||||
|
* precision; as 32 bits in second precision should be enough for statistics.
|
||||||
|
* However, that is not yet implemented.
|
||||||
|
*/
|
||||||
|
typedef struct HyperLogLogState
|
||||||
|
{
|
||||||
|
TimestampTz regs[HLL_N_REGISTERS][HLL_C_BITS + 1];
|
||||||
|
} HyperLogLogState;
|
||||||
|
|
||||||
|
extern void initSHLL(HyperLogLogState *cState);
|
||||||
|
extern void addSHLL(HyperLogLogState *cState, uint32 hash);
|
||||||
|
extern double estimateSHLL(HyperLogLogState *cState, time_t dutration);
|
||||||
|
|
||||||
|
#endif
|
||||||
@@ -427,12 +427,17 @@ pageserver_connect(shardno_t shard_no, int elevel)
|
|||||||
values[n_pgsql_params] = NULL;
|
values[n_pgsql_params] = NULL;
|
||||||
|
|
||||||
shard->conn = PQconnectStartParams(keywords, values, 1);
|
shard->conn = PQconnectStartParams(keywords, values, 1);
|
||||||
if (!shard->conn)
|
if (PQstatus(shard->conn) == CONNECTION_BAD)
|
||||||
{
|
{
|
||||||
neon_shard_log(shard_no, elevel, "Failed to connect to pageserver: out of memory");
|
char *msg = pchomp(PQerrorMessage(shard->conn));
|
||||||
|
CLEANUP_AND_DISCONNECT(shard);
|
||||||
|
ereport(elevel,
|
||||||
|
(errcode(ERRCODE_SQLCLIENT_UNABLE_TO_ESTABLISH_SQLCONNECTION),
|
||||||
|
errmsg(NEON_TAG "[shard %d] could not establish connection to pageserver", shard_no),
|
||||||
|
errdetail_internal("%s", msg)));
|
||||||
|
pfree(msg);
|
||||||
return false;
|
return false;
|
||||||
}
|
}
|
||||||
|
|
||||||
shard->state = PS_Connecting_Startup;
|
shard->state = PS_Connecting_Startup;
|
||||||
/* fallthrough */
|
/* fallthrough */
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -0,0 +1,9 @@
|
|||||||
|
\echo Use "ALTER EXTENSION neon UPDATE TO '1.4'" to load this file. \quit
|
||||||
|
|
||||||
|
CREATE FUNCTION approximate_working_set_size_seconds(duration integer default null)
|
||||||
|
RETURNS integer
|
||||||
|
AS 'MODULE_PATHNAME', 'approximate_working_set_size_seconds'
|
||||||
|
LANGUAGE C PARALLEL SAFE;
|
||||||
|
|
||||||
|
GRANT EXECUTE ON FUNCTION approximate_working_set_size_seconds(integer) TO pg_monitor;
|
||||||
|
|
||||||
@@ -0,0 +1 @@
|
|||||||
|
DROP FUNCTION IF EXISTS approximate_working_set_size_seconds(integer) CASCADE;
|
||||||
@@ -12,6 +12,8 @@
|
|||||||
#include "fmgr.h"
|
#include "fmgr.h"
|
||||||
|
|
||||||
#include "miscadmin.h"
|
#include "miscadmin.h"
|
||||||
|
#include "access/subtrans.h"
|
||||||
|
#include "access/twophase.h"
|
||||||
#include "access/xact.h"
|
#include "access/xact.h"
|
||||||
#include "access/xlog.h"
|
#include "access/xlog.h"
|
||||||
#include "storage/buf_internals.h"
|
#include "storage/buf_internals.h"
|
||||||
@@ -22,10 +24,12 @@
|
|||||||
#include "replication/logical.h"
|
#include "replication/logical.h"
|
||||||
#include "replication/slot.h"
|
#include "replication/slot.h"
|
||||||
#include "replication/walsender.h"
|
#include "replication/walsender.h"
|
||||||
|
#include "storage/proc.h"
|
||||||
#include "storage/procsignal.h"
|
#include "storage/procsignal.h"
|
||||||
#include "tcop/tcopprot.h"
|
#include "tcop/tcopprot.h"
|
||||||
#include "funcapi.h"
|
#include "funcapi.h"
|
||||||
#include "access/htup_details.h"
|
#include "access/htup_details.h"
|
||||||
|
#include "utils/builtins.h"
|
||||||
#include "utils/pg_lsn.h"
|
#include "utils/pg_lsn.h"
|
||||||
#include "utils/guc.h"
|
#include "utils/guc.h"
|
||||||
#include "utils/wait_event.h"
|
#include "utils/wait_event.h"
|
||||||
@@ -266,6 +270,293 @@ LogicalSlotsMonitorMain(Datum main_arg)
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/*
|
||||||
|
* XXX: These private to procarray.c, but we need them here.
|
||||||
|
*/
|
||||||
|
#define PROCARRAY_MAXPROCS (MaxBackends + max_prepared_xacts)
|
||||||
|
#define TOTAL_MAX_CACHED_SUBXIDS \
|
||||||
|
((PGPROC_MAX_CACHED_SUBXIDS + 1) * PROCARRAY_MAXPROCS)
|
||||||
|
|
||||||
|
/*
|
||||||
|
* Restore running-xact information by scanning the CLOG at startup.
|
||||||
|
*
|
||||||
|
* In PostgreSQL, a standby always has to wait for a running-xacts WAL record
|
||||||
|
* to arrive before it can start accepting queries. Furthermore, if there are
|
||||||
|
* transactions with too many subxids (> 64) open to fit in the in-memory
|
||||||
|
* subxids cache, the running-xacts record will be marked as "suboverflowed",
|
||||||
|
* and the standby will need to also wait for the currently in-progress
|
||||||
|
* transactions to finish.
|
||||||
|
*
|
||||||
|
* That's not great in PostgreSQL, because a hot standby does not necessary
|
||||||
|
* open up for queries immediately as you might expect. But it's worse in
|
||||||
|
* Neon: A standby in Neon doesn't need to start WAL replay from a checkpoint
|
||||||
|
* record; it can start at any LSN. Postgres arranges things so that there is
|
||||||
|
* a running-xacts record soon after every checkpoint record, but when you
|
||||||
|
* start from an arbitrary LSN, that doesn't help. If the primary is idle, or
|
||||||
|
* not running at all, it might never write a new running-xacts record,
|
||||||
|
* leaving the replica in a limbo where it can never start accepting queries.
|
||||||
|
*
|
||||||
|
* To mitigate that, we have an additional mechanism to find the running-xacts
|
||||||
|
* information: we scan the CLOG, making note of any XIDs not marked as
|
||||||
|
* committed or aborted. They are added to the Postgres known-assigned XIDs
|
||||||
|
* array by calling ProcArrayApplyRecoveryInfo() in the caller of this
|
||||||
|
* function.
|
||||||
|
*
|
||||||
|
* There is one big limitation with that mechanism: The size of the
|
||||||
|
* known-assigned XIDs is limited, so if there are a lot of in-progress XIDs,
|
||||||
|
* we have to give up. Furthermore, we don't know how many of the in-progress
|
||||||
|
* XIDs are subtransactions, and if we use up all the space in the
|
||||||
|
* known-assigned XIDs array for subtransactions, we might run out of space in
|
||||||
|
* the array later during WAL replay, causing the replica to shut down with
|
||||||
|
* "ERROR: too many KnownAssignedXids". The safe # of XIDs that we can add to
|
||||||
|
* the known-assigned array without risking that error later is very low,
|
||||||
|
* merely PGPROC_MAX_CACHED_SUBXIDS == 64, so we take our chances and use up
|
||||||
|
* to half of the known-assigned XIDs array for the subtransactions, even
|
||||||
|
* though that risks getting the error later.
|
||||||
|
*
|
||||||
|
* Note: It's OK if the recovered list of XIDs includes some transactions that
|
||||||
|
* have crashed in the primary, and hence will never commit. They will be seen
|
||||||
|
* as in-progress, until we see a new next running-acts record with an
|
||||||
|
* oldestActiveXid that invalidates them. That's how the known-assigned XIDs
|
||||||
|
* array always works.
|
||||||
|
*
|
||||||
|
* If scraping the CLOG doesn't succeed for some reason, like the subxid
|
||||||
|
* overflow, Postgres will fall back to waiting for a running-xacts record
|
||||||
|
* like usual.
|
||||||
|
*
|
||||||
|
* Returns true if a complete list of in-progress XIDs was scraped.
|
||||||
|
*/
|
||||||
|
static bool
|
||||||
|
RestoreRunningXactsFromClog(CheckPoint *checkpoint, TransactionId **xids, int *nxids)
|
||||||
|
{
|
||||||
|
TransactionId from;
|
||||||
|
TransactionId till;
|
||||||
|
int max_xcnt;
|
||||||
|
TransactionId *prepared_xids = NULL;
|
||||||
|
int n_prepared_xids;
|
||||||
|
TransactionId *restored_xids = NULL;
|
||||||
|
int n_restored_xids;
|
||||||
|
int next_prepared_idx;
|
||||||
|
|
||||||
|
Assert(*xids == NULL);
|
||||||
|
|
||||||
|
/*
|
||||||
|
* If the checkpoint doesn't have a valid oldestActiveXid, bail out. We
|
||||||
|
* don't know where to start the scan.
|
||||||
|
*
|
||||||
|
* This shouldn't happen, because the pageserver always maintains a valid
|
||||||
|
* oldestActiveXid nowadays. Except when starting at an old point in time
|
||||||
|
* that was ingested before the pageserver was taught to do that.
|
||||||
|
*/
|
||||||
|
if (!TransactionIdIsValid(checkpoint->oldestActiveXid))
|
||||||
|
{
|
||||||
|
elog(LOG, "cannot restore running-xacts from CLOG because oldestActiveXid is not set");
|
||||||
|
goto fail;
|
||||||
|
}
|
||||||
|
|
||||||
|
/*
|
||||||
|
* We will scan the CLOG starting from the oldest active XID.
|
||||||
|
*
|
||||||
|
* In some corner cases, the oldestActiveXid from the last checkpoint
|
||||||
|
* might already have been truncated from the CLOG. That is,
|
||||||
|
* oldestActiveXid might be older than oldestXid. That's possible because
|
||||||
|
* oldestActiveXid is only updated at checkpoints. After the last
|
||||||
|
* checkpoint, the oldest transaction might have committed, and the CLOG
|
||||||
|
* might also have been already truncated. So if oldestActiveXid is older
|
||||||
|
* than oldestXid, start at oldestXid instead. (Otherwise we'd try to
|
||||||
|
* access CLOG segments that have already been truncated away.)
|
||||||
|
*/
|
||||||
|
from = TransactionIdPrecedes(checkpoint->oldestXid, checkpoint->oldestActiveXid)
|
||||||
|
? checkpoint->oldestActiveXid : checkpoint->oldestXid;
|
||||||
|
till = XidFromFullTransactionId(checkpoint->nextXid);
|
||||||
|
|
||||||
|
/*
|
||||||
|
* To avoid "too many KnownAssignedXids" error later during replay, we
|
||||||
|
* limit number of collected transactions. This is a tradeoff: if we are
|
||||||
|
* willing to consume more of the KnownAssignedXids space for the XIDs
|
||||||
|
* now, that allows us to start up, but we might run out of space later.
|
||||||
|
*
|
||||||
|
* The size of the KnownAssignedXids array is TOTAL_MAX_CACHED_SUBXIDS,
|
||||||
|
* which is (PGPROC_MAX_CACHED_SUBXIDS + 1) * PROCARRAY_MAXPROCS). In
|
||||||
|
* PostgreSQL, that's always enough because the primary will always write
|
||||||
|
* an XLOG_XACT_ASSIGNMENT record if a transaction has more than
|
||||||
|
* PGPROC_MAX_CACHED_SUBXIDS subtransactions. Seeing that record allows
|
||||||
|
* the standby to mark the XIDs in pg_subtrans and removing them from the
|
||||||
|
* KnowingAssignedXids array.
|
||||||
|
*
|
||||||
|
* Here, we don't know which XIDs belong to subtransactions that have
|
||||||
|
* already been WAL-logged with an XLOG_XACT_ASSIGNMENT record. If we
|
||||||
|
* wanted to be totally safe and avoid the possibility of getting a "too
|
||||||
|
* many KnownAssignedXids" error later, we would have to limit ourselves
|
||||||
|
* to PGPROC_MAX_CACHED_SUBXIDS, which is not much. And that includes top
|
||||||
|
* transaction IDs too, because we cannot distinguish between top
|
||||||
|
* transaction IDs and subtransactions here.
|
||||||
|
*
|
||||||
|
* Somewhat arbitrarily, we use up to half of KnownAssignedXids. That
|
||||||
|
* strikes a sensible balance between being useful, and risking a "too
|
||||||
|
* many KnownAssignedXids" error later.
|
||||||
|
*/
|
||||||
|
max_xcnt = TOTAL_MAX_CACHED_SUBXIDS / 2;
|
||||||
|
|
||||||
|
/*
|
||||||
|
* Collect XIDs of prepared transactions in an array. This includes only
|
||||||
|
* their top-level XIDs. We assume that StandbyRecoverPreparedTransactions
|
||||||
|
* has already been called, so we can find all the sub-transactions in
|
||||||
|
* pg_subtrans.
|
||||||
|
*/
|
||||||
|
PrescanPreparedTransactions(&prepared_xids, &n_prepared_xids);
|
||||||
|
qsort(prepared_xids, n_prepared_xids, sizeof(TransactionId), xidLogicalComparator);
|
||||||
|
|
||||||
|
/*
|
||||||
|
* Scan the CLOG, collecting in-progress XIDs into 'restored_xids'.
|
||||||
|
*/
|
||||||
|
elog(DEBUG1, "scanning CLOG between %u and %u for in-progress XIDs", from, till);
|
||||||
|
restored_xids = (TransactionId *) palloc(max_xcnt * sizeof(TransactionId));
|
||||||
|
n_restored_xids = 0;
|
||||||
|
next_prepared_idx = 0;
|
||||||
|
for (TransactionId xid = from; xid != till;)
|
||||||
|
{
|
||||||
|
XLogRecPtr xidlsn;
|
||||||
|
XidStatus xidstatus;
|
||||||
|
|
||||||
|
xidstatus = TransactionIdGetStatus(xid, &xidlsn);
|
||||||
|
|
||||||
|
/*
|
||||||
|
* "Merge" the prepared transactions into the restored_xids array as
|
||||||
|
* we go. The prepared transactions array is sorted. This is mostly
|
||||||
|
* a sanity check to ensure that all the prpeared transactions are
|
||||||
|
* seen as in-progress. (There is a check after the loop that we didn't
|
||||||
|
* miss any.)
|
||||||
|
*/
|
||||||
|
if (next_prepared_idx < n_prepared_xids && xid == prepared_xids[next_prepared_idx])
|
||||||
|
{
|
||||||
|
/*
|
||||||
|
* This is a top-level transaction ID of a prepared transaction.
|
||||||
|
* Include it in the array.
|
||||||
|
*/
|
||||||
|
|
||||||
|
/* sanity check */
|
||||||
|
if (xidstatus != TRANSACTION_STATUS_IN_PROGRESS)
|
||||||
|
{
|
||||||
|
elog(LOG, "prepared transaction %u has unexpected status %X, cannot restore running-xacts from CLOG",
|
||||||
|
xid, xidstatus);
|
||||||
|
Assert(false);
|
||||||
|
goto fail;
|
||||||
|
}
|
||||||
|
|
||||||
|
elog(DEBUG1, "XID %u: was next prepared xact (%d / %d)", xid, next_prepared_idx, n_prepared_xids);
|
||||||
|
next_prepared_idx++;
|
||||||
|
}
|
||||||
|
else if (xidstatus == TRANSACTION_STATUS_COMMITTED)
|
||||||
|
{
|
||||||
|
elog(DEBUG1, "XID %u: was committed", xid);
|
||||||
|
goto skip;
|
||||||
|
}
|
||||||
|
else if (xidstatus == TRANSACTION_STATUS_ABORTED)
|
||||||
|
{
|
||||||
|
elog(DEBUG1, "XID %u: was aborted", xid);
|
||||||
|
goto skip;
|
||||||
|
}
|
||||||
|
else if (xidstatus == TRANSACTION_STATUS_IN_PROGRESS)
|
||||||
|
{
|
||||||
|
/*
|
||||||
|
* In-progress transactions are included in the array.
|
||||||
|
*
|
||||||
|
* Except subtransactions of the prepared transactions. They are
|
||||||
|
* already set in pg_subtrans, and hence don't need to be tracked
|
||||||
|
* in the known-assigned XIDs array.
|
||||||
|
*/
|
||||||
|
if (n_prepared_xids > 0)
|
||||||
|
{
|
||||||
|
TransactionId parent = SubTransGetParent(xid);
|
||||||
|
|
||||||
|
if (TransactionIdIsValid(parent))
|
||||||
|
{
|
||||||
|
/*
|
||||||
|
* This is a subtransaction belonging to a prepared
|
||||||
|
* transaction.
|
||||||
|
*
|
||||||
|
* Sanity check that it is in the prepared XIDs array. It
|
||||||
|
* should be, because StandbyRecoverPreparedTransactions
|
||||||
|
* populated pg_subtrans, and no other XID should be set
|
||||||
|
* in it yet. (This also relies on the fact that
|
||||||
|
* StandbyRecoverPreparedTransactions sets the parent of
|
||||||
|
* each subxid to point directly to the top-level XID,
|
||||||
|
* rather than restoring the original subtransaction
|
||||||
|
* hierarchy.)
|
||||||
|
*/
|
||||||
|
if (bsearch(&parent, prepared_xids, next_prepared_idx,
|
||||||
|
sizeof(TransactionId), xidLogicalComparator) == NULL)
|
||||||
|
{
|
||||||
|
elog(LOG, "sub-XID %u has unexpected parent %u, cannot restore running-xacts from CLOG",
|
||||||
|
xid, parent);
|
||||||
|
Assert(false);
|
||||||
|
goto fail;
|
||||||
|
}
|
||||||
|
elog(DEBUG1, "XID %u: was a subtransaction of prepared xid %u", xid, parent);
|
||||||
|
goto skip;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/* include it in the array */
|
||||||
|
elog(DEBUG1, "XID %u: is in progress", xid);
|
||||||
|
}
|
||||||
|
else
|
||||||
|
{
|
||||||
|
/*
|
||||||
|
* SUB_COMMITTED is a transient state used at commit. We don't
|
||||||
|
* expect to see that here.
|
||||||
|
*/
|
||||||
|
elog(LOG, "XID %u has unexpected status %X in pg_xact, cannot restore running-xacts from CLOG",
|
||||||
|
xid, xidstatus);
|
||||||
|
Assert(false);
|
||||||
|
goto fail;
|
||||||
|
}
|
||||||
|
|
||||||
|
if (n_restored_xids >= max_xcnt)
|
||||||
|
{
|
||||||
|
/*
|
||||||
|
* Overflowed. We won't be able to install the RunningTransactions
|
||||||
|
* snapshot.
|
||||||
|
*/
|
||||||
|
elog(LOG, "too many running xacts to restore from the CLOG; oldestXid=%u oldestActiveXid=%u nextXid %u",
|
||||||
|
checkpoint->oldestXid, checkpoint->oldestActiveXid,
|
||||||
|
XidFromFullTransactionId(checkpoint->nextXid));
|
||||||
|
goto fail;
|
||||||
|
}
|
||||||
|
|
||||||
|
restored_xids[n_restored_xids++] = xid;
|
||||||
|
|
||||||
|
skip:
|
||||||
|
TransactionIdAdvance(xid);
|
||||||
|
continue;
|
||||||
|
}
|
||||||
|
|
||||||
|
/* sanity check */
|
||||||
|
if (next_prepared_idx != n_prepared_xids)
|
||||||
|
{
|
||||||
|
elog(LOG, "prepared transaction ID %u was not visited in the CLOG scan, cannot restore running-xacts from CLOG",
|
||||||
|
prepared_xids[next_prepared_idx]);
|
||||||
|
Assert(false);
|
||||||
|
goto fail;
|
||||||
|
}
|
||||||
|
|
||||||
|
elog(LOG, "restored %d running xacts by scanning the CLOG; oldestXid=%u oldestActiveXid=%u nextXid %u",
|
||||||
|
n_restored_xids, checkpoint->oldestXid, checkpoint->oldestActiveXid, XidFromFullTransactionId(checkpoint->nextXid));
|
||||||
|
*nxids = n_restored_xids;
|
||||||
|
*xids = restored_xids;
|
||||||
|
return true;
|
||||||
|
|
||||||
|
fail:
|
||||||
|
*nxids = 0;
|
||||||
|
*xids = NULL;
|
||||||
|
if (restored_xids)
|
||||||
|
pfree(restored_xids);
|
||||||
|
if (prepared_xids)
|
||||||
|
pfree(prepared_xids);
|
||||||
|
return false;
|
||||||
|
}
|
||||||
|
|
||||||
void
|
void
|
||||||
_PG_init(void)
|
_PG_init(void)
|
||||||
{
|
{
|
||||||
@@ -288,6 +579,8 @@ _PG_init(void)
|
|||||||
|
|
||||||
pg_init_extension_server();
|
pg_init_extension_server();
|
||||||
|
|
||||||
|
restore_running_xacts_callback = RestoreRunningXactsFromClog;
|
||||||
|
|
||||||
/*
|
/*
|
||||||
* Important: This must happen after other parts of the extension are
|
* Important: This must happen after other parts of the extension are
|
||||||
* loaded, otherwise any settings to GUCs that were set before the
|
* loaded, otherwise any settings to GUCs that were set before the
|
||||||
|
|||||||
@@ -7,7 +7,7 @@ OBJS = \
|
|||||||
neontest.o
|
neontest.o
|
||||||
|
|
||||||
EXTENSION = neon_test_utils
|
EXTENSION = neon_test_utils
|
||||||
DATA = neon_test_utils--1.1.sql
|
DATA = neon_test_utils--1.3.sql
|
||||||
PGFILEDESC = "neon_test_utils - helpers for neon testing and debugging"
|
PGFILEDESC = "neon_test_utils - helpers for neon testing and debugging"
|
||||||
|
|
||||||
PG_CONFIG = pg_config
|
PG_CONFIG = pg_config
|
||||||
|
|||||||
+19
-1
@@ -41,7 +41,25 @@ RETURNS bytea
|
|||||||
AS 'MODULE_PATHNAME', 'get_raw_page_at_lsn_ex'
|
AS 'MODULE_PATHNAME', 'get_raw_page_at_lsn_ex'
|
||||||
LANGUAGE C PARALLEL UNSAFE;
|
LANGUAGE C PARALLEL UNSAFE;
|
||||||
|
|
||||||
CREATE FUNCTION neon_xlogflush(lsn pg_lsn)
|
CREATE FUNCTION neon_xlogflush(lsn pg_lsn DEFAULT NULL)
|
||||||
RETURNS VOID
|
RETURNS VOID
|
||||||
AS 'MODULE_PATHNAME', 'neon_xlogflush'
|
AS 'MODULE_PATHNAME', 'neon_xlogflush'
|
||||||
LANGUAGE C PARALLEL UNSAFE;
|
LANGUAGE C PARALLEL UNSAFE;
|
||||||
|
|
||||||
|
CREATE FUNCTION trigger_panic()
|
||||||
|
RETURNS VOID
|
||||||
|
AS 'MODULE_PATHNAME', 'trigger_panic'
|
||||||
|
LANGUAGE C PARALLEL UNSAFE;
|
||||||
|
|
||||||
|
CREATE FUNCTION trigger_segfault()
|
||||||
|
RETURNS VOID
|
||||||
|
AS 'MODULE_PATHNAME', 'trigger_segfault'
|
||||||
|
LANGUAGE C PARALLEL UNSAFE;
|
||||||
|
|
||||||
|
-- Alias for `trigger_segfault`, just because `SELECT 💣()` looks fun
|
||||||
|
CREATE OR REPLACE FUNCTION 💣() RETURNS void
|
||||||
|
LANGUAGE plpgsql AS $$
|
||||||
|
BEGIN
|
||||||
|
PERFORM trigger_segfault();
|
||||||
|
END;
|
||||||
|
$$;
|
||||||
@@ -1,6 +1,6 @@
|
|||||||
# neon_test_utils extension
|
# neon_test_utils extension
|
||||||
comment = 'helpers for neon testing and debugging'
|
comment = 'helpers for neon testing and debugging'
|
||||||
default_version = '1.1'
|
default_version = '1.3'
|
||||||
module_pathname = '$libdir/neon_test_utils'
|
module_pathname = '$libdir/neon_test_utils'
|
||||||
relocatable = true
|
relocatable = true
|
||||||
trusted = true
|
trusted = true
|
||||||
|
|||||||
@@ -15,6 +15,7 @@
|
|||||||
#include "access/relation.h"
|
#include "access/relation.h"
|
||||||
#include "access/xact.h"
|
#include "access/xact.h"
|
||||||
#include "access/xlog.h"
|
#include "access/xlog.h"
|
||||||
|
#include "access/xlog_internal.h"
|
||||||
#include "catalog/namespace.h"
|
#include "catalog/namespace.h"
|
||||||
#include "fmgr.h"
|
#include "fmgr.h"
|
||||||
#include "funcapi.h"
|
#include "funcapi.h"
|
||||||
@@ -41,6 +42,8 @@ PG_FUNCTION_INFO_V1(clear_buffer_cache);
|
|||||||
PG_FUNCTION_INFO_V1(get_raw_page_at_lsn);
|
PG_FUNCTION_INFO_V1(get_raw_page_at_lsn);
|
||||||
PG_FUNCTION_INFO_V1(get_raw_page_at_lsn_ex);
|
PG_FUNCTION_INFO_V1(get_raw_page_at_lsn_ex);
|
||||||
PG_FUNCTION_INFO_V1(neon_xlogflush);
|
PG_FUNCTION_INFO_V1(neon_xlogflush);
|
||||||
|
PG_FUNCTION_INFO_V1(trigger_panic);
|
||||||
|
PG_FUNCTION_INFO_V1(trigger_segfault);
|
||||||
|
|
||||||
/*
|
/*
|
||||||
* Linkage to functions in neon module.
|
* Linkage to functions in neon module.
|
||||||
@@ -444,12 +447,68 @@ get_raw_page_at_lsn_ex(PG_FUNCTION_ARGS)
|
|||||||
|
|
||||||
/*
|
/*
|
||||||
* Directly calls XLogFlush(lsn) to flush WAL buffers.
|
* Directly calls XLogFlush(lsn) to flush WAL buffers.
|
||||||
|
*
|
||||||
|
* If 'lsn' is not specified (is NULL), flush all generated WAL.
|
||||||
*/
|
*/
|
||||||
Datum
|
Datum
|
||||||
neon_xlogflush(PG_FUNCTION_ARGS)
|
neon_xlogflush(PG_FUNCTION_ARGS)
|
||||||
{
|
{
|
||||||
XLogRecPtr lsn = PG_GETARG_LSN(0);
|
XLogRecPtr lsn;
|
||||||
|
|
||||||
|
if (RecoveryInProgress())
|
||||||
|
ereport(ERROR,
|
||||||
|
(errcode(ERRCODE_OBJECT_NOT_IN_PREREQUISITE_STATE),
|
||||||
|
errmsg("recovery is in progress"),
|
||||||
|
errhint("cannot flush WAL during recovery.")));
|
||||||
|
|
||||||
|
if (!PG_ARGISNULL(0))
|
||||||
|
lsn = PG_GETARG_LSN(0);
|
||||||
|
else
|
||||||
|
{
|
||||||
|
lsn = GetXLogInsertRecPtr();
|
||||||
|
|
||||||
|
/*---
|
||||||
|
* The LSN returned by GetXLogInsertRecPtr() is the position where the
|
||||||
|
* next inserted record would begin. If the last record ended just at
|
||||||
|
* the page boundary, the next record will begin after the page header
|
||||||
|
* on the next page, but the next page's page header has not been
|
||||||
|
* written yet. If we tried to flush it, XLogFlush() would throw an
|
||||||
|
* error:
|
||||||
|
*
|
||||||
|
* ERROR : xlog flush request %X/%X is not satisfied --- flushed only to %X/%X
|
||||||
|
*
|
||||||
|
* To avoid that, if the insert position points to just after the page
|
||||||
|
* header, back off to page boundary.
|
||||||
|
*/
|
||||||
|
if (lsn % XLOG_BLCKSZ == SizeOfXLogShortPHD &&
|
||||||
|
XLogSegmentOffset(lsn, wal_segment_size) > XLOG_BLCKSZ)
|
||||||
|
lsn -= SizeOfXLogShortPHD;
|
||||||
|
else if (lsn % XLOG_BLCKSZ == SizeOfXLogLongPHD &&
|
||||||
|
XLogSegmentOffset(lsn, wal_segment_size) < XLOG_BLCKSZ)
|
||||||
|
lsn -= SizeOfXLogLongPHD;
|
||||||
|
}
|
||||||
|
|
||||||
XLogFlush(lsn);
|
XLogFlush(lsn);
|
||||||
PG_RETURN_VOID();
|
PG_RETURN_VOID();
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/*
|
||||||
|
* Function to trigger panic.
|
||||||
|
*/
|
||||||
|
Datum
|
||||||
|
trigger_panic(PG_FUNCTION_ARGS)
|
||||||
|
{
|
||||||
|
elog(PANIC, "neon_test_utils: panic");
|
||||||
|
PG_RETURN_VOID();
|
||||||
|
}
|
||||||
|
|
||||||
|
/*
|
||||||
|
* Function to trigger a segfault.
|
||||||
|
*/
|
||||||
|
Datum
|
||||||
|
trigger_segfault(PG_FUNCTION_ARGS)
|
||||||
|
{
|
||||||
|
int *ptr = NULL;
|
||||||
|
*ptr = 42;
|
||||||
|
PG_RETURN_VOID();
|
||||||
|
}
|
||||||
|
|||||||
Generated
+4
-4
@@ -1,4 +1,4 @@
|
|||||||
# This file is automatically @generated by Poetry 1.8.2 and should not be changed by hand.
|
# This file is automatically @generated by Poetry 1.8.3 and should not be changed by hand.
|
||||||
|
|
||||||
[[package]]
|
[[package]]
|
||||||
name = "aiohttp"
|
name = "aiohttp"
|
||||||
@@ -734,13 +734,13 @@ typing-extensions = ">=4.1.0"
|
|||||||
|
|
||||||
[[package]]
|
[[package]]
|
||||||
name = "certifi"
|
name = "certifi"
|
||||||
version = "2023.7.22"
|
version = "2024.7.4"
|
||||||
description = "Python package for providing Mozilla's CA Bundle."
|
description = "Python package for providing Mozilla's CA Bundle."
|
||||||
optional = false
|
optional = false
|
||||||
python-versions = ">=3.6"
|
python-versions = ">=3.6"
|
||||||
files = [
|
files = [
|
||||||
{file = "certifi-2023.7.22-py3-none-any.whl", hash = "sha256:92d6037539857d8206b8f6ae472e8b77db8058fec5937a1ef3f54304089edbb9"},
|
{file = "certifi-2024.7.4-py3-none-any.whl", hash = "sha256:c198e21b1289c2ab85ee4e67bb4b4ef3ead0892059901a8d5b622f24a1101e90"},
|
||||||
{file = "certifi-2023.7.22.tar.gz", hash = "sha256:539cc1d13202e33ca466e88b2807e29f4c13049d6d87031a3c110744495cb082"},
|
{file = "certifi-2024.7.4.tar.gz", hash = "sha256:5a1e7645bc0ec61a09e26c36f6106dd4cf40c6db3a1fb6352b0244e7fb057c7b"},
|
||||||
]
|
]
|
||||||
|
|
||||||
[[package]]
|
[[package]]
|
||||||
|
|||||||
@@ -35,6 +35,7 @@ use proxy::usage_metrics;
|
|||||||
use anyhow::bail;
|
use anyhow::bail;
|
||||||
use proxy::config::{self, ProxyConfig};
|
use proxy::config::{self, ProxyConfig};
|
||||||
use proxy::serverless;
|
use proxy::serverless;
|
||||||
|
use remote_storage::RemoteStorageConfig;
|
||||||
use std::net::SocketAddr;
|
use std::net::SocketAddr;
|
||||||
use std::pin::pin;
|
use std::pin::pin;
|
||||||
use std::sync::Arc;
|
use std::sync::Arc;
|
||||||
@@ -205,8 +206,8 @@ struct ProxyCliArgs {
|
|||||||
/// remote storage configuration for backup metric collection
|
/// remote storage configuration for backup metric collection
|
||||||
/// Encoded as toml (same format as pageservers), eg
|
/// Encoded as toml (same format as pageservers), eg
|
||||||
/// `{bucket_name='the-bucket',bucket_region='us-east-1',prefix_in_bucket='proxy',endpoint='http://minio:9000'}`
|
/// `{bucket_name='the-bucket',bucket_region='us-east-1',prefix_in_bucket='proxy',endpoint='http://minio:9000'}`
|
||||||
#[clap(long, default_value = "{}")]
|
#[clap(long, value_parser = remote_storage_from_toml)]
|
||||||
metric_backup_collection_remote_storage: String,
|
metric_backup_collection_remote_storage: Option<RemoteStorageConfig>,
|
||||||
/// chunk size for backup metric collection
|
/// chunk size for backup metric collection
|
||||||
/// Size of each event is no more than 400 bytes, so 2**22 is about 200MB before the compression.
|
/// Size of each event is no more than 400 bytes, so 2**22 is about 200MB before the compression.
|
||||||
#[clap(long, default_value = "4194304")]
|
#[clap(long, default_value = "4194304")]
|
||||||
@@ -511,9 +512,7 @@ fn build_config(args: &ProxyCliArgs) -> anyhow::Result<&'static ProxyConfig> {
|
|||||||
}
|
}
|
||||||
let backup_metric_collection_config = config::MetricBackupCollectionConfig {
|
let backup_metric_collection_config = config::MetricBackupCollectionConfig {
|
||||||
interval: args.metric_backup_collection_interval,
|
interval: args.metric_backup_collection_interval,
|
||||||
remote_storage_config: remote_storage_from_toml(
|
remote_storage_config: args.metric_backup_collection_remote_storage.clone(),
|
||||||
&args.metric_backup_collection_remote_storage,
|
|
||||||
)?,
|
|
||||||
chunk_size: args.metric_backup_collection_chunk_size,
|
chunk_size: args.metric_backup_collection_chunk_size,
|
||||||
};
|
};
|
||||||
|
|
||||||
|
|||||||
Vendored
+7
@@ -53,6 +53,13 @@ impl<C: Cache, V> Cached<C, V> {
|
|||||||
)
|
)
|
||||||
}
|
}
|
||||||
|
|
||||||
|
pub fn map<U>(self, f: impl FnOnce(V) -> U) -> Cached<C, U> {
|
||||||
|
Cached {
|
||||||
|
token: self.token,
|
||||||
|
value: f(self.value),
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
/// Drop this entry from a cache if it's still there.
|
/// Drop this entry from a cache if it's still there.
|
||||||
pub fn invalidate(self) -> V {
|
pub fn invalidate(self) -> V {
|
||||||
if let Some((cache, info)) = &self.token {
|
if let Some((cache, info)) = &self.token {
|
||||||
|
|||||||
Vendored
+35
-3
@@ -65,6 +65,8 @@ impl<K: Hash + Eq, V> Cache for TimedLru<K, V> {
|
|||||||
struct Entry<T> {
|
struct Entry<T> {
|
||||||
created_at: Instant,
|
created_at: Instant,
|
||||||
expires_at: Instant,
|
expires_at: Instant,
|
||||||
|
ttl: Duration,
|
||||||
|
update_ttl_on_retrieval: bool,
|
||||||
value: T,
|
value: T,
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -122,7 +124,6 @@ impl<K: Hash + Eq, V> TimedLru<K, V> {
|
|||||||
Q: Hash + Eq + ?Sized,
|
Q: Hash + Eq + ?Sized,
|
||||||
{
|
{
|
||||||
let now = Instant::now();
|
let now = Instant::now();
|
||||||
let deadline = now.checked_add(self.ttl).expect("time overflow");
|
|
||||||
|
|
||||||
// Do costly things before taking the lock.
|
// Do costly things before taking the lock.
|
||||||
let mut cache = self.cache.lock();
|
let mut cache = self.cache.lock();
|
||||||
@@ -142,7 +143,8 @@ impl<K: Hash + Eq, V> TimedLru<K, V> {
|
|||||||
let (created_at, expires_at) = (entry.created_at, entry.expires_at);
|
let (created_at, expires_at) = (entry.created_at, entry.expires_at);
|
||||||
|
|
||||||
// Update the deadline and the entry's position in the LRU list.
|
// Update the deadline and the entry's position in the LRU list.
|
||||||
if self.update_ttl_on_retrieval {
|
let deadline = now.checked_add(raw_entry.get().ttl).expect("time overflow");
|
||||||
|
if raw_entry.get().update_ttl_on_retrieval {
|
||||||
raw_entry.get_mut().expires_at = deadline;
|
raw_entry.get_mut().expires_at = deadline;
|
||||||
}
|
}
|
||||||
raw_entry.to_back();
|
raw_entry.to_back();
|
||||||
@@ -162,12 +164,27 @@ impl<K: Hash + Eq, V> TimedLru<K, V> {
|
|||||||
/// existed, return the previous value and its creation timestamp.
|
/// existed, return the previous value and its creation timestamp.
|
||||||
#[tracing::instrument(level = "debug", fields(cache = self.name), skip_all)]
|
#[tracing::instrument(level = "debug", fields(cache = self.name), skip_all)]
|
||||||
fn insert_raw(&self, key: K, value: V) -> (Instant, Option<V>) {
|
fn insert_raw(&self, key: K, value: V) -> (Instant, Option<V>) {
|
||||||
|
self.insert_raw_ttl(key, value, self.ttl, self.update_ttl_on_retrieval)
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Insert an entry to the cache. If an entry with the same key already
|
||||||
|
/// existed, return the previous value and its creation timestamp.
|
||||||
|
#[tracing::instrument(level = "debug", fields(cache = self.name), skip_all)]
|
||||||
|
fn insert_raw_ttl(
|
||||||
|
&self,
|
||||||
|
key: K,
|
||||||
|
value: V,
|
||||||
|
ttl: Duration,
|
||||||
|
update: bool,
|
||||||
|
) -> (Instant, Option<V>) {
|
||||||
let created_at = Instant::now();
|
let created_at = Instant::now();
|
||||||
let expires_at = created_at.checked_add(self.ttl).expect("time overflow");
|
let expires_at = created_at.checked_add(ttl).expect("time overflow");
|
||||||
|
|
||||||
let entry = Entry {
|
let entry = Entry {
|
||||||
created_at,
|
created_at,
|
||||||
expires_at,
|
expires_at,
|
||||||
|
ttl,
|
||||||
|
update_ttl_on_retrieval: update,
|
||||||
value,
|
value,
|
||||||
};
|
};
|
||||||
|
|
||||||
@@ -190,6 +207,21 @@ impl<K: Hash + Eq, V> TimedLru<K, V> {
|
|||||||
}
|
}
|
||||||
|
|
||||||
impl<K: Hash + Eq + Clone, V: Clone> TimedLru<K, V> {
|
impl<K: Hash + Eq + Clone, V: Clone> TimedLru<K, V> {
|
||||||
|
pub fn insert_ttl(&self, key: K, value: V, ttl: Duration) {
|
||||||
|
self.insert_raw_ttl(key, value, ttl, false);
|
||||||
|
}
|
||||||
|
|
||||||
|
pub fn insert_unit(&self, key: K, value: V) -> (Option<V>, Cached<&Self, ()>) {
|
||||||
|
let (created_at, old) = self.insert_raw(key.clone(), value);
|
||||||
|
|
||||||
|
let cached = Cached {
|
||||||
|
token: Some((self, LookupInfo { created_at, key })),
|
||||||
|
value: (),
|
||||||
|
};
|
||||||
|
|
||||||
|
(old, cached)
|
||||||
|
}
|
||||||
|
|
||||||
pub fn insert(&self, key: K, value: V) -> (Option<V>, Cached<&Self>) {
|
pub fn insert(&self, key: K, value: V) -> (Option<V>, Cached<&Self>) {
|
||||||
let (created_at, old) = self.insert_raw(key.clone(), value.clone());
|
let (created_at, old) = self.insert_raw(key.clone(), value.clone());
|
||||||
|
|
||||||
|
|||||||
+2
-6
@@ -399,15 +399,11 @@ impl FromStr for EndpointCacheConfig {
|
|||||||
#[derive(Debug)]
|
#[derive(Debug)]
|
||||||
pub struct MetricBackupCollectionConfig {
|
pub struct MetricBackupCollectionConfig {
|
||||||
pub interval: Duration,
|
pub interval: Duration,
|
||||||
pub remote_storage_config: OptRemoteStorageConfig,
|
pub remote_storage_config: Option<RemoteStorageConfig>,
|
||||||
pub chunk_size: usize,
|
pub chunk_size: usize,
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Hack to avoid clap being smarter. If you don't use this type alias, clap assumes more about the optional state and you get
|
pub fn remote_storage_from_toml(s: &str) -> anyhow::Result<RemoteStorageConfig> {
|
||||||
/// runtime type errors from the value parser we use.
|
|
||||||
pub type OptRemoteStorageConfig = Option<RemoteStorageConfig>;
|
|
||||||
|
|
||||||
pub fn remote_storage_from_toml(s: &str) -> anyhow::Result<OptRemoteStorageConfig> {
|
|
||||||
RemoteStorageConfig::from_toml(&s.parse()?)
|
RemoteStorageConfig::from_toml(&s.parse()?)
|
||||||
}
|
}
|
||||||
|
|
||||||
|
|||||||
@@ -9,7 +9,7 @@ use crate::proxy::retry::CouldRetry;
|
|||||||
|
|
||||||
/// Generic error response with human-readable description.
|
/// Generic error response with human-readable description.
|
||||||
/// Note that we can't always present it to user as is.
|
/// Note that we can't always present it to user as is.
|
||||||
#[derive(Debug, Deserialize)]
|
#[derive(Debug, Deserialize, Clone)]
|
||||||
pub struct ConsoleError {
|
pub struct ConsoleError {
|
||||||
pub error: Box<str>,
|
pub error: Box<str>,
|
||||||
#[serde(skip)]
|
#[serde(skip)]
|
||||||
@@ -82,41 +82,19 @@ impl CouldRetry for ConsoleError {
|
|||||||
.details
|
.details
|
||||||
.error_info
|
.error_info
|
||||||
.map_or(Reason::Unknown, |e| e.reason);
|
.map_or(Reason::Unknown, |e| e.reason);
|
||||||
match reason {
|
|
||||||
// not a transitive error
|
reason.can_retry()
|
||||||
Reason::RoleProtected => false,
|
|
||||||
// on retry, it will still not be found
|
|
||||||
Reason::ResourceNotFound
|
|
||||||
| Reason::ProjectNotFound
|
|
||||||
| Reason::EndpointNotFound
|
|
||||||
| Reason::BranchNotFound => false,
|
|
||||||
// we were asked to go away
|
|
||||||
Reason::RateLimitExceeded
|
|
||||||
| Reason::NonDefaultBranchComputeTimeExceeded
|
|
||||||
| Reason::ActiveTimeQuotaExceeded
|
|
||||||
| Reason::ComputeTimeQuotaExceeded
|
|
||||||
| Reason::WrittenDataQuotaExceeded
|
|
||||||
| Reason::DataTransferQuotaExceeded
|
|
||||||
| Reason::LogicalSizeQuotaExceeded => false,
|
|
||||||
// transitive error. control plane is currently busy
|
|
||||||
// but might be ready soon
|
|
||||||
Reason::RunningOperations => true,
|
|
||||||
Reason::ConcurrencyLimitReached => true,
|
|
||||||
Reason::LockAlreadyTaken => true,
|
|
||||||
// unknown error. better not retry it.
|
|
||||||
Reason::Unknown => false,
|
|
||||||
}
|
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
#[derive(Debug, Deserialize)]
|
#[derive(Debug, Deserialize, Clone)]
|
||||||
pub struct Status {
|
pub struct Status {
|
||||||
pub code: Box<str>,
|
pub code: Box<str>,
|
||||||
pub message: Box<str>,
|
pub message: Box<str>,
|
||||||
pub details: Details,
|
pub details: Details,
|
||||||
}
|
}
|
||||||
|
|
||||||
#[derive(Debug, Deserialize)]
|
#[derive(Debug, Deserialize, Clone)]
|
||||||
pub struct Details {
|
pub struct Details {
|
||||||
pub error_info: Option<ErrorInfo>,
|
pub error_info: Option<ErrorInfo>,
|
||||||
pub retry_info: Option<RetryInfo>,
|
pub retry_info: Option<RetryInfo>,
|
||||||
@@ -199,6 +177,34 @@ impl Reason {
|
|||||||
| Reason::BranchNotFound
|
| Reason::BranchNotFound
|
||||||
)
|
)
|
||||||
}
|
}
|
||||||
|
|
||||||
|
pub fn can_retry(&self) -> bool {
|
||||||
|
match self {
|
||||||
|
// do not retry role protected errors
|
||||||
|
// not a transitive error
|
||||||
|
Reason::RoleProtected => false,
|
||||||
|
// on retry, it will still not be found
|
||||||
|
Reason::ResourceNotFound
|
||||||
|
| Reason::ProjectNotFound
|
||||||
|
| Reason::EndpointNotFound
|
||||||
|
| Reason::BranchNotFound => false,
|
||||||
|
// we were asked to go away
|
||||||
|
Reason::RateLimitExceeded
|
||||||
|
| Reason::NonDefaultBranchComputeTimeExceeded
|
||||||
|
| Reason::ActiveTimeQuotaExceeded
|
||||||
|
| Reason::ComputeTimeQuotaExceeded
|
||||||
|
| Reason::WrittenDataQuotaExceeded
|
||||||
|
| Reason::DataTransferQuotaExceeded
|
||||||
|
| Reason::LogicalSizeQuotaExceeded => false,
|
||||||
|
// transitive error. control plane is currently busy
|
||||||
|
// but might be ready soon
|
||||||
|
Reason::RunningOperations
|
||||||
|
| Reason::ConcurrencyLimitReached
|
||||||
|
| Reason::LockAlreadyTaken => true,
|
||||||
|
// unknown error. better not retry it.
|
||||||
|
Reason::Unknown => false,
|
||||||
|
}
|
||||||
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
#[derive(Copy, Clone, Debug, Deserialize)]
|
#[derive(Copy, Clone, Debug, Deserialize)]
|
||||||
@@ -206,7 +212,7 @@ pub struct RetryInfo {
|
|||||||
pub retry_delay_ms: u64,
|
pub retry_delay_ms: u64,
|
||||||
}
|
}
|
||||||
|
|
||||||
#[derive(Debug, Deserialize)]
|
#[derive(Debug, Deserialize, Clone)]
|
||||||
pub struct UserFacingMessage {
|
pub struct UserFacingMessage {
|
||||||
pub message: Box<str>,
|
pub message: Box<str>,
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -2,7 +2,7 @@
|
|||||||
pub mod mock;
|
pub mod mock;
|
||||||
pub mod neon;
|
pub mod neon;
|
||||||
|
|
||||||
use super::messages::MetricsAuxInfo;
|
use super::messages::{ConsoleError, MetricsAuxInfo};
|
||||||
use crate::{
|
use crate::{
|
||||||
auth::{
|
auth::{
|
||||||
backend::{ComputeCredentialKeys, ComputeUserInfo},
|
backend::{ComputeCredentialKeys, ComputeUserInfo},
|
||||||
@@ -317,8 +317,8 @@ impl NodeInfo {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
pub type NodeInfoCache = TimedLru<EndpointCacheKey, NodeInfo>;
|
pub type NodeInfoCache = TimedLru<EndpointCacheKey, Result<NodeInfo, Box<ConsoleError>>>;
|
||||||
pub type CachedNodeInfo = Cached<&'static NodeInfoCache>;
|
pub type CachedNodeInfo = Cached<&'static NodeInfoCache, NodeInfo>;
|
||||||
pub type CachedRoleSecret = Cached<&'static ProjectInfoCacheImpl, Option<AuthSecret>>;
|
pub type CachedRoleSecret = Cached<&'static ProjectInfoCacheImpl, Option<AuthSecret>>;
|
||||||
pub type CachedAllowedIps = Cached<&'static ProjectInfoCacheImpl, Arc<Vec<IpPattern>>>;
|
pub type CachedAllowedIps = Cached<&'static ProjectInfoCacheImpl, Arc<Vec<IpPattern>>>;
|
||||||
|
|
||||||
|
|||||||
@@ -9,7 +9,7 @@ use super::{
|
|||||||
use crate::{
|
use crate::{
|
||||||
auth::backend::ComputeUserInfo,
|
auth::backend::ComputeUserInfo,
|
||||||
compute,
|
compute,
|
||||||
console::messages::ColdStartInfo,
|
console::messages::{ColdStartInfo, Reason},
|
||||||
http,
|
http,
|
||||||
metrics::{CacheOutcome, Metrics},
|
metrics::{CacheOutcome, Metrics},
|
||||||
rate_limiter::EndpointRateLimiter,
|
rate_limiter::EndpointRateLimiter,
|
||||||
@@ -17,10 +17,10 @@ use crate::{
|
|||||||
};
|
};
|
||||||
use crate::{cache::Cached, context::RequestMonitoring};
|
use crate::{cache::Cached, context::RequestMonitoring};
|
||||||
use futures::TryFutureExt;
|
use futures::TryFutureExt;
|
||||||
use std::sync::Arc;
|
use std::{sync::Arc, time::Duration};
|
||||||
use tokio::time::Instant;
|
use tokio::time::Instant;
|
||||||
use tokio_postgres::config::SslMode;
|
use tokio_postgres::config::SslMode;
|
||||||
use tracing::{error, info, info_span, warn, Instrument};
|
use tracing::{debug, error, info, info_span, warn, Instrument};
|
||||||
|
|
||||||
pub struct Api {
|
pub struct Api {
|
||||||
endpoint: http::Endpoint,
|
endpoint: http::Endpoint,
|
||||||
@@ -273,26 +273,34 @@ impl super::Api for Api {
|
|||||||
) -> Result<CachedNodeInfo, WakeComputeError> {
|
) -> Result<CachedNodeInfo, WakeComputeError> {
|
||||||
let key = user_info.endpoint_cache_key();
|
let key = user_info.endpoint_cache_key();
|
||||||
|
|
||||||
|
macro_rules! check_cache {
|
||||||
|
() => {
|
||||||
|
if let Some(cached) = self.caches.node_info.get(&key) {
|
||||||
|
let (cached, info) = cached.take_value();
|
||||||
|
let info = info.map_err(|c| {
|
||||||
|
info!(key = &*key, "found cached wake_compute error");
|
||||||
|
WakeComputeError::ApiError(ApiError::Console(*c))
|
||||||
|
})?;
|
||||||
|
|
||||||
|
debug!(key = &*key, "found cached compute node info");
|
||||||
|
ctx.set_project(info.aux.clone());
|
||||||
|
return Ok(cached.map(|()| info));
|
||||||
|
}
|
||||||
|
};
|
||||||
|
}
|
||||||
|
|
||||||
// Every time we do a wakeup http request, the compute node will stay up
|
// Every time we do a wakeup http request, the compute node will stay up
|
||||||
// for some time (highly depends on the console's scale-to-zero policy);
|
// for some time (highly depends on the console's scale-to-zero policy);
|
||||||
// The connection info remains the same during that period of time,
|
// The connection info remains the same during that period of time,
|
||||||
// which means that we might cache it to reduce the load and latency.
|
// which means that we might cache it to reduce the load and latency.
|
||||||
if let Some(cached) = self.caches.node_info.get(&key) {
|
check_cache!();
|
||||||
info!(key = &*key, "found cached compute node info");
|
|
||||||
ctx.set_project(cached.aux.clone());
|
|
||||||
return Ok(cached);
|
|
||||||
}
|
|
||||||
|
|
||||||
let permit = self.locks.get_permit(&key).await?;
|
let permit = self.locks.get_permit(&key).await?;
|
||||||
|
|
||||||
// after getting back a permit - it's possible the cache was filled
|
// after getting back a permit - it's possible the cache was filled
|
||||||
// double check
|
// double check
|
||||||
if permit.should_check_cache() {
|
if permit.should_check_cache() {
|
||||||
if let Some(cached) = self.caches.node_info.get(&key) {
|
check_cache!();
|
||||||
info!(key = &*key, "found cached compute node info");
|
|
||||||
ctx.set_project(cached.aux.clone());
|
|
||||||
return Ok(cached);
|
|
||||||
}
|
|
||||||
}
|
}
|
||||||
|
|
||||||
// check rate limit
|
// check rate limit
|
||||||
@@ -300,23 +308,56 @@ impl super::Api for Api {
|
|||||||
.wake_compute_endpoint_rate_limiter
|
.wake_compute_endpoint_rate_limiter
|
||||||
.check(user_info.endpoint.normalize_intern(), 1)
|
.check(user_info.endpoint.normalize_intern(), 1)
|
||||||
{
|
{
|
||||||
info!(key = &*key, "found cached compute node info");
|
|
||||||
return Err(WakeComputeError::TooManyConnections);
|
return Err(WakeComputeError::TooManyConnections);
|
||||||
}
|
}
|
||||||
|
|
||||||
let mut node = permit.release_result(self.do_wake_compute(ctx, user_info).await)?;
|
let node = permit.release_result(self.do_wake_compute(ctx, user_info).await);
|
||||||
ctx.set_project(node.aux.clone());
|
match node {
|
||||||
let cold_start_info = node.aux.cold_start_info;
|
Ok(node) => {
|
||||||
info!("woken up a compute node");
|
ctx.set_project(node.aux.clone());
|
||||||
|
debug!(key = &*key, "created a cache entry for woken compute node");
|
||||||
|
|
||||||
// store the cached node as 'warm'
|
let mut stored_node = node.clone();
|
||||||
node.aux.cold_start_info = ColdStartInfo::WarmCached;
|
// store the cached node as 'warm_cached'
|
||||||
let (_, mut cached) = self.caches.node_info.insert(key.clone(), node);
|
stored_node.aux.cold_start_info = ColdStartInfo::WarmCached;
|
||||||
cached.aux.cold_start_info = cold_start_info;
|
|
||||||
|
|
||||||
info!(key = &*key, "created a cache entry for compute node info");
|
let (_, cached) = self.caches.node_info.insert_unit(key, Ok(stored_node));
|
||||||
|
|
||||||
Ok(cached)
|
Ok(cached.map(|()| node))
|
||||||
|
}
|
||||||
|
Err(err) => match err {
|
||||||
|
WakeComputeError::ApiError(ApiError::Console(err)) => {
|
||||||
|
let Some(status) = &err.status else {
|
||||||
|
return Err(WakeComputeError::ApiError(ApiError::Console(err)));
|
||||||
|
};
|
||||||
|
|
||||||
|
let reason = status
|
||||||
|
.details
|
||||||
|
.error_info
|
||||||
|
.map_or(Reason::Unknown, |x| x.reason);
|
||||||
|
|
||||||
|
// if we can retry this error, do not cache it.
|
||||||
|
if reason.can_retry() {
|
||||||
|
return Err(WakeComputeError::ApiError(ApiError::Console(err)));
|
||||||
|
}
|
||||||
|
|
||||||
|
// at this point, we should only have quota errors.
|
||||||
|
debug!(
|
||||||
|
key = &*key,
|
||||||
|
"created a cache entry for the wake compute error"
|
||||||
|
);
|
||||||
|
|
||||||
|
self.caches.node_info.insert_ttl(
|
||||||
|
key,
|
||||||
|
Err(Box::new(err.clone())),
|
||||||
|
Duration::from_secs(30),
|
||||||
|
);
|
||||||
|
|
||||||
|
Err(WakeComputeError::ApiError(ApiError::Console(err)))
|
||||||
|
}
|
||||||
|
err => return Err(err),
|
||||||
|
},
|
||||||
|
}
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
|||||||
@@ -14,17 +14,14 @@ use parquet::{
|
|||||||
record::RecordWriter,
|
record::RecordWriter,
|
||||||
};
|
};
|
||||||
use pq_proto::StartupMessageParams;
|
use pq_proto::StartupMessageParams;
|
||||||
use remote_storage::{GenericRemoteStorage, RemotePath, TimeoutOrCancel};
|
use remote_storage::{GenericRemoteStorage, RemotePath, RemoteStorageConfig, TimeoutOrCancel};
|
||||||
use serde::ser::SerializeMap;
|
use serde::ser::SerializeMap;
|
||||||
use tokio::{sync::mpsc, time};
|
use tokio::{sync::mpsc, time};
|
||||||
use tokio_util::sync::CancellationToken;
|
use tokio_util::sync::CancellationToken;
|
||||||
use tracing::{debug, info, Span};
|
use tracing::{debug, info, Span};
|
||||||
use utils::backoff;
|
use utils::backoff;
|
||||||
|
|
||||||
use crate::{
|
use crate::{config::remote_storage_from_toml, context::LOG_CHAN_DISCONNECT};
|
||||||
config::{remote_storage_from_toml, OptRemoteStorageConfig},
|
|
||||||
context::LOG_CHAN_DISCONNECT,
|
|
||||||
};
|
|
||||||
|
|
||||||
use super::{RequestMonitoring, LOG_CHAN};
|
use super::{RequestMonitoring, LOG_CHAN};
|
||||||
|
|
||||||
@@ -33,11 +30,11 @@ pub struct ParquetUploadArgs {
|
|||||||
/// Storage location to upload the parquet files to.
|
/// Storage location to upload the parquet files to.
|
||||||
/// Encoded as toml (same format as pageservers), eg
|
/// Encoded as toml (same format as pageservers), eg
|
||||||
/// `{bucket_name='the-bucket',bucket_region='us-east-1',prefix_in_bucket='proxy',endpoint='http://minio:9000'}`
|
/// `{bucket_name='the-bucket',bucket_region='us-east-1',prefix_in_bucket='proxy',endpoint='http://minio:9000'}`
|
||||||
#[clap(long, default_value = "{}", value_parser = remote_storage_from_toml)]
|
#[clap(long, value_parser = remote_storage_from_toml)]
|
||||||
parquet_upload_remote_storage: OptRemoteStorageConfig,
|
parquet_upload_remote_storage: Option<RemoteStorageConfig>,
|
||||||
|
|
||||||
#[clap(long, default_value = "{}", value_parser = remote_storage_from_toml)]
|
#[clap(long, value_parser = remote_storage_from_toml)]
|
||||||
parquet_upload_disconnect_events_remote_storage: OptRemoteStorageConfig,
|
parquet_upload_disconnect_events_remote_storage: Option<RemoteStorageConfig>,
|
||||||
|
|
||||||
/// How many rows to include in a row group
|
/// How many rows to include in a row group
|
||||||
#[clap(long, default_value_t = 8192)]
|
#[clap(long, default_value_t = 8192)]
|
||||||
|
|||||||
@@ -540,8 +540,8 @@ fn helper_create_cached_node_info(cache: &'static NodeInfoCache) -> CachedNodeIn
|
|||||||
},
|
},
|
||||||
allow_self_signed_compute: false,
|
allow_self_signed_compute: false,
|
||||||
};
|
};
|
||||||
let (_, node) = cache.insert("key".into(), node);
|
let (_, node2) = cache.insert_unit("key".into(), Ok(node.clone()));
|
||||||
node
|
node2.map(|()| node)
|
||||||
}
|
}
|
||||||
|
|
||||||
fn helper_create_connect_info(
|
fn helper_create_connect_info(
|
||||||
|
|||||||
@@ -12,7 +12,6 @@ use sd_notify::NotifyState;
|
|||||||
use tokio::runtime::Handle;
|
use tokio::runtime::Handle;
|
||||||
use tokio::signal::unix::{signal, SignalKind};
|
use tokio::signal::unix::{signal, SignalKind};
|
||||||
use tokio::task::JoinError;
|
use tokio::task::JoinError;
|
||||||
use toml_edit::Document;
|
|
||||||
use utils::logging::SecretString;
|
use utils::logging::SecretString;
|
||||||
|
|
||||||
use std::env::{var, VarError};
|
use std::env::{var, VarError};
|
||||||
@@ -29,7 +28,8 @@ use utils::pid_file;
|
|||||||
use metrics::set_build_info_metric;
|
use metrics::set_build_info_metric;
|
||||||
use safekeeper::defaults::{
|
use safekeeper::defaults::{
|
||||||
DEFAULT_CONTROL_FILE_SAVE_INTERVAL, DEFAULT_HEARTBEAT_TIMEOUT, DEFAULT_HTTP_LISTEN_ADDR,
|
DEFAULT_CONTROL_FILE_SAVE_INTERVAL, DEFAULT_HEARTBEAT_TIMEOUT, DEFAULT_HTTP_LISTEN_ADDR,
|
||||||
DEFAULT_MAX_OFFLOADER_LAG_BYTES, DEFAULT_PARTIAL_BACKUP_TIMEOUT, DEFAULT_PG_LISTEN_ADDR,
|
DEFAULT_MAX_OFFLOADER_LAG_BYTES, DEFAULT_PARTIAL_BACKUP_CONCURRENCY,
|
||||||
|
DEFAULT_PARTIAL_BACKUP_TIMEOUT, DEFAULT_PG_LISTEN_ADDR,
|
||||||
};
|
};
|
||||||
use safekeeper::http;
|
use safekeeper::http;
|
||||||
use safekeeper::wal_service;
|
use safekeeper::wal_service;
|
||||||
@@ -125,7 +125,7 @@ struct Args {
|
|||||||
peer_recovery: bool,
|
peer_recovery: bool,
|
||||||
/// Remote storage configuration for WAL backup (offloading to s3) as TOML
|
/// Remote storage configuration for WAL backup (offloading to s3) as TOML
|
||||||
/// inline table, e.g.
|
/// inline table, e.g.
|
||||||
/// {"max_concurrent_syncs" = 17, "max_sync_errors": 13, "bucket_name": "<BUCKETNAME>", "bucket_region":"<REGION>", "concurrency_limit": 119}
|
/// {max_concurrent_syncs = 17, max_sync_errors = 13, bucket_name = "<BUCKETNAME>", bucket_region = "<REGION>", concurrency_limit = 119}
|
||||||
/// Safekeeper offloads WAL to
|
/// Safekeeper offloads WAL to
|
||||||
/// [prefix_in_bucket/]<tenant_id>/<timeline_id>/<segment_file>, mirroring
|
/// [prefix_in_bucket/]<tenant_id>/<timeline_id>/<segment_file>, mirroring
|
||||||
/// structure on the file system.
|
/// structure on the file system.
|
||||||
@@ -191,6 +191,9 @@ struct Args {
|
|||||||
/// Pending updates to control file will be automatically saved after this interval.
|
/// Pending updates to control file will be automatically saved after this interval.
|
||||||
#[arg(long, value_parser = humantime::parse_duration, default_value = DEFAULT_CONTROL_FILE_SAVE_INTERVAL)]
|
#[arg(long, value_parser = humantime::parse_duration, default_value = DEFAULT_CONTROL_FILE_SAVE_INTERVAL)]
|
||||||
control_file_save_interval: Duration,
|
control_file_save_interval: Duration,
|
||||||
|
/// Number of allowed concurrent uploads of partial segments to remote storage.
|
||||||
|
#[arg(long, default_value = DEFAULT_PARTIAL_BACKUP_CONCURRENCY)]
|
||||||
|
partial_backup_concurrency: usize,
|
||||||
}
|
}
|
||||||
|
|
||||||
// Like PathBufValueParser, but allows empty string.
|
// Like PathBufValueParser, but allows empty string.
|
||||||
@@ -344,6 +347,7 @@ async fn main() -> anyhow::Result<()> {
|
|||||||
enable_offload: args.enable_offload,
|
enable_offload: args.enable_offload,
|
||||||
delete_offloaded_wal: args.delete_offloaded_wal,
|
delete_offloaded_wal: args.delete_offloaded_wal,
|
||||||
control_file_save_interval: args.control_file_save_interval,
|
control_file_save_interval: args.control_file_save_interval,
|
||||||
|
partial_backup_concurrency: args.partial_backup_concurrency,
|
||||||
};
|
};
|
||||||
|
|
||||||
// initialize sentry if SENTRY_DSN is provided
|
// initialize sentry if SENTRY_DSN is provided
|
||||||
@@ -441,6 +445,19 @@ async fn start_safekeeper(conf: SafeKeeperConf) -> Result<()> {
|
|||||||
.map(|res| ("WAL service main".to_owned(), res));
|
.map(|res| ("WAL service main".to_owned(), res));
|
||||||
tasks_handles.push(Box::pin(wal_service_handle));
|
tasks_handles.push(Box::pin(wal_service_handle));
|
||||||
|
|
||||||
|
let timeline_housekeeping_handle = current_thread_rt
|
||||||
|
.as_ref()
|
||||||
|
.unwrap_or_else(|| WAL_SERVICE_RUNTIME.handle())
|
||||||
|
.spawn(async move {
|
||||||
|
const TOMBSTONE_TTL: Duration = Duration::from_secs(3600 * 24);
|
||||||
|
loop {
|
||||||
|
tokio::time::sleep(TOMBSTONE_TTL).await;
|
||||||
|
GlobalTimelines::housekeeping(&TOMBSTONE_TTL);
|
||||||
|
}
|
||||||
|
})
|
||||||
|
.map(|res| ("Timeline map housekeeping".to_owned(), res));
|
||||||
|
tasks_handles.push(Box::pin(timeline_housekeeping_handle));
|
||||||
|
|
||||||
if let Some(pg_listener_tenant_only) = pg_listener_tenant_only {
|
if let Some(pg_listener_tenant_only) = pg_listener_tenant_only {
|
||||||
let conf_ = conf.clone();
|
let conf_ = conf.clone();
|
||||||
let wal_service_handle = current_thread_rt
|
let wal_service_handle = current_thread_rt
|
||||||
@@ -548,16 +565,8 @@ fn set_id(workdir: &Utf8Path, given_id: Option<NodeId>) -> Result<NodeId> {
|
|||||||
Ok(my_id)
|
Ok(my_id)
|
||||||
}
|
}
|
||||||
|
|
||||||
// Parse RemoteStorage from TOML table.
|
|
||||||
fn parse_remote_storage(storage_conf: &str) -> anyhow::Result<RemoteStorageConfig> {
|
fn parse_remote_storage(storage_conf: &str) -> anyhow::Result<RemoteStorageConfig> {
|
||||||
// funny toml doesn't consider plain inline table as valid document, so wrap in a key to parse
|
RemoteStorageConfig::from_toml(&storage_conf.parse()?)
|
||||||
let storage_conf_toml = format!("remote_storage = {storage_conf}");
|
|
||||||
let parsed_toml = storage_conf_toml.parse::<Document>()?; // parse
|
|
||||||
let (_, storage_conf_parsed_toml) = parsed_toml.iter().next().unwrap(); // and strip key off again
|
|
||||||
RemoteStorageConfig::from_toml(storage_conf_parsed_toml).and_then(|parsed_config| {
|
|
||||||
// XXX: Don't print the original toml here, there might be some sensitive data
|
|
||||||
parsed_config.context("Incorrectly parsed remote storage toml as no remote storage config")
|
|
||||||
})
|
|
||||||
}
|
}
|
||||||
|
|
||||||
#[test]
|
#[test]
|
||||||
|
|||||||
@@ -52,6 +52,7 @@ pub mod defaults {
|
|||||||
pub const DEFAULT_MAX_OFFLOADER_LAG_BYTES: u64 = 128 * (1 << 20);
|
pub const DEFAULT_MAX_OFFLOADER_LAG_BYTES: u64 = 128 * (1 << 20);
|
||||||
pub const DEFAULT_PARTIAL_BACKUP_TIMEOUT: &str = "15m";
|
pub const DEFAULT_PARTIAL_BACKUP_TIMEOUT: &str = "15m";
|
||||||
pub const DEFAULT_CONTROL_FILE_SAVE_INTERVAL: &str = "300s";
|
pub const DEFAULT_CONTROL_FILE_SAVE_INTERVAL: &str = "300s";
|
||||||
|
pub const DEFAULT_PARTIAL_BACKUP_CONCURRENCY: &str = "5";
|
||||||
}
|
}
|
||||||
|
|
||||||
#[derive(Debug, Clone)]
|
#[derive(Debug, Clone)]
|
||||||
@@ -91,6 +92,7 @@ pub struct SafeKeeperConf {
|
|||||||
pub enable_offload: bool,
|
pub enable_offload: bool,
|
||||||
pub delete_offloaded_wal: bool,
|
pub delete_offloaded_wal: bool,
|
||||||
pub control_file_save_interval: Duration,
|
pub control_file_save_interval: Duration,
|
||||||
|
pub partial_backup_concurrency: usize,
|
||||||
}
|
}
|
||||||
|
|
||||||
impl SafeKeeperConf {
|
impl SafeKeeperConf {
|
||||||
@@ -133,6 +135,7 @@ impl SafeKeeperConf {
|
|||||||
enable_offload: false,
|
enable_offload: false,
|
||||||
delete_offloaded_wal: false,
|
delete_offloaded_wal: false,
|
||||||
control_file_save_interval: Duration::from_secs(1),
|
control_file_save_interval: Duration::from_secs(1),
|
||||||
|
partial_backup_concurrency: 1,
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -72,7 +72,8 @@ pub static WAL_STORAGE_OPERATION_SECONDS: Lazy<HistogramVec> = Lazy::new(|| {
|
|||||||
register_histogram_vec!(
|
register_histogram_vec!(
|
||||||
"safekeeper_wal_storage_operation_seconds",
|
"safekeeper_wal_storage_operation_seconds",
|
||||||
"Seconds spent on WAL storage operations",
|
"Seconds spent on WAL storage operations",
|
||||||
&["operation"]
|
&["operation"],
|
||||||
|
DISK_FSYNC_SECONDS_BUCKETS.to_vec()
|
||||||
)
|
)
|
||||||
.expect("Failed to register safekeeper_wal_storage_operation_seconds histogram vec")
|
.expect("Failed to register safekeeper_wal_storage_operation_seconds histogram vec")
|
||||||
});
|
});
|
||||||
@@ -80,7 +81,8 @@ pub static MISC_OPERATION_SECONDS: Lazy<HistogramVec> = Lazy::new(|| {
|
|||||||
register_histogram_vec!(
|
register_histogram_vec!(
|
||||||
"safekeeper_misc_operation_seconds",
|
"safekeeper_misc_operation_seconds",
|
||||||
"Seconds spent on miscellaneous operations",
|
"Seconds spent on miscellaneous operations",
|
||||||
&["operation"]
|
&["operation"],
|
||||||
|
DISK_FSYNC_SECONDS_BUCKETS.to_vec()
|
||||||
)
|
)
|
||||||
.expect("Failed to register safekeeper_misc_operation_seconds histogram vec")
|
.expect("Failed to register safekeeper_misc_operation_seconds histogram vec")
|
||||||
});
|
});
|
||||||
|
|||||||
@@ -36,7 +36,7 @@ use crate::timeline_guard::ResidenceGuard;
|
|||||||
use crate::timeline_manager::{AtomicStatus, ManagerCtl};
|
use crate::timeline_manager::{AtomicStatus, ManagerCtl};
|
||||||
use crate::timelines_set::TimelinesSet;
|
use crate::timelines_set::TimelinesSet;
|
||||||
use crate::wal_backup::{self};
|
use crate::wal_backup::{self};
|
||||||
use crate::wal_backup_partial::PartialRemoteSegment;
|
use crate::wal_backup_partial::{PartialRemoteSegment, RateLimiter};
|
||||||
use crate::{control_file, safekeeper::UNKNOWN_SERVER_VERSION};
|
use crate::{control_file, safekeeper::UNKNOWN_SERVER_VERSION};
|
||||||
|
|
||||||
use crate::metrics::{FullTimelineInfo, WalStorageMetrics, MISC_OPERATION_SECONDS};
|
use crate::metrics::{FullTimelineInfo, WalStorageMetrics, MISC_OPERATION_SECONDS};
|
||||||
@@ -587,6 +587,7 @@ impl Timeline {
|
|||||||
shared_state: &mut WriteGuardSharedState<'_>,
|
shared_state: &mut WriteGuardSharedState<'_>,
|
||||||
conf: &SafeKeeperConf,
|
conf: &SafeKeeperConf,
|
||||||
broker_active_set: Arc<TimelinesSet>,
|
broker_active_set: Arc<TimelinesSet>,
|
||||||
|
partial_backup_rate_limiter: RateLimiter,
|
||||||
) -> Result<()> {
|
) -> Result<()> {
|
||||||
match fs::metadata(&self.timeline_dir).await {
|
match fs::metadata(&self.timeline_dir).await {
|
||||||
Ok(_) => {
|
Ok(_) => {
|
||||||
@@ -617,7 +618,7 @@ impl Timeline {
|
|||||||
|
|
||||||
return Err(e);
|
return Err(e);
|
||||||
}
|
}
|
||||||
self.bootstrap(conf, broker_active_set);
|
self.bootstrap(conf, broker_active_set, partial_backup_rate_limiter);
|
||||||
Ok(())
|
Ok(())
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -626,6 +627,7 @@ impl Timeline {
|
|||||||
self: &Arc<Timeline>,
|
self: &Arc<Timeline>,
|
||||||
conf: &SafeKeeperConf,
|
conf: &SafeKeeperConf,
|
||||||
broker_active_set: Arc<TimelinesSet>,
|
broker_active_set: Arc<TimelinesSet>,
|
||||||
|
partial_backup_rate_limiter: RateLimiter,
|
||||||
) {
|
) {
|
||||||
let (tx, rx) = self.manager_ctl.bootstrap_manager();
|
let (tx, rx) = self.manager_ctl.bootstrap_manager();
|
||||||
|
|
||||||
@@ -637,6 +639,7 @@ impl Timeline {
|
|||||||
broker_active_set,
|
broker_active_set,
|
||||||
tx,
|
tx,
|
||||||
rx,
|
rx,
|
||||||
|
partial_backup_rate_limiter,
|
||||||
));
|
));
|
||||||
}
|
}
|
||||||
|
|
||||||
|
|||||||
@@ -32,7 +32,7 @@ use crate::{
|
|||||||
timeline_guard::{AccessService, GuardId, ResidenceGuard},
|
timeline_guard::{AccessService, GuardId, ResidenceGuard},
|
||||||
timelines_set::{TimelineSetGuard, TimelinesSet},
|
timelines_set::{TimelineSetGuard, TimelinesSet},
|
||||||
wal_backup::{self, WalBackupTaskHandle},
|
wal_backup::{self, WalBackupTaskHandle},
|
||||||
wal_backup_partial::{self, PartialRemoteSegment},
|
wal_backup_partial::{self, PartialRemoteSegment, RateLimiter},
|
||||||
SafeKeeperConf,
|
SafeKeeperConf,
|
||||||
};
|
};
|
||||||
|
|
||||||
@@ -185,6 +185,7 @@ pub(crate) struct Manager {
|
|||||||
|
|
||||||
// misc
|
// misc
|
||||||
pub(crate) access_service: AccessService,
|
pub(crate) access_service: AccessService,
|
||||||
|
pub(crate) partial_backup_rate_limiter: RateLimiter,
|
||||||
}
|
}
|
||||||
|
|
||||||
/// This task gets spawned alongside each timeline and is responsible for managing the timeline's
|
/// This task gets spawned alongside each timeline and is responsible for managing the timeline's
|
||||||
@@ -197,6 +198,7 @@ pub async fn main_task(
|
|||||||
broker_active_set: Arc<TimelinesSet>,
|
broker_active_set: Arc<TimelinesSet>,
|
||||||
manager_tx: tokio::sync::mpsc::UnboundedSender<ManagerCtlMessage>,
|
manager_tx: tokio::sync::mpsc::UnboundedSender<ManagerCtlMessage>,
|
||||||
mut manager_rx: tokio::sync::mpsc::UnboundedReceiver<ManagerCtlMessage>,
|
mut manager_rx: tokio::sync::mpsc::UnboundedReceiver<ManagerCtlMessage>,
|
||||||
|
partial_backup_rate_limiter: RateLimiter,
|
||||||
) {
|
) {
|
||||||
tli.set_status(Status::Started);
|
tli.set_status(Status::Started);
|
||||||
|
|
||||||
@@ -209,7 +211,14 @@ pub async fn main_task(
|
|||||||
}
|
}
|
||||||
};
|
};
|
||||||
|
|
||||||
let mut mgr = Manager::new(tli, conf, broker_active_set, manager_tx).await;
|
let mut mgr = Manager::new(
|
||||||
|
tli,
|
||||||
|
conf,
|
||||||
|
broker_active_set,
|
||||||
|
manager_tx,
|
||||||
|
partial_backup_rate_limiter,
|
||||||
|
)
|
||||||
|
.await;
|
||||||
|
|
||||||
// Start recovery task which always runs on the timeline.
|
// Start recovery task which always runs on the timeline.
|
||||||
if !mgr.is_offloaded && mgr.conf.peer_recovery_enabled {
|
if !mgr.is_offloaded && mgr.conf.peer_recovery_enabled {
|
||||||
@@ -321,6 +330,7 @@ impl Manager {
|
|||||||
conf: SafeKeeperConf,
|
conf: SafeKeeperConf,
|
||||||
broker_active_set: Arc<TimelinesSet>,
|
broker_active_set: Arc<TimelinesSet>,
|
||||||
manager_tx: tokio::sync::mpsc::UnboundedSender<ManagerCtlMessage>,
|
manager_tx: tokio::sync::mpsc::UnboundedSender<ManagerCtlMessage>,
|
||||||
|
partial_backup_rate_limiter: RateLimiter,
|
||||||
) -> Manager {
|
) -> Manager {
|
||||||
let (is_offloaded, partial_backup_uploaded) = tli.bootstrap_mgr().await;
|
let (is_offloaded, partial_backup_uploaded) = tli.bootstrap_mgr().await;
|
||||||
Manager {
|
Manager {
|
||||||
@@ -339,6 +349,7 @@ impl Manager {
|
|||||||
partial_backup_uploaded,
|
partial_backup_uploaded,
|
||||||
access_service: AccessService::new(manager_tx),
|
access_service: AccessService::new(manager_tx),
|
||||||
tli,
|
tli,
|
||||||
|
partial_backup_rate_limiter,
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -525,6 +536,7 @@ impl Manager {
|
|||||||
self.partial_backup_task = Some(tokio::spawn(wal_backup_partial::main_task(
|
self.partial_backup_task = Some(tokio::spawn(wal_backup_partial::main_task(
|
||||||
self.wal_resident_timeline(),
|
self.wal_resident_timeline(),
|
||||||
self.conf.clone(),
|
self.conf.clone(),
|
||||||
|
self.partial_backup_rate_limiter.clone(),
|
||||||
)));
|
)));
|
||||||
}
|
}
|
||||||
|
|
||||||
|
|||||||
@@ -5,6 +5,7 @@
|
|||||||
use crate::safekeeper::ServerInfo;
|
use crate::safekeeper::ServerInfo;
|
||||||
use crate::timeline::{get_tenant_dir, get_timeline_dir, Timeline, TimelineError};
|
use crate::timeline::{get_tenant_dir, get_timeline_dir, Timeline, TimelineError};
|
||||||
use crate::timelines_set::TimelinesSet;
|
use crate::timelines_set::TimelinesSet;
|
||||||
|
use crate::wal_backup_partial::RateLimiter;
|
||||||
use crate::SafeKeeperConf;
|
use crate::SafeKeeperConf;
|
||||||
use anyhow::{bail, Context, Result};
|
use anyhow::{bail, Context, Result};
|
||||||
use camino::Utf8PathBuf;
|
use camino::Utf8PathBuf;
|
||||||
@@ -14,15 +15,23 @@ use std::collections::HashMap;
|
|||||||
use std::str::FromStr;
|
use std::str::FromStr;
|
||||||
use std::sync::atomic::Ordering;
|
use std::sync::atomic::Ordering;
|
||||||
use std::sync::{Arc, Mutex};
|
use std::sync::{Arc, Mutex};
|
||||||
|
use std::time::{Duration, Instant};
|
||||||
use tracing::*;
|
use tracing::*;
|
||||||
use utils::id::{TenantId, TenantTimelineId, TimelineId};
|
use utils::id::{TenantId, TenantTimelineId, TimelineId};
|
||||||
use utils::lsn::Lsn;
|
use utils::lsn::Lsn;
|
||||||
|
|
||||||
struct GlobalTimelinesState {
|
struct GlobalTimelinesState {
|
||||||
timelines: HashMap<TenantTimelineId, Arc<Timeline>>,
|
timelines: HashMap<TenantTimelineId, Arc<Timeline>>,
|
||||||
|
|
||||||
|
// A tombstone indicates this timeline used to exist has been deleted. These are used to prevent
|
||||||
|
// on-demand timeline creation from recreating deleted timelines. This is only soft-enforced, as
|
||||||
|
// this map is dropped on restart.
|
||||||
|
tombstones: HashMap<TenantTimelineId, Instant>,
|
||||||
|
|
||||||
conf: Option<SafeKeeperConf>,
|
conf: Option<SafeKeeperConf>,
|
||||||
broker_active_set: Arc<TimelinesSet>,
|
broker_active_set: Arc<TimelinesSet>,
|
||||||
load_lock: Arc<tokio::sync::Mutex<TimelineLoadLock>>,
|
load_lock: Arc<tokio::sync::Mutex<TimelineLoadLock>>,
|
||||||
|
partial_backup_rate_limiter: RateLimiter,
|
||||||
}
|
}
|
||||||
|
|
||||||
// Used to prevent concurrent timeline loading.
|
// Used to prevent concurrent timeline loading.
|
||||||
@@ -37,8 +46,12 @@ impl GlobalTimelinesState {
|
|||||||
}
|
}
|
||||||
|
|
||||||
/// Get dependencies for a timeline constructor.
|
/// Get dependencies for a timeline constructor.
|
||||||
fn get_dependencies(&self) -> (SafeKeeperConf, Arc<TimelinesSet>) {
|
fn get_dependencies(&self) -> (SafeKeeperConf, Arc<TimelinesSet>, RateLimiter) {
|
||||||
(self.get_conf().clone(), self.broker_active_set.clone())
|
(
|
||||||
|
self.get_conf().clone(),
|
||||||
|
self.broker_active_set.clone(),
|
||||||
|
self.partial_backup_rate_limiter.clone(),
|
||||||
|
)
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Insert timeline into the map. Returns error if timeline with the same id already exists.
|
/// Insert timeline into the map. Returns error if timeline with the same id already exists.
|
||||||
@@ -58,14 +71,21 @@ impl GlobalTimelinesState {
|
|||||||
.cloned()
|
.cloned()
|
||||||
.ok_or(TimelineError::NotFound(*ttid))
|
.ok_or(TimelineError::NotFound(*ttid))
|
||||||
}
|
}
|
||||||
|
|
||||||
|
fn delete(&mut self, ttid: TenantTimelineId) {
|
||||||
|
self.timelines.remove(&ttid);
|
||||||
|
self.tombstones.insert(ttid, Instant::now());
|
||||||
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
static TIMELINES_STATE: Lazy<Mutex<GlobalTimelinesState>> = Lazy::new(|| {
|
static TIMELINES_STATE: Lazy<Mutex<GlobalTimelinesState>> = Lazy::new(|| {
|
||||||
Mutex::new(GlobalTimelinesState {
|
Mutex::new(GlobalTimelinesState {
|
||||||
timelines: HashMap::new(),
|
timelines: HashMap::new(),
|
||||||
|
tombstones: HashMap::new(),
|
||||||
conf: None,
|
conf: None,
|
||||||
broker_active_set: Arc::new(TimelinesSet::default()),
|
broker_active_set: Arc::new(TimelinesSet::default()),
|
||||||
load_lock: Arc::new(tokio::sync::Mutex::new(TimelineLoadLock)),
|
load_lock: Arc::new(tokio::sync::Mutex::new(TimelineLoadLock)),
|
||||||
|
partial_backup_rate_limiter: RateLimiter::new(1),
|
||||||
})
|
})
|
||||||
});
|
});
|
||||||
|
|
||||||
@@ -79,6 +99,7 @@ impl GlobalTimelines {
|
|||||||
// lock, so use explicit block
|
// lock, so use explicit block
|
||||||
let tenants_dir = {
|
let tenants_dir = {
|
||||||
let mut state = TIMELINES_STATE.lock().unwrap();
|
let mut state = TIMELINES_STATE.lock().unwrap();
|
||||||
|
state.partial_backup_rate_limiter = RateLimiter::new(conf.partial_backup_concurrency);
|
||||||
state.conf = Some(conf);
|
state.conf = Some(conf);
|
||||||
|
|
||||||
// Iterate through all directories and load tenants for all directories
|
// Iterate through all directories and load tenants for all directories
|
||||||
@@ -122,7 +143,7 @@ impl GlobalTimelines {
|
|||||||
/// this function is called during init when nothing else is running, so
|
/// this function is called during init when nothing else is running, so
|
||||||
/// this is fine.
|
/// this is fine.
|
||||||
async fn load_tenant_timelines(tenant_id: TenantId) -> Result<()> {
|
async fn load_tenant_timelines(tenant_id: TenantId) -> Result<()> {
|
||||||
let (conf, broker_active_set) = {
|
let (conf, broker_active_set, partial_backup_rate_limiter) = {
|
||||||
let state = TIMELINES_STATE.lock().unwrap();
|
let state = TIMELINES_STATE.lock().unwrap();
|
||||||
state.get_dependencies()
|
state.get_dependencies()
|
||||||
};
|
};
|
||||||
@@ -145,7 +166,11 @@ impl GlobalTimelines {
|
|||||||
.unwrap()
|
.unwrap()
|
||||||
.timelines
|
.timelines
|
||||||
.insert(ttid, tli.clone());
|
.insert(ttid, tli.clone());
|
||||||
tli.bootstrap(&conf, broker_active_set.clone());
|
tli.bootstrap(
|
||||||
|
&conf,
|
||||||
|
broker_active_set.clone(),
|
||||||
|
partial_backup_rate_limiter.clone(),
|
||||||
|
);
|
||||||
}
|
}
|
||||||
// If we can't load a timeline, it's most likely because of a corrupted
|
// If we can't load a timeline, it's most likely because of a corrupted
|
||||||
// directory. We will log an error and won't allow to delete/recreate
|
// directory. We will log an error and won't allow to delete/recreate
|
||||||
@@ -178,20 +203,27 @@ impl GlobalTimelines {
|
|||||||
_guard: &tokio::sync::MutexGuard<'a, TimelineLoadLock>,
|
_guard: &tokio::sync::MutexGuard<'a, TimelineLoadLock>,
|
||||||
ttid: TenantTimelineId,
|
ttid: TenantTimelineId,
|
||||||
) -> Result<Arc<Timeline>> {
|
) -> Result<Arc<Timeline>> {
|
||||||
let (conf, broker_active_set) = TIMELINES_STATE.lock().unwrap().get_dependencies();
|
let (conf, broker_active_set, partial_backup_rate_limiter) =
|
||||||
|
TIMELINES_STATE.lock().unwrap().get_dependencies();
|
||||||
|
|
||||||
match Timeline::load_timeline(&conf, ttid) {
|
match Timeline::load_timeline(&conf, ttid) {
|
||||||
Ok(timeline) => {
|
Ok(timeline) => {
|
||||||
let tli = Arc::new(timeline);
|
let tli = Arc::new(timeline);
|
||||||
|
|
||||||
// TODO: prevent concurrent timeline creation/loading
|
// TODO: prevent concurrent timeline creation/loading
|
||||||
TIMELINES_STATE
|
{
|
||||||
.lock()
|
let mut state = TIMELINES_STATE.lock().unwrap();
|
||||||
.unwrap()
|
|
||||||
.timelines
|
|
||||||
.insert(ttid, tli.clone());
|
|
||||||
|
|
||||||
tli.bootstrap(&conf, broker_active_set);
|
// We may be have been asked to load a timeline that was previously deleted (e.g. from `pull_timeline.rs`). We trust
|
||||||
|
// that the human doing this manual intervention knows what they are doing, and remove its tombstone.
|
||||||
|
if state.tombstones.remove(&ttid).is_some() {
|
||||||
|
warn!("Un-deleted timeline {ttid}");
|
||||||
|
}
|
||||||
|
|
||||||
|
state.timelines.insert(ttid, tli.clone());
|
||||||
|
}
|
||||||
|
|
||||||
|
tli.bootstrap(&conf, broker_active_set, partial_backup_rate_limiter);
|
||||||
|
|
||||||
Ok(tli)
|
Ok(tli)
|
||||||
}
|
}
|
||||||
@@ -216,18 +248,23 @@ impl GlobalTimelines {
|
|||||||
|
|
||||||
/// Create a new timeline with the given id. If the timeline already exists, returns
|
/// Create a new timeline with the given id. If the timeline already exists, returns
|
||||||
/// an existing timeline.
|
/// an existing timeline.
|
||||||
pub async fn create(
|
pub(crate) async fn create(
|
||||||
ttid: TenantTimelineId,
|
ttid: TenantTimelineId,
|
||||||
server_info: ServerInfo,
|
server_info: ServerInfo,
|
||||||
commit_lsn: Lsn,
|
commit_lsn: Lsn,
|
||||||
local_start_lsn: Lsn,
|
local_start_lsn: Lsn,
|
||||||
) -> Result<Arc<Timeline>> {
|
) -> Result<Arc<Timeline>> {
|
||||||
let (conf, broker_active_set) = {
|
let (conf, broker_active_set, partial_backup_rate_limiter) = {
|
||||||
let state = TIMELINES_STATE.lock().unwrap();
|
let state = TIMELINES_STATE.lock().unwrap();
|
||||||
if let Ok(timeline) = state.get(&ttid) {
|
if let Ok(timeline) = state.get(&ttid) {
|
||||||
// Timeline already exists, return it.
|
// Timeline already exists, return it.
|
||||||
return Ok(timeline);
|
return Ok(timeline);
|
||||||
}
|
}
|
||||||
|
|
||||||
|
if state.tombstones.contains_key(&ttid) {
|
||||||
|
anyhow::bail!("Timeline {ttid} is deleted, refusing to recreate");
|
||||||
|
}
|
||||||
|
|
||||||
state.get_dependencies()
|
state.get_dependencies()
|
||||||
};
|
};
|
||||||
|
|
||||||
@@ -257,7 +294,12 @@ impl GlobalTimelines {
|
|||||||
// Bootstrap is transactional, so if it fails, the timeline will be deleted,
|
// Bootstrap is transactional, so if it fails, the timeline will be deleted,
|
||||||
// and the state on disk should remain unchanged.
|
// and the state on disk should remain unchanged.
|
||||||
if let Err(e) = timeline
|
if let Err(e) = timeline
|
||||||
.init_new(&mut shared_state, &conf, broker_active_set)
|
.init_new(
|
||||||
|
&mut shared_state,
|
||||||
|
&conf,
|
||||||
|
broker_active_set,
|
||||||
|
partial_backup_rate_limiter,
|
||||||
|
)
|
||||||
.await
|
.await
|
||||||
{
|
{
|
||||||
// Note: the most likely reason for init failure is that the timeline
|
// Note: the most likely reason for init failure is that the timeline
|
||||||
@@ -282,17 +324,19 @@ impl GlobalTimelines {
|
|||||||
/// Get a timeline from the global map. If it's not present, it doesn't exist on disk,
|
/// Get a timeline from the global map. If it's not present, it doesn't exist on disk,
|
||||||
/// or was corrupted and couldn't be loaded on startup. Returned timeline is always valid,
|
/// or was corrupted and couldn't be loaded on startup. Returned timeline is always valid,
|
||||||
/// i.e. loaded in memory and not cancelled.
|
/// i.e. loaded in memory and not cancelled.
|
||||||
pub fn get(ttid: TenantTimelineId) -> Result<Arc<Timeline>, TimelineError> {
|
pub(crate) fn get(ttid: TenantTimelineId) -> Result<Arc<Timeline>, TimelineError> {
|
||||||
let res = TIMELINES_STATE.lock().unwrap().get(&ttid);
|
let tli_res = {
|
||||||
|
let state = TIMELINES_STATE.lock().unwrap();
|
||||||
match res {
|
state.get(&ttid)
|
||||||
|
};
|
||||||
|
match tli_res {
|
||||||
Ok(tli) => {
|
Ok(tli) => {
|
||||||
if tli.is_cancelled() {
|
if tli.is_cancelled() {
|
||||||
return Err(TimelineError::Cancelled(ttid));
|
return Err(TimelineError::Cancelled(ttid));
|
||||||
}
|
}
|
||||||
Ok(tli)
|
Ok(tli)
|
||||||
}
|
}
|
||||||
_ => res,
|
_ => tli_res,
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -321,12 +365,26 @@ impl GlobalTimelines {
|
|||||||
|
|
||||||
/// Cancels timeline, then deletes the corresponding data directory.
|
/// Cancels timeline, then deletes the corresponding data directory.
|
||||||
/// If only_local, doesn't remove WAL segments in remote storage.
|
/// If only_local, doesn't remove WAL segments in remote storage.
|
||||||
pub async fn delete(
|
pub(crate) async fn delete(
|
||||||
ttid: &TenantTimelineId,
|
ttid: &TenantTimelineId,
|
||||||
only_local: bool,
|
only_local: bool,
|
||||||
) -> Result<TimelineDeleteForceResult> {
|
) -> Result<TimelineDeleteForceResult> {
|
||||||
let tli_res = TIMELINES_STATE.lock().unwrap().get(ttid);
|
let tli_res = {
|
||||||
match tli_res {
|
let state = TIMELINES_STATE.lock().unwrap();
|
||||||
|
|
||||||
|
if state.tombstones.contains_key(ttid) {
|
||||||
|
// Presence of a tombstone guarantees that a previous deletion has completed and there is no work to do.
|
||||||
|
info!("Timeline {ttid} was already deleted");
|
||||||
|
return Ok(TimelineDeleteForceResult {
|
||||||
|
dir_existed: false,
|
||||||
|
was_active: false,
|
||||||
|
});
|
||||||
|
}
|
||||||
|
|
||||||
|
state.get(ttid)
|
||||||
|
};
|
||||||
|
|
||||||
|
let result = match tli_res {
|
||||||
Ok(timeline) => {
|
Ok(timeline) => {
|
||||||
let was_active = timeline.broker_active.load(Ordering::Relaxed);
|
let was_active = timeline.broker_active.load(Ordering::Relaxed);
|
||||||
|
|
||||||
@@ -336,11 +394,6 @@ impl GlobalTimelines {
|
|||||||
info!("deleting timeline {}, only_local={}", ttid, only_local);
|
info!("deleting timeline {}, only_local={}", ttid, only_local);
|
||||||
let dir_existed = timeline.delete(&mut shared_state, only_local).await?;
|
let dir_existed = timeline.delete(&mut shared_state, only_local).await?;
|
||||||
|
|
||||||
// Remove timeline from the map.
|
|
||||||
// FIXME: re-enable it once we fix the issue with recreation of deleted timelines
|
|
||||||
// https://github.com/neondatabase/neon/issues/3146
|
|
||||||
// TIMELINES_STATE.lock().unwrap().timelines.remove(ttid);
|
|
||||||
|
|
||||||
Ok(TimelineDeleteForceResult {
|
Ok(TimelineDeleteForceResult {
|
||||||
dir_existed,
|
dir_existed,
|
||||||
was_active, // TODO: we probably should remove this field
|
was_active, // TODO: we probably should remove this field
|
||||||
@@ -356,7 +409,14 @@ impl GlobalTimelines {
|
|||||||
was_active: false,
|
was_active: false,
|
||||||
})
|
})
|
||||||
}
|
}
|
||||||
}
|
};
|
||||||
|
|
||||||
|
// Finalize deletion, by dropping Timeline objects and storing smaller tombstones. The tombstones
|
||||||
|
// are used to prevent still-running computes from re-creating the same timeline when they send data,
|
||||||
|
// and to speed up repeated deletion calls by avoiding re-listing objects.
|
||||||
|
TIMELINES_STATE.lock().unwrap().delete(*ttid);
|
||||||
|
|
||||||
|
result
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Deactivates and deletes all timelines for the tenant. Returns map of all timelines which
|
/// Deactivates and deletes all timelines for the tenant. Returns map of all timelines which
|
||||||
@@ -402,19 +462,20 @@ impl GlobalTimelines {
|
|||||||
tenant_id,
|
tenant_id,
|
||||||
))?;
|
))?;
|
||||||
|
|
||||||
// FIXME: we temporarily disabled removing timelines from the map, see `delete_force`
|
|
||||||
// let tlis_after_delete = Self::get_all_for_tenant(*tenant_id);
|
|
||||||
// if !tlis_after_delete.is_empty() {
|
|
||||||
// // Some timelines were created while we were deleting them, returning error
|
|
||||||
// // to the caller, so it can retry later.
|
|
||||||
// bail!(
|
|
||||||
// "failed to delete all timelines for tenant {}: some timelines were created while we were deleting them",
|
|
||||||
// tenant_id
|
|
||||||
// );
|
|
||||||
// }
|
|
||||||
|
|
||||||
Ok(deleted)
|
Ok(deleted)
|
||||||
}
|
}
|
||||||
|
|
||||||
|
pub fn housekeeping(tombstone_ttl: &Duration) {
|
||||||
|
let mut state = TIMELINES_STATE.lock().unwrap();
|
||||||
|
|
||||||
|
// We keep tombstones long enough to have a good chance of preventing rogue computes from re-creating deleted
|
||||||
|
// timelines. If a compute kept running for longer than this TTL (or across a safekeeper restart) then they
|
||||||
|
// may recreate a deleted timeline.
|
||||||
|
let now = Instant::now();
|
||||||
|
state
|
||||||
|
.tombstones
|
||||||
|
.retain(|_, v| now.duration_since(*v) < *tombstone_ttl);
|
||||||
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
#[derive(Clone, Copy, Serialize)]
|
#[derive(Clone, Copy, Serialize)]
|
||||||
|
|||||||
@@ -18,6 +18,8 @@
|
|||||||
//! This way control file stores information about all potentially existing
|
//! This way control file stores information about all potentially existing
|
||||||
//! remote partial segments and can clean them up after uploading a newer version.
|
//! remote partial segments and can clean them up after uploading a newer version.
|
||||||
|
|
||||||
|
use std::sync::Arc;
|
||||||
|
|
||||||
use camino::Utf8PathBuf;
|
use camino::Utf8PathBuf;
|
||||||
use postgres_ffi::{XLogFileName, XLogSegNo, PG_TLI};
|
use postgres_ffi::{XLogFileName, XLogSegNo, PG_TLI};
|
||||||
use remote_storage::RemotePath;
|
use remote_storage::RemotePath;
|
||||||
@@ -27,7 +29,7 @@ use tracing::{debug, error, info, instrument, warn};
|
|||||||
use utils::lsn::Lsn;
|
use utils::lsn::Lsn;
|
||||||
|
|
||||||
use crate::{
|
use crate::{
|
||||||
metrics::{PARTIAL_BACKUP_UPLOADED_BYTES, PARTIAL_BACKUP_UPLOADS},
|
metrics::{MISC_OPERATION_SECONDS, PARTIAL_BACKUP_UPLOADED_BYTES, PARTIAL_BACKUP_UPLOADS},
|
||||||
safekeeper::Term,
|
safekeeper::Term,
|
||||||
timeline::WalResidentTimeline,
|
timeline::WalResidentTimeline,
|
||||||
timeline_manager::StateSnapshot,
|
timeline_manager::StateSnapshot,
|
||||||
@@ -35,6 +37,30 @@ use crate::{
|
|||||||
SafeKeeperConf,
|
SafeKeeperConf,
|
||||||
};
|
};
|
||||||
|
|
||||||
|
#[derive(Clone)]
|
||||||
|
pub struct RateLimiter {
|
||||||
|
semaphore: Arc<tokio::sync::Semaphore>,
|
||||||
|
}
|
||||||
|
|
||||||
|
impl RateLimiter {
|
||||||
|
pub fn new(permits: usize) -> Self {
|
||||||
|
Self {
|
||||||
|
semaphore: Arc::new(tokio::sync::Semaphore::new(permits)),
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
async fn acquire_owned(&self) -> tokio::sync::OwnedSemaphorePermit {
|
||||||
|
let _timer = MISC_OPERATION_SECONDS
|
||||||
|
.with_label_values(&["partial_permit_acquire"])
|
||||||
|
.start_timer();
|
||||||
|
self.semaphore
|
||||||
|
.clone()
|
||||||
|
.acquire_owned()
|
||||||
|
.await
|
||||||
|
.expect("semaphore is closed")
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
#[derive(Debug, Clone, Serialize, Deserialize, PartialEq)]
|
#[derive(Debug, Clone, Serialize, Deserialize, PartialEq)]
|
||||||
pub enum UploadStatus {
|
pub enum UploadStatus {
|
||||||
/// Upload is in progress. This status should be used only for garbage collection,
|
/// Upload is in progress. This status should be used only for garbage collection,
|
||||||
@@ -208,6 +234,9 @@ impl PartialBackup {
|
|||||||
/// Upload the latest version of the partial segment and garbage collect older versions.
|
/// Upload the latest version of the partial segment and garbage collect older versions.
|
||||||
#[instrument(name = "upload", skip_all, fields(name = %prepared.name))]
|
#[instrument(name = "upload", skip_all, fields(name = %prepared.name))]
|
||||||
async fn do_upload(&mut self, prepared: &PartialRemoteSegment) -> anyhow::Result<()> {
|
async fn do_upload(&mut self, prepared: &PartialRemoteSegment) -> anyhow::Result<()> {
|
||||||
|
let _timer = MISC_OPERATION_SECONDS
|
||||||
|
.with_label_values(&["partial_do_upload"])
|
||||||
|
.start_timer();
|
||||||
info!("starting upload {:?}", prepared);
|
info!("starting upload {:?}", prepared);
|
||||||
|
|
||||||
let state_0 = self.state.clone();
|
let state_0 = self.state.clone();
|
||||||
@@ -307,6 +336,7 @@ pub(crate) fn needs_uploading(
|
|||||||
pub async fn main_task(
|
pub async fn main_task(
|
||||||
tli: WalResidentTimeline,
|
tli: WalResidentTimeline,
|
||||||
conf: SafeKeeperConf,
|
conf: SafeKeeperConf,
|
||||||
|
limiter: RateLimiter,
|
||||||
) -> Option<PartialRemoteSegment> {
|
) -> Option<PartialRemoteSegment> {
|
||||||
debug!("started");
|
debug!("started");
|
||||||
let await_duration = conf.partial_backup_timeout;
|
let await_duration = conf.partial_backup_timeout;
|
||||||
@@ -411,6 +441,9 @@ pub async fn main_task(
|
|||||||
continue 'outer;
|
continue 'outer;
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// limit concurrent uploads
|
||||||
|
let _upload_permit = limiter.acquire_owned().await;
|
||||||
|
|
||||||
let prepared = backup.prepare_upload().await;
|
let prepared = backup.prepare_upload().await;
|
||||||
if let Some(seg) = &uploaded_segment {
|
if let Some(seg) = &uploaded_segment {
|
||||||
if seg.eq_without_status(&prepared) {
|
if seg.eq_without_status(&prepared) {
|
||||||
|
|||||||
@@ -187,6 +187,7 @@ pub fn run_server(os: NodeOs, disk: Arc<SafekeeperDisk>) -> Result<()> {
|
|||||||
enable_offload: false,
|
enable_offload: false,
|
||||||
delete_offloaded_wal: false,
|
delete_offloaded_wal: false,
|
||||||
control_file_save_interval: Duration::from_secs(1),
|
control_file_save_interval: Duration::from_secs(1),
|
||||||
|
partial_backup_concurrency: 1,
|
||||||
};
|
};
|
||||||
|
|
||||||
let mut global = GlobalMap::new(disk, conf.clone())?;
|
let mut global = GlobalMap::new(disk, conf.clone())?;
|
||||||
|
|||||||
@@ -10,8 +10,9 @@ use hyper::header::CONTENT_TYPE;
|
|||||||
use hyper::{Body, Request, Response};
|
use hyper::{Body, Request, Response};
|
||||||
use hyper::{StatusCode, Uri};
|
use hyper::{StatusCode, Uri};
|
||||||
use metrics::{BuildInfo, NeonMetrics};
|
use metrics::{BuildInfo, NeonMetrics};
|
||||||
|
use pageserver_api::controller_api::TenantCreateRequest;
|
||||||
use pageserver_api::models::{
|
use pageserver_api::models::{
|
||||||
TenantConfigRequest, TenantCreateRequest, TenantLocationConfigRequest, TenantShardSplitRequest,
|
TenantConfigRequest, TenantLocationConfigRequest, TenantShardSplitRequest,
|
||||||
TenantTimeTravelRequest, TimelineCreateRequest,
|
TenantTimeTravelRequest, TimelineCreateRequest,
|
||||||
};
|
};
|
||||||
use pageserver_api::shard::TenantShardId;
|
use pageserver_api::shard::TenantShardId;
|
||||||
|
|||||||
@@ -1,6 +1,7 @@
|
|||||||
use crate::pageserver_client::PageserverClient;
|
use crate::pageserver_client::PageserverClient;
|
||||||
use crate::persistence::Persistence;
|
use crate::persistence::Persistence;
|
||||||
use crate::service;
|
use crate::service;
|
||||||
|
use pageserver_api::controller_api::PlacementPolicy;
|
||||||
use pageserver_api::models::{
|
use pageserver_api::models::{
|
||||||
LocationConfig, LocationConfigMode, LocationConfigSecondary, TenantConfig,
|
LocationConfig, LocationConfigMode, LocationConfigSecondary, TenantConfig,
|
||||||
};
|
};
|
||||||
@@ -29,6 +30,7 @@ pub(super) struct Reconciler {
|
|||||||
/// of a tenant's state from when we spawned a reconcile task.
|
/// of a tenant's state from when we spawned a reconcile task.
|
||||||
pub(super) tenant_shard_id: TenantShardId,
|
pub(super) tenant_shard_id: TenantShardId,
|
||||||
pub(crate) shard: ShardIdentity,
|
pub(crate) shard: ShardIdentity,
|
||||||
|
pub(crate) placement_policy: PlacementPolicy,
|
||||||
pub(crate) generation: Option<Generation>,
|
pub(crate) generation: Option<Generation>,
|
||||||
pub(crate) intent: TargetState,
|
pub(crate) intent: TargetState,
|
||||||
|
|
||||||
@@ -641,7 +643,7 @@ impl Reconciler {
|
|||||||
generation,
|
generation,
|
||||||
&self.shard,
|
&self.shard,
|
||||||
&self.config,
|
&self.config,
|
||||||
!self.intent.secondary.is_empty(),
|
&self.placement_policy,
|
||||||
);
|
);
|
||||||
match self.observed.locations.get(&node.get_id()) {
|
match self.observed.locations.get(&node.get_id()) {
|
||||||
Some(conf) if conf.conf.as_ref() == Some(&wanted_conf) => {
|
Some(conf) if conf.conf.as_ref() == Some(&wanted_conf) => {
|
||||||
@@ -801,8 +803,15 @@ pub(crate) fn attached_location_conf(
|
|||||||
generation: Generation,
|
generation: Generation,
|
||||||
shard: &ShardIdentity,
|
shard: &ShardIdentity,
|
||||||
config: &TenantConfig,
|
config: &TenantConfig,
|
||||||
has_secondaries: bool,
|
policy: &PlacementPolicy,
|
||||||
) -> LocationConfig {
|
) -> LocationConfig {
|
||||||
|
let has_secondaries = match policy {
|
||||||
|
PlacementPolicy::Attached(0) | PlacementPolicy::Detached | PlacementPolicy::Secondary => {
|
||||||
|
false
|
||||||
|
}
|
||||||
|
PlacementPolicy::Attached(_) => true,
|
||||||
|
};
|
||||||
|
|
||||||
LocationConfig {
|
LocationConfig {
|
||||||
mode: LocationConfigMode::AttachedSingle,
|
mode: LocationConfigMode::AttachedSingle,
|
||||||
generation: generation.into(),
|
generation: generation.into(),
|
||||||
|
|||||||
@@ -32,10 +32,10 @@ use itertools::Itertools;
|
|||||||
use pageserver_api::{
|
use pageserver_api::{
|
||||||
controller_api::{
|
controller_api::{
|
||||||
NodeAvailability, NodeRegisterRequest, NodeSchedulingPolicy, PlacementPolicy,
|
NodeAvailability, NodeRegisterRequest, NodeSchedulingPolicy, PlacementPolicy,
|
||||||
ShardSchedulingPolicy, TenantCreateResponse, TenantCreateResponseShard,
|
ShardSchedulingPolicy, TenantCreateRequest, TenantCreateResponse,
|
||||||
TenantDescribeResponse, TenantDescribeResponseShard, TenantLocateResponse,
|
TenantCreateResponseShard, TenantDescribeResponse, TenantDescribeResponseShard,
|
||||||
TenantPolicyRequest, TenantShardMigrateRequest, TenantShardMigrateResponse,
|
TenantLocateResponse, TenantPolicyRequest, TenantShardMigrateRequest,
|
||||||
UtilizationScore,
|
TenantShardMigrateResponse, UtilizationScore,
|
||||||
},
|
},
|
||||||
models::{SecondaryProgress, TenantConfigRequest, TopTenantShardsRequest},
|
models::{SecondaryProgress, TenantConfigRequest, TopTenantShardsRequest},
|
||||||
};
|
};
|
||||||
@@ -46,10 +46,9 @@ use crate::pageserver_client::PageserverClient;
|
|||||||
use pageserver_api::{
|
use pageserver_api::{
|
||||||
models::{
|
models::{
|
||||||
self, LocationConfig, LocationConfigListResponse, LocationConfigMode,
|
self, LocationConfig, LocationConfigListResponse, LocationConfigMode,
|
||||||
PageserverUtilization, ShardParameters, TenantConfig, TenantCreateRequest,
|
PageserverUtilization, ShardParameters, TenantConfig, TenantLocationConfigRequest,
|
||||||
TenantLocationConfigRequest, TenantLocationConfigResponse, TenantShardLocation,
|
TenantLocationConfigResponse, TenantShardLocation, TenantShardSplitRequest,
|
||||||
TenantShardSplitRequest, TenantShardSplitResponse, TenantTimeTravelRequest,
|
TenantShardSplitResponse, TenantTimeTravelRequest, TimelineCreateRequest, TimelineInfo,
|
||||||
TimelineCreateRequest, TimelineInfo,
|
|
||||||
},
|
},
|
||||||
shard::{ShardCount, ShardIdentity, ShardNumber, ShardStripeSize, TenantShardId},
|
shard::{ShardCount, ShardIdentity, ShardNumber, ShardStripeSize, TenantShardId},
|
||||||
upcall_api::{
|
upcall_api::{
|
||||||
@@ -1391,7 +1390,7 @@ impl Service {
|
|||||||
tenant_shard.generation.unwrap(),
|
tenant_shard.generation.unwrap(),
|
||||||
&tenant_shard.shard,
|
&tenant_shard.shard,
|
||||||
&tenant_shard.config,
|
&tenant_shard.config,
|
||||||
false,
|
&PlacementPolicy::Attached(0),
|
||||||
)),
|
)),
|
||||||
},
|
},
|
||||||
)]);
|
)]);
|
||||||
@@ -3322,7 +3321,7 @@ impl Service {
|
|||||||
generation,
|
generation,
|
||||||
&child_shard,
|
&child_shard,
|
||||||
&config,
|
&config,
|
||||||
matches!(policy, PlacementPolicy::Attached(n) if n > 0),
|
&policy,
|
||||||
)),
|
)),
|
||||||
},
|
},
|
||||||
);
|
);
|
||||||
|
|||||||
@@ -908,12 +908,8 @@ impl TenantShard {
|
|||||||
.generation
|
.generation
|
||||||
.expect("Attempted to enter attached state without a generation");
|
.expect("Attempted to enter attached state without a generation");
|
||||||
|
|
||||||
let wanted_conf = attached_location_conf(
|
let wanted_conf =
|
||||||
generation,
|
attached_location_conf(generation, &self.shard, &self.config, &self.policy);
|
||||||
&self.shard,
|
|
||||||
&self.config,
|
|
||||||
!self.intent.secondary.is_empty(),
|
|
||||||
);
|
|
||||||
match self.observed.locations.get(&node_id) {
|
match self.observed.locations.get(&node_id) {
|
||||||
Some(conf) if conf.conf.as_ref() == Some(&wanted_conf) => {}
|
Some(conf) if conf.conf.as_ref() == Some(&wanted_conf) => {}
|
||||||
Some(_) | None => {
|
Some(_) | None => {
|
||||||
@@ -1099,6 +1095,7 @@ impl TenantShard {
|
|||||||
let mut reconciler = Reconciler {
|
let mut reconciler = Reconciler {
|
||||||
tenant_shard_id: self.tenant_shard_id,
|
tenant_shard_id: self.tenant_shard_id,
|
||||||
shard: self.shard,
|
shard: self.shard,
|
||||||
|
placement_policy: self.policy.clone(),
|
||||||
generation: self.generation,
|
generation: self.generation,
|
||||||
intent: reconciler_intent,
|
intent: reconciler_intent,
|
||||||
detach,
|
detach,
|
||||||
|
|||||||
Some files were not shown because too many files have changed in this diff Show More
Reference in New Issue
Block a user