Merge remote-tracking branch 'origin/main' into overtrue/fix-1905-activation-fence

# Conflicts:
#	crates/ecstore/src/core/pools.rs
This commit is contained in:
overtrue
2026-08-23 02:36:20 +08:00
48 changed files with 3880 additions and 1035 deletions
+2 -2
View File
@@ -1,2 +1,2 @@
sha256-darwin=b4ae71aa894e5c7795ae3eb8116f1777a7601d0f5db3898be2e48faf3329bd9b sha256-darwin=9f767b37ed8b1c82da62ea441462d75487785c8086e56f08fb6f6cd89c6e2e52
sha256-linux=433debd9d9defa832986269abdf0f1d131597b2d7a417ce930e17c1fd47d85ba sha256-linux=fbdaf42b220958d4b1e8880e0f8b5a7992d38e21051bb60596dd4538424757d6
+1
View File
@@ -36,6 +36,7 @@ script-tests: ## Run shell script tests
./scripts/test_manual_transition_runbooks.sh ./scripts/test_manual_transition_runbooks.sh
./scripts/check_embedded_secrets.sh --self-test ./scripts/check_embedded_secrets.sh --self-test
python3 ./scripts/check_test_wiring.py --self-test python3 ./scripts/check_test_wiring.py --self-test
python3 ./scripts/check_scheduled_validation_freshness.py --self-test
python3 ./scripts/s3-tests/test_report_compat.py python3 ./scripts/s3-tests/test_report_compat.py
bash -n ./scripts/validate_object_data_cache_cold_stampede.sh bash -n ./scripts/validate_object_data_cache_cold_stampede.sh
python3 ./scripts/check_object_data_cache_follower_samples.py --self-test python3 ./scripts/check_object_data_cache_follower_samples.py --self-test
@@ -14,9 +14,10 @@
name: "Schedule Failure Issue" name: "Schedule Failure Issue"
description: >- description: >-
Open (or update) a tracking issue when a scheduled workflow run fails. Open (or update) a tracking issue when a scheduled workflow run fails or
does not complete normally.
Dedupes by workflow name: if an open issue titled Dedupes by workflow name: if an open issue titled
"[scheduled-failure] <workflow name>" already exists, the failure is "[scheduled-failure] <workflow name>" already exists, the result is
appended as a comment; otherwise a new issue is created. This is the appended as a comment; otherwise a new issue is created. This is the
single alerting mechanism for all scheduled pipelines (backlog#1149 ci-8). single alerting mechanism for all scheduled pipelines (backlog#1149 ci-8).
@@ -38,6 +39,30 @@ inputs:
Set to an empty string to skip labeling. Set to an empty string to skip labeling.
required: false required: false
default: "infrastructure" default: "infrastructure"
source-run-id:
description: "Run ID to report. Defaults to the current workflow run."
required: false
default: ${{ github.run_id }}
source-run-attempt:
description: "Run attempt to report. Defaults to the current attempt."
required: false
default: ${{ github.run_attempt }}
source-event:
description: "Trigger event of the run being reported."
required: false
default: ${{ github.event_name }}
source-ref-name:
description: "Ref name of the run being reported."
required: false
default: ${{ github.ref_name }}
source-sha:
description: "Commit SHA of the run being reported."
required: false
default: ${{ github.sha }}
details-file:
description: "Optional Markdown file appended to the issue body."
required: false
default: ""
runs: runs:
using: "composite" using: "composite"
@@ -48,17 +73,22 @@ runs:
GH_TOKEN: ${{ inputs.github-token }} GH_TOKEN: ${{ inputs.github-token }}
WORKFLOW_NAME: ${{ inputs.workflow-name }} WORKFLOW_NAME: ${{ inputs.workflow-name }}
ISSUE_LABEL: ${{ inputs.label }} ISSUE_LABEL: ${{ inputs.label }}
SOURCE_RUN_ID: ${{ inputs.source-run-id }}
SOURCE_RUN_ATTEMPT: ${{ inputs.source-run-attempt }}
SOURCE_EVENT: ${{ inputs.source-event }}
SOURCE_REF_NAME: ${{ inputs.source-ref-name }}
SOURCE_SHA: ${{ inputs.source-sha }}
DETAILS_FILE: ${{ inputs.details-file }}
run: | run: |
set -euo pipefail set -euo pipefail
title="[scheduled-failure] ${WORKFLOW_NAME}" title="[scheduled-failure] ${WORKFLOW_NAME}"
run_url="${GITHUB_SERVER_URL}/${GITHUB_REPOSITORY}/actions/runs/${GITHUB_RUN_ID}" run_url="${GITHUB_SERVER_URL}/${GITHUB_REPOSITORY}/actions/runs/${SOURCE_RUN_ID}"
# Failed job names for this run attempt. The alert job runs while the # Inspect the reported run attempt. It can be the current in-workflow
# run as a whole is still in progress, so inspect the jobs that have # failure or a completed run observed by the external watchdog.
# already completed with a non-success conclusion.
failed_jobs="$(gh api \ failed_jobs="$(gh api \
"repos/${GITHUB_REPOSITORY}/actions/runs/${GITHUB_RUN_ID}/attempts/${GITHUB_RUN_ATTEMPT}/jobs" \ "repos/${GITHUB_REPOSITORY}/actions/runs/${SOURCE_RUN_ID}/attempts/${SOURCE_RUN_ATTEMPT}/jobs" \
--paginate \ --paginate \
--jq '.jobs[] --jq '.jobs[]
| select(.conclusion == "failure" or .conclusion == "timed_out" or .conclusion == "cancelled") | select(.conclusion == "failure" or .conclusion == "timed_out" or .conclusion == "cancelled")
@@ -67,15 +97,26 @@ runs:
failed_jobs="- (failed job not recorded yet — see the run page)" failed_jobs="- (failed job not recorded yet — see the run page)"
fi fi
details=""
if [ -n "${DETAILS_FILE}" ]; then
if [ -f "${DETAILS_FILE}" ]; then
details="$(cat "${DETAILS_FILE}")"
else
details="Details file was not available: \`${DETAILS_FILE}\`"
fi
fi
body="$(cat <<EOF body="$(cat <<EOF
Scheduled run of **${WORKFLOW_NAME}** failed. Run of **${WORKFLOW_NAME}** did not complete successfully.
- Run: ${run_url} (attempt ${GITHUB_RUN_ATTEMPT}) - Run: ${run_url} (attempt ${SOURCE_RUN_ATTEMPT})
- Event: \`${GITHUB_EVENT_NAME}\` - Event: \`${SOURCE_EVENT}\`
- Ref: \`${GITHUB_REF_NAME}\` @ \`${GITHUB_SHA}\` - Ref: \`${SOURCE_REF_NAME}\` @ \`${SOURCE_SHA}\`
Failed jobs: Non-success jobs:
${failed_jobs} ${failed_jobs}
${details}
EOF EOF
)" )"
+14
View File
@@ -0,0 +1,14 @@
[
{ "workflow": ".github/workflows/audit.yml", "max_age_hours": 36 },
{ "workflow": ".github/workflows/build.yml", "max_age_hours": 192 },
{ "workflow": ".github/workflows/ci.yml", "max_age_hours": 192 },
{ "workflow": ".github/workflows/coverage.yml", "max_age_hours": 192 },
{ "workflow": ".github/workflows/e2e-replication-nightly.yml", "max_age_hours": 36 },
{ "workflow": ".github/workflows/e2e-s3tests.yml", "max_age_hours": 192 },
{ "workflow": ".github/workflows/fuzz.yml", "max_age_hours": 36 },
{ "workflow": ".github/workflows/mint.yml", "max_age_hours": 192 },
{ "workflow": ".github/workflows/minio-interop.yml", "max_age_hours": 36 },
{ "workflow": ".github/workflows/nightly-gnu.yml", "max_age_hours": 36 },
{ "workflow": ".github/workflows/performance-ab.yml", "max_age_hours": 36 },
{ "workflow": ".github/workflows/runner-hygiene.yml", "max_age_hours": 792 }
]
+1 -1
View File
@@ -46,7 +46,7 @@ on:
# advisory could sit unnoticed for seven days. The check list is unchanged — # advisory could sit unnoticed for seven days. The check list is unchanged —
# splitting it into a light daily advisories-only run and a weekly full run # splitting it into a light daily advisories-only run and a weekly full run
# would create runs where sources/bans/licenses go unverified. # would create runs where sources/bans/licenses go unverified.
- cron: '0 3 * * *' # Daily 03:00 UTC (staggered after the midnight ci/build crons) - cron: '23 3 * * *' # Daily 03:23 UTC
workflow_dispatch: workflow_dispatch:
permissions: permissions:
+21 -1
View File
@@ -52,7 +52,7 @@ on:
- ".dockerignore" - ".dockerignore"
- "flake.lock" - "flake.lock"
schedule: schedule:
- cron: "0 1 * * 0" # Weekly on Sunday 01:00 UTC (staggered after the ci.yml midnight cron) - cron: "13 1 * * 0" # Weekly on Sunday 01:13 UTC
workflow_dispatch: workflow_dispatch:
inputs: inputs:
build_docker: build_docker:
@@ -1032,3 +1032,23 @@ jobs:
echo "🎉 Released $TAG successfully!" echo "🎉 Released $TAG successfully!"
echo "📄 Release URL: ${{ needs.create-release.outputs.release_url }}" echo "📄 Release URL: ${{ needs.create-release.outputs.release_url }}"
alert-on-failure:
name: Alert on scheduled failure
needs: [build-check, prepare-platform-matrix, build-rustfs, build-summary]
if: >-
always() && github.event_name == 'schedule' &&
(contains(needs.*.result, 'failure') || contains(needs.*.result, 'cancelled'))
runs-on: ubuntu-latest
timeout-minutes: 10
permissions:
contents: read
issues: write
steps:
- uses: actions/checkout@9c091bb21b7c1c1d1991bb908d89e4e9dddfe3e0 # v7
with:
persist-credentials: false
- name: Open or update failure-tracking issue
uses: ./.github/actions/schedule-failure-issue
with:
github-token: ${{ secrets.GITHUB_TOKEN }}
+6 -3
View File
@@ -94,6 +94,9 @@ concurrency:
env: env:
CARGO_TERM_COLOR: always CARGO_TERM_COLOR: always
# Swatinem/rust-cache hashes every RUST* variable. Keep this aligned with
# ci.yml or the writer and readers use disjoint cache keys.
RUST_BACKTRACE: 1
jobs: jobs:
# Readers: test-and-lint, test-ilm-integration-serial, build-rustfs-debug-binary, # Readers: test-and-lint, test-ilm-integration-serial, build-rustfs-debug-binary,
@@ -101,7 +104,7 @@ jobs:
warm-ci-dev: warm-ci-dev:
name: Warm ci-dev name: Warm ci-dev
runs-on: sm-standard-4 runs-on: sm-standard-4
timeout-minutes: 90 timeout-minutes: 120
env: env:
FORCE_JAVASCRIPT_ACTIONS_TO_NODE24: "true" FORCE_JAVASCRIPT_ACTIONS_TO_NODE24: "true"
steps: steps:
@@ -191,7 +194,7 @@ jobs:
warm-ci-feat-rio: warm-ci-feat-rio:
name: Warm ci-feat-rio name: Warm ci-feat-rio
runs-on: sm-standard-4 runs-on: sm-standard-4
timeout-minutes: 90 timeout-minutes: 120
env: env:
FORCE_JAVASCRIPT_ACTIONS_TO_NODE24: "true" FORCE_JAVASCRIPT_ACTIONS_TO_NODE24: "true"
steps: steps:
@@ -219,7 +222,7 @@ jobs:
warm-ci-feat-proto: warm-ci-feat-proto:
name: Warm ci-feat-proto name: Warm ci-feat-proto
runs-on: sm-standard-4 runs-on: sm-standard-4
timeout-minutes: 90 timeout-minutes: 120
env: env:
FORCE_JAVASCRIPT_ACTIONS_TO_NODE24: "true" FORCE_JAVASCRIPT_ACTIONS_TO_NODE24: "true"
steps: steps:
+4 -1
View File
@@ -126,7 +126,10 @@ jobs:
run: ./scripts/check_embedded_secrets.sh run: ./scripts/check_embedded_secrets.sh
- name: Check test wiring - name: Check test wiring
run: python3 ./scripts/check_test_wiring.py run: |
python3 ./scripts/check_test_wiring.py --self-test
python3 ./scripts/check_scheduled_validation_freshness.py --self-test
python3 ./scripts/check_test_wiring.py
- name: Check no planning docs committed - name: Check no planning docs committed
run: ./scripts/check_no_planning_docs.sh run: ./scripts/check_no_planning_docs.sh
+39 -2
View File
@@ -59,7 +59,7 @@ on:
merge_group: merge_group:
types: [ checks_requested ] types: [ checks_requested ]
schedule: schedule:
- cron: "0 0 * * 0" # Weekly on Sunday at midnight UTC - cron: "11 0 * * 0" # Weekly on Sunday 00:11 UTC
workflow_dispatch: workflow_dispatch:
permissions: permissions:
@@ -161,7 +161,10 @@ jobs:
run: ./scripts/check_embedded_secrets.sh run: ./scripts/check_embedded_secrets.sh
- name: Check test wiring - name: Check test wiring
run: python3 ./scripts/check_test_wiring.py run: |
python3 ./scripts/check_test_wiring.py --self-test
python3 ./scripts/check_scheduled_validation_freshness.py --self-test
python3 ./scripts/check_test_wiring.py
- name: Check no planning docs committed - name: Check no planning docs committed
run: ./scripts/check_no_planning_docs.sh run: ./scripts/check_no_planning_docs.sh
@@ -1032,3 +1035,37 @@ jobs:
path: artifacts/s3tests-single/** path: artifacts/s3tests-single/**
if-no-files-found: ignore if-no-files-found: ignore
retention-days: 3 retention-days: 3
alert-on-failure:
name: Alert on scheduled failure
needs:
- typos
- quick-checks
- test-and-lint
- test-ilm-integration-serial
- test-and-lint-rio-v2
- test-and-lint-protocols
- build-rustfs-debug-binary
- build-rustfs-debug-binary-rio-v2
- uring-integration
- e2e-tests
- e2e-full
- e2e-tests-rio-v2
- s3-implemented-tests
- s3-lifecycle-behavior-tests
if: >-
always() && github.event_name == 'schedule' &&
(contains(needs.*.result, 'failure') || contains(needs.*.result, 'cancelled'))
runs-on: ubuntu-latest
timeout-minutes: 10
permissions:
contents: read
issues: write
steps:
- uses: actions/checkout@9c091bb21b7c1c1d1991bb908d89e4e9dddfe3e0 # v7
with:
persist-credentials: false
- name: Open or update failure-tracking issue
uses: ./.github/actions/schedule-failure-issue
with:
github-token: ${{ secrets.GITHUB_TOKEN }}
+1 -1
View File
@@ -37,7 +37,7 @@ on:
# build (01:00), e2e-s3tests (02:00), audit (03:00), nix-flake-update # build (01:00), e2e-s3tests (02:00), audit (03:00), nix-flake-update
# (05:00), mint (06:00), and the daily fuzz (02:00), minio-interop (03:17), # (05:00), mint (06:00), and the daily fuzz (02:00), minio-interop (03:17),
# e2e-replication-nightly (04:00) and performance-ab (06:00) lanes. # e2e-replication-nightly (04:00) and performance-ab (06:00) lanes.
- cron: "0 7 * * 0" - cron: "43 7 * * 0"
# Only alert-on-failure needs more than read access; it declares its own # Only alert-on-failure needs more than read access; it declares its own
# job-level `issues: write`. # job-level `issues: write`.
@@ -40,7 +40,7 @@ on:
schedule: schedule:
# 04:00 UTC nightly — staggered clear of fuzz/e2e-s3tests (02:00), # 04:00 UTC nightly — staggered clear of fuzz/e2e-s3tests (02:00),
# stale (01:30) and performance-ab (06:00). # stale (01:30) and performance-ab (06:00).
- cron: "0 4 * * *" - cron: "29 4 * * *"
# Only alert-on-failure needs more than read access; it declares its own # Only alert-on-failure needs more than read access; it declares its own
# job-level `issues: write`. # job-level `issues: write`.
@@ -196,6 +196,9 @@ jobs:
cache-save-if: 'false' cache-save-if: 'false'
install-build-packaging-tools: 'false' install-build-packaging-tools: 'false'
- name: Verify protocol socket oracle
run: ss -tn state CLOSE-WAIT >/dev/null
# The suite owns fixed protocol ports and serializes its internal cases. # The suite owns fixed protocol ports and serializes its internal cases.
- name: Verify protocol e2e membership - name: Verify protocol e2e membership
env: env:
+13 -1
View File
@@ -90,10 +90,14 @@ on:
description: "Optional pytest -m expression" description: "Optional pytest -m expression"
required: false required: false
default: "" default: ""
testexpr:
description: "Optional pytest -k expression"
required: false
default: ""
schedule: schedule:
# Weekly full sweep (Sunday 02:00 UTC): full suite, run against BOTH the # Weekly full sweep (Sunday 02:00 UTC): full suite, run against BOTH the
# single-node and the 4-node distributed topologies (matrix below). # single-node and the 4-node distributed topologies (matrix below).
- cron: "0 2 * * 0" - cron: "19 2 * * 0"
env: env:
# main user # main user
@@ -116,6 +120,7 @@ env:
XDIST: ${{ github.event.inputs.xdist || '4' }} XDIST: ${{ github.event.inputs.xdist || '4' }}
MAXFAIL: ${{ github.event.inputs.maxfail || '0' }} MAXFAIL: ${{ github.event.inputs.maxfail || '0' }}
MARKEXPR: ${{ github.event.inputs.markexpr || '' }} MARKEXPR: ${{ github.event.inputs.markexpr || '' }}
TESTEXPR: ${{ github.event.inputs.testexpr || '' }}
S3_SHARD_COUNT: ${{ github.event_name == 'schedule' && '4' || github.event.inputs.shard-count || '1' }} S3_SHARD_COUNT: ${{ github.event_name == 'schedule' && '4' || github.event.inputs.shard-count || '1' }}
TEST_TIMEOUT: "300" TEST_TIMEOUT: "300"
@@ -269,14 +274,20 @@ jobs:
EOF EOF
cat > haproxy.cfg <<'EOF' cat > haproxy.cfg <<'EOF'
global
log stdout format raw local0 info
defaults defaults
mode http mode http
log global
log-format '%ci:%cp [%tr] %ft %b/%s %TR/%Tw/%Tc/%Tr/%Ta %ST %B %tsc %HM %HP'
timeout connect 5s timeout connect 5s
timeout client 30s timeout client 30s
timeout server 30s timeout server 30s
frontend fe_s3 frontend fe_s3
bind *:9000 bind *:9000
option http-buffer-request
default_backend be_s3 default_backend be_s3
backend be_s3 backend be_s3
@@ -314,6 +325,7 @@ jobs:
XDIST="${XDIST}" \ XDIST="${XDIST}" \
MAXFAIL="${MAXFAIL}" \ MAXFAIL="${MAXFAIL}" \
MARKEXPR="${MARKEXPR}" \ MARKEXPR="${MARKEXPR}" \
TESTEXPR="${TESTEXPR}" \
./scripts/s3-tests/run.sh ./scripts/s3-tests/run.sh
- name: Publish compatibility report - name: Publish compatibility report
+1 -1
View File
@@ -30,7 +30,7 @@ on:
- "Cargo.lock" - "Cargo.lock"
- ".github/workflows/fuzz.yml" - ".github/workflows/fuzz.yml"
schedule: schedule:
- cron: "0 2 * * *" - cron: "17 2 * * *"
workflow_dispatch: workflow_dispatch:
inputs: inputs:
profile: profile:
+18
View File
@@ -121,3 +121,21 @@ jobs:
cargo nextest run --run-ignored ignored-only --no-tests=fail \ cargo nextest run --run-ignored ignored-only --no-tests=fail \
-p "$INTEROP_PACKAGE" --features "$INTEROP_FEATURES" \ -p "$INTEROP_PACKAGE" --features "$INTEROP_FEATURES" \
-E "$INTEROP_FILTER" -E "$INTEROP_FILTER"
alert-on-failure:
name: Alert on scheduled failure
needs: [minio-interop]
if: always() && github.event_name == 'schedule' && contains(needs.*.result, 'failure')
runs-on: ubuntu-latest
timeout-minutes: 10
permissions:
contents: read
issues: write
steps:
- uses: actions/checkout@9c091bb21b7c1c1d1991bb908d89e4e9dddfe3e0 # v7
with:
persist-credentials: false
- name: Open or update failure-tracking issue
uses: ./.github/actions/schedule-failure-issue
with:
github-token: ${{ secrets.GITHUB_TOKEN }}
+3 -10
View File
@@ -45,13 +45,6 @@
# docker-capable self-hosted `dind-sm-standard-2` label was the alternative but # docker-capable self-hosted `dind-sm-standard-2` label was the alternative but
# has fewer cores and reintroduces fleet-state risk for no reliability gain. # has fewer cores and reintroduces fleet-state risk for no reliability gain.
# DISABLED. This workflow is switched off in the repository's Actions settings
# (state: disabled_manually) and does not run on any trigger, including its cron
# and workflow_dispatch. That state lives in GitHub's UI and is invisible when
# reading this file, which has already misled at least one audit — hence this
# banner. Re-enabling is a UI action; anyone doing so should first check that the
# workflow still matches the current CI layout. See rustfs/backlog#1603.
#
name: mint name: mint
on: on:
@@ -70,13 +63,13 @@ on:
- core - core
- full - full
mint-image: mint-image:
description: "Mint image reference" description: "Mint image reference (empty = pinned default)"
required: false required: false
default: "minio/mint:edge" default: ""
schedule: schedule:
# Weekly, after the Sunday s3-tests full sweep (starts 02:00 UTC, up to # Weekly, after the Sunday s3-tests full sweep (starts 02:00 UTC, up to
# 3h) has finished, so the two never contend for the same runner pool. # 3h) has finished, so the two never contend for the same runner pool.
- cron: "0 6 * * 0" - cron: "41 6 * * 0"
env: env:
S3_ACCESS_KEY: rustfsadmin-ci S3_ACCESS_KEY: rustfsadmin-ci
+21 -1
View File
@@ -16,7 +16,7 @@ name: Nightly GNU Build
on: on:
schedule: schedule:
- cron: "0 0 * * *" - cron: "7 0 * * *"
timezone: "Asia/Shanghai" timezone: "Asia/Shanghai"
workflow_dispatch: workflow_dispatch:
@@ -194,3 +194,23 @@ jobs:
- name: Run HA leader failover live checks (three-node Raft cluster in Docker) - name: Run HA leader failover live checks (three-node Raft cluster in Docker)
run: bash scripts/test/vault_ha_kms_live.sh run: bash scripts/test/vault_ha_kms_live.sh
alert-on-failure:
name: Alert on scheduled failure
needs: [build, kms-vault-lane, kms-vault-ha-failover]
if: >-
always() && github.event_name == 'schedule' &&
(contains(needs.*.result, 'failure') || contains(needs.*.result, 'cancelled'))
runs-on: ubuntu-latest
timeout-minutes: 10
permissions:
contents: read
issues: write
steps:
- uses: actions/checkout@9c091bb21b7c1c1d1991bb908d89e4e9dddfe3e0 # v7
with:
persist-credentials: false
- name: Open or update failure-tracking issue
uses: ./.github/actions/schedule-failure-issue
with:
github-token: ${{ secrets.GITHUB_TOKEN }}
+67 -94
View File
@@ -22,22 +22,15 @@
# correctness cost (e.g. the #4221 fsync durability fix) is recorded, not # correctness cost (e.g. the #4221 fsync durability fix) is recorded, not
# blocked (rustfs/backlog#935 correction 1). # blocked (rustfs/backlog#935 correction 1).
# DISABLED. This workflow is switched off in the repository's Actions settings
# (state: disabled_manually) and does not run on any trigger, including its cron
# and workflow_dispatch. That state lives in GitHub's UI and is invisible when
# reading this file, which has already misled at least one audit — hence this
# banner. Re-enabling is a UI action; anyone doing so should first check that the
# workflow still matches the current CI layout. See rustfs/backlog#1603.
#
name: Performance A/B name: Performance A/B
on: on:
schedule: schedule:
- cron: "0 6 * * *" # 06:00 UTC nightly, against main - cron: "31 6 * * *" # 06:31 UTC nightly, against main
workflow_dispatch: workflow_dispatch:
inputs: inputs:
duration: duration:
description: "warp duration per round (short by default to fit the double-build budget)" description: "warp duration per round"
required: false required: false
default: "12s" default: "12s"
type: string type: string
@@ -46,12 +39,8 @@ on:
required: false required: false
default: false default: false
type: boolean type: boolean
push:
# Every main commit pre-builds and caches its release binary (perf-3) so the
# nightly A/B restores a ready baseline instead of paying the double build.
branches: [main]
permissions: permissions:
actions: read
contents: read contents: read
env: env:
@@ -59,83 +48,19 @@ env:
RUST_BACKTRACE: 1 RUST_BACKTRACE: 1
jobs: jobs:
# perf-3: on every push to main, build the release binary once and cache it
# keyed by commit SHA (rustfs-baseline-<sha>). The warp-ab measurements
# restore this instead of paying the ~32min-per-side source
# build. That double build is what pushed the expanded 24-cell nightly past its
# ceiling — 2026-07-11..07-14 all cancelled on the 120min timeout. Incremental
# builds off the shared cargo cache keep each push cheap, and building on the
# same sm-standard-2 runner the A/B measures on guarantees the cached binary is
# ABI-identical. Do NOT source this from build.yml's per-merge artifact: those
# are cancelled ~7/8 of the time and are not a reliable baseline.
build-baseline-cache:
name: Build + cache baseline binary
if: github.event_name == 'push'
runs-on: sm-standard-2
# Latest-wins: consumers only ever restore the binary for the *current*
# origin/main tip, so when pushes land faster than the ~65min build, a
# superseded build's output is dead weight — cancel it instead of stacking
# hour-long jobs on the shared runner pool. A skipped intermediate SHA at
# most costs one same-commit self-heal in the A/B job.
concurrency:
group: perf-baseline-build-main
cancel-in-progress: true
# #4806 put thin LTO + codegen-units=1 on [profile.release], pushing a
# single release build past 60min on this runner — every cache build on
# 2026-07-15 died on the old 60min ceiling ("exceeded the maximum execution
# time of 1h0m0s") and the cache never populated. The measured binary must
# keep the production profile, so the budget absorbs the build instead.
timeout-minutes: 100
steps:
- name: Checkout repository
uses: actions/checkout@9c091bb21b7c1c1d1991bb908d89e4e9dddfe3e0 # v7
with:
persist-credentials: false
- name: Setup Rust environment
uses: ./.github/actions/setup
with:
rust-version: stable
cache-shared-key: warp-ab-${{ hashFiles('**/Cargo.lock') }}
cache-save-if: ${{ github.ref == 'refs/heads/main' }}
- name: Build release rustfs
run: cargo build --release --bin rustfs
- name: Stage binary for cache
run: |
set -euo pipefail
mkdir -p baseline-bin
cp target/release/rustfs baseline-bin/rustfs
- name: Cache baseline binary by SHA
uses: actions/cache/save@55cc8345863c7cc4c66a329aec7e433d2d1c52a9 # v6
with:
path: baseline-bin/rustfs
key: rustfs-baseline-${{ github.sha }}
warp-ab: warp-ab:
name: Warp A/B budget gate name: Warp A/B budget gate
# Always run on schedule / manual dispatch. Never on push — that event only
# feeds build-baseline-cache above.
if: >-
github.event_name == 'schedule' ||
github.event_name == 'workflow_dispatch'
runs-on: sm-standard-2 runs-on: sm-standard-2
# With perf-3's cached baseline binary the common (cache-hit) nightly is # A normal nightly restores the last successful binary and builds only the
# measurement-only and finishes well under 50min. This ceiling stays # candidate; daily access keeps that cache warm. A cache miss may build both
# generous only to absorb the same-commit cache-miss self-heal (~65min # and needs room for the A/B run plus artifact and cache publication.
# single build with the post-#4806 LTO profile + measurement). A timeout timeout-minutes: 180
# surfaces via the alert-on-failure job (it fires on cancelled/timed-out,
# not just failure). perf-6 recalibrates the budget once the noise study
# lands.
timeout-minutes: 120
steps: steps:
- name: Checkout repository - name: Checkout repository
uses: actions/checkout@9c091bb21b7c1c1d1991bb908d89e4e9dddfe3e0 # v7 uses: actions/checkout@9c091bb21b7c1c1d1991bb908d89e4e9dddfe3e0 # v7
with: with:
persist-credentials: false persist-credentials: false
fetch-depth: 0 # baseline is built from origin/main fetch-depth: 0 # baseline may be an earlier successful scheduled head
- name: Setup Rust environment - name: Setup Rust environment
uses: ./.github/actions/setup uses: ./.github/actions/setup
@@ -163,24 +88,55 @@ jobs:
fi fi
echo "allow_regression=$allow" >> "$GITHUB_OUTPUT" echo "allow_regression=$allow" >> "$GITHUB_OUTPUT"
# perf-3: resolve the commits so the cache can be keyed by SHA. The # A failed regression run must keep comparing against the last known-good
# baseline is origin/main; the candidate is the checked-out ref. On the # scheduled head. Otherwise the next nightly would absorb the regression
# nightly (checkout == main) they are the same commit, so one cached binary # into its baseline and turn green without a fix.
# serves both phases and the run does zero source builds. - name: Find last successful scheduled baseline
id: scheduled_baseline
if: github.event_name == 'schedule'
uses: actions/github-script@ed597411d8f924073f98dfc5c65a23a2325f34cd # v8
with:
result-encoding: string
script: |
const { data } = await github.rest.actions.listWorkflowRuns({
owner: context.repo.owner,
repo: context.repo.repo,
workflow_id: "performance-ab.yml",
event: "schedule",
status: "success",
per_page: 1,
});
return data.workflow_runs[0]?.head_sha ?? "";
# Manual runs compare a selected ref with current main. Scheduled runs
# compare current main with the last successful scheduled head. With no
# history, the first run measures the candidate against itself and seeds
# that head only if the complete rig succeeds.
- name: Resolve baseline / candidate commits - name: Resolve baseline / candidate commits
id: commits id: commits
env:
SCHEDULED_BASELINE_SHA: ${{ steps.scheduled_baseline.outputs.result }}
run: | run: |
set -euo pipefail set -euo pipefail
baseline_sha="$(git rev-parse origin/main)"
candidate_sha="$(git rev-parse HEAD)" candidate_sha="$(git rev-parse HEAD)"
if [[ "${{ github.event_name }}" == "schedule" ]]; then
baseline_sha="${SCHEDULED_BASELINE_SHA:-$candidate_sha}"
if ! git merge-base --is-ancestor "$baseline_sha" "$candidate_sha"; then
echo "::error::scheduled baseline $baseline_sha is not an ancestor of candidate $candidate_sha" >&2
exit 1
fi
else
baseline_sha="$(git rev-parse origin/main)"
fi
git cat-file -e "${baseline_sha}^{commit}"
echo "baseline_sha=$baseline_sha" >> "$GITHUB_OUTPUT" echo "baseline_sha=$baseline_sha" >> "$GITHUB_OUTPUT"
echo "candidate_sha=$candidate_sha" >> "$GITHUB_OUTPUT" echo "candidate_sha=$candidate_sha" >> "$GITHUB_OUTPUT"
echo "baseline commit: $baseline_sha" echo "baseline commit: $baseline_sha"
echo "candidate commit: $candidate_sha" echo "candidate commit: $candidate_sha"
# Exact-key restore of the baseline binary built by build-baseline-cache # Exact-key restore of the candidate binary saved by its successful
# when origin/main last landed. A miss (binary evicted or not built yet) # scheduled run. A miss leaves cache-hit unset and falls back to a source
# leaves cache-hit unset and the rig falls back to a source build. # build of that known-good head.
- name: Restore cached baseline binary - name: Restore cached baseline binary
id: baseline_cache id: baseline_cache
uses: actions/cache/restore@55cc8345863c7cc4c66a329aec7e433d2d1c52a9 # v6 uses: actions/cache/restore@55cc8345863c7cc4c66a329aec7e433d2d1c52a9 # v6
@@ -270,11 +226,11 @@ jobs:
elif [[ "$selfheal_built" == "true" ]]; then elif [[ "$selfheal_built" == "true" ]]; then
base_src="source build (cache self-heal, saved as rustfs-baseline-$baseline_sha)" base_src="source build (cache self-heal, saved as rustfs-baseline-$baseline_sha)"
else else
base_src="isolated origin/main source build (saved as rustfs-baseline-$baseline_sha)" base_src="isolated baseline source build (saved as rustfs-baseline-$baseline_sha)"
fi fi
if [[ "$candidate_sha" == "$baseline_sha" ]]; then if [[ "$candidate_sha" == "$baseline_sha" ]]; then
# Nightly on main: the candidate is the same commit as the baseline, # No commits landed since the last successful baseline, so reuse
# so reuse the one binary for both phases and skip all builds. # the one binary for both phases and measure only rig drift.
args+=(--candidate-bin "$base_bin") args+=(--candidate-bin "$base_bin")
cand_src="same binary as baseline (same commit)" cand_src="same binary as baseline (same commit)"
elif [[ "$candidate_built" == "true" ]]; then elif [[ "$candidate_built" == "true" ]]; then
@@ -362,6 +318,23 @@ jobs:
fi fi
} >> "$GITHUB_STEP_SUMMARY" } >> "$GITHUB_STEP_SUMMARY"
- name: Stage successful candidate baseline
if: >-
steps.ab.outputs.status == '0' &&
steps.commits.outputs.baseline_sha != steps.commits.outputs.candidate_sha
run: |
set -euo pipefail
cp candidate-bin/rustfs baseline-bin/rustfs
- name: Cache successful candidate baseline
if: >-
steps.ab.outputs.status == '0' &&
steps.commits.outputs.baseline_sha != steps.commits.outputs.candidate_sha
uses: actions/cache/save@55cc8345863c7cc4c66a329aec7e433d2d1c52a9 # v6
with:
path: baseline-bin/rustfs
key: rustfs-baseline-${{ steps.commits.outputs.candidate_sha }}
# Scheduled failure alerting is handled by the alert-on-failure job below # Scheduled failure alerting is handled by the alert-on-failure job below
# (perf-2 consuming ci-8's schedule-failure-issue composite action). # (perf-2 consuming ci-8's schedule-failure-issue composite action).
+1 -1
View File
@@ -30,7 +30,7 @@ name: Runner Hygiene
on: on:
schedule: schedule:
- cron: "0 6 1 * *" # Monthly, 1st at 06:00 UTC (after the daily audit cron) - cron: "37 6 1 * *" # Monthly, 1st at 06:37 UTC
workflow_dispatch: workflow_dispatch:
permissions: permissions:
@@ -0,0 +1,57 @@
# Copyright 2024 RustFS Team
#
# Licensed under the Apache License, Version 2.0 (the "License");
# you may not use this file except in compliance with the License.
# You may obtain a copy of the License at
#
# http://www.apache.org/licenses/LICENSE-2.0
#
# Unless required by applicable law or agreed to in writing, software
# distributed under the License is distributed on an "AS IS" BASIS,
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
# See the License for the specific language governing permissions and
# limitations under the License.
name: Scheduled Validation Freshness
on:
schedule:
- cron: "47 23 * * *"
workflow_dispatch:
permissions:
contents: read
concurrency:
group: scheduled-validation-freshness
cancel-in-progress: false
jobs:
check-freshness:
name: Check scheduled validation freshness
runs-on: ubuntu-latest
timeout-minutes: 10
permissions:
actions: read
contents: read
issues: write
steps:
- uses: actions/checkout@9c091bb21b7c1c1d1991bb908d89e4e9dddfe3e0 # v7
with:
persist-credentials: false
- name: Check latest scheduled runs
env:
GH_TOKEN: ${{ secrets.GITHUB_TOKEN }}
run: |
set +e
python3 scripts/check_scheduled_validation_freshness.py \
--report "${RUNNER_TEMP}/scheduled-validation-freshness.md"
status=$?
cat "${RUNNER_TEMP}/scheduled-validation-freshness.md" >> "${GITHUB_STEP_SUMMARY}"
exit "${status}"
- name: Open or update freshness issue
if: failure()
uses: ./.github/actions/schedule-failure-issue
with:
github-token: ${{ secrets.GITHUB_TOKEN }}
details-file: ${{ runner.temp }}/scheduled-validation-freshness.md
@@ -0,0 +1,63 @@
# Copyright 2024 RustFS Team
#
# Licensed under the Apache License, Version 2.0 (the "License");
# you may not use this file except in compliance with the License.
# You may obtain a copy of the License at
#
# http://www.apache.org/licenses/LICENSE-2.0
#
# Unless required by applicable law or agreed to in writing, software
# distributed under the License is distributed on an "AS IS" BASIS,
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
# See the License for the specific language governing permissions and
# limitations under the License.
name: Scheduled Validation Watchdog
on:
workflow_run:
workflows:
- "Security Audit"
- "Build and Release"
- "Continuous Integration"
- "coverage"
- "e2e-nightly"
- "e2e-s3tests"
- "Fuzz"
- "mint"
- "minio-interop"
- "Nightly GNU Build"
- "Performance A/B"
- "Runner Hygiene"
types: [completed]
permissions:
contents: read
jobs:
alert-on-incomplete-run:
name: Alert on incomplete scheduled run
if: >-
github.event.workflow_run.event == 'schedule' &&
github.event.workflow_run.conclusion != 'success' &&
github.event.workflow_run.conclusion != 'failure'
runs-on: ubuntu-latest
timeout-minutes: 10
permissions:
actions: read
contents: read
issues: write
steps:
- uses: actions/checkout@9c091bb21b7c1c1d1991bb908d89e4e9dddfe3e0 # v7
with:
persist-credentials: false
- name: Open or update incomplete-run issue
uses: ./.github/actions/schedule-failure-issue
with:
github-token: ${{ secrets.GITHUB_TOKEN }}
workflow-name: ${{ github.event.workflow_run.name }}
source-run-id: ${{ github.event.workflow_run.id }}
source-run-attempt: ${{ github.event.workflow_run.run_attempt }}
source-event: ${{ github.event.workflow_run.event }}
source-ref-name: ${{ github.event.workflow_run.head_branch }}
source-sha: ${{ github.event.workflow_run.head_sha }}
-3
View File
@@ -39,9 +39,6 @@ mod kms_edge_cases_test;
#[cfg(test)] #[cfg(test)]
mod kms_fault_recovery_test; mod kms_fault_recovery_test;
#[cfg(test)]
mod test_runner;
#[cfg(test)] #[cfg(test)]
mod bucket_default_encryption_test; mod bucket_default_encryption_test;
-499
View File
@@ -1,499 +0,0 @@
// Copyright 2024 RustFS Team
//
// Licensed under the Apache License, Version 2.0 (the "License");
// you may not use this file except in compliance with the License.
// You may obtain a copy of the License at
//
// http://www.apache.org/licenses/LICENSE-2.0
//
#![allow(dead_code)]
// Unless required by applicable law or agreed to in writing, software
// distributed under the License is distributed on an "AS IS" BASIS,
// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
// See the License for the specific language governing permissions and
// limitations under the License.
//! Unified KMS test suite runner
//!
//! This module provides a unified interface for running KMS tests with categorization,
//! filtering, and comprehensive reporting capabilities.
use crate::common::init_logging;
use std::time::Instant;
use tokio::time::{Duration, sleep};
use tracing::{debug, error, info, warn};
/// Test category for organization and filtering
#[derive(Debug, Clone, PartialEq, Eq, Hash)]
pub enum TestCategory {
CoreFunctionality,
MultipartEncryption,
EdgeCases,
FaultRecovery,
Comprehensive,
Performance,
}
impl TestCategory {
pub fn as_str(&self) -> &'static str {
match self {
TestCategory::CoreFunctionality => "core-functionality",
TestCategory::MultipartEncryption => "multipart-encryption",
TestCategory::EdgeCases => "edge-cases",
TestCategory::FaultRecovery => "fault-recovery",
TestCategory::Comprehensive => "comprehensive",
TestCategory::Performance => "performance",
}
}
}
/// Test definition with metadata
#[derive(Debug, Clone)]
pub struct TestDefinition {
pub name: String,
pub description: String,
pub category: TestCategory,
pub estimated_duration: Duration,
pub is_critical: bool,
}
impl TestDefinition {
pub fn new(
name: impl Into<String>,
description: impl Into<String>,
category: TestCategory,
estimated_duration: Duration,
is_critical: bool,
) -> Self {
Self {
name: name.into(),
description: description.into(),
category,
estimated_duration,
is_critical,
}
}
}
/// Test execution result
#[derive(Debug, Clone)]
pub struct TestResult {
pub test_name: String,
pub category: TestCategory,
pub success: bool,
pub duration: Duration,
pub error_message: Option<String>,
}
impl TestResult {
pub fn success(test_name: String, category: TestCategory, duration: Duration) -> Self {
Self {
test_name,
category,
success: true,
duration,
error_message: None,
}
}
pub fn failure(test_name: String, category: TestCategory, duration: Duration, error: String) -> Self {
Self {
test_name,
category,
success: false,
duration,
error_message: Some(error),
}
}
}
/// Comprehensive test suite configuration
#[derive(Debug, Clone)]
pub struct TestSuiteConfig {
pub categories: Vec<TestCategory>,
pub include_critical_only: bool,
pub max_duration: Option<Duration>,
pub parallel_execution: bool,
}
impl Default for TestSuiteConfig {
fn default() -> Self {
Self {
categories: vec![
TestCategory::CoreFunctionality,
TestCategory::MultipartEncryption,
TestCategory::EdgeCases,
TestCategory::FaultRecovery,
TestCategory::Comprehensive,
],
include_critical_only: false,
max_duration: None,
parallel_execution: false,
}
}
}
/// Unified KMS test suite runner
pub struct KMSTestSuite {
tests: Vec<TestDefinition>,
config: TestSuiteConfig,
}
impl KMSTestSuite {
/// Create a new test suite with default configuration
pub fn new() -> Self {
let tests = vec![
// Core Functionality Tests
TestDefinition::new(
"test_local_kms_end_to_end",
"End-to-end KMS test with all encryption types",
TestCategory::CoreFunctionality,
Duration::from_secs(60),
true,
),
TestDefinition::new(
"test_local_kms_key_isolation",
"Test KMS key isolation and security",
TestCategory::CoreFunctionality,
Duration::from_secs(45),
true,
),
// Multipart Encryption Tests
TestDefinition::new(
"test_local_kms_multipart_upload",
"Test large file multipart upload with encryption",
TestCategory::MultipartEncryption,
Duration::from_secs(120),
true,
),
TestDefinition::new(
"test_step1_basic_single_file_encryption",
"Basic single file encryption test",
TestCategory::MultipartEncryption,
Duration::from_secs(30),
false,
),
TestDefinition::new(
"test_step2_basic_multipart_upload_without_encryption",
"Basic multipart upload without encryption",
TestCategory::MultipartEncryption,
Duration::from_secs(45),
false,
),
TestDefinition::new(
"test_step3_multipart_upload_with_sse_s3",
"Multipart upload with SSE-S3 encryption",
TestCategory::MultipartEncryption,
Duration::from_secs(60),
true,
),
TestDefinition::new(
"test_step4_large_multipart_upload_with_encryption",
"Large file multipart upload with encryption",
TestCategory::MultipartEncryption,
Duration::from_secs(90),
false,
),
TestDefinition::new(
"test_step5_all_encryption_types_multipart",
"All encryption types multipart test",
TestCategory::MultipartEncryption,
Duration::from_secs(120),
true,
),
// Edge Cases Tests
TestDefinition::new(
"test_kms_zero_byte_file_encryption",
"Test encryption of zero-byte files",
TestCategory::EdgeCases,
Duration::from_secs(20),
false,
),
TestDefinition::new(
"test_kms_single_byte_file_encryption",
"Test encryption of single-byte files",
TestCategory::EdgeCases,
Duration::from_secs(20),
false,
),
TestDefinition::new(
"test_kms_multipart_boundary_conditions",
"Test multipart upload boundary conditions",
TestCategory::EdgeCases,
Duration::from_secs(45),
false,
),
TestDefinition::new(
"test_kms_invalid_key_scenarios",
"Test invalid key scenarios",
TestCategory::EdgeCases,
Duration::from_secs(30),
false,
),
TestDefinition::new(
"test_kms_concurrent_encryption",
"Test concurrent encryption operations",
TestCategory::EdgeCases,
Duration::from_secs(60),
false,
),
TestDefinition::new(
"test_kms_key_validation_security",
"Test key validation security",
TestCategory::EdgeCases,
Duration::from_secs(30),
false,
),
// Fault Recovery Tests
TestDefinition::new(
"test_kms_key_directory_unavailable",
"Test KMS when key directory is unavailable",
TestCategory::FaultRecovery,
Duration::from_secs(45),
false,
),
TestDefinition::new(
"test_kms_corrupted_key_files",
"Test KMS with corrupted key files",
TestCategory::FaultRecovery,
Duration::from_secs(30),
false,
),
TestDefinition::new(
"test_kms_multipart_upload_interruption",
"Test multipart upload interruption recovery",
TestCategory::FaultRecovery,
Duration::from_secs(60),
false,
),
TestDefinition::new(
"test_kms_resource_constraints",
"Test KMS under resource constraints",
TestCategory::FaultRecovery,
Duration::from_secs(90),
false,
),
// Comprehensive Tests
TestDefinition::new(
"test_comprehensive_kms_full_workflow",
"Full KMS workflow comprehensive test",
TestCategory::Comprehensive,
Duration::from_secs(300),
true,
),
TestDefinition::new(
"test_comprehensive_stress_test",
"KMS stress test with large datasets",
TestCategory::Comprehensive,
Duration::from_secs(400),
false,
),
TestDefinition::new(
"test_comprehensive_key_isolation",
"Comprehensive key isolation test",
TestCategory::Comprehensive,
Duration::from_secs(180),
false,
),
TestDefinition::new(
"test_comprehensive_concurrent_operations",
"Comprehensive concurrent operations test",
TestCategory::Comprehensive,
Duration::from_secs(240),
false,
),
TestDefinition::new(
"test_comprehensive_performance_benchmark",
"KMS performance benchmark test",
TestCategory::Comprehensive,
Duration::from_secs(360),
false,
),
];
Self {
tests,
config: TestSuiteConfig::default(),
}
}
/// Configure the test suite
pub fn with_config(mut self, config: TestSuiteConfig) -> Self {
self.config = config;
self
}
/// Filter tests based on category
pub fn filter_by_category(&self, category: &TestCategory) -> Vec<&TestDefinition> {
self.tests.iter().filter(|test| &test.category == category).collect()
}
/// Filter tests based on criticality
pub fn filter_critical_tests(&self) -> Vec<&TestDefinition> {
self.tests.iter().filter(|test| test.is_critical).collect()
}
/// Get test summary by category
pub fn get_category_summary(&self) -> std::collections::HashMap<TestCategory, Vec<&TestDefinition>> {
let mut summary = std::collections::HashMap::new();
for test in &self.tests {
summary.entry(test.category.clone()).or_insert_with(Vec::new).push(test);
}
summary
}
/// Run the complete test suite
pub async fn run_test_suite(&self) -> Vec<TestResult> {
init_logging();
info!("🚀 Starting unified KMS test suite");
let start_time = Instant::now();
let mut results = Vec::new();
// Filter tests based on configuration
let tests_to_run: Vec<&TestDefinition> = self
.tests
.iter()
.filter(|test| self.config.categories.contains(&test.category))
.filter(|test| !self.config.include_critical_only || test.is_critical)
.collect();
info!("📊 Test plan: {} test(s) scheduled", tests_to_run.len());
for (i, test) in tests_to_run.iter().enumerate() {
info!(" {}. {} ({})", i + 1, test.name, test.category.as_str());
}
// Execute tests
for (i, test_def) in tests_to_run.iter().enumerate() {
info!("🧪 Running test {}/{}: {}", i + 1, tests_to_run.len(), test_def.name);
info!(" 📝 Description: {}", test_def.description);
info!(" 🏷️ Category: {}", test_def.category.as_str());
info!(" ⏱️ Estimated duration: {:?}", test_def.estimated_duration);
let test_start = Instant::now();
let result = self.run_single_test(test_def).await;
let test_duration = test_start.elapsed();
match result {
Ok(_) => {
info!("✅ Test passed: {} ({:.2}s)", test_def.name, test_duration.as_secs_f64());
results.push(TestResult::success(test_def.name.clone(), test_def.category.clone(), test_duration));
}
Err(e) => {
error!("❌ Test failed: {} ({:.2}s): {}", test_def.name, test_duration.as_secs_f64(), e);
results.push(TestResult::failure(
test_def.name.clone(),
test_def.category.clone(),
test_duration,
e.to_string(),
));
}
}
// Add delay between tests to avoid resource conflicts
if i < tests_to_run.len() - 1 {
debug!("⏸️ Waiting two seconds before the next test...");
sleep(Duration::from_secs(2)).await;
}
}
let total_duration = start_time.elapsed();
self.print_test_summary(&results, total_duration);
results
}
/// Run a single test by dispatching to the appropriate test function
async fn run_single_test(&self, test_def: &TestDefinition) -> Result<(), Box<dyn std::error::Error + Send + Sync>> {
// This is a placeholder for test dispatch logic
// In a real implementation, this would dispatch to actual test functions
warn!("⚠️ Test '{}' is not implemented in the unified runner; skipping", test_def.name);
Ok(())
}
/// Print comprehensive test summary
fn print_test_summary(&self, results: &[TestResult], total_duration: Duration) {
info!("📊 KMS test suite summary");
info!("⏱️ Total duration: {:.2} seconds", total_duration.as_secs_f64());
info!("📈 Total tests: {}", results.len());
let passed = results.iter().filter(|r| r.success).count();
let failed = results.iter().filter(|r| !r.success).count();
info!("✅ Passed: {}", passed);
info!("❌ Failed: {}", failed);
info!("📊 Success rate: {:.1}%", (passed as f64 / results.len() as f64) * 100.0);
// Summary by category
let mut category_summary: std::collections::HashMap<TestCategory, (usize, usize)> = std::collections::HashMap::new();
for result in results {
let (total, passed_count) = category_summary.entry(result.category.clone()).or_insert((0, 0));
*total += 1;
if result.success {
*passed_count += 1;
}
}
info!("📊 Category summary:");
for (category, (total, passed_count)) in category_summary {
info!(
" 🏷️ {}: {}/{} ({:.1}%)",
category.as_str(),
passed_count,
total,
(passed_count as f64 / total as f64) * 100.0
);
}
// List failed tests
if failed > 0 {
warn!("❌ Failing tests:");
for result in results.iter().filter(|r| !r.success) {
warn!(" - {}: {}", result.test_name, result.error_message.as_deref().unwrap_or("Unknown error"));
}
}
}
}
/// Quick test suite for critical tests only
#[tokio::test]
async fn test_kms_critical_suite() -> Result<(), Box<dyn std::error::Error + Send + Sync>> {
let config = TestSuiteConfig {
categories: vec![TestCategory::CoreFunctionality, TestCategory::MultipartEncryption],
include_critical_only: true,
max_duration: Some(Duration::from_secs(600)), // 10 minutes max
parallel_execution: false,
};
let suite = KMSTestSuite::new().with_config(config);
let results = suite.run_test_suite().await;
let failed_count = results.iter().filter(|r| !r.success).count();
if failed_count > 0 {
return Err(format!("Critical test suite failed: {failed_count} tests failed").into());
}
info!("✅ All critical tests passed");
Ok(())
}
/// Full comprehensive test suite
#[tokio::test]
async fn test_kms_full_suite() -> Result<(), Box<dyn std::error::Error + Send + Sync>> {
let suite = KMSTestSuite::new();
let results = suite.run_test_suite().await;
let total_tests = results.len();
let failed_count = results.iter().filter(|r| !r.success).count();
let success_rate = ((total_tests - failed_count) as f64 / total_tests as f64) * 100.0;
info!("📊 Full suite success rate: {:.1}%", success_rate);
// Allow up to 10% failure rate for non-critical tests
if success_rate < 90.0 {
return Err(format!("Test suite success rate too low: {success_rate:.1}%").into());
}
info!("✅ Full test suite succeeded");
Ok(())
}
+25 -12
View File
@@ -16,6 +16,7 @@ use crate::common::{RustFSTestClusterEnvironment, RustFSTestEnvironment, init_lo
use aws_sdk_s3::primitives::ByteStream; use aws_sdk_s3::primitives::ByteStream;
use http::header::{CONTENT_TYPE, HOST}; use http::header::{CONTENT_TYPE, HOST};
use reqwest::StatusCode; use reqwest::StatusCode;
use rustfs_config::{ENV_DRIVE_ACTIVE_CHECK_INTERVAL_SECS, ENV_NOTIFY_ENABLE};
use rustfs_signer::pre_sign_v4; use rustfs_signer::pre_sign_v4;
use rustfs_utils::egress::ENV_OUTBOUND_ALLOW_ORIGINS; use rustfs_utils::egress::ENV_OUTBOUND_ALLOW_ORIGINS;
use s3s::Body; use s3s::Body;
@@ -976,7 +977,8 @@ async fn test_get_object_lambda_rejects_disabled_target() -> Result<(), Box<dyn
init_logging(); init_logging();
let mut env = RustFSTestEnvironment::new().await?; let mut env = RustFSTestEnvironment::new().await?;
env.start_rustfs_server(vec![]).await?; env.start_rustfs_server_with_env(vec![], &[(ENV_NOTIFY_ENABLE, "true")])
.await?;
let bucket = "object-lambda-e2e-disabled-target"; let bucket = "object-lambda-e2e-disabled-target";
let key = "input.txt"; let key = "input.txt";
@@ -992,17 +994,24 @@ async fn test_get_object_lambda_rejects_disabled_target() -> Result<(), Box<dyn
.send() .send()
.await?; .await?;
configure_webhook_target_with_key_values( let queue_dir = format!("{}/disabled-target-queue", env.temp_dir);
&env, tokio::fs::create_dir_all(&queue_dir).await?;
"transformer", let config_url = format!("{}/rustfs/admin/v3/set-config-kv", env.url);
vec![ let directive = format!(
("endpoint", "http://127.0.0.1:9/transform".to_string()), "notify_webhook:transformer enable=off endpoint=\"http://127.0.0.1:9/transform\" auth_token=\"secret-token\" queue_dir=\"{queue_dir}\""
("auth_token", "secret-token".to_string()), );
("enable", "off".to_string()), let disable_response = signed_request(
], http::Method::PUT,
&config_url,
&env.access_key,
&env.secret_key,
Some(directive.into_bytes()),
Some("text/plain"),
) )
.await?; .await?;
wait_for_target_visibility(&env, "transformer").await?; let disable_status = disable_response.status();
let disable_body = disable_response.text().await?;
assert_eq!(disable_status, StatusCode::OK, "failed to disable target: {disable_body}");
let lambda_url = format!("{}/{}/{}?lambdaArn={}", env.url, bucket, key, urlencoding::encode(lambda_arn)); let lambda_url = format!("{}/{}/{}?lambdaArn={}", env.url, bucket, key, urlencoding::encode(lambda_arn));
let response = signed_request(http::Method::GET, &lambda_url, &env.access_key, &env.secret_key, None, None).await?; let response = signed_request(http::Method::GET, &lambda_url, &env.access_key, &env.secret_key, None, None).await?;
@@ -1021,7 +1030,8 @@ async fn test_configure_object_lambda_target_rejects_invalid_endpoint() -> Resul
init_logging(); init_logging();
let mut env = RustFSTestEnvironment::new().await?; let mut env = RustFSTestEnvironment::new().await?;
env.start_rustfs_server(vec![]).await?; env.start_rustfs_server_with_env(vec![], &[(ENV_NOTIFY_ENABLE, "true")])
.await?;
let bucket = "object-lambda-e2e-invalid-endpoint"; let bucket = "object-lambda-e2e-invalid-endpoint";
@@ -1064,7 +1074,8 @@ async fn test_configure_object_lambda_notify_webhook_rejects_response_header_tim
init_logging(); init_logging();
let mut env = RustFSTestEnvironment::new().await?; let mut env = RustFSTestEnvironment::new().await?;
env.start_rustfs_server(vec![]).await?; env.start_rustfs_server_with_env(vec![], &[(ENV_NOTIFY_ENABLE, "true")])
.await?;
let response = send_configure_webhook_target_request( let response = send_configure_webhook_target_request(
&env, &env,
@@ -1173,6 +1184,8 @@ async fn test_listen_notification_fans_in_remote_node_events() -> Result<(), Box
init_logging(); init_logging();
let mut cluster = RustFSTestClusterEnvironment::new(2).await?; let mut cluster = RustFSTestClusterEnvironment::new(2).await?;
cluster.set_env(ENV_NOTIFY_ENABLE, "true");
cluster.set_env(ENV_DRIVE_ACTIVE_CHECK_INTERVAL_SECS, "1");
cluster.start().await?; cluster.start().await?;
let bucket = "listen-notification-cluster"; let bucket = "listen-notification-cluster";
@@ -15,7 +15,6 @@
use crate::common::{RustFSTestClusterEnvironment, init_logging}; use crate::common::{RustFSTestClusterEnvironment, init_logging};
use aws_sdk_s3::error::SdkError; use aws_sdk_s3::error::SdkError;
use aws_sdk_s3::primitives::ByteStream; use aws_sdk_s3::primitives::ByteStream;
use aws_sdk_s3::types::CompletedMultipartUpload;
use tokio::time::{Duration, sleep}; use tokio::time::{Duration, sleep};
use tracing::info; use tracing::info;
use uuid::Uuid; use uuid::Uuid;
@@ -43,32 +42,18 @@ async fn list_parts_reports_missing_upload(
} }
} }
async fn complete_reports_missing_upload( async fn multipart_listing_reports_missing_upload(
client: &aws_sdk_s3::Client, client: &aws_sdk_s3::Client,
bucket: &str, bucket: &str,
key: &str, key: &str,
upload_id: &str, upload_id: &str,
) -> Result<bool, Box<dyn std::error::Error + Send + Sync>> { ) -> Result<bool, Box<dyn std::error::Error + Send + Sync>> {
let result = client let result = client.list_multipart_uploads().bucket(bucket).prefix(key).send().await?;
.complete_multipart_upload()
.bucket(bucket) Ok(!result
.key(key) .uploads()
.upload_id(upload_id) .iter()
.multipart_upload(CompletedMultipartUpload::builder().build()) .any(|upload| upload.key() == Some(key) && upload.upload_id() == Some(upload_id)))
.send()
.await;
match result {
Ok(_) => Ok(false),
Err(SdkError::ServiceError(err)) => {
let code = err.err().meta().code().unwrap_or("");
if code == "NoSuchUpload" {
Ok(true)
} else {
Err(format!("unexpected complete_multipart_upload service error: code={code}, err={err:?}").into())
}
}
Err(err) => Err(format!("unexpected complete_multipart_upload error: {err:?}").into()),
}
} }
async fn wait_for_cleanup_on_all_nodes( async fn wait_for_cleanup_on_all_nodes(
@@ -81,8 +66,8 @@ async fn wait_for_cleanup_on_all_nodes(
let mut all_cleaned = true; let mut all_cleaned = true;
for (idx, client) in clients.iter().enumerate() { for (idx, client) in clients.iter().enumerate() {
let list_parts_missing = list_parts_reports_missing_upload(client, bucket, key, upload_id).await?; let list_parts_missing = list_parts_reports_missing_upload(client, bucket, key, upload_id).await?;
let complete_missing = complete_reports_missing_upload(client, bucket, key, upload_id).await?; let listing_missing = multipart_listing_reports_missing_upload(client, bucket, key, upload_id).await?;
if !(list_parts_missing && complete_missing) { if !(list_parts_missing && listing_missing) {
info!("stale multipart still visible on node {} at attempt {}", idx, attempt + 1); info!("stale multipart still visible on node {} at attempt {}", idx, attempt + 1);
all_cleaned = false; all_cleaned = false;
break; break;
@@ -146,6 +131,10 @@ async fn test_stale_multipart_cleanup_removes_incomplete_upload_across_cluster()
1, 1,
"multipart upload should be visible before background cleanup" "multipart upload should be visible before background cleanup"
); );
assert!(
!multipart_listing_reports_missing_upload(&clients[2], CLEANUP_BUCKET, &key, &upload_id).await?,
"multipart upload listing should contain the upload before background cleanup"
);
wait_for_cleanup_on_all_nodes(&clients, CLEANUP_BUCKET, &key, &upload_id).await?; wait_for_cleanup_on_all_nodes(&clients, CLEANUP_BUCKET, &key, &upload_id).await?;
+90 -7
View File
@@ -42,8 +42,9 @@ use futures::lock::Mutex;
use metrics::counter; use metrics::counter;
use rustfs_filemeta::{FileInfo, ObjectPartInfo, RawFileInfo}; use rustfs_filemeta::{FileInfo, ObjectPartInfo, RawFileInfo};
use rustfs_io_metrics::internode_metrics::{ use rustfs_io_metrics::internode_metrics::{
INTERNODE_STAGE_READ_VERSION_REQUEST_ENCODE, INTERNODE_STAGE_READ_VERSION_RESPONSE_DECODE, INTERNODE_STAGE_BATCH_READ_VERSION_REQUEST_ENCODE, INTERNODE_STAGE_BATCH_READ_VERSION_RESPONSE_DECODE,
INTERNODE_STAGE_READ_VERSION_RPC_ROUNDTRIP, INTERNODE_STAGE_BATCH_READ_VERSION_RPC_ROUNDTRIP, INTERNODE_STAGE_READ_VERSION_REQUEST_ENCODE,
INTERNODE_STAGE_READ_VERSION_RESPONSE_DECODE, INTERNODE_STAGE_READ_VERSION_RPC_ROUNDTRIP,
}; };
use rustfs_protos::ChannelClass; use rustfs_protos::ChannelClass;
use rustfs_protos::evict_failed_connection; use rustfs_protos::evict_failed_connection;
@@ -98,6 +99,7 @@ const NS_SCANNER_CAPABILITY_PROBE_TIMEOUT: Duration = Duration::from_secs(5);
const REMOTE_DISK_READ_RETRY_BASE_BACKOFF: Duration = Duration::from_millis(50); const REMOTE_DISK_READ_RETRY_BASE_BACKOFF: Duration = Duration::from_millis(50);
const ENV_RUSTFS_METADATA_BATCH_READ: &str = "RUSTFS_METADATA_BATCH_READ"; const ENV_RUSTFS_METADATA_BATCH_READ: &str = "RUSTFS_METADATA_BATCH_READ";
const LEGACY_ENV_RUSTFS_BATCH_METADATA_RPC: &str = "RUSTFS_BATCH_METADATA_RPC"; const LEGACY_ENV_RUSTFS_BATCH_METADATA_RPC: &str = "RUSTFS_BATCH_METADATA_RPC";
const ENV_RUSTFS_GET_METADATA_READ_VERSION_COALESCE: &str = "RUSTFS_GET_METADATA_READ_VERSION_COALESCE";
const BATCH_METADATA_RPC_OFF: &str = "off"; const BATCH_METADATA_RPC_OFF: &str = "off";
const BATCH_METADATA_RPC_AUTO: &str = "auto"; const BATCH_METADATA_RPC_AUTO: &str = "auto";
const BATCH_METADATA_RPC_ON: &str = "on"; const BATCH_METADATA_RPC_ON: &str = "on";
@@ -202,7 +204,8 @@ fn parse_batch_metadata_rpc_mode(raw: &str) -> BatchMetadataRpcMode {
} }
fn batch_metadata_rpc_mode_from_env() -> BatchMetadataRpcMode { fn batch_metadata_rpc_mode_from_env() -> BatchMetadataRpcMode {
rustfs_utils::get_env_opt_str(ENV_RUSTFS_METADATA_BATCH_READ) rustfs_utils::get_env_opt_str(ENV_RUSTFS_GET_METADATA_READ_VERSION_COALESCE)
.or_else(|| rustfs_utils::get_env_opt_str(ENV_RUSTFS_METADATA_BATCH_READ))
.or_else(|| rustfs_utils::get_env_opt_str(LEGACY_ENV_RUSTFS_BATCH_METADATA_RPC)) .or_else(|| rustfs_utils::get_env_opt_str(LEGACY_ENV_RUSTFS_BATCH_METADATA_RPC))
.as_deref() .as_deref()
.map(parse_batch_metadata_rpc_mode) .map(parse_batch_metadata_rpc_mode)
@@ -1826,6 +1829,12 @@ fn record_read_version_stage(stage: &'static str, started_at: Option<Instant>) {
} }
} }
fn record_batch_read_version_stage(stage: &'static str, started_at: Option<Instant>) {
if let Some(started_at) = started_at {
crate::cluster::rpc::runtime_sources::record_remote_disk_grpc_batch_read_version_stage(stage, started_at.elapsed());
}
}
/// Aggregate encoded size (bytes) of a `ReadMultiple` response, preferring the msgpack payloads /// Aggregate encoded size (bytes) of a `ReadMultiple` response, preferring the msgpack payloads
/// and falling back to the JSON compatibility strings. Used to size the RPC for the payload /// and falling back to the JSON compatibility strings. Used to size the RPC for the payload
/// histogram / large-payload alerting (grpc-optimization P0 instrumentation). /// histogram / large-payload alerting (grpc-optimization P0 instrumentation).
@@ -1936,6 +1945,27 @@ fn decode_batch_read_version_response_items(
Ok(batch_read_version_resps) Ok(batch_read_version_resps)
} }
fn batch_read_version_request_payload_len(req: &BatchReadVersionReq, req_json: &str, req_bin: &[u8]) -> usize {
req.items
.iter()
.fold(req_json.len().saturating_add(req_bin.len()), |total, item| {
total
.saturating_add(item.org_volume.len())
.saturating_add(item.volume.len())
.saturating_add(item.path.len())
.saturating_add(item.version_id.len())
})
}
fn batch_read_version_response_payload_len(response: &BatchReadVersionResponse) -> usize {
response
.batch_read_version_resps
.iter()
.map(String::len)
.sum::<usize>()
.saturating_add(response.batch_read_version_resps_bin.iter().map(Bytes::len).sum::<usize>())
}
fn validate_decoded_file_info(file_info: &FileInfo) -> Result<()> { fn validate_decoded_file_info(file_info: &FileInfo) -> Result<()> {
file_info.validate_for_metadata_read().map_err(Into::into) file_info.validate_for_metadata_read().map_err(Into::into)
} }
@@ -2837,14 +2867,19 @@ impl DiskAPI for RemoteDisk {
state = "started", state = "started",
"Remote disk RPC started" "Remote disk RPC started"
); );
let batch_read_version_attribution_enabled = rustfs_io_metrics::get_stage_metrics_enabled();
let encode_started = read_version_stage_timer(batch_read_version_attribution_enabled);
let batch_read_version_req = compat_json(&req)?; let batch_read_version_req = compat_json(&req)?;
let batch_read_version_req_bin = encode_msgpack(&req)?; let batch_read_version_req_bin = encode_msgpack(&req)?;
record_batch_read_version_stage(INTERNODE_STAGE_BATCH_READ_VERSION_REQUEST_ENCODE, encode_started);
let request_payload_bytes = batch_read_version_attribution_enabled
.then(|| batch_read_version_request_payload_len(&req, &batch_read_version_req, &batch_read_version_req_bin));
let batch_result = self let batch_result = self
.execute_with_timeout_for_op( .execute_with_timeout_for_op(
"batch_read_version", "batch_read_version",
move || async move { move || async move {
let disk = self.disk_ref().await; let disk = self.disk_ref().await;
let disk_len = disk.len();
let mut client = self let mut client = self
.get_bulk_client() .get_bulk_client()
.await .await
@@ -2855,9 +2890,20 @@ impl DiskAPI for RemoteDisk {
batch_read_version_req_bin: batch_read_version_req_bin.into(), batch_read_version_req_bin: batch_read_version_req_bin.into(),
}); });
crate::cluster::rpc::runtime_sources::record_remote_disk_grpc_batch_read_version_request();
if let Some(request_payload_bytes) = request_payload_bytes {
crate::cluster::rpc::runtime_sources::record_remote_disk_grpc_batch_read_version_sent_bytes(
request_payload_bytes.saturating_add(disk_len),
);
}
let rpc_started = read_version_stage_timer(batch_read_version_attribution_enabled);
let response = match client.batch_read_version(request).await { let response = match client.batch_read_version(request).await {
Ok(response) => response.into_inner(), Ok(response) => {
record_batch_read_version_stage(INTERNODE_STAGE_BATCH_READ_VERSION_RPC_ROUNDTRIP, rpc_started);
response.into_inner()
}
Err(status) if status.code() == Code::Unimplemented => { Err(status) if status.code() == Code::Unimplemented => {
record_batch_read_version_stage(INTERNODE_STAGE_BATCH_READ_VERSION_RPC_ROUNDTRIP, rpc_started);
if mode.should_fallback_on_unimplemented() { if mode.should_fallback_on_unimplemented() {
record_batch_read_version_gate_decision(mode, BATCH_READ_VERSION_GATE_FALLBACK_UNIMPLEMENTED); record_batch_read_version_gate_decision(mode, BATCH_READ_VERSION_GATE_FALLBACK_UNIMPLEMENTED);
warn!( warn!(
@@ -2874,6 +2920,7 @@ impl DiskAPI for RemoteDisk {
} }
record_batch_read_version_gate_decision(mode, BATCH_READ_VERSION_GATE_UNSUPPORTED_NO_FALLBACK); record_batch_read_version_gate_decision(mode, BATCH_READ_VERSION_GATE_UNSUPPORTED_NO_FALLBACK);
crate::cluster::rpc::runtime_sources::record_remote_disk_grpc_batch_read_version_error();
warn!( warn!(
event = EVENT_REMOTE_DISK_RPC, event = EVENT_REMOTE_DISK_RPC,
component = LOG_COMPONENT_ECSTORE, component = LOG_COMPONENT_ECSTORE,
@@ -2886,14 +2933,33 @@ impl DiskAPI for RemoteDisk {
); );
return Err(Error::from(status)); return Err(Error::from(status));
} }
Err(status) => return Err(Error::from(status)), Err(status) => {
record_batch_read_version_stage(INTERNODE_STAGE_BATCH_READ_VERSION_RPC_ROUNDTRIP, rpc_started);
crate::cluster::rpc::runtime_sources::record_remote_disk_grpc_batch_read_version_error();
return Err(Error::from(status));
}
}; };
if !response.success { if !response.success {
crate::cluster::rpc::runtime_sources::record_remote_disk_grpc_batch_read_version_error();
return Err(response.error.unwrap_or_default().into()); return Err(response.error.unwrap_or_default().into());
} }
decode_batch_read_version_response_items(response, &self.endpoint).map(Some) crate::cluster::rpc::runtime_sources::record_remote_disk_grpc_batch_read_version_recv_bytes(
batch_read_version_response_payload_len(&response),
);
let decode_started = read_version_stage_timer(batch_read_version_attribution_enabled);
match decode_batch_read_version_response_items(response, &self.endpoint) {
Ok(batch_read_version_resps) => {
record_batch_read_version_stage(INTERNODE_STAGE_BATCH_READ_VERSION_RESPONSE_DECODE, decode_started);
Ok(Some(batch_read_version_resps))
}
Err(err) => {
record_batch_read_version_stage(INTERNODE_STAGE_BATCH_READ_VERSION_RESPONSE_DECODE, decode_started);
crate::cluster::rpc::runtime_sources::record_remote_disk_grpc_batch_read_version_error();
Err(err)
}
}
}, },
get_max_timeout_duration(), get_max_timeout_duration(),
) )
@@ -4621,6 +4687,7 @@ mod tests {
} else { } else {
"file version not found".to_string() "file version not found".to_string()
}, },
error_code: if success { 0 } else { DiskError::FileVersionNotFound.to_u32() },
} }
} }
@@ -4740,6 +4807,7 @@ mod tests {
fn batch_metadata_rpc_mode_uses_documented_env_before_legacy_alias() { fn batch_metadata_rpc_mode_uses_documented_env_before_legacy_alias() {
temp_env::with_vars( temp_env::with_vars(
[ [
(ENV_RUSTFS_GET_METADATA_READ_VERSION_COALESCE, None::<&str>),
(ENV_RUSTFS_METADATA_BATCH_READ, Some("auto")), (ENV_RUSTFS_METADATA_BATCH_READ, Some("auto")),
(LEGACY_ENV_RUSTFS_BATCH_METADATA_RPC, Some("on")), (LEGACY_ENV_RUSTFS_BATCH_METADATA_RPC, Some("on")),
], ],
@@ -4749,10 +4817,25 @@ mod tests {
); );
} }
#[test]
fn batch_metadata_rpc_mode_uses_get_coalescer_env_before_batch_env() {
temp_env::with_vars(
[
(ENV_RUSTFS_GET_METADATA_READ_VERSION_COALESCE, Some("on")),
(ENV_RUSTFS_METADATA_BATCH_READ, Some("off")),
(LEGACY_ENV_RUSTFS_BATCH_METADATA_RPC, Some("off")),
],
|| {
assert_eq!(batch_metadata_rpc_mode_from_env(), BatchMetadataRpcMode::On);
},
);
}
#[test] #[test]
fn batch_metadata_rpc_mode_falls_back_to_legacy_env_alias() { fn batch_metadata_rpc_mode_falls_back_to_legacy_env_alias() {
temp_env::with_vars( temp_env::with_vars(
[ [
(ENV_RUSTFS_GET_METADATA_READ_VERSION_COALESCE, None::<&str>),
(ENV_RUSTFS_METADATA_BATCH_READ, None::<&str>), (ENV_RUSTFS_METADATA_BATCH_READ, None::<&str>),
(LEGACY_ENV_RUSTFS_BATCH_METADATA_RPC, Some("on")), (LEGACY_ENV_RUSTFS_BATCH_METADATA_RPC, Some("on")),
], ],
@@ -14,9 +14,10 @@
use rustfs_io_metrics::internode_metrics::{ use rustfs_io_metrics::internode_metrics::{
INTERNODE_MSGPACK_CODEC_JSON, INTERNODE_MSGPACK_CODEC_MSGPACK, INTERNODE_MSGPACK_DIRECTION_RESPONSE, INTERNODE_MSGPACK_CODEC_JSON, INTERNODE_MSGPACK_CODEC_MSGPACK, INTERNODE_MSGPACK_DIRECTION_RESPONSE,
INTERNODE_OPERATION_GRPC_READ_ALL, INTERNODE_OPERATION_GRPC_READ_MULTIPLE, INTERNODE_OPERATION_GRPC_READ_VERSION, INTERNODE_OPERATION_GRPC_BATCH_READ_VERSION, INTERNODE_OPERATION_GRPC_READ_ALL, INTERNODE_OPERATION_GRPC_READ_MULTIPLE,
INTERNODE_OPERATION_GRPC_WRITE_ALL, INTERNODE_OPERATION_PUT_FILE_STREAM, INTERNODE_OPERATION_READ_FILE_STREAM, INTERNODE_OPERATION_GRPC_READ_VERSION, INTERNODE_OPERATION_GRPC_WRITE_ALL, INTERNODE_OPERATION_PUT_FILE_STREAM,
INTERNODE_TRANSPORT_BACKEND_GRPC, INTERNODE_TRANSPORT_BACKEND_TCP_HTTP, global_internode_metrics, INTERNODE_OPERATION_READ_FILE_STREAM, INTERNODE_TRANSPORT_BACKEND_GRPC, INTERNODE_TRANSPORT_BACKEND_TCP_HTTP,
global_internode_metrics,
}; };
use std::time::Duration; use std::time::Duration;
@@ -93,6 +94,59 @@ pub(crate) fn record_remote_disk_grpc_read_version_request() {
); );
} }
pub(crate) fn record_remote_disk_grpc_batch_read_version_request() {
if !rustfs_io_metrics::get_stage_metrics_enabled() {
return;
}
global_internode_metrics().record_outgoing_request_for_operation_and_backend(
INTERNODE_OPERATION_GRPC_BATCH_READ_VERSION,
INTERNODE_TRANSPORT_BACKEND_GRPC,
);
}
pub(crate) fn record_remote_disk_grpc_batch_read_version_stage(stage: &'static str, duration: Duration) {
if !rustfs_io_metrics::get_stage_metrics_enabled() {
return;
}
global_internode_metrics().record_stage_duration_for_operation_and_backend(
INTERNODE_OPERATION_GRPC_BATCH_READ_VERSION,
INTERNODE_TRANSPORT_BACKEND_GRPC,
stage,
duration,
);
}
pub(crate) fn record_remote_disk_grpc_batch_read_version_error() {
if !rustfs_io_metrics::get_stage_metrics_enabled() {
return;
}
global_internode_metrics()
.record_error_for_operation_and_backend(INTERNODE_OPERATION_GRPC_BATCH_READ_VERSION, INTERNODE_TRANSPORT_BACKEND_GRPC);
}
pub(crate) fn record_remote_disk_grpc_batch_read_version_sent_bytes(bytes: usize) {
if !rustfs_io_metrics::get_stage_metrics_enabled() {
return;
}
global_internode_metrics().record_sent_bytes_for_operation_and_backend(
INTERNODE_OPERATION_GRPC_BATCH_READ_VERSION,
INTERNODE_TRANSPORT_BACKEND_GRPC,
bytes,
);
}
pub(crate) fn record_remote_disk_grpc_batch_read_version_recv_bytes(bytes: usize) {
if !rustfs_io_metrics::get_stage_metrics_enabled() {
return;
}
global_internode_metrics().record_recv_bytes_for_operation_and_backend(
INTERNODE_OPERATION_GRPC_BATCH_READ_VERSION,
INTERNODE_TRANSPORT_BACKEND_GRPC,
bytes,
);
record_grpc_payload_size(INTERNODE_OPERATION_GRPC_BATCH_READ_VERSION, bytes);
}
pub(crate) fn record_remote_disk_grpc_read_version_error() { pub(crate) fn record_remote_disk_grpc_read_version_error() {
if !rustfs_io_metrics::get_stage_metrics_enabled() { if !rustfs_io_metrics::get_stage_metrics_enabled() {
return; return;
File diff suppressed because it is too large Load Diff
+69 -25
View File
@@ -44,6 +44,8 @@ pub const PART_TRANSACTION_ROLLBACK: &str = "rollback";
const LOG_COMPONENT_ECSTORE: &str = "ecstore"; const LOG_COMPONENT_ECSTORE: &str = "ecstore";
const LOG_SUBSYSTEM_DISK: &str = "disk"; const LOG_SUBSYSTEM_DISK: &str = "disk";
const EVENT_DISK_PART_ERR_UNCLASSIFIED: &str = "disk_part_err_unclassified"; const EVENT_DISK_PART_ERR_UNCLASSIFIED: &str = "disk_part_err_unclassified";
const ENV_BATCH_READ_VERSION_SERVER_PARALLELISM: &str = "RUSTFS_BATCH_READ_VERSION_SERVER_PARALLELISM";
const BATCH_READ_VERSION_SERVER_PARALLELISM: usize = 4;
pub fn part_transaction_path(part_path: &str) -> String { pub fn part_transaction_path(part_path: &str) -> String {
match part_path.rsplit_once('/') { match part_path.rsplit_once('/') {
@@ -62,6 +64,7 @@ use bytes::Bytes;
use endpoint::Endpoint; use endpoint::Endpoint;
use error::DiskError; use error::DiskError;
use error::{Error, Result}; use error::{Error, Result};
use futures::stream::{self, StreamExt};
use local::LocalDisk; use local::LocalDisk;
use rustfs_filemeta::{FileInfo, ObjectPartInfo, RawFileInfo}; use rustfs_filemeta::{FileInfo, ObjectPartInfo, RawFileInfo};
use rustfs_madmin::info_commands::DiskMetrics; use rustfs_madmin::info_commands::DiskMetrics;
@@ -417,6 +420,14 @@ impl DiskAPI for Disk {
} }
} }
#[tracing::instrument(level = "trace", skip_all)]
async fn batch_read_version(&self, req: BatchReadVersionReq) -> Result<Vec<BatchReadVersionResp>> {
match self {
Disk::Local(local_disk) => local_disk.batch_read_version(req).await,
Disk::Remote(remote_disk) => remote_disk.batch_read_version(req).await,
}
}
#[tracing::instrument(level = "trace", skip_all)] #[tracing::instrument(level = "trace", skip_all)]
async fn read_xl(&self, volume: &str, path: &str, read_data: bool) -> Result<RawFileInfo> { async fn read_xl(&self, volume: &str, path: &str, read_data: bool) -> Result<RawFileInfo> {
match self { match self {
@@ -1028,36 +1039,47 @@ where
D: DiskAPI + ?Sized, D: DiskAPI + ?Sized,
{ {
validate_batch_read_version_item_count(req.items.len())?; validate_batch_read_version_item_count(req.items.len())?;
let parallelism = batch_read_version_server_parallelism();
let mut responses = Vec::with_capacity(req.items.len()); let mut responses = stream::iter(req.items.into_iter().enumerate())
for (index, item) in req.items.iter().enumerate() { .map(|(index, item)| async move {
let response = match disk match disk
.read_version(&item.org_volume, &item.volume, &item.path, &item.version_id, &req.opts) .read_version(&item.org_volume, &item.volume, &item.path, &item.version_id, &req.opts)
.await .await
{ {
Ok(file_info) => BatchReadVersionResp { Ok(file_info) => BatchReadVersionResp {
index, index,
path: item.path.clone(), path: item.path,
version_id: item.version_id.clone(), version_id: item.version_id,
success: true, success: true,
file_info, file_info,
error: String::new(), error: String::new(),
}, error_code: 0,
Err(err) => BatchReadVersionResp { },
index, Err(err) => BatchReadVersionResp {
path: item.path.clone(), index,
version_id: item.version_id.clone(), path: item.path,
success: false, version_id: item.version_id,
file_info: FileInfo::default(), success: false,
error: err.to_string(), file_info: FileInfo::default(),
}, error: err.to_string(),
}; error_code: err.to_u32(),
responses.push(response); },
} }
})
.buffer_unordered(parallelism)
.collect::<Vec<_>>()
.await;
responses.sort_unstable_by_key(|response| response.index);
Ok(responses) Ok(responses)
} }
fn batch_read_version_server_parallelism() -> usize {
rustfs_utils::get_env_usize(ENV_BATCH_READ_VERSION_SERVER_PARALLELISM, BATCH_READ_VERSION_SERVER_PARALLELISM)
.clamp(1, BATCH_READ_VERSION_MAX_ITEMS)
}
#[derive(Debug, Default, Serialize, Deserialize)] #[derive(Debug, Default, Serialize, Deserialize)]
pub struct CheckPartsResp { pub struct CheckPartsResp {
pub results: Vec<usize>, pub results: Vec<usize>,
@@ -1322,6 +1344,8 @@ pub struct BatchReadVersionResp {
pub success: bool, pub success: bool,
pub file_info: FileInfo, pub file_info: FileInfo,
pub error: String, pub error: String,
#[serde(default)]
pub error_code: u32,
} }
pub fn validate_batch_read_version_item_count(item_count: usize) -> Result<()> { pub fn validate_batch_read_version_item_count(item_count: usize) -> Result<()> {
@@ -1417,6 +1441,26 @@ mod tests {
assert!(!partial_valid_location.valid()); assert!(!partial_valid_location.valid());
} }
#[test]
fn batch_read_version_server_parallelism_defaults_to_conservative_four() {
temp_env::with_var(ENV_BATCH_READ_VERSION_SERVER_PARALLELISM, None::<&str>, || {
assert_eq!(batch_read_version_server_parallelism(), 4);
});
}
#[test]
fn batch_read_version_server_parallelism_honors_env_with_bounds() {
temp_env::with_var(ENV_BATCH_READ_VERSION_SERVER_PARALLELISM, Some("8"), || {
assert_eq!(batch_read_version_server_parallelism(), 8);
});
temp_env::with_var(ENV_BATCH_READ_VERSION_SERVER_PARALLELISM, Some("0"), || {
assert_eq!(batch_read_version_server_parallelism(), 1);
});
temp_env::with_var(ENV_BATCH_READ_VERSION_SERVER_PARALLELISM, Some("9999"), || {
assert_eq!(batch_read_version_server_parallelism(), BATCH_READ_VERSION_MAX_ITEMS);
});
}
/// Test FileInfoVersions find_version_index /// Test FileInfoVersions find_version_index
#[test] #[test]
fn test_file_info_versions_find_version_index() { fn test_file_info_versions_find_version_index() {
+8
View File
@@ -81,6 +81,14 @@ pub fn shutdown_background_monitors() {
cluster::rpc::shutdown_background_monitors(); cluster::rpc::shutdown_background_monitors();
} }
/// Publish that the process is ready to serve user-object GET traffic.
///
/// Experimental metadata coalescing is allowed to run only after this point so
/// startup and internal metadata reads keep the original per-disk path.
pub fn mark_get_metadata_read_version_coalescing_service_ready() {
runtime::global::mark_get_metadata_read_version_coalescing_service_ready();
}
#[cfg(test)] #[cfg(test)]
mod rio_tests { mod rio_tests {
#[test] #[test]
+14 -1
View File
@@ -25,7 +25,10 @@ use lazy_static::lazy_static;
use rustfs_lock::client::LockClient; use rustfs_lock::client::LockClient;
use std::{ use std::{
collections::HashMap, collections::HashMap,
sync::{Arc, OnceLock}, sync::{
Arc, OnceLock,
atomic::{AtomicBool, Ordering},
},
time::SystemTime, time::SystemTime,
}; };
use tokio::sync::{OnceCell, RwLock}; use tokio::sync::{OnceCell, RwLock};
@@ -37,6 +40,16 @@ pub const DISK_MIN_INODES: u64 = 1000;
pub const DISK_FILL_FRACTION: f64 = 0.99; pub const DISK_FILL_FRACTION: f64 = 0.99;
pub const DISK_RESERVE_FRACTION: f64 = 0.15; pub const DISK_RESERVE_FRACTION: f64 = 0.15;
static GET_METADATA_READ_VERSION_COALESCING_SERVICE_READY: AtomicBool = AtomicBool::new(false);
pub(crate) fn mark_get_metadata_read_version_coalescing_service_ready() {
GET_METADATA_READ_VERSION_COALESCING_SERVICE_READY.store(true, Ordering::Release);
}
pub(crate) fn get_metadata_read_version_coalescing_service_ready() -> bool {
GET_METADATA_READ_VERSION_COALESCING_SERVICE_READY.load(Ordering::Acquire)
}
// Global singletons for backward compatibility with MinIO port. // Global singletons for backward compatibility with MinIO port.
// These should be migrated to AppContext over time. // These should be migrated to AppContext over time.
// See issue #730 for migration plan. // See issue #730 for migration plan.
+9
View File
@@ -160,6 +160,10 @@ pub struct InstanceContext {
/// workers (scanner/heal/tier/lifecycle) without touching another instance. /// workers (scanner/heal/tier/lifecycle) without touching another instance.
/// Replaces the process-global cancel-token static. /// Replaces the process-global cancel-token static.
background_cancel_token: OnceLock<CancellationToken>, background_cancel_token: OnceLock<CancellationToken>,
/// Serializes decommission data-movement operations with cancellation and
/// a subsequent restart. Readers are held across one object side effect;
/// the transition path takes the writer after cancelling the routine.
decommission_operation_gate: Arc<RwLock<()>>,
/// Resolves object-encryption material at the application boundary. /// Resolves object-encryption material at the application boundary.
object_encryption_resolver: OnceLock<Arc<dyn ObjectEncryptionResolver>>, object_encryption_resolver: OnceLock<Arc<dyn ObjectEncryptionResolver>>,
tier_delete_journal_recovery_stores: std::sync::Mutex<HashSet<Uuid>>, tier_delete_journal_recovery_stores: std::sync::Mutex<HashSet<Uuid>>,
@@ -200,6 +204,7 @@ impl InstanceContext {
local_disk_set_drives: Arc::new(RwLock::new(Vec::new())), local_disk_set_drives: Arc::new(RwLock::new(Vec::new())),
bucket_metadata_sys: std::sync::Mutex::new(None), bucket_metadata_sys: std::sync::Mutex::new(None),
background_cancel_token: OnceLock::new(), background_cancel_token: OnceLock::new(),
decommission_operation_gate: Arc::new(RwLock::new(())),
object_encryption_resolver: OnceLock::new(), object_encryption_resolver: OnceLock::new(),
tier_delete_journal_recovery_stores: std::sync::Mutex::new(HashSet::new()), tier_delete_journal_recovery_stores: std::sync::Mutex::new(HashSet::new()),
transition_transaction_recovery_stores: std::sync::Mutex::new(HashSet::new()), transition_transaction_recovery_stores: std::sync::Mutex::new(HashSet::new()),
@@ -218,6 +223,10 @@ impl InstanceContext {
self.lock_manager.clone() self.lock_manager.clone()
} }
pub(crate) fn decommission_operation_gate(&self) -> Arc<RwLock<()>> {
Arc::clone(&self.decommission_operation_gate)
}
/// Install the application-owned object-encryption resolver once. /// Install the application-owned object-encryption resolver once.
pub fn set_object_encryption_resolver( pub fn set_object_encryption_resolver(
&self, &self,
@@ -53,11 +53,12 @@ use crate::diagnostics::get::{
GetObjectFailureReason, classify_disk_error, get_stage_timer_if_enabled, record_get_object_pipeline_failure, GetObjectFailureReason, classify_disk_error, get_stage_timer_if_enabled, record_get_object_pipeline_failure,
record_get_object_pipeline_failure_for_path, record_get_stage_duration_if_enabled, record_get_object_pipeline_failure_for_path, record_get_stage_duration_if_enabled,
}; };
use crate::disk::disk_store::DiskStoreRenameDataExt; use crate::disk::disk_store::{DiskStoreRenameDataExt, get_drive_metadata_timeout};
use crate::disk::local::DELETE_DATA_DIR_MARKER_PREFIX; use crate::disk::local::DELETE_DATA_DIR_MARKER_PREFIX;
use crate::disk::{ use crate::disk::{
DataDirDeleteStatus, OldCurrentSize, PART_TRANSACTION_NEW_META, PART_TRANSACTION_OLD_META, PART_TRANSACTION_ROLLBACK, BATCH_READ_VERSION_MAX_ITEMS, BatchReadVersionItem, BatchReadVersionReq, BatchReadVersionResp, DataDirDeleteStatus, Disk,
PartTransactionAction, STORAGE_FORMAT_FILE_BACKUP, part_transaction_path, OldCurrentSize, PART_TRANSACTION_NEW_META, PART_TRANSACTION_OLD_META, PART_TRANSACTION_ROLLBACK, PartTransactionAction,
STORAGE_FORMAT_FILE_BACKUP, part_transaction_path,
}; };
use crate::erasure::coding::BitrotReader; use crate::erasure::coding::BitrotReader;
use crate::io_support::bitrot::ShardReader; use crate::io_support::bitrot::ShardReader;
@@ -75,7 +76,7 @@ use std::{
future::Future, future::Future,
pin::Pin, pin::Pin,
sync::{ sync::{
OnceLock, Arc, OnceLock,
atomic::{AtomicUsize, Ordering}, atomic::{AtomicUsize, Ordering},
}, },
task::{Context, Poll}, task::{Context, Poll},
@@ -94,6 +95,242 @@ fn metadata_distribution_key(bucket: &str, object: &str) -> String {
[bucket, object].join("/") [bucket, object].join("/")
} }
fn read_version_coalescing_enabled() -> bool {
let enabled = || {
rustfs_utils::get_env_opt_str(ENV_RUSTFS_GET_METADATA_READ_VERSION_COALESCE)
.is_some_and(|value| value.eq_ignore_ascii_case("auto") || value.eq_ignore_ascii_case("on"))
};
#[cfg(test)]
{
enabled()
}
#[cfg(not(test))]
{
static ENABLED: OnceLock<bool> = OnceLock::new();
*ENABLED.get_or_init(enabled)
}
}
fn read_version_coalescing_delay() -> Duration {
#[cfg(test)]
{
let micros = rustfs_utils::get_env_u64(
ENV_RUSTFS_GET_METADATA_READ_VERSION_COALESCE_DELAY_MICROS,
DEFAULT_GET_METADATA_READ_VERSION_COALESCE_DELAY_MICROS,
);
Duration::from_micros(micros)
}
#[cfg(not(test))]
{
static DELAY: OnceLock<Duration> = OnceLock::new();
*DELAY.get_or_init(|| {
Duration::from_micros(rustfs_utils::get_env_u64(
ENV_RUSTFS_GET_METADATA_READ_VERSION_COALESCE_DELAY_MICROS,
DEFAULT_GET_METADATA_READ_VERSION_COALESCE_DELAY_MICROS,
))
})
}
}
struct CoalescedReadVersionRequest {
item: BatchReadVersionItem,
tx: oneshot::Sender<disk::error::Result<FileInfo>>,
}
#[derive(Clone, Copy, Debug, Eq, Hash, PartialEq)]
struct ReadVersionCoalescerKey {
disk: usize,
incl_free_versions: bool,
read_data: bool,
healing: bool,
}
impl ReadVersionCoalescerKey {
fn new(disk: &DiskStore, opts: &ReadOptions) -> Self {
Self {
disk: Arc::as_ptr(disk) as usize,
incl_free_versions: opts.incl_free_versions,
read_data: opts.read_data,
healing: opts.healing,
}
}
}
#[derive(Default)]
struct ReadVersionCoalescer {
lanes: HashMap<ReadVersionCoalescerKey, Vec<CoalescedReadVersionRequest>>,
}
fn read_version_coalescer() -> &'static Mutex<ReadVersionCoalescer> {
static COALESCER: OnceLock<Mutex<ReadVersionCoalescer>> = OnceLock::new();
COALESCER.get_or_init(|| Mutex::new(ReadVersionCoalescer::default()))
}
fn record_read_version_coalescer_event(event: &'static str, item_count: usize) {
counter!(
METRIC_GET_METADATA_READ_VERSION_COALESCER_TOTAL,
"event" => event,
"item_count" => item_count.to_string()
)
.increment(1);
}
async fn read_version_via_coalescer(
disk: DiskStore,
org_bucket: &str,
bucket: &str,
object: &str,
version_id: &str,
opts: &ReadOptions,
allow_coalescing: bool,
) -> disk::error::Result<FileInfo> {
if !allow_coalescing || !read_version_coalescing_enabled() {
return disk.read_version(org_bucket, bucket, object, version_id, opts).await;
}
if !matches!(disk.as_ref(), Disk::Remote(_)) {
record_read_version_coalescer_event("bypass_non_remote", 1);
return disk.read_version(org_bucket, bucket, object, version_id, opts).await;
}
let (tx, rx) = oneshot::channel();
let item = BatchReadVersionItem {
org_volume: org_bucket.to_string(),
volume: bucket.to_string(),
path: object.to_string(),
version_id: version_id.to_string(),
};
let lane_key = ReadVersionCoalescerKey::new(&disk, opts);
let pending = {
let mut coalescer = read_version_coalescer().lock().await;
let lane = coalescer.lanes.entry(lane_key).or_default();
let schedule_delayed_flush = lane.is_empty();
lane.push(CoalescedReadVersionRequest { item, tx });
if lane.len() >= BATCH_READ_VERSION_MAX_ITEMS {
coalescer.lanes.remove(&lane_key)
} else if schedule_delayed_flush {
let disk = disk.clone();
let task_opts = *opts;
tokio::spawn(async move {
tokio::time::sleep(read_version_coalescing_delay()).await;
flush_read_version_coalescer_lane(lane_key, disk, task_opts).await;
});
None
} else {
None
}
};
if let Some(pending) = pending {
flush_read_version_coalescer_pending(lane_key, disk, *opts, pending).await;
}
rx.await
.unwrap_or_else(|_| Err(DiskError::other("coalesced read_version response channel closed")))
}
async fn flush_read_version_coalescer_lane(lane_key: ReadVersionCoalescerKey, disk: DiskStore, opts: ReadOptions) {
let pending = {
let mut coalescer = read_version_coalescer().lock().await;
coalescer.lanes.remove(&lane_key).unwrap_or_default()
};
flush_read_version_coalescer_pending(lane_key, disk, opts, pending).await;
}
async fn flush_read_version_coalescer_pending(
lane_key: ReadVersionCoalescerKey,
disk: DiskStore,
opts: ReadOptions,
pending: Vec<CoalescedReadVersionRequest>,
) {
if pending.is_empty() {
return;
}
#[cfg(test)]
{
let mut observed_paths = HashSet::new();
for request in &pending {
if observed_paths.insert(request.item.path.as_str()) {
disk_call_counters::record(&request.item.path, disk_call_counters::KIND_BATCH_READ_VERSION, lane_key.disk);
}
}
}
let mut senders = Vec::with_capacity(pending.len());
let mut items = Vec::with_capacity(pending.len());
for request in pending {
senders.push(request.tx);
items.push(request.item);
}
let expected_items = items.clone();
record_read_version_coalescer_event("attempted_batch", items.len());
let result =
match tokio::time::timeout(get_drive_metadata_timeout(), disk.batch_read_version(BatchReadVersionReq { items, opts }))
.await
{
Ok(result) => result,
Err(_) => Err(DiskError::Timeout),
};
match result {
Ok(responses) => {
let results = map_batch_read_version_responses(&expected_items, responses);
for (tx, result) in senders.into_iter().zip(results) {
let _ = tx.send(result);
}
}
Err(err) => {
let message = err.to_string();
for tx in senders {
let _ = tx.send(Err(DiskError::other(message.clone())));
}
}
}
}
fn map_batch_read_version_responses(
expected_items: &[BatchReadVersionItem],
responses: Vec<BatchReadVersionResp>,
) -> Vec<crate::disk::error::Result<FileInfo>> {
let mut results = (0..expected_items.len())
.map(|_| Err(DiskError::other("coalesced read_version response missing")))
.collect::<Vec<_>>();
let mut seen = vec![false; expected_items.len()];
for response in responses {
let Some(expected) = expected_items.get(response.index) else {
continue;
};
let Some(slot) = results.get_mut(response.index) else {
continue;
};
if seen[response.index] {
*slot = Err(DiskError::other("coalesced read_version response duplicate index"));
continue;
}
seen[response.index] = true;
if response.path != expected.path || response.version_id != expected.version_id {
*slot = Err(DiskError::other("coalesced read_version response identity mismatch"));
} else {
*slot = if response.success {
Ok(response.file_info)
} else {
Err(batch_read_version_response_error(response.error_code, response.error))
};
}
}
results
}
fn batch_read_version_response_error(error_code: u32, error: String) -> DiskError {
match DiskError::from_u32(error_code) {
Some(DiskError::Io(_)) | None => DiskError::other(error),
Some(error) => error,
}
}
pub(in crate::set_disk) fn bounded_metadata_fanout_order( pub(in crate::set_disk) fn bounded_metadata_fanout_order(
bucket: &str, bucket: &str,
object: &str, object: &str,
@@ -133,11 +370,15 @@ pub(in crate::set_disk) fn bounded_metadata_fanout_order(
order order
} }
use tokio::io::{AsyncRead, ReadBuf}; use tokio::io::{AsyncRead, ReadBuf};
use tokio::sync::RwLock; use tokio::sync::{Mutex, RwLock, oneshot};
use tokio::task::JoinSet; use tokio::task::JoinSet;
pub(in crate::set_disk) const EVENT_SET_DISK_READ: &str = "set_disk_read"; pub(in crate::set_disk) const EVENT_SET_DISK_READ: &str = "set_disk_read";
pub(in crate::set_disk) const ENV_RUSTFS_GET_DATA_BLOCKS_FIRST_READER_SETUP: &str = "RUSTFS_GET_DATA_BLOCKS_FIRST_READER_SETUP"; pub(in crate::set_disk) const ENV_RUSTFS_GET_DATA_BLOCKS_FIRST_READER_SETUP: &str = "RUSTFS_GET_DATA_BLOCKS_FIRST_READER_SETUP";
const ENV_RUSTFS_GET_METADATA_READ_VERSION_COALESCE: &str = "RUSTFS_GET_METADATA_READ_VERSION_COALESCE";
const ENV_RUSTFS_GET_METADATA_READ_VERSION_COALESCE_DELAY_MICROS: &str = "RUSTFS_GET_METADATA_READ_VERSION_COALESCE_DELAY_MICROS";
const DEFAULT_GET_METADATA_READ_VERSION_COALESCE_DELAY_MICROS: u64 = 200;
const METRIC_GET_METADATA_READ_VERSION_COALESCER_TOTAL: &str = "rustfs_get_metadata_read_version_coalescer_total";
pub(in crate::set_disk) const ENV_RUSTFS_PUT_RENAME_EARLY_ACK_ENABLE: &str = "RUSTFS_PUT_RENAME_EARLY_ACK_ENABLE"; pub(in crate::set_disk) const ENV_RUSTFS_PUT_RENAME_EARLY_ACK_ENABLE: &str = "RUSTFS_PUT_RENAME_EARLY_ACK_ENABLE";
/// Default reader-setup strategy for the GET read path (rustfs/backlog#1215, /// Default reader-setup strategy for the GET read path (rustfs/backlog#1215,
/// #1159, #923). /// #1159, #923).
@@ -2356,6 +2597,7 @@ impl SetDisks {
false, false,
true, true,
0, 0,
false,
) )
.await?; .await?;
Ok((ress, errors)) Ok((ress, errors))
@@ -2386,6 +2628,36 @@ impl SetDisks {
true, true,
caller_allows_early_stop, caller_allows_early_stop,
default_parity_count, default_parity_count,
false,
)
.await
}
#[allow(clippy::too_many_arguments)]
pub(in crate::set_disk) async fn read_all_fileinfo_observed_for_get_object(
disks: &[Option<DiskStore>],
org_bucket: &str,
bucket: &str,
object: &str,
version_id: &str,
read_data: bool,
incl_free_versions: bool,
caller_allows_early_stop: bool,
default_parity_count: usize,
) -> disk::error::Result<(Vec<FileInfo>, Vec<Option<DiskError>>, MetadataFanoutDiagnostics)> {
Self::read_all_fileinfo_inner(
disks,
org_bucket,
bucket,
object,
version_id,
read_data,
false,
incl_free_versions,
true,
caller_allows_early_stop,
default_parity_count,
true,
) )
.await .await
} }
@@ -2408,6 +2680,7 @@ impl SetDisks {
// subset would fail write quorum (backlog#872 regression). // subset would fail write quorum (backlog#872 regression).
caller_allows_early_stop: bool, caller_allows_early_stop: bool,
default_parity_count: usize, default_parity_count: usize,
allow_coalescing: bool,
) -> disk::error::Result<(Vec<FileInfo>, Vec<Option<DiskError>>, MetadataFanoutDiagnostics)> { ) -> disk::error::Result<(Vec<FileInfo>, Vec<Option<DiskError>>, MetadataFanoutDiagnostics)> {
let early_stop_enabled = let early_stop_enabled =
caller_allows_early_stop && observe && (is_get_metadata_early_stop_enabled() || is_version_early_stop_enabled()); caller_allows_early_stop && observe && (is_get_metadata_early_stop_enabled() || is_version_early_stop_enabled());
@@ -2424,6 +2697,7 @@ impl SetDisks {
healing, healing,
incl_free_versions, incl_free_versions,
default_parity_count, default_parity_count,
allow_coalescing,
) )
.await; .await;
} }
@@ -2446,6 +2720,7 @@ impl SetDisks {
healing, healing,
incl_free_versions, incl_free_versions,
observe, observe,
allow_coalescing,
) )
.await .await
} }
@@ -2461,6 +2736,7 @@ impl SetDisks {
healing: bool, healing: bool,
incl_free_versions: bool, incl_free_versions: bool,
observe: bool, observe: bool,
allow_coalescing: bool,
) -> disk::error::Result<(Vec<FileInfo>, Vec<Option<DiskError>>, MetadataFanoutDiagnostics)> { ) -> disk::error::Result<(Vec<FileInfo>, Vec<Option<DiskError>>, MetadataFanoutDiagnostics)> {
let fanout_start = observe.then(Instant::now); let fanout_start = observe.then(Instant::now);
let mut ress = Vec::with_capacity(disks.len()); let mut ress = Vec::with_capacity(disks.len());
@@ -2492,7 +2768,7 @@ impl SetDisks {
if let Some(delay) = slowtail_fault.as_ref().and_then(|fault| fault.delay_for_disk(disk_index)) { if let Some(delay) = slowtail_fault.as_ref().and_then(|fault| fault.delay_for_disk(disk_index)) {
tokio::time::sleep(delay).await; tokio::time::sleep(delay).await;
} }
disk.read_version(&org_bucket, &bucket, &object, &version_id, &task_opts) read_version_via_coalescer(disk, &org_bucket, &bucket, &object, &version_id, &task_opts, allow_coalescing)
.await .await
} else { } else {
Err(DiskError::DiskNotFound) Err(DiskError::DiskNotFound)
@@ -2559,6 +2835,7 @@ impl SetDisks {
healing: bool, healing: bool,
incl_free_versions: bool, incl_free_versions: bool,
default_parity_count: usize, default_parity_count: usize,
allow_coalescing: bool,
) -> disk::error::Result<(Vec<FileInfo>, Vec<Option<DiskError>>, MetadataFanoutDiagnostics)> { ) -> disk::error::Result<(Vec<FileInfo>, Vec<Option<DiskError>>, MetadataFanoutDiagnostics)> {
let fanout_start = Instant::now(); let fanout_start = Instant::now();
let mut ress = vec![FileInfo::default(); disks.len()]; let mut ress = vec![FileInfo::default(); disks.len()];
@@ -2607,7 +2884,7 @@ impl SetDisks {
if let Some(delay) = slowtail_fault.as_ref().and_then(|fault| fault.delay_for_disk(index)) { if let Some(delay) = slowtail_fault.as_ref().and_then(|fault| fault.delay_for_disk(index)) {
tokio::time::sleep(delay).await; tokio::time::sleep(delay).await;
} }
disk.read_version(&org_bucket, &bucket, &object, &version_id, &task_opts) read_version_via_coalescer(disk, &org_bucket, &bucket, &object, &version_id, &task_opts, allow_coalescing)
.await .await
} else { } else {
Err(DiskError::DiskNotFound) Err(DiskError::DiskNotFound)
@@ -5737,6 +6014,7 @@ pub(crate) mod disk_call_counters {
/// Kind label for the per-disk `read_version` metadata RPC. /// Kind label for the per-disk `read_version` metadata RPC.
pub const KIND_READ_VERSION: &str = "read_version"; pub const KIND_READ_VERSION: &str = "read_version";
pub const KIND_BATCH_READ_VERSION: &str = "batch_read_version";
/// Registry key: (object, kind, disk_index). /// Registry key: (object, kind, disk_index).
type CountKey = (String, String, usize); type CountKey = (String, String, usize);
@@ -6460,6 +6738,286 @@ mod tests {
drop(dirs); drop(dirs);
} }
#[tokio::test(flavor = "multi_thread", worker_threads = 4)]
async fn metadata_read_version_coalescer_bypasses_local_disks() {
const DISKS: usize = 4;
let bucket = "coalesced-read-version-local-bypass-bucket";
let object_a = "coalesced-local-object-a";
let object_b = "coalesced-local-object-b";
let (dirs, disks) = call_counter_local_disks(bucket, DISKS).await;
install_metadata_fanout_fileinfo(&disks, bucket, object_a, None).await;
install_metadata_fanout_fileinfo(&disks, bucket, object_b, None).await;
temp_env::async_with_vars(
[
(ENV_RUSTFS_GET_METADATA_READ_VERSION_COALESCE, Some("auto")),
(ENV_RUSTFS_GET_METADATA_READ_VERSION_COALESCE_DELAY_MICROS, Some("5000")),
],
async {
let calls = disk_call_counters::observe(object_a);
let disks_a = disks.clone();
let disks_b = disks.clone();
let read_a = tokio::spawn(async move {
SetDisks::read_all_fileinfo_observed_for_get_object(
&disks_a, "", bucket, object_a, "", false, false, false, 2,
)
.await
.map(|(file_infos, errors, _)| (file_infos, errors))
});
tokio::task::yield_now().await;
let read_b = tokio::spawn(async move {
SetDisks::read_all_fileinfo_observed_for_get_object(
&disks_b, "", bucket, object_b, "", false, false, false, 2,
)
.await
.map(|(file_infos, errors, _)| (file_infos, errors))
});
let (metadata_a, errs_a) = read_a
.await
.expect("first read task should not panic")
.expect("first coalesced read should resolve");
let (metadata_b, errs_b) = read_b
.await
.expect("second read task should not panic")
.expect("second coalesced read should resolve");
assert_eq!(metadata_a.iter().filter(|fi| fi.name == object_a).count(), DISKS);
assert_eq!(metadata_b.iter().filter(|fi| fi.name == object_b).count(), DISKS);
assert!(errs_a.iter().all(Option::is_none));
assert!(errs_b.iter().all(Option::is_none));
assert_eq!(
calls.total(disk_call_counters::KIND_READ_VERSION),
DISKS as u64,
"local disks still execute the ordinary per-disk read_version path"
);
assert_eq!(
calls.total(disk_call_counters::KIND_BATCH_READ_VERSION),
0,
"GET coalescing targets internode RPC count only and must not batch local disk reads"
);
},
)
.await;
drop(dirs);
}
#[tokio::test]
async fn metadata_read_version_coalescer_requires_get_object_intent() {
const DISKS: usize = 4;
let bucket = "coalesced-read-version-default-bypass-bucket";
let object = "default-bypass-object";
let (dirs, disks) = call_counter_local_disks(bucket, DISKS).await;
install_metadata_fanout_fileinfo(&disks, bucket, object, None).await;
temp_env::async_with_vars([(ENV_RUSTFS_GET_METADATA_READ_VERSION_COALESCE, Some("auto"))], async {
let calls = disk_call_counters::observe(object);
let (metadata, errs) = SetDisks::read_all_fileinfo(&disks, "", bucket, object, "", false, false, false)
.await
.expect("default metadata read should resolve");
assert_eq!(metadata.iter().filter(|fi| fi.name == object).count(), DISKS);
assert!(errs.iter().all(Option::is_none));
assert_eq!(calls.total(disk_call_counters::KIND_READ_VERSION), DISKS as u64);
assert_eq!(
calls.total(disk_call_counters::KIND_BATCH_READ_VERSION),
0,
"non-GET metadata paths must bypass coalescer even when the env gate is enabled"
);
})
.await;
drop(dirs);
}
#[test]
fn batch_read_version_response_mapping_preserves_index_and_errors() {
let expected_items = vec![
BatchReadVersionItem {
org_volume: String::new(),
volume: "bucket".to_string(),
path: "object-a".to_string(),
version_id: "v-a".to_string(),
},
BatchReadVersionItem {
org_volume: String::new(),
volume: "bucket".to_string(),
path: "object-b".to_string(),
version_id: "v-b".to_string(),
},
BatchReadVersionItem {
org_volume: String::new(),
volume: "bucket".to_string(),
path: "object-c".to_string(),
version_id: "v-c".to_string(),
},
];
let ok_file_info = FileInfo {
name: "object-a".to_string(),
..Default::default()
};
let responses = vec![
BatchReadVersionResp {
index: 2,
path: "object-c".to_string(),
version_id: "v-c".to_string(),
success: false,
file_info: FileInfo::default(),
error: "disk read failed".to_string(),
error_code: 0,
},
BatchReadVersionResp {
index: 0,
path: "object-a".to_string(),
version_id: "v-a".to_string(),
success: true,
file_info: ok_file_info,
error: String::new(),
error_code: 0,
},
];
let mut results = map_batch_read_version_responses(&expected_items, responses).into_iter();
let first = results
.next()
.expect("slot 0 should exist")
.expect("slot 0 should map the success response by index");
assert_eq!(first.name, "object-a");
let missing = results
.next()
.expect("slot 1 should exist")
.expect_err("slot 1 should stay missing");
assert!(
missing.to_string().contains("response missing"),
"unexpected missing response error: {missing}"
);
let failed = results
.next()
.expect("slot 2 should exist")
.expect_err("slot 2 should map the response error");
assert!(failed.to_string().contains("disk read failed"), "unexpected per-item error: {failed}");
assert!(results.next().is_none());
}
#[test]
fn batch_read_version_response_mapping_preserves_typed_not_found_errors() {
let expected_items = vec![
BatchReadVersionItem {
org_volume: String::new(),
volume: "bucket".to_string(),
path: "object-a".to_string(),
version_id: "v-a".to_string(),
},
BatchReadVersionItem {
org_volume: String::new(),
volume: "bucket".to_string(),
path: "object-b".to_string(),
version_id: "v-b".to_string(),
},
];
let results = map_batch_read_version_responses(
&expected_items,
vec![
BatchReadVersionResp {
index: 0,
path: "object-a".to_string(),
version_id: "v-a".to_string(),
success: false,
file_info: FileInfo::default(),
error: DiskError::FileNotFound.to_string(),
error_code: DiskError::FileNotFound.to_u32(),
},
BatchReadVersionResp {
index: 1,
path: "object-b".to_string(),
version_id: "v-b".to_string(),
success: false,
file_info: FileInfo::default(),
error: DiskError::FileVersionNotFound.to_string(),
error_code: DiskError::FileVersionNotFound.to_u32(),
},
],
);
assert!(matches!(results.first().expect("slot 0 should exist"), Err(DiskError::FileNotFound)));
assert!(matches!(
results.get(1).expect("slot 1 should exist"),
Err(DiskError::FileVersionNotFound)
));
}
#[test]
fn batch_read_version_response_mapping_rejects_identity_mismatch_and_duplicate_index() {
let expected_items = vec![BatchReadVersionItem {
org_volume: String::new(),
volume: "bucket".to_string(),
path: "object-a".to_string(),
version_id: "v-a".to_string(),
}];
let mismatched = map_batch_read_version_responses(
&expected_items,
vec![BatchReadVersionResp {
index: 0,
path: "object-b".to_string(),
version_id: "v-a".to_string(),
success: true,
file_info: FileInfo {
name: "object-b".to_string(),
..Default::default()
},
error: String::new(),
error_code: 0,
}],
)
.pop()
.expect("slot 0 should exist")
.expect_err("identity mismatch should fail closed");
assert!(
mismatched.to_string().contains("identity mismatch"),
"unexpected mismatch error: {mismatched}"
);
let duplicate = map_batch_read_version_responses(
&expected_items,
vec![
BatchReadVersionResp {
index: 0,
path: "object-a".to_string(),
version_id: "v-a".to_string(),
success: true,
file_info: FileInfo {
name: "object-a".to_string(),
..Default::default()
},
error: String::new(),
error_code: 0,
},
BatchReadVersionResp {
index: 0,
path: "object-a".to_string(),
version_id: "v-a".to_string(),
success: true,
file_info: FileInfo {
name: "object-a".to_string(),
..Default::default()
},
error: String::new(),
error_code: 0,
},
],
)
.pop()
.expect("slot 0 should exist")
.expect_err("duplicate response index should fail closed");
assert!(
duplicate.to_string().contains("duplicate index"),
"unexpected duplicate error: {duplicate}"
);
}
/// Isolation guard: unobserved objects record nothing (so parallel tests do /// Isolation guard: unobserved objects record nothing (so parallel tests do
/// not inflate one another), and a scope clears its own counts on drop. /// not inflate one another), and a scope clears its own counts on drop.
#[tokio::test] #[tokio::test]
+1 -1
View File
@@ -1294,7 +1294,7 @@ impl crate::storage_api_contracts::object::ObjectIO for SetDisks {
(prepared.snapshot, prepared.object_info) (prepared.snapshot, prepared.object_info)
} else { } else {
match self match self
.get_object_fileinfo( .get_object_fileinfo_for_get_object_reader(
bucket, bucket,
object, object,
opts, opts,
+66 -14
View File
@@ -259,10 +259,33 @@ impl SetDisks {
read_data: bool, read_data: bool,
caller_allows_early_stop: bool, caller_allows_early_stop: bool,
) -> Result<GetObjectFileInfo> { ) -> Result<GetObjectFileInfo> {
self.get_object_fileinfo_gated(bucket, object, opts, read_data, caller_allows_early_stop) self.get_object_fileinfo_gated_inner(bucket, object, opts, read_data, caller_allows_early_stop, false)
.await .await
} }
#[tracing::instrument(level = "debug", skip(self))]
#[hotpath::measure(impl_type = "SetDisks")]
pub(super) async fn get_object_fileinfo_for_get_object_reader(
&self,
bucket: &str,
object: &str,
opts: &ObjectOptions,
read_data: bool,
caller_allows_early_stop: bool,
) -> Result<GetObjectFileInfo> {
let allow_read_version_coalescing = !crate::bucket::utils::is_meta_bucketname(bucket)
&& crate::runtime::global::get_metadata_read_version_coalescing_service_ready();
self.get_object_fileinfo_gated_inner(
bucket,
object,
opts,
read_data,
caller_allows_early_stop,
allow_read_version_coalescing,
)
.await
}
/// Like `get_object_fileinfo`, but `allow_early_stop=false` forces the full /// Like `get_object_fileinfo`, but `allow_early_stop=false` forces the full
/// quorum fanout. Read-before-write callers (object tagging) must use this: /// quorum fanout. Read-before-write callers (object tagging) must use this:
/// the returned online-disk set is the write target, and the early-stop /// the returned online-disk set is the write target, and the early-stop
@@ -275,6 +298,20 @@ impl SetDisks {
opts: &ObjectOptions, opts: &ObjectOptions,
read_data: bool, read_data: bool,
allow_early_stop: bool, allow_early_stop: bool,
) -> Result<GetObjectFileInfo> {
self.get_object_fileinfo_gated_inner(bucket, object, opts, read_data, allow_early_stop, false)
.await
}
#[allow(clippy::too_many_arguments)]
async fn get_object_fileinfo_gated_inner(
&self,
bucket: &str,
object: &str,
opts: &ObjectOptions,
read_data: bool,
allow_early_stop: bool,
allow_read_version_coalescing: bool,
) -> Result<GetObjectFileInfo> { ) -> Result<GetObjectFileInfo> {
let vid = opts.version_id.clone().unwrap_or_default(); let vid = opts.version_id.clone().unwrap_or_default();
let stage_metrics_enabled = rustfs_io_metrics::get_stage_metrics_enabled(); let stage_metrics_enabled = rustfs_io_metrics::get_stage_metrics_enabled();
@@ -337,19 +374,34 @@ impl SetDisks {
// read_all_fileinfo_observed (see read_all_fileinfo_early_stop in // read_all_fileinfo_observed (see read_all_fileinfo_early_stop in
// core/io_primitives.rs); unsafe requests and callers that opt out // core/io_primitives.rs); unsafe requests and callers that opt out
// (allow_early_stop=false) fall back to full-wait. // (allow_early_stop=false) fall back to full-wait.
let (mut parts_metadata, errs, metadata_fanout_diagnostics) = Self::read_all_fileinfo_observed( let (mut parts_metadata, errs, metadata_fanout_diagnostics) = if allow_read_version_coalescing {
&disks, Self::read_all_fileinfo_observed_for_get_object(
"", &disks,
bucket, "",
object, bucket,
vid.as_str(), object,
read_data, vid.as_str(),
false, read_data,
opts.incl_free_versions, opts.incl_free_versions,
allow_early_stop, allow_early_stop,
self.default_parity_count, self.default_parity_count,
) )
.await?; .await?
} else {
Self::read_all_fileinfo_observed(
&disks,
"",
bucket,
object,
vid.as_str(),
read_data,
false,
opts.incl_free_versions,
allow_early_stop,
self.default_parity_count,
)
.await?
};
let metadata_metrics_path = if crate::bucket::utils::is_meta_bucketname(bucket) { let metadata_metrics_path = if crate::bucket::utils::is_meta_bucketname(bucket) {
GET_OBJECT_PATH_INTERNAL_META GET_OBJECT_PATH_INTERNAL_META
} else { } else {
+386 -1
View File
@@ -859,6 +859,7 @@ fn lifecycle_delete_all_test_failure(phase: crate::object_api::LifecycleDeleteAl
#[cfg(test)] #[cfg(test)]
mod tests { mod tests {
use super::*; use super::*;
use crate::bucket::replication::{ReplicationStatusType, VersionPurgeStatusType};
use crate::config::storageclass::{CLASS_RRS, CLASS_STANDARD, lookup_config_for_pools_without_env}; use crate::config::storageclass::{CLASS_RRS, CLASS_STANDARD, lookup_config_for_pools_without_env};
use crate::disk::error::DiskError; use crate::disk::error::DiskError;
use crate::layout::endpoint::Endpoint; use crate::layout::endpoint::Endpoint;
@@ -1423,6 +1424,14 @@ mod tests {
} }
} }
fn object_info_with_identity(unix_ts: i64, delete_marker: bool, version_id: Uuid, etag: Option<String>) -> ObjectInfo {
ObjectInfo {
version_id: Some(version_id),
etag,
..object_info_with_mod_time(unix_ts, delete_marker)
}
}
#[test] #[test]
fn resolve_latest_object_info_candidates_returns_latest_delete_marker() { fn resolve_latest_object_info_candidates_returns_latest_delete_marker() {
let candidates = vec![ let candidates = vec![
@@ -1446,7 +1455,7 @@ mod tests {
} }
#[test] #[test]
fn resolve_latest_object_info_candidates_prefers_higher_pool_idx_on_equal_mod_time() { fn resolve_latest_object_info_candidates_prefers_higher_pool_idx_on_equal_mod_time_for_equivalent_candidates() {
let candidates = vec![ let candidates = vec![
LatestObjectInfoCandidate { LatestObjectInfoCandidate {
info: Some(object_info_with_mod_time(10, false)), info: Some(object_info_with_mod_time(10, false)),
@@ -1466,6 +1475,382 @@ mod tests {
assert_eq!(idx, 1); assert_eq!(idx, 1);
} }
#[test]
fn resolve_latest_object_info_candidates_keeps_index_fallback_for_fully_equivalent_identities() {
let candidates = vec![
LatestObjectInfoCandidate {
info: Some(object_info_with_identity(10, false, Uuid::from_u128(1), Some("etag-a".to_string()))),
idx: 2,
err: None,
},
LatestObjectInfoCandidate {
info: Some(object_info_with_identity(10, false, Uuid::from_u128(1), Some("etag-a".to_string()))),
idx: 7,
err: None,
},
];
let (info, idx) = resolve_latest_object_info_candidates(candidates, "bucket", "object", &ObjectOptions::default())
.expect("equivalent replicas must resolve deterministically");
assert_eq!(idx, 7);
assert_eq!(info.version_id, Some(Uuid::from_u128(1)));
}
#[test]
fn resolve_latest_object_info_candidates_rejects_equal_time_version_id_conflict() {
let candidates = vec![
LatestObjectInfoCandidate {
info: Some(object_info_with_identity(10, false, Uuid::from_u128(1), Some("etag-a".to_string()))),
idx: 0,
err: None,
},
LatestObjectInfoCandidate {
info: Some(object_info_with_identity(10, false, Uuid::from_u128(2), Some("etag-a".to_string()))),
idx: 1,
err: None,
},
];
let err = resolve_latest_object_info_candidates(candidates, "bucket", "object", &ObjectOptions::default())
.expect_err("divergent version ids must not silently resolve to the higher pool index");
assert_eq!(err, Error::ErasureReadQuorum);
}
#[test]
fn resolve_latest_object_info_candidates_rejects_equal_time_etag_conflict() {
let candidates = vec![
LatestObjectInfoCandidate {
info: Some(object_info_with_identity(10, false, Uuid::from_u128(1), Some("etag-old".to_string()))),
idx: 0,
err: None,
},
LatestObjectInfoCandidate {
info: Some(object_info_with_identity(10, false, Uuid::from_u128(1), Some("etag-new".to_string()))),
idx: 1,
err: None,
},
];
let err = resolve_latest_object_info_candidates(candidates, "bucket", "object", &ObjectOptions::default())
.expect_err("divergent etags must not silently resolve to the higher pool index");
assert_eq!(err, Error::ErasureReadQuorum);
}
#[test]
fn resolve_latest_object_info_candidates_rejects_equal_time_delete_marker_conflict() {
let candidates = vec![
LatestObjectInfoCandidate {
info: Some(object_info_with_identity(10, false, Uuid::from_u128(1), None)),
idx: 0,
err: None,
},
LatestObjectInfoCandidate {
info: Some(object_info_with_identity(10, true, Uuid::from_u128(1), Some("etag-a".to_string()))),
idx: 1,
err: None,
},
];
let err = resolve_latest_object_info_candidates(candidates, "bucket", "object", &ObjectOptions::default())
.expect_err("a delete marker tied with a live version must not be masked by the pool index");
assert_eq!(err, Error::ErasureReadQuorum);
}
fn assert_equal_time_identity_conflict(left: ObjectInfo, right: ObjectInfo) {
let err = resolve_latest_object_info_candidates(
vec![
LatestObjectInfoCandidate {
info: Some(left),
idx: 0,
err: None,
},
LatestObjectInfoCandidate {
info: Some(right),
idx: 1,
err: None,
},
],
"bucket",
"object",
&ObjectOptions::default(),
)
.expect_err("equal-time identity divergence must fail closed");
assert_eq!(err, Error::ErasureReadQuorum);
}
#[test]
fn resolve_latest_object_info_candidates_rejects_equal_time_payload_identity_conflicts() {
let base = object_info_with_identity(10, false, Uuid::from_u128(1), Some("etag-a".to_string()));
let mut data_dir = base.clone();
data_dir.data_dir = Some(Uuid::from_u128(2));
assert_equal_time_identity_conflict(base.clone(), data_dir);
let mut size = base.clone();
size.size = 1;
assert_equal_time_identity_conflict(base.clone(), size);
let mut actual_size = base.clone();
actual_size.actual_size = 1;
assert_equal_time_identity_conflict(base.clone(), actual_size);
let mut checksum = base.clone();
checksum.checksum = Some(bytes::Bytes::from_static(b"checksum"));
assert_equal_time_identity_conflict(base.clone(), checksum);
let mut parts = base.clone();
parts.parts = std::sync::Arc::new(vec![rustfs_filemeta::ObjectPartInfo {
etag: "part-etag".to_string(),
number: 1,
size: 1,
..Default::default()
}]);
assert_equal_time_identity_conflict(base.clone(), parts);
let mut transition = base;
transition.transitioned_object.tier = "tier-a".to_string();
assert_equal_time_identity_conflict(
object_info_with_identity(10, false, Uuid::from_u128(1), Some("etag-a".to_string())),
transition,
);
}
#[test]
fn resolve_latest_object_info_candidates_accepts_internal_metadata_aliases() {
let base = object_info_with_identity(10, false, Uuid::from_u128(1), Some("etag-a".to_string()));
let mut rustfs_alias = base.clone();
rustfs_alias.user_defined = std::sync::Arc::new(std::collections::HashMap::from([(
"x-rustfs-internal-compression".to_string(),
"zstd".to_string(),
)]));
let mut minio_alias = base.clone();
minio_alias.user_defined = std::sync::Arc::new(std::collections::HashMap::from([(
"X-MINIO-INTERNAL-COMPRESSION".to_string(),
"zstd".to_string(),
)]));
let (_, idx) = resolve_latest_object_info_candidates(
vec![
LatestObjectInfoCandidate {
info: Some(rustfs_alias),
idx: 0,
err: None,
},
LatestObjectInfoCandidate {
info: Some(minio_alias),
idx: 1,
err: None,
},
],
"bucket",
"object",
&ObjectOptions::default(),
)
.expect("same-value internal aliases should resolve");
assert_eq!(idx, 1);
let mut dual_alias = base.clone();
dual_alias.user_defined = std::sync::Arc::new(std::collections::HashMap::from([
("x-rustfs-internal-compression".to_string(), "zstd".to_string()),
("x-minio-internal-compression".to_string(), "zstd".to_string()),
]));
let mut single_alias = base;
single_alias.user_defined = std::sync::Arc::new(std::collections::HashMap::from([(
"x-rustfs-internal-compression".to_string(),
"zstd".to_string(),
)]));
let (_, idx) = resolve_latest_object_info_candidates(
vec![
LatestObjectInfoCandidate {
info: Some(dual_alias),
idx: 0,
err: None,
},
LatestObjectInfoCandidate {
info: Some(single_alias),
idx: 1,
err: None,
},
],
"bucket",
"object",
&ObjectOptions::default(),
)
.expect("dual-key and single-key internal metadata should resolve");
assert_eq!(idx, 1);
}
#[test]
fn resolve_latest_object_info_candidates_rejects_different_internal_metadata_alias_values() {
let base = object_info_with_identity(10, false, Uuid::from_u128(1), Some("etag-a".to_string()));
let mut rustfs_alias = base.clone();
rustfs_alias.user_defined = std::sync::Arc::new(std::collections::HashMap::from([(
"x-rustfs-internal-compression".to_string(),
"zstd".to_string(),
)]));
let mut minio_alias = base;
minio_alias.user_defined = std::sync::Arc::new(std::collections::HashMap::from([(
"x-minio-internal-compression".to_string(),
"snappy".to_string(),
)]));
assert_equal_time_identity_conflict(rustfs_alias, minio_alias);
}
#[test]
fn resolve_latest_object_info_candidates_preserves_dynamic_internal_metadata_identity_case() {
for suffix_prefix in ["replication-reset-", "replication-delete-marker-version-"] {
let base = object_info_with_identity(10, false, Uuid::from_u128(1), Some("etag-a".to_string()));
let mut rustfs_alias = base.clone();
rustfs_alias.user_defined = std::sync::Arc::new(std::collections::HashMap::from([(
format!(
"X-RUSTFS-INTERNAL-{}{suffix}",
suffix_prefix.to_uppercase(),
suffix = "arn:aws:s3:::Bucket"
),
"value".to_string(),
)]));
let mut minio_alias = base.clone();
minio_alias.user_defined = std::sync::Arc::new(std::collections::HashMap::from([(
format!("x-minio-internal-{suffix_prefix}arn:aws:s3:::Bucket"),
"value".to_string(),
)]));
let (_, idx) = resolve_latest_object_info_candidates(
vec![
LatestObjectInfoCandidate {
info: Some(rustfs_alias.clone()),
idx: 0,
err: None,
},
LatestObjectInfoCandidate {
info: Some(minio_alias),
idx: 1,
err: None,
},
],
"bucket",
"object",
&ObjectOptions::default(),
)
.expect("dynamic internal aliases with the same target should resolve");
assert_eq!(idx, 1);
let mut different_target_case = base;
different_target_case.user_defined = std::sync::Arc::new(std::collections::HashMap::from([(
format!("x-minio-internal-{suffix_prefix}arn:aws:s3:::bucket"),
"value".to_string(),
)]));
assert_equal_time_identity_conflict(rustfs_alias, different_target_case);
}
}
#[test]
fn resolve_latest_object_info_candidates_rejects_conflicting_internal_metadata_aliases_in_one_candidate() {
let base = object_info_with_identity(10, false, Uuid::from_u128(1), Some("etag-a".to_string()));
let mut first = base.clone();
first.user_defined = std::sync::Arc::new(std::collections::HashMap::from([
("x-rustfs-internal-compression".to_string(), "zstd".to_string()),
("x-minio-internal-compression".to_string(), "snappy".to_string()),
]));
let mut second = base;
second.user_defined = first.user_defined.clone();
assert_equal_time_identity_conflict(first, second);
}
#[test]
fn resolve_latest_object_info_candidates_rejects_replication_identity_conflict() {
let base = object_info_with_identity(10, false, Uuid::from_u128(1), Some("etag-a".to_string()));
let mut replication = base.clone();
replication.replication_status_internal = Some("PENDING".to_string());
replication.replication_status = ReplicationStatusType::Pending;
assert_equal_time_identity_conflict(base.clone(), replication);
let mut purge = base.clone();
purge.version_purge_status_internal = Some("PENDING".to_string());
purge.version_purge_status = VersionPurgeStatusType::Pending;
assert_equal_time_identity_conflict(base.clone(), purge);
let mut decision = base;
decision.replication_decision = "replicate".to_string();
assert_equal_time_identity_conflict(
object_info_with_identity(10, false, Uuid::from_u128(1), Some("etag-a".to_string())),
decision,
);
}
#[test]
fn resolve_latest_object_info_candidates_rejects_none_vs_unix_epoch_mod_time() {
let mut without_mod_time = object_info_with_identity(0, false, Uuid::from_u128(1), Some("etag-a".to_string()));
without_mod_time.mod_time = None;
let with_unix_epoch = object_info_with_identity(0, false, Uuid::from_u128(1), Some("etag-a".to_string()));
assert_equal_time_identity_conflict(without_mod_time, with_unix_epoch);
}
#[test]
fn resolve_latest_object_info_candidates_ignores_older_identity_conflicts() {
let latest = object_info_with_identity(20, false, Uuid::from_u128(1), Some("etag-latest".to_string()));
let mut older = object_info_with_identity(10, true, Uuid::from_u128(2), Some("etag-old".to_string()));
older.data_dir = Some(Uuid::from_u128(2));
let (info, idx) = resolve_latest_object_info_candidates(
vec![
LatestObjectInfoCandidate {
info: Some(latest),
idx: 0,
err: None,
},
LatestObjectInfoCandidate {
info: Some(older),
idx: 9,
err: None,
},
],
"bucket",
"object",
&ObjectOptions::default(),
)
.expect("older identity divergence must not affect the latest candidate");
assert_eq!(idx, 0);
assert_eq!(
info.mod_time,
Some(OffsetDateTime::from_unix_timestamp(20).expect("operation should succeed"))
);
}
#[test]
fn resolve_latest_object_info_candidates_ignores_not_found_pools_when_resolving() {
let candidates = vec![
LatestObjectInfoCandidate {
info: Some(object_info_with_identity(10, false, Uuid::from_u128(1), Some("etag-a".to_string()))),
idx: 0,
err: None,
},
LatestObjectInfoCandidate {
info: None,
idx: 1,
err: Some(Error::ObjectNotFound("bucket".to_string(), "object".to_string())),
},
];
let (info, idx) = resolve_latest_object_info_candidates(candidates, "bucket", "object", &ObjectOptions::default())
.expect("not-found pools must not block resolution of found candidates");
assert_eq!(idx, 0);
assert_eq!(info.version_id, Some(Uuid::from_u128(1)));
}
#[test] #[test]
fn resolve_latest_object_info_candidates_returns_non_not_found_error() { fn resolve_latest_object_info_candidates_returns_non_not_found_error() {
let err = resolve_latest_object_info_candidates( let err = resolve_latest_object_info_candidates(
+146 -21
View File
@@ -12,10 +12,14 @@
// See the License for the specific language governing permissions and // See the License for the specific language governing permissions and
// limitations under the License. // limitations under the License.
use std::cmp::Ordering; use std::collections::HashMap;
use crate::error::{Error, Result, StorageError, is_err_object_not_found, is_err_version_not_found}; use crate::error::{Error, Result, StorageError, is_err_object_not_found, is_err_version_not_found};
use crate::object_api::{ObjectInfo, ObjectOptions}; use crate::object_api::{ObjectInfo, ObjectOptions};
use rustfs_utils::http::metadata_compat::{
SUFFIX_REPLICATION_DELETE_MARKER_VERSION_ARN_PREFIX, SUFFIX_REPLICATION_RESET_ARN_PREFIX,
strip_internal_prefix_preserving_case,
};
use rustfs_utils::path::decode_dir_object; use rustfs_utils::path::decode_dir_object;
use time::OffsetDateTime; use time::OffsetDateTime;
@@ -137,37 +141,158 @@ pub(super) fn rebalance_disk_set_lookup_error(pool_idx: usize, set_idx: usize, p
)) ))
} }
fn latest_candidate_mod_time(candidate: &LatestObjectInfoCandidate) -> Option<OffsetDateTime> {
candidate
.info
.as_ref()
.map(|info| info.mod_time.unwrap_or(OffsetDateTime::UNIX_EPOCH))
}
fn same_transition_identity(left: &ObjectInfo, right: &ObjectInfo) -> bool {
left.transition_version_state == right.transition_version_state
&& left.transitioned_object.name == right.transitioned_object.name
&& left.transitioned_object.version_id == right.transitioned_object.version_id
&& left.transitioned_object.tier == right.transitioned_object.tier
&& left.transitioned_object.free_version == right.transitioned_object.free_version
&& left.transitioned_object.status == right.transitioned_object.status
}
#[derive(PartialEq, Eq)]
struct LatestUserDefinedIdentity {
internal: HashMap<String, String>,
other: HashMap<String, String>,
}
fn normalize_internal_identity_suffix(key: &str) -> Option<String> {
let suffix = strip_internal_prefix_preserving_case(key)?;
for dynamic_prefix in [
SUFFIX_REPLICATION_RESET_ARN_PREFIX,
SUFFIX_REPLICATION_DELETE_MARKER_VERSION_ARN_PREFIX,
] {
let prefix_len = dynamic_prefix.len();
if let (Some(prefix), Some(remainder)) = (suffix.get(..prefix_len), suffix.get(prefix_len..))
&& prefix.eq_ignore_ascii_case(dynamic_prefix)
{
return Some(format!("{dynamic_prefix}{remainder}"));
}
}
Some(suffix.to_lowercase())
}
fn normalize_user_defined_identity(user_defined: &HashMap<String, String>) -> Option<LatestUserDefinedIdentity> {
let mut identity = LatestUserDefinedIdentity {
internal: HashMap::with_capacity(user_defined.len()),
other: HashMap::with_capacity(user_defined.len()),
};
for (key, value) in user_defined {
if let Some(suffix) = normalize_internal_identity_suffix(key) {
if identity
.internal
.insert(suffix, value.clone())
.is_some_and(|previous| previous != *value)
{
return None;
}
} else {
identity.other.insert(key.clone(), value.clone());
}
}
Some(identity)
}
fn same_user_defined_identity(left: &ObjectInfo, right: &ObjectInfo) -> bool {
match (
normalize_user_defined_identity(&left.user_defined),
normalize_user_defined_identity(&right.user_defined),
) {
(Some(left), Some(right)) => left == right,
_ => false,
}
}
/// Pool-specific erasure geometry is intentionally excluded: `get_object_info`
/// returns each pool's own `data_blocks`/`parity_blocks`, so those values can
/// differ for the same object version while the selected winner still carries
/// the chosen pool's layout. `put_object_reader` is also intentionally
/// excluded because it is a transient request handle that `ObjectInfo::clone`
/// drops. Every other ObjectInfo field is part of the production-visible
/// identity and must agree before the pool index can provide a deterministic
/// tie-break.
fn same_latest_object_info_identity(left: &ObjectInfo, right: &ObjectInfo) -> bool {
left.bucket == right.bucket
&& left.name == right.name
&& left.storage_class == right.storage_class
&& left.mod_time == right.mod_time
&& left.size == right.size
&& left.actual_size == right.actual_size
&& left.is_dir == right.is_dir
&& same_user_defined_identity(left, right)
&& left.user_tags == right.user_tags
&& left.version_id == right.version_id
&& left.data_dir == right.data_dir
&& left.delete_marker == right.delete_marker
&& same_transition_identity(left, right)
&& left.restore_ongoing == right.restore_ongoing
&& left.restore_expires == right.restore_expires
&& left.parts == right.parts
&& left.is_latest == right.is_latest
&& left.content_type == right.content_type
&& left.content_encoding == right.content_encoding
&& left.expires == right.expires
&& left.num_versions == right.num_versions
&& left.successor_mod_time == right.successor_mod_time
&& left.etag == right.etag
&& left.inlined == right.inlined
&& left.metadata_only == right.metadata_only
&& left.version_only == right.version_only
&& left.replication_status_internal == right.replication_status_internal
&& left.replication_status == right.replication_status
&& left.version_purge_status_internal == right.version_purge_status_internal
&& left.version_purge_status == right.version_purge_status
&& left.replication_decision == right.replication_decision
&& left.checksum == right.checksum
}
pub(super) fn resolve_latest_object_info_candidates( pub(super) fn resolve_latest_object_info_candidates(
mut candidates: Vec<LatestObjectInfoCandidate>, candidates: Vec<LatestObjectInfoCandidate>,
bucket: &str, bucket: &str,
object: &str, object: &str,
opts: &ObjectOptions, opts: &ObjectOptions,
) -> Result<(ObjectInfo, usize)> { ) -> Result<(ObjectInfo, usize)> {
candidates.sort_by(|a, b| { let latest_mod_time = candidates.iter().filter_map(latest_candidate_mod_time).max();
let a_mod = if let Some(info) = &a.info {
info.mod_time.unwrap_or(OffsetDateTime::UNIX_EPOCH) if let Some(latest_mod_time) = latest_mod_time {
} else { let mut latest_candidates = candidates
OffsetDateTime::UNIX_EPOCH .into_iter()
.filter(|candidate| latest_candidate_mod_time(candidate) == Some(latest_mod_time))
.collect::<Vec<_>>();
latest_candidates.sort_by(|left, right| right.idx.cmp(&left.idx));
let Some(winner) = latest_candidates.first() else {
return Err(Error::ErasureReadQuorum);
};
let Some(winner_info) = winner.info.as_ref() else {
return Err(Error::ErasureReadQuorum);
}; };
let b_mod = if let Some(info) = &b.info { if latest_candidates.iter().skip(1).any(|candidate| {
info.mod_time.unwrap_or(OffsetDateTime::UNIX_EPOCH) candidate
} else { .info
OffsetDateTime::UNIX_EPOCH .as_ref()
}; .is_none_or(|info| !same_latest_object_info_identity(winner_info, info))
}) {
if a_mod == b_mod { return Err(Error::ErasureReadQuorum);
return if a.idx < b.idx { Ordering::Greater } else { Ordering::Less };
} }
b_mod.cmp(&a_mod) return Ok((winner_info.clone(), winner.idx));
}); }
for candidate in candidates { for candidate in candidates {
if let Some(info) = candidate.info {
return Ok((info, candidate.idx));
}
if let Some(err) = candidate.err if let Some(err) = candidate.err
&& !is_err_object_not_found(&err) && !is_err_object_not_found(&err)
&& !is_err_version_not_found(&err) && !is_err_version_not_found(&err)
@@ -54,6 +54,13 @@ pub const INTERNODE_STAGE_READ_VERSION_RESPONSE_JSON_ENCODE: &str = "read_versio
pub const INTERNODE_STAGE_READ_VERSION_RESPONSE_MSGPACK_ENCODE: &str = "read_version_response_msgpack_encode"; pub const INTERNODE_STAGE_READ_VERSION_RESPONSE_MSGPACK_ENCODE: &str = "read_version_response_msgpack_encode";
pub const INTERNODE_STAGE_READ_VERSION_RPC_ROUNDTRIP: &str = "read_version_rpc_roundtrip"; pub const INTERNODE_STAGE_READ_VERSION_RPC_ROUNDTRIP: &str = "read_version_rpc_roundtrip";
pub const INTERNODE_STAGE_READ_VERSION_RESPONSE_DECODE: &str = "read_version_response_decode"; pub const INTERNODE_STAGE_READ_VERSION_RESPONSE_DECODE: &str = "read_version_response_decode";
pub const INTERNODE_STAGE_BATCH_READ_VERSION_REQUEST_ENCODE: &str = "batch_read_version_request_encode";
pub const INTERNODE_STAGE_BATCH_READ_VERSION_REQUEST_DECODE: &str = "batch_read_version_request_decode";
pub const INTERNODE_STAGE_BATCH_READ_VERSION_DISK_READ: &str = "batch_read_version_disk_read";
pub const INTERNODE_STAGE_BATCH_READ_VERSION_RESPONSE_JSON_ENCODE: &str = "batch_read_version_response_json_encode";
pub const INTERNODE_STAGE_BATCH_READ_VERSION_RESPONSE_MSGPACK_ENCODE: &str = "batch_read_version_response_msgpack_encode";
pub const INTERNODE_STAGE_BATCH_READ_VERSION_RPC_ROUNDTRIP: &str = "batch_read_version_rpc_roundtrip";
pub const INTERNODE_STAGE_BATCH_READ_VERSION_RESPONSE_DECODE: &str = "batch_read_version_response_decode";
const OPERATION_LABEL: &str = "operation"; const OPERATION_LABEL: &str = "operation";
const BACKEND_LABEL: &str = "backend"; const BACKEND_LABEL: &str = "backend";
+40 -2
View File
@@ -195,7 +195,8 @@ pub struct BucketPolicyArgs<'a> {
#[derive(Serialize, Deserialize, Clone, Default, Debug)] #[derive(Serialize, Deserialize, Clone, Default, Debug)]
#[serde(deny_unknown_fields)] #[serde(deny_unknown_fields)]
pub struct BucketPolicy { pub struct BucketPolicy {
#[serde(default, rename = "Id", skip_serializing_if = "ID::is_empty")] // RUSTFS_COMPAT_TODO(rustfs-6339): accept bucket policies persisted with the legacy "ID" key. Remove after migration tooling rewrites every retained legacy bucket policy.
#[serde(default, rename = "Id", alias = "ID", skip_serializing_if = "ID::is_empty")]
pub id: ID, pub id: ID,
#[serde(rename = "Version")] #[serde(rename = "Version")]
pub version: String, pub version: String,
@@ -2786,7 +2787,7 @@ mod test {
let parsed: serde_json::Value = serde_json::from_str(&json).expect("Should parse"); let parsed: serde_json::Value = serde_json::from_str(&json).expect("Should parse");
// Verify empty fields are omitted // Verify empty fields are omitted
assert!(!parsed.as_object().unwrap().contains_key("ID"), "Empty ID should be omitted"); assert!(parsed.get("Id").is_none(), "Empty ID should be omitted");
let statement = &parsed["Statement"][0]; let statement = &parsed["Statement"][0];
assert!(!statement.as_object().unwrap().contains_key("Sid"), "Empty Sid should be omitted"); assert!(!statement.as_object().unwrap().contains_key("Sid"), "Empty Sid should be omitted");
@@ -2809,6 +2810,43 @@ mod test {
assert_eq!(statement["Principal"]["AWS"], "*"); assert_eq!(statement["Principal"]["AWS"], "*");
} }
#[test]
fn test_bucket_policy_deserializes_legacy_id() {
let legacy_policy = br#"{"ID":"","Version":"2012-10-17","Statement":[{"Sid":"","Effect":"Allow","Principal":{"AWS":["*"]},"Action":["s3:GetObject"],"NotAction":[],"Resource":["arn:aws:s3:::bucket/*"],"NotResource":[],"Condition":{}}]}"#;
let policy: BucketPolicy =
serde_json::from_slice(legacy_policy).expect("bucket policy with legacy ID should deserialize");
assert!(policy.id.is_empty());
policy.is_valid().expect("legacy bucket policy should remain valid");
let policy: BucketPolicy = serde_json::from_str(r#"{"ID":"legacy-policy","Version":"2012-10-17","Statement":[]}"#)
.expect("non-empty legacy ID should deserialize");
assert_eq!(policy.id.0, "legacy-policy");
let serialized = serde_json::to_value(&policy).expect("bucket policy should serialize");
assert_eq!(serialized["Id"], "legacy-policy");
assert!(serialized.get("ID").is_none(), "legacy ID spelling should not be serialized");
}
#[test]
fn test_bucket_policy_legacy_id_alias_remains_strict() {
let unknown_field = r#"{"Version":"2012-10-17","Statement":[],"Unexpected":true}"#;
let error =
serde_json::from_str::<BucketPolicy>(unknown_field).expect_err("unrelated unknown fields should remain rejected");
assert!(
error.to_string().contains("unknown field `Unexpected`"),
"unexpected deserialization error: {error}"
);
let duplicate_id = r#"{"Id":"current-policy","ID":"legacy-policy","Version":"2012-10-17","Statement":[]}"#;
let error = serde_json::from_str::<BucketPolicy>(duplicate_id)
.expect_err("canonical and legacy ID fields should not be accepted together");
assert!(
error.to_string().contains("duplicate field `Id`"),
"unexpected deserialization error: {error}"
);
}
#[test] #[test]
fn test_existing_object_tag_condition_helpers() { fn test_existing_object_tag_condition_helpers() {
let identity_policy = Policy::parse_config( let identity_policy = Policy::parse_config(
@@ -12,6 +12,7 @@ for later deletion.
## Open Items ## Open Items
- `rustfs-6339` legacy bucket policy ID casing: earlier RustFS releases persisted the top-level policy identifier as "ID", while current writes use the S3-compatible "Id" spelling. Readers accept both spellings so retained bucket metadata remains usable after upgrade. Remove the legacy alias after migration tooling has rewritten every retained bucket policy using "ID".
- `table-publication-fence-v1` table publication fencing: nodes that predate table and table-bucket publication fences can mutate live files while a new node is publishing a catalog pointer. New nodes retain exact object guards until the operator confirms that every serving node uses the new fences. Fleet confirmation also requires non-overlapping active warehouse prefixes and lifecycle workers that exclude table buckets. Remove the exact live-file fallback and the fleet-confirmation gate after the minimum supported RustFS release acquires table fences for registered-table mutations and table-bucket fences for unresolved-prefix mutations. - `table-publication-fence-v1` table publication fencing: nodes that predate table and table-bucket publication fences can mutate live files while a new node is publishing a catalog pointer. New nodes retain exact object guards until the operator confirms that every serving node uses the new fences. Fleet confirmation also requires non-overlapping active warehouse prefixes and lifecycle workers that exclude table buckets. Remove the exact live-file fallback and the fleet-confirmation gate after the minimum supported RustFS release acquires table fences for registered-table mutations and table-bucket fences for unresolved-prefix mutations.
- `table-catalog-strong-snapshot-v1` durable strong catalog snapshot compatibility: version 1 writes continue during mixed-version rollout until operators confirm that every serving node reads version 2, and version 1 table/view identifier collisions remain available only for cleanup. Remove version 1 writes and collision cleanup after the minimum supported RustFS release reads version 2 and every retained durable strong snapshot is collision-free and has been upgraded to version 2. - `table-catalog-strong-snapshot-v1` durable strong catalog snapshot compatibility: version 1 writes continue during mixed-version rollout until operators confirm that every serving node reads version 2, and version 1 table/view identifier collisions remain available only for cleanup. Remove version 1 writes and collision cleanup after the minimum supported RustFS release reads version 2 and every retained durable strong snapshot is collision-free and has been upgraded to version 2.
- `table-catalog-migration-fence-v1` durable strong migration fence compatibility: version 1 "PREPARING" fences did not distinguish a known-absent global strong snapshot from an unknown baseline, so retries read them but fail closed if the global snapshot is missing. Version 2 preserves the same JSON shape and records the pre-migration global snapshot ETag in the existing target_snapshot_etag field while the fence is "PREPARING". Remove version 1 reads after every supported direct-upgrade source writes version 2 fences and operators have completed or cancelled every older in-progress backing migration. - `table-catalog-migration-fence-v1` durable strong migration fence compatibility: version 1 "PREPARING" fences did not distinguish a known-absent global strong snapshot from an unknown baseline, so retries read them but fail closed if the global snapshot is missing. Version 2 preserves the same JSON shape and records the pre-migration global snapshot ETag in the existing target_snapshot_etag field while the fence is "PREPARING". Remove version 1 reads after every supported direct-upgrade source writes version 2 fences and operators have completed or cancelled every older in-progress backing migration.
+2 -2
View File
@@ -58,7 +58,7 @@
| heal_erasure_disk_rebuild_test | 4 | 🌙 | | heal_erasure_disk_rebuild_test | 4 | 🌙 |
| inline_fast_path_cluster_test | 16 | | | inline_fast_path_cluster_test | 16 | |
| internode_rpc_signature_e2e_test | 5 | | | internode_rpc_signature_e2e_test | 5 | |
| kms | 48 | | | kms | 46 | |
| leading_slash_key_test | 2 | ✅ | | leading_slash_key_test | 2 | ✅ |
| lifecycle_regression_test | 4 | | | lifecycle_regression_test | 4 | |
| list_buckets_auth_test | 1 | ✅ | | list_buckets_auth_test | 1 | ✅ |
@@ -99,4 +99,4 @@
| tls_hot_reload_test | 1 | ✅ | | tls_hot_reload_test | 1 | ✅ |
| version_id_regression_test | 10 | ✅ | | version_id_regression_test | 10 | ✅ |
**Total listed: 577 tests across 82 modules · PR smoke: 163 tests / 36 modules · merge/main full: 455 tests / 73 modules · nightly replication: 55 tests · nightly cluster faults: 28 tests / 7 modules · nightly protocols: 16 tests** · updated 2026-08-21. **Total listed: 575 tests across 82 modules · PR smoke: 163 tests / 36 modules · merge/main full: 453 tests / 73 modules · nightly replication: 55 tests · nightly cluster faults: 28 tests / 7 modules · nightly protocols: 16 tests** · updated 2026-08-23.
+5
View File
@@ -20,6 +20,7 @@ use crate::storage_api::server::readiness::contract::admin::StorageAdminApi;
use crate::storage_api::server::readiness::{Endpoint, EndpointServerPools, is_dist_erasure}; use crate::storage_api::server::readiness::{Endpoint, EndpointServerPools, is_dist_erasure};
#[cfg(test)] #[cfg(test)]
use crate::storage_api::server::readiness::{Endpoints, PoolEndpoints}; use crate::storage_api::server::readiness::{Endpoints, PoolEndpoints};
use crate::storage_api::startup::shutdown::mark_get_metadata_read_version_coalescing_service_ready;
use bytes::Bytes; use bytes::Bytes;
use http::HeaderValue; use http::HeaderValue;
use http::{Request as HttpRequest, Response, StatusCode}; use http::{Request as HttpRequest, Response, StatusCode};
@@ -212,6 +213,9 @@ where
if readiness_gate_blocks_path(path, &readiness) { if readiness_gate_blocks_path(path, &readiness) {
return Ok(service_not_ready_response(readiness.current_stage())); return Ok(service_not_ready_response(readiness.current_stage()));
} }
if !is_probe_path(path) && readiness.is_ready() {
mark_get_metadata_read_version_coalescing_service_ready();
}
let resp = inner.call(req).await?; let resp = inner.call(req).await?;
// System is ready, forward to the actual S3/RPC handlers // System is ready, forward to the actual S3/RPC handlers
// Transparently converts any response body into a BoxBody, and then Trace/Cors/Compression continues to work // Transparently converts any response body into a BoxBody, and then Trace/Cors/Compression continues to work
@@ -232,6 +236,7 @@ pub async fn publish_ready_when_runtime_ready(
collect_node_readiness, collect_node_readiness,
|dependency_readiness| { |dependency_readiness| {
readiness.mark_stage(rustfs_common::SystemStage::FullReady); readiness.mark_stage(rustfs_common::SystemStage::FullReady);
mark_get_metadata_read_version_coalescing_service_ready();
if let Some(state_manager) = state_manager { if let Some(state_manager) = state_manager {
state_manager.update(ServiceState::Ready); state_manager.update(ServiceState::Ready);
} }
+99 -15
View File
@@ -23,10 +23,12 @@ use bytes::Bytes;
use rustfs_filemeta::FileInfo; use rustfs_filemeta::FileInfo;
use rustfs_io_metrics::internode_metrics::{ use rustfs_io_metrics::internode_metrics::{
INTERNODE_MSGPACK_CODEC_JSON, INTERNODE_MSGPACK_CODEC_MSGPACK, INTERNODE_MSGPACK_DIRECTION_REQUEST, INTERNODE_MSGPACK_CODEC_JSON, INTERNODE_MSGPACK_CODEC_MSGPACK, INTERNODE_MSGPACK_DIRECTION_REQUEST,
INTERNODE_OPERATION_GRPC_READ_ALL, INTERNODE_OPERATION_GRPC_READ_VERSION, INTERNODE_OPERATION_GRPC_WRITE_ALL, INTERNODE_OPERATION_GRPC_BATCH_READ_VERSION, INTERNODE_OPERATION_GRPC_READ_ALL, INTERNODE_OPERATION_GRPC_READ_VERSION,
INTERNODE_STAGE_READ_VERSION_DISK_READ, INTERNODE_STAGE_READ_VERSION_REQUEST_DECODE, INTERNODE_OPERATION_GRPC_WRITE_ALL, INTERNODE_STAGE_BATCH_READ_VERSION_DISK_READ,
INTERNODE_STAGE_READ_VERSION_RESPONSE_JSON_ENCODE, INTERNODE_STAGE_READ_VERSION_RESPONSE_MSGPACK_ENCODE, INTERNODE_STAGE_BATCH_READ_VERSION_REQUEST_DECODE, INTERNODE_STAGE_BATCH_READ_VERSION_RESPONSE_JSON_ENCODE,
INTERNODE_TRANSPORT_BACKEND_GRPC, global_internode_metrics, INTERNODE_STAGE_BATCH_READ_VERSION_RESPONSE_MSGPACK_ENCODE, INTERNODE_STAGE_READ_VERSION_DISK_READ,
INTERNODE_STAGE_READ_VERSION_REQUEST_DECODE, INTERNODE_STAGE_READ_VERSION_RESPONSE_JSON_ENCODE,
INTERNODE_STAGE_READ_VERSION_RESPONSE_MSGPACK_ENCODE, INTERNODE_TRANSPORT_BACKEND_GRPC, global_internode_metrics,
}; };
use rustfs_protos::proto_gen::node_service::*; use rustfs_protos::proto_gen::node_service::*;
use serde::de::DeserializeOwned; use serde::de::DeserializeOwned;
@@ -242,24 +244,42 @@ fn record_read_version_stage(stage: &'static str, started_at: Option<Instant>) {
} }
} }
fn record_batch_read_version_stage(stage: &'static str, started_at: Option<Instant>) {
if let Some(started_at) = started_at {
global_internode_metrics().record_stage_duration_for_operation_and_backend(
INTERNODE_OPERATION_GRPC_BATCH_READ_VERSION,
INTERNODE_TRANSPORT_BACKEND_GRPC,
stage,
started_at.elapsed(),
);
}
}
fn encode_batch_read_version_response_payloads( fn encode_batch_read_version_response_payloads(
batch_read_version_resps: &[BatchReadVersionResp], batch_read_version_resps: &[BatchReadVersionResp],
request_decoded_from_msgpack: bool, request_decoded_from_msgpack: bool,
) -> std::result::Result<(Vec<String>, Vec<Bytes>), DiskError> { ) -> std::result::Result<(Vec<String>, Vec<Bytes>), DiskError> {
let attribution_enabled = rustfs_io_metrics::get_stage_metrics_enabled();
let mut batch_read_version_resps_json = Vec::with_capacity(batch_read_version_resps.len()); let mut batch_read_version_resps_json = Vec::with_capacity(batch_read_version_resps.len());
let mut batch_read_version_resps_bin = Vec::with_capacity(batch_read_version_resps.len()); let json_encode_started = internode_stage_timer(attribution_enabled);
for batch_read_version_resp in batch_read_version_resps { for batch_read_version_resp in batch_read_version_resps {
batch_read_version_resps_json.push( batch_read_version_resps_json.push(
compat_response_json(batch_read_version_resp, request_decoded_from_msgpack) compat_response_json(batch_read_version_resp, request_decoded_from_msgpack)
.map_err(|err| DiskError::other(format!("encode BatchReadVersionResp json failed: {err}")))?, .map_err(|err| DiskError::other(format!("encode BatchReadVersionResp json failed: {err}")))?,
); );
}
record_batch_read_version_stage(INTERNODE_STAGE_BATCH_READ_VERSION_RESPONSE_JSON_ENCODE, json_encode_started);
let mut batch_read_version_resps_bin = Vec::with_capacity(batch_read_version_resps.len());
let msgpack_encode_started = internode_stage_timer(attribution_enabled);
for batch_read_version_resp in batch_read_version_resps {
batch_read_version_resps_bin.push(Bytes::from(encode_msgpack_with_capacity( batch_read_version_resps_bin.push(Bytes::from(encode_msgpack_with_capacity(
batch_read_version_resp, batch_read_version_resp,
"BatchReadVersionResp", "BatchReadVersionResp",
FILE_INFO_MSGPACK_ENCODE_CAPACITY_HINT, FILE_INFO_MSGPACK_ENCODE_CAPACITY_HINT,
)?)); )?));
} }
record_batch_read_version_stage(INTERNODE_STAGE_BATCH_READ_VERSION_RESPONSE_MSGPACK_ENCODE, msgpack_encode_started);
Ok((batch_read_version_resps_json, batch_read_version_resps_bin)) Ok((batch_read_version_resps_json, batch_read_version_resps_bin))
} }
@@ -485,15 +505,37 @@ impl NodeService {
&self, &self,
request: Request<BatchReadVersionRequest>, request: Request<BatchReadVersionRequest>,
) -> Result<Response<BatchReadVersionResponse>, Status> { ) -> Result<Response<BatchReadVersionResponse>, Status> {
let attribution_enabled = rustfs_io_metrics::get_stage_metrics_enabled();
let request = request.into_inner(); let request = request.into_inner();
if attribution_enabled {
let metrics = global_internode_metrics();
metrics.record_incoming_request_for_operation_and_backend(
INTERNODE_OPERATION_GRPC_BATCH_READ_VERSION,
INTERNODE_TRANSPORT_BACKEND_GRPC,
);
metrics.record_recv_bytes_for_operation_and_backend(
INTERNODE_OPERATION_GRPC_BATCH_READ_VERSION,
INTERNODE_TRANSPORT_BACKEND_GRPC,
request
.disk
.len()
.saturating_add(request.batch_read_version_req.len())
.saturating_add(request.batch_read_version_req_bin.len()),
);
}
if let Some(disk) = self.find_disk(&request.disk).await { if let Some(disk) = self.find_disk(&request.disk).await {
let decode_started = internode_stage_timer(attribution_enabled);
let decoded_batch_read_version_req: DecodedRpcPayload<BatchReadVersionReq> = match decode_msgpack_or_json_with_source( let decoded_batch_read_version_req: DecodedRpcPayload<BatchReadVersionReq> = match decode_msgpack_or_json_with_source(
&request.batch_read_version_req_bin, &request.batch_read_version_req_bin,
&request.batch_read_version_req, &request.batch_read_version_req,
"BatchReadVersionReq", "BatchReadVersionReq",
) { ) {
Ok(batch_read_version_req) => batch_read_version_req, Ok(batch_read_version_req) => {
record_batch_read_version_stage(INTERNODE_STAGE_BATCH_READ_VERSION_REQUEST_DECODE, decode_started);
batch_read_version_req
}
Err(err) => { Err(err) => {
record_batch_read_version_stage(INTERNODE_STAGE_BATCH_READ_VERSION_REQUEST_DECODE, decode_started);
return Ok(Response::new(BatchReadVersionResponse { return Ok(Response::new(BatchReadVersionResponse {
success: false, success: false,
batch_read_version_resps: Vec::new(), batch_read_version_resps: Vec::new(),
@@ -514,8 +556,10 @@ impl NodeService {
})); }));
} }
let disk_read_started = internode_stage_timer(attribution_enabled);
match disk.batch_read_version(batch_read_version_req).await { match disk.batch_read_version(batch_read_version_req).await {
Ok(batch_read_version_resps) => { Ok(batch_read_version_resps) => {
record_batch_read_version_stage(INTERNODE_STAGE_BATCH_READ_VERSION_DISK_READ, disk_read_started);
let (batch_read_version_resps, batch_read_version_resps_bin) = let (batch_read_version_resps, batch_read_version_resps_bin) =
match encode_batch_read_version_response_payloads(&batch_read_version_resps, request_decoded_from_msgpack) match encode_batch_read_version_response_payloads(&batch_read_version_resps, request_decoded_from_msgpack)
{ {
@@ -537,12 +581,15 @@ impl NodeService {
error: None, error: None,
})) }))
} }
Err(err) => Ok(Response::new(BatchReadVersionResponse { Err(err) => {
success: false, record_batch_read_version_stage(INTERNODE_STAGE_BATCH_READ_VERSION_DISK_READ, disk_read_started);
batch_read_version_resps: Vec::new(), Ok(Response::new(BatchReadVersionResponse {
batch_read_version_resps_bin: Vec::new(), success: false,
error: Some(err.into()), batch_read_version_resps: Vec::new(),
})), batch_read_version_resps_bin: Vec::new(),
error: Some(err.into()),
}))
}
} }
} else { } else {
Ok(Response::new(BatchReadVersionResponse { Ok(Response::new(BatchReadVersionResponse {
@@ -722,9 +769,9 @@ impl NodeService {
&self, &self,
request: Request<ReadVersionRequest>, request: Request<ReadVersionRequest>,
) -> Result<Response<ReadVersionResponse>, Status> { ) -> Result<Response<ReadVersionResponse>, Status> {
let request = request.into_inner();
let metrics = global_internode_metrics(); let metrics = global_internode_metrics();
let read_version_attribution_enabled = rustfs_io_metrics::get_stage_metrics_enabled(); let read_version_attribution_enabled = rustfs_io_metrics::get_stage_metrics_enabled();
let request = request.into_inner();
if read_version_attribution_enabled { if read_version_attribution_enabled {
metrics.record_incoming_request_for_operation_and_backend( metrics.record_incoming_request_for_operation_and_backend(
INTERNODE_OPERATION_GRPC_READ_VERSION, INTERNODE_OPERATION_GRPC_READ_VERSION,
@@ -1635,6 +1682,7 @@ mod tests {
encode_batch_read_version_response_payloads, encode_delete_versions_errors, encode_file_info_msgpack, encode_msgpack, encode_batch_read_version_response_payloads, encode_delete_versions_errors, encode_file_info_msgpack, encode_msgpack,
encode_msgpack_named, encode_read_multiple_response_payloads, encode_rename_data_response_payloads, encode_msgpack_named, encode_read_multiple_response_payloads, encode_rename_data_response_payloads,
}; };
use crate::storage::DiskError;
use crate::storage::rpc::node_service::make_server; use crate::storage::rpc::node_service::make_server;
use crate::storage::storage_api::ReadMultipleResp; use crate::storage::storage_api::ReadMultipleResp;
use crate::storage::storage_api::RenameDataResp; use crate::storage::storage_api::RenameDataResp;
@@ -2028,7 +2076,8 @@ mod tests {
path: "object-a".to_string(), path: "object-a".to_string(),
version_id: "version-a".to_string(), version_id: "version-a".to_string(),
success: false, success: false,
error: "file version not found".to_string(), error: DiskError::FileVersionNotFound.to_string(),
error_code: DiskError::FileVersionNotFound.to_u32(),
..Default::default() ..Default::default()
}]; }];
@@ -2044,8 +2093,43 @@ mod tests {
.expect("msgpack batch read version response should decode"); .expect("msgpack batch read version response should decode");
assert_eq!(json_decoded.index, responses[0].index); assert_eq!(json_decoded.index, responses[0].index);
assert_eq!(json_decoded.error_code, responses[0].error_code);
assert_eq!(msgpack_decoded.path, responses[0].path); assert_eq!(msgpack_decoded.path, responses[0].path);
assert_eq!(msgpack_decoded.error, responses[0].error); assert_eq!(msgpack_decoded.error, responses[0].error);
assert_eq!(msgpack_decoded.error_code, responses[0].error_code);
}
#[test]
fn batch_read_version_response_decode_accepts_legacy_payload_without_error_code() {
#[derive(Serialize)]
struct LegacyBatchReadVersionResp {
index: usize,
path: String,
version_id: String,
success: bool,
file_info: FileInfo,
error: String,
}
let legacy = LegacyBatchReadVersionResp {
index: 2,
path: "object-legacy".to_string(),
version_id: "version-legacy".to_string(),
success: false,
file_info: FileInfo::default(),
error: "legacy error".to_string(),
};
let legacy_json = serde_json::to_string(&legacy).expect("legacy json should encode");
let legacy_msgpack = encode_msgpack(&legacy, "LegacyBatchReadVersionResp").expect("legacy msgpack should encode");
let json_decoded: BatchReadVersionResp =
decode_msgpack_or_json(&[], &legacy_json, "BatchReadVersionResp").expect("legacy json should decode");
let msgpack_decoded: BatchReadVersionResp =
decode_msgpack_or_json(&legacy_msgpack, "", "BatchReadVersionResp").expect("legacy msgpack should decode");
assert_eq!(json_decoded.error_code, 0);
assert_eq!(msgpack_decoded.error_code, 0);
assert_eq!(msgpack_decoded.error, legacy.error);
} }
#[test] #[test]
+4
View File
@@ -1102,6 +1102,10 @@ pub(crate) fn shutdown_background_monitors() {
rustfs_ecstore::shutdown_background_monitors(); rustfs_ecstore::shutdown_background_monitors();
} }
pub(crate) fn mark_get_metadata_read_version_coalescing_service_ready() {
rustfs_ecstore::mark_get_metadata_read_version_coalescing_service_ready();
}
pub(crate) fn set_global_rustfs_port(value: u16) { pub(crate) fn set_global_rustfs_port(value: u16) {
ecstore_global::set_global_rustfs_port(value); ecstore_global::set_global_rustfs_port(value);
} }
+2 -1
View File
@@ -284,7 +284,8 @@ pub(crate) mod startup {
pub(crate) mod shutdown { pub(crate) mod shutdown {
pub(crate) use crate::storage::storage_api::{ pub(crate) use crate::storage::storage_api::{
shutdown_background_monitors, shutdown_background_services, store_compression_total_in_backend, mark_get_metadata_read_version_coalescing_service_ready, shutdown_background_monitors, shutdown_background_services,
store_compression_total_in_backend,
}; };
} }
@@ -0,0 +1,263 @@
#!/usr/bin/env python3
"""Fail when a critical scheduled validation has not started recently."""
from __future__ import annotations
import argparse
from datetime import datetime, timedelta, timezone
import json
import os
from pathlib import Path
import re
import sys
import tempfile
import unittest
from unittest import mock
from urllib.parse import quote, urlencode
from urllib.request import Request, urlopen
ROOT = Path(__file__).resolve().parents[1]
def load_validations(path: Path) -> list[tuple[str, int]]:
data = json.loads(path.read_text())
if not isinstance(data, list) or not data:
raise ValueError("scheduled validation config must be a non-empty list")
validations: list[tuple[str, int]] = []
seen: set[str] = set()
for item in data:
if not isinstance(item, dict):
raise ValueError("scheduled validation entries must be objects")
workflow = item.get("workflow")
max_age_hours = item.get("max_age_hours")
if not isinstance(workflow, str) or not re.fullmatch(
r"\.github/workflows/[a-z0-9-]+\.yml", workflow
):
raise ValueError(f"invalid scheduled validation workflow: {workflow!r}")
if workflow in seen:
raise ValueError(f"duplicate scheduled validation workflow: {workflow}")
if (
not isinstance(max_age_hours, int)
or isinstance(max_age_hours, bool)
or max_age_hours <= 0
):
raise ValueError(f"invalid max_age_hours for {workflow}: {max_age_hours!r}")
seen.add(workflow)
validations.append((workflow, max_age_hours))
return validations
def parse_timestamp(value: object) -> datetime:
if not isinstance(value, str):
raise ValueError(f"invalid run timestamp: {value!r}")
parsed = datetime.fromisoformat(value.replace("Z", "+00:00"))
if parsed.tzinfo is None:
raise ValueError(f"run timestamp has no timezone: {value!r}")
return parsed.astimezone(timezone.utc)
def stale_reason(
run: dict[str, object] | None, now: datetime, max_age_hours: int
) -> str | None:
if run is None:
return "no scheduled run has been recorded"
created_at = parse_timestamp(run.get("created_at"))
age = now - created_at
if age > timedelta(hours=max_age_hours):
return f"last scheduled run is {age.total_seconds() / 3600:.1f}h old"
return None
def fetch_latest_scheduled_run(
repository: str, workflow: str, token: str, api_url: str
) -> dict[str, object] | None:
owner, repo = repository.split("/", 1)
workflow_name = Path(workflow).name
endpoint = (
f"{api_url.rstrip('/')}/repos/{quote(owner, safe='')}/{quote(repo, safe='')}"
f"/actions/workflows/{quote(workflow_name, safe='')}/runs?"
+ urlencode({"event": "schedule", "per_page": 1})
)
request = Request(
endpoint,
headers={
"Accept": "application/vnd.github+json",
"Authorization": f"Bearer {token}",
"X-GitHub-Api-Version": "2022-11-28",
},
)
with urlopen(request, timeout=30) as response:
payload = json.load(response)
runs = payload.get("workflow_runs")
if not isinstance(runs, list):
raise ValueError(f"GitHub returned no workflow_runs list for {workflow}")
if not runs:
return None
if not isinstance(runs[0], dict):
raise ValueError(f"GitHub returned an invalid workflow run for {workflow}")
return runs[0]
def write_report(path: Path, failures: list[tuple[str, int, str, str]]) -> None:
lines = ["## Scheduled validation freshness"]
if not failures:
lines.append("")
lines.append("All critical scheduled validations have a recent scheduled run.")
else:
lines.extend(
[
"",
"The following critical validations are stale or could not be inspected:",
"",
"| Workflow | Limit | Result | Last run |",
"| --- | ---: | --- | --- |",
]
)
for workflow, max_age_hours, reason, run_url in failures:
link = f"[open]({run_url})" if run_url else ""
lines.append(f"| `{workflow}` | {max_age_hours}h | {reason} | {link} |")
path.write_text("\n".join(lines) + "\n")
def check_freshness(
config: Path, report: Path, repository: str, token: str, api_url: str
) -> int:
now = datetime.now(timezone.utc)
failures: list[tuple[str, int, str, str]] = []
for workflow, max_age_hours in load_validations(config):
try:
run = fetch_latest_scheduled_run(repository, workflow, token, api_url)
reason = stale_reason(run, now, max_age_hours)
if reason is not None:
run_url = str(run.get("html_url", "")) if run else ""
failures.append((workflow, max_age_hours, reason, run_url))
except Exception as error:
failures.append(
(workflow, max_age_hours, f"inspection failed: {error}", "")
)
write_report(report, failures)
return 1 if failures else 0
class SelfTests(unittest.TestCase):
NOW = datetime(2026, 8, 22, 12, tzinfo=timezone.utc)
def test_freshness_boundaries(self) -> None:
at_limit = {"created_at": "2026-08-21T00:00:00Z"}
past_limit = {"created_at": "2026-08-20T23:59:59Z"}
self.assertIsNone(stale_reason(at_limit, self.NOW, 36))
self.assertIsNotNone(stale_reason(past_limit, self.NOW, 36))
self.assertIsNotNone(stale_reason(None, self.NOW, 36))
def test_config_rejects_duplicate_and_invalid_entries(self) -> None:
with tempfile.TemporaryDirectory() as tmp:
path = Path(tmp) / "validations.json"
path.write_text(
json.dumps(
[
{"workflow": ".github/workflows/ci.yml", "max_age_hours": 36},
{"workflow": ".github/workflows/ci.yml", "max_age_hours": 0},
]
)
)
with self.assertRaises(ValueError):
load_validations(path)
path.write_text(
json.dumps(
[{"workflow": ".github/workflows/ci.yml", "max_age_hours": 0}]
)
)
with self.assertRaises(ValueError):
load_validations(path)
def test_check_reports_missing_runs(self) -> None:
with tempfile.TemporaryDirectory() as tmp:
root = Path(tmp)
config = root / "validations.json"
report = root / "report.md"
config.write_text(
json.dumps(
[
{"workflow": ".github/workflows/ci.yml", "max_age_hours": 36},
{"workflow": ".github/workflows/fuzz.yml", "max_age_hours": 36},
{"workflow": ".github/workflows/mint.yml", "max_age_hours": 36},
]
)
)
with mock.patch(
__name__ + ".fetch_latest_scheduled_run",
side_effect=[
{"created_at": "2999-01-01T00:00:00Z"},
None,
RuntimeError("API unavailable"),
],
):
self.assertEqual(
check_freshness(
config,
report,
"rustfs/rustfs",
"token",
"https://api.github.test",
),
1,
)
contents = report.read_text()
self.assertIn(".github/workflows/fuzz.yml", contents)
self.assertIn("inspection failed: API unavailable", contents)
self.assertNotIn(".github/workflows/ci.yml`", contents)
config.write_text(
json.dumps(
[{"workflow": ".github/workflows/ci.yml", "max_age_hours": 36}]
)
)
with mock.patch(
__name__ + ".fetch_latest_scheduled_run",
return_value={"created_at": "2999-01-01T00:00:00Z"},
):
self.assertEqual(
check_freshness(
config,
report,
"rustfs/rustfs",
"token",
"https://api.github.test",
),
0,
)
self.assertIn("All critical scheduled validations", report.read_text())
def main() -> int:
parser = argparse.ArgumentParser(description=__doc__)
parser.add_argument(
"--config", type=Path, default=ROOT / ".github/scheduled-validations.json"
)
parser.add_argument("--report", type=Path)
parser.add_argument("--self-test", action="store_true")
args = parser.parse_args()
if args.self_test:
load_validations(args.config)
suite = unittest.defaultTestLoader.loadTestsFromTestCase(SelfTests)
return (
0 if unittest.TextTestRunner(verbosity=2).run(suite).wasSuccessful() else 1
)
if args.report is None:
parser.error("--report is required unless --self-test is used")
repository = os.environ.get("GITHUB_REPOSITORY", "")
token = os.environ.get("GH_TOKEN", "")
api_url = os.environ.get("GITHUB_API_URL", "https://api.github.com")
if not re.fullmatch(r"[^/\s]+/[^/\s]+", repository):
parser.error("GITHUB_REPOSITORY must be owner/repository")
if not token:
parser.error("GH_TOKEN is required")
return check_freshness(args.config, args.report, repository, token, api_url)
if __name__ == "__main__":
raise SystemExit(main())
+561 -1
View File
@@ -10,11 +10,17 @@ import sys
import tempfile import tempfile
import tomllib import tomllib
import unittest import unittest
from datetime import datetime, timezone
from unittest import mock from unittest import mock
from pathlib import Path from pathlib import Path
from zoneinfo import ZoneInfo, ZoneInfoNotFoundError
ROOT = Path(__file__).resolve().parents[1] ROOT = Path(__file__).resolve().parents[1]
SCHEDULED_ALERT_WORKFLOWS = tuple(
item["workflow"]
for item in json.loads((ROOT / ".github/scheduled-validations.json").read_text())
)
def words(value: str) -> set[str]: def words(value: str) -> set[str]:
@@ -252,6 +258,292 @@ def check_profile_definitions(root: Path) -> list[str]:
return errors return errors
def yaml_block(lines: list[str], key: str, indent: int) -> list[str] | None:
try:
start = lines.index(f"{' ' * indent}{key}:") + 1
except ValueError:
return None
end = next(
(
index
for index in range(start, len(lines))
if lines[index].strip()
and not lines[index].lstrip().startswith("#")
and len(lines[index]) - len(lines[index].lstrip()) <= indent
),
len(lines),
)
return lines[start:end]
def workflow_step_block(job_lines: list[str], action: str) -> tuple[int, list[str]] | None:
uses_index = next(
(
index
for index, line in enumerate(job_lines)
if (
line.split("#", 1)[0].strip() == f"- uses: {action}"
and len(line) - len(line.lstrip()) == 6
)
or (
line.split("#", 1)[0].strip() == f"uses: {action}"
and len(line) - len(line.lstrip()) == 8
)
),
None,
)
if uses_index is None:
return None
start = next(
(
index
for index in range(uses_index, -1, -1)
if job_lines[index].lstrip().startswith("- ")
),
uses_index,
)
indent = len(job_lines[start]) - len(job_lines[start].lstrip())
end = next(
(
index
for index in range(start + 1, len(job_lines))
if len(job_lines[index]) - len(job_lines[index].lstrip()) == indent
and job_lines[index].lstrip().startswith("- ")
),
len(job_lines),
)
return start, job_lines[start:end]
def alert_step_errors(
job_lines: list[str],
expected_action_if: str | None,
required_permissions: tuple[str, ...],
required_action_tokens: tuple[str, ...],
) -> list[str]:
checkout = workflow_step_block(job_lines, "actions/checkout@9c091bb21b7c1c1d1991bb908d89e4e9dddfe3e0")
action = workflow_step_block(job_lines, "./.github/actions/schedule-failure-issue")
errors: list[str] = []
permissions = yaml_block(job_lines, "permissions", 4)
permission_text = "\n".join(line.split("#", 1)[0] for line in permissions or [])
missing_permissions = [token for token in required_permissions if token not in permission_text]
if missing_permissions:
errors.append("alert job permissions missing " + ", ".join(missing_permissions))
if checkout is None:
errors.append("checkout step is missing")
if action is None:
errors.append("local alert action step is missing")
if checkout is None or action is None:
return errors
if checkout[0] >= action[0]:
errors.append("checkout must run before the local alert action")
checkout_ifs = [line.strip() for line in checkout[1] if line.strip().startswith("if:")]
if checkout_ifs:
errors.append("checkout step must not be conditional")
action_ifs = [line.strip() for line in action[1] if line.strip().startswith("if:")]
expected_ifs = [] if expected_action_if is None else [expected_action_if]
if action_ifs != expected_ifs:
errors.append("alert action has an invalid step condition")
action_text = "\n".join(line.split("#", 1)[0] for line in action[1])
missing_action_tokens = [token for token in required_action_tokens if token not in action_text]
if missing_action_tokens:
errors.append("alert action inputs missing " + ", ".join(missing_action_tokens))
return errors
def schedule_utc_slots(hour: int, minute: int, timezone_name: str | None) -> set[tuple[int, int]]:
if timezone_name is None:
return {(hour, minute)}
zone = ZoneInfo(timezone_name)
return {
(utc.hour, utc.minute)
for year in (2025, 2026)
for month in range(1, 13)
for utc in [datetime(year, month, 1, hour, minute, tzinfo=zone).astimezone(timezone.utc)]
}
def check_scheduled_alerts(root: Path) -> list[str]:
errors: list[str] = []
schedule_slots: dict[tuple[int, int], list[str]] = {}
for relative in SCHEDULED_ALERT_WORKFLOWS:
path = root / relative
try:
lines = path.read_text().splitlines()
except FileNotFoundError:
errors.append(f"{relative}: missing scheduled validation workflow")
continue
on_block = yaml_block(lines, "on", 0)
schedule_block = yaml_block(on_block or [], "schedule", 2)
schedule_lines = schedule_block or []
cron_indices = [index for index, line in enumerate(schedule_lines) if re.match(r"^\s*-\s+cron:", line)]
if not cron_indices:
errors.append(f"{relative}: missing simple numeric schedule")
else:
for position, cron_index in enumerate(cron_indices):
cron_line = schedule_lines[cron_index]
schedule = re.match(r"^\s*-\s+cron:\s*[\"']?(\d+)\s+(\d+)\s+", cron_line)
if not schedule:
errors.append(f"{relative}: missing simple numeric schedule")
continue
minute, hour = map(int, schedule.groups())
if minute == 0:
errors.append(f"{relative}: scheduled validation must avoid minute zero")
entry_end = cron_indices[position + 1] if position + 1 < len(cron_indices) else len(schedule_lines)
entry = "\n".join(schedule_lines[cron_index + 1 : entry_end])
timezone_match = re.search(r"^\s*timezone:\s*[\"']?([^\"'\s]+)", entry, re.MULTILINE)
timezone_name = timezone_match.group(1) if timezone_match else None
try:
utc_slots = schedule_utc_slots(hour, minute, timezone_name)
except ZoneInfoNotFoundError:
errors.append(f"{relative}: unknown schedule timezone {timezone_name}")
continue
for slot in utc_slots:
schedule_slots.setdefault(slot, []).append(relative)
job_lines = yaml_block(lines, "alert-on-failure", 2)
if job_lines is None:
errors.append(f"{relative}: missing alert-on-failure job")
continue
job = "\n".join(line.split("#", 1)[0] for line in job_lines)
required = (
"always()",
"github.event_name == 'schedule'",
"contains(needs.*.result, 'failure')",
"issues: write",
"uses: actions/checkout@9c091bb21b7c1c1d1991bb908d89e4e9dddfe3e0",
"uses: ./.github/actions/schedule-failure-issue",
"github-token: ${{ secrets.GITHUB_TOKEN }}",
)
missing = [token for token in required if token not in job]
if missing:
errors.append(f"{relative}: alert-on-failure missing {', '.join(missing)}")
else:
errors.extend(
f"{relative}: {error}"
for error in alert_step_errors(job_lines, None, ("issues: write",), ("github-token: ${{ secrets.GITHUB_TOKEN }}",))
)
for (hour, minute), workflows in schedule_slots.items():
if len(workflows) > 1:
errors.append(
f"scheduled validations share {hour:02d}:{minute:02d} UTC: {', '.join(workflows)}"
)
watchdog_path = root / ".github/workflows/scheduled-validation-watchdog.yml"
try:
watchdog_lines = watchdog_path.read_text().splitlines()
except FileNotFoundError:
errors.append(".github/workflows/scheduled-validation-watchdog.yml: missing completion watchdog")
return errors
watchdog_on = yaml_block(watchdog_lines, "on", 0)
watchdog_run = yaml_block(watchdog_on or [], "workflow_run", 2)
watchdog_workflows = yaml_block(watchdog_run or [], "workflows", 4)
if watchdog_workflows is None:
errors.append(".github/workflows/scheduled-validation-watchdog.yml: missing workflow_run workflows")
return errors
watchdog_sources = "\n".join(line.split("#", 1)[0] for line in watchdog_workflows)
for relative in SCHEDULED_ALERT_WORKFLOWS:
path = root / relative
if not path.is_file():
continue
source = path.read_text()
match = re.search(r"^name:\s*[\"']?([^\"'\n]+)", source, re.MULTILINE)
if not match:
errors.append(f"{relative}: missing workflow name")
elif f'- "{match.group(1).strip()}"' not in watchdog_sources:
errors.append(f"{relative}: missing from scheduled completion watchdog")
watchdog_job_lines = yaml_block(watchdog_lines, "alert-on-incomplete-run", 2)
if watchdog_job_lines is None:
errors.append(".github/workflows/scheduled-validation-watchdog.yml: missing alert-on-incomplete-run job")
return errors
watchdog_job = "\n".join(line.split("#", 1)[0] for line in watchdog_job_lines)
required = (
"github.event.workflow_run.event == 'schedule'",
"github.event.workflow_run.conclusion != 'success'",
"github.event.workflow_run.conclusion != 'failure'",
"actions: read",
"issues: write",
"uses: actions/checkout@9c091bb21b7c1c1d1991bb908d89e4e9dddfe3e0",
"uses: ./.github/actions/schedule-failure-issue",
"github-token: ${{ secrets.GITHUB_TOKEN }}",
"workflow-name: ${{ github.event.workflow_run.name }}",
"source-run-id: ${{ github.event.workflow_run.id }}",
"source-run-attempt: ${{ github.event.workflow_run.run_attempt }}",
"source-event: ${{ github.event.workflow_run.event }}",
"source-ref-name: ${{ github.event.workflow_run.head_branch }}",
"source-sha: ${{ github.event.workflow_run.head_sha }}",
)
missing = [token for token in required if token not in watchdog_job]
if missing:
errors.append(
".github/workflows/scheduled-validation-watchdog.yml: missing " + ", ".join(missing)
)
else:
errors.extend(
".github/workflows/scheduled-validation-watchdog.yml: " + error
for error in alert_step_errors(
watchdog_job_lines,
None,
("actions: read", "issues: write"),
(
"github-token: ${{ secrets.GITHUB_TOKEN }}",
"workflow-name: ${{ github.event.workflow_run.name }}",
"source-run-id: ${{ github.event.workflow_run.id }}",
"source-run-attempt: ${{ github.event.workflow_run.run_attempt }}",
"source-event: ${{ github.event.workflow_run.event }}",
"source-ref-name: ${{ github.event.workflow_run.head_branch }}",
"source-sha: ${{ github.event.workflow_run.head_sha }}",
),
)
)
freshness_path = root / ".github/workflows/scheduled-validation-freshness.yml"
try:
freshness_lines = freshness_path.read_text().splitlines()
except FileNotFoundError:
errors.append(".github/workflows/scheduled-validation-freshness.yml: missing freshness check")
return errors
freshness_job_lines = yaml_block(freshness_lines, "check-freshness", 2)
if freshness_job_lines is None:
errors.append(".github/workflows/scheduled-validation-freshness.yml: missing check-freshness job")
return errors
freshness_job = "\n".join(line.split("#", 1)[0] for line in freshness_job_lines)
required = (
"python3 scripts/check_scheduled_validation_freshness.py",
"actions: read",
"issues: write",
"if: failure()",
"uses: actions/checkout@9c091bb21b7c1c1d1991bb908d89e4e9dddfe3e0",
"uses: ./.github/actions/schedule-failure-issue",
"github-token: ${{ secrets.GITHUB_TOKEN }}",
"details-file: ${{ runner.temp }}/scheduled-validation-freshness.md",
)
missing = [token for token in required if token not in freshness_job]
if missing:
errors.append(
".github/workflows/scheduled-validation-freshness.yml: missing " + ", ".join(missing)
)
else:
errors.extend(
".github/workflows/scheduled-validation-freshness.yml: " + error
for error in alert_step_errors(
freshness_job_lines,
"if: failure()",
("actions: read", "issues: write"),
(
"github-token: ${{ secrets.GITHUB_TOKEN }}",
"details-file: ${{ runner.temp }}/scheduled-validation-freshness.md",
),
)
)
if not (root / "scripts/check_scheduled_validation_freshness.py").is_file():
errors.append("scripts/check_scheduled_validation_freshness.py: missing freshness checker")
return errors
def check_profile_listing(root: Path, profile: str, listing: Path) -> list[str]: def check_profile_listing(root: Path, profile: str, listing: Path) -> list[str]:
try: try:
expected_digest = profile_selection(root, profile) expected_digest = profile_selection(root, profile)
@@ -281,6 +573,7 @@ def validate(root: Path) -> list[str]:
errors.extend(check_runner_selection(root)) errors.extend(check_runner_selection(root))
errors.extend(check_s3_tests_runner(root)) errors.extend(check_s3_tests_runner(root))
errors.extend(check_profile_definitions(root)) errors.extend(check_profile_definitions(root))
errors.extend(check_scheduled_alerts(root))
return errors return errors
@@ -363,6 +656,7 @@ class SelfTests(unittest.TestCase):
mock.patch(__name__ + ".check_fuzz_targets", return_value=[]), mock.patch(__name__ + ".check_fuzz_targets", return_value=[]),
mock.patch(__name__ + ".check_runner_selection", return_value=[]), mock.patch(__name__ + ".check_runner_selection", return_value=[]),
mock.patch(__name__ + ".check_profile_definitions", return_value=[]), mock.patch(__name__ + ".check_profile_definitions", return_value=[]),
mock.patch(__name__ + ".check_scheduled_alerts", return_value=[]),
): ):
self.assertEqual(len(validate(root)), 1) self.assertEqual(len(validate(root)), 1)
@@ -413,6 +707,272 @@ class SelfTests(unittest.TestCase):
with mock.patch.object(sys, "platform", "linux"): with mock.patch.object(sys, "platform", "linux"):
self.assertEqual(len(check_profile_listing(root, "e2e-full", listing)), 1) self.assertEqual(len(check_profile_listing(root, "e2e-full", listing)), 1)
def test_scheduled_alerts_require_completion_watchdog(self) -> None:
with tempfile.TemporaryDirectory() as tmp:
root = Path(tmp)
alert = (
" alert-on-failure:\n"
" if: always() && github.event_name == 'schedule' && "
"contains(needs.*.result, 'failure')\n"
" permissions:\n"
" issues: write\n"
" steps:\n"
" - uses: actions/checkout@9c091bb21b7c1c1d1991bb908d89e4e9dddfe3e0\n"
" - uses: ./.github/actions/schedule-failure-issue\n"
" with:\n"
" github-token: ${{ secrets.GITHUB_TOKEN }}\n"
)
names: list[str] = []
for index, relative in enumerate(SCHEDULED_ALERT_WORKFLOWS, start=1):
path = root / relative
path.parent.mkdir(parents=True, exist_ok=True)
names.append(path.stem)
path.write_text(
f'name: "{path.stem}"\n'
f'on:\n schedule:\n - cron: "{index} {index} * * *"\n'
f'jobs:\n{alert}'
)
watchdog = root / ".github/workflows/scheduled-validation-watchdog.yml"
watchdog.write_text(
"on:\n workflow_run:\n workflows:\n"
+ "\n".join(f' - "{name}"' for name in names)
+ "\njobs:\n"
+ " alert-on-incomplete-run:\n"
+ " github.event.workflow_run.event == 'schedule'\n"
+ " github.event.workflow_run.conclusion != 'success'\n"
+ " github.event.workflow_run.conclusion != 'failure'\n"
+ " permissions:\n"
+ " actions: read\n"
+ " issues: write\n"
+ " steps:\n"
+ " - uses: actions/checkout@9c091bb21b7c1c1d1991bb908d89e4e9dddfe3e0\n"
+ " - uses: ./.github/actions/schedule-failure-issue\n"
+ " with:\n"
+ " github-token: ${{ secrets.GITHUB_TOKEN }}\n"
+ " workflow-name: ${{ github.event.workflow_run.name }}\n"
+ " source-run-id: ${{ github.event.workflow_run.id }}\n"
+ " source-run-attempt: ${{ github.event.workflow_run.run_attempt }}\n"
+ " source-event: ${{ github.event.workflow_run.event }}\n"
+ " source-ref-name: ${{ github.event.workflow_run.head_branch }}\n"
+ " source-sha: ${{ github.event.workflow_run.head_sha }}\n"
)
freshness = root / ".github/workflows/scheduled-validation-freshness.yml"
freshness.write_text(
"jobs:\n"
" check-freshness:\n"
" permissions:\n"
" actions: read\n"
" issues: write\n"
" steps:\n"
" - uses: actions/checkout@9c091bb21b7c1c1d1991bb908d89e4e9dddfe3e0\n"
" - run: python3 scripts/check_scheduled_validation_freshness.py\n"
" - uses: ./.github/actions/schedule-failure-issue\n"
" if: failure()\n"
" with:\n"
" github-token: ${{ secrets.GITHUB_TOKEN }}\n"
" details-file: ${{ runner.temp }}/scheduled-validation-freshness.md\n"
)
checker = root / "scripts/check_scheduled_validation_freshness.py"
checker.parent.mkdir()
checker.write_text("")
self.assertEqual(check_scheduled_alerts(root), [])
first = root / SCHEDULED_ALERT_WORKFLOWS[0]
mutations = (
("contains(needs.*.result, 'failure')", "false"),
("issues: write", "issues: read"),
(
"uses: actions/checkout@9c091bb21b7c1c1d1991bb908d89e4e9dddfe3e0",
"uses: actions/checkout@missing",
),
(
" - uses: actions/checkout@9c091bb21b7c1c1d1991bb908d89e4e9dddfe3e0\n",
" - uses: actions/checkout@9c091bb21b7c1c1d1991bb908d89e4e9dddfe3e0\n"
" if: github.event_name == 'workflow_dispatch'\n",
),
(
" - uses: ./.github/actions/schedule-failure-issue\n",
" - uses: ./.github/actions/schedule-failure-issue\n"
" if: github.event_name == 'workflow_dispatch'\n",
),
("uses: ./.github/actions/schedule-failure-issue", "uses: actions/checkout@v7"),
("github-token: ${{ secrets.GITHUB_TOKEN }}", "github-token: missing"),
)
for required, replacement in mutations:
original = first.read_text()
first.write_text(original.replace(required, replacement))
self.assertEqual(len(check_scheduled_alerts(root)), 1)
first.write_text(original)
first_original = first.read_text()
real_steps = (
" - uses: actions/checkout@9c091bb21b7c1c1d1991bb908d89e4e9dddfe3e0\n"
" - uses: ./.github/actions/schedule-failure-issue\n"
" with:\n"
" github-token: ${{ secrets.GITHUB_TOKEN }}\n"
)
first.write_text(
first_original.replace(
real_steps,
" - run: |\n"
" : <<'MARKER'\n"
" uses: actions/checkout@9c091bb21b7c1c1d1991bb908d89e4e9dddfe3e0\n"
" MARKER\n"
" - run: |\n"
" : <<'MARKER'\n"
" uses: ./.github/actions/schedule-failure-issue\n"
" github-token: ${{ secrets.GITHUB_TOKEN }}\n"
" MARKER\n",
)
)
self.assertTrue(check_scheduled_alerts(root))
first.write_text(
first_original.replace(
real_steps,
" - uses: ./.github/actions/schedule-failure-issue\n"
" with:\n"
" github-token: ${{ secrets.GITHUB_TOKEN }}\n"
" - uses: actions/checkout@9c091bb21b7c1c1d1991bb908d89e4e9dddfe3e0\n",
)
)
self.assertEqual(len(check_scheduled_alerts(root)), 1)
first.write_text(first_original)
watchdog_mutations = (
("actions: read", "actions: none"),
("issues: write", "issues: read"),
(
"uses: actions/checkout@9c091bb21b7c1c1d1991bb908d89e4e9dddfe3e0",
"uses: actions/checkout@missing",
),
(
" - uses: actions/checkout@9c091bb21b7c1c1d1991bb908d89e4e9dddfe3e0\n",
" - uses: actions/checkout@9c091bb21b7c1c1d1991bb908d89e4e9dddfe3e0\n"
" if: github.event_name == 'workflow_dispatch'\n",
),
(
" - uses: ./.github/actions/schedule-failure-issue\n",
" - uses: ./.github/actions/schedule-failure-issue\n"
" if: github.event_name == 'workflow_dispatch'\n",
),
("uses: ./.github/actions/schedule-failure-issue", "uses: actions/checkout@v7"),
("github-token: ${{ secrets.GITHUB_TOKEN }}", "github-token: missing"),
("source-event: ${{ github.event.workflow_run.event }}", "source-event: watchdog"),
(
"source-ref-name: ${{ github.event.workflow_run.head_branch }}",
"source-ref-name: main",
),
("source-sha: ${{ github.event.workflow_run.head_sha }}", "source-sha: missing"),
)
for required, replacement in watchdog_mutations:
original = watchdog.read_text()
watchdog.write_text(original.replace(required, replacement))
self.assertEqual(len(check_scheduled_alerts(root)), 1)
watchdog.write_text(original)
watchdog_original = watchdog.read_text()
watchdog.write_text(
watchdog_original.replace("issues: write", "issues: read")
+ " decoy:\n permissions:\n issues: write\n"
)
self.assertEqual(len(check_scheduled_alerts(root)), 1)
watchdog.write_text(watchdog_original)
first_original = first.read_text()
first.write_text(
first_original.replace(' schedule:\n - cron: "1 1 * * *"\n', "")
+ ' decoy:\n strategy:\n matrix:\n cron:\n - "1 1 * * *"\n'
+ ' runs-on: ubuntu-latest\n steps:\n - run: true\n'
)
self.assertEqual(len(check_scheduled_alerts(root)), 1)
first.write_text(first_original)
first.write_text(
first_original.replace(
' - cron: "1 1 * * *"\n',
' - cron: "1 1 * * *"\n - cron: "0 5 * * *"\n',
)
)
self.assertEqual(len(check_scheduled_alerts(root)), 1)
first.write_text(
first_original.replace(
' - cron: "1 1 * * *"\n',
' - cron: "1 1 * * *"\n - cron: "2 2 * * *"\n',
)
)
self.assertEqual(len(check_scheduled_alerts(root)), 1)
first.write_text(first_original)
watchdog.write_text(
watchdog_original.replace(f' - "{names[0]}"\n', "")
+ f' decoy:\n strategy:\n matrix:\n workflow:\n - "{names[0]}"\n'
+ ' runs-on: ubuntu-latest\n steps:\n - run: true\n'
)
self.assertEqual(len(check_scheduled_alerts(root)), 1)
watchdog.write_text(watchdog_original)
watchdog.write_text(watchdog_original.replace(f' - "{names[0]}"\n', ""))
self.assertEqual(len(check_scheduled_alerts(root)), 1)
watchdog.write_text(watchdog_original)
original = first.read_text()
first.write_text(re.sub(r'- cron: "\d+ \d+', '- cron: "0 0', original, count=1))
self.assertEqual(len(check_scheduled_alerts(root)), 1)
first.write_text(original)
second = root / SCHEDULED_ALERT_WORKFLOWS[1]
second_original = second.read_text()
second.write_text(re.sub(r'- cron: "\d+ \d+', '- cron: "1 1', second_original, count=1))
self.assertEqual(len(check_scheduled_alerts(root)), 1)
second.write_text(second_original)
first.write_text(
first_original.replace(
' - cron: "1 1 * * *"\n',
' - cron: "7 0 * * *"\n timezone: "Asia/Shanghai"\n',
)
)
second.write_text(
second_original.replace(
' - cron: "2 2 * * *"\n',
' - cron: "2 2 * * *"\n - cron: "7 16 * * *"\n',
)
)
self.assertEqual(len(check_scheduled_alerts(root)), 1)
first.write_text(first_original)
second.write_text(second_original)
freshness_original = freshness.read_text()
freshness.write_text(freshness_original.replace("details-file:", "report-file:"))
self.assertEqual(len(check_scheduled_alerts(root)), 1)
freshness.write_text(
freshness_original.replace("github-token: ${{ secrets.GITHUB_TOKEN }}", "github-token: missing")
)
self.assertEqual(len(check_scheduled_alerts(root)), 1)
freshness.write_text(
freshness_original.replace(
"uses: actions/checkout@9c091bb21b7c1c1d1991bb908d89e4e9dddfe3e0",
"uses: actions/checkout@missing",
)
)
self.assertEqual(len(check_scheduled_alerts(root)), 1)
freshness.write_text(
freshness_original.replace(
" - uses: actions/checkout@9c091bb21b7c1c1d1991bb908d89e4e9dddfe3e0\n",
" - uses: actions/checkout@9c091bb21b7c1c1d1991bb908d89e4e9dddfe3e0\n"
" if: github.event_name == 'workflow_dispatch'\n",
)
)
self.assertEqual(len(check_scheduled_alerts(root)), 1)
freshness.write_text(
freshness_original.replace("if: failure()", "if: github.event_name == 'workflow_dispatch'")
)
self.assertEqual(len(check_scheduled_alerts(root)), 1)
freshness.write_text(
freshness_original.replace("issues: write", "issues: read")
+ " decoy:\n permissions:\n issues: write\n"
)
self.assertEqual(len(check_scheduled_alerts(root)), 1)
def main() -> int: def main() -> int:
if sys.argv[1:] == ["--self-test"]: if sys.argv[1:] == ["--self-test"]:
suite = unittest.defaultTestLoader.loadTestsFromTestCase(SelfTests) suite = unittest.defaultTestLoader.loadTestsFromTestCase(SelfTests)
@@ -436,7 +996,7 @@ def main() -> int:
for error in errors: for error in errors:
print(f"ERROR: {error}", file=sys.stderr) print(f"ERROR: {error}", file=sys.stderr)
return 1 return 1
print("OK: e2e modules, runner selection, fuzz matrices, profiles, and bounded diagnostics are wired") print("OK: e2e modules, runner selection, fuzz matrices, profiles, and scheduled alerts are wired")
return 0 return 0
+22
View File
@@ -64,3 +64,25 @@ if [ "$status" -ne 0 ]; then
fi fi
echo "OK: every ./.github/actions/setup call states cache-save-if explicitly" echo "OK: every ./.github/actions/setup call states cache-save-if explicitly"
# rust-cache hashes every CARGO*, CC*, CFLAGS*, CXX*, CMAKE*, and RUST*
# variable that is present when the setup action runs. The dedicated writer
# and the CI readers therefore need identical workflow-level compiler env.
compiler_env() {
awk '
/^env:[[:space:]]*$/ { in_env = 1; next }
in_env && /^[^[:space:]]/ { exit }
in_env && /^ (CARGO|CC|CFLAGS|CXX|CMAKE|RUST)[A-Z0-9_]*:/ { print }
' "$1" | sort
}
ci_env="$(compiler_env .github/workflows/ci.yml)"
warm_env="$(compiler_env .github/workflows/cache-warm.yml)"
if [ "$ci_env" != "$warm_env" ]; then
echo "CI and cache-warm compiler environments differ; rust-cache keys will not match:" >&2
diff -u <(printf '%s\n' "$ci_env") <(printf '%s\n' "$warm_env") >&2 || true
exit 1
fi
echo "OK: cache-warm and CI compiler environments match"
@@ -13,8 +13,29 @@ require_absent_pattern() {
fi fi
} }
require_present_pattern() {
local pattern="$1"
local description="$2"
if ! grep -Eq -- "$pattern" "$workflow"; then
echo "invalid performance A/B workflow contract: $description" >&2
exit 1
fi
}
require_absent_pattern '(^|[^[:alnum:]_])pull_request(_target)?([^[:alnum:]_]|$)' "the workflow must not contain PR event handling" require_absent_pattern '(^|[^[:alnum:]_])pull_request(_target)?([^[:alnum:]_]|$)' "the workflow must not contain PR event handling"
require_absent_pattern 'pull-requests[[:space:]]*:[[:space:]]*write' "the workflow must not receive PR write permission" require_absent_pattern 'pull-requests[[:space:]]*:[[:space:]]*write' "the workflow must not receive PR write permission"
require_absent_pattern 'permissions[[:space:]]*:[[:space:]]*write-all' "the workflow must not receive broad write permission" require_absent_pattern 'permissions[[:space:]]*:[[:space:]]*write-all' "the workflow must not receive broad write permission"
require_absent_pattern '^[[:space:]]*push:' "the workflow must not spend a release build on every main push"
require_present_pattern 'listWorkflowRuns' "the scheduled baseline must come from workflow history"
require_present_pattern 'status:[[:space:]]*"success"' "the scheduled baseline must be a successful run"
require_present_pattern 'SCHEDULED_BASELINE_SHA' "the resolved scheduled baseline must reach the comparison"
require_present_pattern "SCHEDULED_BASELINE_SHA:-\\\$candidate_sha" "the first scheduled run must seed from its verified candidate"
require_present_pattern 'git merge-base --is-ancestor' "the scheduled baseline must stay on candidate history"
require_present_pattern 'Cache successful candidate baseline' "a successful candidate must become the next cached baseline"
if ! sed -n '/^ warp-ab:/,/^ alert-on-failure:/p' "$workflow" | grep -Eq '^ timeout-minutes:[[:space:]]*180([[:space:]]|$)'; then
echo "invalid performance A/B workflow contract: the cold-cache path must fit both builds, the A/B run, and evidence publication" >&2
exit 1
fi
echo "Performance A/B workflow trust boundary ok." echo "Performance A/B workflow contract ok."