Compare commits

..

8 Commits

Author SHA1 Message Date
overtrue a8cecf6462 test(e2e): register verified Darwin test membership 2026-09-05 20:18:52 +08:00
overtrue 980f3abbd3 chore: merge main PR evidence guidance 2026-09-05 19:00:05 +08:00
overtrue e36650827b chore: integrate shared quick checks for E2E validation 2026-09-05 18:59:07 +08:00
overtrue 09c8e10d5e feat(test): verify the E2E server build and source identity 2026-09-05 18:58:01 +08:00
overtrue 0ff03c596c chore: merge main after ECStore compile repair 2026-09-05 18:46:20 +08:00
overtrue a74919db8e fix(ci): reject dependencies on required quick checks 2026-09-05 18:17:13 +08:00
overtrue e77c6f0ca5 fix(ci): install actionlint from its verified release 2026-09-05 17:42:14 +08:00
overtrue 3149411943 fix(ci): share quick checks and lint workflows 2026-09-05 17:37:20 +08:00
148 changed files with 3028 additions and 16740 deletions
+1 -1
View File
@@ -1,2 +1,2 @@
sha256-darwin=a881fd7d3f5cb94654221ca85b8b30cce1b95e608824a55a15339cbc294e6d34 sha256-darwin=85ee9b9e66e916dbf51857298637b6bcd1c23c2b066aaa4c171ce961c2d002b5
sha256-linux=a2933d83dfe74ffa03410a0959333a1c48288b8469ca9f17273d449d7510c24b sha256-linux=a2933d83dfe74ffa03410a0959333a1c48288b8469ca9f17273d449d7510c24b
+2 -3
View File
@@ -3,10 +3,9 @@
.NOTPARALLEL: pre-commit pre-pr dev-check .NOTPARALLEL: pre-commit pre-pr dev-check
.PHONY: setup-hooks .PHONY: setup-hooks
setup-hooks: ## Install the configured pre-commit hooks setup-hooks: ## Set up git hooks
@echo "🔧 Setting up git hooks..." @echo "🔧 Setting up git hooks..."
pre-commit validate-config chmod +x .git/hooks/pre-commit
pre-commit install
@echo "✅ Git hooks setup complete!" @echo "✅ Git hooks setup complete!"
.PHONY: doc-paths-check .PHONY: doc-paths-check
+1 -1
View File
@@ -38,10 +38,10 @@ script-tests: ## Run shell script tests
./scripts/test_python_bin.sh ./scripts/test_python_bin.sh
./scripts/check_embedded_secrets.sh --self-test ./scripts/check_embedded_secrets.sh --self-test
$(RUSTFS_PYTHON_BIN) ./scripts/check_test_wiring.py --self-test $(RUSTFS_PYTHON_BIN) ./scripts/check_test_wiring.py --self-test
$(RUSTFS_PYTHON_BIN) ./scripts/test_e2e_binary.py
$(RUSTFS_PYTHON_BIN) ./scripts/check_security_coverage.py --self-test $(RUSTFS_PYTHON_BIN) ./scripts/check_security_coverage.py --self-test
$(RUSTFS_PYTHON_BIN) ./scripts/check_scheduled_validation_freshness.py --self-test $(RUSTFS_PYTHON_BIN) ./scripts/check_scheduled_validation_freshness.py --self-test
$(RUSTFS_PYTHON_BIN) ./scripts/test_security_workflow.py $(RUSTFS_PYTHON_BIN) ./scripts/test_security_workflow.py
$(RUSTFS_PYTHON_BIN) ./scripts/test_nightly_candidate.py
$(RUSTFS_PYTHON_BIN) ./scripts/s3-tests/test_report_compat.py $(RUSTFS_PYTHON_BIN) ./scripts/s3-tests/test_report_compat.py
bash -n ./scripts/validate_object_data_cache_cold_stampede.sh bash -n ./scripts/validate_object_data_cache_cold_stampede.sh
$(RUSTFS_PYTHON_BIN) ./scripts/check_object_data_cache_follower_samples.py --self-test $(RUSTFS_PYTHON_BIN) ./scripts/check_object_data_cache_follower_samples.py --self-test
+1 -5
View File
@@ -94,17 +94,13 @@ runs:
shell: bash shell: bash
run: ./scripts/check_embedded_secrets.sh run: ./scripts/check_embedded_secrets.sh
- name: Run script contract tests
shell: bash
run: make script-tests
- name: Check test wiring - name: Check test wiring
shell: bash shell: bash
run: | run: |
python3 ./scripts/check_test_wiring.py --self-test python3 ./scripts/check_test_wiring.py --self-test
python3 ./scripts/test_e2e_binary.py
python3 ./scripts/check_scheduled_validation_freshness.py --self-test python3 ./scripts/check_scheduled_validation_freshness.py --self-test
python3 ./scripts/test_security_workflow.py python3 ./scripts/test_security_workflow.py
python3 ./scripts/test_nightly_candidate.py
python3 ./scripts/check_test_wiring.py python3 ./scripts/check_test_wiring.py
- name: Check no planning docs committed - name: Check no planning docs committed
+14 -10
View File
@@ -582,13 +582,15 @@ jobs:
install-build-packaging-tools: 'false' install-build-packaging-tools: 'false'
- name: Build debug binary - name: Build debug binary
run: cargo build -p rustfs --bins --features e2e-test-hooks run: python3 scripts/e2e_binary.py build --bins --features e2e-test-hooks
- name: Upload debug binary - name: Upload debug binary
uses: actions/upload-artifact@b7c566a772e6b6bfb58ed0dc250532a479d7789f # v6 uses: actions/upload-artifact@b7c566a772e6b6bfb58ed0dc250532a479d7789f # v6
with: with:
name: rustfs-debug-binary name: rustfs-debug-binary
path: target/debug/rustfs path: |
target/debug/rustfs
target/debug/rustfs.e2e.json
if-no-files-found: error if-no-files-found: error
retention-days: 1 retention-days: 1
@@ -620,13 +622,15 @@ jobs:
install-build-packaging-tools: 'false' install-build-packaging-tools: 'false'
- name: Build debug binary with rio-v2 - name: Build debug binary with rio-v2
run: cargo build -p rustfs --bins --features rio-v2,e2e-test-hooks run: python3 scripts/e2e_binary.py build --bins --features rio-v2,e2e-test-hooks
- name: Upload debug binary - name: Upload debug binary
uses: actions/upload-artifact@b7c566a772e6b6bfb58ed0dc250532a479d7789f # v6 uses: actions/upload-artifact@b7c566a772e6b6bfb58ed0dc250532a479d7789f # v6
with: with:
name: rustfs-debug-binary-rio-v2 name: rustfs-debug-binary-rio-v2
path: target/debug/rustfs path: |
target/debug/rustfs
target/debug/rustfs.e2e.json
if-no-files-found: error if-no-files-found: error
retention-days: 1 retention-days: 1
@@ -775,7 +779,7 @@ jobs:
NEXTEST_ARCHIVE: ${{ runner.temp }}/rustfs-e2e-smoke.tar.zst NEXTEST_ARCHIVE: ${{ runner.temp }}/rustfs-e2e-smoke.tar.zst
RUSTFS_E2E_LOG_DIR: ${{ runner.temp }}/rustfs-e2e-smoke-logs RUSTFS_E2E_LOG_DIR: ${{ runner.temp }}/rustfs-e2e-smoke-logs
run: | run: |
cargo nextest run --profile e2e-smoke --archive-file "${NEXTEST_ARCHIVE}" \ python3 scripts/e2e_binary.py run --features e2e-test-hooks -- cargo nextest run --profile e2e-smoke --archive-file "${NEXTEST_ARCHIVE}" \
--status-level all --final-status-level all --failure-output final --status-level all --final-status-level all --failure-output final
- name: Upload e2e smoke diagnostics - name: Upload e2e smoke diagnostics
@@ -811,7 +815,7 @@ jobs:
RUSTFS_TEST_PORT="$(python3 -c 'import socket; s=socket.socket(); s.bind(("127.0.0.1", 0)); print(s.getsockname()[1]); s.close()')" RUSTFS_TEST_PORT="$(python3 -c 'import socket; s=socket.socket(); s.bind(("127.0.0.1", 0)); print(s.getsockname()[1]); s.close()')"
RUSTFS_TEST_PORT="${RUSTFS_TEST_PORT}" \ RUSTFS_TEST_PORT="${RUSTFS_TEST_PORT}" \
RUSTFS_TEST_LOG="${RUN_ROOT}/rustfs.log" \ RUSTFS_TEST_LOG="${RUN_ROOT}/rustfs.log" \
./scripts/e2e-run.sh ./target/debug/rustfs "${RUN_ROOT}/data" python3 scripts/e2e_binary.py run --features e2e-test-hooks -- ./scripts/e2e-run.sh ./target/debug/rustfs "${RUN_ROOT}/data"
- name: Upload test logs - name: Upload test logs
if: failure() if: failure()
@@ -913,7 +917,7 @@ jobs:
# extend that filter, never add ad-hoc e2e jobs here. Reuses the downloaded # extend that filter, never add ad-hoc e2e jobs here. Reuses the downloaded
# debug binary; each test spawns its own rustfs server on a random port. # debug binary; each test spawns its own rustfs server on a random port.
- name: Run e2e full suite - name: Run e2e full suite
run: cargo nextest run --profile e2e-full -p e2e_test run: python3 scripts/e2e_binary.py run --features e2e-test-hooks -- cargo nextest run --profile e2e-full -p e2e_test
- name: Upload junit - name: Upload junit
if: always() if: always()
@@ -974,7 +978,7 @@ jobs:
- name: Run end-to-end tests - name: Run end-to-end tests
run: | run: |
s3s-e2e --version s3s-e2e --version
./scripts/e2e-run.sh ./target/debug/rustfs /tmp/rustfs python3 scripts/e2e_binary.py run --features rio-v2,e2e-test-hooks -- ./scripts/e2e-run.sh ./target/debug/rustfs /tmp/rustfs
- name: Upload test logs - name: Upload test logs
if: failure() if: failure()
@@ -1017,7 +1021,7 @@ jobs:
S3_PORT="${S3_PORT}" \ S3_PORT="${S3_PORT}" \
DATA_ROOT="${RUN_ROOT}" \ DATA_ROOT="${RUN_ROOT}" \
S3TESTS_CONF=artifacts/s3tests-single/s3tests.conf \ S3TESTS_CONF=artifacts/s3tests-single/s3tests.conf \
./scripts/s3-tests/run.sh python3 scripts/e2e_binary.py run --features e2e-test-hooks -- ./scripts/s3-tests/run.sh
- name: Upload s3 test artifacts - name: Upload s3 test artifacts
if: always() if: always()
@@ -1099,7 +1103,7 @@ jobs:
S3_PORT="${S3_PORT}" \ S3_PORT="${S3_PORT}" \
DATA_ROOT="${RUN_ROOT}" \ DATA_ROOT="${RUN_ROOT}" \
S3TESTS_CONF=artifacts/s3tests-single/s3tests.conf \ S3TESTS_CONF=artifacts/s3tests-single/s3tests.conf \
./scripts/s3-tests/run.sh python3 scripts/e2e_binary.py run --features e2e-test-hooks -- ./scripts/s3-tests/run.sh
- name: Upload s3 test artifacts - name: Upload s3 test artifacts
if: always() if: always()
+9 -11
View File
@@ -89,14 +89,10 @@ jobs:
- name: Verify awscurl - name: Verify awscurl
run: test -x "$AWSCURL_PATH" run: test -x "$AWSCURL_PATH"
# Build the rustfs binary once up front. The e2e tests spawn it as a # Build once and carry its source/binary identity into the test invocation.
# child process (crates/e2e_test/src/common.rs) and will build it on
# demand otherwise, but a single explicit build avoids several parallel
# nextest test processes racing to build it at once.
- name: Build rustfs binary - name: Build rustfs binary
run: | run: |
cargo build -p rustfs --bins python3 scripts/e2e_binary.py build --bins
: > target/debug/rustfs.features
- name: Verify replication e2e membership - name: Verify replication e2e membership
env: env:
@@ -108,7 +104,7 @@ jobs:
- name: Run replication e2e nightly suite - name: Run replication e2e nightly suite
env: env:
RUSTFS_E2E_LOG_DIR: ${{ runner.temp }}/rustfs-e2e-repl-nightly-logs RUSTFS_E2E_LOG_DIR: ${{ runner.temp }}/rustfs-e2e-repl-nightly-logs
run: cargo nextest run --profile e2e-repl-nightly -p e2e_test run: python3 scripts/e2e_binary.py run -- cargo nextest run --profile e2e-repl-nightly -p e2e_test
- name: Upload nextest junit report - name: Upload nextest junit report
if: always() if: always()
@@ -144,8 +140,7 @@ jobs:
- name: Build rustfs binary - name: Build rustfs binary
run: | run: |
cargo build -p rustfs --bins --features e2e-test-hooks python3 scripts/e2e_binary.py build --bins --features e2e-test-hooks
: > target/debug/rustfs.features
- name: Verify cluster fault e2e membership - name: Verify cluster fault e2e membership
env: env:
@@ -157,7 +152,7 @@ jobs:
- name: Run cluster fault e2e nightly suite - name: Run cluster fault e2e nightly suite
env: env:
RUSTFS_E2E_LOG_DIR: ${{ runner.temp }}/rustfs-e2e-nightly-logs RUSTFS_E2E_LOG_DIR: ${{ runner.temp }}/rustfs-e2e-nightly-logs
run: cargo nextest run --profile e2e-nightly -p e2e_test run: python3 scripts/e2e_binary.py run --features e2e-test-hooks -- cargo nextest run --profile e2e-nightly -p e2e_test
- name: Upload cluster fault diagnostics - name: Upload cluster fault diagnostics
if: always() if: always()
@@ -198,6 +193,9 @@ jobs:
sudo apt-get install -y -qq iproute2 sudo apt-get install -y -qq iproute2
ss -tn state CLOSE-WAIT >/dev/null ss -tn state CLOSE-WAIT >/dev/null
- name: Build protocol server
run: python3 scripts/e2e_binary.py build --features "$RUSTFS_BUILD_FEATURES"
# The suite owns fixed protocol ports and serializes its internal cases. # The suite owns fixed protocol ports and serializes its internal cases.
- name: Verify protocol e2e membership - name: Verify protocol e2e membership
env: env:
@@ -210,7 +208,7 @@ jobs:
env: env:
RUSTFS_E2E_LOG_DIR: ${{ runner.temp }}/rustfs-protocol-e2e-logs RUSTFS_E2E_LOG_DIR: ${{ runner.temp }}/rustfs-protocol-e2e-logs
run: >- run: >-
cargo nextest run -j 1 --profile e2e-protocols -p e2e_test --no-capture python3 scripts/e2e_binary.py run --features "$RUSTFS_BUILD_FEATURES" -- cargo nextest run -j 1 --profile e2e-protocols -p e2e_test --no-capture
- name: Upload protocol diagnostics - name: Upload protocol diagnostics
if: always() if: always()
+7 -27
View File
@@ -19,9 +19,7 @@ on:
paths: paths:
- ".github/workflows/e2e-upgrade.yml" - ".github/workflows/e2e-upgrade.yml"
- "crates/e2e_test/src/common.rs" - "crates/e2e_test/src/common.rs"
- "crates/e2e_test/src/fake_s3_target/**"
- "crates/e2e_test/src/lib.rs" - "crates/e2e_test/src/lib.rs"
- "crates/e2e_test/src/replication_extension_test.rs"
- "crates/e2e_test/src/upgrade_compatibility_test.rs" - "crates/e2e_test/src/upgrade_compatibility_test.rs"
- "crates/ecstore/**" - "crates/ecstore/**"
- "crates/filemeta/**" - "crates/filemeta/**"
@@ -46,9 +44,9 @@ concurrency:
env: env:
CARGO_TERM_COLOR: always CARGO_TERM_COLOR: always
RUST_BACKTRACE: 1 RUST_BACKTRACE: 1
UPGRADE_SOURCE_VERSION: 1.0.0-rc.5 UPGRADE_SOURCE_VERSION: 1.0.0-rc.2
UPGRADE_SOURCE_ASSET: rustfs-linux-x86_64-gnu-v1.0.0-rc.5.zip UPGRADE_SOURCE_ASSET: rustfs-linux-x86_64-gnu-v1.0.0-rc.2.zip
UPGRADE_SOURCE_SHA256: 3ee8df71e8edcfada533be452c4135868f697bc515460ae97b027313eade7a3d UPGRADE_SOURCE_SHA256: 7c789386bf85278f865b8e0d359bf4edb84d5aa408cc3fa54a18c25ca74cd6e7
jobs: jobs:
upgrade: upgrade:
@@ -57,31 +55,14 @@ jobs:
fail-fast: false fail-fast: false
matrix: matrix:
include: include:
# The two `_from_rc2_` tests keep their names: they assert - name: Direct upgrade from rc.2
# release-independent object contracts and pass unchanged against the
# newer pinned source, so renaming them would only churn history and
# the CI required-check names. UPGRADE_SOURCE_VERSION above is the
# single source of truth for which release they actually run against.
- name: Direct upgrade from the previous release
cache_key: e2e-direct-upgrade cache_key: e2e-direct-upgrade
test: direct_upgrade_from_rc2_preserves_object_contracts test: direct_upgrade_from_rc2_preserves_object_contracts
artifact: direct-upgrade artifact: direct-upgrade
- name: Mixed-version rolling upgrade from the previous release - name: Mixed-version rolling upgrade from rc.2
cache_key: e2e-mixed-version-upgrade cache_key: e2e-mixed-version-upgrade
test: rolling_upgrade_from_rc2_preserves_mixed_version_contracts test: rolling_upgrade_from_rc2_preserves_mixed_version_contracts
artifact: mixed-version-upgrade artifact: mixed-version-upgrade
- name: Bucket configuration survives the upgrade
cache_key: e2e-bucket-config-upgrade
test: direct_upgrade_from_previous_release_preserves_bucket_configuration
artifact: bucket-config-upgrade
- name: Rollback reads current bucket metadata
cache_key: e2e-bucket-config-rollback
test: rollback_to_previous_release_reads_current_bucket_metadata
artifact: bucket-config-rollback
- name: ODM configuration recovery after rc.5 rollback
cache_key: e2e-odm-config-rollback
test: rc5_rollback_requires_restoring_odm_configuration
artifact: odm-config-rollback
runs-on: ubuntu-latest runs-on: ubuntu-latest
timeout-minutes: 60 timeout-minutes: 60
env: env:
@@ -117,12 +98,11 @@ jobs:
- name: Build current RustFS binary - name: Build current RustFS binary
run: | run: |
cargo build --locked -p rustfs --bin rustfs python3 scripts/e2e_binary.py build
: > target/debug/rustfs.features
- name: Run upgrade compatibility test - name: Run upgrade compatibility test
run: | run: |
cargo test --locked -p e2e_test \ python3 scripts/e2e_binary.py run -- cargo test --locked -p e2e_test \
"upgrade_compatibility_test::${{ matrix.test }}" \ "upgrade_compatibility_test::${{ matrix.test }}" \
-- --ignored --exact --nocapture -- --ignored --exact --nocapture
+8 -51
View File
@@ -166,9 +166,8 @@ jobs:
# e.g. https://dl.rustfs.com/artifacts/rustfs/packages/nightly/... . # e.g. https://dl.rustfs.com/artifacts/rustfs/packages/nightly/... .
# Skipped when the R2 secrets are not configured (artifact-only mode). # Skipped when the R2 secrets are not configured (artifact-only mode).
- name: Upload DEB to Cloudflare R2 - name: Upload DEB to Cloudflare R2
id: publish if: env.R2_ACCESS_KEY_ID != ''
env: env:
DEB_FILE: ${{ steps.deb.outputs.deb_file }}
R2_ACCESS_KEY_ID: ${{ secrets.R2_ACCESS_KEY_ID }} R2_ACCESS_KEY_ID: ${{ secrets.R2_ACCESS_KEY_ID }}
R2_SECRET_ACCESS_KEY: ${{ secrets.R2_SECRET_ACCESS_KEY }} R2_SECRET_ACCESS_KEY: ${{ secrets.R2_SECRET_ACCESS_KEY }}
R2_ENDPOINT: ${{ secrets.R2_ENDPOINT }} R2_ENDPOINT: ${{ secrets.R2_ENDPOINT }}
@@ -183,70 +182,28 @@ jobs:
exit 0 exit 0
fi fi
if ! command -v aws >/dev/null 2>&1; then
sudo apt-get update && sudo apt-get install -y -qq awscli
fi
export AWS_ACCESS_KEY_ID="$R2_ACCESS_KEY_ID" export AWS_ACCESS_KEY_ID="$R2_ACCESS_KEY_ID"
export AWS_SECRET_ACCESS_KEY="$R2_SECRET_ACCESS_KEY" export AWS_SECRET_ACCESS_KEY="$R2_SECRET_ACCESS_KEY"
export AWS_DEFAULT_REGION="auto" export AWS_DEFAULT_REGION="auto"
SOURCE_SHA="$(git rev-parse HEAD)" DEB_FILE="${{ steps.deb.outputs.deb_file }}"
if [[ "${SOURCE_SHA}" != "${GITHUB_SHA}" ]]; then
echo "Checkout SHA does not match the nightly build run" >&2
exit 1
fi
DEB_SHA256="$(sha256sum "${DEB_FILE}" | cut -d ' ' -f 1)"
CANDIDATE_KEY="artifacts/rustfs/packages/nightly/runs/${GITHUB_RUN_ID}/${GITHUB_RUN_ATTEMPT}/${DEB_SHA256}/rustfs.deb"
CANDIDATE_URL="https://dl.rustfs.com/${CANDIDATE_KEY}"
# Old AWS CLI models lack conditional PutObject support. Never fall
# back to an overwriting upload for a candidate.
AWS_CLI=aws
if ! "${AWS_CLI}" s3api put-object --generate-cli-skeleton input | jq -e 'has("IfNoneMatch")' >/dev/null; then
sudo apt-get update
sudo apt-get install -y -qq python3-venv
AWS_CLI_DIR="$(mktemp -d "${RUNNER_TEMP}/nightly-awscli.XXXXXX")"
trap 'rm -rf "${AWS_CLI_DIR}"' EXIT
python3 -m venv "${AWS_CLI_DIR}"
"${AWS_CLI_DIR}/bin/python" -m pip install --disable-pip-version-check 'awscli==1.44.79'
AWS_CLI="${AWS_CLI_DIR}/bin/aws"
fi
"${AWS_CLI}" s3api put-object --generate-cli-skeleton input | jq -e 'has("IfNoneMatch")' >/dev/null
"${AWS_CLI}" --version
"${AWS_CLI}" s3api put-object --bucket "${R2_BUCKET}" --key "${CANDIDATE_KEY}" \
--body "${DEB_FILE}" --if-none-match '*' --endpoint-url "${R2_ENDPOINT}"
PUBLISHED_SHA256="$(curl -fsSL --retry 3 --connect-timeout 15 --max-time 300 "${CANDIDATE_URL}" | sha256sum | cut -d ' ' -f 1)"
if [[ "${PUBLISHED_SHA256}" != "${DEB_SHA256}" ]]; then
echo "Published candidate checksum does not match the built package" >&2
exit 1
fi
R2_PREFIX="s3://${R2_BUCKET}/artifacts/rustfs/packages/nightly/" R2_PREFIX="s3://${R2_BUCKET}/artifacts/rustfs/packages/nightly/"
echo "📤 Uploading ${DEB_FILE} to ${R2_PREFIX}" echo "📤 Uploading ${DEB_FILE} to ${R2_PREFIX}"
"${AWS_CLI}" s3 cp "${DEB_FILE}" "${R2_PREFIX}" --endpoint-url "$R2_ENDPOINT" --only-show-errors aws s3 cp "${DEB_FILE}" "${R2_PREFIX}" --endpoint-url "$R2_ENDPOINT" --only-show-errors
# Stable "latest" alias so tests can fetch the newest nightly # Stable "latest" alias so tests can fetch the newest nightly
# without knowing today's date. # without knowing today's date.
echo "📤 Uploading latest alias" echo "📤 Uploading latest alias"
"${AWS_CLI}" s3 cp "${DEB_FILE}" "${R2_PREFIX}rustfs-nightly-latest.deb" \ aws s3 cp "${DEB_FILE}" "${R2_PREFIX}rustfs-nightly-latest.deb" \
--endpoint-url "$R2_ENDPOINT" --only-show-errors --endpoint-url "$R2_ENDPOINT" --only-show-errors
echo "✅ R2 upload complete" echo "✅ R2 upload complete"
CANDIDATE_FILE="${RUNNER_TEMP}/nightly-candidate-${GITHUB_RUN_ID}-${GITHUB_RUN_ATTEMPT}.json"
jq -n --arg source_sha "${SOURCE_SHA}" \
--argjson build_run_id "${GITHUB_RUN_ID}" --argjson build_run_attempt "${GITHUB_RUN_ATTEMPT}" \
--arg package_url "${CANDIDATE_URL}" --arg package_sha256 "${DEB_SHA256}" \
'{schema: 1, source_sha: $source_sha, build_run_id: $build_run_id, build_run_attempt: $build_run_attempt, package_url: $package_url, package_sha256: $package_sha256}' \
> "${CANDIDATE_FILE}"
echo "candidate_file=${CANDIDATE_FILE}" >> "${GITHUB_OUTPUT}"
- name: Upload nightly candidate manifest
if: ${{ steps.publish.outputs.candidate_file != '' }}
uses: actions/upload-artifact@b7c566a772e6b6bfb58ed0dc250532a479d7789f # v6
with:
name: nightly-candidate-${{ github.run_id }}-${{ github.run_attempt }}
path: ${{ steps.publish.outputs.candidate_file }}
if-no-files-found: error
# Live-Vault lane for the rustfs-kms suite (rustfs/backlog#1774). # Live-Vault lane for the rustfs-kms suite (rustfs/backlog#1774).
# #
# RUSTFS_KMS_VAULT_TOKEN is the single switch that adds the Vault KV2 and # RUSTFS_KMS_VAULT_TOKEN is the single switch that adds the Vault KV2 and
@@ -132,7 +132,7 @@ jobs:
s3api create-bucket --bucket "${RUSTFS_ODM_INTEROP_BUCKET}" s3api create-bucket --bucket "${RUSTFS_ODM_INTEROP_BUCKET}"
- name: Build the RustFS binary under test - name: Build the RustFS binary under test
run: cargo build --locked -p rustfs --bins run: python3 scripts/e2e_binary.py build --bins
# The lane selects tests by module, so a rename would quietly shrink it. # The lane selects tests by module, so a rename would quietly shrink it.
# The committed digest in .config/e2e-odm-interop-selection.txt fails # The committed digest in .config/e2e-odm-interop-selection.txt fails
@@ -143,7 +143,7 @@ jobs:
python3 ./scripts/check_test_wiring.py --check-profile e2e-odm-interop "${NEXTEST_LISTING}" python3 ./scripts/check_test_wiring.py --check-profile e2e-odm-interop "${NEXTEST_LISTING}"
- name: Run the interop cases against MinIO - name: Run the interop cases against MinIO
run: cargo nextest run --profile e2e-odm-interop -p e2e_test --no-tests=fail run: python3 scripts/e2e_binary.py run -- cargo nextest run --profile e2e-odm-interop -p e2e_test --no-tests=fail
- name: Build the MinIO interop report - name: Build the MinIO interop report
if: always() if: always()
@@ -251,7 +251,7 @@ jobs:
- name: Build the RustFS binary under test - name: Build the RustFS binary under test
if: steps.credentials.outputs.present == 'true' if: steps.credentials.outputs.present == 'true'
run: cargo build --locked -p rustfs --bins run: python3 scripts/e2e_binary.py build --bins
# A filterset that matches nothing is valid, so the count is asserted # A filterset that matches nothing is valid, so the count is asserted
# rather than inferred from a green run. # rather than inferred from a green run.
@@ -272,7 +272,7 @@ jobs:
- name: Run the three-case minimum - name: Run the three-case minimum
if: steps.credentials.outputs.present == 'true' if: steps.credentials.outputs.present == 'true'
run: | run: |
cargo nextest run --profile e2e-odm-interop -p e2e_test \ python3 scripts/e2e_binary.py run -- cargo nextest run --profile e2e-odm-interop -p e2e_test \
-E "${CLOUD_CASE_FILTER}" --no-tests=fail -E "${CLOUD_CASE_FILTER}" --no-tests=fail
- name: Build the ${{ matrix.provider }} interop report - name: Build the ${{ matrix.provider }} interop report
+15 -2
View File
@@ -14,8 +14,8 @@
# Functional chain driver: runs the ten functional suites in a fixed order # Functional chain driver: runs the ten functional suites in a fixed order
# (upgrade -> s3 -> kms -> tier -> storage -> heal -> pool -> security -> # (upgrade -> s3 -> kms -> tier -> storage -> heal -> pool -> security ->
# replication -> performance). Each suite attempts the next handoff even # replication, with performance on its own runner in parallel) and guarantees
# when its tests fail. # the chain keeps moving even when individual suites fail.
# #
# Each suite workflow can still be dispatched standalone (workflow_dispatch); # Each suite workflow can still be dispatched standalone (workflow_dispatch);
# only chain-triggered runs forward to the next suite via repository_dispatch, # only chain-triggered runs forward to the next suite via repository_dispatch,
@@ -59,3 +59,16 @@ jobs:
gh api --method POST repos/rustfs/rustfs/dispatches \ gh api --method POST repos/rustfs/rustfs/dispatches \
-f event_type='rustfs-chain-upgrade' \ -f event_type='rustfs-chain-upgrade' \
-F 'client_payload[from_suite]=nightly-build' -F 'client_payload[from_suite]=nightly-build'
- name: Dispatch performance suite (parallel, own runner)
env:
GH_TOKEN: ${{ secrets.PF_TESTING_GH_TOKEN }}
run: |
set -euo pipefail
if [ -z "${GH_TOKEN:-}" ]; then
echo "PF_TESTING_GH_TOKEN is not configured; cannot dispatch performance" >&2
exit 1
fi
gh api --method POST repos/rustfs/rustfs/dispatches \
-f event_type='rustfs-chain-performance' \
-F 'client_payload[from_suite]=nightly-build'
+37 -70
View File
@@ -54,26 +54,14 @@ env:
jobs: jobs:
heal-test: heal-test:
runs-on: smoke-testing runs-on: smoke-testing
# Requirement: a failing suite must not fail the workflow; failures
# are filed to rustfs/backlog and the chain continues.
continue-on-error: true
timeout-minutes: 480 timeout-minutes: 480
# Standalone manual run, or one link of the nightly functional chain # Standalone manual run, or one link of the nightly functional chain
# (storage -> heal -> pool). Pool expansion no longer re-runs heal. # (storage -> heal -> pool). Pool expansion no longer re-runs heal.
if: ${{ github.event_name == 'workflow_dispatch' || github.event_name == 'repository_dispatch' }} if: ${{ github.event_name == 'workflow_dispatch' || github.event_name == 'repository_dispatch' }}
steps: steps:
- name: Initialize functional evidence
id: evidence
run: |
set -euo pipefail
umask 077
FUNCTIONAL_ARTIFACTS_DIR="${RUNNER_TEMP}/rustfs-heal-${GITHUB_RUN_ID}-${GITHUB_RUN_ATTEMPT}"
mkdir -- "${FUNCTIONAL_ARTIFACTS_DIR}" "${FUNCTIONAL_ARTIFACTS_DIR}-scratch"
{
printf 'FUNCTIONAL_ARTIFACTS_DIR=%s\n' "${FUNCTIONAL_ARTIFACTS_DIR}"
printf 'LOG_FILE=%s/suite.log\n' "${FUNCTIONAL_ARTIFACTS_DIR}"
printf 'RUSTFS_WARP_LOG_FILE=%s/warp.log\n' "${FUNCTIONAL_ARTIFACTS_DIR}"
printf 'REPORT_FILE=%s/report.md\n' "${FUNCTIONAL_ARTIFACTS_DIR}"
printf 'TMPDIR=%s-scratch\n' "${FUNCTIONAL_ARTIFACTS_DIR}"
} >> "${GITHUB_ENV}"
# auto-testing is private: clone it with the dedicated PF token (not # auto-testing is private: clone it with the dedicated PF token (not
# GITHUB_TOKEN) and retry transient GitHub/network failures. # GITHUB_TOKEN) and retry transient GitHub/network failures.
- name: Checkout auto-testing scripts (with retry) - name: Checkout auto-testing scripts (with retry)
@@ -129,7 +117,7 @@ jobs:
else else
ARGS+=(--package-url "${{ env.RUSTFS_NIGHTLY_PACKAGE_URL }}") ARGS+=(--package-url "${{ env.RUSTFS_NIGHTLY_PACKAGE_URL }}")
fi fi
./auto-testing/rustfs_heal_test.sh "${ARGS[@]}" --log-file "${LOG_FILE}" ./auto-testing/rustfs_heal_test.sh "${ARGS[@]}"
- name: Preflight checks - name: Preflight checks
run: | run: |
@@ -139,7 +127,7 @@ jobs:
else else
ARGS+=(--package-url "${{ env.RUSTFS_NIGHTLY_PACKAGE_URL }}") ARGS+=(--package-url "${{ env.RUSTFS_NIGHTLY_PACKAGE_URL }}")
fi fi
./auto-testing/rustfs_heal_test.sh "${ARGS[@]}" --log-file "${LOG_FILE}" ./auto-testing/rustfs_heal_test.sh "${ARGS[@]}"
- name: Run heal test (write -> outage -> heal -> verify) - name: Run heal test (write -> outage -> heal -> verify)
id: test id: test
@@ -149,10 +137,13 @@ jobs:
--endpoint "${{ env.RUSTFS_API_ENDPOINT }}" \ --endpoint "${{ env.RUSTFS_API_ENDPOINT }}" \
--stop-node-gb "${{ inputs.stop_node_gb || '15' }}" \ --stop-node-gb "${{ inputs.stop_node_gb || '15' }}" \
--warp-stop-gb "${{ inputs.warp_stop_gb || '40' }}" \ --warp-stop-gb "${{ inputs.warp_stop_gb || '40' }}" \
--log-file "${LOG_FILE}" --log-file /tmp/rustfs-heal-test.log
- name: Generate report - name: Generate report
if: ${{ always() && steps.evidence.outcome == 'success' }} if: always()
env:
LOG_FILE: /tmp/rustfs-heal-test.log
REPORT_FILE: /tmp/rustfs-heal-report.md
run: | run: |
set -euo pipefail set -euo pipefail
PACKAGE_URL='${{ inputs.package_url }}' PACKAGE_URL='${{ inputs.package_url }}'
@@ -161,9 +152,8 @@ jobs:
else else
PACKAGE_SOURCE="${RUSTFS_NIGHTLY_PACKAGE_URL}" PACKAGE_SOURCE="${RUSTFS_NIGHTLY_PACKAGE_URL}"
fi fi
STEPS_TABLE="${FUNCTIONAL_ARTIFACTS_DIR}/steps.md" STEPS_TABLE="/tmp/rustfs-heal-steps.md"
CASE_RESULT=success python3 - "${LOG_FILE}" "${STEPS_TABLE}" <<'PY'
python3 - "${LOG_FILE}" "${STEPS_TABLE}" <<'PY' || CASE_RESULT=failure
import re import re
import sys import sys
@@ -175,7 +165,6 @@ jobs:
steps = {} steps = {}
order = [] order = []
status_rank = {'SKIP': 0, 'PASS': 1, 'FAIL': 2}
version = None version = None
version_node = None version_node = None
verdict = None verdict = None
@@ -189,15 +178,14 @@ jobs:
n, desc, status = m.group(1), m.group(2), m.group(3) n, desc, status = m.group(1), m.group(2), m.group(3)
if n not in steps: if n not in steps:
order.append(n) order.append(n)
if n not in steps or status_rank[status] > status_rank[steps[n][1]]: steps[n] = (desc, status) # later lines win (fail after pass)
steps[n] = (desc, status)
continue continue
m = ver_re.match(line) m = ver_re.match(line)
if m: if m:
version, version_node = m.group(1), m.group(2) version, version_node = m.group(1), m.group(2)
continue continue
m = result_re.match(line) m = result_re.match(line)
if m and verdict != 'FAIL': if m:
verdict, verdict_detail = m.group(1), m.group(2) verdict, verdict_detail = m.group(1), m.group(2)
except FileNotFoundError: except FileNotFoundError:
pass pass
@@ -217,43 +205,30 @@ jobs:
out.write(f'| {n} | {desc} | {status} |\n') out.write(f'| {n} | {desc} | {status} |\n')
if not order: if not order:
out.write('| - | - | NOT RUN (no step result lines found) |\n') out.write('| - | - | NOT RUN (no step result lines found) |\n')
complete = set(steps) == {str(n) for n in range(1, 8)}
sys.exit(0 if complete and verdict != 'FAIL' and all(status == 'PASS' for _, status in steps.values()) else 1)
PY PY
RESULT=failure
if [ '${{ steps.test.outcome }}' = 'success' ] && [ "${CASE_RESULT}" = 'success' ]; then
RESULT=success
fi
{ {
echo "# RustFS heal test report" echo "# RustFS heal test report"
echo "" echo ""
echo "- Run: ${{ github.server_url }}/${{ github.repository }}/actions/runs/${{ github.run_id }}" echo "- Run: ${{ github.server_url }}/${{ github.repository }}/actions/runs/${{ github.run_id }}"
echo "- Attempt: ${GITHUB_RUN_ATTEMPT}"
echo "- Workflow Commit: ${GITHUB_SHA}"
echo "- Trigger: ${{ github.event_name }}" echo "- Trigger: ${{ github.event_name }}"
echo "- Package: ${PACKAGE_SOURCE}" echo "- Package: ${PACKAGE_SOURCE}"
echo "- Test Step Outcome: ${RESULT}" echo "- Test Step Outcome: ${{ steps.test.outcome }}"
echo "- Suite Step Outcome: ${{ steps.test.outcome }}"
echo "" echo ""
if [ "${RESULT}" = "success" ]; then cat "${STEPS_TABLE}" || true
cat "${STEPS_TABLE}" echo ""
echo "" echo "## Log tail"
echo "## Log tail" echo '```text'
echo '```text' tail -n 200 "${LOG_FILE}" || true
tail -n 200 "${LOG_FILE}" echo '```'
echo '```'
else
echo "The suite or evidence validation failed. See this run's artifact for partial step results and suite.log."
fi
} | tee "${REPORT_FILE}" } | tee "${REPORT_FILE}"
cat "${REPORT_FILE}" >> "${GITHUB_STEP_SUMMARY}" cat "${REPORT_FILE}" >> "${GITHUB_STEP_SUMMARY}"
[ "${RESULT}" = "success" ]
- name: Upload functional report to dashboard - name: Upload functional report to dashboard
if: ${{ always() && steps.evidence.outcome == 'success' }} if: always()
continue-on-error: true continue-on-error: true
env: env:
GH_TOKEN: ${{ env.PF_TESTING_GH_TOKEN }} GH_TOKEN: ${{ env.PF_TESTING_GH_TOKEN }}
REPORT_FILE: /tmp/rustfs-heal-report.md
SUITE: heal SUITE: heal
run: | run: |
set -euo pipefail set -euo pipefail
@@ -263,32 +238,28 @@ jobs:
fi fi
DATE="$(date -u +%Y-%m-%d)" DATE="$(date -u +%Y-%m-%d)"
REPORT_PATH="functional-reports/${SUITE}/${DATE}.md" REPORT_PATH="functional-reports/${SUITE}/${DATE}.md"
# Base64-encode the report into a temp file and feed it to jq via CONTENT="$(python3 -c 'import base64,sys;print(base64.b64encode(open(sys.argv[1],"rb").read()).decode())' "${REPORT_FILE}")"
# --rawfile: large reports (e.g. pool) exceed the OS argv limit and
# make `jq --arg content "${CONTENT}"` fail with "Argument list too long".
B64_FILE="$(mktemp)"
python3 -c 'import base64,sys;print(base64.b64encode(open(sys.argv[1],"rb").read()).decode())' "${REPORT_FILE}" > "${B64_FILE}"
SHA="$(gh api "repos/rustfs/dashboard/contents/${REPORT_PATH}" -q '.sha' 2>/dev/null || true)" SHA="$(gh api "repos/rustfs/dashboard/contents/${REPORT_PATH}" -q '.sha' 2>/dev/null || true)"
if [ -n "${SHA}" ]; then if [ -n "${SHA}" ]; then
jq -n --arg msg "report(${SUITE}): ${DATE}" --rawfile content "${B64_FILE}" --arg sha "${SHA}" \ jq -n --arg msg "report(${SUITE}): ${DATE}" --arg content "${CONTENT}" --arg sha "${SHA}" \
'{message:$msg, content:($content|rtrimstr("\n")), sha:$sha}' \ '{message:$msg, content:$content, sha:$sha}' \
| gh api --method PUT "repos/rustfs/dashboard/contents/${REPORT_PATH}" --input - >/dev/null | gh api --method PUT "repos/rustfs/dashboard/contents/${REPORT_PATH}" --input - >/dev/null
else else
jq -n --arg msg "report(${SUITE}): ${DATE}" --rawfile content "${B64_FILE}" \ jq -n --arg msg "report(${SUITE}): ${DATE}" --arg content "${CONTENT}" \
'{message:$msg, content:($content|rtrimstr("\n"))}' \ '{message:$msg, content:$content}' \
| gh api --method PUT "repos/rustfs/dashboard/contents/${REPORT_PATH}" --input - >/dev/null | gh api --method PUT "repos/rustfs/dashboard/contents/${REPORT_PATH}" --input - >/dev/null
fi fi
rm -f "${B64_FILE}"
- name: File failure issue in rustfs/backlog - name: File failure issue in rustfs/backlog
if: ${{ always() && (failure() || steps.test.outcome == 'failure' || steps.test.outcome == 'cancelled') }} if: ${{ always() && (failure() || steps.test.outcome == 'failure' || steps.test.outcome == 'cancelled') }}
continue-on-error: true continue-on-error: true
env: env:
GH_TOKEN: ${{ secrets.PF_TESTING_GH_TOKEN }} GH_TOKEN: ${{ secrets.PF_TESTING_GH_TOKEN }}
EVIDENCE_OUTCOME: ${{ steps.evidence.outcome }}
SUITE: 'heal' SUITE: 'heal'
SUITE_LABEL: 'Heal' SUITE_LABEL: 'Heal'
RUN_URL: ${{ github.server_url }}/${{ github.repository }}/actions/runs/${{ github.run_id }} RUN_URL: ${{ github.server_url }}/${{ github.repository }}/actions/runs/${{ github.run_id }}
REPORT_FILE: '/tmp/rustfs-heal-report.md'
LOG_FILE: '/tmp/rustfs-heal-test.log'
run: | run: |
set -euo pipefail set -euo pipefail
if [ -z "${GH_TOKEN:-}" ]; then if [ -z "${GH_TOKEN:-}" ]; then
@@ -316,16 +287,14 @@ jobs:
echo "" echo ""
echo "- Suite: \`${SUITE}\`" echo "- Suite: \`${SUITE}\`"
echo "- Run: ${RUN_URL}" echo "- Run: ${RUN_URL}"
echo "- Attempt: ${GITHUB_RUN_ATTEMPT}"
echo "- Workflow Commit: ${GITHUB_SHA}"
echo "- Trigger: ${GITHUB_EVENT_NAME}" echo "- Trigger: ${GITHUB_EVENT_NAME}"
echo "- Date: $(date -u +%Y-%m-%d)" echo "- Date: $(date -u +%Y-%m-%d)"
echo "" echo ""
echo "## Report (errors and symptoms)" echo "## Report (errors and symptoms)"
echo "" echo ""
if [ "${EVIDENCE_OUTCOME}" = "success" ] && [ -s "${REPORT_FILE}" ]; then if [ -s "${REPORT_FILE}" ]; then
redact < "${REPORT_FILE}" redact < "${REPORT_FILE}"
elif [ "${EVIDENCE_OUTCOME}" = "success" ] && [ -s "${LOG_FILE:-}" ]; then elif [ -s "${LOG_FILE:-}" ]; then
echo "(report file missing; log tail below)" echo "(report file missing; log tail below)"
echo "" echo ""
tail -n 200 "${LOG_FILE}" | redact tail -n 200 "${LOG_FILE}" | redact
@@ -341,16 +310,14 @@ jobs:
echo "filed backlog issue for suite ${SUITE}" echo "filed backlog issue for suite ${SUITE}"
- name: Upload test logs - name: Upload test logs
if: ${{ always() && steps.evidence.outcome == 'success' }} if: always()
uses: actions/upload-artifact@b7c566a772e6b6bfb58ed0dc250532a479d7789f # v6 uses: actions/upload-artifact@b7c566a772e6b6bfb58ed0dc250532a479d7789f # v6
with: with:
name: rustfs-heal-test-${{ github.run_id }}-${{ github.run_attempt }} name: rustfs-heal-test-${{ github.run_id }}
path: | path: |
${{ env.FUNCTIONAL_ARTIFACTS_DIR }}/report.md /tmp/rustfs-heal-test*.log
${{ env.FUNCTIONAL_ARTIFACTS_DIR }}/suite.log /tmp/rustfs-warp.*.log
${{ env.FUNCTIONAL_ARTIFACTS_DIR }}/warp.log if-no-files-found: warn
${{ env.FUNCTIONAL_ARTIFACTS_DIR }}/steps.md
if-no-files-found: error
- name: Cleanup environment (after) - name: Cleanup environment (after)
if: ${{ always() && inputs.cleanup_after != 'false' }} if: ${{ always() && inputs.cleanup_after != 'false' }}
+81 -64
View File
@@ -49,28 +49,10 @@ env:
jobs: jobs:
kms-test: kms-test:
runs-on: smoke-testing runs-on: smoke-testing
continue-on-error: true
timeout-minutes: 420 timeout-minutes: 420
if: ${{ github.event_name == 'workflow_dispatch' || github.event_name == 'repository_dispatch' }} if: ${{ github.event_name == 'workflow_dispatch' || github.event_name == 'repository_dispatch' }}
steps: steps:
- name: Checkout repository (for report parser)
uses: actions/checkout@9c091bb21b7c1c1d1991bb908d89e4e9dddfe3e0 # v7
with:
persist-credentials: false
- name: Initialize functional evidence
id: evidence
run: |
set -euo pipefail
umask 077
FUNCTIONAL_ARTIFACTS_DIR="${RUNNER_TEMP}/rustfs-kms-${GITHUB_RUN_ID}-${GITHUB_RUN_ATTEMPT}"
mkdir -- "${FUNCTIONAL_ARTIFACTS_DIR}" "${FUNCTIONAL_ARTIFACTS_DIR}-scratch"
{
printf 'FUNCTIONAL_ARTIFACTS_DIR=%s\n' "${FUNCTIONAL_ARTIFACTS_DIR}"
printf 'LOG_FILE=%s/suite.log\n' "${FUNCTIONAL_ARTIFACTS_DIR}"
printf 'REPORT_FILE=%s/report.md\n' "${FUNCTIONAL_ARTIFACTS_DIR}"
printf 'TMPDIR=%s-scratch\n' "${FUNCTIONAL_ARTIFACTS_DIR}"
} >> "${GITHUB_ENV}"
# auto-testing is private: clone it with the dedicated PF token (not # auto-testing is private: clone it with the dedicated PF token (not
# GITHUB_TOKEN) and retry transient GitHub/network failures. # GITHUB_TOKEN) and retry transient GitHub/network failures.
- name: Checkout auto-testing scripts (with retry) - name: Checkout auto-testing scripts (with retry)
@@ -127,6 +109,9 @@ jobs:
- name: Run KMS suite - name: Run KMS suite
id: test id: test
continue-on-error: true
env:
LOG_FILE: /tmp/rustfs-kms.log
run: | run: |
set -euo pipefail set -euo pipefail
chmod +x auto-testing/rustfs-kms-test.sh chmod +x auto-testing/rustfs-kms-test.sh
@@ -156,7 +141,10 @@ jobs:
./auto-testing/rustfs-kms-test.sh "${ARGS[@]}" ./auto-testing/rustfs-kms-test.sh "${ARGS[@]}"
- name: Generate report - name: Generate report
if: ${{ always() && steps.evidence.outcome == 'success' }} if: always()
env:
LOG_FILE: /tmp/rustfs-kms.log
REPORT_FILE: /tmp/rustfs-kms-report.md
run: | run: |
set -euo pipefail set -euo pipefail
PACKAGE_URL='${{ inputs.package_url }}' PACKAGE_URL='${{ inputs.package_url }}'
@@ -168,43 +156,79 @@ jobs:
else else
PACKAGE_SOURCE="${RUSTFS_NIGHTLY_PACKAGE_URL}" PACKAGE_SOURCE="${RUSTFS_NIGHTLY_PACKAGE_URL}"
fi fi
CASE_TABLE="${FUNCTIONAL_ARTIFACTS_DIR}/cases.md" CASE_TABLE="/tmp/rustfs-kms-cases.md"
CASE_RESULT=success python3 - "${LOG_FILE}" "${CASE_TABLE}" <<'PY'
python3 scripts/functional_case_report.py "${LOG_FILE}" "${CASE_TABLE}" || CASE_RESULT=failure import re
RESULT=failure import sys
if [ '${{ steps.test.outcome }}' = 'success' ] && [ "${CASE_RESULT}" = 'success' ]; then
RESULT=success log_file, out_file = sys.argv[1], sys.argv[2]
fi ansi = re.compile(r'\x1b\[[0-9;]*m')
start_re = re.compile(r'^---\s+([A-Z]+-[0-9]+)\s+(.+?)\s+---$')
done_re = re.compile(r'^\[(PASS|FAIL|UNSUPPORTED)\]\s+([A-Z]+-[0-9]+)\b')
rows = []
index = {}
try:
with open(log_file, 'r', encoding='utf-8', errors='replace') as fh:
for raw in fh:
line = ansi.sub('', raw).strip()
m = start_re.match(line)
if m:
case_id, name = m.group(1), m.group(2)
if case_id not in index:
index[case_id] = len(rows)
rows.append([case_id, name, 'RUNNING'])
continue
m = done_re.match(line)
if m:
status, case_id = m.group(1), m.group(2)
if case_id in index:
rows[index[case_id]][2] = status
else:
rows.append([case_id, case_id, status])
index[case_id] = len(rows) - 1
except FileNotFoundError:
rows = []
counts = {'PASS': 0, 'FAIL': 0, 'UNSUPPORTED': 0, 'RUNNING': 0}
for _, _, status in rows:
counts[status] = counts.get(status, 0) + 1
with open(out_file, 'w', encoding='utf-8') as out:
out.write('## Case Summary\n\n')
out.write(f"- Total: {len(rows)}\\n")
out.write(f"- PASS: {counts.get('PASS', 0)}\\n")
out.write(f"- FAIL: {counts.get('FAIL', 0)}\\n")
out.write(f"- UNSUPPORTED: {counts.get('UNSUPPORTED', 0)}\\n")
out.write('\\n')
out.write('| Case | Name | Status |\\n')
out.write('| --- | --- | --- |\\n')
for case_id, name, status in rows:
out.write(f'| {case_id} | {name} | {status} |\\n')
PY
{ {
echo "# RustFS KMS test report" echo "# RustFS KMS test report"
echo "" echo ""
echo "- Run: ${{ github.server_url }}/${{ github.repository }}/actions/runs/${{ github.run_id }}" echo "- Run: ${{ github.server_url }}/${{ github.repository }}/actions/runs/${{ github.run_id }}"
echo "- Attempt: ${GITHUB_RUN_ATTEMPT}"
echo "- Workflow Commit: ${GITHUB_SHA}"
echo "- Trigger: ${{ github.event_name }}" echo "- Trigger: ${{ github.event_name }}"
echo "- Package: ${PACKAGE_SOURCE}" echo "- Package: ${PACKAGE_SOURCE}"
echo "- Test Step Outcome: ${RESULT}" echo "- Test Step Outcome: ${{ steps.test.outcome }}"
echo "- Suite Step Outcome: ${{ steps.test.outcome }}"
echo "" echo ""
if [ "${RESULT}" = "success" ]; then cat "${CASE_TABLE}" || true
cat "${CASE_TABLE}" echo ""
echo "" echo "## Log tail"
echo "## Log tail" echo '```text'
echo '```text' tail -n 200 "${LOG_FILE}" || true
tail -n 200 "${LOG_FILE}" echo '```'
echo '```'
else
echo "The suite or evidence validation failed. See this run's artifact for partial case results and suite.log."
fi
} | tee "${REPORT_FILE}" } | tee "${REPORT_FILE}"
cat "${REPORT_FILE}" >> "${GITHUB_STEP_SUMMARY}" cat "${REPORT_FILE}" >> "${GITHUB_STEP_SUMMARY}"
[ "${RESULT}" = "success" ]
- name: Upload functional report to dashboard - name: Upload functional report to dashboard
if: ${{ always() && steps.evidence.outcome == 'success' }} if: always()
continue-on-error: true continue-on-error: true
env: env:
GH_TOKEN: ${{ env.PF_TESTING_GH_TOKEN }} GH_TOKEN: ${{ env.PF_TESTING_GH_TOKEN }}
REPORT_FILE: /tmp/rustfs-kms-report.md
SUITE: kms SUITE: kms
run: | run: |
set -euo pipefail set -euo pipefail
@@ -214,32 +238,28 @@ jobs:
fi fi
DATE="$(date -u +%Y-%m-%d)" DATE="$(date -u +%Y-%m-%d)"
REPORT_PATH="functional-reports/${SUITE}/${DATE}.md" REPORT_PATH="functional-reports/${SUITE}/${DATE}.md"
# Base64-encode the report into a temp file and feed it to jq via CONTENT="$(python3 -c 'import base64,sys;print(base64.b64encode(open(sys.argv[1],"rb").read()).decode())' "${REPORT_FILE}")"
# --rawfile: large reports (e.g. pool) exceed the OS argv limit and
# make `jq --arg content "${CONTENT}"` fail with "Argument list too long".
B64_FILE="$(mktemp)"
python3 -c 'import base64,sys;print(base64.b64encode(open(sys.argv[1],"rb").read()).decode())' "${REPORT_FILE}" > "${B64_FILE}"
SHA="$(gh api "repos/rustfs/dashboard/contents/${REPORT_PATH}" -q '.sha' 2>/dev/null || true)" SHA="$(gh api "repos/rustfs/dashboard/contents/${REPORT_PATH}" -q '.sha' 2>/dev/null || true)"
if [ -n "${SHA}" ]; then if [ -n "${SHA}" ]; then
jq -n --arg msg "report(${SUITE}): ${DATE}" --rawfile content "${B64_FILE}" --arg sha "${SHA}" \ jq -n --arg msg "report(${SUITE}): ${DATE}" --arg content "${CONTENT}" --arg sha "${SHA}" \
'{message:$msg, content:($content|rtrimstr("\n")), sha:$sha}' \ '{message:$msg, content:$content, sha:$sha}' \
| gh api --method PUT "repos/rustfs/dashboard/contents/${REPORT_PATH}" --input - >/dev/null | gh api --method PUT "repos/rustfs/dashboard/contents/${REPORT_PATH}" --input - >/dev/null
else else
jq -n --arg msg "report(${SUITE}): ${DATE}" --rawfile content "${B64_FILE}" \ jq -n --arg msg "report(${SUITE}): ${DATE}" --arg content "${CONTENT}" \
'{message:$msg, content:($content|rtrimstr("\n"))}' \ '{message:$msg, content:$content}' \
| gh api --method PUT "repos/rustfs/dashboard/contents/${REPORT_PATH}" --input - >/dev/null | gh api --method PUT "repos/rustfs/dashboard/contents/${REPORT_PATH}" --input - >/dev/null
fi fi
rm -f "${B64_FILE}"
- name: File failure issue in rustfs/backlog - name: File failure issue in rustfs/backlog
if: ${{ always() && (failure() || steps.test.outcome == 'failure' || steps.test.outcome == 'cancelled') }} if: ${{ always() && (failure() || steps.test.outcome == 'failure' || steps.test.outcome == 'cancelled') }}
continue-on-error: true continue-on-error: true
env: env:
GH_TOKEN: ${{ secrets.PF_TESTING_GH_TOKEN }} GH_TOKEN: ${{ secrets.PF_TESTING_GH_TOKEN }}
EVIDENCE_OUTCOME: ${{ steps.evidence.outcome }}
SUITE: 'kms' SUITE: 'kms'
SUITE_LABEL: 'KMS' SUITE_LABEL: 'KMS'
RUN_URL: ${{ github.server_url }}/${{ github.repository }}/actions/runs/${{ github.run_id }} RUN_URL: ${{ github.server_url }}/${{ github.repository }}/actions/runs/${{ github.run_id }}
REPORT_FILE: '/tmp/rustfs-kms-report.md'
LOG_FILE: '/tmp/rustfs-kms.log'
run: | run: |
set -euo pipefail set -euo pipefail
if [ -z "${GH_TOKEN:-}" ]; then if [ -z "${GH_TOKEN:-}" ]; then
@@ -267,16 +287,14 @@ jobs:
echo "" echo ""
echo "- Suite: \`${SUITE}\`" echo "- Suite: \`${SUITE}\`"
echo "- Run: ${RUN_URL}" echo "- Run: ${RUN_URL}"
echo "- Attempt: ${GITHUB_RUN_ATTEMPT}"
echo "- Workflow Commit: ${GITHUB_SHA}"
echo "- Trigger: ${GITHUB_EVENT_NAME}" echo "- Trigger: ${GITHUB_EVENT_NAME}"
echo "- Date: $(date -u +%Y-%m-%d)" echo "- Date: $(date -u +%Y-%m-%d)"
echo "" echo ""
echo "## Report (errors and symptoms)" echo "## Report (errors and symptoms)"
echo "" echo ""
if [ "${EVIDENCE_OUTCOME}" = "success" ] && [ -s "${REPORT_FILE}" ]; then if [ -s "${REPORT_FILE}" ]; then
redact < "${REPORT_FILE}" redact < "${REPORT_FILE}"
elif [ "${EVIDENCE_OUTCOME}" = "success" ] && [ -s "${LOG_FILE:-}" ]; then elif [ -s "${LOG_FILE:-}" ]; then
echo "(report file missing; log tail below)" echo "(report file missing; log tail below)"
echo "" echo ""
tail -n 200 "${LOG_FILE}" | redact tail -n 200 "${LOG_FILE}" | redact
@@ -292,15 +310,14 @@ jobs:
echo "filed backlog issue for suite ${SUITE}" echo "filed backlog issue for suite ${SUITE}"
- name: Upload report and logs - name: Upload report and logs
if: ${{ always() && steps.evidence.outcome == 'success' }} if: always()
uses: actions/upload-artifact@b7c566a772e6b6bfb58ed0dc250532a479d7789f # v6 uses: actions/upload-artifact@b7c566a772e6b6bfb58ed0dc250532a479d7789f # v6
with: with:
name: rustfs-kms-test-${{ github.run_id }}-${{ github.run_attempt }} name: rustfs-kms-test-${{ github.run_id }}
path: | path: |
${{ env.FUNCTIONAL_ARTIFACTS_DIR }}/report.md /tmp/rustfs-kms.log
${{ env.FUNCTIONAL_ARTIFACTS_DIR }}/suite.log /tmp/rustfs-kms-report.md
${{ env.FUNCTIONAL_ARTIFACTS_DIR }}/cases.md if-no-files-found: warn
if-no-files-found: error
- name: Cleanup environment (after) - name: Cleanup environment (after)
if: always() if: always()
+32 -48
View File
@@ -49,16 +49,17 @@ on:
type: boolean type: boolean
default: true default: true
repository_dispatch: repository_dispatch:
# Chain handoff: dispatched when the replication suite finishes. # Chain entry: dispatched by rustfs-functional-chain.yml (runs on its own
# pf-testing runner, in parallel with the shared-VM chain).
types: [rustfs-chain-performance] types: [rustfs-chain-performance]
permissions: permissions:
contents: read contents: read
# The default performance nodes overlap the other suites' remote VMs, even # Dedicated pf-testing runner/environment: own concurrency group so perf runs
# though the runner differs. Hold the shared lock through cleanup as well. # never block (or are blocked by) the pool-expansion / heal tests.
concurrency: concurrency:
group: rustfs-shared-functional-tests group: rustfs-performance-test
cancel-in-progress: false cancel-in-progress: false
defaults: defaults:
@@ -75,33 +76,22 @@ env:
# Package used by the nightly run (workflow_dispatch inputs are empty for # Package used by the nightly run (workflow_dispatch inputs are empty for
# workflow_run events), i.e. the latest nightly deb published by nightly-gnu.yml. # workflow_run events), i.e. the latest nightly deb published by nightly-gnu.yml.
RUSTFS_NIGHTLY_PACKAGE_URL: ${{ vars.RUSTFS_NIGHTLY_PACKAGE_URL || 'https://dl.rustfs.com/artifacts/rustfs/packages/nightly/rustfs-nightly-latest.deb' }} RUSTFS_NIGHTLY_PACKAGE_URL: ${{ vars.RUSTFS_NIGHTLY_PACKAGE_URL || 'https://dl.rustfs.com/artifacts/rustfs/packages/nightly/rustfs-nightly-latest.deb' }}
# Fixed benchmark result directory so later steps can read summary.md
RUSTFS_RESULT_DIR: /tmp/rustfs-perf-results
# Cross-repo token for uploading reports to rustfs/dashboard (set in repo settings) # Cross-repo token for uploading reports to rustfs/dashboard (set in repo settings)
PF_TESTING_GH_TOKEN: ${{ secrets.PF_TESTING_GH_TOKEN }} PF_TESTING_GH_TOKEN: ${{ secrets.PF_TESTING_GH_TOKEN }}
jobs: jobs:
performance-test: performance-test:
runs-on: pf-testing runs-on: pf-testing
# Requirement: a failing benchmark must not fail the workflow;
# failures are filed to rustfs/backlog.
continue-on-error: true
timeout-minutes: 900 timeout-minutes: 900
# Run on manual dispatch, or when the nightly build completed successfully. # Run on manual dispatch, or when the nightly build completed successfully.
# Skipped when nightly failed. # Skipped when nightly failed.
if: ${{ github.event_name == 'workflow_dispatch' || github.event_name == 'repository_dispatch' }} if: ${{ github.event_name == 'workflow_dispatch' || github.event_name == 'repository_dispatch' }}
steps: steps:
- name: Initialize functional evidence
id: evidence
run: |
set -euo pipefail
umask 077
FUNCTIONAL_ARTIFACTS_DIR="${RUNNER_TEMP}/rustfs-performance-${GITHUB_RUN_ID}-${GITHUB_RUN_ATTEMPT}"
mkdir -- "${FUNCTIONAL_ARTIFACTS_DIR}" "${FUNCTIONAL_ARTIFACTS_DIR}-scratch"
{
printf 'FUNCTIONAL_ARTIFACTS_DIR=%s\n' "${FUNCTIONAL_ARTIFACTS_DIR}"
printf 'LOG_FILE=%s/suite.log\n' "${FUNCTIONAL_ARTIFACTS_DIR}"
printf 'REPORT_FILE=%s/report.md\n' "${FUNCTIONAL_ARTIFACTS_DIR}"
printf 'TMPDIR=%s-scratch\n' "${FUNCTIONAL_ARTIFACTS_DIR}"
printf 'RUSTFS_RESULT_DIR=%s/results\n' "${FUNCTIONAL_ARTIFACTS_DIR}"
printf 'VERSION_FILE=%s/version.txt\n' "${FUNCTIONAL_ARTIFACTS_DIR}"
} >> "${GITHUB_ENV}"
# auto-testing is private: clone it with the dedicated PF token (not # auto-testing is private: clone it with the dedicated PF token (not
# GITHUB_TOKEN) and retry transient GitHub/network failures. # GITHUB_TOKEN) and retry transient GitHub/network failures.
- name: Checkout auto-testing scripts (with retry) - name: Checkout auto-testing scripts (with retry)
@@ -133,7 +123,7 @@ jobs:
if: ${{ inputs.cleanup_before != 'false' }} if: ${{ inputs.cleanup_before != 'false' }}
run: | run: |
chmod +x auto-testing/rustfs_performance_test.sh chmod +x auto-testing/rustfs_performance_test.sh
./auto-testing/rustfs_performance_test.sh --step 1 -y --log-file "${LOG_FILE:-/dev/null}" ./auto-testing/rustfs_performance_test.sh --step 1 -y
- name: Install RustFS package & start cluster (4x4) - name: Install RustFS package & start cluster (4x4)
run: | run: |
@@ -143,7 +133,7 @@ jobs:
else else
ARGS+=(--package-url "${{ env.RUSTFS_NIGHTLY_PACKAGE_URL }}") ARGS+=(--package-url "${{ env.RUSTFS_NIGHTLY_PACKAGE_URL }}")
fi fi
./auto-testing/rustfs_performance_test.sh "${ARGS[@]}" --log-file "${LOG_FILE}" ./auto-testing/rustfs_performance_test.sh "${ARGS[@]}"
- name: Preflight checks - name: Preflight checks
run: | run: |
@@ -153,7 +143,7 @@ jobs:
else else
ARGS+=(--package-url "${{ env.RUSTFS_NIGHTLY_PACKAGE_URL }}") ARGS+=(--package-url "${{ env.RUSTFS_NIGHTLY_PACKAGE_URL }}")
fi fi
./auto-testing/rustfs_performance_test.sh "${ARGS[@]}" --log-file "${LOG_FILE}" ./auto-testing/rustfs_performance_test.sh "${ARGS[@]}"
- name: Run benchmark (GET/PUT/MIXED) - name: Run benchmark (GET/PUT/MIXED)
id: benchmark id: benchmark
@@ -166,15 +156,17 @@ jobs:
--step 5 -y \ --step 5 -y \
--warp-duration "${{ inputs.warp_duration || '5m' }}" \ --warp-duration "${{ inputs.warp_duration || '5m' }}" \
--warp-concurrency "${{ inputs.warp_concurrency || '64' }}" \ --warp-concurrency "${{ inputs.warp_concurrency || '64' }}" \
--log-file "${LOG_FILE}" --log-file /tmp/rustfs-perf-test.log
- name: Analyze results - name: Analyze results
if: ${{ steps.benchmark.conclusion == 'success' }} if: ${{ steps.benchmark.conclusion == 'success' }}
run: | run: |
./auto-testing/rustfs_performance_test.sh --step 6 -y --log-file "${LOG_FILE:-/dev/null}" ./auto-testing/rustfs_performance_test.sh --step 6 -y
- name: Collect RustFS version info - name: Collect RustFS version info
if: ${{ steps.benchmark.conclusion == 'success' }} if: ${{ steps.benchmark.conclusion == 'success' }}
env:
VERSION_FILE: /tmp/rustfs-version.txt
run: | run: |
set -euo pipefail set -euo pipefail
read -r -a NODES <<< "${RUSTFS_NODES}" read -r -a NODES <<< "${RUSTFS_NODES}"
@@ -194,6 +186,7 @@ jobs:
env: env:
GH_TOKEN: ${{ env.PF_TESTING_GH_TOKEN }} GH_TOKEN: ${{ env.PF_TESTING_GH_TOKEN }}
RESULT_DIR: ${{ env.RUSTFS_RESULT_DIR }} RESULT_DIR: ${{ env.RUSTFS_RESULT_DIR }}
VERSION_FILE: /tmp/rustfs-version.txt
run: | run: |
set -euo pipefail set -euo pipefail
if [ -z "${GH_TOKEN:-}" ]; then if [ -z "${GH_TOKEN:-}" ]; then
@@ -201,7 +194,7 @@ jobs:
exit 0 exit 0
fi fi
SUMMARY="${RESULT_DIR}/summary.md" SUMMARY="${RESULT_DIR}/summary.md"
[ -s "${SUMMARY}" ] || { echo "summary.md not found at ${SUMMARY}"; exit 1; } [ -f "${SUMMARY}" ] || { echo "summary.md not found at ${SUMMARY}"; exit 1; }
DATE="$(date -u +%Y-%m-%d)" DATE="$(date -u +%Y-%m-%d)"
REPORT_PATH="reports/${DATE}.md" REPORT_PATH="reports/${DATE}.md"
{ {
@@ -209,8 +202,6 @@ jobs:
echo "" echo ""
echo "- **Date**: ${DATE}" echo "- **Date**: ${DATE}"
echo "- **Run**: ${{ github.server_url }}/${{ github.repository }}/actions/runs/${{ github.run_id }}" echo "- **Run**: ${{ github.server_url }}/${{ github.repository }}/actions/runs/${{ github.run_id }}"
echo "- **Attempt**: ${GITHUB_RUN_ATTEMPT}"
echo "- **Workflow Commit**: ${GITHUB_SHA}"
echo "- **Trigger**: ${{ github.event_name }}" echo "- **Trigger**: ${{ github.event_name }}"
echo "- **Package**: ${{ inputs.package_url || 'nightly (R2 latest)' }}" echo "- **Package**: ${{ inputs.package_url || 'nightly (R2 latest)' }}"
echo "" echo ""
@@ -220,8 +211,8 @@ jobs:
echo '```text' echo '```text'
cat "${VERSION_FILE}" cat "${VERSION_FILE}"
echo '```' echo '```'
} > "${REPORT_FILE}" } > /tmp/rustfs-perf-report.md
CONTENT="$(python3 -c 'import base64,sys; print(base64.b64encode(open(sys.argv[1],"rb").read()).decode())' "${REPORT_FILE}")" CONTENT="$(python3 -c 'import base64; print(base64.b64encode(open("/tmp/rustfs-perf-report.md","rb").read()).decode())')"
SHA="$(gh api "repos/rustfs/dashboard/contents/${REPORT_PATH}" -q '.sha' 2>/dev/null || true)" SHA="$(gh api "repos/rustfs/dashboard/contents/${REPORT_PATH}" -q '.sha' 2>/dev/null || true)"
if [ -n "${SHA}" ]; then if [ -n "${SHA}" ]; then
jq -n --arg msg "report: ${DATE}" --arg content "${CONTENT}" --arg sha "${SHA}" \ jq -n --arg msg "report: ${DATE}" --arg content "${CONTENT}" --arg sha "${SHA}" \
@@ -240,10 +231,11 @@ jobs:
continue-on-error: true continue-on-error: true
env: env:
GH_TOKEN: ${{ secrets.PF_TESTING_GH_TOKEN }} GH_TOKEN: ${{ secrets.PF_TESTING_GH_TOKEN }}
EVIDENCE_OUTCOME: ${{ steps.evidence.outcome }}
SUITE: 'performance' SUITE: 'performance'
SUITE_LABEL: 'Performance' SUITE_LABEL: 'Performance'
RUN_URL: ${{ github.server_url }}/${{ github.repository }}/actions/runs/${{ github.run_id }} RUN_URL: ${{ github.server_url }}/${{ github.repository }}/actions/runs/${{ github.run_id }}
REPORT_FILE: '/tmp/rustfs-perf-report.md'
LOG_FILE: '/tmp/rustfs-perf-test.log'
run: | run: |
set -euo pipefail set -euo pipefail
if [ -z "${GH_TOKEN:-}" ]; then if [ -z "${GH_TOKEN:-}" ]; then
@@ -271,16 +263,14 @@ jobs:
echo "" echo ""
echo "- Suite: \`${SUITE}\`" echo "- Suite: \`${SUITE}\`"
echo "- Run: ${RUN_URL}" echo "- Run: ${RUN_URL}"
echo "- Attempt: ${GITHUB_RUN_ATTEMPT}"
echo "- Workflow Commit: ${GITHUB_SHA}"
echo "- Trigger: ${GITHUB_EVENT_NAME}" echo "- Trigger: ${GITHUB_EVENT_NAME}"
echo "- Date: $(date -u +%Y-%m-%d)" echo "- Date: $(date -u +%Y-%m-%d)"
echo "" echo ""
echo "## Report (errors and symptoms)" echo "## Report (errors and symptoms)"
echo "" echo ""
if [ "${EVIDENCE_OUTCOME}" = "success" ] && [ -s "${REPORT_FILE}" ]; then if [ -s "${REPORT_FILE}" ]; then
redact < "${REPORT_FILE}" redact < "${REPORT_FILE}"
elif [ "${EVIDENCE_OUTCOME}" = "success" ] && [ -s "${LOG_FILE:-}" ]; then elif [ -s "${LOG_FILE:-}" ]; then
echo "(report file missing; log tail below)" echo "(report file missing; log tail below)"
echo "" echo ""
tail -n 200 "${LOG_FILE}" | redact tail -n 200 "${LOG_FILE}" | redact
@@ -296,26 +286,20 @@ jobs:
echo "filed backlog issue for suite ${SUITE}" echo "filed backlog issue for suite ${SUITE}"
- name: Upload test logs & results - name: Upload test logs & results
if: ${{ always() && steps.evidence.outcome == 'success' }} if: always()
uses: actions/upload-artifact@b7c566a772e6b6bfb58ed0dc250532a479d7789f # v6 uses: actions/upload-artifact@b7c566a772e6b6bfb58ed0dc250532a479d7789f # v6
with: with:
name: rustfs-perf-test-${{ github.run_id }}-${{ github.run_attempt }} name: rustfs-perf-test-${{ github.run_id }}
path: | path: |
${{ env.FUNCTIONAL_ARTIFACTS_DIR }}/report.md /tmp/rustfs-perf-test*.log
${{ env.FUNCTIONAL_ARTIFACTS_DIR }}/suite.log /tmp/rustfs-perf-results/**
${{ env.FUNCTIONAL_ARTIFACTS_DIR }}/version.txt /tmp/rustfs-version.txt
${{ env.FUNCTIONAL_ARTIFACTS_DIR }}/results/master.log if-no-files-found: warn
${{ env.FUNCTIONAL_ARTIFACTS_DIR }}/results/summary.md
${{ env.FUNCTIONAL_ARTIFACTS_DIR }}/results/summary.tsv
${{ env.FUNCTIONAL_ARTIFACTS_DIR }}/results/get_*.txt
${{ env.FUNCTIONAL_ARTIFACTS_DIR }}/results/put_*.txt
${{ env.FUNCTIONAL_ARTIFACTS_DIR }}/results/mixed_*.txt
if-no-files-found: error
- name: Reset test environment (after) - name: Reset test environment (after)
if: ${{ always() && inputs.cleanup_after != 'false' }} if: ${{ always() && inputs.cleanup_after != 'false' }}
run: | run: |
./auto-testing/rustfs_performance_test.sh --step 7 -y --log-file "${LOG_FILE:-/dev/null}" ./auto-testing/rustfs_performance_test.sh --step 7 -y
- name: Notify on failure - name: Notify on failure
if: failure() if: failure()
+8 -10
View File
@@ -76,6 +76,9 @@ jobs:
pool-expansion-test: pool-expansion-test:
name: Pool expansion / decommission test name: Pool expansion / decommission test
runs-on: smoke-testing runs-on: smoke-testing
# Requirement: a failing suite must not fail the workflow; failures
# are filed to rustfs/backlog and the chain continues.
continue-on-error: true
timeout-minutes: 360 timeout-minutes: 360
if: ${{ github.event_name == 'workflow_dispatch' || github.event_name == 'repository_dispatch' }} if: ${{ github.event_name == 'workflow_dispatch' || github.event_name == 'repository_dispatch' }}
env: env:
@@ -539,22 +542,17 @@ jobs:
fi fi
DATE="$(date -u +%Y-%m-%d)" DATE="$(date -u +%Y-%m-%d)"
REPORT_PATH="functional-reports/${SUITE}/${DATE}.md" REPORT_PATH="functional-reports/${SUITE}/${DATE}.md"
# Base64-encode the report into a temp file and feed it to jq via CONTENT="$(python3 -c 'import base64,sys;print(base64.b64encode(open(sys.argv[1],"rb").read()).decode())' "${REPORT_FILE}")"
# --rawfile: large reports (e.g. pool) exceed the OS argv limit and
# make `jq --arg content "${CONTENT}"` fail with "Argument list too long".
B64_FILE="$(mktemp)"
python3 -c 'import base64,sys;print(base64.b64encode(open(sys.argv[1],"rb").read()).decode())' "${REPORT_FILE}" > "${B64_FILE}"
SHA="$(gh api "repos/rustfs/dashboard/contents/${REPORT_PATH}" -q '.sha' 2>/dev/null || true)" SHA="$(gh api "repos/rustfs/dashboard/contents/${REPORT_PATH}" -q '.sha' 2>/dev/null || true)"
if [ -n "${SHA}" ]; then if [ -n "${SHA}" ]; then
jq -n --arg msg "report(${SUITE}): ${DATE}" --rawfile content "${B64_FILE}" --arg sha "${SHA}" \ jq -n --arg msg "report(${SUITE}): ${DATE}" --arg content "${CONTENT}" --arg sha "${SHA}" \
'{message:$msg, content:($content|rtrimstr("\n")), sha:$sha}' \ '{message:$msg, content:$content, sha:$sha}' \
| gh api --method PUT "repos/rustfs/dashboard/contents/${REPORT_PATH}" --input - >/dev/null | gh api --method PUT "repos/rustfs/dashboard/contents/${REPORT_PATH}" --input - >/dev/null
else else
jq -n --arg msg "report(${SUITE}): ${DATE}" --rawfile content "${B64_FILE}" \ jq -n --arg msg "report(${SUITE}): ${DATE}" --arg content "${CONTENT}" \
'{message:$msg, content:($content|rtrimstr("\n"))}' \ '{message:$msg, content:$content}' \
| gh api --method PUT "repos/rustfs/dashboard/contents/${REPORT_PATH}" --input - >/dev/null | gh api --method PUT "repos/rustfs/dashboard/contents/${REPORT_PATH}" --input - >/dev/null
fi fi
rm -f "${B64_FILE}"
- name: File failure issue in rustfs/backlog - name: File failure issue in rustfs/backlog
if: ${{ always() && (failure() || steps.pool_test.outcome == 'failure' || steps.pool_test.outcome == 'cancelled') }} if: ${{ always() && (failure() || steps.pool_test.outcome == 'failure' || steps.pool_test.outcome == 'cancelled') }}
+90 -107
View File
@@ -34,7 +34,8 @@ on:
- site - site
default: all default: all
repository_dispatch: repository_dispatch:
# Chain handoff: dispatched when the security suite finishes. # Chain handoff: dispatched when the security suite finishes. This is the
# last link of the functional chain.
types: [rustfs-chain-replication] types: [rustfs-chain-replication]
permissions: permissions:
@@ -61,28 +62,12 @@ env:
jobs: jobs:
replication-test: replication-test:
runs-on: smoke-testing runs-on: smoke-testing
# A failed replication run must not break the chain or the workflow: the
# failure is reported to rustfs/backlog instead (see the issue step).
continue-on-error: true
timeout-minutes: 360 timeout-minutes: 360
if: ${{ github.event_name == 'workflow_dispatch' || github.event_name == 'repository_dispatch' }} if: ${{ github.event_name == 'workflow_dispatch' || github.event_name == 'repository_dispatch' }}
steps: steps:
- name: Checkout repository (for report parser)
uses: actions/checkout@9c091bb21b7c1c1d1991bb908d89e4e9dddfe3e0 # v7
with:
persist-credentials: false
- name: Initialize functional evidence
id: evidence
run: |
set -euo pipefail
umask 077
FUNCTIONAL_ARTIFACTS_DIR="${RUNNER_TEMP}/rustfs-replication-${GITHUB_RUN_ID}-${GITHUB_RUN_ATTEMPT}"
mkdir -- "${FUNCTIONAL_ARTIFACTS_DIR}" "${FUNCTIONAL_ARTIFACTS_DIR}-scratch"
{
printf 'FUNCTIONAL_ARTIFACTS_DIR=%s\n' "${FUNCTIONAL_ARTIFACTS_DIR}"
printf 'LOG_FILE=%s/suite.log\n' "${FUNCTIONAL_ARTIFACTS_DIR}"
printf 'REPORT_FILE=%s/report.md\n' "${FUNCTIONAL_ARTIFACTS_DIR}"
printf 'TMPDIR=%s-scratch\n' "${FUNCTIONAL_ARTIFACTS_DIR}"
} >> "${GITHUB_ENV}"
# auto-testing is private: clone it with the dedicated PF token (not # auto-testing is private: clone it with the dedicated PF token (not
# GITHUB_TOKEN) and retry transient GitHub/network failures. # GITHUB_TOKEN) and retry transient GitHub/network failures.
- name: Checkout auto-testing scripts (with retry) - name: Checkout auto-testing scripts (with retry)
@@ -131,6 +116,9 @@ jobs:
- name: Run replication suite - name: Run replication suite
id: test id: test
continue-on-error: true
env:
LOG_FILE: /tmp/rustfs-replication.log
run: | run: |
set -euo pipefail set -euo pipefail
chmod +x auto-testing/rustfs-replication-test.sh chmod +x auto-testing/rustfs-replication-test.sh
@@ -153,7 +141,10 @@ jobs:
./auto-testing/rustfs-replication-test.sh "${ARGS[@]}" ./auto-testing/rustfs-replication-test.sh "${ARGS[@]}"
- name: Generate report - name: Generate report
if: ${{ always() && steps.evidence.outcome == 'success' }} if: always()
env:
LOG_FILE: /tmp/rustfs-replication.log
REPORT_FILE: /tmp/rustfs-replication-report.md
run: | run: |
set -euo pipefail set -euo pipefail
PACKAGE_URL='${{ inputs.package_url }}' PACKAGE_URL='${{ inputs.package_url }}'
@@ -175,44 +166,80 @@ jobs:
RUSTFS_VERSION_INFO="${DETECTED_VERSION}" RUSTFS_VERSION_INFO="${DETECTED_VERSION}"
fi fi
fi fi
CASE_TABLE="${FUNCTIONAL_ARTIFACTS_DIR}/cases.md" CASE_TABLE="/tmp/rustfs-replication-cases.md"
CASE_RESULT=success python3 - "${LOG_FILE}" "${CASE_TABLE}" <<'PY'
python3 scripts/functional_case_report.py "${LOG_FILE}" "${CASE_TABLE}" || CASE_RESULT=failure import re
RESULT=failure import sys
if [ '${{ steps.test.outcome }}' = 'success' ] && [ "${CASE_RESULT}" = 'success' ]; then
RESULT=success log_file, out_file = sys.argv[1], sys.argv[2]
fi ansi = re.compile(r'\x1b\[[0-9;]*m')
start_re = re.compile(r'^---\s+([A-Z0-9]+-[0-9]+)\s+(.+?)\s+---$')
done_re = re.compile(r'^\[(PASS|FAIL|UNSUPPORTED)\]\s+([A-Z0-9]+-[0-9]+)\b')
rows = []
index = {}
try:
with open(log_file, 'r', encoding='utf-8', errors='replace') as fh:
for raw in fh:
line = ansi.sub('', raw).strip()
m = start_re.match(line)
if m:
case_id, name = m.group(1), m.group(2)
if case_id not in index:
index[case_id] = len(rows)
rows.append([case_id, name, 'RUNNING'])
continue
m = done_re.match(line)
if m:
status, case_id = m.group(1), m.group(2)
if case_id in index:
rows[index[case_id]][2] = status
else:
rows.append([case_id, case_id, status])
index[case_id] = len(rows) - 1
except FileNotFoundError:
rows = []
counts = {'PASS': 0, 'FAIL': 0, 'UNSUPPORTED': 0, 'RUNNING': 0}
for _, _, status in rows:
counts[status] = counts.get(status, 0) + 1
with open(out_file, 'w', encoding='utf-8') as out:
out.write('## Case Summary\n\n')
out.write(f"- Total: {len(rows)}\\n")
out.write(f"- PASS: {counts.get('PASS', 0)}\\n")
out.write(f"- FAIL: {counts.get('FAIL', 0)}\\n")
out.write(f"- UNSUPPORTED: {counts.get('UNSUPPORTED', 0)}\\n")
out.write('\\n')
out.write('| Case | Name | Status |\\n')
out.write('| --- | --- | --- |\\n')
for case_id, name, status in rows:
out.write(f'| {case_id} | {name} | {status} |\\n')
PY
{ {
echo "# RustFS replication test report" echo "# RustFS replication test report"
echo "" echo ""
echo "- Run: ${{ github.server_url }}/${{ github.repository }}/actions/runs/${{ github.run_id }}" echo "- Run: ${{ github.server_url }}/${{ github.repository }}/actions/runs/${{ github.run_id }}"
echo "- Attempt: ${GITHUB_RUN_ATTEMPT}"
echo "- Workflow Commit: ${GITHUB_SHA}"
echo "- Trigger: ${{ github.event_name }}" echo "- Trigger: ${{ github.event_name }}"
echo "- Package: ${PACKAGE_SOURCE}" echo "- Package: ${PACKAGE_SOURCE}"
echo "- RustFS Version: ${RUSTFS_VERSION_INFO}" echo "- RustFS Version: ${RUSTFS_VERSION_INFO}"
echo "- Test Step Outcome: ${RESULT}" echo "- Test Step Outcome: ${{ steps.test.outcome }}"
echo "- Suite Step Outcome: ${{ steps.test.outcome }}"
echo "" echo ""
if [ "${RESULT}" = "success" ]; then cat "${CASE_TABLE}" || true
cat "${CASE_TABLE}" echo ""
echo "" echo "## Log tail"
echo "## Log tail" echo '```text'
echo '```text' tail -n 200 "${LOG_FILE}" || true
tail -n 200 "${LOG_FILE}" echo '```'
echo '```'
else
echo "The suite or evidence validation failed. See this run's artifact for partial case results and suite.log."
fi
} | tee "${REPORT_FILE}" } | tee "${REPORT_FILE}"
cat "${REPORT_FILE}" >> "${GITHUB_STEP_SUMMARY}" cat "${REPORT_FILE}" >> "${GITHUB_STEP_SUMMARY}"
[ "${RESULT}" = "success" ]
- name: Upload functional report to dashboard - name: Upload functional report to dashboard
if: ${{ always() && steps.evidence.outcome == 'success' }} if: always()
continue-on-error: true continue-on-error: true
env: env:
GH_TOKEN: ${{ env.PF_TESTING_GH_TOKEN }} GH_TOKEN: ${{ env.PF_TESTING_GH_TOKEN }}
REPORT_FILE: /tmp/rustfs-replication-report.md
SUITE: replication SUITE: replication
run: | run: |
set -euo pipefail set -euo pipefail
@@ -222,32 +249,28 @@ jobs:
fi fi
DATE="$(date -u +%Y-%m-%d)" DATE="$(date -u +%Y-%m-%d)"
REPORT_PATH="functional-reports/${SUITE}/${DATE}.md" REPORT_PATH="functional-reports/${SUITE}/${DATE}.md"
# Base64-encode the report into a temp file and feed it to jq via CONTENT="$(python3 -c 'import base64,sys;print(base64.b64encode(open(sys.argv[1],"rb").read()).decode())' "${REPORT_FILE}")"
# --rawfile: large reports (e.g. pool) exceed the OS argv limit and
# make `jq --arg content "${CONTENT}"` fail with "Argument list too long".
B64_FILE="$(mktemp)"
python3 -c 'import base64,sys;print(base64.b64encode(open(sys.argv[1],"rb").read()).decode())' "${REPORT_FILE}" > "${B64_FILE}"
SHA="$(gh api "repos/rustfs/dashboard/contents/${REPORT_PATH}" -q '.sha' 2>/dev/null || true)" SHA="$(gh api "repos/rustfs/dashboard/contents/${REPORT_PATH}" -q '.sha' 2>/dev/null || true)"
if [ -n "${SHA}" ]; then if [ -n "${SHA}" ]; then
jq -n --arg msg "report(${SUITE}): ${DATE}" --rawfile content "${B64_FILE}" --arg sha "${SHA}" \ jq -n --arg msg "report(${SUITE}): ${DATE}" --arg content "${CONTENT}" --arg sha "${SHA}" \
'{message:$msg, content:($content|rtrimstr("\n")), sha:$sha}' \ '{message:$msg, content:$content, sha:$sha}' \
| gh api --method PUT "repos/rustfs/dashboard/contents/${REPORT_PATH}" --input - >/dev/null | gh api --method PUT "repos/rustfs/dashboard/contents/${REPORT_PATH}" --input - >/dev/null
else else
jq -n --arg msg "report(${SUITE}): ${DATE}" --rawfile content "${B64_FILE}" \ jq -n --arg msg "report(${SUITE}): ${DATE}" --arg content "${CONTENT}" \
'{message:$msg, content:($content|rtrimstr("\n"))}' \ '{message:$msg, content:$content}' \
| gh api --method PUT "repos/rustfs/dashboard/contents/${REPORT_PATH}" --input - >/dev/null | gh api --method PUT "repos/rustfs/dashboard/contents/${REPORT_PATH}" --input - >/dev/null
fi fi
rm -f "${B64_FILE}"
- name: File failure issue in rustfs/backlog - name: File failure issue in rustfs/backlog
if: ${{ always() && (failure() || steps.test.outcome == 'failure' || steps.test.outcome == 'cancelled') }} if: ${{ always() && (failure() || steps.test.outcome == 'failure' || steps.test.outcome == 'cancelled') }}
continue-on-error: true continue-on-error: true
env: env:
GH_TOKEN: ${{ secrets.PF_TESTING_GH_TOKEN }} GH_TOKEN: ${{ secrets.PF_TESTING_GH_TOKEN }}
EVIDENCE_OUTCOME: ${{ steps.evidence.outcome }}
SUITE: 'replication' SUITE: 'replication'
SUITE_LABEL: 'Replication (bucket + site)' SUITE_LABEL: 'Replication (bucket + site)'
RUN_URL: ${{ github.server_url }}/${{ github.repository }}/actions/runs/${{ github.run_id }} RUN_URL: ${{ github.server_url }}/${{ github.repository }}/actions/runs/${{ github.run_id }}
REPORT_FILE: '/tmp/rustfs-replication-report.md'
LOG_FILE: '/tmp/rustfs-replication.log'
run: | run: |
set -euo pipefail set -euo pipefail
if [ -z "${GH_TOKEN:-}" ]; then if [ -z "${GH_TOKEN:-}" ]; then
@@ -275,16 +298,14 @@ jobs:
echo "" echo ""
echo "- Suite: \`${SUITE}\`" echo "- Suite: \`${SUITE}\`"
echo "- Run: ${RUN_URL}" echo "- Run: ${RUN_URL}"
echo "- Attempt: ${GITHUB_RUN_ATTEMPT}"
echo "- Workflow Commit: ${GITHUB_SHA}"
echo "- Trigger: ${GITHUB_EVENT_NAME}" echo "- Trigger: ${GITHUB_EVENT_NAME}"
echo "- Date: $(date -u +%Y-%m-%d)" echo "- Date: $(date -u +%Y-%m-%d)"
echo "" echo ""
echo "## Report (errors and symptoms)" echo "## Report (errors and symptoms)"
echo "" echo ""
if [ "${EVIDENCE_OUTCOME}" = "success" ] && [ -s "${REPORT_FILE}" ]; then if [ -s "${REPORT_FILE}" ]; then
redact < "${REPORT_FILE}" redact < "${REPORT_FILE}"
elif [ "${EVIDENCE_OUTCOME}" = "success" ] && [ -s "${LOG_FILE:-}" ]; then elif [ -s "${LOG_FILE:-}" ]; then
echo "(report file missing; log tail below)" echo "(report file missing; log tail below)"
echo "" echo ""
tail -n 200 "${LOG_FILE}" | redact tail -n 200 "${LOG_FILE}" | redact
@@ -300,15 +321,14 @@ jobs:
echo "filed backlog issue for suite ${SUITE}" echo "filed backlog issue for suite ${SUITE}"
- name: Upload report and logs - name: Upload report and logs
if: ${{ always() && steps.evidence.outcome == 'success' }} if: always()
uses: actions/upload-artifact@b7c566a772e6b6bfb58ed0dc250532a479d7789f # v6 uses: actions/upload-artifact@b7c566a772e6b6bfb58ed0dc250532a479d7789f # v6
with: with:
name: rustfs-replication-${{ github.run_id }}-${{ github.run_attempt }} name: rustfs-replication-${{ github.run_id }}
path: | path: |
${{ env.FUNCTIONAL_ARTIFACTS_DIR }}/report.md /tmp/rustfs-replication.log
${{ env.FUNCTIONAL_ARTIFACTS_DIR }}/suite.log /tmp/rustfs-replication-report.md
${{ env.FUNCTIONAL_ARTIFACTS_DIR }}/cases.md if-no-files-found: warn
if-no-files-found: error
- name: Cleanup environment (after) - name: Cleanup environment (after)
if: always() if: always()
@@ -329,50 +349,13 @@ jobs:
' '
done done
- name: "Continue functional chain (next: Performance)" - name: Chain complete
# Replication is the last link of the functional chain: nothing to
# dispatch after it. This step just records that the chain finished.
if: ${{ always() && github.event_name == 'repository_dispatch' }} if: ${{ always() && github.event_name == 'repository_dispatch' }}
env:
GH_TOKEN: ${{ secrets.PF_TESTING_GH_TOKEN }}
run: | run: |
set -uo pipefail echo "Functional chain complete: replication (final suite) finished."
if [ -z "${GH_TOKEN:-}" ]; then echo "from_suite=security trigger=${{ github.event_name }} outcome=${{ steps.test.outcome }}"
echo "PF_TESTING_GH_TOKEN is not configured; cannot dispatch the next suite" >&2
exit 1
fi
DISPATCHED=0
for attempt in 1 2 3; do
if gh api --method POST repos/rustfs/rustfs/dispatches \
-f event_type='rustfs-chain-performance' \
-F 'client_payload[from_suite]=replication'; then
echo "dispatched next suite Performance (attempt ${attempt})"
DISPATCHED=1
break
fi
echo "dispatch attempt ${attempt} failed; retrying in ${attempt}0s" >&2
sleep "${attempt}0"
done
if [ "${DISPATCHED:-0}" -ne 1 ]; then
echo "ERROR: functional chain stalled: could not dispatch Performance after 3 attempts" >&2
TITLE="[functional][chain] stalled after replication (run ${GITHUB_RUN_ID})"
BODY_FILE="$(mktemp)"
trap 'rm -f "${BODY_FILE}"' EXIT
{
echo "The functional chain could not hand off from **replication** to **Performance** after 3 attempts."
echo ""
echo "- Failed suite job: ${GITHUB_SERVER_URL}/${GITHUB_REPOSITORY}/actions/runs/${GITHUB_RUN_ID}"
echo "- Expected next event: 'rustfs-chain-performance'"
echo "- Likely cause: PF_TESTING_GH_TOKEN lacks contents:write on rustfs/rustfs, or the GitHub API was unavailable."
echo "- Recovery: re-dispatch manually with"
FENCE="$(printf "\x60\x60\x60")"; echo " ${FENCE}"
echo " gh api --method POST repos/rustfs/rustfs/dispatches -f event_type='rustfs-chain-performance'"
FENCE="$(printf "\x60\x60\x60")"; echo " ${FENCE}"
} > "${BODY_FILE}"
gh issue create -R rustfs/backlog --title "${TITLE}" \
--body-file "${BODY_FILE}" --label functional-test \
|| gh issue create -R rustfs/backlog --title "${TITLE}" --body-file "${BODY_FILE}" \
|| echo "could not file the stall alert issue either; check the token" >&2
exit 1
fi
- name: Notify on failure - name: Notify on failure
if: failure() if: failure()
+84 -64
View File
@@ -37,28 +37,10 @@ env:
jobs: jobs:
s3-compat-test: s3-compat-test:
runs-on: smoke-testing runs-on: smoke-testing
continue-on-error: true
timeout-minutes: 360 timeout-minutes: 360
if: ${{ github.event_name == 'workflow_dispatch' || github.event_name == 'repository_dispatch' }} if: ${{ github.event_name == 'workflow_dispatch' || github.event_name == 'repository_dispatch' }}
steps: steps:
- name: Checkout repository (for report parser)
uses: actions/checkout@9c091bb21b7c1c1d1991bb908d89e4e9dddfe3e0 # v7
with:
persist-credentials: false
- name: Initialize functional evidence
id: evidence
run: |
set -euo pipefail
umask 077
FUNCTIONAL_ARTIFACTS_DIR="${RUNNER_TEMP}/rustfs-s3-compat-${GITHUB_RUN_ID}-${GITHUB_RUN_ATTEMPT}"
mkdir -- "${FUNCTIONAL_ARTIFACTS_DIR}" "${FUNCTIONAL_ARTIFACTS_DIR}-scratch"
{
printf 'FUNCTIONAL_ARTIFACTS_DIR=%s\n' "${FUNCTIONAL_ARTIFACTS_DIR}"
printf 'LOG_FILE=%s/suite.log\n' "${FUNCTIONAL_ARTIFACTS_DIR}"
printf 'REPORT_FILE=%s/report.md\n' "${FUNCTIONAL_ARTIFACTS_DIR}"
printf 'TMPDIR=%s-scratch\n' "${FUNCTIONAL_ARTIFACTS_DIR}"
} >> "${GITHUB_ENV}"
# auto-testing is private: clone it with the dedicated PF token (not # auto-testing is private: clone it with the dedicated PF token (not
# GITHUB_TOKEN) and retry transient GitHub/network failures. # GITHUB_TOKEN) and retry transient GitHub/network failures.
- name: Checkout auto-testing scripts (with retry) - name: Checkout auto-testing scripts (with retry)
@@ -106,6 +88,9 @@ jobs:
- name: Run S3 compatibility suite - name: Run S3 compatibility suite
id: test id: test
continue-on-error: true
env:
LOG_FILE: /tmp/rustfs-s3-compat.log
run: | run: |
set -euo pipefail set -euo pipefail
chmod +x auto-testing/rustfs-s3-compat-test.sh chmod +x auto-testing/rustfs-s3-compat-test.sh
@@ -122,7 +107,10 @@ jobs:
./auto-testing/rustfs-s3-compat-test.sh "${ARGS[@]}" ./auto-testing/rustfs-s3-compat-test.sh "${ARGS[@]}"
- name: Generate report - name: Generate report
if: ${{ always() && steps.evidence.outcome == 'success' }} if: always()
env:
LOG_FILE: /tmp/rustfs-s3-compat.log
REPORT_FILE: /tmp/rustfs-s3-compat-report.md
run: | run: |
set -euo pipefail set -euo pipefail
PACKAGE_URL='${{ inputs.package_url }}' PACKAGE_URL='${{ inputs.package_url }}'
@@ -144,44 +132,83 @@ jobs:
RUSTFS_VERSION_INFO="${DETECTED_VERSION}" RUSTFS_VERSION_INFO="${DETECTED_VERSION}"
fi fi
fi fi
CASE_TABLE="${FUNCTIONAL_ARTIFACTS_DIR}/cases.md" CASE_TABLE="/tmp/rustfs-s3-compat-cases.md"
CASE_RESULT=success python3 - "${LOG_FILE}" "${CASE_TABLE}" <<'PY'
python3 scripts/functional_case_report.py "${LOG_FILE}" "${CASE_TABLE}" || CASE_RESULT=failure import re
RESULT=failure import sys
if [ '${{ steps.test.outcome }}' = 'success' ] && [ "${CASE_RESULT}" = 'success' ]; then
RESULT=success log_file, out_file = sys.argv[1], sys.argv[2]
fi ansi = re.compile(r'\x1b\[[0-9;]*m')
start_re = re.compile(r'^---\s+([A-Z0-9]+-[0-9]+)\s+(.+?)\s+---$')
done_re = re.compile(r'^\[(PASS|FAIL|UNSUPPORTED)\]\s+([A-Z0-9]+-[0-9]+)\b')
rows = []
index = {}
current = None
try:
with open(log_file, 'r', encoding='utf-8', errors='replace') as fh:
for raw in fh:
line = ansi.sub('', raw).strip()
m = start_re.match(line)
if m:
case_id, name = m.group(1), m.group(2)
current = case_id
if case_id not in index:
index[case_id] = len(rows)
rows.append([case_id, name, 'RUNNING'])
continue
m = done_re.match(line)
if m:
status, case_id = m.group(1), m.group(2)
if case_id in index:
rows[index[case_id]][2] = status
else:
rows.append([case_id, case_id, status])
index[case_id] = len(rows) - 1
current = None
except FileNotFoundError:
rows = []
counts = {'PASS': 0, 'FAIL': 0, 'UNSUPPORTED': 0, 'RUNNING': 0}
for _, _, status in rows:
counts[status] = counts.get(status, 0) + 1
with open(out_file, 'w', encoding='utf-8') as out:
out.write('## Case Summary\n\n')
out.write(f"- Total: {len(rows)}\\n")
out.write(f"- PASS: {counts.get('PASS', 0)}\\n")
out.write(f"- FAIL: {counts.get('FAIL', 0)}\\n")
out.write(f"- UNSUPPORTED: {counts.get('UNSUPPORTED', 0)}\\n")
out.write('\\n')
out.write('| Case | Name | Status |\\n')
out.write('| --- | --- | --- |\\n')
for case_id, name, status in rows:
out.write(f'| {case_id} | {name} | {status} |\\n')
PY
{ {
echo "# RustFS S3 compatibility test report" echo "# RustFS S3 compatibility test report"
echo "" echo ""
echo "- Run: ${{ github.server_url }}/${{ github.repository }}/actions/runs/${{ github.run_id }}" echo "- Run: ${{ github.server_url }}/${{ github.repository }}/actions/runs/${{ github.run_id }}"
echo "- Attempt: ${GITHUB_RUN_ATTEMPT}"
echo "- Workflow Commit: ${GITHUB_SHA}"
echo "- Trigger: ${{ github.event_name }}" echo "- Trigger: ${{ github.event_name }}"
echo "- Package: ${PACKAGE_SOURCE}" echo "- Package: ${PACKAGE_SOURCE}"
echo "- RustFS Version: ${RUSTFS_VERSION_INFO}" echo "- RustFS Version: ${RUSTFS_VERSION_INFO}"
echo "- Test Step Outcome: ${RESULT}" echo "- Test Step Outcome: ${{ steps.test.outcome }}"
echo "- Suite Step Outcome: ${{ steps.test.outcome }}"
echo "" echo ""
if [ "${RESULT}" = "success" ]; then cat "${CASE_TABLE}" || true
cat "${CASE_TABLE}" echo ""
echo "" echo "## Log tail"
echo "## Log tail" echo '```text'
echo '```text' tail -n 200 "${LOG_FILE}" || true
tail -n 200 "${LOG_FILE}" echo '```'
echo '```'
else
echo "The suite or evidence validation failed. See this run's artifact for partial case results and suite.log."
fi
} | tee "${REPORT_FILE}" } | tee "${REPORT_FILE}"
cat "${REPORT_FILE}" >> "${GITHUB_STEP_SUMMARY}" cat "${REPORT_FILE}" >> "${GITHUB_STEP_SUMMARY}"
[ "${RESULT}" = "success" ]
- name: Upload functional report to dashboard - name: Upload functional report to dashboard
if: ${{ always() && steps.evidence.outcome == 'success' }} if: always()
continue-on-error: true continue-on-error: true
env: env:
GH_TOKEN: ${{ env.PF_TESTING_GH_TOKEN }} GH_TOKEN: ${{ env.PF_TESTING_GH_TOKEN }}
REPORT_FILE: /tmp/rustfs-s3-compat-report.md
SUITE: s3 SUITE: s3
run: | run: |
set -euo pipefail set -euo pipefail
@@ -191,32 +218,28 @@ jobs:
fi fi
DATE="$(date -u +%Y-%m-%d)" DATE="$(date -u +%Y-%m-%d)"
REPORT_PATH="functional-reports/${SUITE}/${DATE}.md" REPORT_PATH="functional-reports/${SUITE}/${DATE}.md"
# Base64-encode the report into a temp file and feed it to jq via CONTENT="$(python3 -c 'import base64,sys;print(base64.b64encode(open(sys.argv[1],"rb").read()).decode())' "${REPORT_FILE}")"
# --rawfile: large reports (e.g. pool) exceed the OS argv limit and
# make `jq --arg content "${CONTENT}"` fail with "Argument list too long".
B64_FILE="$(mktemp)"
python3 -c 'import base64,sys;print(base64.b64encode(open(sys.argv[1],"rb").read()).decode())' "${REPORT_FILE}" > "${B64_FILE}"
SHA="$(gh api "repos/rustfs/dashboard/contents/${REPORT_PATH}" -q '.sha' 2>/dev/null || true)" SHA="$(gh api "repos/rustfs/dashboard/contents/${REPORT_PATH}" -q '.sha' 2>/dev/null || true)"
if [ -n "${SHA}" ]; then if [ -n "${SHA}" ]; then
jq -n --arg msg "report(${SUITE}): ${DATE}" --rawfile content "${B64_FILE}" --arg sha "${SHA}" \ jq -n --arg msg "report(${SUITE}): ${DATE}" --arg content "${CONTENT}" --arg sha "${SHA}" \
'{message:$msg, content:($content|rtrimstr("\n")), sha:$sha}' \ '{message:$msg, content:$content, sha:$sha}' \
| gh api --method PUT "repos/rustfs/dashboard/contents/${REPORT_PATH}" --input - >/dev/null | gh api --method PUT "repos/rustfs/dashboard/contents/${REPORT_PATH}" --input - >/dev/null
else else
jq -n --arg msg "report(${SUITE}): ${DATE}" --rawfile content "${B64_FILE}" \ jq -n --arg msg "report(${SUITE}): ${DATE}" --arg content "${CONTENT}" \
'{message:$msg, content:($content|rtrimstr("\n"))}' \ '{message:$msg, content:$content}' \
| gh api --method PUT "repos/rustfs/dashboard/contents/${REPORT_PATH}" --input - >/dev/null | gh api --method PUT "repos/rustfs/dashboard/contents/${REPORT_PATH}" --input - >/dev/null
fi fi
rm -f "${B64_FILE}"
- name: File failure issue in rustfs/backlog - name: File failure issue in rustfs/backlog
if: ${{ always() && (failure() || steps.test.outcome == 'failure' || steps.test.outcome == 'cancelled') }} if: ${{ always() && (failure() || steps.test.outcome == 'failure' || steps.test.outcome == 'cancelled') }}
continue-on-error: true continue-on-error: true
env: env:
GH_TOKEN: ${{ secrets.PF_TESTING_GH_TOKEN }} GH_TOKEN: ${{ secrets.PF_TESTING_GH_TOKEN }}
EVIDENCE_OUTCOME: ${{ steps.evidence.outcome }}
SUITE: 's3' SUITE: 's3'
SUITE_LABEL: 'S3 compatibility' SUITE_LABEL: 'S3 compatibility'
RUN_URL: ${{ github.server_url }}/${{ github.repository }}/actions/runs/${{ github.run_id }} RUN_URL: ${{ github.server_url }}/${{ github.repository }}/actions/runs/${{ github.run_id }}
REPORT_FILE: '/tmp/rustfs-s3-compat-report.md'
LOG_FILE: '/tmp/rustfs-s3-compat.log'
run: | run: |
set -euo pipefail set -euo pipefail
if [ -z "${GH_TOKEN:-}" ]; then if [ -z "${GH_TOKEN:-}" ]; then
@@ -244,16 +267,14 @@ jobs:
echo "" echo ""
echo "- Suite: \`${SUITE}\`" echo "- Suite: \`${SUITE}\`"
echo "- Run: ${RUN_URL}" echo "- Run: ${RUN_URL}"
echo "- Attempt: ${GITHUB_RUN_ATTEMPT}"
echo "- Workflow Commit: ${GITHUB_SHA}"
echo "- Trigger: ${GITHUB_EVENT_NAME}" echo "- Trigger: ${GITHUB_EVENT_NAME}"
echo "- Date: $(date -u +%Y-%m-%d)" echo "- Date: $(date -u +%Y-%m-%d)"
echo "" echo ""
echo "## Report (errors and symptoms)" echo "## Report (errors and symptoms)"
echo "" echo ""
if [ "${EVIDENCE_OUTCOME}" = "success" ] && [ -s "${REPORT_FILE}" ]; then if [ -s "${REPORT_FILE}" ]; then
redact < "${REPORT_FILE}" redact < "${REPORT_FILE}"
elif [ "${EVIDENCE_OUTCOME}" = "success" ] && [ -s "${LOG_FILE:-}" ]; then elif [ -s "${LOG_FILE:-}" ]; then
echo "(report file missing; log tail below)" echo "(report file missing; log tail below)"
echo "" echo ""
tail -n 200 "${LOG_FILE}" | redact tail -n 200 "${LOG_FILE}" | redact
@@ -269,15 +290,14 @@ jobs:
echo "filed backlog issue for suite ${SUITE}" echo "filed backlog issue for suite ${SUITE}"
- name: Upload report and logs - name: Upload report and logs
if: ${{ always() && steps.evidence.outcome == 'success' }} if: always()
uses: actions/upload-artifact@b7c566a772e6b6bfb58ed0dc250532a479d7789f # v6 uses: actions/upload-artifact@b7c566a772e6b6bfb58ed0dc250532a479d7789f # v6
with: with:
name: rustfs-s3-compat-${{ github.run_id }}-${{ github.run_attempt }} name: rustfs-s3-compat-${{ github.run_id }}
path: | path: |
${{ env.FUNCTIONAL_ARTIFACTS_DIR }}/report.md /tmp/rustfs-s3-compat.log
${{ env.FUNCTIONAL_ARTIFACTS_DIR }}/suite.log /tmp/rustfs-s3-compat-report.md
${{ env.FUNCTIONAL_ARTIFACTS_DIR }}/cases.md if-no-files-found: warn
if-no-files-found: error
- name: Cleanup environment (after) - name: Cleanup environment (after)
if: always() if: always()
+10 -22
View File
@@ -77,14 +77,10 @@ jobs:
timeout-minutes: 360 timeout-minutes: 360
if: ${{ github.event_name == 'workflow_dispatch' || github.event_name == 'repository_dispatch' }} if: ${{ github.event_name == 'workflow_dispatch' || github.event_name == 'repository_dispatch' }}
steps: steps:
# Checkout the repository into its own subdirectory. Checking out at
# the workspace root would wipe the auto-testing clone above (that is
# exactly how run 33934141181 lost rustfs-security-test.sh).
- name: Checkout repository (for the OIDC live gate script) - name: Checkout repository (for the OIDC live gate script)
uses: actions/checkout@9c091bb21b7c1c1d1991bb908d89e4e9dddfe3e0 # v7 uses: actions/checkout@9c091bb21b7c1c1d1991bb908d89e4e9dddfe3e0 # v7
with: with:
persist-credentials: false persist-credentials: false
path: rustfs-repo
- name: Initialize security evidence - name: Initialize security evidence
id: evidence id: evidence
@@ -92,7 +88,7 @@ jobs:
set -euo pipefail set -euo pipefail
umask 077 umask 077
SECURITY_ARTIFACTS_DIR="${RUNNER_TEMP}/rustfs-security-${GITHUB_RUN_ID}-${GITHUB_RUN_ATTEMPT}" SECURITY_ARTIFACTS_DIR="${RUNNER_TEMP}/rustfs-security-${GITHUB_RUN_ID}-${GITHUB_RUN_ATTEMPT}"
mkdir -- "${SECURITY_ARTIFACTS_DIR}" "${SECURITY_ARTIFACTS_DIR}-scratch" mkdir -- "${SECURITY_ARTIFACTS_DIR}"
printf 'SECURITY_ARTIFACTS_DIR=%s\n' "${SECURITY_ARTIFACTS_DIR}" >> "${GITHUB_ENV}" printf 'SECURITY_ARTIFACTS_DIR=%s\n' "${SECURITY_ARTIFACTS_DIR}" >> "${GITHUB_ENV}"
# auto-testing is private: clone it with the dedicated PF token (not # auto-testing is private: clone it with the dedicated PF token (not
@@ -148,8 +144,8 @@ jobs:
continue-on-error: true continue-on-error: true
env: env:
REPORT_FILE: ${{ env.SECURITY_ARTIFACTS_DIR }}/suite-report.md REPORT_FILE: ${{ env.SECURITY_ARTIFACTS_DIR }}/suite-report.md
TMPDIR: ${{ env.SECURITY_ARTIFACTS_DIR }}-scratch TMPDIR: ${{ env.SECURITY_ARTIFACTS_DIR }}
RUSTFS_SECURITY_OIDC_LIVE_SCRIPT: ${{ github.workspace }}/rustfs-repo/scripts/test/oidc_keycloak_live.sh RUSTFS_SECURITY_OIDC_LIVE_SCRIPT: ${{ github.workspace }}/scripts/test/oidc_keycloak_live.sh
run: | run: |
set -euo pipefail set -euo pipefail
chmod +x auto-testing/rustfs-security-test.sh chmod +x auto-testing/rustfs-security-test.sh
@@ -172,7 +168,7 @@ jobs:
else else
ARGS+=(--package-url "${RUSTFS_NIGHTLY_PACKAGE_URL}") ARGS+=(--package-url "${RUSTFS_NIGHTLY_PACKAGE_URL}")
fi fi
GITHUB_STEP_SUMMARY=/dev/null ./auto-testing/rustfs-security-test.sh "${ARGS[@]}" 2>&1 | tee "${SECURITY_ARTIFACTS_DIR}/suite.log" GITHUB_STEP_SUMMARY=/dev/null ./auto-testing/rustfs-security-test.sh "${ARGS[@]}"
- name: Generate report - name: Generate report
id: report id: report
@@ -223,22 +219,17 @@ jobs:
fi fi
DATE="$(date -u +%Y-%m-%d)" DATE="$(date -u +%Y-%m-%d)"
REPORT_PATH="functional-reports/${SUITE}/${DATE}.md" REPORT_PATH="functional-reports/${SUITE}/${DATE}.md"
# Base64-encode the report into a temp file and feed it to jq via CONTENT="$(python3 -c 'import base64,sys;print(base64.b64encode(open(sys.argv[1],"rb").read()).decode())' "${REPORT_FILE}")"
# --rawfile: large reports (e.g. pool) exceed the OS argv limit and
# make `jq --arg content "${CONTENT}"` fail with "Argument list too long".
B64_FILE="$(mktemp)"
python3 -c 'import base64,sys;print(base64.b64encode(open(sys.argv[1],"rb").read()).decode())' "${REPORT_FILE}" > "${B64_FILE}"
SHA="$(gh api "repos/rustfs/dashboard/contents/${REPORT_PATH}" -q '.sha' 2>/dev/null || true)" SHA="$(gh api "repos/rustfs/dashboard/contents/${REPORT_PATH}" -q '.sha' 2>/dev/null || true)"
if [ -n "${SHA}" ]; then if [ -n "${SHA}" ]; then
jq -n --arg msg "report(${SUITE}): ${DATE}" --rawfile content "${B64_FILE}" --arg sha "${SHA}" \ jq -n --arg msg "report(${SUITE}): ${DATE}" --arg content "${CONTENT}" --arg sha "${SHA}" \
'{message:$msg, content:($content|rtrimstr("\n")), sha:$sha}' \ '{message:$msg, content:$content, sha:$sha}' \
| gh api --method PUT "repos/rustfs/dashboard/contents/${REPORT_PATH}" --input - >/dev/null | gh api --method PUT "repos/rustfs/dashboard/contents/${REPORT_PATH}" --input - >/dev/null
else else
jq -n --arg msg "report(${SUITE}): ${DATE}" --rawfile content "${B64_FILE}" \ jq -n --arg msg "report(${SUITE}): ${DATE}" --arg content "${CONTENT}" \
'{message:$msg, content:($content|rtrimstr("\n"))}' \ '{message:$msg, content:$content}' \
| gh api --method PUT "repos/rustfs/dashboard/contents/${REPORT_PATH}" --input - >/dev/null | gh api --method PUT "repos/rustfs/dashboard/contents/${REPORT_PATH}" --input - >/dev/null
fi fi
rm -f "${B64_FILE}"
- name: File failure issue in rustfs/backlog - name: File failure issue in rustfs/backlog
if: ${{ always() && (failure() || steps.test.outcome == 'failure' || steps.test.outcome == 'cancelled') }} if: ${{ always() && (failure() || steps.test.outcome == 'failure' || steps.test.outcome == 'cancelled') }}
@@ -305,10 +296,7 @@ jobs:
uses: actions/upload-artifact@b7c566a772e6b6bfb58ed0dc250532a479d7789f # v6 uses: actions/upload-artifact@b7c566a772e6b6bfb58ed0dc250532a479d7789f # v6
with: with:
name: rustfs-security-test-${{ github.run_id }}-${{ github.run_attempt }} name: rustfs-security-test-${{ github.run_id }}-${{ github.run_attempt }}
path: | path: ${{ env.SECURITY_ARTIFACTS_DIR }}/
${{ env.SECURITY_ARTIFACTS_DIR }}/report.md
${{ env.SECURITY_ARTIFACTS_DIR }}/suite.log
${{ env.SECURITY_ARTIFACTS_DIR }}/suite-report.md
if-no-files-found: error if-no-files-found: error
retention-days: 3 retention-days: 3
+84 -64
View File
@@ -46,28 +46,10 @@ env:
jobs: jobs:
storage-test: storage-test:
runs-on: smoke-testing runs-on: smoke-testing
continue-on-error: true
timeout-minutes: 360 timeout-minutes: 360
if: ${{ github.event_name == 'workflow_dispatch' || github.event_name == 'repository_dispatch' }} if: ${{ github.event_name == 'workflow_dispatch' || github.event_name == 'repository_dispatch' }}
steps: steps:
- name: Checkout repository (for report parser)
uses: actions/checkout@9c091bb21b7c1c1d1991bb908d89e4e9dddfe3e0 # v7
with:
persist-credentials: false
- name: Initialize functional evidence
id: evidence
run: |
set -euo pipefail
umask 077
FUNCTIONAL_ARTIFACTS_DIR="${RUNNER_TEMP}/rustfs-storage-${GITHUB_RUN_ID}-${GITHUB_RUN_ATTEMPT}"
mkdir -- "${FUNCTIONAL_ARTIFACTS_DIR}" "${FUNCTIONAL_ARTIFACTS_DIR}-scratch"
{
printf 'FUNCTIONAL_ARTIFACTS_DIR=%s\n' "${FUNCTIONAL_ARTIFACTS_DIR}"
printf 'LOG_FILE=%s/suite.log\n' "${FUNCTIONAL_ARTIFACTS_DIR}"
printf 'REPORT_FILE=%s/report.md\n' "${FUNCTIONAL_ARTIFACTS_DIR}"
printf 'TMPDIR=%s-scratch\n' "${FUNCTIONAL_ARTIFACTS_DIR}"
} >> "${GITHUB_ENV}"
# auto-testing is private: clone it with the dedicated PF token (not # auto-testing is private: clone it with the dedicated PF token (not
# GITHUB_TOKEN) and retry transient GitHub/network failures. # GITHUB_TOKEN) and retry transient GitHub/network failures.
- name: Checkout auto-testing scripts (with retry) - name: Checkout auto-testing scripts (with retry)
@@ -115,6 +97,9 @@ jobs:
- name: Run storage engine suite - name: Run storage engine suite
id: test id: test
continue-on-error: true
env:
LOG_FILE: /tmp/rustfs-storage.log
run: | run: |
set -euo pipefail set -euo pipefail
chmod +x auto-testing/rustfs-storage-test.sh chmod +x auto-testing/rustfs-storage-test.sh
@@ -137,7 +122,10 @@ jobs:
./auto-testing/rustfs-storage-test.sh "${ARGS[@]}" ./auto-testing/rustfs-storage-test.sh "${ARGS[@]}"
- name: Generate report - name: Generate report
if: ${{ always() && steps.evidence.outcome == 'success' }} if: always()
env:
LOG_FILE: /tmp/rustfs-storage.log
REPORT_FILE: /tmp/rustfs-storage-report.md
run: | run: |
set -euo pipefail set -euo pipefail
PACKAGE_URL='${{ inputs.package_url }}' PACKAGE_URL='${{ inputs.package_url }}'
@@ -159,44 +147,83 @@ jobs:
RUSTFS_VERSION_INFO="${DETECTED_VERSION}" RUSTFS_VERSION_INFO="${DETECTED_VERSION}"
fi fi
fi fi
CASE_TABLE="${FUNCTIONAL_ARTIFACTS_DIR}/cases.md" CASE_TABLE="/tmp/rustfs-storage-cases.md"
CASE_RESULT=success python3 - "${LOG_FILE}" "${CASE_TABLE}" <<'PY'
python3 scripts/functional_case_report.py "${LOG_FILE}" "${CASE_TABLE}" || CASE_RESULT=failure import re
RESULT=failure import sys
if [ '${{ steps.test.outcome }}' = 'success' ] && [ "${CASE_RESULT}" = 'success' ]; then
RESULT=success log_file, out_file = sys.argv[1], sys.argv[2]
fi ansi = re.compile(r'\x1b\[[0-9;]*m')
start_re = re.compile(r'^---\s+([A-Z0-9]+-[0-9]+)\s+(.+?)\s+---$')
done_re = re.compile(r'^\[(PASS|FAIL|UNSUPPORTED)\]\s+([A-Z0-9]+-[0-9]+)\b')
rows = []
index = {}
current = None
try:
with open(log_file, 'r', encoding='utf-8', errors='replace') as fh:
for raw in fh:
line = ansi.sub('', raw).strip()
m = start_re.match(line)
if m:
case_id, name = m.group(1), m.group(2)
current = case_id
if case_id not in index:
index[case_id] = len(rows)
rows.append([case_id, name, 'RUNNING'])
continue
m = done_re.match(line)
if m:
status, case_id = m.group(1), m.group(2)
if case_id in index:
rows[index[case_id]][2] = status
else:
rows.append([case_id, case_id, status])
index[case_id] = len(rows) - 1
current = None
except FileNotFoundError:
rows = []
counts = {'PASS': 0, 'FAIL': 0, 'UNSUPPORTED': 0, 'RUNNING': 0}
for _, _, status in rows:
counts[status] = counts.get(status, 0) + 1
with open(out_file, 'w', encoding='utf-8') as out:
out.write('## Case Summary\n\n')
out.write(f"- Total: {len(rows)}\\n")
out.write(f"- PASS: {counts.get('PASS', 0)}\\n")
out.write(f"- FAIL: {counts.get('FAIL', 0)}\\n")
out.write(f"- UNSUPPORTED: {counts.get('UNSUPPORTED', 0)}\\n")
out.write('\\n')
out.write('| Case | Name | Status |\\n')
out.write('| --- | --- | --- |\\n')
for case_id, name, status in rows:
out.write(f'| {case_id} | {name} | {status} |\\n')
PY
{ {
echo "# RustFS storage engine test report" echo "# RustFS storage engine test report"
echo "" echo ""
echo "- Run: ${{ github.server_url }}/${{ github.repository }}/actions/runs/${{ github.run_id }}" echo "- Run: ${{ github.server_url }}/${{ github.repository }}/actions/runs/${{ github.run_id }}"
echo "- Attempt: ${GITHUB_RUN_ATTEMPT}"
echo "- Workflow Commit: ${GITHUB_SHA}"
echo "- Trigger: ${{ github.event_name }}" echo "- Trigger: ${{ github.event_name }}"
echo "- Package: ${PACKAGE_SOURCE}" echo "- Package: ${PACKAGE_SOURCE}"
echo "- RustFS Version: ${RUSTFS_VERSION_INFO}" echo "- RustFS Version: ${RUSTFS_VERSION_INFO}"
echo "- Test Step Outcome: ${RESULT}" echo "- Test Step Outcome: ${{ steps.test.outcome }}"
echo "- Suite Step Outcome: ${{ steps.test.outcome }}"
echo "" echo ""
if [ "${RESULT}" = "success" ]; then cat "${CASE_TABLE}" || true
cat "${CASE_TABLE}" echo ""
echo "" echo "## Log tail"
echo "## Log tail" echo '```text'
echo '```text' tail -n 200 "${LOG_FILE}" || true
tail -n 200 "${LOG_FILE}" echo '```'
echo '```'
else
echo "The suite or evidence validation failed. See this run's artifact for partial case results and suite.log."
fi
} | tee "${REPORT_FILE}" } | tee "${REPORT_FILE}"
cat "${REPORT_FILE}" >> "${GITHUB_STEP_SUMMARY}" cat "${REPORT_FILE}" >> "${GITHUB_STEP_SUMMARY}"
[ "${RESULT}" = "success" ]
- name: Upload functional report to dashboard - name: Upload functional report to dashboard
if: ${{ always() && steps.evidence.outcome == 'success' }} if: always()
continue-on-error: true continue-on-error: true
env: env:
GH_TOKEN: ${{ env.PF_TESTING_GH_TOKEN }} GH_TOKEN: ${{ env.PF_TESTING_GH_TOKEN }}
REPORT_FILE: /tmp/rustfs-storage-report.md
SUITE: storage SUITE: storage
run: | run: |
set -euo pipefail set -euo pipefail
@@ -206,32 +233,28 @@ jobs:
fi fi
DATE="$(date -u +%Y-%m-%d)" DATE="$(date -u +%Y-%m-%d)"
REPORT_PATH="functional-reports/${SUITE}/${DATE}.md" REPORT_PATH="functional-reports/${SUITE}/${DATE}.md"
# Base64-encode the report into a temp file and feed it to jq via CONTENT="$(python3 -c 'import base64,sys;print(base64.b64encode(open(sys.argv[1],"rb").read()).decode())' "${REPORT_FILE}")"
# --rawfile: large reports (e.g. pool) exceed the OS argv limit and
# make `jq --arg content "${CONTENT}"` fail with "Argument list too long".
B64_FILE="$(mktemp)"
python3 -c 'import base64,sys;print(base64.b64encode(open(sys.argv[1],"rb").read()).decode())' "${REPORT_FILE}" > "${B64_FILE}"
SHA="$(gh api "repos/rustfs/dashboard/contents/${REPORT_PATH}" -q '.sha' 2>/dev/null || true)" SHA="$(gh api "repos/rustfs/dashboard/contents/${REPORT_PATH}" -q '.sha' 2>/dev/null || true)"
if [ -n "${SHA}" ]; then if [ -n "${SHA}" ]; then
jq -n --arg msg "report(${SUITE}): ${DATE}" --rawfile content "${B64_FILE}" --arg sha "${SHA}" \ jq -n --arg msg "report(${SUITE}): ${DATE}" --arg content "${CONTENT}" --arg sha "${SHA}" \
'{message:$msg, content:($content|rtrimstr("\n")), sha:$sha}' \ '{message:$msg, content:$content, sha:$sha}' \
| gh api --method PUT "repos/rustfs/dashboard/contents/${REPORT_PATH}" --input - >/dev/null | gh api --method PUT "repos/rustfs/dashboard/contents/${REPORT_PATH}" --input - >/dev/null
else else
jq -n --arg msg "report(${SUITE}): ${DATE}" --rawfile content "${B64_FILE}" \ jq -n --arg msg "report(${SUITE}): ${DATE}" --arg content "${CONTENT}" \
'{message:$msg, content:($content|rtrimstr("\n"))}' \ '{message:$msg, content:$content}' \
| gh api --method PUT "repos/rustfs/dashboard/contents/${REPORT_PATH}" --input - >/dev/null | gh api --method PUT "repos/rustfs/dashboard/contents/${REPORT_PATH}" --input - >/dev/null
fi fi
rm -f "${B64_FILE}"
- name: File failure issue in rustfs/backlog - name: File failure issue in rustfs/backlog
if: ${{ always() && (failure() || steps.test.outcome == 'failure' || steps.test.outcome == 'cancelled') }} if: ${{ always() && (failure() || steps.test.outcome == 'failure' || steps.test.outcome == 'cancelled') }}
continue-on-error: true continue-on-error: true
env: env:
GH_TOKEN: ${{ secrets.PF_TESTING_GH_TOKEN }} GH_TOKEN: ${{ secrets.PF_TESTING_GH_TOKEN }}
EVIDENCE_OUTCOME: ${{ steps.evidence.outcome }}
SUITE: 'storage' SUITE: 'storage'
SUITE_LABEL: 'Storage engine' SUITE_LABEL: 'Storage engine'
RUN_URL: ${{ github.server_url }}/${{ github.repository }}/actions/runs/${{ github.run_id }} RUN_URL: ${{ github.server_url }}/${{ github.repository }}/actions/runs/${{ github.run_id }}
REPORT_FILE: '/tmp/rustfs-storage-report.md'
LOG_FILE: '/tmp/rustfs-storage.log'
run: | run: |
set -euo pipefail set -euo pipefail
if [ -z "${GH_TOKEN:-}" ]; then if [ -z "${GH_TOKEN:-}" ]; then
@@ -259,16 +282,14 @@ jobs:
echo "" echo ""
echo "- Suite: \`${SUITE}\`" echo "- Suite: \`${SUITE}\`"
echo "- Run: ${RUN_URL}" echo "- Run: ${RUN_URL}"
echo "- Attempt: ${GITHUB_RUN_ATTEMPT}"
echo "- Workflow Commit: ${GITHUB_SHA}"
echo "- Trigger: ${GITHUB_EVENT_NAME}" echo "- Trigger: ${GITHUB_EVENT_NAME}"
echo "- Date: $(date -u +%Y-%m-%d)" echo "- Date: $(date -u +%Y-%m-%d)"
echo "" echo ""
echo "## Report (errors and symptoms)" echo "## Report (errors and symptoms)"
echo "" echo ""
if [ "${EVIDENCE_OUTCOME}" = "success" ] && [ -s "${REPORT_FILE}" ]; then if [ -s "${REPORT_FILE}" ]; then
redact < "${REPORT_FILE}" redact < "${REPORT_FILE}"
elif [ "${EVIDENCE_OUTCOME}" = "success" ] && [ -s "${LOG_FILE:-}" ]; then elif [ -s "${LOG_FILE:-}" ]; then
echo "(report file missing; log tail below)" echo "(report file missing; log tail below)"
echo "" echo ""
tail -n 200 "${LOG_FILE}" | redact tail -n 200 "${LOG_FILE}" | redact
@@ -284,15 +305,14 @@ jobs:
echo "filed backlog issue for suite ${SUITE}" echo "filed backlog issue for suite ${SUITE}"
- name: Upload report and logs - name: Upload report and logs
if: ${{ always() && steps.evidence.outcome == 'success' }} if: always()
uses: actions/upload-artifact@b7c566a772e6b6bfb58ed0dc250532a479d7789f # v6 uses: actions/upload-artifact@b7c566a772e6b6bfb58ed0dc250532a479d7789f # v6
with: with:
name: rustfs-storage-${{ github.run_id }}-${{ github.run_attempt }} name: rustfs-storage-${{ github.run_id }}
path: | path: |
${{ env.FUNCTIONAL_ARTIFACTS_DIR }}/report.md /tmp/rustfs-storage.log
${{ env.FUNCTIONAL_ARTIFACTS_DIR }}/suite.log /tmp/rustfs-storage-report.md
${{ env.FUNCTIONAL_ARTIFACTS_DIR }}/cases.md if-no-files-found: warn
if-no-files-found: error
- name: Cleanup environment (after) - name: Cleanup environment (after)
if: always() if: always()
+8 -10
View File
@@ -61,6 +61,9 @@ env:
jobs: jobs:
tier-test: tier-test:
runs-on: smoke-testing runs-on: smoke-testing
# Requirement: a failing suite must not fail the workflow; failures
# are filed to rustfs/backlog and the chain continues.
continue-on-error: true
timeout-minutes: 420 timeout-minutes: 420
if: ${{ github.event_name == 'workflow_dispatch' || github.event_name == 'repository_dispatch' }} if: ${{ github.event_name == 'workflow_dispatch' || github.event_name == 'repository_dispatch' }}
steps: steps:
@@ -377,22 +380,17 @@ jobs:
fi fi
DATE="$(date -u +%Y-%m-%d)" DATE="$(date -u +%Y-%m-%d)"
REPORT_PATH="functional-reports/${SUITE}/${DATE}.md" REPORT_PATH="functional-reports/${SUITE}/${DATE}.md"
# Base64-encode the report into a temp file and feed it to jq via CONTENT="$(python3 -c 'import base64,sys;print(base64.b64encode(open(sys.argv[1],"rb").read()).decode())' "${REPORT_FILE}")"
# --rawfile: large reports (e.g. pool) exceed the OS argv limit and
# make `jq --arg content "${CONTENT}"` fail with "Argument list too long".
B64_FILE="$(mktemp)"
python3 -c 'import base64,sys;print(base64.b64encode(open(sys.argv[1],"rb").read()).decode())' "${REPORT_FILE}" > "${B64_FILE}"
SHA="$(gh api "repos/rustfs/dashboard/contents/${REPORT_PATH}" -q '.sha' 2>/dev/null || true)" SHA="$(gh api "repos/rustfs/dashboard/contents/${REPORT_PATH}" -q '.sha' 2>/dev/null || true)"
if [ -n "${SHA}" ]; then if [ -n "${SHA}" ]; then
jq -n --arg msg "report(${SUITE}): ${DATE}" --rawfile content "${B64_FILE}" --arg sha "${SHA}" \ jq -n --arg msg "report(${SUITE}): ${DATE}" --arg content "${CONTENT}" --arg sha "${SHA}" \
'{message:$msg, content:($content|rtrimstr("\n")), sha:$sha}' \ '{message:$msg, content:$content, sha:$sha}' \
| gh api --method PUT "repos/rustfs/dashboard/contents/${REPORT_PATH}" --input - >/dev/null | gh api --method PUT "repos/rustfs/dashboard/contents/${REPORT_PATH}" --input - >/dev/null
else else
jq -n --arg msg "report(${SUITE}): ${DATE}" --rawfile content "${B64_FILE}" \ jq -n --arg msg "report(${SUITE}): ${DATE}" --arg content "${CONTENT}" \
'{message:$msg, content:($content|rtrimstr("\n"))}' \ '{message:$msg, content:$content}' \
| gh api --method PUT "repos/rustfs/dashboard/contents/${REPORT_PATH}" --input - >/dev/null | gh api --method PUT "repos/rustfs/dashboard/contents/${REPORT_PATH}" --input - >/dev/null
fi fi
rm -f "${B64_FILE}"
- name: Verify required tier evidence - name: Verify required tier evidence
id: evidence_verify id: evidence_verify
+103 -68
View File
@@ -79,28 +79,10 @@ env:
jobs: jobs:
upgrade-test: upgrade-test:
runs-on: smoke-testing runs-on: smoke-testing
continue-on-error: true
timeout-minutes: 420 timeout-minutes: 420
if: ${{ github.event_name == 'workflow_dispatch' || github.event_name == 'repository_dispatch' }} if: ${{ github.event_name == 'workflow_dispatch' || github.event_name == 'repository_dispatch' }}
steps: steps:
- name: Checkout repository (for report parser)
uses: actions/checkout@9c091bb21b7c1c1d1991bb908d89e4e9dddfe3e0 # v7
with:
persist-credentials: false
- name: Initialize functional evidence
id: evidence
run: |
set -euo pipefail
umask 077
FUNCTIONAL_ARTIFACTS_DIR="${RUNNER_TEMP}/rustfs-upgrade-${GITHUB_RUN_ID}-${GITHUB_RUN_ATTEMPT}"
mkdir -- "${FUNCTIONAL_ARTIFACTS_DIR}" "${FUNCTIONAL_ARTIFACTS_DIR}-scratch"
{
printf 'FUNCTIONAL_ARTIFACTS_DIR=%s\n' "${FUNCTIONAL_ARTIFACTS_DIR}"
printf 'LOG_FILE=%s/suite.log\n' "${FUNCTIONAL_ARTIFACTS_DIR}"
printf 'REPORT_FILE=%s/report.md\n' "${FUNCTIONAL_ARTIFACTS_DIR}"
printf 'TMPDIR=%s-scratch\n' "${FUNCTIONAL_ARTIFACTS_DIR}"
} >> "${GITHUB_ENV}"
# auto-testing is private: clone it with the dedicated PF token (not # auto-testing is private: clone it with the dedicated PF token (not
# GITHUB_TOKEN) and retry transient GitHub/network failures. # GITHUB_TOKEN) and retry transient GitHub/network failures.
- name: Checkout auto-testing scripts (with retry) - name: Checkout auto-testing scripts (with retry)
@@ -160,7 +142,9 @@ jobs:
- name: Run upgrade compatibility suite - name: Run upgrade compatibility suite
id: test id: test
continue-on-error: true
env: env:
LOG_FILE: /tmp/rustfs-upgrade.log
GH_TOKEN: ${{ secrets.PF_TESTING_GH_TOKEN }} GH_TOKEN: ${{ secrets.PF_TESTING_GH_TOKEN }}
run: | run: |
set -euo pipefail set -euo pipefail
@@ -218,7 +202,10 @@ jobs:
./auto-testing/rustfs-upgrade-test.sh "${ARGS[@]}" ./auto-testing/rustfs-upgrade-test.sh "${ARGS[@]}"
- name: Generate report - name: Generate report
if: ${{ always() && steps.evidence.outcome == 'success' }} if: always()
env:
LOG_FILE: /tmp/rustfs-upgrade.log
REPORT_FILE: /tmp/rustfs-upgrade-report.md
run: | run: |
set -euo pipefail set -euo pipefail
FROM_URL='${{ inputs.from_url }}' FROM_URL='${{ inputs.from_url }}'
@@ -239,47 +226,103 @@ jobs:
else else
TO_SOURCE="${RUSTFS_NIGHTLY_PACKAGE_URL}" TO_SOURCE="${RUSTFS_NIGHTLY_PACKAGE_URL}"
fi fi
CASE_TABLE="${FUNCTIONAL_ARTIFACTS_DIR}/cases.md" CASE_TABLE="/tmp/rustfs-upgrade-cases.md"
MATRIX_TABLE="${FUNCTIONAL_ARTIFACTS_DIR}/matrix.md" MATRIX_TABLE="/tmp/rustfs-upgrade-matrix.md"
CASE_RESULT=success python3 - "${LOG_FILE}" "${CASE_TABLE}" "${MATRIX_TABLE}" <<'PY'
python3 scripts/functional_case_report.py "${LOG_FILE}" "${CASE_TABLE}" "${MATRIX_TABLE}" || CASE_RESULT=failure import re
RESULT=failure import sys
if [ '${{ steps.test.outcome }}' = 'success' ] && [ "${CASE_RESULT}" = 'success' ]; then
RESULT=success log_file, out_file, matrix_file = sys.argv[1], sys.argv[2], sys.argv[3]
fi ansi = re.compile(r'\x1b\[[0-9;]*m')
start_re = re.compile(r'^---\s+([A-Z]+-[0-9]+)\s+(.+?)\s+---$')
done_re = re.compile(r'^\[(PASS|FAIL|UNSUPPORTED)\]\s+([A-Z]+-[0-9]+)\b')
topo_re = re.compile(
r'^\[UPG-TOPO\]\s+(\S+)\s+(\S+)\s+(\S+)\s+(\S+)\s+PASS=(\d+)\s+FAIL=(\d+)\s*$')
rows = []
index = {}
topo_rows = []
try:
with open(log_file, 'r', encoding='utf-8', errors='replace') as fh:
for raw in fh:
line = ansi.sub('', raw).strip()
m = topo_re.match(line)
if m:
topo_rows.append(m.groups())
continue
m = start_re.match(line)
if m:
case_id, name = m.group(1), m.group(2)
if case_id not in index:
index[case_id] = len(rows)
rows.append([case_id, name, 'RUNNING'])
continue
m = done_re.match(line)
if m:
status, case_id = m.group(1), m.group(2)
if case_id in index:
rows[index[case_id]][2] = status
else:
rows.append([case_id, case_id, status])
index[case_id] = len(rows) - 1
except FileNotFoundError:
rows = []
counts = {'PASS': 0, 'FAIL': 0, 'UNSUPPORTED': 0, 'RUNNING': 0}
for _, _, status in rows:
counts[status] = counts.get(status, 0) + 1
with open(out_file, 'w', encoding='utf-8') as out:
out.write('## Case Summary\n\n')
out.write(f"- Total: {len(rows)}\\n")
out.write(f"- PASS: {counts.get('PASS', 0)}\\n")
out.write(f"- FAIL: {counts.get('FAIL', 0)}\\n")
out.write(f"- UNSUPPORTED: {counts.get('UNSUPPORTED', 0)}\\n")
out.write('\\n')
out.write('| Case | Name | Status |\\n')
out.write('| --- | --- | --- |\\n')
for case_id, name, status in rows:
out.write(f'| {case_id} | {name} | {status} |\\n')
# Upgrade matrix: one row per topology/backend with the versions
# captured on the nodes (rustfs --version) and the aggregated
# result. The dashboard renders this table directly.
with open(matrix_file, 'w', encoding='utf-8') as out:
out.write('## Upgrade Matrix\n\n')
out.write('| Topology | KMS Backend | From Version | To Version | Result |\n')
out.write('| --- | --- | --- | --- | --- |\n')
for topo, backend, old_v, new_v, npass, nfail in topo_rows:
result = 'PASS' if nfail == '0' else 'FAIL'
out.write(f'| {topo} | {backend} | {old_v} | {new_v} | {result} (PASS={npass} FAIL={nfail}) |\n')
if not topo_rows:
out.write('| - | - | - | - | NOT RUN (suite failed before upgrade) |\n')
PY
{ {
echo "# RustFS upgrade compatibility report" echo "# RustFS upgrade compatibility report"
echo "" echo ""
echo "- Run: ${{ github.server_url }}/${{ github.repository }}/actions/runs/${{ github.run_id }}" echo "- Run: ${{ github.server_url }}/${{ github.repository }}/actions/runs/${{ github.run_id }}"
echo "- Attempt: ${GITHUB_RUN_ATTEMPT}"
echo "- Workflow Commit: ${GITHUB_SHA}"
echo "- Trigger: ${{ github.event_name }}" echo "- Trigger: ${{ github.event_name }}"
echo "- From: ${FROM_SOURCE}" echo "- From: ${FROM_SOURCE}"
echo "- To: ${TO_SOURCE}" echo "- To: ${TO_SOURCE}"
echo "- Test Step Outcome: ${RESULT}" echo "- Test Step Outcome: ${{ steps.test.outcome }}"
echo "- Suite Step Outcome: ${{ steps.test.outcome }}"
echo "" echo ""
if [ "${RESULT}" = "success" ]; then cat "${MATRIX_TABLE}" || true
cat "${MATRIX_TABLE}" echo ""
echo "" cat "${CASE_TABLE}" || true
cat "${CASE_TABLE}" echo ""
echo "" echo "## Log tail"
echo "## Log tail" echo '```text'
echo '```text' tail -n 200 "${LOG_FILE}" || true
tail -n 200 "${LOG_FILE}" echo '```'
echo '```'
else
echo "The suite or evidence validation failed. See this run's artifact for partial case results and suite.log."
fi
} | tee "${REPORT_FILE}" } | tee "${REPORT_FILE}"
cat "${REPORT_FILE}" >> "${GITHUB_STEP_SUMMARY}" cat "${REPORT_FILE}" >> "${GITHUB_STEP_SUMMARY}"
[ "${RESULT}" = "success" ]
- name: Upload functional report to dashboard - name: Upload functional report to dashboard
if: ${{ always() && steps.evidence.outcome == 'success' }} if: always()
continue-on-error: true continue-on-error: true
env: env:
GH_TOKEN: ${{ env.PF_TESTING_GH_TOKEN }} GH_TOKEN: ${{ env.PF_TESTING_GH_TOKEN }}
REPORT_FILE: /tmp/rustfs-upgrade-report.md
SUITE: upgrade SUITE: upgrade
run: | run: |
set -euo pipefail set -euo pipefail
@@ -289,32 +332,28 @@ jobs:
fi fi
DATE="$(date -u +%Y-%m-%d)" DATE="$(date -u +%Y-%m-%d)"
REPORT_PATH="functional-reports/${SUITE}/${DATE}.md" REPORT_PATH="functional-reports/${SUITE}/${DATE}.md"
# Base64-encode the report into a temp file and feed it to jq via CONTENT="$(python3 -c 'import base64,sys;print(base64.b64encode(open(sys.argv[1],"rb").read()).decode())' "${REPORT_FILE}")"
# --rawfile: large reports (e.g. pool) exceed the OS argv limit and
# make `jq --arg content "${CONTENT}"` fail with "Argument list too long".
B64_FILE="$(mktemp)"
python3 -c 'import base64,sys;print(base64.b64encode(open(sys.argv[1],"rb").read()).decode())' "${REPORT_FILE}" > "${B64_FILE}"
SHA="$(gh api "repos/rustfs/dashboard/contents/${REPORT_PATH}" -q '.sha' 2>/dev/null || true)" SHA="$(gh api "repos/rustfs/dashboard/contents/${REPORT_PATH}" -q '.sha' 2>/dev/null || true)"
if [ -n "${SHA}" ]; then if [ -n "${SHA}" ]; then
jq -n --arg msg "report(${SUITE}): ${DATE}" --rawfile content "${B64_FILE}" --arg sha "${SHA}" \ jq -n --arg msg "report(${SUITE}): ${DATE}" --arg content "${CONTENT}" --arg sha "${SHA}" \
'{message:$msg, content:($content|rtrimstr("\n")), sha:$sha}' \ '{message:$msg, content:$content, sha:$sha}' \
| gh api --method PUT "repos/rustfs/dashboard/contents/${REPORT_PATH}" --input - >/dev/null | gh api --method PUT "repos/rustfs/dashboard/contents/${REPORT_PATH}" --input - >/dev/null
else else
jq -n --arg msg "report(${SUITE}): ${DATE}" --rawfile content "${B64_FILE}" \ jq -n --arg msg "report(${SUITE}): ${DATE}" --arg content "${CONTENT}" \
'{message:$msg, content:($content|rtrimstr("\n"))}' \ '{message:$msg, content:$content}' \
| gh api --method PUT "repos/rustfs/dashboard/contents/${REPORT_PATH}" --input - >/dev/null | gh api --method PUT "repos/rustfs/dashboard/contents/${REPORT_PATH}" --input - >/dev/null
fi fi
rm -f "${B64_FILE}"
- name: File failure issue in rustfs/backlog - name: File failure issue in rustfs/backlog
if: ${{ always() && (failure() || steps.test.outcome == 'failure' || steps.test.outcome == 'cancelled') }} if: ${{ always() && (failure() || steps.test.outcome == 'failure' || steps.test.outcome == 'cancelled') }}
continue-on-error: true continue-on-error: true
env: env:
GH_TOKEN: ${{ secrets.PF_TESTING_GH_TOKEN }} GH_TOKEN: ${{ secrets.PF_TESTING_GH_TOKEN }}
EVIDENCE_OUTCOME: ${{ steps.evidence.outcome }}
SUITE: 'upgrade' SUITE: 'upgrade'
SUITE_LABEL: 'Upgrade compatibility' SUITE_LABEL: 'Upgrade compatibility'
RUN_URL: ${{ github.server_url }}/${{ github.repository }}/actions/runs/${{ github.run_id }} RUN_URL: ${{ github.server_url }}/${{ github.repository }}/actions/runs/${{ github.run_id }}
REPORT_FILE: '/tmp/rustfs-upgrade-report.md'
LOG_FILE: '/tmp/rustfs-upgrade.log'
run: | run: |
set -euo pipefail set -euo pipefail
if [ -z "${GH_TOKEN:-}" ]; then if [ -z "${GH_TOKEN:-}" ]; then
@@ -342,16 +381,14 @@ jobs:
echo "" echo ""
echo "- Suite: \`${SUITE}\`" echo "- Suite: \`${SUITE}\`"
echo "- Run: ${RUN_URL}" echo "- Run: ${RUN_URL}"
echo "- Attempt: ${GITHUB_RUN_ATTEMPT}"
echo "- Workflow Commit: ${GITHUB_SHA}"
echo "- Trigger: ${GITHUB_EVENT_NAME}" echo "- Trigger: ${GITHUB_EVENT_NAME}"
echo "- Date: $(date -u +%Y-%m-%d)" echo "- Date: $(date -u +%Y-%m-%d)"
echo "" echo ""
echo "## Report (errors and symptoms)" echo "## Report (errors and symptoms)"
echo "" echo ""
if [ "${EVIDENCE_OUTCOME}" = "success" ] && [ -s "${REPORT_FILE}" ]; then if [ -s "${REPORT_FILE}" ]; then
redact < "${REPORT_FILE}" redact < "${REPORT_FILE}"
elif [ "${EVIDENCE_OUTCOME}" = "success" ] && [ -s "${LOG_FILE:-}" ]; then elif [ -s "${LOG_FILE:-}" ]; then
echo "(report file missing; log tail below)" echo "(report file missing; log tail below)"
echo "" echo ""
tail -n 200 "${LOG_FILE}" | redact tail -n 200 "${LOG_FILE}" | redact
@@ -367,16 +404,14 @@ jobs:
echo "filed backlog issue for suite ${SUITE}" echo "filed backlog issue for suite ${SUITE}"
- name: Upload report and logs - name: Upload report and logs
if: ${{ always() && steps.evidence.outcome == 'success' }} if: always()
uses: actions/upload-artifact@b7c566a772e6b6bfb58ed0dc250532a479d7789f # v6 uses: actions/upload-artifact@b7c566a772e6b6bfb58ed0dc250532a479d7789f # v6
with: with:
name: rustfs-upgrade-test-${{ github.run_id }}-${{ github.run_attempt }} name: rustfs-upgrade-test-${{ github.run_id }}
path: | path: |
${{ env.FUNCTIONAL_ARTIFACTS_DIR }}/report.md /tmp/rustfs-upgrade-report.md
${{ env.FUNCTIONAL_ARTIFACTS_DIR }}/suite.log /tmp/rustfs-upgrade.*/*
${{ env.FUNCTIONAL_ARTIFACTS_DIR }}/cases.md if-no-files-found: ignore
${{ env.FUNCTIONAL_ARTIFACTS_DIR }}/matrix.md
if-no-files-found: error
retention-days: 3 retention-days: 3
- name: Cleanup environment (after) - name: Cleanup environment (after)
+3 -3
View File
@@ -3,9 +3,9 @@
repos: repos:
- repo: local - repo: local
hooks: hooks:
- id: rustfs-fmt-check - id: rustfs-dev-check
name: Rust formatting name: rustfs dev-check
entry: cargo fmt --all --check entry: make dev-check
language: system language: system
types: [rust] types: [rust]
pass_filenames: false pass_filenames: false
+1 -4
View File
@@ -18,10 +18,7 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0
- Read paths: an object at or below `policy.inline_max_bytes` (16 MiB by default) is teed to the client and to the local store in a single source read; a larger object or a Range read streams through and a background pull stores the whole object. A HEAD miss is proxied to the source and stores nothing (`policy.head = local_only` disables it). Every source-backed response carries `x-rustfs-on-demand-migration: source` - Read paths: an object at or below `policy.inline_max_bytes` (16 MiB by default) is teed to the client and to the local store in a single source read; a larger object or a Range read streams through and a background pull stores the whole object. A HEAD miss is proxied to the source and stores nothing (`policy.head = local_only` disables it). Every source-backed response carries `x-rustfs-on-demand-migration: source`
- Protections: a per-source circuit breaker, a per-key negative cache, singleflight per key, a concurrency limit and a bounded pull queue shared by the inline and background paths, an optional bandwidth limit, an anti-loop request marker, and the shared outbound-endpoint (SSRF) policy - Protections: a per-source circuit breaker, a per-key negative cache, singleflight per key, a concurrency limit and a bounded pull queue shared by the inline and background paths, an optional bandwidth limit, an anti-loop request marker, and the shared outbound-endpoint (SSRF) policy
- Metrics under `rustfs_on_demand_migration_*` (`requests_total`, `pulled_bytes_total`, `pulled_objects_total`, `pull_failures_total`, `inflight_pulls`, `queue_depth`, `source_latency_seconds_*`, `breaker_state`), mirrored per node by the admin status route - Metrics under `rustfs_on_demand_migration_*` (`requests_total`, `pulled_bytes_total`, `pulled_objects_total`, `pull_failures_total`, `inflight_pulls`, `queue_depth`, `source_latency_seconds_*`, `breaker_state`), mirrored per node by the admin status route
- Listings: `ListObjects` v1 remains local with ordinary key markers. `ListObjectsV2` can merge source objects when `policy.list_through = true`; this is off by default - Limitations: listings show only local objects (the source is not merged into `ListObjectsV2`); PUT and DELETE never reach the source; a source object updated after it was pulled is not re-fetched; SSE-C source objects are unsupported and answer 424; `Last-Modified` on a pulled object is the local write time, with the source timestamp kept in metadata
- Upgrade and rollback: finish upgrading every node before enabling ODM. An rc.5 node that writes bucket configuration drops the ODM fields from metadata; neither a later restart nor moving the service out of ECStore recovers them. Before rollback, disable ODM and securely retain the original full configuration and credentials. After every node returns to a compatible version, restore and validate that configuration. Redacted exports cannot replace the credential backup; source-only objects are unavailable through RustFS while ODM is disabled. See the upgrade and rollback section of `docs/operations/on-demand-migration.md`
- Optional Google dependencies: default and `full` server builds retain native GCS support. `cargo build -p rustfs --no-default-features --features ftps,webdav` excludes Google SDKs while preserving configuration decoding and redaction; native GCS ODM and tier operations require the `gcs` feature. Do not use that build with existing GCS-tiered data
- Limitations: PUT and DELETE never reach the source; a source object updated after it was pulled is not re-fetched; SSE-C source objects are unsupported and answer 424; `Last-Modified` on a pulled object is the local write time, with the source timestamp kept in metadata
- **NATS JetStream Publish Path**: Opt-in at-least-once delivery for the NATS notify and audit targets. A NATS Core publish flushes to the connection without awaiting a broker acknowledgement, so an event can be lost across a broker restart or a reconnect after the send queue has already cleared it. A queued event now clears only after the JetStream `PublishAck`, so bucket notifications survive those interruptions. Off by default and byte-identical to the NATS Core path when disabled. - **NATS JetStream Publish Path**: Opt-in at-least-once delivery for the NATS notify and audit targets. A NATS Core publish flushes to the connection without awaiting a broker acknowledgement, so an event can be lost across a broker restart or a reconnect after the send queue has already cleared it. A queued event now clears only after the JetStream `PublishAck`, so bucket notifications survive those interruptions. Off by default and byte-identical to the NATS Core path when disabled.
- Three configuration keys per target: `JETSTREAM_ENABLE`, `JETSTREAM_STREAM_NAME`, and `JETSTREAM_ACK_TIMEOUT_SECS`, under the `RUSTFS_NOTIFY_NATS_` and `RUSTFS_AUDIT_NATS_` prefixes - Three configuration keys per target: `JETSTREAM_ENABLE`, `JETSTREAM_STREAM_NAME`, and `JETSTREAM_ACK_TIMEOUT_SECS`, under the `RUSTFS_NOTIFY_NATS_` and `RUSTFS_AUDIT_NATS_` prefixes
- Durable store-and-forward with a stable dedup id sent as the `Nats-Msg-Id` header, so a replay after a crash is collapsed by the server duplicate window - Durable store-and-forward with a stable dedup id sent as the `Nats-Msg-Id` header, so a replay after a crash is collapsed by the server duplicate window
+37 -11
View File
@@ -109,17 +109,24 @@ affected boundaries and risks. CI still runs its configured repository gates.
### 🔒 Git Pre-commit Hooks (optional) ### 🔒 Git Pre-commit Hooks (optional)
The optional hook uses the checked-in `.pre-commit-config.yaml`. Install [pre-commit](https://pre-commit.com/#installation), then run this from the checkout or a linked worktree: Git hooks are **not** versioned in this repository, so a fresh clone has no
active pre-commit hook. If you add your own `.git/hooks/pre-commit` (a good
choice is a one-liner that runs `make pre-commit`), you can mark it executable
with:
```bash ```bash
make setup-hooks make setup-hooks
``` ```
The hook runs `cargo fmt --all --check` when staged files include Rust source. It does not compile the workspace or run tests. Fix formatting with `cargo fmt --all`, inspect and stage the result, then commit again. Or manually:
`pre-commit install` resolves Git's hook directory for linked worktrees and preserves an existing hook in migration mode. If you use `core.hooksPath`, keep that hook manager and integrate `pre-commit run` there; the installer refuses to silently replace that configuration. ```bash
chmod +x .git/hooks/pre-commit
```
A local hook provides early formatting feedback. With or without it, follow the verification tiers in `AGENTS.md`, run relevant behavioral tests, and satisfy the CI merge gates. `make pre-commit` and `make dev-check` remain explicit broader commands. With or without a hook, follow the verification tiers in `AGENTS.md`. Run the
applicable scoped checks, and reserve `make pre-pr` for broad cross-module
changes whose impact cannot be bounded by those checks.
### 📝 Formatting Configuration ### 📝 Formatting Configuration
@@ -131,11 +138,31 @@ fn_call_width = 90
single_line_let_else_max_width = 100 single_line_let_else_max_width = 100
``` ```
### 🚫 Commit Prevention
If you set up a pre-commit hook and your code doesn't meet the formatting requirements, the hook will:
1. **Block the commit** and show clear error messages
2. **Provide exact commands** to fix the issues
3. **Guide you through** the resolution process
Example output when formatting fails:
```
❌ Code formatting check failed!
💡 Please run 'cargo fmt --all' to format your code before committing.
🔧 Quick fix:
cargo fmt --all
git add .
git commit
```
### 🔄 Development Workflow ### 🔄 Development Workflow
1. **Make your changes** 1. **Make your changes**
2. **Format your code**: `make fmt` or `cargo fmt --all` 2. **Format your code**: `make fmt` or `cargo fmt --all`
3. **Select relevant checks** using the validation tier in `AGENTS.md`; use `make pre-commit` when its broader fast gate adds useful coverage 3. **Run the fast gate**: `make pre-commit` (no clippy, no tests)
4. **Commit your changes**: `git commit -m "your message"` 4. **Commit your changes**: `git commit -m "your message"`
5. **Complete the applicable multi-role adversarial review** for non-exempt changes (see `AGENTS.md`) 5. **Complete the applicable multi-role adversarial review** for non-exempt changes (see `AGENTS.md`)
6. **Run applicable scoped checks before opening/updating a PR**; consider 6. **Run applicable scoped checks before opening/updating a PR**; consider
@@ -179,12 +206,11 @@ Configure your IDE to:
#### Pre-commit hook not running? #### Pre-commit hook not running?
```bash ```bash
pre-commit validate-config # Check if hook is executable
pre-commit run --all-files ls -la .git/hooks/pre-commit
# Inspect any configured hook manager; do not overwrite it.
git config --get core.hooksPath # Make it executable if needed
# Install if no separate hook manager is configured. chmod +x .git/hooks/pre-commit
make setup-hooks
``` ```
#### Formatting issues? #### Formatting issues?
Generated
+34 -67
View File
@@ -315,7 +315,7 @@ dependencies = [
"strum", "strum",
"thiserror 2.0.20", "thiserror 2.0.20",
"uuid", "uuid",
"zstd 0.13.3", "zstd",
] ]
[[package]] [[package]]
@@ -508,7 +508,7 @@ dependencies = [
"arrow-select", "arrow-select",
"flatbuffers", "flatbuffers",
"lz4_flex", "lz4_flex",
"zstd 0.13.3", "zstd",
] ]
[[package]] [[package]]
@@ -1679,7 +1679,7 @@ version = "0.10.4"
source = "registry+https://github.com/rust-lang/crates.io-index" source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "3078c7629b62d3f0439517fa394996acacc5cbc91c5a20d8c658e77abd503a71" checksum = "3078c7629b62d3f0439517fa394996acacc5cbc91c5a20d8c658e77abd503a71"
dependencies = [ dependencies = [
"generic-array 0.14.9", "generic-array 0.14.7",
] ]
[[package]] [[package]]
@@ -1698,7 +1698,7 @@ version = "0.3.3"
source = "registry+https://github.com/rust-lang/crates.io-index" source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "a8894febbff9f758034a5b8e12d87918f56dfc64a8e1fe757d65e29041538d93" checksum = "a8894febbff9f758034a5b8e12d87918f56dfc64a8e1fe757d65e29041538d93"
dependencies = [ dependencies = [
"generic-array 0.14.9", "generic-array 0.14.7",
] ]
[[package]] [[package]]
@@ -2110,7 +2110,7 @@ version = "0.4.4"
source = "registry+https://github.com/rust-lang/crates.io-index" source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "773f3b9af64447d2ce9850330c473515014aa235e6a783b02db81ff39e4a3dad" checksum = "773f3b9af64447d2ce9850330c473515014aa235e6a783b02db81ff39e4a3dad"
dependencies = [ dependencies = [
"crypto-common 0.1.6", "crypto-common 0.1.7",
"inout 0.1.4", "inout 0.1.4",
] ]
@@ -2249,8 +2249,8 @@ dependencies = [
"liblzma", "liblzma",
"lz4", "lz4",
"memchr", "memchr",
"zstd 0.13.3", "zstd",
"zstd-safe 7.3.0", "zstd-safe",
] ]
[[package]] [[package]]
@@ -2580,7 +2580,7 @@ version = "0.5.5"
source = "registry+https://github.com/rust-lang/crates.io-index" source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "0dc92fb57ca44df6db8059111ab3af99a63d5d0f8375d9972e319a379c6bab76" checksum = "0dc92fb57ca44df6db8059111ab3af99a63d5d0f8375d9972e319a379c6bab76"
dependencies = [ dependencies = [
"generic-array 0.14.9", "generic-array 0.14.7",
"rand_core 0.6.4", "rand_core 0.6.4",
"subtle", "subtle",
"zeroize", "zeroize",
@@ -2605,11 +2605,11 @@ dependencies = [
[[package]] [[package]]
name = "crypto-common" name = "crypto-common"
version = "0.1.6" version = "0.1.7"
source = "registry+https://github.com/rust-lang/crates.io-index" source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "1bfb12502f3fc46cca1bb51ac28df9d618d813cdc3d2f25b9fe775a34af26bb3" checksum = "78c8292055d1c1df0cce5d180393dc8cce0abec0a7102adb6c7b1eef6016d60a"
dependencies = [ dependencies = [
"generic-array 0.14.9", "generic-array 0.14.7",
"typenum", "typenum",
] ]
@@ -3901,7 +3901,7 @@ checksum = "9ed9a281f7bc9b7576e61468ba615a66a5c8cfdff42420a70aa82701a3b1e292"
dependencies = [ dependencies = [
"block-buffer 0.10.4", "block-buffer 0.10.4",
"const-oid 0.9.6", "const-oid 0.9.6",
"crypto-common 0.1.6", "crypto-common 0.1.7",
"subtle", "subtle",
] ]
@@ -4067,7 +4067,7 @@ dependencies = [
"uuid", "uuid",
"walkdir", "walkdir",
"zip", "zip",
"zstd 0.14.0", "zstd",
] ]
[[package]] [[package]]
@@ -4166,7 +4166,7 @@ dependencies = [
"crypto-bigint 0.5.5", "crypto-bigint 0.5.5",
"digest 0.10.7", "digest 0.10.7",
"ff 0.13.1", "ff 0.13.1",
"generic-array 0.14.9", "generic-array 0.14.7",
"group 0.13.0", "group 0.13.0",
"hkdf 0.12.4", "hkdf 0.12.4",
"pem-rfc7468 0.7.0", "pem-rfc7468 0.7.0",
@@ -4499,7 +4499,7 @@ checksum = "94e7099f6313ecacbe1256e8ff9d617b75d1bcb16a6fddef94866d225a01a14a"
dependencies = [ dependencies = [
"io-lifetimes 2.0.4", "io-lifetimes 2.0.4",
"rustix", "rustix",
"windows-sys 0.59.0", "windows-sys 0.52.0",
] ]
[[package]] [[package]]
@@ -4622,9 +4622,9 @@ dependencies = [
[[package]] [[package]]
name = "generic-array" name = "generic-array"
version = "0.14.9" version = "0.14.7"
source = "registry+https://github.com/rust-lang/crates.io-index" source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "4bb6743198531e02858aeaea5398fcc883e71851fcbcb5a2f773e2fb6cb1edf2" checksum = "85649ca51fd72272d7821adaf274ad91c288277713d9c18820d8499a7ff69e9a"
dependencies = [ dependencies = [
"typenum", "typenum",
"version_check", "version_check",
@@ -4637,7 +4637,7 @@ version = "1.4.5"
source = "registry+https://github.com/rust-lang/crates.io-index" source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "337d46834ee672ab3e48caca2cb0c78cc174fb12b3a68d0d88f99a0519a5e36e" checksum = "337d46834ee672ab3e48caca2cb0c78cc174fb12b3a68d0d88f99a0519a5e36e"
dependencies = [ dependencies = [
"generic-array 0.14.9", "generic-array 0.14.7",
"rustversion", "rustversion",
"typenum", "typenum",
] ]
@@ -5319,9 +5319,9 @@ dependencies = [
[[package]] [[package]]
name = "hotpath-macros" name = "hotpath-macros"
version = "0.25.1" version = "0.25.0"
source = "registry+https://github.com/rust-lang/crates.io-index" source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "846bde0d9600d98434e1aac376977d7718bfe3d2f5312a041b7c59a6a466c51a" checksum = "929b2285d2cd21b2733a7fb6ebc843bb4f83dbd1db0122f5f9ebb9567b1e2613"
dependencies = [ dependencies = [
"proc-macro2", "proc-macro2",
"quote", "quote",
@@ -5660,7 +5660,7 @@ source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "879f10e63c20629ecabbb64a8010319738c66a5cd0c29b02d63d272b03751d01" checksum = "879f10e63c20629ecabbb64a8010319738c66a5cd0c29b02d63d272b03751d01"
dependencies = [ dependencies = [
"block-padding 0.3.3", "block-padding 0.3.3",
"generic-array 0.14.9", "generic-array 0.14.7",
] ]
[[package]] [[package]]
@@ -5693,7 +5693,7 @@ source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "20fd6de4ccfcc187e38bc21cfa543cb5a302cb86a8b114eb7f0bf0dc9f8ac00f" checksum = "20fd6de4ccfcc187e38bc21cfa543cb5a302cb86a8b114eb7f0bf0dc9f8ac00f"
dependencies = [ dependencies = [
"io-lifetimes 3.0.1", "io-lifetimes 3.0.1",
"windows-sys 0.60.2", "windows-sys 0.52.0",
] ]
[[package]] [[package]]
@@ -5971,7 +5971,7 @@ dependencies = [
"lz4", "lz4",
"snap", "snap",
"uuid", "uuid",
"zstd 0.13.3", "zstd",
] ]
[[package]] [[package]]
@@ -7115,7 +7115,7 @@ version = "5.0.0"
source = "registry+https://github.com/rust-lang/crates.io-index" source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "51e219e79014df21a225b1860a479e2dcd7cbd9130f4defd4bd0e191ea31d67d" checksum = "51e219e79014df21a225b1860a479e2dcd7cbd9130f4defd4bd0e191ea31d67d"
dependencies = [ dependencies = [
"base64 0.22.1", "base64 0.21.7",
"chrono", "chrono",
"getrandom 0.2.17", "getrandom 0.2.17",
"http 1.5.0", "http 1.5.0",
@@ -7658,7 +7658,7 @@ dependencies = [
"snap", "snap",
"tokio", "tokio",
"twox-hash", "twox-hash",
"zstd 0.13.3", "zstd",
] ]
[[package]] [[package]]
@@ -9493,8 +9493,6 @@ dependencies = [
"atomic_enum", "atomic_enum",
"aws-config", "aws-config",
"aws-sdk-s3", "aws-sdk-s3",
"aws-smithy-runtime-api",
"aws-smithy-types",
"axum", "axum",
"base64-simd", "base64-simd",
"bytes", "bytes",
@@ -9503,13 +9501,11 @@ dependencies = [
"clap", "clap",
"const-str", "const-str",
"datafusion", "datafusion",
"faster-hex",
"flatbuffers", "flatbuffers",
"flate2", "flate2",
"futures", "futures",
"futures-lite", "futures-lite",
"futures-util", "futures-util",
"google-cloud-auth",
"hashbrown 0.17.1", "hashbrown 0.17.1",
"hex-simd", "hex-simd",
"hmac 0.13.0", "hmac 0.13.0",
@@ -9528,7 +9524,6 @@ dependencies = [
"metrics", "metrics",
"metrics-util", "metrics-util",
"mime_guess", "mime_guess",
"moka",
"opentelemetry", "opentelemetry",
"opentelemetry_sdk", "opentelemetry_sdk",
"p256 0.14.0", "p256 0.14.0",
@@ -9621,10 +9616,9 @@ dependencies = [
"urlencoding", "urlencoding",
"uuid", "uuid",
"x509-parser", "x509-parser",
"xxhash-rust",
"zeroize", "zeroize",
"zip", "zip",
"zstd 0.14.0", "zstd",
] ]
[[package]] [[package]]
@@ -10234,7 +10228,7 @@ dependencies = [
"thiserror 2.0.20", "thiserror 2.0.20",
"walkdir", "walkdir",
"zip", "zip",
"zstd 0.14.0", "zstd",
] ]
[[package]] [[package]]
@@ -10401,7 +10395,7 @@ dependencies = [
"tracing-opentelemetry", "tracing-opentelemetry",
"tracing-subscriber", "tracing-subscriber",
"url", "url",
"zstd 0.14.0", "zstd",
] ]
[[package]] [[package]]
@@ -10991,7 +10985,7 @@ dependencies = [
"transform-stream", "transform-stream",
"url", "url",
"windows", "windows",
"zstd 0.14.0", "zstd",
] ]
[[package]] [[package]]
@@ -11405,7 +11399,7 @@ checksum = "d3e97a565f76233a6003f9f5c54be1d9c5bdfa3eccfb189469f11ec4901c47dc"
dependencies = [ dependencies = [
"base16ct 0.2.0", "base16ct 0.2.0",
"der 0.7.10", "der 0.7.10",
"generic-array 0.14.9", "generic-array 0.14.7",
"pkcs8 0.10.2", "pkcs8 0.10.2",
"subtle", "subtle",
"zeroize", "zeroize",
@@ -12405,7 +12399,7 @@ source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "32497e9a4c7b38532efcdebeef879707aa9f794296a4f0244f6f69e9bc8574bd" checksum = "32497e9a4c7b38532efcdebeef879707aa9f794296a4f0244f6f69e9bc8574bd"
dependencies = [ dependencies = [
"fastrand", "fastrand",
"getrandom 0.4.3", "getrandom 0.3.4",
"once_cell", "once_cell",
"rustix", "rustix",
"windows-sys 0.61.2", "windows-sys 0.61.2",
@@ -13657,15 +13651,6 @@ dependencies = [
"windows-targets 0.52.6", "windows-targets 0.52.6",
] ]
[[package]]
name = "windows-sys"
version = "0.59.0"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "1e38bc4d79ed67fd075bcc251a1c39b32a1776bbe92e5bef1f0bf1f8c531853b"
dependencies = [
"windows-targets 0.52.6",
]
[[package]] [[package]]
name = "windows-sys" name = "windows-sys"
version = "0.60.2" version = "0.60.2"
@@ -13838,7 +13823,7 @@ source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "3f3fd376f71958b862e7afb20cfe5a22830e1963462f3a17f49d82a6c1d1f42d" checksum = "3f3fd376f71958b862e7afb20cfe5a22830e1963462f3a17f49d82a6c1d1f42d"
dependencies = [ dependencies = [
"bitflags 2.13.1", "bitflags 2.13.1",
"windows-sys 0.59.0", "windows-sys 0.52.0",
] ]
[[package]] [[package]]
@@ -14110,7 +14095,7 @@ dependencies = [
"typed-path", "typed-path",
"zeroize", "zeroize",
"zopfli", "zopfli",
"zstd 0.13.3", "zstd",
] ]
[[package]] [[package]]
@@ -14143,16 +14128,7 @@ version = "0.13.3"
source = "registry+https://github.com/rust-lang/crates.io-index" source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "e91ee311a569c327171651566e07972200e76fcfe2242a4fa446149a3881c08a" checksum = "e91ee311a569c327171651566e07972200e76fcfe2242a4fa446149a3881c08a"
dependencies = [ dependencies = [
"zstd-safe 7.3.0", "zstd-safe",
]
[[package]]
name = "zstd"
version = "0.14.0"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "bf06bd8162af0734b344780deb55b42a2429ae430870d13fcc12f238e880fe6e"
dependencies = [
"zstd-safe 8.0.0",
] ]
[[package]] [[package]]
@@ -14164,15 +14140,6 @@ dependencies = [
"zstd-sys", "zstd-sys",
] ]
[[package]]
name = "zstd-safe"
version = "8.0.0"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "ae42c0555055784c70058d19ba8e275528e8a99a706684868ace5da4e716a4ab"
dependencies = [
"zstd-sys",
]
[[package]] [[package]]
name = "zstd-sys" name = "zstd-sys"
version = "2.1.0+zstd.1.5.7" version = "2.1.0+zstd.1.5.7"
+6 -7
View File
@@ -199,10 +199,10 @@ serde_urlencoded = "0.7.1"
# matching stable releases are not available yet, while previous stable lines # matching stable releases are not available yet, while previous stable lines
# have incompatible APIs. Keep them exact-pinned and monitor upstream for stable # have incompatible APIs. Keep them exact-pinned and monitor upstream for stable
# releases. # releases.
aes-gcm = { version = "0.11.1" } aes-gcm = { version = "=0.11.1" }
argon2 = { version = "0.6.0" } argon2 = { version = "=0.6.0" }
blake2 = "0.11.0" blake2 = "=0.11.0"
chacha20poly1305 = { version = "0.11.0" } chacha20poly1305 = { version = "=0.11.0" }
crc-fast = "1.10.0" crc-fast = "1.10.0"
hmac = { version = "0.13.0" } hmac = { version = "0.13.0" }
jsonwebtoken = { version = "11.0.0" } jsonwebtoken = { version = "11.0.0" }
@@ -343,7 +343,7 @@ windows = { version = "0.62.2" }
windows-sys = "0.61.2" windows-sys = "0.61.2"
xxhash-rust = { version = "0.8.18" } xxhash-rust = { version = "0.8.18" }
zip = "8.6.0" zip = "8.6.0"
zstd = "0.14.0" zstd = "0.13.3"
# Observability and Metrics # Observability and Metrics
metrics = "0.24.6" metrics = "0.24.6"
@@ -371,8 +371,7 @@ dav-server = "0.11.0"
# Performance Analysis and Memory Profiling # Performance Analysis and Memory Profiling
rustfs-mimalloc = { version = "0.5.3" } rustfs-mimalloc = { version = "0.5.3" }
# Preserve Unicode focus filters until rustfs/backlog#2302 is resolved. hotpath = { version = "0.25.0", default-features = false }
hotpath = { version = "=0.25.0", default-features = false }
# Snapshot testing for output format regression detection # Snapshot testing for output format regression detection
insta = { version = "1.48" } insta = { version = "1.48" }
-15
View File
@@ -130,21 +130,6 @@ Scanner cycle budget controls:
- timeout returns S3 `SlowDown`, so clients should use normal SDK retry handling. - timeout returns S3 `SlowDown`, so clients should use normal SDK retry handling.
- this is not a fdatasync or group-commit switch. Track fdatasync batching separately with `rustfs_s3_put_object_rename_fdatasync_batch_files`. - this is not a fdatasync or group-commit switch. Track fdatasync batching separately with `rustfs_s3_put_object_rename_fdatasync_batch_files`.
## Remote tier timeout environment variables
- `RUSTFS_TIER_REMOTE_CONNECT_TIMEOUT_SECS`
- remote tier TCP connect timeout.
- default is `10`.
- must be positive; zero fails tier client initialization, while an invalid integer is logged and falls back to the default.
- `RUSTFS_TIER_REMOTE_REQUEST_TIMEOUT_SECS`
- remote tier request timeout through response headers.
- default is `86400` so large transition uploads keep a production-safe budget.
- must be positive; zero fails tier client initialization, while an invalid integer is logged and falls back to the default. Very large values are accepted and act as a correspondingly long budget.
- `RUSTFS_TIER_REMOTE_RESPONSE_BODY_IDLE_TIMEOUT_SECS`
- maximum idle time between remote tier response-body chunks.
- default is `60`; the timer resets only when non-empty body data keeps progressing.
- must be positive; zero fails tier client initialization, while an invalid integer is logged and falls back to the default.
## Drive timeout environment variables ## Drive timeout environment variables
- `RUSTFS_DRIVE_METADATA_TIMEOUT_SECS` - `RUSTFS_DRIVE_METADATA_TIMEOUT_SECS`
-32
View File
@@ -137,28 +137,6 @@ pub const DEFAULT_TIER_REMOTE_VERSION_STATE_FLEET_CONFIRMED: bool = false;
const _: () = assert!(!DEFAULT_TIER_REMOTE_VERSION_STATE_WRITE); const _: () = assert!(!DEFAULT_TIER_REMOTE_VERSION_STATE_WRITE);
const _: () = assert!(!DEFAULT_TIER_REMOTE_VERSION_STATE_FLEET_CONFIRMED); const _: () = assert!(!DEFAULT_TIER_REMOTE_VERSION_STATE_FLEET_CONFIRMED);
/// Environment variable for remote tier TCP connect timeout in seconds.
pub const ENV_TIER_REMOTE_CONNECT_TIMEOUT_SECS: &str = "RUSTFS_TIER_REMOTE_CONNECT_TIMEOUT_SECS";
/// Default remote tier TCP connect timeout in seconds.
pub const DEFAULT_TIER_REMOTE_CONNECT_TIMEOUT_SECS: u64 = 10;
/// Environment variable for the remote tier request timeout in seconds.
///
/// This bounds upload/download request progress through response headers. The
/// default is intentionally large so multi-TiB transition uploads keep their
/// previous production budget while black-hole remotes no longer wait forever.
pub const ENV_TIER_REMOTE_REQUEST_TIMEOUT_SECS: &str = "RUSTFS_TIER_REMOTE_REQUEST_TIMEOUT_SECS";
/// Default remote tier request timeout in seconds.
pub const DEFAULT_TIER_REMOTE_REQUEST_TIMEOUT_SECS: u64 = 24 * 60 * 60;
/// Environment variable for remote tier response-body idle timeout in seconds.
///
/// The timer is re-armed on every non-empty response-body chunk, so slow but
/// progressing remotes can continue while silent response bodies are cancelled.
pub const ENV_TIER_REMOTE_RESPONSE_BODY_IDLE_TIMEOUT_SECS: &str = "RUSTFS_TIER_REMOTE_RESPONSE_BODY_IDLE_TIMEOUT_SECS";
/// Default remote tier response-body idle timeout in seconds.
pub const DEFAULT_TIER_REMOTE_RESPONSE_BODY_IDLE_TIMEOUT_SECS: u64 = 60;
/// Request the object-transaction fencing contract used by storage-owned /// Request the object-transaction fencing contract used by storage-owned
/// cleanup receipts and lock-window optimizations. /// cleanup receipts and lock-window optimizations.
/// ///
@@ -834,16 +812,6 @@ mod remote_version_state_tests {
); );
} }
#[test]
fn remote_tier_timeout_env_names_are_stable() {
assert_eq!(super::ENV_TIER_REMOTE_CONNECT_TIMEOUT_SECS, "RUSTFS_TIER_REMOTE_CONNECT_TIMEOUT_SECS");
assert_eq!(super::ENV_TIER_REMOTE_REQUEST_TIMEOUT_SECS, "RUSTFS_TIER_REMOTE_REQUEST_TIMEOUT_SECS");
assert_eq!(
super::ENV_TIER_REMOTE_RESPONSE_BODY_IDLE_TIMEOUT_SECS,
"RUSTFS_TIER_REMOTE_RESPONSE_BODY_IDLE_TIMEOUT_SECS"
);
}
#[test] #[test]
fn data_movement_part_checksum_gate_uses_stable_environment_names() { fn data_movement_part_checksum_gate_uses_stable_environment_names() {
assert_eq!(super::ENV_DATA_MOVEMENT_PART_CHECKSUMS_WRITE, "RUSTFS_DATA_MOVEMENT_PART_CHECKSUMS_WRITE"); assert_eq!(super::ENV_DATA_MOVEMENT_PART_CHECKSUMS_WRITE, "RUSTFS_DATA_MOVEMENT_PART_CHECKSUMS_WRITE");
+33 -43
View File
@@ -1,7 +1,7 @@
# e2e_test # e2e_test
End-to-end test suite for RustFS. Each test spawns a **real `rustfs` binary** End-to-end test suite for RustFS. Each test spawns a **real `rustfs` binary**
(built on demand from the workspace) and drives it over the network with the (built and identified before the test invocation) and drives it over the network with the
AWS SDK (`aws-sdk-s3`), raw HTTP (`reqwest` / `awscurl`), or a protocol client AWS SDK (`aws-sdk-s3`), raw HTTP (`reqwest` / `awscurl`), or a protocol client
(FTPS / WebDAV / SFTP). This is the black-box integration layer: exhaustive (FTPS / WebDAV / SFTP). This is the black-box integration layer: exhaustive
end-to-end behavior lives here, unit behavior stays in the source crates end-to-end behavior lives here, unit behavior stays in the source crates
@@ -31,32 +31,28 @@ Registered in [`src/lib.rs`](src/lib.rs). Grouped by concern:
## How to run ## How to run
All commands assume repo root. `cargo test` triggers an on-demand build of the All commands assume repo root and Python 3.9 or newer on Linux or macOS. Build the server once through the provenance entry point, then run the test command through the same script:
`rustfs` binary from [`src/common.rs`](src/common.rs) (`rustfs_binary_path`) on
first use — the first invocation is slow, later ones reuse the binary.
```bash ```bash
# Whole crate (default = ignored tests skipped) python3 scripts/e2e_binary.py build --features e2e-test-hooks
cargo nextest run -p e2e_test
# Whole crate (ignored tests remain skipped)
python3 scripts/e2e_binary.py run --features e2e-test-hooks -- cargo nextest run -p e2e_test
# One module # One module
cargo nextest run -p e2e_test -E 'test(list_objects_v2_pagination_test)' python3 scripts/e2e_binary.py run --features e2e-test-hooks -- cargo nextest run -p e2e_test -E 'test(list_objects_v2_pagination_test)'
# PR smoke subset (see "CI smoke subset" below)
cargo nextest run --profile e2e-smoke -p e2e_test
# ILM serial lane — ignored lifecycle tests, single-threaded (mirrors CI)
cargo nextest run -j1 --run-ignored ignored-only -p rustfs-scanner -p rustfs \
-E 'binary(lifecycle_integration_test) or (package(rustfs) and test(lifecycle_transition_api_test))'
# PR smoke subset
python3 scripts/e2e_binary.py run --features e2e-test-hooks -- cargo nextest run --profile e2e-smoke -p e2e_test
``` ```
The protocols suite has its own contract (fixed bind ports 90229301, `build` records the source contents, HEAD, resolved Cargo features, profile, toolchain, and binary SHA-256 beside the executable in `rustfs.e2e.json`. `run` validates that identity before and after the command, preserves command failures, and removes its temporary run receipt on completion. The Rust harness checks that receipt before starting each server; it never compiles a server inside a test process. Source or binary changes during a run invalidate the result, even when the test command succeeds. Use an isolated worktree and keep it unchanged until the command finishes.
single-worker execution, feature-gated scheduling) documented in
[`src/protocols/README.md`](src/protocols/README.md). `RUSTFS_BUILD_FEATURES` The additional `--features` arguments must match between `build` and `run`; Cargo defaults remain enabled. The wrapper supplies `RUSTFS_BUILD_FEATURES` from Cargo's resolved feature list, including features enabled by `full`. Protocol helpers require a subset of that list. `CARGO_TARGET_DIR` and `--profile release` are supported. An in-workspace target directory must be Git-ignored; tracked files are always included in the source identity. `build --bins` preserves CI lanes that compile all RustFS binary targets. For a downloaded artifact, copy both the executable and its sidecar, then use `run`; do not generate a new identity for an arbitrary prebuilt binary. `CARGO_BIN_EXE_rustfs` cannot override the verified executable.
selects which features the spawned binary is built with; leave it unset to run
every protocol entry. Use the exact profile command under Each build/run holds an exclusive `rustfs.e2e.lock` marker beside the binary; concurrent wrappers fail immediately. Use a private target directory and do not run ordinary Cargo builds against it while tests are active: Cargo does not honor this marker. Interrupted runs fail and terminate their command group. After an uncatchable kill, inspect the PID recorded in a leftover marker and remove it only after confirming its owner has stopped. Embedded file symlinks are hashed through their target; embedded directory symlinks are rejected because their contents cannot be enumerated safely by this entry point.
[Troubleshooting](#troubleshooting) for CI-equivalent execution.
The protocols suite has its own fixed-port and single-worker contract in [`src/protocols/README.md`](src/protocols/README.md). Use its command under [Troubleshooting](#troubleshooting).
### `#[ignore]` semantics ### `#[ignore]` semantics
@@ -122,7 +118,7 @@ via `create_s3_client(idx)` / `create_all_clients()`. See
| `wait_for_server_ready` | Poll readiness before issuing requests | | `wait_for_server_ready` | Poll readiness before issuing requests |
| `create_s3_client` / `create_test_bucket` / `delete_test_bucket` | aws-sdk-s3 client + bucket lifecycle | | `create_s3_client` / `create_test_bucket` / `delete_test_bucket` | aws-sdk-s3 client + bucket lifecycle |
| `find_available_port` | Random free port (isolation primitive) | | `find_available_port` | Random free port (isolation primitive) |
| `rustfs_binary_path` / `_with_features` | Locate/build the binary; honors `RUSTFS_BUILD_FEATURES` | | `rustfs_binary_path` / `_with_features` | Verify this run's binary receipt and required feature subset |
| `requested_rustfs_build_features` / `rustfs_build_feature_enabled` | Feature-gate a test to what the binary was built with | | `requested_rustfs_build_features` / `rustfs_build_feature_enabled` | Feature-gate a test to what the binary was built with |
| `execute_awscurl` / `awscurl_post` / `_get` / `_put` / `_delete` / `awscurl_post_sts_form_urlencoded` | Admin/STS API calls via `awscurl`; missing binaries are test failures | | `execute_awscurl` / `awscurl_post` / `_get` / `_put` / `_delete` / `awscurl_post_sts_form_urlencoded` | Admin/STS API calls via `awscurl`; missing binaries are test failures |
| `replication_fast_env` | Env vars that shrink replication timers (from repl-4); pass to `start_rustfs_server_with_env` | | `replication_fast_env` | Env vars that shrink replication timers (from repl-4); pass to `start_rustfs_server_with_env` |
@@ -185,32 +181,26 @@ the wiring source of truth. Committed test-ID digests under
**Reproduce a CI failure locally** — run the exact profile/lane: **Reproduce a CI failure locally** — run the exact profile/lane:
```bash ```bash
# Smoke (e2e-tests job) — includes the 20 fast replication tests # Smoke, full, and cluster lanes share a server with fault-test hooks.
cargo nextest run --profile e2e-smoke -p e2e_test python3 scripts/e2e_binary.py build --features e2e-test-hooks
# Full single-node merge/main lane python3 scripts/e2e_binary.py run --features e2e-test-hooks -- cargo nextest run --profile e2e-smoke -p e2e_test
cargo nextest run --profile e2e-full -p e2e_test python3 scripts/e2e_binary.py run --features e2e-test-hooks -- cargo nextest run --profile e2e-full -p e2e_test
# Cluster fault nightly lane python3 scripts/e2e_binary.py run --features e2e-test-hooks -- cargo nextest run --profile e2e-nightly -p e2e_test
cargo nextest run --profile e2e-nightly -p e2e_test
# Replication nightly lane; awscurl is required for STS paths # Replication nightly uses the default server; awscurl is required for STS.
cargo nextest run --profile e2e-repl-nightly -p e2e_test python3 scripts/e2e_binary.py build
# Fixed-port protocol nightly lane python3 scripts/e2e_binary.py run -- cargo nextest run --profile e2e-repl-nightly -p e2e_test
RUSTFS_BUILD_FEATURES=ftps,webdav,sftp \
cargo nextest run -j 1 --profile e2e-protocols -p e2e_test --no-capture # Protocol nightly owns fixed ports.
# ILM serial lane python3 scripts/e2e_binary.py build --features ftps,webdav,sftp
python3 scripts/e2e_binary.py run --features ftps,webdav,sftp -- cargo nextest run -j 1 --profile e2e-protocols -p e2e_test --no-capture
# The ILM serial lane does not use this server harness.
cargo nextest run -j1 --run-ignored ignored-only -p rustfs-scanner -p rustfs \ cargo nextest run -j1 --run-ignored ignored-only -p rustfs-scanner -p rustfs \
-E 'binary(lifecycle_integration_test) or (package(rustfs) and test(lifecycle_transition_api_test))' -E 'binary(lifecycle_integration_test) or (package(rustfs) and test(lifecycle_transition_api_test))'
# s3s-e2e black box
./scripts/e2e-run.sh ./target/debug/rustfs /tmp/rustfs-e2e-data
``` ```
**Stale binary.** Tests build the `rustfs` binary once and reuse it. To avoid **Stale or unverified binary.** Re-run the matching `build` command after changing source or features, then invoke tests through `run`. A missing receipt, copied old executable, or mismatched build identity is a prerequisite failure. Bare Cargo invocations that start a server deliberately fail; unit tests that do not start a server can still run directly.
rebuilding while iterating on tests, `common.rs` reuses an existing binary when
running *inside* the e2e test process even if sources changed
(`can_reuse_inside_e2e`, [`src/common.rs`](src/common.rs) line 98). Downside: if
you changed **server** code, force a rebuild with
`cargo build -p rustfs` (or `touch` a source file outside the reuse window)
before re-running, or CI's freshly built artifact will diverge from your local
one.
**Port already in use / orphan processes.** A hard-killed run can leak a **Port already in use / orphan processes.** A hard-killed run can leak a
`rustfs` child holding its port. Find and kill it: `rustfs` child holding its port. Find and kill it:
+123 -149
View File
@@ -31,7 +31,6 @@ use rustfs_signer::constants::UNSIGNED_PAYLOAD;
use rustfs_signer::sign_v4; use rustfs_signer::sign_v4;
use s3s::Body; use s3s::Body;
use serde_json; use serde_json;
use std::ffi::OsStr;
use std::fs as stdfs; use std::fs as stdfs;
use std::io::ErrorKind; use std::io::ErrorKind;
use std::net::SocketAddr; use std::net::SocketAddr;
@@ -44,7 +43,6 @@ use tokio::net::TcpStream;
use tokio::time::sleep; use tokio::time::sleep;
use tracing::{error, info, warn}; use tracing::{error, info, warn};
use uuid::Uuid; use uuid::Uuid;
use walkdir::WalkDir;
// Common constants for all E2E tests // Common constants for all E2E tests
pub const DEFAULT_ACCESS_KEY: &str = "rustfsadmin"; pub const DEFAULT_ACCESS_KEY: &str = "rustfsadmin";
@@ -365,59 +363,75 @@ fn resolve_rustfs_binary_path(workspace: &Path, configured_target_dir: Option<&P
path path
} }
/// Resolve the RustFS binary relative to the workspace, optionally requesting build features. /// Resolve the server verified by `scripts/e2e_binary.py run` for this test invocation.
/// Requested features are a required subset of the server's resolved Cargo features.
pub fn rustfs_binary_path_with_features(requested_features: Option<&str>) -> PathBuf { pub fn rustfs_binary_path_with_features(requested_features: Option<&str>) -> PathBuf {
if let Some(path) = std::env::var_os("CARGO_BIN_EXE_rustfs") {
return PathBuf::from(path);
}
let requested_features = requested_features.and_then(normalize_rustfs_build_features);
let workspace = workspace_root(); let workspace = workspace_root();
let configured_target_dir = std::env::var_os("CARGO_TARGET_DIR").map(PathBuf::from); let configured_target_dir = std::env::var_os("CARGO_TARGET_DIR").map(PathBuf::from);
let binary_path = resolve_rustfs_binary_path(&workspace, configured_target_dir.as_deref()); let binary_path = std::env::var_os("CARGO_BIN_EXE_rustfs")
.map(PathBuf::from)
.unwrap_or_else(|| resolve_rustfs_binary_path(&workspace, configured_target_dir.as_deref()));
let receipt_path = std::env::var_os("RUSTFS_E2E_BINARY_RECEIPT").map(PathBuf::from);
receipt_path
.ok_or_else(|| std::io::Error::new(ErrorKind::NotFound, "missing E2E run receipt"))
.and_then(|receipt| verify_e2e_binary_receipt(&receipt, &workspace, &binary_path, requested_features))
.unwrap_or_else(|error| {
panic!(
"E2E server prerequisite failed: {error}. Build with `python3 scripts/e2e_binary.py build --features <features>` and run tests with `python3 scripts/e2e_binary.py run --features <features> -- cargo nextest run ...`"
)
})
}
let features_match = binary_features_match(&binary_path, requested_features.as_deref()); #[derive(serde::Deserialize)]
let source_is_newer = workspace_sources_newer_than_binary(&binary_path); #[serde(deny_unknown_fields)]
let can_reuse_inside_e2e = running_inside_e2e_test_binary() && requested_features.is_none() && features_match; struct E2eBinaryReceipt {
if binary_path.is_file() && features_match && (!source_is_newer || can_reuse_inside_e2e) { schema: u32,
if source_is_newer { workspace: PathBuf,
warn!( binary: PathBuf,
"RustFS binary at {:?} appears older than workspace sources; reusing it inside cargo test to avoid nested builds", size: u64,
binary_path modified_ns: u128,
); features: Vec<String>,
} }
info!("Using existing RustFS binary at {:?}", binary_path);
return binary_path; fn verify_e2e_binary_receipt(
receipt_path: &Path,
workspace: &Path,
binary_path: &Path,
requested_features: Option<&str>,
) -> std::io::Result<PathBuf> {
let receipt: E2eBinaryReceipt = serde_json::from_slice(&stdfs::read(receipt_path)?)?;
let binary = binary_path.canonicalize()?;
let metadata = binary.metadata()?;
let modified_ns = metadata
.modified()?
.duration_since(std::time::UNIX_EPOCH)
.map_err(std::io::Error::other)?
.as_nanos();
// The runner hashes source and binary before/after the entire suite. Each
// nextest process checks only this invocation's path, features, and file stat.
if receipt.schema != 1
|| receipt.workspace != workspace.canonicalize()?
|| receipt.binary != binary
|| !metadata.is_file()
|| receipt.size != metadata.len()
|| receipt.modified_ns != modified_ns
{
return Err(std::io::Error::new(
ErrorKind::InvalidData,
"E2E server differs from this run's verified binary",
));
} }
if let Some(requested) = requested_features.and_then(normalize_rustfs_build_features)
info!("Building RustFS binary to ensure it's up to date..."); && requested
build_rustfs_binary(requested_features.as_deref(), &binary_path); .split(',')
.any(|feature| !receipt.features.iter().any(|actual| actual == feature))
info!("Using RustFS binary at {:?}", binary_path); {
binary_path return Err(std::io::Error::new(
} ErrorKind::InvalidInput,
"E2E server is missing a requested build feature",
fn workspace_sources_newer_than_binary(binary_path: &PathBuf) -> bool { ));
let Ok(binary_meta) = std::fs::metadata(binary_path) else { }
return true; Ok(binary)
};
let Ok(binary_modified) = binary_meta.modified() else {
return true;
};
let workspace = workspace_root();
let watch_roots = [
workspace.join("Cargo.toml"),
workspace.join("Cargo.lock"),
workspace.join("rustfs"),
workspace.join("crates"),
];
watch_roots.iter().any(|path| path_is_newer_than(binary_modified, path))
}
fn running_inside_e2e_test_binary() -> bool {
std::env::var("CARGO_PKG_NAME").is_ok_and(|value| value == "e2e_test")
} }
pub fn requested_rustfs_build_features() -> Option<String> { pub fn requested_rustfs_build_features() -> Option<String> {
@@ -447,96 +461,6 @@ pub fn rustfs_build_feature_enabled(requested_features: Option<&str>, required_f
.any(|feature| feature.eq_ignore_ascii_case(RUSTFS_FULL_FEATURE) || feature.eq_ignore_ascii_case(required_feature)) .any(|feature| feature.eq_ignore_ascii_case(RUSTFS_FULL_FEATURE) || feature.eq_ignore_ascii_case(required_feature))
} }
fn rustfs_binary_features_stamp_path(binary_path: &Path) -> PathBuf {
binary_path.with_extension("features")
}
fn binary_features_match(binary_path: &Path, requested_features: Option<&str>) -> bool {
let stamp_path = rustfs_binary_features_stamp_path(binary_path);
let recorded = stdfs::read_to_string(stamp_path)
.ok()
.and_then(|value| normalize_rustfs_build_features(&value));
let requested = requested_features.and_then(normalize_rustfs_build_features);
match requested.as_deref() {
Some(features) => recorded.as_deref() == Some(features),
None => recorded.is_none(),
}
}
fn path_is_newer_than(binary_modified: std::time::SystemTime, path: &Path) -> bool {
if path.is_file() {
return std::fs::metadata(path)
.and_then(|meta| meta.modified())
.map(|modified| modified > binary_modified)
.unwrap_or(false);
}
if !path.is_dir() {
return false;
}
WalkDir::new(path)
.into_iter()
.filter_entry(|entry| {
let name = entry.file_name();
name != OsStr::new("target") && name != OsStr::new(".git")
})
.filter_map(Result::ok)
.filter(|entry| entry.file_type().is_file())
.any(|entry| {
std::fs::metadata(entry.path())
.and_then(|meta| meta.modified())
.map(|modified| modified > binary_modified)
.unwrap_or(false)
})
}
/// Build the RustFS binary using cargo
fn build_rustfs_binary(requested_features: Option<&str>, binary_path: &Path) {
let workspace = workspace_root();
info!("Building RustFS binary from workspace: {:?}", workspace);
let _profile = if cfg!(debug_assertions) {
info!("Building in debug mode");
"dev"
} else {
info!("Building in release mode");
"release"
};
let mut cmd = Command::new("cargo");
cmd.current_dir(&workspace).args(["build", "--bin", "rustfs"]);
if let Some(features) = requested_features {
cmd.arg("--features").arg(features);
info!("Building with features: {}", features);
}
if !cfg!(debug_assertions) {
cmd.arg("--release");
}
info!(
"Executing: cargo build --bin rustfs {}",
if cfg!(debug_assertions) { "" } else { "--release" }
);
let output = cmd.output().expect("Failed to execute cargo build command");
if !output.status.success() {
let stderr = String::from_utf8_lossy(&output.stderr);
panic!("Failed to build RustFS binary. Error: {stderr}");
}
let stamp_path = rustfs_binary_features_stamp_path(binary_path);
if let Err(err) = stdfs::write(&stamp_path, requested_features.unwrap_or_default()) {
warn!("Failed to write RustFS feature stamp {:?}: {}", stamp_path, err);
}
info!("✅ RustFS binary built successfully");
}
fn awscurl_binary_path() -> PathBuf { fn awscurl_binary_path() -> PathBuf {
std::env::var_os("AWSCURL_PATH") std::env::var_os("AWSCURL_PATH")
.map(PathBuf::from) .map(PathBuf::from)
@@ -2073,16 +1997,66 @@ mod tests {
} }
#[test] #[test]
fn binary_feature_stamp_matching_uses_normalized_features() { fn explicit_binary_without_run_receipt_is_rejected() {
let binary_path = std::env::temp_dir().join(format!("rustfs-feature-stamp-test-{}", Uuid::new_v4())); const CHILD_ENV: &str = "RUSTFS_E2E_RECEIPT_TEST_CHILD";
let stamp_path = rustfs_binary_features_stamp_path(&binary_path); if std::env::var_os(CHILD_ENV).is_some() {
rustfs_binary_path_with_features(None);
return;
}
let executable = std::env::current_exe().expect("locate isolated test process");
let output = Command::new(&executable)
.args([
"--exact",
"common::tests::explicit_binary_without_run_receipt_is_rejected",
"--nocapture",
])
.env(CHILD_ENV, "1")
.env("CARGO_BIN_EXE_rustfs", &executable)
.env_remove("RUSTFS_E2E_BINARY_RECEIPT")
.output()
.expect("run the missing-receipt scenario with isolated environment variables");
assert!(!output.status.success(), "an explicit binary must not bypass run verification");
assert!(String::from_utf8_lossy(&output.stderr).contains("missing E2E run receipt"));
}
stdfs::write(&stamp_path, " SFTP, ftps ").expect("write feature stamp"); #[test]
assert!(binary_features_match(&binary_path, Some("sftp,ftps"))); fn e2e_run_receipt_rejects_replaced_binary_and_missing_features() {
assert!(binary_features_match(&binary_path, Some(" SFTP, FTPS "))); let directory = std::env::temp_dir().join(format!("rustfs-e2e-receipt-test-{}", Uuid::new_v4()));
assert!(!binary_features_match(&binary_path, Some("sftp"))); stdfs::create_dir(&directory).expect("create receipt fixture");
let binary = directory.join("rustfs");
stdfs::remove_file(stamp_path).ok(); let receipt = directory.join("receipt.json");
stdfs::write(&binary, "server").expect("write fixture binary");
let metadata = binary.metadata().expect("stat fixture binary");
let record = serde_json::json!({
"schema": 1,
"workspace": directory.canonicalize().expect("canonical workspace"),
"binary": binary.canonicalize().expect("canonical binary"),
"size": metadata.len(),
"modified_ns": metadata.modified().expect("modified time").duration_since(std::time::UNIX_EPOCH).expect("positive timestamp").as_nanos(),
"features": ["default", "full", "ftps", "webdav", "sftp"]
});
stdfs::write(&receipt, serde_json::to_vec(&record).expect("serialize receipt")).expect("write receipt");
verify_e2e_binary_receipt(&receipt, &directory, &binary, Some("sftp,webdav")).expect("resolved feature subset");
verify_e2e_binary_receipt(&receipt, &directory, &binary, Some("full")).expect("full was actually requested");
assert_eq!(
verify_e2e_binary_receipt(&receipt, &directory, &binary, Some("rio-v2"))
.expect_err("full does not enable rio-v2")
.kind(),
ErrorKind::InvalidInput
);
let other = directory.join("old-server");
stdfs::write(&other, "server").expect("write alternate binary");
assert!(verify_e2e_binary_receipt(&receipt, &directory, &other, None).is_err());
stdfs::write(&binary, "different server").expect("replace fixture binary");
assert!(verify_e2e_binary_receipt(&receipt, &directory, &binary, None).is_err());
stdfs::remove_file(&receipt).expect("remove expired receipt");
assert_eq!(
verify_e2e_binary_receipt(&receipt, &directory, &binary, None)
.expect_err("expired receipt")
.kind(),
ErrorKind::NotFound
);
stdfs::remove_dir_all(directory).expect("remove receipt fixture");
} }
/// Build a cluster environment struct in-memory (no ports, no processes) so /// Build a cluster environment struct in-memory (no ports, no processes) so
@@ -20,10 +20,9 @@
//! journal (`count_requests`) carries the assertion in every one of them. //! journal (`count_requests`) carries the assertion in every one of them.
use super::common::{BoxError, OdmTestEnv, RawResponse, SeedObject, start_configured_env}; use super::common::{BoxError, OdmTestEnv, RawResponse, SeedObject, start_configured_env};
use crate::fake_s3_target::{FaultAction, Operation}; use crate::fake_s3_target::Operation;
use aws_sdk_s3::types::{BucketVersioningStatus, VersioningConfiguration}; use aws_sdk_s3::types::{BucketVersioningStatus, VersioningConfiguration};
use bytes::Bytes; use bytes::Bytes;
use futures::{StreamExt, TryStreamExt};
use std::time::Duration; use std::time::Duration;
type TestResult = Result<(), BoxError>; type TestResult = Result<(), BoxError>;
@@ -146,38 +145,14 @@ async fn test_odm_range_burst_overflows_the_pull_queue_without_failing_clients()
.await?; .await?;
let body = payload(128 * 1024); let body = payload(128 * 1024);
let blocker = "queue/blocker.bin";
env.seed_source(SOURCE_BUCKET, &[SeedObject::new(blocker, body.clone())]);
// The one-chunk range completes immediately; its full background pull
// occupies the only slot while the remaining requests fill the queue.
env.source.inject_for_key(
Operation::GetObject,
blocker,
FaultAction::SlowSendBody {
chunk_bytes: 1024,
delay: Duration::from_millis(100),
},
2,
);
let response = env
.raw_object_request(http::Method::GET, bucket, blocker, &[("range", "bytes=0-1023")])
.await?;
assert_eq!(response.status, 206);
assert_eq!(response.body, body.slice(0..1024));
env.wait_for_status_counter(bucket, "/inflight_pulls", 1, SETTLE).await?;
let keys: Vec<String> = (0..REQUESTS).map(|index| format!("queue/object-{index:03}.bin")).collect(); let keys: Vec<String> = (0..REQUESTS).map(|index| format!("queue/object-{index:03}.bin")).collect();
let seeds: Vec<SeedObject> = keys.iter().map(|key| SeedObject::new(key.clone(), body.clone())).collect(); let seeds: Vec<SeedObject> = keys.iter().map(|key| SeedObject::new(key.clone(), body.clone())).collect();
env.seed_source(SOURCE_BUCKET, &seeds); env.seed_source(SOURCE_BUCKET, &seeds);
// Bound source connections below the fixture's limit while still let responses: Vec<RawResponse> = futures::future::try_join_all(
// submitting all 100 requests to the eight-slot background queue.
let responses: Vec<RawResponse> = futures::stream::iter(
keys.iter() keys.iter()
.map(|key| env.raw_object_request(http::Method::GET, bucket, key, &[("range", "bytes=0-1023")])), .map(|key| env.raw_object_request(http::Method::GET, bucket, key, &[("range", "bytes=0-1023")])),
) )
.buffered(16)
.try_collect()
.await?; .await?;
for (key, response) in keys.iter().zip(&responses) { for (key, response) in keys.iter().zip(&responses) {
assert_eq!(response.status, 206, "{key}: {}", String::from_utf8_lossy(&response.body)); assert_eq!(response.status, 206, "{key}: {}", String::from_utf8_lossy(&response.body));
@@ -193,15 +168,6 @@ async fn test_odm_range_burst_overflows_the_pull_queue_without_failing_clients()
.wait_for_status_counter(bucket, "/counters/pull_failures_total/queue_full", 1, SETTLE) .wait_for_status_counter(bucket, "/counters/pull_failures_total/queue_full", 1, SETTLE)
.await?; .await?;
assert!(queue_full > 0, "a 100-deep burst must overflow an 8-slot queue"); assert!(queue_full > 0, "a 100-deep burst must overflow an 8-slot queue");
let queue_full = usize::try_from(queue_full)?;
assert!(queue_full <= REQUESTS);
env.wait_for_status_counter(
bucket,
"/counters/pulled_objects_total/background",
u64::try_from(REQUESTS + 1 - queue_full)?,
SETTLE,
)
.await?;
let ranged_reads: usize = keys.iter().map(|key| source_get_count(&env, key)).sum(); let ranged_reads: usize = keys.iter().map(|key| source_get_count(&env, key)).sum();
assert!( assert!(
@@ -209,6 +175,9 @@ async fn test_odm_range_burst_overflows_the_pull_queue_without_failing_clients()
"every reader is served from the source: {ranged_reads} GETs for {REQUESTS} readers" "every reader is served from the source: {ranged_reads} GETs for {REQUESTS} readers"
); );
let dropped = keys.iter().filter(|key| source_get_count(&env, key) == 1).count(); let dropped = keys.iter().filter(|key| source_get_count(&env, key) == 1).count();
assert_eq!(dropped, queue_full, "only overflowed keys remain without a background GET"); assert!(
dropped > 0,
"the overflowed keys are the ones with no backfill GET, but every key got one"
);
Ok(()) Ok(())
} }
@@ -265,13 +265,16 @@ async fn list_through_rejects_a_tampered_continuation_token() -> TestResult {
let decoded = String::from_utf8(base64_simd::STANDARD.decode_to_vec(token.as_bytes())?)?; let decoded = String::from_utf8(base64_simd::STANDARD.decode_to_vec(token.as_bytes())?)?;
assert!(decoded.contains("\"t\":\"odm-list\""), "the merged token is an envelope: {decoded}"); assert!(decoded.contains("\"t\":\"odm-list\""), "the merged token is an envelope: {decoded}");
let tampered = base64_simd::STANDARD.encode_to_string(decoded.replace("\"v\":1", "\"v\":3").as_bytes()); let tampered = base64_simd::STANDARD.encode_to_string(decoded.replace("\"v\":1", "\"v\":2").as_bytes());
assert_ne!(tampered, token, "the test must change the token version"); let rejected = env
let query = serde_urlencoded::to_string([("continuation-token", tampered.as_str())])?; .raw_list_objects_v2(bucket, &format!("continuation-token={tampered}"))
let rejected = env.raw_list_objects_v2(bucket, &query).await?; .await?;
let error_body = String::from_utf8_lossy(&rejected.body); assert_eq!(
assert_eq!(rejected.status, 400, "a bumped token version is a client error: {}", error_body); rejected.status,
assert!(error_body.contains("<Code>InvalidArgument</Code>"), "{error_body}"); 400,
"a bumped token version is a client error: {}",
String::from_utf8_lossy(&rejected.body)
);
Ok(()) Ok(())
} }
+3 -5
View File
@@ -17,15 +17,13 @@ Use the canonical CI-equivalent protocol command in the parent
For targeted debugging of the core suite only: For targeted debugging of the core suite only:
```bash ```bash
RUSTFS_BUILD_FEATURES=ftps,webdav,sftp cargo test --package e2e_test test_protocol_core_suite -- --test-threads=1 --nocapture python3 scripts/e2e_binary.py build --features ftps,webdav,sftp
python3 scripts/e2e_binary.py run --features ftps,webdav,sftp -- cargo test --package e2e_test test_protocol_core_suite -- --test-threads=1 --nocapture
``` ```
This targeted command does not cover the full `e2e-protocols` profile. This targeted command does not cover the full `e2e-protocols` profile.
`RUSTFS_BUILD_FEATURES` controls which features the test rustfs binary is `e2e_binary.py` supplies `RUSTFS_BUILD_FEATURES` from the verified server's resolved Cargo features. The protocol runner schedules only entries present in that feature list; helpers check that their required features are available without rebuilding the server.
built with. When this variable is set, the protocol test runner schedules
only entries whose protocol is present in the requested feature list. Leave
it unset to run every protocol entry.
`--test-threads=1` is required because every entry spawns a rustfs server `--test-threads=1` is required because every entry spawns a rustfs server
on fixed bind ports. on fixed bind ports.
@@ -12,34 +12,21 @@
// See the License for the specific language governing permissions and // See the License for the specific language governing permissions and
// limitations under the License. // limitations under the License.
use crate::common::{ use crate::common::{RustFSTestClusterEnvironment, RustFSTestEnvironment, init_logging, rustfs_binary_path};
RustFSTestClusterEnvironment, RustFSTestEnvironment, admin_request, init_logging, replication_fast_env, rustfs_binary_path,
};
use crate::fake_s3_target::{BucketMode, FAKE_ACCESS_KEY, FAKE_SECRET_KEY, FakeS3Target};
use crate::on_demand_migration::common::{ODM_SERVER_ENV, OdmTestEnv, SeedObject};
use crate::replication_extension_test::{
LOOPBACK_REPLICATION_TARGET_ENV, ReplicationTargetOptions, put_bucket_replication, set_replication_target_with_options,
};
use aws_sdk_s3::Client; use aws_sdk_s3::Client;
use aws_sdk_s3::error::ProvideErrorMetadata; use aws_sdk_s3::error::ProvideErrorMetadata;
use aws_sdk_s3::primitives::ByteStream; use aws_sdk_s3::primitives::ByteStream;
use aws_sdk_s3::types::{ use aws_sdk_s3::types::{
BucketLifecycleConfiguration, BucketVersioningStatus, CompletedMultipartUpload, CompletedPart, DefaultRetention, BucketVersioningStatus, CompletedMultipartUpload, CompletedPart, ServerSideEncryption, VersioningConfiguration,
ExpirationStatus, LifecycleExpiration, LifecycleRule, LifecycleRuleFilter, ObjectLockConfiguration, ObjectLockEnabled,
ObjectLockRetentionMode, ObjectLockRule, PublicAccessBlockConfiguration, ServerSideEncryption, ServerSideEncryptionByDefault,
ServerSideEncryptionConfiguration, ServerSideEncryptionRule, Tag, Tagging, VersioningConfiguration,
}; };
use http::{Method, StatusCode};
use std::path::{Path, PathBuf}; use std::path::{Path, PathBuf};
use std::time::Duration; use std::time::Duration;
use tokio::task::JoinSet; use tokio::task::JoinSet;
use tokio::time::{Instant, sleep}; use tokio::time::{Instant, sleep};
type TestResult = Result<(), Box<dyn std::error::Error + Send + Sync>>; type TestResult = Result<(), Box<dyn std::error::Error + Send + Sync>>;
type BoxError = Box<dyn std::error::Error + Send + Sync>;
const SOURCE_BINARY_ENV: &str = "RUSTFS_UPGRADE_SOURCE_BINARY"; const SOURCE_BINARY_ENV: &str = "RUSTFS_UPGRADE_SOURCE_BINARY";
const RC5_COMMIT: &str = "40a2470feb567201165a5b809b7598bb4b1f68f5";
const SSE_MASTER_KEY_ENV: &str = "RUSTFS_SSE_S3_MASTER_KEY"; const SSE_MASTER_KEY_ENV: &str = "RUSTFS_SSE_S3_MASTER_KEY";
const SSE_MASTER_KEY: &str = "QkJCQkJCQkJCQkJCQkJCQkJCQkJCQkJCQkJCQkJCQkI="; const SSE_MASTER_KEY: &str = "QkJCQkJCQkJCQkJCQkJCQkJCQkJCQkJCQkJCQkJCQkI=";
const PLAIN_BUCKET: &str = "upgrade-plain-data"; const PLAIN_BUCKET: &str = "upgrade-plain-data";
@@ -53,32 +40,6 @@ const MULTIPART_UPLOADS_PER_WORKER: usize = 16;
// comfortably covers that window plus CI scheduling jitter. // comfortably covers that window plus CI scheduling jitter.
const LISTING_CONVERGENCE_TIMEOUT: Duration = Duration::from_secs(30); const LISTING_CONVERGENCE_TIMEOUT: Duration = Duration::from_secs(30);
// Bucket-configuration upgrade/rollback scenarios (rustfs#7172, #7183, #7089).
const CONFIG_PLAIN_BUCKET: &str = "upgrade-config-plain";
const CONFIG_ENCRYPTED_BUCKET: &str = "upgrade-config-encrypted";
const CONFIG_REPLICATED_BUCKET: &str = "upgrade-config-replicated";
const CONFIG_LOCKED_BUCKET: &str = "upgrade-config-locked";
const CONFIG_REPLICA_BUCKET: &str = "upgrade-config-replica";
const ROLLBACK_BUCKET: &str = "rollback-config-data";
const ROLLBACK_REPLICA_BUCKET: &str = "rollback-config-replica";
const BUCKET_QUOTA_BYTES: u64 = 64 * 1024 * 1024;
const LIFECYCLE_RULE_ID: &str = "upgrade-expire-logs";
const LIFECYCLE_PREFIX: &str = "logs/";
const LIFECYCLE_DAYS: i32 = 30;
const BUCKET_TAG_KEY: &str = "owner";
const BUCKET_TAG_VALUE: &str = "upgrade-compatibility";
const OBJECT_LOCK_DAYS: i32 = 1;
// `set-bucket-quota` answers 503 until the scanner has made the bucket's usage
// authoritative; the quota test uses the same 30s budget.
const QUOTA_READINESS_TIMEOUT: Duration = Duration::from_secs(30);
// Quota admission fails closed while a freshly started server has neither
// authoritative usage nor a persisted degraded baseline for the bucket
// (rustfs#5716), so a write to a quota-enabled bucket is retryable-503 for that
// window. It is a restart property, not an upgrade property — the same window
// opens on the very first start — so the write assertions ride it out instead
// of treating it as an upgrade failure.
const QUOTA_ADMISSION_WARMUP_TIMEOUT: Duration = Duration::from_secs(90);
fn source_binary() -> Result<PathBuf, Box<dyn std::error::Error + Send + Sync>> { fn source_binary() -> Result<PathBuf, Box<dyn std::error::Error + Send + Sync>> {
let path = std::env::var_os(SOURCE_BINARY_ENV) let path = std::env::var_os(SOURCE_BINARY_ENV)
.map(PathBuf::from) .map(PathBuf::from)
@@ -279,93 +240,6 @@ async fn exercise_mixed_cluster(
Ok(()) Ok(())
} }
/// Pins the published old writer's limitation and the supported recovery
/// procedure. This is not a promise that mixed-version ODM is supported.
/// Replace the loss assertion when ODM gains independent persistence;
/// preserving configuration across rc.5 writes is then an improvement.
#[tokio::test]
#[ignore = "requires the pinned 1.0.0-rc.5 release binary"]
async fn rc5_rollback_requires_restoring_odm_configuration() -> TestResult {
init_logging();
let previous_binary = source_binary()?;
let version = tokio::process::Command::new(&previous_binary)
.arg("--version")
.output()
.await?;
assert!(version.status.success(), "previous binary must report its version");
assert!(
String::from_utf8(version.stdout)?.contains(RC5_COMMIT),
"this compatibility scenario requires the published rc.5 writer"
);
let mut env = OdmTestEnv::start().await?;
let bucket = "odm-rc5-rollback";
let source_bucket = "odm-rc5-source";
env.source.create_bucket_with_mode(source_bucket, BucketMode::Unversioned);
env.seed_source(
source_bucket,
&[SeedObject::new(
"source-only",
bytes::Bytes::from_static(b"source read after recovery"),
)],
);
env.rustfs.create_test_bucket(bucket).await?;
let saved_config = env.fake_source_spec(source_bucket);
assert_eq!(env.configure_source(bucket, &saved_config).await?.status, 200);
let before = env.get_config(bucket).await?;
assert_eq!(before.status, 200);
let expected_config = before
.json()?
.get("config")
.cloned()
.ok_or("configuration response omitted config")?;
env.client
.put_object()
.bucket(bucket)
.key("local")
.body(ByteStream::from_static(b"local data survives rollback"))
.send()
.await?;
env.rustfs.restart_server_preserving_data(vec![], ODM_SERVER_ENV).await?;
let restarted = env.get_config(bucket).await?;
assert_eq!(restarted.status, 200, "a current writer preserves ODM across restart");
assert_eq!(restarted.json()?.get("config"), Some(&expected_config));
restart_from_binary(&mut env.rustfs, &previous_binary, &[]).await?;
env.client
.put_bucket_tagging()
.bucket(bucket)
.tagging(
Tagging::builder()
.tag_set(Tag::builder().key("writer").value("rc5").build()?)
.build()?,
)
.send()
.await?;
env.rustfs.restart_server_preserving_data(vec![], ODM_SERVER_ENV).await?;
let missing = env.get_config(bucket).await?;
assert_eq!(missing.status, 404, "rc.5 rewrites metadata without ODM keys");
assert!(missing.body.contains("NoSuchConfiguration"));
assert_eq!(read_object(&env.client, bucket, "local", None).await?.1, b"local data survives rollback");
let tags = env.client.get_bucket_tagging().bucket(bucket).send().await?;
assert!(tags.tag_set().iter().any(|tag| tag.key() == "writer" && tag.value() == "rc5"));
assert_eq!(
env.configure_source(bucket, &saved_config).await?.status,
200,
"restore from saved full configuration"
);
env.rustfs.restart_server_preserving_data(vec![], ODM_SERVER_ENV).await?;
let restored = env.get_config(bucket).await?;
assert_eq!(restored.status, 200, "restored ODM configuration persists");
assert_eq!(restored.json()?.get("config"), Some(&expected_config));
env.wait_until_source_consulted(bucket).await?;
assert_eq!(
read_object(&env.client, bucket, "source-only", None).await?.1,
b"source read after recovery"
);
Ok(())
}
#[tokio::test] #[tokio::test]
#[ignore = "requires a pinned previous RustFS release binary"] #[ignore = "requires a pinned previous RustFS release binary"]
async fn direct_upgrade_from_rc2_preserves_object_contracts() -> TestResult { async fn direct_upgrade_from_rc2_preserves_object_contracts() -> TestResult {
@@ -555,653 +429,3 @@ async fn rolling_upgrade_from_rc2_preserves_mixed_version_contracts() -> TestRes
Ok(()) Ok(())
} }
/// Child-process environment shared by both bucket-configuration scenarios.
///
/// The replication target is an in-process fake bound to `127.0.0.1`, which
/// `set-remote-target` rejects as an SSRF risk without the loopback opt-in, and
/// the proxy bypass keeps a developer's `HTTP_PROXY` from intercepting the
/// server's outbound health check.
fn bucket_config_server_env() -> Vec<(&'static str, &'static str)> {
let mut env = vec![
(SSE_MASTER_KEY_ENV, SSE_MASTER_KEY),
("NO_PROXY", "127.0.0.1,localhost"),
("HTTP_PROXY", ""),
("HTTPS_PROXY", ""),
// Shorten the scanner cycle so the bucket's usage becomes authoritative
// in seconds; both `set-bucket-quota` and quota admission block on it.
("RUSTFS_SCANNER_CYCLE", "1"),
("RUSTFS_SCANNER_START_DELAY_SECS", "0"),
];
env.extend_from_slice(LOOPBACK_REPLICATION_TARGET_ENV);
env.extend(replication_fast_env());
env
}
/// Restart `env` in place on the same data directory using an explicit binary.
///
/// [`RustFSTestEnvironment::restart_server_preserving_data`] always relaunches
/// the workspace build, which is the upgrade direction only. The rollback
/// scenario needs the reverse: stop the current build and bring the pinned
/// previous release up on the metadata that build just wrote.
async fn restart_from_binary(env: &mut RustFSTestEnvironment, binary: &Path, server_env: &[(&str, &str)]) -> TestResult {
env.stop_server();
env.start_rustfs_server_from_binary(binary, vec![], server_env).await
}
async fn set_bucket_quota(env: &RustFSTestEnvironment, bucket: &str, quota_bytes: u64) -> TestResult {
let path = format!("/rustfs/admin/v3/quota/{bucket}");
let body = serde_json::json!({ "quota": quota_bytes, "quota_type": "HARD" }).to_string();
let deadline = Instant::now() + QUOTA_READINESS_TIMEOUT;
loop {
let (status, response) =
admin_request(&env.url, Method::PUT, &path, Some(body.clone()), &env.access_key, &env.secret_key).await?;
if status.is_success() {
return Ok(());
}
if status != StatusCode::SERVICE_UNAVAILABLE || Instant::now() >= deadline {
return Err(format!("setting the quota of {bucket} failed: {status} {response}").into());
}
sleep(Duration::from_millis(500)).await;
}
}
/// PUT into a quota-enabled bucket, riding out the post-start quota-admission
/// warm-up described on [`QUOTA_ADMISSION_WARMUP_TIMEOUT`].
///
/// Only `ServiceUnavailable` is retried: any other failure, and a warm-up that
/// never ends, is a genuine regression and surfaces as an error.
async fn put_object_through_quota_warmup(client: &Client, bucket: &str, key: &str, body: &'static [u8]) -> TestResult {
let deadline = Instant::now() + QUOTA_ADMISSION_WARMUP_TIMEOUT;
loop {
let result = client
.put_object()
.bucket(bucket)
.key(key)
.body(ByteStream::from_static(body))
.send()
.await;
let error = match result {
Ok(_) => return Ok(()),
Err(error) => error,
};
let retryable = error.as_service_error().and_then(ProvideErrorMetadata::code) == Some("ServiceUnavailable");
if !retryable || Instant::now() >= deadline {
return Err(format!("PUT {bucket}/{key} failed after the quota warm-up window: {error}").into());
}
sleep(Duration::from_millis(500)).await;
}
}
async fn get_bucket_quota(env: &RustFSTestEnvironment, bucket: &str) -> Result<Option<u64>, BoxError> {
let path = format!("/rustfs/admin/v3/quota/{bucket}");
let (status, response) = admin_request(&env.url, Method::GET, &path, None, &env.access_key, &env.secret_key).await?;
if status != StatusCode::OK {
return Err(format!("reading the quota of {bucket} failed: {status} {response}").into());
}
let quota: serde_json::Value = serde_json::from_str(&response)?;
Ok(quota.get("quota").and_then(serde_json::Value::as_u64))
}
/// `GET /rustfs/admin/v3/list-remote-targets?bucket=...`.
///
/// Returns an error for any non-200, because rustfs#7172 made this endpoint
/// fail closed on a `bucket-targets.json` blob the running build cannot parse.
/// An upgrade that misreads a blob written by the previous release therefore
/// shows up here as an error, and a silently dropped target shows up as an
/// empty list — the caller must distinguish the two.
async fn list_remote_targets(env: &RustFSTestEnvironment, bucket: &str) -> Result<Vec<serde_json::Value>, BoxError> {
let path = format!("/rustfs/admin/v3/list-remote-targets?bucket={}", urlencoding::encode(bucket));
let (status, response) = admin_request(&env.url, Method::GET, &path, None, &env.access_key, &env.secret_key).await?;
if status != StatusCode::OK {
return Err(format!("list-remote-targets for {bucket} failed: {status} {response}").into());
}
Ok(serde_json::from_str(&response)?)
}
/// Assert that `bucket` still carries exactly the replication target `arn`.
async fn assert_remote_target_preserved(env: &RustFSTestEnvironment, bucket: &str, arn: &str, context: &str) -> TestResult {
let targets = list_remote_targets(env, bucket).await?;
assert_eq!(
targets.len(),
1,
"{context}: list-remote-targets must still report the single configured target, got {targets:?}"
);
assert_eq!(
targets[0].get("arn").and_then(serde_json::Value::as_str),
Some(arn),
"{context}: the target ARN changed across the restart: {targets:?}"
);
Ok(())
}
/// Configure a replication target on `bucket` pointing at the in-process fake,
/// then attach an enabled replication rule for it. Returns the target ARN.
async fn configure_replication(
env: &RustFSTestEnvironment,
bucket: &str,
target: &FakeS3Target,
target_bucket: &str,
) -> Result<String, BoxError> {
let arn = set_replication_target_with_options(
env,
bucket,
ReplicationTargetOptions {
endpoint: &target.address(),
access_key: FAKE_ACCESS_KEY,
secret_key: FAKE_SECRET_KEY,
target_bucket,
secure: false,
skip_tls_verify: false,
ca_cert_pem: None,
},
)
.await?;
put_bucket_replication(env, bucket, &arn).await?;
Ok(arn)
}
async fn put_default_sse_s3_encryption(client: &Client, bucket: &str) -> TestResult {
let configuration = ServerSideEncryptionConfiguration::builder()
.rules(
ServerSideEncryptionRule::builder()
.apply_server_side_encryption_by_default(
ServerSideEncryptionByDefault::builder()
.sse_algorithm(ServerSideEncryption::Aes256)
.build()?,
)
.build(),
)
.build()?;
client
.put_bucket_encryption()
.bucket(bucket)
.server_side_encryption_configuration(configuration)
.send()
.await?;
Ok(())
}
async fn assert_default_sse_s3_encryption(client: &Client, bucket: &str, context: &str) -> TestResult {
let response = client.get_bucket_encryption().bucket(bucket).send().await?;
let rules = response
.server_side_encryption_configuration()
.ok_or("GetBucketEncryption omitted the configuration")?
.rules();
assert_eq!(rules.len(), 1, "{context}: expected exactly one encryption rule, got {rules:?}");
assert_eq!(
rules[0]
.apply_server_side_encryption_by_default()
.map(ServerSideEncryptionByDefault::sse_algorithm),
Some(&ServerSideEncryption::Aes256),
"{context}: the default encryption algorithm changed"
);
Ok(())
}
async fn put_bucket_tag(client: &Client, bucket: &str) -> TestResult {
let tagging = Tagging::builder()
.tag_set(Tag::builder().key(BUCKET_TAG_KEY).value(BUCKET_TAG_VALUE).build()?)
.build()?;
client.put_bucket_tagging().bucket(bucket).tagging(tagging).send().await?;
Ok(())
}
async fn assert_bucket_tag(client: &Client, bucket: &str, context: &str) -> TestResult {
let tags = client.get_bucket_tagging().bucket(bucket).send().await?;
let tag_set = tags.tag_set();
assert_eq!(tag_set.len(), 1, "{context}: expected exactly one bucket tag, got {tag_set:?}");
assert_eq!(tag_set[0].key(), BUCKET_TAG_KEY, "{context}: bucket tag key changed");
assert_eq!(tag_set[0].value(), BUCKET_TAG_VALUE, "{context}: bucket tag value changed");
Ok(())
}
async fn assert_versioning_enabled(client: &Client, bucket: &str, context: &str) -> TestResult {
let versioning = client.get_bucket_versioning().bucket(bucket).send().await?;
assert_eq!(
versioning.status(),
Some(&BucketVersioningStatus::Enabled),
"{context}: versioning is no longer Enabled on {bucket}"
);
Ok(())
}
fn bucket_policy_document(bucket: &str) -> serde_json::Value {
serde_json::json!({
"Version": "2012-10-17",
"Statement": [{
"Sid": "UpgradePublicRead",
"Effect": "Allow",
"Principal": { "AWS": ["*"] },
"Action": ["s3:GetObject"],
"Resource": [format!("arn:aws:s3:::{bucket}/public/*")]
}]
})
}
/// `GET .../on-demand-migration/{bucket}/status`.
///
/// The migration module defaults on from rustfs#7089, so a bucket that never
/// configured a source must still answer `configured: false` rather than
/// engaging the migration path.
async fn assert_migration_not_configured(env: &RustFSTestEnvironment, bucket: &str) -> TestResult {
let path = format!("/rustfs/admin/v3/on-demand-migration/{bucket}/status");
let (status, response) = admin_request(&env.url, Method::GET, &path, None, &env.access_key, &env.secret_key).await?;
assert_eq!(
status,
StatusCode::OK,
"the migration status endpoint must answer for an unconfigured bucket: {status} {response}"
);
let body: serde_json::Value = serde_json::from_str(&response)?;
assert_eq!(
body.get("configured"),
Some(&serde_json::Value::Bool(false)),
"a bucket upgraded from the previous release must not look migration-configured: {body}"
);
Ok(())
}
/// A GET for a key that was never written must be a plain `NoSuchKey`.
///
/// With the migration module on by default this is the cheap proof that an
/// unconfigured bucket never consults a source: any migration engagement would
/// surface as a different status or error code here.
async fn assert_missing_key_is_no_such_key(client: &Client, bucket: &str, key: &str) -> TestResult {
let error = client
.get_object()
.bucket(bucket)
.key(key)
.send()
.await
.expect_err("a key that was never written must not be readable");
assert_eq!(
error.raw_response().map(|response| response.status().as_u16()),
Some(404),
"a missing key must stay a 404 on a bucket with no migration configuration"
);
assert_eq!(
error.as_service_error().and_then(ProvideErrorMetadata::code),
Some("NoSuchKey"),
"a missing key must stay NoSuchKey on a bucket with no migration configuration"
);
Ok(())
}
/// Bucket configuration written by the pinned previous release must survive an
/// upgrade to the current build unchanged, and must keep working.
///
/// This pins the three on-disk surfaces the on-demand-migration series moved:
///
/// * `BucketMetadata` grew two msgpack keys (encoded map length 44 -> 46), so
/// every configuration read below decodes a 44-key blob on 46-key code.
/// * rustfs#7172 made an unreadable `bucket-targets.json` / encryption /
/// public-access-block / quota blob "present but unreadable" instead of
/// silently defaulting, and made `list-remote-targets` fail closed on it. A
/// replication target configured by the old release must therefore still be
/// *listed*, not dropped and not an error.
/// * rustfs#7183 made the object write path refuse a PUT when the bucket's
/// encryption configuration cannot be read, so a misparsed SSE config would
/// turn every PUT to that bucket into a 500.
///
/// Not covered on purpose: on-demand-migration configuration itself, which the
/// previous release has no public API for — the reverse direction is asserted
/// instead (an upgraded bucket reports `configured: false`).
#[tokio::test]
#[ignore = "requires a pinned previous RustFS release binary"]
async fn direct_upgrade_from_previous_release_preserves_bucket_configuration() -> TestResult {
init_logging();
let previous_binary = source_binary()?;
// In-process: the fake target outlives both server processes, so the
// replication target stays reachable across the upgrade.
let replication_target = FakeS3Target::start().await?;
replication_target.create_bucket(CONFIG_REPLICA_BUCKET);
let mut env = RustFSTestEnvironment::new().await?;
let server_env = bucket_config_server_env();
env.start_rustfs_server_from_binary(&previous_binary, vec![], &server_env)
.await?;
let old_client = env.create_s3_client();
env.create_test_bucket(CONFIG_PLAIN_BUCKET).await?;
env.create_test_bucket(CONFIG_ENCRYPTED_BUCKET).await?;
env.create_test_bucket(CONFIG_REPLICATED_BUCKET).await?;
old_client
.create_bucket()
.bucket(CONFIG_LOCKED_BUCKET)
.object_lock_enabled_for_bucket(true)
.send()
.await?;
// Plain bucket: policy, tags, lifecycle, quota.
let policy = bucket_policy_document(CONFIG_PLAIN_BUCKET);
old_client
.put_bucket_policy()
.bucket(CONFIG_PLAIN_BUCKET)
.policy(policy.to_string())
.send()
.await?;
put_bucket_tag(&old_client, CONFIG_PLAIN_BUCKET).await?;
old_client
.put_bucket_lifecycle_configuration()
.bucket(CONFIG_PLAIN_BUCKET)
.lifecycle_configuration(
BucketLifecycleConfiguration::builder()
.rules(
LifecycleRule::builder()
.id(LIFECYCLE_RULE_ID)
.status(ExpirationStatus::Enabled)
.filter(LifecycleRuleFilter::builder().prefix(LIFECYCLE_PREFIX).build())
.expiration(LifecycleExpiration::builder().days(LIFECYCLE_DAYS).build())
.build()?,
)
.build()?,
)
.send()
.await?;
set_bucket_quota(&env, CONFIG_PLAIN_BUCKET, BUCKET_QUOTA_BYTES).await?;
// Encrypted bucket: SSE-S3 default encryption plus a fully restrictive
// public access block, both of which rustfs#7172 now fails closed on.
put_default_sse_s3_encryption(&old_client, CONFIG_ENCRYPTED_BUCKET).await?;
old_client
.put_public_access_block()
.bucket(CONFIG_ENCRYPTED_BUCKET)
.public_access_block_configuration(
PublicAccessBlockConfiguration::builder()
.block_public_acls(true)
.ignore_public_acls(true)
.block_public_policy(true)
.restrict_public_buckets(true)
.build(),
)
.send()
.await?;
// Replicated bucket: versioning, a validated remote target, a rule.
enable_versioning(&old_client, CONFIG_REPLICATED_BUCKET).await?;
let target_arn = configure_replication(&env, CONFIG_REPLICATED_BUCKET, &replication_target, CONFIG_REPLICA_BUCKET).await?;
assert_remote_target_preserved(&env, CONFIG_REPLICATED_BUCKET, &target_arn, "before the upgrade").await?;
// Object-lock bucket: a default GOVERNANCE retention on a fresh bucket.
old_client
.put_object_lock_configuration()
.bucket(CONFIG_LOCKED_BUCKET)
.object_lock_configuration(
ObjectLockConfiguration::builder()
.object_lock_enabled(ObjectLockEnabled::Enabled)
.rule(
ObjectLockRule::builder()
.default_retention(
DefaultRetention::builder()
.mode(ObjectLockRetentionMode::Governance)
.days(OBJECT_LOCK_DAYS)
.build(),
)
.build(),
)
.build(),
)
.send()
.await?;
let plain_key = "plain/written-by-previous";
let plain_bytes = b"plain object written by the previous RustFS release";
put_object_through_quota_warmup(&old_client, CONFIG_PLAIN_BUCKET, plain_key, plain_bytes).await?;
let encrypted_key = "encrypted/written-by-previous";
let encrypted_bytes = b"default-encrypted object written by the previous RustFS release";
old_client
.put_object()
.bucket(CONFIG_ENCRYPTED_BUCKET)
.key(encrypted_key)
.body(ByteStream::from_static(encrypted_bytes))
.send()
.await?;
assert_eq!(
read_object(&old_client, CONFIG_ENCRYPTED_BUCKET, encrypted_key, None)
.await?
.0,
Some(ServerSideEncryption::Aes256),
"the previous release must apply the bucket default encryption it just accepted"
);
// The multipart object lives in the default-encrypted bucket so the
// upgraded build has to reassemble parts *and* re-derive the object key.
let multipart_key = "encrypted/multipart-written-by-previous";
let multipart_parts = vec![vec![b'm'; 5 * 1024 * 1024], b"final multipart bytes".to_vec()];
let multipart_bytes = multipart_parts.concat();
write_multipart(&old_client, CONFIG_ENCRYPTED_BUCKET, multipart_key, &multipart_parts).await?;
let versioned_key = "versioned/written-by-previous";
let versioned_bytes = b"versioned object written by the previous RustFS release";
let versioned_id = old_client
.put_object()
.bucket(CONFIG_REPLICATED_BUCKET)
.key(versioned_key)
.body(ByteStream::from_static(versioned_bytes))
.send()
.await?
.version_id()
.ok_or("versioned PUT omitted version ID")?
.to_string();
env.restart_server_preserving_data(vec![], &server_env).await?;
let new_client = env.create_s3_client();
// Every configuration must read back unchanged on the upgraded build.
let upgraded_policy = new_client.get_bucket_policy().bucket(CONFIG_PLAIN_BUCKET).send().await?;
let upgraded_policy: serde_json::Value =
serde_json::from_str(upgraded_policy.policy().ok_or("GetBucketPolicy omitted the document")?)?;
assert_eq!(upgraded_policy, policy, "the bucket policy changed across the upgrade");
assert_bucket_tag(&new_client, CONFIG_PLAIN_BUCKET, "after the upgrade").await?;
let lifecycle = new_client
.get_bucket_lifecycle_configuration()
.bucket(CONFIG_PLAIN_BUCKET)
.send()
.await?;
let rules = lifecycle.rules();
assert_eq!(rules.len(), 1, "the lifecycle rule count changed across the upgrade: {rules:?}");
assert_eq!(rules[0].id(), Some(LIFECYCLE_RULE_ID));
assert_eq!(rules[0].status(), &ExpirationStatus::Enabled);
assert_eq!(
rules[0].expiration().and_then(LifecycleExpiration::days),
Some(LIFECYCLE_DAYS),
"the lifecycle expiration changed across the upgrade"
);
assert_eq!(
get_bucket_quota(&env, CONFIG_PLAIN_BUCKET).await?,
Some(BUCKET_QUOTA_BYTES),
"the bucket quota changed across the upgrade"
);
assert_default_sse_s3_encryption(&new_client, CONFIG_ENCRYPTED_BUCKET, "after the upgrade").await?;
let public_access_block = new_client
.get_public_access_block()
.bucket(CONFIG_ENCRYPTED_BUCKET)
.send()
.await?;
let public_access_block = public_access_block
.public_access_block_configuration()
.ok_or("GetPublicAccessBlock omitted the configuration")?;
assert_eq!(public_access_block.block_public_acls(), Some(true));
assert_eq!(public_access_block.ignore_public_acls(), Some(true));
assert_eq!(public_access_block.block_public_policy(), Some(true));
assert_eq!(public_access_block.restrict_public_buckets(), Some(true));
assert_versioning_enabled(&new_client, CONFIG_REPLICATED_BUCKET, "after the upgrade").await?;
// rustfs#7172: neither an empty list nor an error is acceptable here.
assert_remote_target_preserved(&env, CONFIG_REPLICATED_BUCKET, &target_arn, "after the upgrade").await?;
let replication = new_client
.get_bucket_replication()
.bucket(CONFIG_REPLICATED_BUCKET)
.send()
.await?;
let replication_rules = replication
.replication_configuration()
.ok_or("GetBucketReplication omitted the configuration")?
.rules();
assert_eq!(
replication_rules.len(),
1,
"the replication rule count changed across the upgrade: {replication_rules:?}"
);
assert_eq!(
replication_rules[0].destination().map(|destination| destination.bucket()),
Some(target_arn.as_str()),
"the replication rule no longer points at the configured target"
);
let object_lock = new_client
.get_object_lock_configuration()
.bucket(CONFIG_LOCKED_BUCKET)
.send()
.await?;
let object_lock = object_lock
.object_lock_configuration()
.ok_or("GetObjectLockConfiguration omitted the configuration")?;
assert_eq!(object_lock.object_lock_enabled(), Some(&ObjectLockEnabled::Enabled));
let retention = object_lock
.rule()
.and_then(ObjectLockRule::default_retention)
.ok_or("the object lock configuration lost its default retention")?;
assert_eq!(retention.mode(), Some(&ObjectLockRetentionMode::Governance));
assert_eq!(retention.days(), Some(OBJECT_LOCK_DAYS));
// rustfs#7183: a PUT into the default-encrypted bucket must still succeed
// and still come back encrypted.
let post_upgrade_encrypted_key = "encrypted/written-after-upgrade";
let post_upgrade_encrypted_bytes = b"default-encrypted object written by the current RustFS build";
new_client
.put_object()
.bucket(CONFIG_ENCRYPTED_BUCKET)
.key(post_upgrade_encrypted_key)
.body(ByteStream::from_static(post_upgrade_encrypted_bytes))
.send()
.await?;
let (encryption, body) = read_object(&new_client, CONFIG_ENCRYPTED_BUCKET, post_upgrade_encrypted_key, None).await?;
assert_eq!(
encryption,
Some(ServerSideEncryption::Aes256),
"a PUT after the upgrade lost the bucket default encryption"
);
assert_eq!(body, post_upgrade_encrypted_bytes);
let post_upgrade_plain_key = "plain/written-after-upgrade";
let post_upgrade_plain_bytes = b"plain object written by the current RustFS build";
put_object_through_quota_warmup(&new_client, CONFIG_PLAIN_BUCKET, post_upgrade_plain_key, post_upgrade_plain_bytes).await?;
let (encryption, body) = read_object(&new_client, CONFIG_PLAIN_BUCKET, post_upgrade_plain_key, None).await?;
assert_eq!(encryption, None, "a bucket without default encryption must not encrypt a PUT");
assert_eq!(body, post_upgrade_plain_bytes);
// Every object written by the previous release reads back byte-identical.
assert_eq!(read_object(&new_client, CONFIG_PLAIN_BUCKET, plain_key, None).await?.1, plain_bytes);
let (encryption, body) = read_object(&new_client, CONFIG_ENCRYPTED_BUCKET, encrypted_key, None).await?;
assert_eq!(encryption, Some(ServerSideEncryption::Aes256));
assert_eq!(body, encrypted_bytes);
let (encryption, body) = read_object(&new_client, CONFIG_ENCRYPTED_BUCKET, multipart_key, None).await?;
assert_eq!(encryption, Some(ServerSideEncryption::Aes256));
assert_eq!(body, multipart_bytes, "the multipart object did not survive the upgrade");
assert_eq!(
read_object(&new_client, CONFIG_REPLICATED_BUCKET, versioned_key, Some(&versioned_id))
.await?
.1,
versioned_bytes
);
// rustfs#7089: the migration module is on by default, but a bucket that
// never configured a source behaves exactly as before.
assert_migration_not_configured(&env, CONFIG_PLAIN_BUCKET).await?;
assert_missing_key_is_no_such_key(&new_client, CONFIG_PLAIN_BUCKET, "plain/never-written").await?;
replication_target.shutdown().await;
Ok(())
}
/// Rolling back to the pinned previous release must still read the bucket
/// metadata the current build wrote.
///
/// This is the other half of the `BucketMetadata` 44 -> 46 key change: the
/// current build writes a 46-key msgpack map with `OnDemandMigrationConfigJSON`
/// and `OnDemandMigrationConfigUpdatedAt`, and the previous release's decoder
/// has to skip those two unknown keys instead of failing the whole blob. If it
/// did not, every configuration read below would come back empty or error and
/// the rollback would silently discard the bucket's configuration.
#[tokio::test]
#[ignore = "requires a pinned previous RustFS release binary"]
async fn rollback_to_previous_release_reads_current_bucket_metadata() -> TestResult {
init_logging();
let previous_binary = source_binary()?;
let replication_target = FakeS3Target::start().await?;
replication_target.create_bucket(ROLLBACK_REPLICA_BUCKET);
let mut env = RustFSTestEnvironment::new().await?;
let server_env = bucket_config_server_env();
env.start_rustfs_server_with_env(vec![], &server_env).await?;
let new_client = env.create_s3_client();
env.create_test_bucket(ROLLBACK_BUCKET).await?;
enable_versioning(&new_client, ROLLBACK_BUCKET).await?;
put_default_sse_s3_encryption(&new_client, ROLLBACK_BUCKET).await?;
put_bucket_tag(&new_client, ROLLBACK_BUCKET).await?;
let target_arn = configure_replication(&env, ROLLBACK_BUCKET, &replication_target, ROLLBACK_REPLICA_BUCKET).await?;
assert_remote_target_preserved(&env, ROLLBACK_BUCKET, &target_arn, "before the rollback").await?;
let single_key = "rollback/single";
let single_bytes = b"single-part object written by the current RustFS build";
let single_version = new_client
.put_object()
.bucket(ROLLBACK_BUCKET)
.key(single_key)
.body(ByteStream::from_static(single_bytes))
.send()
.await?
.version_id()
.ok_or("versioned PUT omitted version ID")?
.to_string();
let multipart_key = "rollback/multipart";
let multipart_parts = vec![vec![b'r'; 5 * 1024 * 1024], b"final rollback bytes".to_vec()];
let multipart_bytes = multipart_parts.concat();
write_multipart(&new_client, ROLLBACK_BUCKET, multipart_key, &multipart_parts).await?;
restart_from_binary(&mut env, &previous_binary, &server_env).await?;
let old_client = env.create_s3_client();
assert_versioning_enabled(&old_client, ROLLBACK_BUCKET, "after the rollback").await?;
assert_default_sse_s3_encryption(&old_client, ROLLBACK_BUCKET, "after the rollback").await?;
assert_bucket_tag(&old_client, ROLLBACK_BUCKET, "after the rollback").await?;
assert_remote_target_preserved(&env, ROLLBACK_BUCKET, &target_arn, "after the rollback").await?;
let (encryption, body) = read_object(&old_client, ROLLBACK_BUCKET, single_key, Some(&single_version)).await?;
assert_eq!(encryption, Some(ServerSideEncryption::Aes256));
assert_eq!(body, single_bytes);
let (encryption, body) = read_object(&old_client, ROLLBACK_BUCKET, multipart_key, None).await?;
assert_eq!(encryption, Some(ServerSideEncryption::Aes256));
assert_eq!(body, multipart_bytes, "the multipart object did not survive the rollback");
// A PUT on the rolled-back release must still honour the encryption
// configuration it decoded out of the current build's metadata blob.
let post_rollback_key = "rollback/written-after-rollback";
let post_rollback_bytes = b"object written by the previous RustFS release after the rollback";
old_client
.put_object()
.bucket(ROLLBACK_BUCKET)
.key(post_rollback_key)
.body(ByteStream::from_static(post_rollback_bytes))
.send()
.await?;
let (encryption, body) = read_object(&old_client, ROLLBACK_BUCKET, post_rollback_key, None).await?;
assert_eq!(
encryption,
Some(ServerSideEncryption::Aes256),
"the rolled-back release lost the bucket default encryption"
);
assert_eq!(body, post_rollback_bytes);
replication_target.shutdown().await;
Ok(())
}
+2 -3
View File
@@ -31,7 +31,6 @@ workspace = true
[features] [features]
default = [] default = []
gcs = ["dep:google-cloud-storage", "dep:google-cloud-auth"]
# Compiles the controlled list-objects namespace-journal chaos injector into a # Compiles the controlled list-objects namespace-journal chaos injector into a
# production binary (it is always available to tests). Off by default so the # production binary (it is always available to tests). Off by default so the
# RUSTFS_LIST_OBJECTS_NAMESPACE_JOURNAL_CHAOS_* env vars cannot rewrite journal # RUSTFS_LIST_OBJECTS_NAMESPACE_JOURNAL_CHAOS_* env vars cannot rewrite journal
@@ -213,8 +212,8 @@ aws-smithy-runtime-api = { workspace = true, features = ["http-1x"] }
parking_lot = { workspace = true } parking_lot = { workspace = true }
base64-simd.workspace = true base64-simd.workspace = true
serde_urlencoded.workspace = true serde_urlencoded.workspace = true
google-cloud-storage = { workspace = true, optional = true } google-cloud-storage = { workspace = true }
google-cloud-auth = { workspace = true, optional = true } google-cloud-auth = { workspace = true }
faster-hex = { workspace = true } faster-hex = { workspace = true }
ratelimit = { workspace = true } ratelimit = { workspace = true }
aws-smithy-http-client = { workspace = true, default-features = false, features = ["rustls-aws-lc"] } aws-smithy-http-client = { workspace = true, default-features = false, features = ["rustls-aws-lc"] }
+57 -26
View File
@@ -69,13 +69,6 @@ pub mod bucket {
}; };
} }
pub mod recovery_control {
pub use crate::bucket::lifecycle::recovery_control::{
IlmRecoveryClassification, IlmRecoveryControlPage, IlmRecoveryControlView, IlmRecoveryProtocol,
inspect_recovery_control, list_recovery_controls,
};
}
pub mod transition_transaction { pub mod transition_transaction {
pub use crate::bucket::lifecycle::transition_transaction::{ pub use crate::bucket::lifecycle::transition_transaction::{
TransitionOperatorDeleteResult, TransitionOperatorError, TransitionOperatorProbe, TransitionOperatorStatus, TransitionOperatorDeleteResult, TransitionOperatorError, TransitionOperatorProbe, TransitionOperatorStatus,
@@ -153,23 +146,67 @@ pub mod bucket {
}; };
} }
pub mod on_demand_migration {
pub use crate::bucket::on_demand_migration::{
ApplyOutcome, BREAKER_FAILURE_THRESHOLD, BREAKER_FAILURE_WINDOW, BREAKER_HALF_OPEN_MAX_PROBES, BREAKER_OPEN_DURATION,
Breaker, BreakerState, BreakerTransition, BreakerVerdict, BucketOdmState, GLOBAL_ON_DEMAND_MIGRATION_SYS, GaugeGuard,
LastSourceError, LatencyBucketSnapshot, NEGATIVE_CACHE_MAX_ENTRIES, NegativeCache, OdmBucketSnapshot, OdmLookup,
OdmOp, OdmOutcome, OdmStateError, OdmStats, OdmStatsSnapshot, OnDemandMigrationSys, PullError, PullFailureReason,
PullFollower, PullLeader, PullOutcome, PullPath, PullResult, PullSlot, SOURCE_LATENCY_BUCKET_BOUNDS_MS,
SourceLatencySnapshot, source_client_spec,
};
pub use crate::bucket::on_demand_migration::{
ConfigPublishHook, FilterConfig, HeadPolicy, ON_DEMAND_MIGRATION_CONFIG_HOOK, ON_DEMAND_MIGRATION_CONFIG_VERSION,
OnDemandMigrationConfig, OnDemandMigrationConfigError, PathStyle, PolicyConfig, Provider, RangeGetPolicy,
SourceConfig, SourceCredentials, SourceErrorPolicy, SourceTimeout, TlsConfig, ValidationContext,
};
pub use crate::bucket::on_demand_migration::{
EnqueueOutcome, LocalObject, MAX_MULTIPART_PARTS, OdmWriteBack, PULL_MAX_RETRIES, PULL_RETRY_BASE_DELAYS,
PullCompletion, PullQueue, PullReason, PullSource, QueuedPullOutcome, SourceBody, SourceIdleGuard, WriteBackBody,
WriteBackError, WriteBackOutcome, WriteBackPart, WriteBackRequest, commit_inline, commit_inline_with,
idle_guarded_body,
};
pub use crate::bucket::on_demand_migration::{
FetchRequest, LIST_THROUGH_TOKEN_VERSION, ListEntryKey, ListThroughCursor, ListThroughMerger, ListThroughToken,
ListThroughTokenError, MAX_LIST_FETCHES_PER_SIDE, MergeOutcome, MergePick, MergeSide, SOURCE_LIST_MAX_RATE_WAIT,
SOURCE_LIST_RATE_PER_SEC, SourceListPlan, SourceListRateLimiter, decode_continuation_token, source_list_plan,
};
pub mod backfill {
pub use crate::bucket::on_demand_migration::backfill::{
BACKFILL_CHECKPOINT_FILE, BACKFILL_CHECKPOINT_FORMAT_VERSION, BACKFILL_FAILED_KEYS_CAPACITY, BACKFILL_LEASE,
BACKFILL_LEASE_LOCK_PREFIX, BACKFILL_LIST_PAGE_SIZE, BACKFILL_RECOVERY_INTERVAL, BACKFILL_SAVE_EVERY_KEYS,
BACKFILL_SAVE_INTERVAL, BackfillCheckpoint, BackfillContext, BackfillContextFactory, BackfillError,
BackfillLastError, BackfillOwner, BackfillRecoveryStats, BackfillRequest, BackfillRunner, BackfillState,
BucketBackfillContext, LocalBackfillObject, PriorityPullPermits, PullPermit, PullPriority, SkipExisting,
StoredCheckpoint, SysBackfillContexts, global_backfill_runner, install_global_backfill_runner, key_hash,
read_checkpoint, run_backfill_recovery_loop, spawn_backfill_recovery_loop,
};
}
pub mod source_client {
pub use crate::bucket::on_demand_migration::source_client::{
SourceClient, SourceClientSpec, SourceError, SourceGet, SourceHead, SourceListRequest, SourceObject, SourcePage,
SourceProbe, SourceProvider, SourceSse, SourceTimeouts, USER_AGENT_SUFFIX, is_multipart_etag, range_header_value,
resolve_path_style,
};
}
}
pub mod metadata_sys { pub mod metadata_sys {
#[cfg(feature = "test-util")]
pub use crate::bucket::metadata_sys::ConfigWriteLockProbe;
pub use crate::bucket::metadata_sys::{ pub use crate::bucket::metadata_sys::{
BUCKET_CONFIG_PUBLISH_HOOK, BucketConfigPublishHook, BucketMetadataMutationGuard, BucketMetadataSys, BucketMetadataMutationGuard, BucketMetadataSys, ObjectLockConfigState, acquire_bucket_metadata_transaction_lock,
ObjectLockConfigState, acquire_bucket_metadata_transaction_lock,
acquire_bucket_metadata_transaction_lock_for_incarnation, acquire_scanner_bucket_incarnation_fence, acquire_bucket_metadata_transaction_lock_for_incarnation, acquire_scanner_bucket_incarnation_fence,
capture_bucket_metadata_incarnation, delete, delete_if_incarnation, delete_under_transaction_lock, get, capture_bucket_metadata_incarnation, delete, delete_if_incarnation, delete_under_transaction_lock, get,
get_accelerate_config, get_bucket_policy, get_bucket_policy_raw, get_bucket_targets_config, get_config_from_disk, get_accelerate_config, get_bucket_policy, get_bucket_policy_raw, get_bucket_targets_config, get_config_from_disk,
get_cors_config, get_durability_config, get_global_bucket_metadata_sys, get_lifecycle_config, get_logging_config, get_cors_config, get_durability_config, get_global_bucket_metadata_sys, get_lifecycle_config, get_logging_config,
get_notification_config, get_object_lock_config, get_object_lock_config_state, get_on_demand_migration_config, get_notification_config, get_object_lock_config, get_object_lock_config_state, get_on_demand_migration_config,
get_on_demand_migration_config_in, get_public_access_block_config, get_quota_config, get_replication_config, get_public_access_block_config, get_quota_config, get_replication_config, get_request_payment_config, get_sse_config,
get_request_payment_config, get_sse_config, get_tagging_config, get_versioning_config, get_website_config, get_tagging_config, get_versioning_config, get_website_config, init_bucket_metadata_sys, list_bucket_targets,
init_bucket_metadata_sys, list_bucket_targets, reload_bucket_metadata, remove_bucket_metadata, set_bucket_metadata, reload_bucket_metadata, remove_bucket_metadata, set_bucket_metadata, update,
update, update_bucket_targets_under_transaction_lock, update_config_with, update_if_incarnation, update_bucket_targets_under_transaction_lock, update_config_with, update_if_incarnation, update_quota_if_incarnation,
update_quota_if_incarnation, update_under_transaction_lock, update_under_transaction_lock,
}; };
#[cfg(feature = "test-util")]
pub use crate::bucket::metadata_sys::{ConfigWriteLockProbe, test_support};
} }
pub mod migration { pub mod migration {
@@ -212,7 +249,7 @@ pub mod bucket {
pub mod remote_s3_client { pub mod remote_s3_client {
pub use crate::bucket::remote_s3_client::{ pub use crate::bucket::remote_s3_client::{
PathStyle, RemoteCredentials, RemoteS3ClientError, RemoteS3EndpointSpec, RemoteS3RetryPolicy, build_remote_s3_client, PathStyle, RemoteCredentials, RemoteS3ClientError, RemoteS3EndpointSpec, RemoteS3RetryPolicy, build_remote_s3_client,
build_remote_s3_config, validate_remote_endpoint, validate_target_ca_pem, validate_remote_endpoint,
}; };
} }
@@ -458,9 +495,9 @@ pub mod object {
ObjectInfo, ObjectLockConfigSnapshot, ObjectMutationHook, ObjectOptions, PutObjReader, QuotaAdmission, ObjectInfo, ObjectLockConfigSnapshot, ObjectMutationHook, ObjectOptions, PutObjReader, QuotaAdmission,
RangedDecompressReader, ReadEncryptionMaterial, ReadEncryptionMode, ReadEncryptionRequest, RangedDecompressReader, ReadEncryptionMaterial, ReadEncryptionMode, ReadEncryptionRequest,
SCANNER_PUBLICATION_LEASE_FENCE_METADATA_KEY, ScannerPublicationCommitScope, ScannerPublicationCommitStartError, SCANNER_PUBLICATION_LEASE_FENCE_METADATA_KEY, ScannerPublicationCommitScope, ScannerPublicationCommitStartError,
ScannerPublicationCommitState, StreamConsumer, WriteCompletion, get_object_body_cache_plaintext_len, ScannerPublicationCommitState, StreamConsumer, get_object_body_cache_plaintext_len, lookup_get_object_body_cache_hook,
lookup_get_object_body_cache_hook, register_get_object_body_cache_hook, register_object_mutation_hook, register_get_object_body_cache_hook, register_object_mutation_hook, unregister_get_object_body_cache_hook,
unregister_get_object_body_cache_hook, unregister_object_mutation_hook, unregister_object_mutation_hook,
}; };
pub use crate::store::{ pub use crate::store::{
PrepareSelectObjectSnapshotError, PreparedGetObjectReader, SelectObjectSnapshot, SelectObjectSnapshotReadError, PrepareSelectObjectSnapshotError, PreparedGetObjectReader, SelectObjectSnapshot, SelectObjectSnapshotReadError,
@@ -524,12 +561,6 @@ pub mod set_disk {
pub mod test_util { pub mod test_util {
pub use crate::bucket::quota::reservation::fail_next_quota_ledger_save_for_test; pub use crate::bucket::quota::reservation::fail_next_quota_ledger_save_for_test;
pub use crate::set_disk::{MultipartCommitBarrier, MultipartCommitPause, PutObjectCommitBarrier, PutObjectCommitPause}; pub use crate::set_disk::{MultipartCommitBarrier, MultipartCommitPause, PutObjectCommitBarrier, PutObjectCommitPause};
/// Keep a namespace commit pending until the returned owner is dropped.
#[must_use]
pub fn hold_namespace_commit(store: &crate::store::ECStore) -> impl Send + Sync {
store.ctx.begin_namespace_commit()
}
} }
} }
@@ -22,7 +22,7 @@ use super::{
bucket_lifecycle_ops::{ bucket_lifecycle_ops::{
ManualTransitionQueueSnapshot, ManualTransitionRunReport, decode_manual_transition_continuation_token, ManualTransitionQueueSnapshot, ManualTransitionRunReport, decode_manual_transition_continuation_token,
}, },
manual_transition_job, recovery_control, tier_delete_journal, transition_transaction, manual_transition_job, tier_delete_journal, transition_transaction,
}; };
use crate::error::{Error, Result}; use crate::error::{Error, Result};
use crate::services::tier::tier_probe_intent; use crate::services::tier::tier_probe_intent;
@@ -41,7 +41,6 @@ pub(crate) enum DurableIlmRecordKind {
ManualTransitionScope, ManualTransitionScope,
ManualTransitionTask, ManualTransitionTask,
ManualTransitionWorkerResult, ManualTransitionWorkerResult,
RecoveryControl,
} }
#[derive(Debug, Clone, Copy, PartialEq, Eq)] #[derive(Debug, Clone, Copy, PartialEq, Eq)]
@@ -106,14 +105,8 @@ pub(crate) const MANUAL_TRANSITION_WORKER_RESULT_NAMESPACE: DurableIlmNamespace
max_record_size: manual_transition_job::MAX_MANUAL_TRANSITION_WORKER_RESULT_RECORD_SIZE, max_record_size: manual_transition_job::MAX_MANUAL_TRANSITION_WORKER_RESULT_RECORD_SIZE,
kind: DurableIlmRecordKind::ManualTransitionWorkerResult, kind: DurableIlmRecordKind::ManualTransitionWorkerResult,
}; };
pub(crate) const RECOVERY_CONTROL_NAMESPACE: DurableIlmNamespace = DurableIlmNamespace {
name: "recovery-control",
prefix: recovery_control::ILM_RECOVERY_CONTROL_PREFIX,
max_record_size: recovery_control::MAX_ILM_RECOVERY_CONTROL_SIZE,
kind: DurableIlmRecordKind::RecoveryControl,
};
pub(crate) const DURABLE_ILM_NAMESPACES: [DurableIlmNamespace; 10] = [ pub(crate) const DURABLE_ILM_NAMESPACES: [DurableIlmNamespace; 9] = [
TIER_DELETE_JOURNAL_NAMESPACE, TIER_DELETE_JOURNAL_NAMESPACE,
TIER_DELETE_JOURNAL_V6_NAMESPACE, TIER_DELETE_JOURNAL_V6_NAMESPACE,
TIER_DELETE_DISPATCH_MANIFEST_NAMESPACE, TIER_DELETE_DISPATCH_MANIFEST_NAMESPACE,
@@ -123,7 +116,6 @@ pub(crate) const DURABLE_ILM_NAMESPACES: [DurableIlmNamespace; 10] = [
MANUAL_TRANSITION_SCOPE_NAMESPACE, MANUAL_TRANSITION_SCOPE_NAMESPACE,
MANUAL_TRANSITION_TASK_NAMESPACE, MANUAL_TRANSITION_TASK_NAMESPACE,
MANUAL_TRANSITION_WORKER_RESULT_NAMESPACE, MANUAL_TRANSITION_WORKER_RESULT_NAMESPACE,
RECOVERY_CONTROL_NAMESPACE,
]; ];
#[derive(Debug, Clone, PartialEq, Eq)] #[derive(Debug, Clone, PartialEq, Eq)]
@@ -249,18 +241,6 @@ pub(crate) enum DurableIlmRecordCheckpoint {
ManualTransitionWorkerResult { ManualTransitionWorkerResult {
content_sha256: String, content_sha256: String,
}, },
RecoveryControl {
content_sha256: String,
identity_sha256: String,
source_generation_sha256: String,
first_seen_at_unix_nanos: i64,
revision: u64,
classification: recovery_control::IlmRecoveryClassification,
attempt_count: u64,
consecutive_failure_count: u32,
#[serde(default, skip_serializing_if = "Option::is_none")]
owner_fence_sha256: Option<String>,
},
} }
impl DurableIlmRecordCheckpoint { impl DurableIlmRecordCheckpoint {
@@ -274,8 +254,7 @@ impl DurableIlmRecordCheckpoint {
| Self::ManualTransitionJob { content_sha256, .. } | Self::ManualTransitionJob { content_sha256, .. }
| Self::ManualTransitionScope { content_sha256, .. } | Self::ManualTransitionScope { content_sha256, .. }
| Self::ManualTransitionTask { content_sha256 } | Self::ManualTransitionTask { content_sha256 }
| Self::ManualTransitionWorkerResult { content_sha256 } | Self::ManualTransitionWorkerResult { content_sha256 } => content_sha256,
| Self::RecoveryControl { content_sha256, .. } => content_sha256,
} }
} }
@@ -549,51 +528,6 @@ impl DurableIlmRecordCheckpoint {
.. ..
}, },
) => previous_identity == next_identity && next_updated_at > previous_updated_at, ) => previous_identity == next_identity && next_updated_at > previous_updated_at,
(
Self::RecoveryControl {
identity_sha256: previous_identity,
source_generation_sha256: previous_generation,
first_seen_at_unix_nanos: previous_first_seen,
revision: previous_revision,
classification: previous_classification,
attempt_count: previous_attempts,
consecutive_failure_count: previous_failures,
owner_fence_sha256: previous_owner,
..
},
Self::RecoveryControl {
identity_sha256: next_identity,
source_generation_sha256: next_generation,
first_seen_at_unix_nanos: next_first_seen,
revision: next_revision,
classification: next_classification,
attempt_count: next_attempts,
consecutive_failure_count: next_failures,
owner_fence_sha256: next_owner,
..
},
) => {
let adjacent = previous_identity == next_identity
&& previous_first_seen == next_first_seen
&& previous_revision.checked_add(1) == Some(*next_revision);
let claim = next_owner.is_some()
&& *previous_classification == recovery_control::IlmRecoveryClassification::Retrying
&& *next_classification == recovery_control::IlmRecoveryClassification::Retrying
&& previous_attempts.checked_add(1) == Some(*next_attempts)
&& previous_failures == next_failures;
let source_refresh = previous_owner.is_some()
&& previous_owner == next_owner
&& *previous_classification == recovery_control::IlmRecoveryClassification::Retrying
&& *next_classification == recovery_control::IlmRecoveryClassification::Retrying
&& previous_attempts == next_attempts
&& previous_failures == next_failures
&& previous_generation != next_generation;
let completion = previous_owner.is_some()
&& next_owner.is_none()
&& previous_generation == next_generation
&& previous_attempts == next_attempts;
adjacent && (claim || source_refresh || completion)
}
_ => false, _ => false,
}; };
@@ -619,14 +553,6 @@ impl DurableIlmRecordCheckpoint {
{ {
return false; return false;
} }
if let Self::RecoveryControl { classification, .. } = terminal
&& !matches!(
classification,
recovery_control::IlmRecoveryClassification::Terminal | recovery_control::IlmRecoveryClassification::Abandoned
)
{
return false;
}
if self == terminal || self.validate_successor(terminal).is_ok() { if self == terminal || self.validate_successor(terminal).is_ok() {
return true; return true;
} }
@@ -726,32 +652,6 @@ impl DurableIlmRecordCheckpoint {
.is_some_and(|distance| tier_probe_state_reaches(*previous_state, *terminal_state, distance)) .is_some_and(|distance| tier_probe_state_reaches(*previous_state, *terminal_state, distance))
&& (!previous_remote_version_known || previous_remote_version == terminal_remote_version) && (!previous_remote_version_known || previous_remote_version == terminal_remote_version)
} }
(
Self::RecoveryControl {
identity_sha256: previous_identity,
source_generation_sha256: previous_generation,
first_seen_at_unix_nanos: previous_first_seen,
revision: previous_revision,
attempt_count: previous_attempts,
..
},
Self::RecoveryControl {
identity_sha256: terminal_identity,
source_generation_sha256: terminal_generation,
first_seen_at_unix_nanos: terminal_first_seen,
revision: terminal_revision,
attempt_count: terminal_attempts,
classification:
recovery_control::IlmRecoveryClassification::Terminal | recovery_control::IlmRecoveryClassification::Abandoned,
..
},
) => {
previous_identity == terminal_identity
&& (previous_generation == terminal_generation || terminal_attempts > previous_attempts)
&& previous_first_seen == terminal_first_seen
&& terminal_revision > previous_revision
&& terminal_attempts >= previous_attempts
}
_ => false, _ => false,
} }
} }
@@ -1319,35 +1219,6 @@ pub(crate) fn validate_durable_ilm_record(path: &str, data: &[u8]) -> Result<Val
}, },
) )
} }
DurableIlmRecordKind::RecoveryControl => {
let (protocol, control_id) = recovery_control::recovery_control_id_from_record_object_name(path)
.map_err(|err| Error::other(err.to_string()))?;
let control =
recovery_control::IlmRecoveryControl::decode(&control_id, data).map_err(|err| Error::other(err.to_string()))?;
let canonical = recovery_control::recovery_control_record_object_name(protocol, &control_id)
.map_err(|err| Error::other(err.to_string()))?;
if canonical != path || control.identity.protocol != protocol {
return Err(Error::other("ILM recovery control path is not canonical"));
}
let identity_sha256 = checkpoint_hash(&control.identity)?;
let source_generation_sha256 = checkpoint_hash(&control.observed_source_generation)?;
let owner_fence_sha256 = control.owner.as_ref().map(checkpoint_hash).transpose()?;
(
"control_id",
control_id,
DurableIlmRecordCheckpoint::RecoveryControl {
content_sha256,
identity_sha256,
source_generation_sha256,
first_seen_at_unix_nanos: control.first_seen_at_unix_nanos,
revision: control.revision,
classification: control.classification,
attempt_count: control.attempt_count,
consecutive_failure_count: control.consecutive_failure_count,
owner_fence_sha256,
},
)
}
DurableIlmRecordKind::ManualTransitionJob => { DurableIlmRecordKind::ManualTransitionJob => {
let job_id = manual_transition_job::manual_transition_job_id_from_record_object_name(path) let job_id = manual_transition_job::manual_transition_job_id_from_record_object_name(path)
.map_err(|err| Error::other(err.to_string()))?; .map_err(|err| Error::other(err.to_string()))?;
@@ -1541,94 +1412,6 @@ mod tests {
.checkpoint .checkpoint
} }
fn recovery_control_fixture() -> recovery_control::IlmRecoveryControl {
let source_path = "ilm/transition-transactions/records/12/34/1234567890abcdef1234567890abcdef.json";
let generation = recovery_control::IlmRecoverySourceGeneration::new(
transition_transaction::TRANSITION_TRANSACTION_SCHEMA,
"source-etag",
"a".repeat(64),
vec![recovery_control::IlmRecoverySourceCopy {
authority: "pool-0/set-0".to_string(),
canonical_path: source_path.to_string(),
etag: "source-etag".to_string(),
encoded_len: 128,
content_sha256: "a".repeat(64),
}],
)
.expect("source generation should build");
recovery_control::IlmRecoveryControl::new(
recovery_control::IlmRecoveryControlIdentity {
protocol: recovery_control::IlmRecoveryProtocol::TransitionTransaction,
canonical_source_path: source_path.to_string(),
stable_operation_identity: "12345678-90ab-cdef-1234-567890abcdef".to_string(),
record_class: "transition_transaction_v1".to_string(),
},
generation,
recovery_control::IlmRecoveryClassification::Retrying,
1_000_000_000,
recovery_control::IlmRecoveryErrorCode::None,
)
.expect("recovery control should build")
}
fn recovery_control_checkpoint(control: &recovery_control::IlmRecoveryControl) -> DurableIlmRecordCheckpoint {
let control_id = control.identity.source_operation_digest().expect("control id should derive");
let path = recovery_control::recovery_control_record_object_name(control.identity.protocol, &control_id)
.expect("control path should build");
let encoded = control.encode().expect("control should encode");
let namespace = classify_durable_ilm_record(&path)
.expect("recovery control namespace should classify")
.expect("recovery control should be durable");
assert_eq!(namespace, &RECOVERY_CONTROL_NAMESPACE);
validate_durable_ilm_record(&path, &encoded)
.expect("recovery control should validate")
.checkpoint
}
#[test]
fn recovery_control_checkpoint_tracks_claim_retry_and_terminal_generations() {
let initial_control = recovery_control_fixture();
let initial = recovery_control_checkpoint(&initial_control);
let mut claimed_control = initial_control;
let mut advanced_generation = claimed_control.observed_source_generation.clone();
advanced_generation.source_schema = "rustfs-transition-transaction-v2".to_string();
claimed_control
.claim_for_source_generation("node-a", Uuid::new_v4(), 2_000_000_000, 300_000_000_000, advanced_generation)
.expect("control should claim");
let claimed = recovery_control_checkpoint(&claimed_control);
initial.validate_successor(&claimed).expect("claim should advance receipt");
let mut retry_control = claimed_control;
retry_control
.record_retryable_failure(3_000_000_000, recovery_control::IlmRecoveryErrorCode::BackendTimeout)
.expect("retry should persist");
let retry = recovery_control_checkpoint(&retry_control);
claimed.validate_successor(&retry).expect("retry should advance receipt");
let ready_at = retry_control
.next_attempt_at_unix_nanos
.expect("retry deadline should persist");
let mut terminal_control = retry_control;
terminal_control
.claim("node-b", Uuid::new_v4(), ready_at, 300_000_000_000)
.expect("retry should claim");
let reclaimed = recovery_control_checkpoint(&terminal_control);
retry.validate_successor(&reclaimed).expect("reclaim should advance receipt");
terminal_control
.finish_attempt(
recovery_control::IlmRecoveryClassification::Terminal,
recovery_control::IlmRecoveryErrorCode::None,
)
.expect("control should terminate");
let terminal = recovery_control_checkpoint(&terminal_control);
reclaimed
.validate_successor(&terminal)
.expect("terminal state should advance receipt");
assert!(initial.is_predecessor_of_terminal(&terminal));
assert!(!initial.is_predecessor_of_terminal(&retry));
}
#[test] #[test]
fn tier_probe_intent_checkpoint_tracks_exact_monotonic_generations() { fn tier_probe_intent_checkpoint_tracks_exact_monotonic_generations() {
let initial_intent = tier_probe_intent_fixture(); let initial_intent = tier_probe_intent_fixture();
@@ -24,7 +24,6 @@ pub(crate) use metadata_boundary::{LifecycleExpiryConfigs, get_expiry_configs, g
mod object_handlers_common; mod object_handlers_common;
mod object_lock_boundary; mod object_lock_boundary;
pub use self::core as lifecycle; pub use self::core as lifecycle;
pub mod recovery_control;
mod replication_sink; mod replication_sink;
pub mod rule; pub mod rule;
mod runtime_boundary; mod runtime_boundary;
File diff suppressed because it is too large Load Diff
@@ -23,11 +23,6 @@ use uuid::Uuid;
use crate::bucket::lifecycle::config_boundary; use crate::bucket::lifecycle::config_boundary;
use crate::bucket::lifecycle::durable_namespace::TRANSITION_TRANSACTION_NAMESPACE; use crate::bucket::lifecycle::durable_namespace::TRANSITION_TRANSACTION_NAMESPACE;
use crate::bucket::lifecycle::lifecycle::TRANSITION_COMPLETE; use crate::bucket::lifecycle::lifecycle::TRANSITION_COMPLETE;
use crate::bucket::lifecycle::recovery_control::{
IlmRecoveryClassification, IlmRecoveryControl, IlmRecoveryControlIdentity, IlmRecoveryErrorCode, IlmRecoveryProtocol,
ObservedIlmRecoveryControl, load_recovery_control, observe_recovery_source, recovery_control_record_object_name,
save_recovery_control_if_absent, save_recovery_control_if_current,
};
use crate::bucket::lifecycle::tier_sweeper::{ use crate::bucket::lifecycle::tier_sweeper::{
delete_confirmed_transition_candidate_exact_with_lease_idempotent, delete_confirmed_transition_candidate_exact_with_lease_idempotent,
delete_object_from_remote_tier_idempotent_with_manager_and_identity, delete_object_from_remote_tier_idempotent_with_manager_and_identity,
@@ -49,7 +44,6 @@ const EVENT_LIFECYCLE_TRANSITION_TRANSACTION_RECOVERY: &str = "lifecycle_transit
pub const DEFAULT_TRANSITION_TRANSACTION_RECOVERY_LIMIT: usize = 1_000; pub const DEFAULT_TRANSITION_TRANSACTION_RECOVERY_LIMIT: usize = 1_000;
const TRANSITION_TRANSACTION_RECOVERY_INTERVAL: Duration = Duration::from_secs(60); const TRANSITION_TRANSACTION_RECOVERY_INTERVAL: Duration = Duration::from_secs(60);
const TRANSITION_TRANSACTION_RECOVERY_TIMEOUT: Duration = Duration::from_secs(300); const TRANSITION_TRANSACTION_RECOVERY_TIMEOUT: Duration = Duration::from_secs(300);
const TRANSITION_RECOVERY_CONTROL_LEASE_NANOS: i64 = 15 * 60 * 1_000_000_000;
pub const TRANSITION_TRANSACTION_SCHEMA: &str = "rustfs-transition-transaction-v1"; pub const TRANSITION_TRANSACTION_SCHEMA: &str = "rustfs-transition-transaction-v1";
pub const TRANSITION_TRANSACTION_PREFIX: &str = "ilm/transition-transactions"; pub const TRANSITION_TRANSACTION_PREFIX: &str = "ilm/transition-transactions";
pub const TRANSITION_TRANSACTION_RECORD_PREFIX: &str = TRANSITION_TRANSACTION_NAMESPACE.prefix; pub const TRANSITION_TRANSACTION_RECORD_PREFIX: &str = TRANSITION_TRANSACTION_NAMESPACE.prefix;
@@ -743,8 +737,6 @@ pub enum TransitionTransactionRecoveryOutcome {
RemoteCandidateDeleted, RemoteCandidateDeleted,
RecordDeleted, RecordDeleted,
Retained, Retained,
RetainedAmbiguous(IlmRecoveryErrorCode),
OperatorRequired(IlmRecoveryErrorCode),
} }
#[cfg(test)] #[cfg(test)]
@@ -825,80 +817,6 @@ async fn pause_before_transition_recovery_claim(transaction_id: Uuid) {
} }
} }
#[cfg(test)]
#[derive(Default)]
struct TransitionRecoveryTerminalBarrierState {
transaction_id: Uuid,
arrived: tokio::sync::Notify,
release: tokio::sync::Notify,
}
#[cfg(test)]
pub(crate) struct TransitionRecoveryTerminalBarrier {
state: Arc<TransitionRecoveryTerminalBarrierState>,
}
#[cfg(test)]
static TRANSITION_RECOVERY_TERMINAL_BARRIER: std::sync::OnceLock<
std::sync::Mutex<Option<Arc<TransitionRecoveryTerminalBarrierState>>>,
> = std::sync::OnceLock::new();
#[cfg(test)]
impl TransitionRecoveryTerminalBarrier {
pub(crate) fn install(transaction_id: Uuid) -> Self {
let state = Arc::new(TransitionRecoveryTerminalBarrierState {
transaction_id,
..Default::default()
});
let mut slot = TRANSITION_RECOVERY_TERMINAL_BARRIER
.get_or_init(|| std::sync::Mutex::new(None))
.lock()
.expect("transition recovery terminal barrier mutex should not poison");
assert!(
slot.is_none(),
"transition recovery terminal barrier must be installed by one test at a time"
);
*slot = Some(Arc::clone(&state));
drop(slot);
Self { state }
}
pub(crate) async fn wait_until_paused(&self) {
tokio::time::timeout(Duration::from_secs(30), self.state.arrived.notified())
.await
.expect("transition recovery should persist terminal control before source cleanup");
}
}
#[cfg(test)]
impl Drop for TransitionRecoveryTerminalBarrier {
fn drop(&mut self) {
self.state.release.notify_one();
let mut slot = TRANSITION_RECOVERY_TERMINAL_BARRIER
.get_or_init(|| std::sync::Mutex::new(None))
.lock()
.expect("transition recovery terminal barrier mutex should not poison");
if slot.as_ref().is_some_and(|state| Arc::ptr_eq(state, &self.state)) {
*slot = None;
}
}
}
#[cfg(test)]
async fn pause_after_transition_recovery_terminal(transaction_id: Uuid) {
let barrier = TRANSITION_RECOVERY_TERMINAL_BARRIER
.get_or_init(|| std::sync::Mutex::new(None))
.lock()
.expect("transition recovery terminal barrier mutex should not poison")
.as_ref()
.filter(|barrier| barrier.transaction_id == transaction_id)
.cloned();
if let Some(barrier) = barrier {
barrier.arrived.notify_one();
barrier.release.notified().await;
}
}
#[derive(Debug, Clone, PartialEq, Eq, Serialize)] #[derive(Debug, Clone, PartialEq, Eq, Serialize)]
#[serde(rename_all = "snake_case")] #[serde(rename_all = "snake_case")]
pub enum TransitionOperatorProbe { pub enum TransitionOperatorProbe {
@@ -1102,35 +1020,17 @@ fn transition_transaction_id_from_record_object_name(object: &str) -> Result<Uui
let suffix = object let suffix = object
.strip_prefix(&prefix) .strip_prefix(&prefix)
.ok_or(TransitionTransactionError::Corrupt("transaction record path has wrong prefix"))?; .ok_or(TransitionTransactionError::Corrupt("transaction record path has wrong prefix"))?;
let mut parts = suffix.split('/'); let file_name = suffix
let shard_a = parts .rsplit('/')
.next() .next()
.ok_or(TransitionTransactionError::Corrupt("transaction record path is incomplete"))?; .ok_or(TransitionTransactionError::Corrupt("transaction record path is incomplete"))?;
let shard_b = parts
.next()
.ok_or(TransitionTransactionError::Corrupt("transaction record path is incomplete"))?;
let file_name = parts
.next()
.ok_or(TransitionTransactionError::Corrupt("transaction record path is incomplete"))?;
if parts.next().is_some() {
return Err(TransitionTransactionError::Corrupt("transaction record path is not canonical"));
}
let transaction_key = file_name let transaction_key = file_name
.strip_suffix(".json") .strip_suffix(".json")
.ok_or(TransitionTransactionError::Corrupt("transaction record path has wrong suffix"))?; .ok_or(TransitionTransactionError::Corrupt("transaction record path has wrong suffix"))?;
if transaction_key.len() != 32 if transaction_key.len() != 32 || !transaction_key.bytes().all(|byte| byte.is_ascii_hexdigit()) {
|| !transaction_key
.bytes()
.all(|byte| byte.is_ascii_hexdigit() && !byte.is_ascii_uppercase())
|| shard_a != &transaction_key[..2]
|| shard_b != &transaction_key[2..4]
{
return Err(TransitionTransactionError::Corrupt("transaction record path has invalid transaction id")); return Err(TransitionTransactionError::Corrupt("transaction record path has invalid transaction id"));
} }
Uuid::parse_str(transaction_key) Uuid::parse_str(transaction_key).map_err(|_| TransitionTransactionError::Corrupt("transaction record path has invalid uuid"))
.ok()
.filter(|transaction_id| !transaction_id.is_nil())
.ok_or(TransitionTransactionError::Corrupt("transaction record path has invalid uuid"))
} }
pub async fn process_transition_transaction_record( pub async fn process_transition_transaction_record(
@@ -1155,27 +1055,6 @@ async fn process_transition_transaction_record_at(
) -> EcstoreResult<TransitionTransactionRecoveryOutcome> { ) -> EcstoreResult<TransitionTransactionRecoveryOutcome> {
let record_name = let record_name =
transition_transaction_record_object_name(observed.transaction_id).map_err(transition_transaction_store_error)?; transition_transaction_record_object_name(observed.transaction_id).map_err(transition_transaction_store_error)?;
let now_unix_nanos =
i64::try_from(now_unix_nanos).map_err(|_| Error::other("transition transaction recovery timestamp does not fit i64"))?;
let recovery_control_identity = transition_recovery_control_identity(observed, &record_name);
let recovery_control_id = recovery_control_identity
.source_operation_digest()
.map_err(|err| Error::other(err.to_string()))?;
let control_record_name =
recovery_control_record_object_name(IlmRecoveryProtocol::TransitionTransaction, &recovery_control_id)
.map_err(|err| Error::other(err.to_string()))?;
let control_lock = if transition_state_needs_recovery_control(observed, now_unix_nanos) {
Some(
api.new_ns_lock(RUSTFS_META_BUCKET, &format!("{control_record_name}.recovery-lock"))
.await?,
)
} else {
None
};
let _control_guard = match &control_lock {
Some(lock) => Some(lock.get_write_lock(crate::set_disk::get_lock_acquire_timeout()).await?),
None => None,
};
// The synthetic key avoids nesting the recovery lock with the config // The synthetic key avoids nesting the recovery lock with the config
// object's own I/O lock. Holding it across the bounded source proof and // object's own I/O lock. Holding it across the bounded source proof and
// remote DELETE elects one destructive recovery worker across nodes. // remote DELETE elects one destructive recovery worker across nodes.
@@ -1194,400 +1073,55 @@ async fn process_transition_transaction_record_at(
return Ok(TransitionTransactionRecoveryOutcome::Retained); return Ok(TransitionTransactionRecoveryOutcome::Retained);
} }
let mut recovery_control = if transition_state_needs_recovery_control(&current, now_unix_nanos) { match current.state {
if cleanup_terminal_transition_recovery_control(
api.clone(),
&current,
&record_name,
&recovery_control_identity,
&recovery_control_id,
)
.await?
{
return Ok(TransitionTransactionRecoveryOutcome::RecordDeleted);
}
match claim_transition_recovery_control(
api.clone(),
&current,
&record_name,
recovery_control_identity,
&recovery_control_id,
now_unix_nanos,
)
.await?
{
Some(control) => Some(control),
None => return Ok(TransitionTransactionRecoveryOutcome::Retained),
}
} else {
None
};
let recovery = match current.state {
TransitionTransactionState::Uploaded => { TransitionTransactionState::Uploaded => {
if transition_transaction_ownership_is_active(&current, i128::from(now_unix_nanos)) { if transition_transaction_ownership_is_active(&current, now_unix_nanos) {
Ok(TransitionTransactionRecoveryOutcome::Retained) return Ok(TransitionTransactionRecoveryOutcome::Retained);
} else { }
let mut cleanup = current.clone(); let mut cleanup = current.clone();
cleanup cleanup
.mark_cleanup_pending( .mark_cleanup_pending(
current.fence(), current.fence(),
TransitionCleanupProof { TransitionCleanupProof {
transaction_id: current.transaction_id, transaction_id: current.transaction_id,
write_id: current.write_id, write_id: current.write_id,
remote_object: current.remote_object.clone(), remote_object: current.remote_object.clone(),
remote_version: current.remote_version.clone(), remote_version: current.remote_version.clone(),
backend_fingerprint: current.backend_fingerprint, backend_fingerprint: current.backend_fingerprint,
decision: TransitionCleanupDecision::UploadAbortedBeforeLocalCommit, decision: TransitionCleanupDecision::UploadAbortedBeforeLocalCommit,
}, },
) )
.map_err(transition_transaction_store_error)?; .map_err(transition_transaction_store_error)?;
#[cfg(test)] #[cfg(test)]
pause_before_transition_recovery_claim(current.transaction_id).await; pause_before_transition_recovery_claim(current.transaction_id).await;
match save_transition_transaction_record_if_current(api.clone(), &current, &cleanup).await { match save_transition_transaction_record_if_current(api.clone(), &current, &cleanup).await {
Ok(()) => recover_cleanup_pending(api.clone(), &cleanup).await, Ok(()) => recover_cleanup_pending(api, &cleanup).await,
Err(Error::PreconditionFailed) | Err(Error::ConfigNotFound) => { Err(Error::PreconditionFailed) | Err(Error::ConfigNotFound) => Ok(TransitionTransactionRecoveryOutcome::Retained),
Ok(TransitionTransactionRecoveryOutcome::Retained) Err(err) => Err(err),
}
Err(err) => Err(err),
}
} }
} }
TransitionTransactionState::CleanupPending => recover_cleanup_pending(api.clone(), &current).await, TransitionTransactionState::CleanupPending => recover_cleanup_pending(api, &current).await,
TransitionTransactionState::LocalCommitStarted => match local_commit_matches_transaction(api.clone(), &current).await { TransitionTransactionState::LocalCommitStarted => match local_commit_matches_transaction(api.clone(), &current).await {
Ok(true) => Ok(TransitionTransactionRecoveryOutcome::RecordDeleted), Ok(true) => {
Ok(false) => Ok(TransitionTransactionRecoveryOutcome::OperatorRequired( delete_transition_transaction_record(api, &current).await?;
IlmRecoveryErrorCode::LocalCommitAmbiguous, Ok(TransitionTransactionRecoveryOutcome::RecordDeleted)
)), }
Err(err) if transition_source_is_missing(&err) => Ok(TransitionTransactionRecoveryOutcome::OperatorRequired( Ok(false) => Ok(TransitionTransactionRecoveryOutcome::Retained),
IlmRecoveryErrorCode::LocalCommitAmbiguous, Err(err) if transition_source_is_missing(&err) => Ok(TransitionTransactionRecoveryOutcome::Retained),
)),
Err(err) => Err(err), Err(err) => Err(err),
}, },
TransitionTransactionState::AbortedNoRemote | TransitionTransactionState::Committed => { TransitionTransactionState::AbortedNoRemote | TransitionTransactionState::Committed => {
delete_transition_transaction_record(api, &current).await?;
Ok(TransitionTransactionRecoveryOutcome::RecordDeleted) Ok(TransitionTransactionRecoveryOutcome::RecordDeleted)
} }
TransitionTransactionState::UploadOutcomeUnknown => { TransitionTransactionState::UploadOutcomeUnknown => {
if transition_transaction_ownership_is_active(&current, i128::from(now_unix_nanos)) { if transition_transaction_ownership_is_active(&current, now_unix_nanos) {
Ok(TransitionTransactionRecoveryOutcome::Retained) Ok(TransitionTransactionRecoveryOutcome::Retained)
} else { } else {
recover_unknown_upload_outcome(api.clone(), &current).await recover_unknown_upload_outcome(api, &current).await
} }
} }
TransitionTransactionState::UploadStarted => { TransitionTransactionState::UploadStarted => Ok(TransitionTransactionRecoveryOutcome::Retained),
if transition_transaction_ownership_is_active(&current, i128::from(now_unix_nanos)) {
Ok(TransitionTransactionRecoveryOutcome::Retained)
} else {
Ok(TransitionTransactionRecoveryOutcome::RetainedAmbiguous(
IlmRecoveryErrorCode::RemoteVersionUnknown,
))
}
}
};
if let Some(mut control) = recovery_control.take() {
let source_to_delete = if matches!(
recovery,
Ok(TransitionTransactionRecoveryOutcome::RemoteCandidateDeleted
| TransitionTransactionRecoveryOutcome::RecordDeleted)
) {
let refreshed =
refresh_transition_recovery_control_source(api.clone(), control, &record_name, current.transaction_id).await?;
control = refreshed.0;
refreshed.1
} else {
None
};
persist_transition_recovery_result(api.clone(), control, &recovery, now_unix_nanos).await?;
if let Some(source) = source_to_delete {
#[cfg(test)]
pause_after_transition_recovery_terminal(source.transaction_id).await;
delete_transition_transaction_record(api, &source).await?;
}
} else if matches!(
recovery,
Ok(TransitionTransactionRecoveryOutcome::RemoteCandidateDeleted | TransitionTransactionRecoveryOutcome::RecordDeleted)
) {
delete_transition_transaction_record(api, &current).await?;
}
recovery
}
fn transition_recovery_control_identity(transaction: &TransitionTransaction, record_name: &str) -> IlmRecoveryControlIdentity {
IlmRecoveryControlIdentity {
protocol: IlmRecoveryProtocol::TransitionTransaction,
canonical_source_path: record_name.to_string(),
stable_operation_identity: transaction.transaction_id.to_string(),
record_class: "transition_transaction_v1".to_string(),
}
}
#[cfg(test)]
pub(crate) fn transition_recovery_control_id(transaction: &TransitionTransaction) -> Result<String> {
let record_name = transition_transaction_record_object_name(transaction.transaction_id)?;
transition_recovery_control_identity(transaction, &record_name)
.source_operation_digest()
.map_err(|_| TransitionTransactionError::Corrupt("transition recovery control identity is invalid"))
}
fn transition_state_needs_recovery_control(transaction: &TransitionTransaction, now_unix_nanos: i64) -> bool {
now_unix_nanos >= transaction.not_after_unix_nanos
&& !matches!(
transaction.state,
TransitionTransactionState::AbortedNoRemote | TransitionTransactionState::Committed
)
}
async fn cleanup_terminal_transition_recovery_control(
api: Arc<ECStore>,
transaction: &TransitionTransaction,
record_name: &str,
identity: &IlmRecoveryControlIdentity,
control_id: &str,
) -> EcstoreResult<bool> {
let observed = match load_recovery_control(api.clone(), IlmRecoveryProtocol::TransitionTransaction, control_id).await {
Ok(observed) => observed,
Err(Error::ConfigNotFound) => return Ok(false),
Err(err) => return Err(err),
};
if observed.control.classification != IlmRecoveryClassification::Terminal {
return Ok(false);
}
let source = observe_recovery_source(api.clone(), record_name, TRANSITION_TRANSACTION_SCHEMA).await?;
let exact_source = source.is_consistent()
&& source.generation == observed.control.observed_source_generation
&& source.canonical_data.as_deref().is_some_and(|data| {
TransitionTransaction::decode(transaction.transaction_id, data).is_ok_and(|decoded| decoded == *transaction)
});
if observed.control.identity != *identity || !exact_source {
return Ok(false);
}
delete_transition_transaction_record(api, transaction).await?;
Ok(true)
}
async fn claim_transition_recovery_control(
api: Arc<ECStore>,
transaction: &TransitionTransaction,
record_name: &str,
identity: IlmRecoveryControlIdentity,
control_id: &str,
now_unix_nanos: i64,
) -> EcstoreResult<Option<ObservedIlmRecoveryControl>> {
let existing = match load_recovery_control(api.clone(), IlmRecoveryProtocol::TransitionTransaction, control_id).await {
Ok(control) => Some(control),
Err(Error::ConfigNotFound) => None,
Err(err) => return Err(err),
};
if let Some(observed) = existing.as_ref() {
if observed.control.identity != identity {
return Ok(None);
}
if observed
.control
.owner
.as_ref()
.is_some_and(|owner| owner.lease_expires_at_unix_nanos <= now_unix_nanos)
{
let mut expired = observed.control.clone();
expired
.record_expired_attempt(now_unix_nanos)
.map_err(|err| Error::other(err.to_string()))?;
save_recovery_control_if_current(api, observed, &expired).await?;
return Ok(None);
}
if !observed.control.should_attempt_at(now_unix_nanos) {
return Ok(None);
}
}
let source = match observe_recovery_source(api.clone(), record_name, TRANSITION_TRANSACTION_SCHEMA).await {
Ok(source) => source,
Err(err) => {
if let Some(observed) = existing {
persist_transition_recovery_source_failure(api, observed, now_unix_nanos).await?;
return Ok(None);
}
return Err(err);
}
};
let source_matches = source.is_consistent()
&& source.canonical_data.as_deref().is_some_and(|data| {
TransitionTransaction::decode(transaction.transaction_id, data).is_ok_and(|observed| observed == *transaction)
});
let source_error = if source_matches {
IlmRecoveryErrorCode::None
} else if source.canonical_data.is_some() {
IlmRecoveryErrorCode::SourceGenerationChanged
} else {
IlmRecoveryErrorCode::SourceDivergent
};
let mut observed = match existing {
Some(control) => control,
None => {
let candidate = IlmRecoveryControl::new(
identity.clone(),
source.generation.clone(),
if source_matches {
IlmRecoveryClassification::Retrying
} else {
IlmRecoveryClassification::Corrupt
},
now_unix_nanos,
source_error,
)
.map_err(|err| Error::other(err.to_string()))?;
match save_recovery_control_if_absent(api.clone(), &candidate).await {
Ok(()) | Err(Error::PreconditionFailed) => {}
Err(err) => return Err(err),
}
load_recovery_control(api.clone(), IlmRecoveryProtocol::TransitionTransaction, control_id).await?
}
};
if observed.control.identity != identity || !observed.control.should_attempt_at(now_unix_nanos) {
return Ok(None);
}
let mut claimed = observed.control.clone();
claimed
.claim_for_source_generation(
api.id.to_string(),
Uuid::new_v4(),
now_unix_nanos,
TRANSITION_RECOVERY_CONTROL_LEASE_NANOS,
source.generation,
)
.map_err(|err| Error::other(err.to_string()))?;
save_recovery_control_if_current(api.clone(), &observed, &claimed).await?;
observed = load_recovery_control(api.clone(), IlmRecoveryProtocol::TransitionTransaction, control_id).await?;
if observed.control != claimed {
return Err(Error::PreconditionFailed);
}
if !source_matches {
let mut corrupt = observed.control.clone();
corrupt
.finish_attempt(IlmRecoveryClassification::Corrupt, source_error)
.map_err(|err| Error::other(err.to_string()))?;
save_recovery_control_if_current(api, &observed, &corrupt).await?;
return Ok(None);
}
Ok(Some(observed))
}
async fn persist_transition_recovery_source_failure(
api: Arc<ECStore>,
observed: ObservedIlmRecoveryControl,
now_unix_nanos: i64,
) -> EcstoreResult<()> {
let mut claimed = observed.control.clone();
claimed
.claim(
api.id.to_string(),
Uuid::new_v4(),
now_unix_nanos,
TRANSITION_RECOVERY_CONTROL_LEASE_NANOS,
)
.map_err(|err| Error::other(err.to_string()))?;
save_recovery_control_if_current(api.clone(), &observed, &claimed).await?;
let claimed = load_recovery_control(
api.clone(),
IlmRecoveryProtocol::TransitionTransaction,
&claimed
.identity
.source_operation_digest()
.map_err(|err| Error::other(err.to_string()))?,
)
.await?;
let mut failed = claimed.control.clone();
failed
.record_retryable_failure(now_unix_nanos, IlmRecoveryErrorCode::SourceUnavailable)
.map_err(|err| Error::other(err.to_string()))?;
save_recovery_control_if_current(api, &claimed, &failed).await
}
async fn refresh_transition_recovery_control_source(
api: Arc<ECStore>,
mut observed: ObservedIlmRecoveryControl,
record_name: &str,
transaction_id: Uuid,
) -> EcstoreResult<(ObservedIlmRecoveryControl, Option<TransitionTransaction>)> {
let transaction = match load_transition_transaction_record(api.clone(), transaction_id).await {
Ok(transaction) => transaction,
Err(Error::ConfigNotFound) => return Ok((observed, None)),
Err(err) => return Err(err),
};
let source = observe_recovery_source(api.clone(), record_name, TRANSITION_TRANSACTION_SCHEMA).await?;
let exact_source = source.is_consistent()
&& source
.canonical_data
.as_deref()
.is_some_and(|data| TransitionTransaction::decode(transaction_id, data).is_ok_and(|decoded| decoded == transaction));
if !exact_source {
return Err(Error::PreconditionFailed);
}
if observed.control.observed_source_generation != source.generation {
let mut refreshed = observed.control.clone();
refreshed
.refresh_owned_source_generation(source.generation)
.map_err(|err| Error::other(err.to_string()))?;
save_recovery_control_if_current(api.clone(), &observed, &refreshed).await?;
observed = load_recovery_control(
api,
IlmRecoveryProtocol::TransitionTransaction,
&refreshed
.identity
.source_operation_digest()
.map_err(|err| Error::other(err.to_string()))?,
)
.await?;
if observed.control != refreshed {
return Err(Error::PreconditionFailed);
}
}
Ok((observed, Some(transaction)))
}
async fn persist_transition_recovery_result(
api: Arc<ECStore>,
observed: ObservedIlmRecoveryControl,
recovery: &EcstoreResult<TransitionTransactionRecoveryOutcome>,
now_unix_nanos: i64,
) -> EcstoreResult<()> {
let mut next = observed.control.clone();
match recovery {
Ok(
TransitionTransactionRecoveryOutcome::RemoteCandidateDeleted | TransitionTransactionRecoveryOutcome::RecordDeleted,
) => next
.finish_attempt(IlmRecoveryClassification::Terminal, IlmRecoveryErrorCode::None)
.map_err(|err| Error::other(err.to_string()))?,
Ok(TransitionTransactionRecoveryOutcome::Retained) => next
.record_retryable_failure(now_unix_nanos, IlmRecoveryErrorCode::SourceGenerationChanged)
.map_err(|err| Error::other(err.to_string()))?,
Ok(TransitionTransactionRecoveryOutcome::RetainedAmbiguous(code)) => next
.finish_attempt(IlmRecoveryClassification::RetainedAmbiguous, *code)
.map_err(|err| Error::other(err.to_string()))?,
Ok(TransitionTransactionRecoveryOutcome::OperatorRequired(code)) => next
.finish_attempt(IlmRecoveryClassification::OperatorRequired, *code)
.map_err(|err| Error::other(err.to_string()))?,
Err(err) => next
.record_retryable_failure(now_unix_nanos, transition_recovery_error_code(err))
.map_err(|err| Error::other(err.to_string()))?,
}
save_recovery_control_if_current(api, &observed, &next).await
}
fn transition_recovery_error_code(err: &Error) -> IlmRecoveryErrorCode {
match err {
Error::PreconditionFailed => IlmRecoveryErrorCode::CasConflict,
Error::ConfigNotFound
| Error::FileNotFound
| Error::FileVersionNotFound
| Error::ObjectNotFound(_, _)
| Error::VersionNotFound(_, _, _)
| Error::BucketNotFound(_) => IlmRecoveryErrorCode::SourceUnavailable,
Error::SlowDown => IlmRecoveryErrorCode::BackendThrottled,
_ => IlmRecoveryErrorCode::Unknown,
} }
} }
@@ -1600,7 +1134,10 @@ async fn recover_cleanup_pending(
transaction: &TransitionTransaction, transaction: &TransitionTransaction,
) -> EcstoreResult<TransitionTransactionRecoveryOutcome> { ) -> EcstoreResult<TransitionTransactionRecoveryOutcome> {
match local_commit_matches_transaction(api.clone(), transaction).await { match local_commit_matches_transaction(api.clone(), transaction).await {
Ok(true) => Ok(TransitionTransactionRecoveryOutcome::RecordDeleted), Ok(true) => {
delete_transition_transaction_record(api, transaction).await?;
Ok(TransitionTransactionRecoveryOutcome::RecordDeleted)
}
Ok(false) => delete_unreferenced_transition_candidate(api, transaction).await, Ok(false) => delete_unreferenced_transition_candidate(api, transaction).await,
Err(err) if transition_source_is_missing(&err) => delete_unreferenced_transition_candidate(api, transaction).await, Err(err) if transition_source_is_missing(&err) => delete_unreferenced_transition_candidate(api, transaction).await,
Err(err) => Err(err), Err(err) => Err(err),
@@ -1620,6 +1157,7 @@ async fn delete_unreferenced_transition_candidate(
return Ok(TransitionTransactionRecoveryOutcome::Retained); return Ok(TransitionTransactionRecoveryOutcome::Retained);
} }
delete_transition_remote_candidate(api.clone(), &current).await?; delete_transition_remote_candidate(api.clone(), &current).await?;
delete_transition_transaction_record(api, &current).await?;
Ok(TransitionTransactionRecoveryOutcome::RemoteCandidateDeleted) Ok(TransitionTransactionRecoveryOutcome::RemoteCandidateDeleted)
} }
@@ -1640,26 +1178,24 @@ async fn recover_unknown_upload_outcome(
.await .await
.map_err(Error::other)? .map_err(Error::other)?
{ {
TransitionCandidateProbe::Missing => Ok(TransitionTransactionRecoveryOutcome::RecordDeleted), TransitionCandidateProbe::Missing => {
delete_transition_transaction_record(api, transaction).await?;
Ok(TransitionTransactionRecoveryOutcome::RecordDeleted)
}
TransitionCandidateProbe::UnversionedPresent => { TransitionCandidateProbe::UnversionedPresent => {
cleanup_recovered_unknown_upload_candidate(api, transaction, TransitionRemoteVersion::unversioned()).await cleanup_recovered_unknown_upload_candidate(api, transaction, TransitionRemoteVersion::unversioned()).await
} }
TransitionCandidateProbe::VersionedPresent(version_id) TransitionCandidateProbe::VersionedPresent(version_id)
if Uuid::parse_str(&version_id).is_ok_and(|version_id| version_id.is_nil()) => if Uuid::parse_str(&version_id).is_ok_and(|version_id| version_id.is_nil()) =>
{ {
Ok(TransitionTransactionRecoveryOutcome::RetainedAmbiguous( Ok(TransitionTransactionRecoveryOutcome::Retained)
IlmRecoveryErrorCode::RemoteVersionUnknown,
))
} }
TransitionCandidateProbe::VersionedPresent(version_id) => { TransitionCandidateProbe::VersionedPresent(version_id) => {
cleanup_recovered_unknown_upload_candidate(api, transaction, TransitionRemoteVersion::versioned(version_id)).await cleanup_recovered_unknown_upload_candidate(api, transaction, TransitionRemoteVersion::versioned(version_id)).await
} }
TransitionCandidateProbe::Ambiguous => Ok(TransitionTransactionRecoveryOutcome::RetainedAmbiguous( TransitionCandidateProbe::Ambiguous | TransitionCandidateProbe::Unsupported => {
IlmRecoveryErrorCode::RemoteProbeAmbiguous, Ok(TransitionTransactionRecoveryOutcome::Retained)
)), }
TransitionCandidateProbe::Unsupported => Ok(TransitionTransactionRecoveryOutcome::RetainedAmbiguous(
IlmRecoveryErrorCode::RemoteProbeUnsupported,
)),
} }
} }
@@ -1787,11 +1323,6 @@ async fn recover_transition_transaction_records_with_now(
false, false,
) )
.await?; .await?;
if list.is_truncated && list.next_continuation_token.is_none() {
return Err(Error::other(
"transition transaction recovery returned a truncated page without a continuation marker",
));
}
let mut stats = TransitionTransactionRecoveryStats { let mut stats = TransitionTransactionRecoveryStats {
scanned: 0, scanned: 0,
@@ -1850,11 +1381,7 @@ async fn recover_transition_transaction_records_with_now(
) => { ) => {
stats.recovered += 1; stats.recovered += 1;
} }
Ok( Ok(TransitionTransactionRecoveryOutcome::Retained) => {
TransitionTransactionRecoveryOutcome::Retained
| TransitionTransactionRecoveryOutcome::RetainedAmbiguous(_)
| TransitionTransactionRecoveryOutcome::OperatorRequired(_),
) => {
stats.retained += 1; stats.retained += 1;
debug!( debug!(
event = EVENT_LIFECYCLE_TRANSITION_TRANSACTION_RECOVERY, event = EVENT_LIFECYCLE_TRANSITION_TRANSACTION_RECOVERY,
@@ -1982,74 +1509,11 @@ fn state_requires_known_remote_version(state: TransitionTransactionState) -> boo
#[cfg(test)] #[cfg(test)]
mod tests { mod tests {
use std::collections::HashMap; use std::collections::HashMap;
use std::sync::atomic::{AtomicBool, Ordering};
use super::*; use super::*;
const BACKEND_FINGERPRINT: [u8; 32] = [7; 32]; const BACKEND_FINGERPRINT: [u8; 32] = [7; 32];
struct RecoveryAttemptDropGuard(Arc<AtomicBool>);
impl Drop for RecoveryAttemptDropGuard {
fn drop(&mut self) {
self.0.store(true, Ordering::SeqCst);
}
}
async fn pending_recovery_attempt(started: Arc<tokio::sync::Notify>, dropped: Arc<AtomicBool>) -> EcstoreResult<()> {
let _drop_guard = RecoveryAttemptDropGuard(dropped);
started.notify_one();
std::future::pending().await
}
#[tokio::test(start_paused = true)]
async fn transition_recovery_timeout_and_cancellation_drop_inflight_attempts() {
let timeout_started = Arc::new(tokio::sync::Notify::new());
let timeout_dropped = Arc::new(AtomicBool::new(false));
let timeout_task = tokio::spawn({
let started = Arc::clone(&timeout_started);
let dropped = Arc::clone(&timeout_dropped);
async move {
await_transition_transaction_recovery(
&CancellationToken::new(),
TRANSITION_TRANSACTION_RECOVERY_TIMEOUT,
pending_recovery_attempt(started, dropped),
)
.await
}
});
timeout_started.notified().await;
tokio::time::advance(TRANSITION_TRANSACTION_RECOVERY_TIMEOUT).await;
let timed_out = timeout_task.await.expect("timeout wrapper task should join");
assert!(matches!(timed_out, Some(Err(_))), "outer timeout should fail the recovery pass");
assert!(timeout_dropped.load(Ordering::SeqCst), "outer timeout must drop its in-flight attempt");
let cancel_token = CancellationToken::new();
let cancel_started = Arc::new(tokio::sync::Notify::new());
let cancel_dropped = Arc::new(AtomicBool::new(false));
let cancel_task = tokio::spawn({
let cancel_token = cancel_token.clone();
let started = Arc::clone(&cancel_started);
let dropped = Arc::clone(&cancel_dropped);
async move {
await_transition_transaction_recovery(
&cancel_token,
TRANSITION_TRANSACTION_RECOVERY_TIMEOUT,
pending_recovery_attempt(started, dropped),
)
.await
}
});
cancel_started.notified().await;
cancel_token.cancel();
let cancelled = cancel_task.await.expect("cancellation wrapper task should join");
assert!(cancelled.is_none(), "outer cancellation should stop the recovery loop");
assert!(
cancel_dropped.load(Ordering::SeqCst),
"outer cancellation must drop its in-flight attempt"
);
}
#[derive(Default)] #[derive(Default)]
struct MemoryTransactionStore { struct MemoryTransactionStore {
records: HashMap<Uuid, Vec<u8>>, records: HashMap<Uuid, Vec<u8>>,
@@ -2504,19 +1968,5 @@ mod tests {
transition_transaction_record_object_name(Uuid::nil()), transition_transaction_record_object_name(Uuid::nil()),
Err(TransitionTransactionError::Corrupt("transaction_id is nil")) Err(TransitionTransactionError::Corrupt("transaction_id is nil"))
)); ));
assert_eq!(
transition_transaction_id_from_record_object_name(&object).expect("canonical record path should parse"),
transaction_id
);
for malformed in [
object.to_ascii_uppercase(),
object.replace("/aa/aa/", "/ff/aa/"),
object.replace("/aa/aa/", "/aa/aa/extra/"),
] {
assert!(matches!(
transition_transaction_id_from_record_object_name(&malformed),
Err(TransitionTransactionError::Corrupt(_))
));
}
} }
} }
+64 -25
View File
@@ -489,17 +489,28 @@ impl BucketMetadata {
!self.bucket_targets_config_json.is_empty() && self.bucket_target_config.is_none() !self.bucket_targets_config_json.is_empty() && self.bucket_target_config.is_none()
} }
/// Opaque application-owned configuration with its persisted update time. /// Parsed per-bucket durability override, if a valid one is stored.
/// Empty bytes mean absent or cleared; decoding belongs to the consumer. ///
pub fn on_demand_migration_config(&self) -> Option<(&[u8], OffsetDateTime)> { /// Absent/empty/unparsable payloads all mean "no override" (the bucket
(!self.on_demand_migration_config_json.is_empty()).then_some(( /// follows the global durability mode); a parse failure is logged so a
self.on_demand_migration_config_json.as_slice(), /// corrupted entry cannot silently change fsync behavior.
self.on_demand_migration_config_updated_at, /// Parsed on-demand migration config, if one is stored.
)) ///
/// `Ok(None)` means no config (absent or cleared). A stored payload that
/// does not parse is an error, never a default: the runtime must not
/// pull from a source it cannot describe.
pub fn on_demand_migration_config(
&self,
) -> std::result::Result<
Option<super::on_demand_migration::OnDemandMigrationConfig>,
super::on_demand_migration::OnDemandMigrationConfigError,
> {
if self.on_demand_migration_config_json.is_empty() {
return Ok(None);
}
super::on_demand_migration::OnDemandMigrationConfig::from_json(&self.on_demand_migration_config_json).map(Some)
} }
/// Parsed per-bucket durability override, if a valid one is stored.
/// Invalid payloads follow the global mode after logging a parse failure.
pub fn durability_config(&self) -> Option<super::durability::BucketDurabilityConfig> { pub fn durability_config(&self) -> Option<super::durability::BucketDurabilityConfig> {
if self.durability_config_json.is_empty() { if self.durability_config_json.is_empty() {
return None; return None;
@@ -905,6 +916,13 @@ impl BucketMetadata {
self.durability_config_updated_at = updated; self.durability_config_updated_at = updated;
} }
BUCKET_ON_DEMAND_MIGRATION_CONFIG => { BUCKET_ON_DEMAND_MIGRATION_CONFIG => {
// Structural check only (shape, unknown fields); the
// deployment-relative rules run in the admin handler with a
// `ValidationContext`. A blob this build cannot read must not
// be persisted for every later reader to trip over.
if !data.is_empty() {
super::on_demand_migration::OnDemandMigrationConfig::from_json(&data).map_err(Error::other)?;
}
self.on_demand_migration_config_json = data; self.on_demand_migration_config_json = data;
self.on_demand_migration_config_updated_at = updated; self.on_demand_migration_config_updated_at = updated;
} }
@@ -1960,30 +1978,51 @@ mod test {
const ODM_JSON: &[u8] = br#"{"version":1,"enabled":true,"source":{"provider":"minio","endpoint":"https://legacy.example.com:9000","region":"auto","bucket":"legacy-bucket","credentials":{"access_key":"AK","secret_key":"SK"}}}"#; const ODM_JSON: &[u8] = br#"{"version":1,"enabled":true,"source":{"provider":"minio","endpoint":"https://legacy.example.com:9000","region":"auto","bucket":"legacy-bucket","credentials":{"access_key":"AK","secret_key":"SK"}}}"#;
/// The metadata codec preserves application-owned bytes and timestamps. /// rustfs/backlog#2148: the on-demand migration config is a RustFS
/// extension entry that round-trips through `update_config` and the
/// msgpack codec, clears on delete, and never parses corruption into a
/// default.
#[test] #[test]
fn on_demand_migration_config_round_trips_and_tracks_updates() { fn on_demand_migration_config_round_trips_and_tracks_updates() {
use crate::bucket::on_demand_migration::{OnDemandMigrationConfig, OnDemandMigrationConfigError};
let mut bm = BucketMetadata::new("odm-bucket"); let mut bm = BucketMetadata::new("odm-bucket");
assert_eq!(bm.on_demand_migration_config(), None, "fresh metadata carries no config"); assert_eq!(bm.on_demand_migration_config(), Ok(None), "fresh metadata carries no config");
let expected = OnDemandMigrationConfig::from_json(ODM_JSON).unwrap();
bm.update_config(BUCKET_ON_DEMAND_MIGRATION_CONFIG, ODM_JSON.to_vec()) bm.update_config(BUCKET_ON_DEMAND_MIGRATION_CONFIG, ODM_JSON.to_vec())
.expect("opaque config is accepted"); .expect("valid config is accepted");
let stamped = bm.on_demand_migration_config_updated_at; assert_ne!(bm.on_demand_migration_config_updated_at, OffsetDateTime::UNIX_EPOCH);
assert_ne!(stamped, OffsetDateTime::UNIX_EPOCH); assert_eq!(bm.on_demand_migration_config(), Ok(Some(expected.clone())));
assert_eq!(bm.on_demand_migration_config(), Some((ODM_JSON, stamped)));
let back = BucketMetadata::unmarshal(&bm.marshal_msg().unwrap()).unwrap(); let back = BucketMetadata::unmarshal(&bm.marshal_msg().unwrap()).unwrap();
assert_eq!(back.on_demand_migration_config_json, bm.on_demand_migration_config_json); assert_eq!(back.on_demand_migration_config_json, bm.on_demand_migration_config_json);
assert_eq!(back.on_demand_migration_config_updated_at.unix_timestamp(), stamped.unix_timestamp()); assert_eq!(
back.on_demand_migration_config_updated_at.unix_timestamp(),
bm.on_demand_migration_config_updated_at.unix_timestamp()
);
assert_eq!(back.on_demand_migration_config(), Ok(Some(expected)));
// A blob this build cannot read is rejected at the write boundary
// rather than persisted for every reader to trip over.
let before = bm.on_demand_migration_config_json.clone();
assert!(
bm.update_config(BUCKET_ON_DEMAND_MIGRATION_CONFIG, br#"{"source":{"provider":"s3"},"bogus":1}"#.to_vec())
.is_err()
);
assert_eq!(bm.on_demand_migration_config_json, before, "a rejected update leaves the blob untouched");
// Delete clears the entry.
let stamped = bm.on_demand_migration_config_updated_at;
bm.update_config(BUCKET_ON_DEMAND_MIGRATION_CONFIG, Vec::new()).unwrap(); bm.update_config(BUCKET_ON_DEMAND_MIGRATION_CONFIG, Vec::new()).unwrap();
assert!(bm.on_demand_migration_config_json.is_empty()); assert!(bm.on_demand_migration_config_json.is_empty());
assert_eq!(bm.on_demand_migration_config(), None); assert_eq!(bm.on_demand_migration_config(), Ok(None));
assert!(bm.on_demand_migration_config_updated_at >= stamped); assert!(bm.on_demand_migration_config_updated_at >= stamped);
bm.update_config(BUCKET_ON_DEMAND_MIGRATION_CONFIG, b"not-json".to_vec())
.unwrap(); // Corruption that bypassed `update_config` (disk, another writer)
let back = BucketMetadata::unmarshal(&bm.marshal_msg().unwrap()).unwrap(); // is a typed error, never a default.
assert_eq!( bm.on_demand_migration_config_json = b"not-json".to_vec();
back.on_demand_migration_config_json, b"not-json", assert!(matches!(bm.on_demand_migration_config(), Err(OnDemandMigrationConfigError::Malformed(_))));
"metadata must not reinterpret application bytes"
);
} }
/// rustfs/backlog#2148: a `.metadata.bin` written before the on-demand /// rustfs/backlog#2148: a `.metadata.bin` written before the on-demand
@@ -1995,7 +2034,7 @@ mod test {
let mut bm = BucketMetadata::unmarshal(&blob[4..]).expect("unmarshal MinIO bucket metadata"); let mut bm = BucketMetadata::unmarshal(&blob[4..]).expect("unmarshal MinIO bucket metadata");
assert!(bm.on_demand_migration_config_json.is_empty()); assert!(bm.on_demand_migration_config_json.is_empty());
assert_eq!(bm.on_demand_migration_config_updated_at, OffsetDateTime::UNIX_EPOCH); assert_eq!(bm.on_demand_migration_config_updated_at, OffsetDateTime::UNIX_EPOCH);
assert_eq!(bm.on_demand_migration_config(), None); assert_eq!(bm.on_demand_migration_config(), Ok(None));
bm.default_timestamps(); bm.default_timestamps();
assert_ne!(bm.created, OffsetDateTime::UNIX_EPOCH, "fixture must carry a real creation time"); assert_ne!(bm.created, OffsetDateTime::UNIX_EPOCH, "fixture must carry a real creation time");
+98 -58
View File
@@ -19,6 +19,7 @@ use super::quota::BucketQuota;
use super::target::BucketTargets; use super::target::BucketTargets;
use crate::bucket::bucket_target_sys::BucketTargetSys; use crate::bucket::bucket_target_sys::BucketTargetSys;
use crate::bucket::metadata::{load_bucket_metadata_parse, load_bucket_metadata_parse_with_presence}; use crate::bucket::metadata::{load_bucket_metadata_parse, load_bucket_metadata_parse_with_presence};
use crate::bucket::on_demand_migration::{ON_DEMAND_MIGRATION_CONFIG_HOOK, OnDemandMigrationConfig};
use crate::bucket::utils::is_meta_bucketname; use crate::bucket::utils::is_meta_bucketname;
use crate::disk::RUSTFS_META_BUCKET; use crate::disk::RUSTFS_META_BUCKET;
use crate::error::{Error, Result, is_err_bucket_not_found, is_err_strict_volume_not_found}; use crate::error::{Error, Result, is_err_bucket_not_found, is_err_strict_volume_not_found};
@@ -48,11 +49,6 @@ use tokio_util::sync::CancellationToken;
use tracing::{error, warn}; use tracing::{error, warn};
use uuid::Uuid; use uuid::Uuid;
/// Opaque bucket configuration notifications for application-owned services.
/// `None` withdraws a configuration; consumers validate nonempty bytes.
pub type BucketConfigPublishHook = Box<dyn Fn(&str, &str, Option<(&[u8], OffsetDateTime, Uuid)>) + Send + Sync>;
pub static BUCKET_CONFIG_PUBLISH_HOOK: std::sync::OnceLock<BucketConfigPublishHook> = std::sync::OnceLock::new();
const BUCKET_METADATA_REFRESH_INTERVAL: Duration = Duration::from_secs(15 * 60); const BUCKET_METADATA_REFRESH_INTERVAL: Duration = Duration::from_secs(15 * 60);
#[cfg(any(test, feature = "test-util"))] #[cfg(any(test, feature = "test-util"))]
@@ -399,21 +395,39 @@ fn clear_bucket_durability(bucket: &str) {
crate::disk::local::bucket_durability::set(bucket, None); crate::disk::local::bucket_durability::set(bucket, None);
} }
/// Publish application-owned bytes on every cache install path. /// Publish the bucket's on-demand migration config (or its absence) to the
/// runtime registered in `ON_DEMAND_MIGRATION_CONFIG_HOOK`.
///
/// Called from the same five cache-install paths as
/// [`sync_bucket_durability`]. A stored payload this build cannot parse is
/// published as `None`: the runtime must stop pulling for that bucket rather
/// than keep an older config or guess.
fn sync_on_demand_migration(bucket: &str, bm: &BucketMetadata) { fn sync_on_demand_migration(bucket: &str, bm: &BucketMetadata) {
if let Some(hook) = BUCKET_CONFIG_PUBLISH_HOOK.get() { let Some(hook) = ON_DEMAND_MIGRATION_CONFIG_HOOK.get() else {
hook( return;
bucket, };
super::metadata::BUCKET_ON_DEMAND_MIGRATION_CONFIG, match bm.on_demand_migration_config() {
bm.on_demand_migration_config() Ok(config) => hook(bucket, config.as_ref()),
.map(|(bytes, stamp)| (bytes, stamp, bm.bucket_incarnation_id)), Err(err) => {
); warn!(
event = "bucket_metadata_parse_failed",
component = "ecstore",
subsystem = "bucket_metadata",
bucket = %bucket,
config = "on_demand_migration",
error = %err,
"Failed to parse bucket metadata config"
);
hook(bucket, None);
}
} }
} }
/// Withdraw a bucket's on-demand migration config when its metadata leaves
/// the cache.
fn clear_on_demand_migration(bucket: &str) { fn clear_on_demand_migration(bucket: &str) {
if let Some(hook) = BUCKET_CONFIG_PUBLISH_HOOK.get() { if let Some(hook) = ON_DEMAND_MIGRATION_CONFIG_HOOK.get() {
hook(bucket, super::metadata::BUCKET_ON_DEMAND_MIGRATION_CONFIG, None); hook(bucket, None);
} }
} }
@@ -1035,21 +1049,15 @@ pub async fn get_durability_config(
} }
/// The bucket's on-demand migration config with its update time, or /// The bucket's on-demand migration config with its update time, or
/// `Ok(None)` when the bucket has none. Bytes are opaque to the metadata owner. /// `Ok(None)` when the bucket has none. A stored payload that does not parse
pub async fn get_on_demand_migration_config(bucket: &str) -> Result<Option<(Vec<u8>, OffsetDateTime)>> { /// is a typed error (`OnDemandMigrationConfigError` inside `Error::Io`).
pub async fn get_on_demand_migration_config(bucket: &str) -> Result<Option<(OnDemandMigrationConfig, OffsetDateTime)>> {
let bucket_meta_sys_lock = get_bucket_metadata_sys()?; let bucket_meta_sys_lock = get_bucket_metadata_sys()?;
let bucket_meta_sys = bucket_meta_sys_lock.read().await; let bucket_meta_sys = bucket_meta_sys_lock.read().await;
bucket_meta_sys.get_on_demand_migration_config(bucket).await bucket_meta_sys.get_on_demand_migration_config(bucket).await
} }
/// Resolve opaque configuration from the store's own metadata system.
pub async fn get_on_demand_migration_config_in(api: &ECStore, bucket: &str) -> Result<Option<(Vec<u8>, OffsetDateTime)>> {
let sys = bucket_metadata_sys_of(&api.ctx)?;
let lock = sys.read().await;
lock.get_on_demand_migration_config(bucket).await
}
pub async fn get_quota_config(bucket: &str) -> Result<(BucketQuota, OffsetDateTime)> { pub async fn get_quota_config(bucket: &str) -> Result<(BucketQuota, OffsetDateTime)> {
let bucket_meta_sys_lock = get_bucket_metadata_sys()?; let bucket_meta_sys_lock = get_bucket_metadata_sys()?;
let bucket_meta_sys = bucket_meta_sys_lock.read().await; let bucket_meta_sys = bucket_meta_sys_lock.read().await;
@@ -2571,27 +2579,29 @@ impl BucketMetadataSys {
} }
/// See [`get_on_demand_migration_config`]. /// See [`get_on_demand_migration_config`].
pub async fn get_on_demand_migration_config(&self, bucket: &str) -> Result<Option<(Vec<u8>, OffsetDateTime)>> { pub async fn get_on_demand_migration_config(
&self,
bucket: &str,
) -> Result<Option<(OnDemandMigrationConfig, OffsetDateTime)>> {
let (bm, _) = self.get_config(bucket).await?; let (bm, _) = self.get_config(bucket).await?;
Ok(bm let config = bm.on_demand_migration_config().map_err(Error::other)?;
.on_demand_migration_config() Ok(config.map(|config| (config, bm.on_demand_migration_config_updated_at)))
.map(|(bytes, updated_at)| (bytes.to_vec(), updated_at)))
} }
} }
/// Test-only fixture shared with sibling modules (e.g. the quota checker /// Test-only fixture shared with sibling modules (e.g. the quota checker
/// tests): a 4-disk `ECStore` on an isolated instance context, so tests /// tests): a 4-disk `ECStore` on an isolated instance context, so tests
/// exercising the metadata system never touch ambient process state. /// exercising the metadata system never touch ambient process state.
#[cfg(any(test, feature = "test-util"))] #[cfg(test)]
pub mod test_support { pub(crate) mod test_support {
use super::*; use super::*;
use crate::disk::endpoint::Endpoint; use crate::disk::endpoint::Endpoint;
use crate::layout::endpoints::{EndpointServerPools, Endpoints, PoolEndpoints}; use crate::layout::endpoints::{EndpointServerPools, Endpoints, PoolEndpoints};
use crate::runtime::instance::InstanceContext; use crate::runtime::instance::InstanceContext;
use crate::store::init_local_disks_with_instance_ctx; use crate::store::init_local_disks_with_instance_ctx;
pub async fn isolated_store_over_temp_disks() -> (Vec<tempfile::TempDir>, Arc<ECStore>) { pub(crate) async fn isolated_store_over_temp_disks() -> (Vec<tempfile::TempDir>, Arc<ECStore>) {
let mut dirs = Vec::with_capacity(4); let mut dirs = Vec::with_capacity(4);
let mut endpoints = Vec::with_capacity(4); let mut endpoints = Vec::with_capacity(4);
for disk_idx in 0..4 { for disk_idx in 0..4 {
@@ -4375,26 +4385,19 @@ mod tests {
const ODM_JSON: &[u8] = br#"{"source":{"provider":"minio","endpoint":"https://legacy.example.com:9000","region":"auto","bucket":"legacy-bucket","credentials":{"access_key":"AK","secret_key":"SK"}}}"#; const ODM_JSON: &[u8] = br#"{"source":{"provider":"minio","endpoint":"https://legacy.example.com:9000","region":"auto","bucket":"legacy-bucket","credentials":{"access_key":"AK","secret_key":"SK"}}}"#;
type RecordedOdmConfig = Option<(Vec<u8>, OffsetDateTime, Uuid)>;
type RecordedOdmHookCall = (String, RecordedOdmConfig);
/// Every `(bucket, config)` the recording hook has seen. Tests filter by /// Every `(bucket, config)` the recording hook has seen. Tests filter by
/// their own bucket name; the hook is process-wide and set once. /// their own bucket name; the hook is process-wide and set once.
static ODM_HOOK_CALLS: std::sync::Mutex<Vec<RecordedOdmHookCall>> = std::sync::Mutex::new(Vec::new()); static ODM_HOOK_CALLS: std::sync::Mutex<Vec<(String, Option<OnDemandMigrationConfig>)>> = std::sync::Mutex::new(Vec::new());
fn install_recording_odm_hook() { fn install_recording_odm_hook() {
BUCKET_CONFIG_PUBLISH_HOOK.get_or_init(|| { ON_DEMAND_MIGRATION_CONFIG_HOOK.get_or_init(|| {
Box::new(|bucket, config_file, config| { Box::new(|bucket, config| {
assert_eq!(config_file, super::super::metadata::BUCKET_ON_DEMAND_MIGRATION_CONFIG); ODM_HOOK_CALLS.lock().unwrap().push((bucket.to_string(), config.cloned()));
ODM_HOOK_CALLS.lock().unwrap().push((
bucket.to_string(),
config.map(|(bytes, stamp, incarnation)| (bytes.to_vec(), stamp, incarnation)),
));
}) })
}); });
} }
fn odm_hook_calls(bucket: &str) -> Vec<RecordedOdmConfig> { fn odm_hook_calls(bucket: &str) -> Vec<Option<OnDemandMigrationConfig>> {
ODM_HOOK_CALLS ODM_HOOK_CALLS
.lock() .lock()
.unwrap() .unwrap()
@@ -4404,6 +4407,54 @@ mod tests {
.collect() .collect()
} }
/// rustfs/backlog#2148: the accessor reports absence as `Ok(None)` and a
/// stored payload it cannot parse as a typed error, never as a default
/// and never as `ConfigNotFound`.
#[tokio::test]
async fn get_on_demand_migration_config_distinguishes_absent_from_corrupt() {
use crate::bucket::on_demand_migration::OnDemandMigrationConfigError;
let (_dirs, ecstore) = isolated_store_over_temp_disks().await;
let sys = BucketMetadataSys::new(ecstore);
let bucket = "odm-accessor";
sys.set(bucket.to_string(), Arc::new(BucketMetadata::new(bucket))).await;
assert_eq!(sys.get_on_demand_migration_config(bucket).await.unwrap(), None);
let mut corrupt = BucketMetadata::new(bucket);
corrupt.on_demand_migration_config_json = br#"{"source":{"provider":"s3"},"bogus":1}"#.to_vec();
sys.set(bucket.to_string(), Arc::new(corrupt)).await;
let err = sys
.get_on_demand_migration_config(bucket)
.await
.expect_err("corrupt config must not read as a default");
assert_ne!(err, Error::ConfigNotFound, "corruption must not be reported as absence");
let typed = match &err {
Error::Io(io) => io
.get_ref()
.and_then(|source| source.downcast_ref::<OnDemandMigrationConfigError>()),
_ => None,
};
assert!(
matches!(typed, Some(OnDemandMigrationConfigError::Malformed(_))),
"typed parse error must survive the Result boundary, got: {err:?}"
);
let mut valid = BucketMetadata::new(bucket);
valid
.update_config(crate::bucket::metadata::BUCKET_ON_DEMAND_MIGRATION_CONFIG, ODM_JSON.to_vec())
.unwrap();
let stamped = valid.on_demand_migration_config_updated_at;
sys.set(bucket.to_string(), Arc::new(valid)).await;
let (config, updated_at) = sys
.get_on_demand_migration_config(bucket)
.await
.unwrap()
.expect("stored config is returned");
assert_eq!(config, OnDemandMigrationConfig::from_json(ODM_JSON).unwrap());
assert_eq!(updated_at, stamped);
}
/// rustfs/backlog#2148: the publish hook fires on every path that /// rustfs/backlog#2148: the publish hook fires on every path that
/// installs bucket metadata into the cache (set, initial load, peer /// installs bucket metadata into the cache (set, initial load, peer
/// reload, refresh loop, lazy load) and withdraws on removal, mirroring /// reload, refresh loop, lazy load) and withdraws on removal, mirroring
@@ -4417,22 +4468,15 @@ mod tests {
for dir in &dirs { for dir in &dirs {
std::fs::create_dir_all(dir.path().join(bucket)).expect("physical bucket should exist"); std::fs::create_dir_all(dir.path().join(bucket)).expect("physical bucket should exist");
} }
let expected = OnDemandMigrationConfig::from_json(ODM_JSON).unwrap();
let incarnation = Uuid::new_v4();
let expect_publish = |before: usize, label: &str| { let expect_publish = |before: usize, label: &str| {
let calls = odm_hook_calls(bucket); let calls = odm_hook_calls(bucket);
assert_eq!(calls.len(), before + 1, "{label} must publish exactly once"); assert_eq!(calls.len(), before + 1, "{label} must publish exactly once");
assert_eq!( assert_eq!(calls.last().unwrap().as_ref(), Some(&expected), "{label} must publish the stored config");
calls.last().unwrap().as_ref().map(|(bytes, _, _)| bytes.as_slice()),
Some(ODM_JSON),
"{label} must publish the stored bytes"
);
assert_eq!(calls.last().unwrap().as_ref().map(|(_, _, id)| *id), Some(incarnation));
}; };
// set (via persist_new_and_set, which installs through `set`). // set (via persist_new_and_set, which installs through `set`).
let mut bm = BucketMetadata::new(bucket); let mut bm = BucketMetadata::new(bucket);
bm.bucket_incarnation_id = incarnation;
bm.update_config(crate::bucket::metadata::BUCKET_ON_DEMAND_MIGRATION_CONFIG, ODM_JSON.to_vec()) bm.update_config(crate::bucket::metadata::BUCKET_ON_DEMAND_MIGRATION_CONFIG, ODM_JSON.to_vec())
.unwrap(); .unwrap();
let writer = BucketMetadataSys::new(ecstore.clone()); let writer = BucketMetadataSys::new(ecstore.clone());
@@ -4474,18 +4518,14 @@ mod tests {
assert_eq!(calls.len(), before + 1, "remove must withdraw exactly once"); assert_eq!(calls.len(), before + 1, "remove must withdraw exactly once");
assert_eq!(calls.last().unwrap(), &None); assert_eq!(calls.last().unwrap(), &None);
// Opaque bytes reach the application even if they are not valid JSON. // A corrupt payload is withdrawn, never published as a config.
let mut corrupt = BucketMetadata::new(bucket); let mut corrupt = BucketMetadata::new(bucket);
corrupt.on_demand_migration_config_json = b"not-json".to_vec(); corrupt.on_demand_migration_config_json = b"not-json".to_vec();
let before = odm_hook_calls(bucket).len(); let before = odm_hook_calls(bucket).len();
lazy.set(bucket.to_string(), Arc::new(corrupt)).await; lazy.set(bucket.to_string(), Arc::new(corrupt)).await;
let calls = odm_hook_calls(bucket); let calls = odm_hook_calls(bucket);
assert_eq!(calls.len(), before + 1); assert_eq!(calls.len(), before + 1);
assert_eq!( assert_eq!(calls.last().unwrap(), &None, "unreadable config must publish absence");
calls.last().unwrap().as_ref().map(|(bytes, _, _)| bytes.as_slice()),
Some(b"not-json".as_slice()),
"the application validates opaque config bytes"
);
} }
#[tokio::test] #[tokio::test]
+1
View File
@@ -26,6 +26,7 @@ mod metadata_test;
pub mod migration; pub mod migration;
mod msgp_decode; mod msgp_decode;
pub mod object_lock; pub mod object_lock;
pub mod on_demand_migration;
pub mod policy_sys; pub mod policy_sys;
pub mod quota; pub mod quota;
pub mod remote_s3_client; pub mod remote_s3_client;
@@ -25,8 +25,8 @@
//! [`BACKFILL_SAVE_INTERVAL`], and at every page end, with an `If-Match` //! [`BACKFILL_SAVE_INTERVAL`], and at every page end, with an `If-Match`
//! compare-and-set so a concurrent cancel or takeover is never overwritten. //! compare-and-set so a concurrent cancel or takeover is never overwritten.
//! - The `continuation_token` only advances once every pull queued from the //! - The `continuation_token` only advances once every pull queued from the
//! page before it has succeeded. After a failure it stays at that page, //! page before it has reported back, so a crash re-lists at most one page
//! so crash recovery cannot skip failed pulls (existing keys are skipped). //! (already-present keys are then skipped, never re-pulled).
//! - The owner holds a lease of [`BACKFILL_LEASE`] renewed by every save. The //! - The owner holds a lease of [`BACKFILL_LEASE`] renewed by every save. The
//! recovery loop ([`run_backfill_recovery_loop`]) scans the buckets this //! recovery loop ([`run_backfill_recovery_loop`]) scans the buckets this
//! node has an ODM state for every [`BACKFILL_RECOVERY_INTERVAL`] and takes //! node has an ODM state for every [`BACKFILL_RECOVERY_INTERVAL`] and takes
@@ -45,12 +45,16 @@
use super::pull::{EnqueueOutcome, PullReason, QueuedPullOutcome}; use super::pull::{EnqueueOutcome, PullReason, QueuedPullOutcome};
use super::source_client::{SourceError, SourcePage}; use super::source_client::{SourceError, SourcePage};
use super::storage_api::{
BUCKET_META_PREFIX, ECStore, HTTPPreconditions, NamespaceLocking as _, ObjectOperations as _, ObjectOptions,
RUSTFS_META_BUCKET, StorageError, WriteCompletion, get_lock_acquire_timeout, get_on_demand_migration_config_in,
local_node_name, read_config_with_metadata, save_config_with_opts,
};
use super::sys::{BucketOdmState, OnDemandMigrationSys}; use super::sys::{BucketOdmState, OnDemandMigrationSys};
use crate::bucket::metadata_sys::bucket_metadata_sys_of;
use crate::config::com::{read_config_with_metadata, save_config_with_opts};
use crate::disk::{BUCKET_META_PREFIX, RUSTFS_META_BUCKET};
use crate::error::Error as StorageError;
use crate::object_api::ObjectOptions;
use crate::runtime::sources::local_node_name;
use crate::set_disk::get_lock_acquire_timeout;
use crate::storage_api_contracts::{namespace::NamespaceLocking as _, object::HTTPPreconditions, object::ObjectOperations as _};
use crate::store::ECStore;
use async_trait::async_trait; use async_trait::async_trait;
use futures::StreamExt; use futures::StreamExt;
use futures::stream::FuturesUnordered; use futures::stream::FuturesUnordered;
@@ -363,15 +367,14 @@ pub struct LocalBackfillObject {
pub source_etag: Option<String>, pub source_etag: Option<String>,
} }
/// Shared report of a new or coalesced pull; absent only when not admitted. /// Receiver of one queued pull's report; `None` when the pull was coalesced
pub type PullReport = Option<super::pull::QueuedPullReport>; /// into one already running.
pub type PullReport = Option<oneshot::Receiver<QueuedPullOutcome>>;
/// Everything the job needs from its bucket, so the loop can run against a /// Everything the job needs from its bucket, so the loop can run against a
/// mock in unit tests. Production: [`BucketBackfillContext`]. /// mock in unit tests. Production: [`BucketBackfillContext`].
#[async_trait] #[async_trait]
pub trait BackfillContext: Send + Sync { pub trait BackfillContext: Send + Sync {
/// The bucket incarnation captured by this context.
fn incarnation_id(&self) -> Uuid;
/// One source page in the local key namespace. /// One source page in the local key namespace.
async fn list_page(&self, prefix: Option<&str>, token: Option<&str>, max_keys: i32) -> Result<SourcePage, SourceError>; async fn list_page(&self, prefix: Option<&str>, token: Option<&str>, max_keys: i32) -> Result<SourcePage, SourceError>;
/// Whether the breaker admits source traffic right now. /// Whether the breaker admits source traffic right now.
@@ -413,10 +416,6 @@ impl BucketBackfillContext {
#[async_trait] #[async_trait]
impl BackfillContext for BucketBackfillContext { impl BackfillContext for BucketBackfillContext {
fn incarnation_id(&self) -> Uuid {
self.state.incarnation_id()
}
async fn list_page(&self, prefix: Option<&str>, token: Option<&str>, max_keys: i32) -> Result<SourcePage, SourceError> { async fn list_page(&self, prefix: Option<&str>, token: Option<&str>, max_keys: i32) -> Result<SourcePage, SourceError> {
let client = self.state.client().map_err(|err| SourceError::Unsupported(err.to_string()))?; let client = self.state.client().map_err(|err| SourceError::Unsupported(err.to_string()))?;
let started = Instant::now(); let started = Instant::now();
@@ -476,10 +475,12 @@ impl BackfillContext for BucketBackfillContext {
} }
async fn config_updated_at(&self) -> Result<Option<OffsetDateTime>, StorageError> { async fn config_updated_at(&self) -> Result<Option<OffsetDateTime>, StorageError> {
Ok( let sys = bucket_metadata_sys_of(&self.api.ctx)?;
super::config::decode_stored_config(get_on_demand_migration_config_in(&self.api, self.state.bucket()).await?)? let guard = sys.read().await;
.map(|(_, updated_at)| updated_at), Ok(guard
) .get_on_demand_migration_config(self.state.bucket())
.await?
.map(|(_, updated_at)| updated_at))
} }
} }
@@ -667,36 +668,8 @@ pub async fn read_checkpoint(api: &Arc<ECStore>, bucket: &str) -> Result<Option<
async fn write_checkpoint( async fn write_checkpoint(
api: &Arc<ECStore>, api: &Arc<ECStore>,
bucket: &str, bucket: &str,
incarnation_id: Uuid,
checkpoint: &BackfillCheckpoint, checkpoint: &BackfillCheckpoint,
expected_etag: Option<&str>, expected_etag: Option<&str>,
) -> Result<String, BackfillError> {
let api = Arc::clone(api);
let bucket = bucket.to_string();
let checkpoint = checkpoint.clone();
let expected_etag = expected_etag.map(str::to_string);
// The storage commit owns detached work. Keep its user-bucket fence alive
// even when a caller aborts its waiter before the erasure tail has drained.
tokio::spawn(async move {
let fence = api.acquire_bucket_incarnation_fence(&bucket, incarnation_id).await?;
let mut opts = ObjectOptions::default();
fence.attach_to_object_options(&mut opts);
let result = write_checkpoint_while_fenced(&api, &bucket, &checkpoint, expected_etag.as_deref(), opts).await;
drop(fence);
result
})
.await
.map_err(|err| StorageError::other(format!("backfill checkpoint task failed: {err}")))?
}
/// The caller holds the destination bucket's lifecycle fence through the CAS
/// write and its read-back, including the drained erasure write tail.
async fn write_checkpoint_while_fenced(
api: &Arc<ECStore>,
bucket: &str,
checkpoint: &BackfillCheckpoint,
expected_etag: Option<&str>,
mut opts: ObjectOptions,
) -> Result<String, BackfillError> { ) -> Result<String, BackfillError> {
let data = checkpoint.to_json()?; let data = checkpoint.to_json()?;
let preconditions = match expected_etag { let preconditions = match expected_etag {
@@ -709,9 +682,12 @@ async fn write_checkpoint_while_fenced(
..Default::default() ..Default::default()
}, },
}; };
opts.max_parity = true; let opts = ObjectOptions {
opts.write_completion = WriteCompletion::TailDrained; max_parity: true,
opts.http_preconditions = Some(preconditions); write_completion: crate::object_api::WriteCompletion::TailDrained,
http_preconditions: Some(preconditions),
..Default::default()
};
match save_config_with_opts(Arc::clone(api), &checkpoint_path(bucket), data, &opts).await { match save_config_with_opts(Arc::clone(api), &checkpoint_path(bucket), data, &opts).await {
Ok(()) => {} Ok(()) => {}
Err(StorageError::PreconditionFailed) => return Err(BackfillError::Conflict(bucket.to_string())), Err(StorageError::PreconditionFailed) => return Err(BackfillError::Conflict(bucket.to_string())),
@@ -882,14 +858,7 @@ impl BackfillRunner {
}); });
} }
let checkpoint = BackfillCheckpoint::new(&request, config_updated_at, &self.node, now); let checkpoint = BackfillCheckpoint::new(&request, config_updated_at, &self.node, now);
let etag = write_checkpoint( let etag = write_checkpoint(&self.api, bucket, &checkpoint, stored.as_ref().map(|s| s.etag.as_str())).await?;
&self.api,
bucket,
context.incarnation_id(),
&checkpoint,
stored.as_ref().map(|s| s.etag.as_str()),
)
.await?;
info!( info!(
event = EVENT_ODM_BACKFILL_STATE, event = EVENT_ODM_BACKFILL_STATE,
component = LOG_COMPONENT_ECSTORE, component = LOG_COMPONENT_ECSTORE,
@@ -924,42 +893,30 @@ impl BackfillRunner {
} }
return Ok(handle.snapshot.lock().clone()); return Ok(handle.snapshot.lock().clone());
} }
let incarnation_id = self.api.bucket_incarnation_id_from_disk(bucket).await?; let _lock = self.lease_lock(bucket, get_lock_acquire_timeout()).await?;
let lock = self.lease_lock(bucket, get_lock_acquire_timeout()).await?; let Some(stored) = read_checkpoint(&self.api, bucket).await? else {
let api = Arc::clone(&self.api); return Err(BackfillError::NotFound(bucket.to_string()));
let bucket = bucket.to_string(); };
tokio::spawn(async move { if !stored.checkpoint.state.is_active() {
let _lock = lock; return Ok(stored.checkpoint);
let fence = api.acquire_bucket_incarnation_fence(&bucket, incarnation_id).await?; }
let mut opts = ObjectOptions::default(); let mut checkpoint = stored.checkpoint;
fence.attach_to_object_options(&mut opts); let now = OffsetDateTime::now_utc();
let Some(stored) = read_checkpoint(&api, &bucket).await? else { checkpoint.state = BackfillState::Cancelled;
return Err(BackfillError::NotFound(bucket.to_string())); checkpoint.updated_at = now;
}; write_checkpoint(&self.api, bucket, &checkpoint, Some(&stored.etag)).await?;
if !stored.checkpoint.state.is_active() { info!(
return Ok(stored.checkpoint); event = EVENT_ODM_BACKFILL_STATE,
} component = LOG_COMPONENT_ECSTORE,
let mut checkpoint = stored.checkpoint; subsystem = LOG_SUBSYSTEM_ON_DEMAND_MIGRATION,
let now = OffsetDateTime::now_utc(); state = checkpoint.state.as_str(),
checkpoint.state = BackfillState::Cancelled; result = "cancelled",
checkpoint.updated_at = now; bucket = %bucket,
write_checkpoint_while_fenced(&api, &bucket, &checkpoint, Some(&stored.etag), opts).await?; job_id = %checkpoint.job_id,
info!( owner = %checkpoint.owner.as_ref().map(|o| o.node.as_str()).unwrap_or_default(),
event = EVENT_ODM_BACKFILL_STATE, "On-demand migration backfill job cancelled remotely"
component = LOG_COMPONENT_ECSTORE, );
subsystem = LOG_SUBSYSTEM_ON_DEMAND_MIGRATION, Ok(checkpoint)
state = checkpoint.state.as_str(),
result = "cancelled",
bucket = %bucket,
job_id = %checkpoint.job_id,
owner = %checkpoint.owner.as_ref().map(|o| o.node.as_str()).unwrap_or_default(),
"On-demand migration backfill job cancelled remotely"
);
drop(fence);
Ok(checkpoint)
})
.await
.map_err(|err| StorageError::other(format!("backfill cancellation task failed: {err}")))?
} }
/// Latest checkpoint: the in-memory progress of a local job, else the /// Latest checkpoint: the in-memory progress of a local job, else the
@@ -1052,7 +1009,7 @@ impl BackfillRunner {
checkpoint.state = BackfillState::Cancelled; checkpoint.state = BackfillState::Cancelled;
checkpoint.updated_at = now; checkpoint.updated_at = now;
checkpoint.record_failure("config_changed", None, now); checkpoint.record_failure("config_changed", None, now);
write_checkpoint(&self.api, bucket, context.incarnation_id(), &checkpoint, Some(&stored.etag)).await?; write_checkpoint(&self.api, bucket, &checkpoint, Some(&stored.etag)).await?;
info!( info!(
event = EVENT_ODM_BACKFILL_STATE, event = EVENT_ODM_BACKFILL_STATE,
component = LOG_COMPONENT_ECSTORE, component = LOG_COMPONENT_ECSTORE,
@@ -1071,7 +1028,7 @@ impl BackfillRunner {
node: self.node.clone(), node: self.node.clone(),
lease_until: now + BACKFILL_LEASE, lease_until: now + BACKFILL_LEASE,
}); });
let etag = write_checkpoint(&self.api, bucket, context.incarnation_id(), &checkpoint, Some(&stored.etag)).await?; let etag = write_checkpoint(&self.api, bucket, &checkpoint, Some(&stored.etag)).await?;
warn!( warn!(
event = EVENT_ODM_BACKFILL_LEASE_TAKEOVER, event = EVENT_ODM_BACKFILL_LEASE_TAKEOVER,
component = LOG_COMPONENT_ECSTORE, component = LOG_COMPONENT_ECSTORE,
@@ -1234,11 +1191,9 @@ impl Job {
} }
async fn main_loop(&mut self) -> Result<(), Stop> { async fn main_loop(&mut self) -> Result<(), Stop> {
let mut cursor = self.checkpoint.continuation_token.clone();
let failed_at_resume = self.checkpoint.failed;
loop { loop {
self.check_cancel()?; self.check_cancel()?;
let page = self.list_page(cursor.as_deref()).await?; let page = self.list_page().await?;
for object in &page.objects { for object in &page.objects {
self.check_cancel()?; self.check_cancel()?;
self.checkpoint.listed += 1; self.checkpoint.listed += 1;
@@ -1250,13 +1205,10 @@ impl Job {
self.drain_ready(); self.drain_ready();
self.tick(false).await?; self.tick(false).await?;
} }
// A persisted cursor certifies successful work, not just listing // Only advance the cursor once every pull of this page reported
// progress. Keep it at the first failed page for crash recovery. // back, so a takeover re-lists at most this page.
self.drain_all().await?; self.drain_all().await?;
cursor = page.next_continuation_token; self.checkpoint.continuation_token = page.next_continuation_token.clone();
if self.checkpoint.failed == failed_at_resume {
self.checkpoint.continuation_token = cursor.clone();
}
self.tick(true).await?; self.tick(true).await?;
if !page.is_truncated { if !page.is_truncated {
return Ok(()); return Ok(());
@@ -1271,7 +1223,7 @@ impl Job {
} }
} }
async fn list_page(&mut self, cursor: Option<&str>) -> Result<SourcePage, Stop> { async fn list_page(&mut self) -> Result<SourcePage, Stop> {
let mut attempt = 0; let mut attempt = 0;
loop { loop {
while !self.context.source_available() { while !self.context.source_available() {
@@ -1279,7 +1231,7 @@ impl Job {
self.tick(false).await?; self.tick(false).await?;
} }
let prefix = self.checkpoint.prefix.clone(); let prefix = self.checkpoint.prefix.clone();
let token = cursor.map(str::to_string); let token = self.checkpoint.continuation_token.clone();
match self match self
.context .context
.list_page(prefix.as_deref(), token.as_deref(), BACKFILL_LIST_PAGE_SIZE) .list_page(prefix.as_deref(), token.as_deref(), BACKFILL_LIST_PAGE_SIZE)
@@ -1353,10 +1305,9 @@ impl Job {
} }
loop { loop {
match self.context.enqueue(key) { match self.context.enqueue(key) {
(EnqueueOutcome::Enqueued | EnqueueOutcome::Coalesced, report) => { (EnqueueOutcome::Enqueued, report) => {
self.checkpoint.enqueued += 1; self.checkpoint.enqueued += 1;
let rx = report.ok_or(Stop::Unavailable)?; if let Some(rx) = report {
{
let key = key.to_string(); let key = key.to_string();
self.outstanding.push(Box::pin(async move { (key, rx.await) })); self.outstanding.push(Box::pin(async move { (key, rx.await) }));
} }
@@ -1371,6 +1322,11 @@ impl Job {
); );
return Ok(()); return Ok(());
} }
(EnqueueOutcome::Coalesced, _) => {
// Someone else pulls it; its result is not ours to count.
self.checkpoint.enqueued += 1;
return Ok(());
}
(EnqueueOutcome::QueueFull, _) => { (EnqueueOutcome::QueueFull, _) => {
// Wait, never drop: one completion frees a slot. // Wait, never drop: one completion frees a slot.
if self.outstanding.is_empty() { if self.outstanding.is_empty() {
@@ -1510,8 +1466,7 @@ impl Job {
lease_until: now + BACKFILL_LEASE, lease_until: now + BACKFILL_LEASE,
}); });
} }
let etag = let etag = write_checkpoint(&self.api, &self.bucket, &self.checkpoint, Some(&self.etag)).await?;
write_checkpoint(&self.api, &self.bucket, self.context.incarnation_id(), &self.checkpoint, Some(&self.etag)).await?;
self.etag = etag; self.etag = etag;
self.keys_since_save = 0; self.keys_since_save = 0;
self.last_save = Instant::now(); self.last_save = Instant::now();
@@ -1523,7 +1478,7 @@ impl Job {
/// Spawns [`run_backfill_recovery_loop`] on the store's shutdown token; /// Spawns [`run_backfill_recovery_loop`] on the store's shutdown token;
/// `false` (nothing spawned) when the store has no background token. /// `false` (nothing spawned) when the store has no background token.
pub fn spawn_backfill_recovery_loop(runner: Arc<BackfillRunner>) -> bool { pub fn spawn_backfill_recovery_loop(runner: Arc<BackfillRunner>) -> bool {
let Some(cancel) = runner.api.background_cancel_token() else { let Some(cancel) = runner.api.ctx.background_cancel_token() else {
return false; return false;
}; };
tokio::spawn(run_backfill_recovery_loop(runner, cancel)); tokio::spawn(run_backfill_recovery_loop(runner, cancel));
@@ -1551,13 +1506,10 @@ pub async fn run_backfill_recovery_loop(runner: Arc<BackfillRunner>, cancel: Can
#[cfg(test)] #[cfg(test)]
mod tests { mod tests {
use super::super::storage_api::test_support::{
BUCKET_LIFECYCLE_LOCK_OBJECT, BucketOperations as _, PutObjectCommitBarrier, PutObjectCommitPause,
isolated_store_over_temp_disks,
};
use super::*; use super::*;
use crate::on_demand_migration::source_client::SourceObject; use crate::bucket::metadata_sys::test_support::isolated_store_over_temp_disks;
use crate::on_demand_migration::sys::PullError; use crate::bucket::on_demand_migration::source_client::SourceObject;
use crate::bucket::on_demand_migration::sys::PullError;
use std::collections::{BTreeSet, HashSet}; use std::collections::{BTreeSet, HashSet};
use std::sync::atomic::AtomicBool; use std::sync::atomic::AtomicBool;
@@ -1680,7 +1632,6 @@ mod tests {
/// Scripted source + local store + queue with a controllable report path. /// Scripted source + local store + queue with a controllable report path.
struct MockContext { struct MockContext {
incarnation_id: Mutex<Option<Uuid>>,
objects: Vec<SourceObject>, objects: Vec<SourceObject>,
page_size: usize, page_size: usize,
local: Mutex<HashMap<String, LocalBackfillObject>>, local: Mutex<HashMap<String, LocalBackfillObject>>,
@@ -1689,7 +1640,6 @@ mod tests {
queue_capacity: usize, queue_capacity: usize,
pending: Mutex<Vec<(String, oneshot::Sender<QueuedPullOutcome>)>>, pending: Mutex<Vec<(String, oneshot::Sender<QueuedPullOutcome>)>>,
fail_keys: HashSet<String>, fail_keys: HashSet<String>,
coalesced: bool,
auto_complete: AtomicBool, auto_complete: AtomicBool,
cancel: CancellationToken, cancel: CancellationToken,
config_updated_at: Mutex<Option<OffsetDateTime>>, config_updated_at: Mutex<Option<OffsetDateTime>>,
@@ -1709,7 +1659,6 @@ mod tests {
}) })
.collect(); .collect();
Arc::new(Self { Arc::new(Self {
incarnation_id: Mutex::new(None),
objects, objects,
page_size, page_size,
local: Mutex::new(HashMap::new()), local: Mutex::new(HashMap::new()),
@@ -1718,7 +1667,6 @@ mod tests {
queue_capacity: usize::MAX, queue_capacity: usize::MAX,
pending: Mutex::new(Vec::new()), pending: Mutex::new(Vec::new()),
fail_keys: HashSet::new(), fail_keys: HashSet::new(),
coalesced: false,
auto_complete: AtomicBool::new(true), auto_complete: AtomicBool::new(true),
cancel: CancellationToken::new(), cancel: CancellationToken::new(),
config_updated_at: Mutex::new(Some(ts(1_700_000_000))), config_updated_at: Mutex::new(Some(ts(1_700_000_000))),
@@ -1744,10 +1692,6 @@ mod tests {
#[async_trait] #[async_trait]
impl BackfillContext for MockContext { impl BackfillContext for MockContext {
fn incarnation_id(&self) -> Uuid {
self.incarnation_id.lock().expect("test bucket initialized")
}
async fn list_page(&self, prefix: Option<&str>, token: Option<&str>, max_keys: i32) -> Result<SourcePage, SourceError> { async fn list_page(&self, prefix: Option<&str>, token: Option<&str>, max_keys: i32) -> Result<SourcePage, SourceError> {
if let Some(err) = self.list_error.lock().take() { if let Some(err) = self.list_error.lock().take() {
return Err(err); return Err(err);
@@ -1802,12 +1746,7 @@ mod tests {
} else { } else {
self.pending.lock().push((key.to_string(), tx)); self.pending.lock().push((key.to_string(), tx));
} }
let outcome = if self.coalesced { (EnqueueOutcome::Enqueued, Some(rx))
EnqueueOutcome::Coalesced
} else {
EnqueueOutcome::Enqueued
};
(outcome, Some(futures::FutureExt::shared(rx)))
} }
fn cancel_token(&self) -> CancellationToken { fn cancel_token(&self) -> CancellationToken {
@@ -1840,17 +1779,12 @@ mod tests {
context: Arc<MockContext>, context: Arc<MockContext>,
) -> (Vec<tempfile::TempDir>, Arc<ECStore>, Arc<BackfillRunner>) { ) -> (Vec<tempfile::TempDir>, Arc<ECStore>, Arc<BackfillRunner>) {
let (dirs, store) = isolated_store_over_temp_disks().await; let (dirs, store) = isolated_store_over_temp_disks().await;
super::super::storage_api::test_support::init_bucket_metadata_sys(Arc::clone(&store), Vec::new()).await; // The isolated store has no bucket metadata system; the checkpoint
store // only needs the bucket's directory under the metadata volume.
.make_bucket(bucket, &Default::default()) for dir in &dirs {
.await std::fs::create_dir_all(dir.path().join(RUSTFS_META_BUCKET).join(BUCKET_META_PREFIX).join(bucket))
.expect("create test bucket"); .expect("test bucket metadata directory");
*context.incarnation_id.lock() = Some( }
store
.bucket_incarnation_id_from_disk(bucket)
.await
.expect("test bucket identity"),
);
let runner = runner_on(node, bucket, context, Arc::clone(&store)); let runner = runner_on(node, bucket, context, Arc::clone(&store));
(dirs, store, runner) (dirs, store, runner)
} }
@@ -1860,129 +1794,6 @@ mod tests {
BackfillRunner::new(store, node, Arc::new(contexts)) BackfillRunner::new(store, node, Arc::new(contexts))
} }
#[tokio::test]
async fn cancelled_checkpoint_waiter_keeps_bucket_fenced_until_commit_finishes() {
for (suffix, pause) in [
("before", PutObjectCommitPause::BeforeQuotaRename),
("after", PutObjectCommitPause::AfterRenameQuorum),
] {
let bucket = format!("backfill-cancel-tail-{suffix}");
let context = MockContext::new(0, 1);
let (_dirs, store, _runner) = runner_with("node-a", &bucket, Arc::clone(&context)).await;
let original_incarnation = context.incarnation_id();
let checkpoint = BackfillCheckpoint::new(&BackfillRequest::default(), ts(1_700_000_000), "node-a", ts(1_700_000_001));
let barrier = PutObjectCommitBarrier::install(RUSTFS_META_BUCKET, &checkpoint_path(&bucket), pause);
let writer_api = Arc::clone(&store);
let writer_bucket = bucket.clone();
let waiter = tokio::spawn(async move {
write_checkpoint(&writer_api, &writer_bucket, original_incarnation, &checkpoint, None).await
});
barrier.wait_until_paused().await;
waiter.abort();
assert!(waiter.await.expect_err("caller aborted").is_cancelled());
let lifecycle_lock = store
.new_ns_lock(&bucket, BUCKET_LIFECYCLE_LOCK_OBJECT)
.await
.expect("lifecycle lock");
{
let mut probe = Box::pin(lifecycle_lock.get_write_lock(Duration::from_secs(1)));
assert!(
futures::poll!(probe.as_mut()).is_pending(),
"lifecycle writer must first try to acquire the lock"
);
assert!(
tokio::time::timeout(Duration::from_millis(100), probe.as_mut())
.await
.is_err(),
"the checkpoint owner must retain the user bucket lifecycle read lock after caller cancellation"
);
}
let delete_api = Arc::clone(&store);
let delete_bucket = bucket.clone();
let mut deletion = tokio::spawn(async move { delete_api.delete_bucket(&delete_bucket, &Default::default()).await });
assert!(
tokio::time::timeout(Duration::from_millis(100), &mut deletion).await.is_err(),
"DeleteBucket must wait for the checkpoint owner after its caller aborts"
);
barrier.release();
tokio::time::timeout(Duration::from_secs(10), deletion)
.await
.expect("commit must drain and release its lifecycle guard")
.expect("delete task")
.expect("delete original bucket");
store
.make_bucket(&bucket, &Default::default())
.await
.expect("recreate bucket");
assert_ne!(
original_incarnation,
store.bucket_incarnation_id_from_disk(&bucket).await.expect("new identity")
);
assert!(
read_checkpoint(&store, &bucket)
.await
.expect("read recreated bucket")
.is_none(),
"no old checkpoint may outlive bucket deletion"
);
}
}
#[tokio::test]
async fn stale_checkpoint_writer_cannot_resurrect_or_overwrite_a_recreated_bucket() {
let bucket = "backfill-incarnation";
let context = MockContext::new(0, 1);
let (_dirs, store, _runner) = runner_with("node-a", bucket, Arc::clone(&context)).await;
let old_incarnation = context.incarnation_id();
let old = BackfillCheckpoint::new(&BackfillRequest::default(), ts(1_700_000_000), "node-a", ts(1_700_000_001));
let old_etag = write_checkpoint(&store, bucket, old_incarnation, &old, None)
.await
.expect("old checkpoint");
store
.delete_bucket(bucket, &Default::default())
.await
.expect("delete original bucket");
store.make_bucket(bucket, &Default::default()).await.expect("recreate bucket");
let current_incarnation = store.bucket_incarnation_id_from_disk(bucket).await.expect("new identity");
assert_ne!(old_incarnation, current_incarnation);
assert!(
read_checkpoint(&store, bucket)
.await
.expect("read after recreation")
.is_none()
);
for expected_etag in [None, Some(old_etag.as_str())] {
let error = write_checkpoint(&store, bucket, old_incarnation, &old, expected_etag)
.await
.expect_err("stale writer rejected");
assert!(matches!(error, BackfillError::Storage(StorageError::BucketNotFound(_))));
}
assert!(
read_checkpoint(&store, bucket)
.await
.expect("stale writer left no checkpoint")
.is_none()
);
let current = BackfillCheckpoint::new(&BackfillRequest::default(), ts(1_700_000_000), "node-b", ts(1_700_000_002));
let current_etag = write_checkpoint(&store, bucket, current_incarnation, &current, None)
.await
.expect("current checkpoint");
let error = write_checkpoint(&store, bucket, old_incarnation, &old, Some(&current_etag))
.await
.expect_err("old identity cannot overwrite a matching ETag");
assert!(matches!(error, BackfillError::Storage(StorageError::BucketNotFound(_))));
let stored = read_checkpoint(&store, bucket)
.await
.expect("read current checkpoint")
.expect("current checkpoint remains");
assert_eq!(stored.etag, current_etag);
assert_eq!(stored.checkpoint, current);
}
#[tokio::test] #[tokio::test]
async fn full_backfill_lists_pages_and_counts_every_key() { async fn full_backfill_lists_pages_and_counts_every_key() {
let bucket = "backfill-full"; let bucket = "backfill-full";
@@ -2101,7 +1912,7 @@ mod tests {
#[tokio::test] #[tokio::test]
async fn failed_pulls_are_counted_hashed_and_finish_with_failures() { async fn failed_pulls_are_counted_hashed_and_finish_with_failures() {
let bucket = "backfill-failed"; let bucket = "backfill-failed";
let mut context = MockContext::new(5, 2); let mut context = MockContext::new(5, 1000);
Arc::get_mut(&mut context) Arc::get_mut(&mut context)
.expect("unshared") .expect("unshared")
.fail_keys .fail_keys
@@ -2116,52 +1927,12 @@ mod tests {
.checkpoint; .checkpoint;
assert_eq!(cp.state, BackfillState::CompletedWithFailures); assert_eq!(cp.state, BackfillState::CompletedWithFailures);
assert_eq!((cp.pulled, cp.failed), (4, 1)); assert_eq!((cp.pulled, cp.failed), (4, 1));
assert_eq!(cp.continuation_token.as_deref(), Some("2"), "retain the first failed page for recovery");
assert_eq!(cp.failed_keys, vec![key_hash("k/00002")]); assert_eq!(cp.failed_keys, vec![key_hash("k/00002")]);
let last = cp.last_error.expect("last error"); let last = cp.last_error.expect("last error");
assert_eq!(last.class, "local_write"); assert_eq!(last.class, "local_write");
assert_eq!(last.key_hash.as_deref(), Some(key_hash("k/00002").as_str())); assert_eq!(last.key_hash.as_deref(), Some(key_hash("k/00002").as_str()));
} }
#[tokio::test]
async fn coalesced_pulls_block_the_checkpoint_and_report_failures() {
let bucket = "backfill-coalesced";
let mut context = MockContext::new(1, 1);
{
let ctx = Arc::get_mut(&mut context).expect("unshared");
ctx.coalesced = true;
ctx.auto_complete = AtomicBool::new(false);
ctx.fail_keys.insert("k/00000".to_string());
}
let (_dirs, store, runner) = runner_with("node-a", bucket, Arc::clone(&context)).await;
runner.start(bucket, BackfillRequest::default()).await.expect("start");
tokio::time::timeout(Duration::from_secs(10), async {
while context.pending.lock().is_empty() {
tokio::task::yield_now().await;
}
})
.await
.expect("job enqueued");
assert!(runner.is_running_locally(bucket), "coalescing is not completion");
let cp = read_checkpoint(&store, bucket)
.await
.expect("read")
.expect("checkpoint")
.checkpoint;
assert!(cp.state.is_active());
assert!(cp.continuation_token.is_none());
context.complete_pending();
runner.wait_until_idle(bucket).await;
let cp = read_checkpoint(&store, bucket)
.await
.expect("read")
.expect("checkpoint")
.checkpoint;
assert_eq!(cp.state, BackfillState::CompletedWithFailures);
assert_eq!((cp.enqueued, cp.pulled, cp.failed), (1, 0, 1));
assert_eq!(cp.failed_keys, vec![key_hash("k/00000")]);
}
#[tokio::test] #[tokio::test]
async fn listing_failure_marks_the_job_failed_with_the_error_class() { async fn listing_failure_marks_the_job_failed_with_the_error_class() {
let bucket = "backfill-list-error"; let bucket = "backfill-list-error";
@@ -2322,7 +2093,7 @@ mod tests {
node: "node-a".to_string(), node: "node-a".to_string(),
lease_until: now - Duration::from_secs(120), lease_until: now - Duration::from_secs(120),
}); });
let etag = write_checkpoint(&store, bucket, context.incarnation_id(), &crashed, None) let etag = write_checkpoint(&store, bucket, &crashed, None)
.await .await
.expect("seed checkpoint"); .expect("seed checkpoint");
@@ -2333,7 +2104,7 @@ mod tests {
lease_until: now + Duration::from_secs(60), lease_until: now + Duration::from_secs(60),
}); });
live.updated_at = now; live.updated_at = now;
let etag = write_checkpoint(&store, bucket, context.incarnation_id(), &live, Some(&etag)) let etag = write_checkpoint(&store, bucket, &live, Some(&etag))
.await .await
.expect("live lease"); .expect("live lease");
assert_eq!(runner.recover_once().await.taken_over, 0, "unexpired lease must not be taken over"); assert_eq!(runner.recover_once().await.taken_over, 0, "unexpired lease must not be taken over");
@@ -2350,7 +2121,7 @@ mod tests {
lease_until: now - Duration::from_secs(1), lease_until: now - Duration::from_secs(1),
}); });
expired.updated_at = now + Duration::from_millis(1); expired.updated_at = now + Duration::from_millis(1);
write_checkpoint(&store, bucket, context.incarnation_id(), &expired, Some(&etag)) write_checkpoint(&store, bucket, &expired, Some(&etag))
.await .await
.expect("expire lease"); .expect("expire lease");
let stats = runner.recover_once().await; let stats = runner.recover_once().await;
@@ -2374,68 +2145,6 @@ mod tests {
assert_eq!(runner.recover_once().await.taken_over, 0, "a finished job is not recovered"); assert_eq!(runner.recover_once().await.taken_over, 0, "a finished job is not recovered");
} }
#[tokio::test]
async fn recovery_advances_past_historical_failures_but_pins_new_failures() {
let bucket = "backfill-takeover-failed";
let mut context = MockContext::new(8, 2);
{
let ctx = Arc::get_mut(&mut context).expect("unshared");
ctx.auto_complete = AtomicBool::new(false);
ctx.fail_keys.insert("k/00004".to_string());
}
let (_dirs, store, runner) = runner_with("node-b", bucket, Arc::clone(&context)).await;
let crashed_at = OffsetDateTime::now_utc() - Duration::from_secs(300);
let mut crashed = BackfillCheckpoint::new(&BackfillRequest::default(), ts(1_700_000_000), "node-a", crashed_at);
crashed.continuation_token = Some("2".to_string());
crashed.failed = 1;
crashed.record_failure("local_write", Some("k/00002"), crashed_at);
write_checkpoint(&store, bucket, context.incarnation_id(), &crashed, None)
.await
.expect("seed failed page with an expired lease");
assert_eq!(runner.recover_once().await.taken_over, 1);
for (page_start, durable_token, failures) in [(2, "2", 1), (4, "4", 1), (6, "4", 2)] {
tokio::time::timeout(Duration::from_secs(10), async {
loop {
if context.pending.lock().len() == 2 {
break;
}
tokio::task::yield_now().await;
}
})
.await
.expect("resumed page enqueued before its reports complete");
assert_eq!(
context.pending.lock().iter().map(|(key, _)| key.clone()).collect::<Vec<_>>(),
vec![format!("k/{page_start:05}"), format!("k/{:05}", page_start + 1)]
);
let cp = read_checkpoint(&store, bucket)
.await
.expect("read persisted page boundary")
.expect("checkpoint")
.checkpoint;
assert_eq!(cp.job_id, crashed.job_id);
assert_eq!(cp.owner.as_ref().map(|owner| owner.node.as_str()), Some("node-b"));
assert_eq!(cp.continuation_token.as_deref(), Some(durable_token));
assert_eq!(cp.failed, failures);
context.complete_pending();
}
runner.wait_until_idle(bucket).await;
let cp = read_checkpoint(&store, bucket)
.await
.expect("read completed checkpoint")
.expect("checkpoint")
.checkpoint;
assert_eq!(cp.state, BackfillState::CompletedWithFailures);
assert_eq!((cp.pulled, cp.failed), (5, 2));
assert_eq!(cp.continuation_token.as_deref(), Some("4"));
assert_eq!(cp.failed_keys, vec![key_hash("k/00002"), key_hash("k/00004")]);
assert_eq!(
context.list_requests.lock().as_slice(),
&[Some("2".to_string()), Some("4".to_string()), Some("6".to_string())]
);
}
#[tokio::test] #[tokio::test]
async fn recovery_cancels_a_job_whose_config_changed_and_reclaims_own_node_jobs() { async fn recovery_cancels_a_job_whose_config_changed_and_reclaims_own_node_jobs() {
let bucket = "backfill-recovery-config"; let bucket = "backfill-recovery-config";
@@ -2445,9 +2154,7 @@ mod tests {
// Same node name, unexpired lease: only a restart can produce this. // Same node name, unexpired lease: only a restart can produce this.
let own = BackfillCheckpoint::new(&BackfillRequest::default(), ts(1_700_000_000), "node-a", now); let own = BackfillCheckpoint::new(&BackfillRequest::default(), ts(1_700_000_000), "node-a", now);
let etag = write_checkpoint(&store, bucket, context.incarnation_id(), &own, None) let etag = write_checkpoint(&store, bucket, &own, None).await.expect("seed");
.await
.expect("seed");
assert_eq!(runner.recover_once().await.taken_over, 1, "own-node running job is reclaimed at once"); assert_eq!(runner.recover_once().await.taken_over, 1, "own-node running job is reclaimed at once");
runner.wait_until_idle(bucket).await; runner.wait_until_idle(bucket).await;
let cp = read_checkpoint(&store, bucket) let cp = read_checkpoint(&store, bucket)
@@ -2465,7 +2172,7 @@ mod tests {
node: "node-z".to_string(), node: "node-z".to_string(),
lease_until: now - Duration::from_secs(1), lease_until: now - Duration::from_secs(1),
}); });
write_checkpoint(&store, bucket, context.incarnation_id(), &stale, Some(&stored.etag)) write_checkpoint(&store, bucket, &stale, Some(&stored.etag))
.await .await
.expect("seed stale"); .expect("seed stale");
let stats = runner.recover_once().await; let stats = runner.recover_once().await;
@@ -14,44 +14,22 @@
//! Bucket-level On-Demand Migration configuration: wire model (JSON stored //! Bucket-level On-Demand Migration configuration: wire model (JSON stored
//! under `on-demand-migration.json`), pure validation, credential redaction, //! under `on-demand-migration.json`), pure validation, credential redaction,
//! and persisted-config decoding (rustfs/backlog#2148). //! and the publish hook the runtime registers into (rustfs/backlog#2148).
//! //!
//! The persisted blob is not encrypted; it shares the trust boundary of //! The persisted blob is not encrypted; it shares the trust boundary of
//! `bucket-targets.json` and `tier-config.bin`. //! `bucket-targets.json` and `tier-config.bin`.
use serde::{Deserialize, Serialize}; use serde::{Deserialize, Serialize};
use std::fmt; use std::fmt;
use std::sync::OnceLock;
use url::Url; use url::Url;
/// Decode bytes only at the service boundary, preserving typed corruption errors.
pub(super) fn decode_stored_config(
stored: Option<(Vec<u8>, time::OffsetDateTime)>,
) -> Result<Option<(OnDemandMigrationConfig, time::OffsetDateTime)>, super::storage_api::StorageError> {
stored
.map(|(bytes, updated_at)| {
OnDemandMigrationConfig::from_json(&bytes)
.map(|config| (config, updated_at))
.map_err(super::storage_api::StorageError::other)
})
.transpose()
}
pub(crate) async fn get_config(
bucket: &str,
) -> Result<Option<(OnDemandMigrationConfig, time::OffsetDateTime)>, super::storage_api::StorageError> {
decode_stored_config(super::storage_api::get_on_demand_migration_config(bucket).await?)
}
/// The only wire version this build reads and writes. /// The only wire version this build reads and writes.
pub const ON_DEMAND_MIGRATION_CONFIG_VERSION: u32 = 1; pub const ON_DEMAND_MIGRATION_CONFIG_VERSION: u32 = 1;
const REDACTED: &str = "REDACTED"; const REDACTED: &str = "REDACTED";
const AUTO_REGION: &str = "auto"; const AUTO_REGION: &str = "auto";
const AUTO_REGION_FALLBACK: &str = "us-east-1"; const AUTO_REGION_FALLBACK: &str = "us-east-1";
/// Public Azure Blob host suffix; the account name is the first label.
pub const AZURE_BLOB_SUFFIX: &str = "blob.core.windows.net";
/// Public Google Cloud Storage endpoint for the native provider.
pub const GCS_DEFAULT_ENDPOINT: &str = "https://storage.googleapis.com";
const KIB: u64 = 1024; const KIB: u64 = 1024;
const MIB: u64 = 1024 * KIB; const MIB: u64 = 1024 * KIB;
@@ -97,25 +75,14 @@ pub struct SourceConfig {
pub bucket: String, pub bucket: String,
#[serde(default)] #[serde(default)]
pub path_style: PathStyle, pub path_style: PathStyle,
/// `None` means anonymous access to a public source bucket. Only the /// `None` means anonymous access to a public source bucket.
/// SigV4 providers read it; `azure` and `gcs_native` carry their own
/// credentials in `azure` / `gcs`.
#[serde(default)] #[serde(default)]
pub credentials: Option<SourceCredentials>, pub credentials: Option<SourceCredentials>,
#[serde(default)] #[serde(default)]
pub tls: TlsConfig, pub tls: TlsConfig,
/// Required for [`Provider::Azure`] and rejected for every other
/// provider.
#[serde(default)]
pub azure: Option<AzureSourceConfig>,
/// Required for [`Provider::GcsNative`] and rejected for every other
/// provider. [`Provider::Gcs`] keeps using `credentials` because it
/// speaks the S3 interoperability API.
#[serde(default)]
pub gcs: Option<GcsSourceConfig>,
} }
/// Source vendor family. /// Source vendor family. `azure` is deliberately absent from this version.
#[derive(Debug, Clone, Copy, PartialEq, Eq, Serialize, Deserialize)] #[derive(Debug, Clone, Copy, PartialEq, Eq, Serialize, Deserialize)]
#[serde(rename_all = "lowercase")] #[serde(rename_all = "lowercase")]
pub enum Provider { pub enum Provider {
@@ -127,12 +94,6 @@ pub enum Provider {
R2, R2,
/// GCS XML interoperability API with HMAC keys. /// GCS XML interoperability API with HMAC keys.
Gcs, Gcs,
/// Native Azure Blob service; parameters in `source.azure`.
Azure,
/// Native GCS JSON API with a service-account key; parameters in
/// `source.gcs`.
#[serde(rename = "gcs_native")]
GcsNative,
} }
impl Provider { impl Provider {
@@ -144,22 +105,13 @@ impl Provider {
Provider::Rustfs => "rustfs", Provider::Rustfs => "rustfs",
Provider::R2 => "r2", Provider::R2 => "r2",
Provider::Gcs => "gcs", Provider::Gcs => "gcs",
Provider::Azure => "azure",
Provider::GcsNative => "gcs_native",
} }
} }
/// Providers that do not speak S3 and therefore ignore `region`,
/// `path_style` and `credentials`.
pub fn is_native(&self) -> bool {
matches!(self, Provider::Azure | Provider::GcsNative)
}
/// Providers whose SDKs accept `region = "auto"`; RustFS maps it to /// Providers whose SDKs accept `region = "auto"`; RustFS maps it to
/// `us-east-1` for signing. The native providers never sign with a /// `us-east-1` for signing.
/// region, so they accept it as well.
fn accepts_auto_region(&self) -> bool { fn accepts_auto_region(&self) -> bool {
matches!(self, Provider::R2 | Provider::Minio | Provider::Rustfs) || self.is_native() matches!(self, Provider::R2 | Provider::Minio | Provider::Rustfs)
} }
} }
@@ -212,73 +164,6 @@ impl fmt::Debug for SourceCredentials {
} }
} }
/// Native Azure Blob source parameters. The container is `source.bucket`,
/// so a config never carries two names for the same container. Exactly one
/// of `account_key` and `sas_token` must be set: the account key signs with
/// Shared Key, the SAS token is appended to every request URL.
#[derive(Clone, PartialEq, Eq, Serialize, Deserialize)]
#[serde(deny_unknown_fields)]
pub struct AzureSourceConfig {
/// Storage account name; also derives the default `blob.core.windows.net`
/// endpoint when `source.endpoint` is absent.
pub account: String,
/// Base64 shared key of the storage account.
#[serde(default)]
pub account_key: Option<String>,
/// SAS query string without the leading `?`.
#[serde(default)]
pub sas_token: Option<String>,
}
impl AzureSourceConfig {
/// A copy safe to return to admin clients or log: both secrets are
/// replaced by `REDACTED`, and whether each is set stays visible.
pub fn redacted(&self) -> Self {
Self {
account: self.account.clone(),
account_key: self.account_key.as_ref().map(|_| REDACTED.to_string()),
sas_token: self.sas_token.as_ref().map(|_| REDACTED.to_string()),
}
}
}
impl fmt::Debug for AzureSourceConfig {
fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result {
f.debug_struct("AzureSourceConfig")
.field("account", &self.account)
.field("account_key", &self.account_key.as_ref().map(|_| REDACTED))
.field("sas_token", &self.sas_token.as_ref().map(|_| REDACTED))
.finish()
}
}
/// Native Google Cloud Storage source parameters. The bucket is
/// `source.bucket`; only the service-account key lives here.
#[derive(Clone, PartialEq, Eq, Serialize, Deserialize)]
#[serde(deny_unknown_fields)]
pub struct GcsSourceConfig {
/// Service-account key JSON, verbatim as downloaded from Google Cloud.
pub service_account_json: String,
}
impl GcsSourceConfig {
/// A copy safe to return to admin clients or log: the whole key JSON is
/// a secret (it embeds the private key), so it is replaced wholesale.
pub fn redacted(&self) -> Self {
Self {
service_account_json: REDACTED.to_string(),
}
}
}
impl fmt::Debug for GcsSourceConfig {
fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result {
f.debug_struct("GcsSourceConfig")
.field("service_account_json", &REDACTED)
.finish()
}
}
#[derive(Debug, Clone, PartialEq, Eq, Default, Serialize, Deserialize)] #[derive(Debug, Clone, PartialEq, Eq, Default, Serialize, Deserialize)]
#[serde(deny_unknown_fields)] #[serde(deny_unknown_fields)]
pub struct TlsConfig { pub struct TlsConfig {
@@ -469,14 +354,6 @@ pub enum OnDemandMigrationConfigError {
InvalidBucket(&'static str), InvalidBucket(&'static str),
#[error("source credentials field {0} must not be empty")] #[error("source credentials field {0} must not be empty")]
EmptyCredential(&'static str), EmptyCredential(&'static str),
#[error("source.{0} is required for provider {1}")]
MissingProviderBlock(&'static str, Provider),
#[error("source.{0} is not valid for provider {1}")]
UnexpectedProviderBlock(&'static str, Provider),
/// Carries only the reason: the block holds account keys, SAS tokens and
/// service-account JSON, so no value of it is ever echoed.
#[error("source.{0} is invalid: {1}")]
InvalidProviderBlock(&'static str, &'static str),
#[error("source tls.ca_cert_pem is not a PEM certificate")] #[error("source tls.ca_cert_pem is not a PEM certificate")]
InvalidCaCert, InvalidCaCert,
#[error("filter.{0} must be null or a non-empty string")] #[error("filter.{0} must be null or a non-empty string")]
@@ -511,8 +388,6 @@ impl OnDemandMigrationConfig {
pub fn redacted(&self) -> Self { pub fn redacted(&self) -> Self {
let mut copy = self.clone(); let mut copy = self.clone();
copy.source.credentials = self.source.credentials.as_ref().map(SourceCredentials::redacted); copy.source.credentials = self.source.credentials.as_ref().map(SourceCredentials::redacted);
copy.source.azure = self.source.azure.as_ref().map(AzureSourceConfig::redacted);
copy.source.gcs = self.source.gcs.as_ref().map(GcsSourceConfig::redacted);
copy copy
} }
@@ -558,12 +433,6 @@ impl SourceConfig {
match (&self.endpoint, self.provider) { match (&self.endpoint, self.provider) {
(Some(endpoint), _) => endpoint.clone(), (Some(endpoint), _) => endpoint.clone(),
(None, Provider::Aws) => format!("https://s3.{}.amazonaws.com", self.region), (None, Provider::Aws) => format!("https://s3.{}.amazonaws.com", self.region),
(None, Provider::Azure) => self
.azure
.as_ref()
.map(|azure| format!("https://{}.{AZURE_BLOB_SUFFIX}", azure.account))
.unwrap_or_default(),
(None, Provider::GcsNative) => GCS_DEFAULT_ENDPOINT.to_string(),
(None, _) => String::new(), (None, _) => String::new(),
} }
} }
@@ -579,8 +448,6 @@ impl SourceConfig {
} }
fn validate(&self) -> Result<(), OnDemandMigrationConfigError> { fn validate(&self) -> Result<(), OnDemandMigrationConfigError> {
self.validate_provider_block()?;
if self.region.is_empty() { if self.region.is_empty() {
return Err(OnDemandMigrationConfigError::EmptyRegion); return Err(OnDemandMigrationConfigError::EmptyRegion);
} }
@@ -599,9 +466,6 @@ impl SourceConfig {
)); ));
} }
} }
// Both native providers derive a fixed endpoint; Azure's is built
// from the account name, already checked by `validate_provider_block`.
None if self.provider.is_native() => {}
None => return Err(OnDemandMigrationConfigError::MissingEndpoint(self.provider)), None => return Err(OnDemandMigrationConfigError::MissingEndpoint(self.provider)),
} }
@@ -632,84 +496,6 @@ impl SourceConfig {
Ok(()) Ok(())
} }
/// The provider-specific block must be present for exactly its own
/// provider: a stray `azure` block on an `s3` source would otherwise be
/// accepted, stored, and silently ignored by the client builder.
fn validate_provider_block(&self) -> Result<(), OnDemandMigrationConfigError> {
let missing = OnDemandMigrationConfigError::MissingProviderBlock;
let unexpected = OnDemandMigrationConfigError::UnexpectedProviderBlock;
let invalid = OnDemandMigrationConfigError::InvalidProviderBlock;
if self.provider != Provider::Azure && self.azure.is_some() {
return Err(unexpected("azure", self.provider));
}
if self.provider != Provider::GcsNative && self.gcs.is_some() {
return Err(unexpected("gcs", self.provider));
}
match self.provider {
Provider::Azure => {
let azure = self.azure.as_ref().ok_or(missing("azure", self.provider))?;
if azure.account.is_empty() {
return Err(invalid("azure", "account must not be empty"));
}
// The account feeds a hostname when the endpoint is derived:
// keep it to label characters so it cannot rewrite the host.
if !azure.account.bytes().all(|b| b.is_ascii_alphanumeric() || b == b'-') {
return Err(invalid("azure", "account contains characters outside [A-Za-z0-9-]"));
}
match (azure.account_key.as_deref(), azure.sas_token.as_deref()) {
(Some(_), Some(_)) => return Err(invalid("azure", "account_key and sas_token are mutually exclusive")),
(None, None) => return Err(invalid("azure", "one of account_key and sas_token is required")),
(Some(key), None) => {
if key.is_empty() {
return Err(invalid("azure", "account_key must not be empty"));
}
// Decoded here so a mistyped key fails at the admin
// boundary instead of on the first source request.
if base64_simd::STANDARD.decode_to_vec(key.as_bytes()).is_err() {
return Err(invalid("azure", "account_key is not base64"));
}
}
(None, Some(sas)) => {
if sas.is_empty() {
return Err(invalid("azure", "sas_token must not be empty"));
}
if sas.starts_with('?') {
return Err(invalid("azure", "sas_token must not start with '?'"));
}
if sas.chars().any(char::is_whitespace) {
return Err(invalid("azure", "sas_token must not contain whitespace"));
}
}
}
}
Provider::GcsNative => {
let gcs = self.gcs.as_ref().ok_or(missing("gcs", self.provider))?;
let key: serde_json::Value = serde_json::from_str(&gcs.service_account_json)
.map_err(|_| invalid("gcs", "service_account_json is not valid JSON"))?;
let Some(object) = key.as_object() else {
return Err(invalid("gcs", "service_account_json is not a JSON object"));
};
if object.get("type").and_then(serde_json::Value::as_str) != Some("service_account") {
return Err(invalid("gcs", "service_account_json is not a service_account key"));
}
for field in ["client_email", "private_key"] {
if object
.get(field)
.and_then(serde_json::Value::as_str)
.is_none_or(str::is_empty)
{
return Err(invalid("gcs", "service_account_json is missing client_email or private_key"));
}
}
}
Provider::S3 | Provider::Aws | Provider::Minio | Provider::Rustfs | Provider::R2 | Provider::Gcs => {}
}
Ok(())
}
} }
fn validate_endpoint(endpoint: &str) -> Result<(), OnDemandMigrationConfigError> { fn validate_endpoint(endpoint: &str) -> Result<(), OnDemandMigrationConfigError> {
@@ -804,6 +590,16 @@ impl EndpointKey {
} }
} }
/// Signature of the runtime publish hook: called with the bucket name and
/// its parsed config (`None` when absent, cleared, or unreadable) every time
/// the bucket's metadata is installed into or removed from the cache.
pub type ConfigPublishHook = Box<dyn Fn(&str, Option<&OnDemandMigrationConfig>) + Send + Sync>;
/// Registration point for the runtime (`OnDemandMigrationSys`). Until it is
/// set, metadata publishes are no-ops for ODM, so this crate carries no
/// runtime dependency and the config layer stays inert.
pub static ON_DEMAND_MIGRATION_CONFIG_HOOK: OnceLock<ConfigPublishHook> = OnceLock::new();
#[cfg(test)] #[cfg(test)]
mod tests { mod tests {
use super::*; use super::*;
@@ -903,15 +699,7 @@ mod tests {
), ),
( (
"provider enum", "provider enum",
r#"{"source":{"provider":"swift","endpoint":"https://h","region":"r","bucket":"b"}}"#, r#"{"source":{"provider":"azure","endpoint":"https://h","region":"r","bucket":"b"}}"#,
),
(
"azure block",
r#"{"source":{"provider":"azure","region":"auto","bucket":"b","azure":{"account":"acct","account_key":"a2V5","extra":1}}}"#,
),
(
"gcs block",
r#"{"source":{"provider":"gcs_native","region":"auto","bucket":"b","gcs":{"service_account_json":"{}","extra":1}}}"#,
), ),
] { ] {
let err = OnDemandMigrationConfig::from_json(json.as_bytes()).expect_err(label); let err = OnDemandMigrationConfig::from_json(json.as_bytes()).expect_err(label);
@@ -1032,201 +820,9 @@ mod tests {
"{provider}" "{provider}"
); );
} }
// The native providers never sign with a region, so "auto" is the
// honest value to write for them.
for cfg in [azure_cfg(), gcs_native_cfg()] {
assert_eq!(cfg.source.region, "auto");
cfg.validate(empty_ctx())
.unwrap_or_else(|err| panic!("{}: {err}", cfg.source.provider));
}
assert_eq!(sample().source.effective_region(), "us-west-1"); assert_eq!(sample().source.effective_region(), "us-west-1");
} }
const SERVICE_ACCOUNT_JSON: &str = r#"{"type":"service_account","project_id":"p","client_email":"a@b.iam.gserviceaccount.com","private_key":"-----BEGIN PRIVATE KEY-----\nsecret\n-----END PRIVATE KEY-----"}"#;
fn azure_cfg() -> OnDemandMigrationConfig {
let mut cfg = sample();
cfg.source.provider = Provider::Azure;
cfg.source.endpoint = None;
cfg.source.region = "auto".to_string();
cfg.source.credentials = None;
cfg.source.azure = Some(AzureSourceConfig {
account: "legacyaccount".to_string(),
account_key: Some("c2VjcmV0LWtleQ==".to_string()),
sas_token: None,
});
cfg
}
fn gcs_native_cfg() -> OnDemandMigrationConfig {
let mut cfg = sample();
cfg.source.provider = Provider::GcsNative;
cfg.source.endpoint = None;
cfg.source.region = "auto".to_string();
cfg.source.credentials = None;
cfg.source.gcs = Some(GcsSourceConfig {
service_account_json: SERVICE_ACCOUNT_JSON.to_string(),
});
cfg
}
#[test]
fn native_providers_derive_their_endpoint_and_round_trip_on_the_wire() {
let azure = azure_cfg();
assert_eq!(azure.source.effective_endpoint(), "https://legacyaccount.blob.core.windows.net");
let gcs = gcs_native_cfg();
assert_eq!(gcs.source.effective_endpoint(), "https://storage.googleapis.com");
for cfg in [azure_cfg(), gcs_native_cfg()] {
let json = cfg.to_json().expect("config must serialize");
assert_eq!(OnDemandMigrationConfig::from_json(&json).expect("config must parse"), cfg);
}
// The wire labels are part of the admin contract.
assert!(
String::from_utf8(azure_cfg().to_json().expect("json"))
.expect("utf8")
.contains(r#""provider":"azure""#)
);
assert!(
String::from_utf8(gcs_native_cfg().to_json().expect("json"))
.expect("utf8")
.contains(r#""provider":"gcs_native""#)
);
}
#[test]
fn an_explicit_endpoint_overrides_the_derived_native_one() {
// Azurite and fake-gcs-server are addressed this way.
let mut cfg = azure_cfg();
cfg.source.endpoint = Some("http://azurite.example.com:10000".to_string());
cfg.validate(empty_ctx()).expect("an explicit native endpoint is allowed");
assert_eq!(cfg.source.effective_endpoint(), "http://azurite.example.com:10000");
cfg.source.endpoint = Some("http://azurite.example.com:10000/devstoreaccount1".to_string());
assert!(
matches!(cfg.validate(empty_ctx()), Err(OnDemandMigrationConfigError::InvalidEndpoint(_))),
"a native endpoint is still an origin"
);
}
#[test]
fn a_provider_block_belongs_to_exactly_its_own_provider() {
let mut cfg = sample();
cfg.source.azure = azure_cfg().source.azure;
assert_eq!(
cfg.validate(empty_ctx()),
Err(OnDemandMigrationConfigError::UnexpectedProviderBlock("azure", Provider::S3))
);
let mut cfg = sample();
cfg.source.gcs = gcs_native_cfg().source.gcs;
assert_eq!(
cfg.validate(empty_ctx()),
Err(OnDemandMigrationConfigError::UnexpectedProviderBlock("gcs", Provider::S3))
);
let mut cfg = azure_cfg();
cfg.source.azure = None;
assert_eq!(
cfg.validate(empty_ctx()),
Err(OnDemandMigrationConfigError::MissingProviderBlock("azure", Provider::Azure))
);
let mut cfg = gcs_native_cfg();
cfg.source.gcs = None;
assert_eq!(
cfg.validate(empty_ctx()),
Err(OnDemandMigrationConfigError::MissingProviderBlock("gcs", Provider::GcsNative))
);
}
#[test]
fn azure_block_rules() {
let with = |account: &str, key: Option<&str>, sas: Option<&str>| {
let mut cfg = azure_cfg();
cfg.source.azure = Some(AzureSourceConfig {
account: account.to_string(),
account_key: key.map(str::to_string),
sas_token: sas.map(str::to_string),
});
cfg.validate(empty_ctx())
};
with("legacyaccount", None, Some("sv=2021-08-06&sig=abc%3D")).expect("a SAS token is a complete credential");
with("legacyaccount", Some("c2VjcmV0LWtleQ=="), None).expect("an account key is a complete credential");
for (label, result) in [
("empty account", with("", Some("c2VjcmV0LWtleQ=="), None)),
// The account becomes the first label of the derived hostname.
("account with a dot", with("legacy.account", Some("c2VjcmV0LWtleQ=="), None)),
("account with a slash", with("legacy/account", Some("c2VjcmV0LWtleQ=="), None)),
("no credential", with("legacyaccount", None, None)),
("both credentials", with("legacyaccount", Some("c2VjcmV0LWtleQ=="), Some("sv=1"))),
("empty key", with("legacyaccount", Some(""), None)),
("key that is not base64", with("legacyaccount", Some("not base64!"), None)),
("empty sas", with("legacyaccount", None, Some(""))),
("sas with a leading question mark", with("legacyaccount", None, Some("?sv=1"))),
("sas with whitespace", with("legacyaccount", None, Some("sv=1 &sig=a"))),
] {
assert!(
matches!(result, Err(OnDemandMigrationConfigError::InvalidProviderBlock("azure", _))),
"{label}: {result:?}"
);
}
}
#[test]
fn gcs_native_block_requires_a_usable_service_account_key() {
let with = |json: &str| {
let mut cfg = gcs_native_cfg();
cfg.source.gcs = Some(GcsSourceConfig {
service_account_json: json.to_string(),
});
cfg.validate(empty_ctx())
};
with(SERVICE_ACCOUNT_JSON).expect("a service-account key is accepted");
for (label, json) in [
("empty", ""),
("not json", "not json"),
("not an object", "[]"),
("wrong type", r#"{"type":"authorized_user","client_email":"a@b","private_key":"k"}"#),
("no private key", r#"{"type":"service_account","client_email":"a@b"}"#),
("empty client email", r#"{"type":"service_account","client_email":"","private_key":"k"}"#),
] {
let result = with(json);
assert!(
matches!(result, Err(OnDemandMigrationConfigError::InvalidProviderBlock("gcs", _))),
"{label}: {result:?}"
);
}
}
#[test]
fn native_secrets_never_survive_redaction_or_debug() {
let mut azure = azure_cfg();
azure.source.azure.as_mut().expect("block").sas_token = Some("sv=2021-08-06&sig=top-secret".to_string());
azure.source.azure.as_mut().expect("block").account_key = None;
let gcs = gcs_native_cfg();
for rendered in [
format!("{:?}", azure.redacted()),
format!("{azure:?}"),
String::from_utf8(azure.redacted().to_json().expect("json")).expect("utf8"),
] {
assert!(!rendered.contains("top-secret"), "{rendered}");
assert!(rendered.contains("legacyaccount"), "the account name is not a secret: {rendered}");
}
for rendered in [
format!("{:?}", gcs.redacted()),
format!("{gcs:?}"),
String::from_utf8(gcs.redacted().to_json().expect("json")).expect("utf8"),
] {
assert!(!rendered.contains("PRIVATE KEY-----"), "{rendered}");
assert!(!rendered.contains("gserviceaccount"), "{rendered}");
}
}
#[test] #[test]
fn bucket_rules() { fn bucket_rules() {
let mut cfg = sample(); let mut cfg = sample();
@@ -1509,55 +1105,4 @@ mod tests {
assert!(!rendered.contains("topsecret"), "{rendered}"); assert!(!rendered.contains("topsecret"), "{rendered}");
assert!(!rendered.contains("SK"), "{rendered}"); assert!(!rendered.contains("SK"), "{rendered}");
} }
/// rustfs/backlog#2148: the accessor reports absence as `Ok(None)` and a
/// stored payload it cannot parse as a typed error, never as a default
/// and never as `ConfigNotFound`.
#[tokio::test]
async fn get_on_demand_migration_config_distinguishes_absent_from_corrupt() {
use super::super::storage_api::StorageError as Error;
use super::super::storage_api::test_support::{
BUCKET_ON_DEMAND_MIGRATION_CONFIG, BucketMetadata, BucketMetadataSys, isolated_store_over_temp_disks,
};
use std::sync::Arc;
const ODM_JSON: &[u8] = br#"{"source":{"provider":"minio","endpoint":"https://legacy.example.com:9000","region":"auto","bucket":"legacy-bucket","credentials":{"access_key":"AK","secret_key":"SK"}}}"#;
let (_dirs, ecstore) = isolated_store_over_temp_disks().await;
let sys = BucketMetadataSys::new(ecstore);
let bucket = "odm-accessor";
sys.set(bucket.to_string(), Arc::new(BucketMetadata::new(bucket))).await;
assert_eq!(
decode_stored_config(sys.get_on_demand_migration_config(bucket).await.unwrap()).unwrap(),
None
);
let mut corrupt = BucketMetadata::new(bucket);
corrupt.on_demand_migration_config_json = br#"{"source":{"provider":"s3"},"bogus":1}"#.to_vec();
sys.set(bucket.to_string(), Arc::new(corrupt)).await;
let err = decode_stored_config(sys.get_on_demand_migration_config(bucket).await.unwrap())
.expect_err("corrupt config must not read as a default");
assert_ne!(err, Error::ConfigNotFound, "corruption must not be reported as absence");
let typed = match &err {
Error::Io(io) => io
.get_ref()
.and_then(|source| source.downcast_ref::<OnDemandMigrationConfigError>()),
_ => None,
};
assert!(
matches!(typed, Some(OnDemandMigrationConfigError::Malformed(_))),
"typed parse error must survive the Result boundary, got: {err:?}"
);
let mut valid = BucketMetadata::new(bucket);
valid
.update_config(BUCKET_ON_DEMAND_MIGRATION_CONFIG, ODM_JSON.to_vec())
.unwrap();
let stamped = valid.on_demand_migration_config_updated_at;
sys.set(bucket.to_string(), Arc::new(valid)).await;
let (config, updated_at) = decode_stored_config(sys.get_on_demand_migration_config(bucket).await.unwrap())
.unwrap()
.expect("stored config is returned");
assert_eq!(config, OnDemandMigrationConfig::from_json(ODM_JSON).unwrap());
assert_eq!(updated_at, stamped);
}
} }
@@ -25,21 +25,13 @@ use parking_lot::Mutex;
use serde::{Deserialize, Serialize}; use serde::{Deserialize, Serialize};
use std::time::{Duration, Instant}; use std::time::{Duration, Instant};
/// The continuation-token version used by ordinary progressing pages. /// The only continuation-token envelope version this build reads and writes.
pub const LIST_THROUGH_TOKEN_VERSION: u32 = 1; pub const LIST_THROUGH_TOKEN_VERSION: u32 = 1;
const LIST_THROUGH_PROGRESS_TOKEN_VERSION: u32 = 2;
/// The sixteenth consecutive merged page without a key or new EOF fails.
/// This also bounds legitimate sparse listings; it is not a cycle detector.
pub const MAX_LIST_NO_PROGRESS_PAGES: u8 = 16;
/// Envelope marker. A bucket that is *not* merging hands out the local /// Envelope marker. A bucket that is *not* merging hands out the local
/// listing's own marker, so the decoder needs a positive signal before it /// listing's own marker, so the decoder needs a positive signal before it
/// treats an opaque token as a merged one. /// treats an opaque token as a merged one.
const LIST_THROUGH_TOKEN_TAG: &str = "odm-list"; const LIST_THROUGH_TOKEN_TAG: &str = "odm-list";
// Object keys cannot contain NUL (bucket::utils::is_valid_object_prefix),
// so this framing cannot collide with a local key used as an opaque marker.
const LIST_THROUGH_TOKEN_PREFIX: &str = "\0odm-list:";
/// Pages fetched per side per request: the first page, plus at most one refill /// Pages fetched per side per request: the first page, plus at most one refill
/// when the first one was mostly consumed by the previous page. Two pages of /// when the first one was mostly consumed by the previous page. Two pages of
@@ -94,7 +86,8 @@ pub struct MergePick {
} }
/// The continuation-token envelope. Opaque to clients: it is serialized as /// The continuation-token envelope. Opaque to clients: it is serialized as
/// framed JSON and then base64-encoded by the same helper as a local marker. /// JSON and then base64-encoded by the same helper that encodes a plain local
/// marker, so the wire shape is `base64(json)`.
/// ///
/// A `null` cursor with `done = false` means "list that side from the start"; /// A `null` cursor with `done = false` means "list that side from the start";
/// `done = true` means the side is finished and must not be listed again. /// `done = true` means the side is finished and must not be listed again.
@@ -118,10 +111,6 @@ pub struct ListThroughToken {
/// common prefix compares as itself, never as its members. /// common prefix compares as itself, never as its members.
#[serde(default)] #[serde(default)]
pub last_key: Option<String>, pub last_key: Option<String>,
/// Consecutive empty truncated merged pages, present only in v2 tokens.
/// Ordinary v1 tokens retain their original serialized shape.
#[serde(default, skip_serializing_if = "Option::is_none")]
pub no_progress: Option<u8>,
} }
impl ListThroughToken { impl ListThroughToken {
@@ -134,14 +123,13 @@ impl ListThroughToken {
source: source.token, source: source.token,
source_done: source.done, source_done: source.done,
last_key, last_key,
no_progress: None,
} }
} }
pub fn encode(&self) -> String { pub fn encode(&self) -> String {
// The envelope is built here from owned strings, so serialization // The envelope is built here from owned strings, so serialization
// cannot fail; the fallback keeps the signature infallible. // cannot fail; the fallback keeps the signature infallible.
format!("{LIST_THROUGH_TOKEN_PREFIX}{}", serde_json::to_string(self).unwrap_or_default()) serde_json::to_string(self).unwrap_or_default()
} }
} }
@@ -165,35 +153,24 @@ pub enum ListThroughTokenError {
/// Classifies an already base64-decoded continuation token. /// Classifies an already base64-decoded continuation token.
/// ///
/// Only a framed JSON object is read as a merged token; /// Only a JSON object carrying the envelope marker is read as a merged token;
/// anything else is a local marker, so a bucket that turns `list_through` off /// anything else is a local marker, so a bucket that turns `list_through` off
/// keeps paginating with the tokens it handed out. A token that *is* an /// keeps paginating with the tokens it handed out. A token that *is* an
/// envelope but was tampered with (unknown version, unknown field, truncated /// envelope but was tampered with (unknown version, unknown field, truncated
/// JSON) is an error, never a silent fallback. /// JSON) is an error, never a silent fallback.
pub fn decode_continuation_token(decoded: &str) -> Result<ListThroughCursor, ListThroughTokenError> { pub fn decode_continuation_token(decoded: &str) -> Result<ListThroughCursor, ListThroughTokenError> {
let Some(payload) = decoded.strip_prefix(LIST_THROUGH_TOKEN_PREFIX) else { if !decoded.starts_with('{') {
return Ok(ListThroughCursor::Local(decoded.to_string()));
}
let Ok(value) = serde_json::from_str::<serde_json::Value>(decoded) else {
// Not JSON at all: an object key may legitimately start with '{'.
return Ok(ListThroughCursor::Local(decoded.to_string())); return Ok(ListThroughCursor::Local(decoded.to_string()));
}; };
let value = serde_json::from_str::<serde_json::Value>(payload).map_err(|_| ListThroughTokenError::Malformed)?;
if value.get("t").and_then(serde_json::Value::as_str) != Some(LIST_THROUGH_TOKEN_TAG) { if value.get("t").and_then(serde_json::Value::as_str) != Some(LIST_THROUGH_TOKEN_TAG) {
return Err(ListThroughTokenError::Malformed); return Ok(ListThroughCursor::Local(decoded.to_string()));
} }
match value.get("v").and_then(serde_json::Value::as_u64) { match value.get("v").and_then(serde_json::Value::as_u64) {
Some(version) if version == u64::from(LIST_THROUGH_TOKEN_VERSION) => { Some(version) if version == u64::from(LIST_THROUGH_TOKEN_VERSION) => {}
// v1 readers reject this field even when it is null or zero.
if value.get("no_progress").is_some() {
return Err(ListThroughTokenError::Malformed);
}
}
Some(version) if version == u64::from(LIST_THROUGH_PROGRESS_TOKEN_VERSION) => {
if !value
.get("no_progress")
.and_then(serde_json::Value::as_u64)
.is_some_and(|count| (1..u64::from(MAX_LIST_NO_PROGRESS_PAGES)).contains(&count))
{
return Err(ListThroughTokenError::Malformed);
}
}
Some(version) => return Err(ListThroughTokenError::UnsupportedVersion(version.min(u64::from(u32::MAX)) as u32)), Some(version) => return Err(ListThroughTokenError::UnsupportedVersion(version.min(u64::from(u32::MAX)) as u32)),
None => return Err(ListThroughTokenError::Malformed), None => return Err(ListThroughTokenError::Malformed),
} }
@@ -311,8 +288,6 @@ pub enum ListPageError {
Empty, Empty,
#[error("truncated listing repeats a continuation token")] #[error("truncated listing repeats a continuation token")]
Repeated, Repeated,
#[error("listing exhausted its consecutive no-progress page budget")]
NoProgress(MergeSide),
} }
pub(crate) fn validate_list_page(is_truncated: bool, token: Option<&str>, next_token: Option<&str>) -> Result<(), ListPageError> { pub(crate) fn validate_list_page(is_truncated: bool, token: Option<&str>, next_token: Option<&str>) -> Result<(), ListPageError> {
@@ -377,7 +352,6 @@ pub struct MergeOutcome {
#[derive(Debug)] #[derive(Debug)]
pub struct ListThroughMerger { pub struct ListThroughMerger {
max_keys: usize, max_keys: usize,
no_progress: Option<u8>,
last_key: Option<String>, last_key: Option<String>,
local: SideState, local: SideState,
source: SideState, source: SideState,
@@ -397,7 +371,6 @@ impl ListThroughMerger {
}; };
Self { Self {
max_keys, max_keys,
no_progress: token.and_then(|token| token.no_progress),
last_key, last_key,
local, local,
source, source,
@@ -463,18 +436,13 @@ impl ListThroughMerger {
Ok(()) Ok(())
} }
/// `issue_progress_tokens` allows a v1 chain to start carrying a budget. pub fn finish(self) -> MergeOutcome {
/// An existing v2 budget is always enforced, including on reader-only nodes.
/// Borrowing lets a source failure re-merge the fetched local buffers.
pub fn finish(&self, issue_progress_tokens: bool) -> Result<MergeOutcome, ListPageError> {
let Self { let Self {
max_keys, max_keys,
no_progress,
last_key, last_key,
local, local,
source, source,
} = self; } = self;
let max_keys = *max_keys;
// A side with more pages behind it can only be trusted up to the last // A side with more pages behind it can only be trusted up to the last
// key it handed over: past that horizon the other side's entries could // key it handed over: past that horizon the other side's entries could
@@ -540,44 +508,12 @@ impl ListThroughMerger {
let source_left = !source.disabled && (!source_cursor.done || consumed_source < source.entries.len()); let source_left = !source.disabled && (!source_cursor.done || consumed_source < source.entries.len());
let is_truncated = local_left || source_left; let is_truncated = local_left || source_left;
let reached_eof = (!local.start.done && local_cursor.done) || (!source.start.done && source_cursor.done); let last_key = consumed_key.or(last_key);
let next_no_progress = if !is_truncated || !picks.is_empty() || reached_eof { MergeOutcome {
None
} else if max_keys == 0 {
// A zero-sized request cannot consume entries. Preserve an existing
// budget without spending it or starting a new one.
*no_progress
} else if issue_progress_tokens || no_progress.is_some() {
let count = no_progress.unwrap_or(0).saturating_add(1);
if count >= MAX_LIST_NO_PROGRESS_PAGES {
// An empty truncated side closes the merge horizon. Local
// failure takes precedence; disabling the source cannot fix it.
let side = if local.more && local.entries.is_empty() {
MergeSide::Local
} else if !source.disabled && source.more && source.entries.is_empty() {
MergeSide::Source
} else {
MergeSide::Local
};
return Err(ListPageError::NoProgress(side));
}
Some(count)
} else {
None
};
let last_key = consumed_key.or_else(|| last_key.clone());
Ok(MergeOutcome {
picks, picks,
is_truncated, is_truncated,
next_token: is_truncated.then(|| { next_token: is_truncated.then(|| ListThroughToken::new(local_cursor, source_cursor, last_key)),
let mut token = ListThroughToken::new(local_cursor, source_cursor, last_key); }
if let Some(count) = next_no_progress {
token.v = LIST_THROUGH_PROGRESS_TOKEN_VERSION;
token.no_progress = Some(count);
}
token
}),
})
} }
} }
@@ -705,7 +641,7 @@ mod tests {
.push_page(fetch.side, kept, truncated, next) .push_page(fetch.side, kept, truncated, next)
.expect("reference provider pages must advance"); .expect("reference provider pages must advance");
} }
let outcome = merger.finish(false).expect("valid merge outcome"); let outcome = merger.finish();
assert_eq!(outcome.is_truncated, outcome.next_token.is_some()); assert_eq!(outcome.is_truncated, outcome.next_token.is_some());
if outcome.is_truncated { if outcome.is_truncated {
assert_ne!(outcome.next_token, token, "every truncated merged page must make progress"); assert_ne!(outcome.next_token, token, "every truncated merged page must make progress");
@@ -788,7 +724,7 @@ mod tests {
.push_page(MergeSide::Local, vec![ListEntryKey::object("a")], false, None) .push_page(MergeSide::Local, vec![ListEntryKey::object("a")], false, None)
.expect("local EOF is valid"); .expect("local EOF is valid");
assert_eq!(merger.next_fetch(), None); assert_eq!(merger.next_fetch(), None);
let outcome = merger.finish(false).expect("valid merge outcome"); let outcome = merger.finish();
assert_eq!(outcome.picks.len(), 1); assert_eq!(outcome.picks.len(), 1);
assert!(!outcome.is_truncated); assert!(!outcome.is_truncated);
assert!(outcome.next_token.is_none()); assert!(outcome.next_token.is_none());
@@ -804,7 +740,6 @@ mod tests {
source: Some("source-1".to_string()), source: Some("source-1".to_string()),
source_done: false, source_done: false,
last_key: Some("a".to_string()), last_key: Some("a".to_string()),
no_progress: None,
}; };
let mut merger = ListThroughMerger::new(1, Some(&resume)); let mut merger = ListThroughMerger::new(1, Some(&resume));
merger.disable_source(); merger.disable_source();
@@ -816,7 +751,7 @@ mod tests {
Some("local-2".to_string()), Some("local-2".to_string()),
) )
.expect("local cursor advances"); .expect("local cursor advances");
let outcome = merger.finish(false).expect("valid merge outcome"); let outcome = merger.finish();
assert!(outcome.is_truncated); assert!(outcome.is_truncated);
let token = outcome.next_token.expect("truncated page carries a token"); let token = outcome.next_token.expect("truncated page carries a token");
assert_eq!(token.source.as_deref(), Some("source-1"), "the source cursor must not move"); assert_eq!(token.source.as_deref(), Some("source-1"), "the source cursor must not move");
@@ -895,7 +830,7 @@ mod tests {
.expect("opaque cursor advances regardless of sort order"); .expect("opaque cursor advances regardless of sort order");
} }
assert!(merger.next_fetch().is_none(), "two source fetches exhaust the request budget"); assert!(merger.next_fetch().is_none(), "two source fetches exhaust the request budget");
let outcome = merger.finish(false).expect("valid merge outcome"); let outcome = merger.finish();
assert!(outcome.picks.is_empty()); assert!(outcome.picks.is_empty());
assert!(outcome.is_truncated); assert!(outcome.is_truncated);
let token = outcome.next_token.expect("empty progressing page has a cursor"); let token = outcome.next_token.expect("empty progressing page has a cursor");
@@ -905,7 +840,7 @@ mod tests {
merger merger
.push_page(MergeSide::Source, vec![ListEntryKey::object("result")], false, None) .push_page(MergeSide::Source, vec![ListEntryKey::object("result")], false, None)
.expect("source EOF"); .expect("source EOF");
let outcome = merger.finish(false).expect("valid merge outcome"); let outcome = merger.finish();
assert_eq!( assert_eq!(
outcome.picks, outcome.picks,
vec![MergePick { vec![MergePick {
@@ -952,7 +887,7 @@ mod tests {
Err(ListPageError::Repeated) Err(ListPageError::Repeated)
); );
merger.disable_source(); merger.disable_source();
let outcome = merger.finish(false).expect("valid merge outcome"); let outcome = merger.finish();
assert_eq!( assert_eq!(
outcome.picks, outcome.picks,
vec![MergePick { vec![MergePick {
@@ -1044,287 +979,21 @@ mod tests {
let encoded = token.encode(); let encoded = token.encode();
assert_eq!(decode_continuation_token(&encoded), Ok(ListThroughCursor::Merged(Box::new(token)))); assert_eq!(decode_continuation_token(&encoded), Ok(ListThroughCursor::Merged(Box::new(token))));
let bumped = encoded.replace("\"v\":1", "\"v\":3"); let bumped = encoded.replace("\"v\":1", "\"v\":2");
assert_eq!(decode_continuation_token(&bumped), Err(ListThroughTokenError::UnsupportedVersion(3))); assert_eq!(decode_continuation_token(&bumped), Err(ListThroughTokenError::UnsupportedVersion(2)));
let extra = encoded.replace("{", "{\"x\":1,"); let extra = encoded.replace("{", "{\"x\":1,");
assert_eq!(decode_continuation_token(&extra), Err(ListThroughTokenError::Malformed)); assert_eq!(decode_continuation_token(&extra), Err(ListThroughTokenError::Malformed));
let truncated = &encoded[..encoded.len() - 3]; let truncated = &encoded[..encoded.len() - 3];
assert_eq!(decode_continuation_token(truncated), Err(ListThroughTokenError::Malformed)); assert_eq!(decode_continuation_token(truncated), Ok(ListThroughCursor::Local(truncated.to_string())));
let no_version = "\0odm-list:{\"t\":\"odm-list\"}"; let no_version = "{\"t\":\"odm-list\"}";
assert_eq!(decode_continuation_token(no_version), Err(ListThroughTokenError::Malformed)); assert_eq!(decode_continuation_token(no_version), Err(ListThroughTokenError::Malformed));
} }
fn progress_token(count: Option<u8>, local_done: bool, source_done: bool) -> ListThroughToken {
let mut token = ListThroughToken::new(
SideCursor {
token: None,
done: local_done,
},
SideCursor {
token: Some("A".into()),
done: source_done,
},
Some("last-key".into()),
);
if let Some(count) = count {
token.v = LIST_THROUGH_PROGRESS_TOKEN_VERSION;
token.no_progress = Some(count);
}
token
}
fn push_empty_pages(merger: &mut ListThroughMerger, side: MergeSide) {
for _ in 0..MAX_LIST_FETCHES_PER_SIDE {
let fetch = merger.next_fetch().expect("empty truncated side must be fetched");
assert_eq!(fetch.side, side);
let next = format!("{}:next", fetch.token.unwrap_or_default());
merger
.push_page(side, vec![], true, Some(next))
.expect("opaque cursor advances");
}
}
#[test]
fn progress_tokens_preserve_v1_bytes_and_validate_v2_counts() {
fn framed(payload: &str) -> String {
format!("{LIST_THROUGH_TOKEN_PREFIX}{payload}")
}
let token = progress_token(None, true, false);
assert_eq!(
token.encode(),
concat!(
"\0odm-list:",
r#"{"t":"odm-list","v":1,"local":null,"local_done":true,"source":"A","source_done":false,"last_key":"last-key"}"#
)
);
for count in 1..MAX_LIST_NO_PROGRESS_PAGES {
let token = progress_token(Some(count), true, false);
assert_eq!(decode_continuation_token(&token.encode()), Ok(ListThroughCursor::Merged(Box::new(token))));
}
for version in [1, 2] {
for value in ["null", "0", "16", "-1", "1.5", "256", "18446744073709551616", "\"1\""] {
let encoded = framed(&format!(r#"{{"t":"odm-list","v":{version},"no_progress":{value}}}"#));
assert_eq!(decode_continuation_token(&encoded), Err(ListThroughTokenError::Malformed), "{encoded}");
}
}
for payload in [
r#"{"t":"odm-list","v":1,"no_progress":1}"#,
r#"{"t":"odm-list","v":2}"#,
r#"{"t":"odm-list","v":2,"no_progress":1,"extra":true}"#,
] {
let encoded = framed(payload);
assert_eq!(decode_continuation_token(&encoded), Err(ListThroughTokenError::Malformed), "{encoded}");
}
}
#[test]
fn reader_only_nodes_do_not_start_a_budget_but_mixed_readers_preserve_one() {
let mut token = progress_token(None, true, false);
for _ in 0..MAX_LIST_NO_PROGRESS_PAGES {
let mut merger = ListThroughMerger::new(2, Some(&token));
push_empty_pages(&mut merger, MergeSide::Source);
token = merger
.finish(false)
.expect("reader-only v1 behavior")
.next_token
.expect("truncated cursor");
assert_eq!(token.v, 1);
assert_eq!(token.no_progress, None);
}
for count in 1..=MAX_LIST_NO_PROGRESS_PAGES {
let mut merger = ListThroughMerger::new(2, Some(&token));
push_empty_pages(&mut merger, MergeSide::Source);
assert!(merger.next_fetch().is_none(), "the per-request two-fetch limit stays intact");
let outcome = merger.finish(count % 2 == 1);
if count == MAX_LIST_NO_PROGRESS_PAGES {
assert_eq!(outcome, Err(ListPageError::NoProgress(MergeSide::Source)));
break;
}
token = outcome.expect("budget not exhausted").next_token.expect("truncated cursor");
assert_eq!(token.no_progress, Some(count));
let ListThroughCursor::Merged(decoded) = decode_continuation_token(&token.encode()).expect("round-trip v2") else {
panic!("merged cursor expected");
};
token = *decoded;
}
}
#[test]
fn objects_and_common_prefixes_reset_a_budget_at_the_boundary() {
for entry in [ListEntryKey::object("result"), ListEntryKey::prefix("result/")] {
for issue_tokens in [false, true] {
let resume = progress_token(Some(MAX_LIST_NO_PROGRESS_PAGES - 1), true, false);
let mut merger = ListThroughMerger::new(2, Some(&resume));
merger
.push_page(MergeSide::Source, vec![], true, Some("B".into()))
.expect("empty advancing page");
merger
.push_page(MergeSide::Source, vec![entry.clone()], true, Some("C".into()))
.expect("real progress");
let outcome = merger
.finish(issue_tokens)
.expect("real progress does not exhaust the budget");
assert_eq!(
outcome.picks,
vec![MergePick {
side: MergeSide::Source,
index: 0
}]
);
let next = outcome.next_token.expect("source remains truncated");
assert_eq!(next.last_key.as_deref(), Some(entry.name.as_str()));
assert_eq!(next.v, 1);
assert_eq!(next.no_progress, None);
assert!(!next.encode().contains("no_progress"));
}
}
}
#[test]
fn only_a_new_eof_transition_resets_the_empty_page_budget() {
for finished_side in [MergeSide::Local, MergeSide::Source] {
let resume = progress_token(Some(MAX_LIST_NO_PROGRESS_PAGES - 1), false, false);
let mut merger = ListThroughMerger::new(2, Some(&resume));
if finished_side == MergeSide::Local {
merger
.push_page(MergeSide::Local, vec![], false, None)
.expect("new local EOF");
push_empty_pages(&mut merger, MergeSide::Source);
} else {
push_empty_pages(&mut merger, MergeSide::Local);
merger
.push_page(MergeSide::Source, vec![], false, None)
.expect("new source EOF");
}
let next = merger
.finish(false)
.expect("new EOF is progress")
.next_token
.expect("other side truncated");
assert_eq!(next.no_progress, None);
assert_eq!(next.v, 1);
assert_eq!(next.local_done, finished_side == MergeSide::Local);
assert_eq!(next.source_done, finished_side == MergeSide::Source);
let mut merger = ListThroughMerger::new(2, Some(&next));
let remaining = if finished_side == MergeSide::Local {
MergeSide::Source
} else {
MergeSide::Local
};
push_empty_pages(&mut merger, remaining);
let next = merger
.finish(true)
.expect("a new budget starts")
.next_token
.expect("truncated");
assert_eq!(next.no_progress, Some(1), "an already-done side cannot reset every page");
}
let resume = progress_token(Some(MAX_LIST_NO_PROGRESS_PAGES - 1), true, false);
let mut merger = ListThroughMerger::new(2, Some(&resume));
merger.push_page(MergeSide::Source, vec![], false, None).expect("final EOF");
let outcome = merger.finish(false).expect("EOF succeeds at the budget boundary");
assert!(!outcome.is_truncated);
assert!(outcome.next_token.is_none());
}
#[test]
fn filtered_duplicates_cannot_reset_the_no_progress_budget() {
let resume = progress_token(Some(MAX_LIST_NO_PROGRESS_PAGES - 1), true, false);
let mut merger = ListThroughMerger::new(2, Some(&resume));
for next in ["B", "C"] {
let entries = [ListEntryKey::object("last-key"), ListEntryKey::object("earlier")]
.into_iter()
.filter(|entry| merger.accepts(&entry.name))
.collect::<Vec<_>>();
assert!(entries.is_empty(), "both provider entries were already consumed");
merger
.push_page(MergeSide::Source, entries, true, Some(next.into()))
.expect("advancing cursor");
}
assert_eq!(merger.finish(false), Err(ListPageError::NoProgress(MergeSide::Source)));
}
#[test]
fn no_progress_is_attributed_to_local_when_source_cannot_unblock_it() {
for source_mode in ["disabled", "done", "empty", "data"] {
let resume = progress_token(Some(MAX_LIST_NO_PROGRESS_PAGES - 1), false, source_mode == "done");
let mut merger = ListThroughMerger::new(2, Some(&resume));
if source_mode == "disabled" {
merger.disable_source();
}
push_empty_pages(&mut merger, MergeSide::Local);
match source_mode {
"empty" => push_empty_pages(&mut merger, MergeSide::Source),
"data" => merger
.push_page(MergeSide::Source, vec![ListEntryKey::object("source")], false, None)
.expect("source data"),
_ => {}
}
assert_eq!(merger.finish(false), Err(ListPageError::NoProgress(MergeSide::Local)), "{source_mode}");
}
}
#[test]
fn source_budget_failure_remerges_local_objects_and_prefixes_without_refetching() {
let resume = progress_token(Some(MAX_LIST_NO_PROGRESS_PAGES - 1), false, false);
let mut merger = ListThroughMerger::new(2, Some(&resume));
merger
.push_page(MergeSide::Local, vec![ListEntryKey::object("local")], true, Some("L1".into()))
.expect("local object");
merger
.push_page(MergeSide::Local, vec![ListEntryKey::prefix("prefix/")], true, Some("L2".into()))
.expect("local prefix");
push_empty_pages(&mut merger, MergeSide::Source);
assert_eq!(merger.finish(false), Err(ListPageError::NoProgress(MergeSide::Source)));
merger.disable_source();
assert!(merger.next_fetch().is_none(), "fallback does not perform another fetch");
let outcome = merger.finish(false).expect("local data makes progress");
assert_eq!(
outcome.picks,
vec![
MergePick {
side: MergeSide::Local,
index: 0
},
MergePick {
side: MergeSide::Local,
index: 1
}
]
);
let token = outcome.next_token.expect("remaining local page");
assert_eq!(token.local.as_deref(), Some("L2"));
assert_eq!(token.source.as_deref(), Some("A"));
assert_eq!(token.last_key.as_deref(), Some("prefix/"));
assert_eq!(token.no_progress, None);
assert_eq!(token.v, 1);
}
#[test]
fn a_zero_sized_merge_preserves_an_existing_budget() {
let resume = progress_token(Some(MAX_LIST_NO_PROGRESS_PAGES - 1), true, false);
let mut merger = ListThroughMerger::new(0, Some(&resume));
merger
.push_page(MergeSide::Source, vec![ListEntryKey::object("result")], true, Some("B".into()))
.expect("source page");
let outcome = merger.finish(false).expect("a zero-sized request cannot consume entries");
assert!(outcome.picks.is_empty());
assert_eq!(outcome.next_token.expect("unconsumed source").no_progress, resume.no_progress);
}
#[test] #[test]
fn a_plain_local_marker_stays_local() { fn a_plain_local_marker_stays_local() {
for marker in [
r#"{"t":"odm-list","v":1}"#,
r#"{"t":"odm-list","v":2,"local_done":true}"#,
r#"{"t":"odm-list"}"#,
] {
assert_eq!(decode_continuation_token(marker), Ok(ListThroughCursor::Local(marker.to_string())));
}
assert_eq!( assert_eq!(
decode_continuation_token("photos/2024/01.jpg"), decode_continuation_token("photos/2024/01.jpg"),
Ok(ListThroughCursor::Local("photos/2024/01.jpg".to_string())) Ok(ListThroughCursor::Local("photos/2024/01.jpg".to_string()))
@@ -19,45 +19,30 @@
//! client, and the per-node runtime (`sys`) that turns configs into live //! client, and the per-node runtime (`sys`) that turns configs into live
//! clients guarded by a breaker, a negative cache, singleflight and a pull //! clients guarded by a breaker, a negative cache, singleflight and a pull
//! concurrency limit (rustfs/backlog#2147). //! concurrency limit (rustfs/backlog#2147).
//!
//! A source is reached through one `SourceBackend`: the S3 dialect for every
//! S3-compatible provider, and a native backend for the providers that have no
//! S3 API (`azure`, `gcs_native`).
pub mod azure;
#[cfg(test)]
mod backend_contract;
pub mod backfill; pub mod backfill;
pub mod breaker; pub mod breaker;
pub mod config; pub mod config;
#[cfg(feature = "gcs")]
pub mod gcs;
pub mod list_through; pub mod list_through;
mod metrics;
mod native_http;
pub mod negative_cache; pub mod negative_cache;
pub mod pull; pub mod pull;
pub mod source_client; pub mod source_client;
pub mod stats; pub mod stats;
mod storage_api;
pub mod sys; pub mod sys;
#[cfg(test)]
mod test_http_fixture;
pub use breaker::{ pub use breaker::{
BREAKER_FAILURE_THRESHOLD, BREAKER_FAILURE_WINDOW, BREAKER_HALF_OPEN_MAX_PROBES, BREAKER_OPEN_DURATION, Breaker, BREAKER_FAILURE_THRESHOLD, BREAKER_FAILURE_WINDOW, BREAKER_HALF_OPEN_MAX_PROBES, BREAKER_OPEN_DURATION, Breaker,
BreakerState, BreakerTransition, BreakerVerdict, BreakerState, BreakerTransition, BreakerVerdict,
}; };
pub use config::{ pub use config::{
AzureSourceConfig, FilterConfig, GcsSourceConfig, HeadPolicy, ON_DEMAND_MIGRATION_CONFIG_VERSION, OnDemandMigrationConfig, ConfigPublishHook, FilterConfig, HeadPolicy, ON_DEMAND_MIGRATION_CONFIG_HOOK, ON_DEMAND_MIGRATION_CONFIG_VERSION,
OnDemandMigrationConfigError, PathStyle, PolicyConfig, Provider, RangeGetPolicy, SourceConfig, SourceCredentials, OnDemandMigrationConfig, OnDemandMigrationConfigError, PathStyle, PolicyConfig, Provider, RangeGetPolicy, SourceConfig,
SourceErrorPolicy, SourceTimeout, TlsConfig, ValidationContext, SourceCredentials, SourceErrorPolicy, SourceTimeout, TlsConfig, ValidationContext,
}; };
pub use list_through::{ pub use list_through::{
FetchRequest, LIST_THROUGH_TOKEN_VERSION, ListEntryKey, ListPageError, ListThroughCursor, ListThroughMerger, FetchRequest, LIST_THROUGH_TOKEN_VERSION, ListEntryKey, ListThroughCursor, ListThroughMerger, ListThroughToken,
ListThroughToken, ListThroughTokenError, MAX_LIST_FETCHES_PER_SIDE, MAX_LIST_NO_PROGRESS_PAGES, MergeOutcome, MergePick, ListThroughTokenError, MAX_LIST_FETCHES_PER_SIDE, MergeOutcome, MergePick, MergeSide, SOURCE_LIST_MAX_RATE_WAIT,
MergeSide, SOURCE_LIST_MAX_RATE_WAIT, SOURCE_LIST_RATE_PER_SEC, SourceListPlan, SourceListRateLimiter, SOURCE_LIST_RATE_PER_SEC, SourceListPlan, SourceListRateLimiter, decode_continuation_token, source_list_plan,
decode_continuation_token, source_list_plan,
}; };
pub use negative_cache::{NEGATIVE_CACHE_MAX_ENTRIES, NegativeCache}; pub use negative_cache::{NEGATIVE_CACHE_MAX_ENTRIES, NegativeCache};
pub use pull::{ pub use pull::{
@@ -65,19 +50,11 @@ pub use pull::{
PullQueue, PullReason, PullSource, QueuedPullOutcome, SourceBody, SourceIdleGuard, WriteBackBody, WriteBackError, PullQueue, PullReason, PullSource, QueuedPullOutcome, SourceBody, SourceIdleGuard, WriteBackBody, WriteBackError,
WriteBackOutcome, WriteBackPart, WriteBackRequest, commit_inline, commit_inline_with, idle_guarded_body, WriteBackOutcome, WriteBackPart, WriteBackRequest, commit_inline, commit_inline_with, idle_guarded_body,
}; };
pub use source_client::{
SourceClient, SourceError, SourceGet, SourceHead, SourceListRequest, SourceObject, SourcePage, SourceSse, is_multipart_etag,
};
pub use stats::{ pub use stats::{
GaugeGuard, LastSourceError, LatencyBucketSnapshot, OdmOp, OdmOutcome, OdmStats, OdmStatsSnapshot, PullFailureReason, GaugeGuard, LastSourceError, LatencyBucketSnapshot, OdmOp, OdmOutcome, OdmStats, OdmStatsSnapshot, PullFailureReason,
PullPath, SOURCE_LATENCY_BUCKET_BOUNDS_MS, SourceLatencySnapshot, PullPath, SOURCE_LATENCY_BUCKET_BOUNDS_MS, SourceLatencySnapshot,
}; };
pub use sys::{ pub use sys::{
ApplyOutcome, BucketOdmState, GLOBAL_ON_DEMAND_MIGRATION_SYS, OdmBucketSnapshot, OdmLookup, OdmStateError, ApplyOutcome, BucketOdmState, GLOBAL_ON_DEMAND_MIGRATION_SYS, OdmBucketSnapshot, OdmLookup, OdmStateError,
OnDemandMigrationSys, PullError, PullFollower, PullLeader, PullOutcome, PullResult, PullSlot, source_backend_spec, OnDemandMigrationSys, PullError, PullFollower, PullLeader, PullOutcome, PullResult, PullSlot, source_client_spec,
source_client_spec,
}; };
pub(crate) fn register_metrics() {
metrics::register();
}
@@ -46,10 +46,10 @@ use super::stats::{PullFailureReason, PullPath};
use super::sys::{BucketOdmState, OnDemandMigrationSys, PullError, PullOutcome, PullSlot}; use super::sys::{BucketOdmState, OnDemandMigrationSys, PullError, PullOutcome, PullSlot};
use async_trait::async_trait; use async_trait::async_trait;
use bytes::Bytes; use bytes::Bytes;
use futures::{FutureExt, Stream, StreamExt, future::Shared}; use futures::{Stream, StreamExt};
use parking_lot::Mutex; use parking_lot::Mutex;
use rand::RngExt; use rand::RngExt;
use std::collections::HashMap; use std::collections::{HashMap, HashSet};
use std::fmt; use std::fmt;
use std::io; use std::io;
use std::pin::Pin; use std::pin::Pin;
@@ -133,8 +133,6 @@ pub enum QueuedPullOutcome {
Failed(PullError), Failed(PullError),
} }
pub type QueuedPullReport = Shared<oneshot::Receiver<QueuedPullOutcome>>;
/// Result of [`PullQueue::enqueue`]. /// Result of [`PullQueue::enqueue`].
#[derive(Clone, Copy, Debug, PartialEq, Eq, Hash)] #[derive(Clone, Copy, Debug, PartialEq, Eq, Hash)]
pub enum EnqueueOutcome { pub enum EnqueueOutcome {
@@ -243,8 +241,6 @@ impl PullSource for SourceClient {
#[derive(Clone, Debug)] #[derive(Clone, Debug)]
pub struct WriteBackRequest { pub struct WriteBackRequest {
pub bucket: String, pub bucket: String,
/// Identity captured with the source configuration, retained through cleanup.
pub bucket_incarnation_id: uuid::Uuid,
pub key: String, pub key: String,
/// Source HEAD/GET of the whole object. /// Source HEAD/GET of the whole object.
pub head: SourceHead, pub head: SourceHead,
@@ -255,7 +251,6 @@ pub struct WriteBackRequest {
pub preserve_etag: bool, pub preserve_etag: bool,
/// `policy.emit_events`. /// `policy.emit_events`.
pub emit_events: bool, pub emit_events: bool,
pub respect_delete_marker: bool,
/// Source tags to copy (`policy.copy_tags`), `None` to skip. /// Source tags to copy (`policy.copy_tags`), `None` to skip.
pub tags: Option<HashMap<String, String>>, pub tags: Option<HashMap<String, String>>,
} }
@@ -265,14 +260,12 @@ impl WriteBackRequest {
let config = state.config(); let config = state.config();
Self { Self {
bucket: state.bucket().to_string(), bucket: state.bucket().to_string(),
bucket_incarnation_id: state.incarnation_id(),
key: key.to_string(), key: key.to_string(),
head, head,
source_label: format!("{}:{}", config.source.provider.as_str(), config.source.bucket), source_label: format!("{}:{}", config.source.provider.as_str(), config.source.bucket),
pulled_at: OffsetDateTime::now_utc(), pulled_at: OffsetDateTime::now_utc(),
preserve_etag: config.policy.preserve_etag, preserve_etag: config.policy.preserve_etag,
emit_events: config.policy.emit_events, emit_events: config.policy.emit_events,
respect_delete_marker: config.policy.respect_local_delete_marker,
tags, tags,
} }
} }
@@ -360,7 +353,7 @@ pub trait OdmWriteBack: Send + Sync {
parts: Vec<WriteBackPart>, parts: Vec<WriteBackPart>,
) -> Result<WriteBackOutcome, WriteBackError>; ) -> Result<WriteBackOutcome, WriteBackError>;
async fn abort_multipart_upload(&self, request: &WriteBackRequest, upload_id: &str) -> Result<(), WriteBackError>; async fn abort_multipart_upload(&self, bucket: &str, key: &str, upload_id: &str) -> Result<(), WriteBackError>;
} }
/// Why the pump stopped feeding the write-back before EOF. /// Why the pump stopped feeding the write-back before EOF.
@@ -665,7 +658,9 @@ async fn write_multipart(
Err(err) => Err(err), Err(err) => Err(err),
}; };
if completed.is_err() if completed.is_err()
&& let Err(abort_err) = write_back.abort_multipart_upload(request, &upload_id).await && let Err(abort_err) = write_back
.abort_multipart_upload(&request.bucket, &request.key, &upload_id)
.await
{ {
debug!( debug!(
event = EVENT_ODM_PULL_FAILED, event = EVENT_ODM_PULL_FAILED,
@@ -835,7 +830,7 @@ pub struct PullQueue {
bucket: String, bucket: String,
tx: mpsc::Sender<PullJob>, tx: mpsc::Sender<PullJob>,
/// Keys queued or running; the job removes its key when it ends. /// Keys queued or running; the job removes its key when it ends.
pending: Mutex<HashMap<String, QueuedPullReport>>, pending: Mutex<HashSet<String>>,
capacity: usize, capacity: usize,
cancel: CancellationToken, cancel: CancellationToken,
stats: Arc<super::stats::OdmStats>, stats: Arc<super::stats::OdmStats>,
@@ -874,7 +869,7 @@ impl PullQueue {
let queue = Arc::new(Self { let queue = Arc::new(Self {
bucket: state.bucket().to_string(), bucket: state.bucket().to_string(),
tx, tx,
pending: Mutex::new(HashMap::new()), pending: Mutex::new(HashSet::new()),
capacity, capacity,
cancel: state.cancel_token(), cancel: state.cancel_token(),
stats: Arc::clone(state.stats()), stats: Arc::clone(state.stats()),
@@ -908,24 +903,29 @@ impl PullQueue {
self.enqueue_with_report(key, reason).0 self.enqueue_with_report(key, reason).0
} }
/// [`Self::enqueue`] with a shared report, including for coalesced pulls. /// [`Self::enqueue`] that also hands back the job's report channel when
pub fn enqueue_with_report(&self, key: &str, reason: PullReason) -> (EnqueueOutcome, Option<QueuedPullReport>) { /// a new job was queued (`Coalesced` pulls report to their first
/// requester only).
pub fn enqueue_with_report(
&self,
key: &str,
reason: PullReason,
) -> (EnqueueOutcome, Option<oneshot::Receiver<QueuedPullOutcome>>) {
if self.cancel.is_cancelled() { if self.cancel.is_cancelled() {
return (EnqueueOutcome::Unavailable, None); return (EnqueueOutcome::Unavailable, None);
} }
let mut pending = self.pending.lock(); let mut pending = self.pending.lock();
if let Some(report) = pending.get(key) { if pending.contains(key) {
return (EnqueueOutcome::Coalesced, Some(report.clone())); return (EnqueueOutcome::Coalesced, None);
} }
let (report_tx, report_rx) = oneshot::channel(); let (report_tx, report_rx) = oneshot::channel();
let report_rx = report_rx.shared();
match self.tx.try_send(PullJob { match self.tx.try_send(PullJob {
key: key.to_string(), key: key.to_string(),
reason, reason,
report: Some(report_tx), report: Some(report_tx),
}) { }) {
Ok(()) => { Ok(()) => {
pending.insert(key.to_string(), report_rx.clone()); pending.insert(key.to_string());
(EnqueueOutcome::Enqueued, Some(report_rx)) (EnqueueOutcome::Enqueued, Some(report_rx))
} }
Err(TrySendError::Full(_)) => { Err(TrySendError::Full(_)) => {
@@ -1072,7 +1072,7 @@ impl BucketOdmState {
self: &Arc<Self>, self: &Arc<Self>,
key: &str, key: &str,
reason: PullReason, reason: PullReason,
) -> (EnqueueOutcome, Option<QueuedPullReport>) { ) -> (EnqueueOutcome, Option<oneshot::Receiver<QueuedPullOutcome>>) {
match self.pull_queue() { match self.pull_queue() {
Some(queue) => queue.enqueue_with_report(key, reason), Some(queue) => queue.enqueue_with_report(key, reason),
None => (EnqueueOutcome::Unavailable, None), None => (EnqueueOutcome::Unavailable, None),
@@ -1094,7 +1094,7 @@ impl OnDemandMigrationSys {
#[cfg(test)] #[cfg(test)]
mod tests { mod tests {
use super::*; use super::*;
use crate::on_demand_migration::config::{ use crate::bucket::on_demand_migration::config::{
FilterConfig, OnDemandMigrationConfig, PathStyle as ConfigPathStyle, PolicyConfig, Provider, SourceConfig, FilterConfig, OnDemandMigrationConfig, PathStyle as ConfigPathStyle, PolicyConfig, Provider, SourceConfig,
SourceCredentials, TlsConfig, SourceCredentials, TlsConfig,
}; };
@@ -1119,8 +1119,6 @@ mod tests {
session_token: None, session_token: None,
}), }),
tls: TlsConfig::default(), tls: TlsConfig::default(),
azure: None,
gcs: None,
}, },
filter: FilterConfig::default(), filter: FilterConfig::default(),
policy: PolicyConfig::default(), policy: PolicyConfig::default(),
@@ -1341,7 +1339,7 @@ mod tests {
}) })
} }
async fn abort_multipart_upload(&self, _request: &WriteBackRequest, upload_id: &str) -> Result<(), WriteBackError> { async fn abort_multipart_upload(&self, _bucket: &str, _key: &str, upload_id: &str) -> Result<(), WriteBackError> {
self.aborted.lock().push(upload_id.to_string()); self.aborted.lock().push(upload_id.to_string());
Ok(()) Ok(())
} }
@@ -1401,21 +1399,13 @@ mod tests {
assert_eq!(queue.capacity(), 1024); assert_eq!(queue.capacity(), 1024);
let mut outcomes = HashMap::new(); let mut outcomes = HashMap::new();
let mut shared_report = None;
for _ in 0..100 { for _ in 0..100 {
let (outcome, report) = queue.enqueue_with_report("a", PullReason::RangeGet); *outcomes.entry(queue.enqueue("a", PullReason::RangeGet)).or_insert(0) += 1;
*outcomes.entry(outcome).or_insert(0) += 1;
shared_report = report;
} }
assert_eq!(outcomes.get(&EnqueueOutcome::Enqueued), Some(&1)); assert_eq!(outcomes.get(&EnqueueOutcome::Enqueued), Some(&1));
assert_eq!(outcomes.get(&EnqueueOutcome::Coalesced), Some(&99)); assert_eq!(outcomes.get(&EnqueueOutcome::Coalesced), Some(&99));
assert_eq!(queue.pending_keys(), 1); assert_eq!(queue.pending_keys(), 1);
assert_eq!(
shared_report.expect("coalesced report").await,
Ok(QueuedPullOutcome::Stored { size: 1000 })
);
wait_until("first pull to finish", || queue.pending_keys() == 0).await; wait_until("first pull to finish", || queue.pending_keys() == 0).await;
assert_eq!(source.head_calls.load(Ordering::SeqCst), 1); assert_eq!(source.head_calls.load(Ordering::SeqCst), 1);
assert_eq!(source.get_calls.load(Ordering::SeqCst), 1); assert_eq!(source.get_calls.load(Ordering::SeqCst), 1);
@@ -1448,23 +1438,6 @@ mod tests {
assert_eq!(queue.enqueue("a", PullReason::RangeGet), EnqueueOutcome::Unavailable); assert_eq!(queue.enqueue("a", PullReason::RangeGet), EnqueueOutcome::Unavailable);
} }
#[tokio::test]
async fn coalesced_enqueues_share_failure_reports() {
let sys = OnDemandMigrationSys::new();
let state = enabled_state(&sys, &config()).await;
let source = MockSource::with_object("missing", 1000, BodyKind::Bytes(body_bytes(1000)));
let queue = PullQueue::start(Arc::clone(&state), source, Arc::new(MockWriteBack::default()));
let (first, first_report) = queue.enqueue_with_report("absent", PullReason::RangeGet);
let (second, second_report) = queue.enqueue_with_report("absent", PullReason::Backfill);
assert_eq!(first, EnqueueOutcome::Enqueued);
assert_eq!(second, EnqueueOutcome::Coalesced);
let (first, second) = tokio::join!(first_report.expect("leader report"), second_report.expect("coalesced report"));
assert_eq!(first, second);
assert!(matches!(first, Ok(QueuedPullOutcome::Failed(_))));
sys.remove(BUCKET);
queue.wait_until_stopped().await;
}
#[tokio::test] #[tokio::test]
async fn queue_full_is_reported_and_cancel_drains_without_leaking_tasks() { async fn queue_full_is_reported_and_cancel_drains_without_leaking_tasks() {
let sys = OnDemandMigrationSys::new(); let sys = OnDemandMigrationSys::new();
@@ -1494,23 +1467,16 @@ mod tests {
wait_until("dispatcher to wait for a slot", || state.stats().queue_depth() == 1).await; wait_until("dispatcher to wait for a slot", || state.stats().queue_depth() == 1).await;
assert_eq!(queue.enqueue("c", PullReason::LargeObject), EnqueueOutcome::Enqueued); assert_eq!(queue.enqueue("c", PullReason::LargeObject), EnqueueOutcome::Enqueued);
assert_eq!(queue.enqueue("d", PullReason::LargeObject), EnqueueOutcome::QueueFull); assert_eq!(queue.enqueue("d", PullReason::LargeObject), EnqueueOutcome::QueueFull);
let (coalesced, canceled_report) = queue.enqueue_with_report("c", PullReason::LargeObject); assert_eq!(queue.enqueue("c", PullReason::LargeObject), EnqueueOutcome::Coalesced);
assert_eq!(coalesced, EnqueueOutcome::Coalesced);
assert_eq!(queue.pending_keys(), 3); assert_eq!(queue.pending_keys(), 3);
assert_eq!(failures(&state).get("queue_full"), Some(&1)); assert_eq!(failures(&state).get("queue_full"), Some(&1));
assert!(!queue.is_stopped()); assert!(!queue.is_stopped());
assert_eq!(sys.remove(BUCKET), crate::on_demand_migration::ApplyOutcome::Removed); assert_eq!(sys.remove(BUCKET), crate::bucket::on_demand_migration::ApplyOutcome::Removed);
tokio::time::timeout(Duration::from_secs(5), queue.wait_until_stopped()) tokio::time::timeout(Duration::from_secs(5), queue.wait_until_stopped())
.await .await
.expect("dispatcher and in-flight job must exit after cancel"); .expect("dispatcher and in-flight job must exit after cancel");
assert!(queue.is_stopped()); assert!(queue.is_stopped());
assert!(
tokio::time::timeout(Duration::from_secs(5), canceled_report.expect("coalesced cancellation report"))
.await
.expect("cancellation closes the report")
.is_err()
);
assert_eq!(queue.pending_keys(), 0); assert_eq!(queue.pending_keys(), 0);
assert_eq!(state.inflight_keys(), 0); assert_eq!(state.inflight_keys(), 0);
assert_eq!(state.stats().inflight_pulls(), 0); assert_eq!(state.stats().inflight_pulls(), 0);
@@ -25,14 +25,11 @@
//! Client-supplied `If-*`, `Authorization`, `Host` and SSE-C headers are never //! Client-supplied `If-*`, `Authorization`, `Host` and SSE-C headers are never
//! forwarded: v1 rejects SSE-C source objects outright. //! forwarded: v1 rejects SSE-C source objects outright.
use super::azure::AzureSourceBackend;
#[cfg(feature = "gcs")]
use super::gcs::GcsNativeSourceBackend;
use super::list_through::{ListPageError, validate_list_page}; use super::list_through::{ListPageError, validate_list_page};
use super::storage_api::HTTPRangeSpec; use crate::bucket::remote_s3_client::{
use super::storage_api::remote_s3_client::{
PathStyle, RemoteCredentials, RemoteS3ClientError, RemoteS3EndpointSpec, RemoteS3RetryPolicy, build_remote_s3_config, PathStyle, RemoteCredentials, RemoteS3ClientError, RemoteS3EndpointSpec, RemoteS3RetryPolicy, build_remote_s3_config,
}; };
use crate::storage_api_contracts::range::HTTPRangeSpec;
use aws_sdk_s3::Client as S3Client; use aws_sdk_s3::Client as S3Client;
use aws_sdk_s3::error::{ProvideErrorMetadata, SdkError}; use aws_sdk_s3::error::{ProvideErrorMetadata, SdkError};
use aws_sdk_s3::operation::get_object::GetObjectOutput; use aws_sdk_s3::operation::get_object::GetObjectOutput;
@@ -68,10 +65,6 @@ pub enum SourceProvider {
/// Generic S3-compatible service. /// Generic S3-compatible service.
#[default] #[default]
S3, S3,
/// Native Azure Blob service; not an S3 dialect.
Azure,
/// Native GCS JSON API with a service-account key; not an S3 dialect.
GcsNative,
} }
impl SourceProvider { impl SourceProvider {
@@ -83,8 +76,6 @@ impl SourceProvider {
"minio" => Some(Self::Minio), "minio" => Some(Self::Minio),
"rustfs" => Some(Self::Rustfs), "rustfs" => Some(Self::Rustfs),
"s3" => Some(Self::S3), "s3" => Some(Self::S3),
"azure" => Some(Self::Azure),
"gcs_native" => Some(Self::GcsNative),
_ => None, _ => None,
} }
} }
@@ -97,8 +88,6 @@ impl SourceProvider {
Self::Minio => "minio", Self::Minio => "minio",
Self::Rustfs => "rustfs", Self::Rustfs => "rustfs",
Self::S3 => "s3", Self::S3 => "s3",
Self::Azure => "azure",
Self::GcsNative => "gcs_native",
} }
} }
@@ -164,75 +153,12 @@ pub struct SourceClientSpec {
/// Wire requests one logical source call may cost. The pull pipeline and /// Wire requests one logical source call may cost. The pull pipeline and
/// the backfill job own the retry budget (`pull.rs` `PULL_MAX_RETRIES`, /// the backfill job own the retry budget (`pull.rs` `PULL_MAX_RETRIES`,
/// `backfill.rs` `LIST_MAX_RETRIES`) and the breaker counts logical calls, /// `backfill.rs` `LIST_MAX_RETRIES`) and the breaker counts logical calls,
/// so ODM declares [`RemoteS3RetryPolicy::Disabled`]. An ambiguous HEAD /// so ODM declares [`RemoteS3RetryPolicy::Disabled`] and keeps one counted
/// 404 additionally probes the bucket before declaring a key absent. /// failure equal to one request against a struggling source.
pub retry: RemoteS3RetryPolicy, pub retry: RemoteS3RetryPolicy,
/// Bytes per second the pull pipeline may consume from this source; /// Bytes per second the pull pipeline may consume from this source;
/// `None` means unlimited. Enforced by the consumer, not by this client. /// `None` means unlimited. Enforced by the consumer, not by this client.
pub bandwidth_limit: Option<NonZeroU64>, pub bandwidth_limit: Option<NonZeroU64>,
/// Which [`SourceBackend`] to build. The S3 variant reads `region`,
/// `path_style` and `credentials`; the native variants ignore all three
/// and carry their own credentials.
pub backend: SourceBackendSpec,
}
/// Provider-specific half of [`SourceClientSpec`].
#[derive(Clone, Debug, Default, PartialEq, Eq)]
pub enum SourceBackendSpec {
#[default]
S3,
Azure(AzureSourceSpec),
Gcs(GcsSourceSpec),
}
/// Native Azure Blob parameters. The container is [`SourceClientSpec::bucket`].
#[derive(Clone, PartialEq, Eq)]
pub struct AzureSourceSpec {
pub account: String,
pub auth: AzureAuth,
}
impl fmt::Debug for AzureSourceSpec {
fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result {
f.debug_struct("AzureSourceSpec")
.field("account", &self.account)
.field("auth", &self.auth)
.finish()
}
}
/// How Azure requests are authorized.
#[derive(Clone, PartialEq, Eq)]
pub enum AzureAuth {
/// Base64 storage-account key, signed per request with Shared Key.
SharedKey(String),
/// SAS query string without the leading `?`, appended to every URL.
Sas(String),
}
impl fmt::Debug for AzureAuth {
fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result {
// Both variants are secrets; only the scheme may be rendered.
f.write_str(match self {
Self::SharedKey(_) => "SharedKey(REDACTED)",
Self::Sas(_) => "Sas(REDACTED)",
})
}
}
/// Native GCS parameters. The bucket is [`SourceClientSpec::bucket`].
#[derive(Clone, PartialEq, Eq)]
pub struct GcsSourceSpec {
/// Service-account key JSON.
pub service_account_json: String,
}
impl fmt::Debug for GcsSourceSpec {
fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result {
f.debug_struct("GcsSourceSpec")
.field("service_account_json", &"REDACTED")
.finish()
}
} }
impl SourceClientSpec { impl SourceClientSpec {
@@ -335,9 +261,8 @@ const THROTTLE_CODES: &[&str] = &[
"RequestLimitExceeded", "RequestLimitExceeded",
"TooManyRequests", "TooManyRequests",
"RequestThrottled", "RequestThrottled",
"ServerBusy",
]; ];
const NOT_FOUND_CODES: &[&str] = &["NoSuchKey", "BlobNotFound"]; const NOT_FOUND_CODES: &[&str] = &["NoSuchKey", "NotFound", "NoSuchBucket", "NoSuchVersion"];
const ACCESS_DENIED_CODES: &[&str] = &[ const ACCESS_DENIED_CODES: &[&str] = &[
"AccessDenied", "AccessDenied",
"InvalidAccessKeyId", "InvalidAccessKeyId",
@@ -345,10 +270,9 @@ const ACCESS_DENIED_CODES: &[&str] = &[
"AllAccessDisabled", "AllAccessDisabled",
"ExpiredToken", "ExpiredToken",
"InvalidToken", "InvalidToken",
"AuthorizationPermissionMismatch",
]; ];
pub(super) fn classify_status(status: u16, code: Option<&str>, message: String) -> SourceError { fn classify_status(status: u16, code: Option<&str>, message: String) -> SourceError {
if let Some(code) = code { if let Some(code) = code {
if THROTTLE_CODES.contains(&code) { if THROTTLE_CODES.contains(&code) {
return SourceError::Throttled; return SourceError::Throttled;
@@ -361,6 +285,7 @@ pub(super) fn classify_status(status: u16, code: Option<&str>, message: String)
} }
} }
match status { match status {
404 => SourceError::NotFound,
401 | 403 => SourceError::AccessDenied, 401 | 403 => SourceError::AccessDenied,
429 | 503 => SourceError::Throttled, 429 | 503 => SourceError::Throttled,
500..=599 => SourceError::ServerError(status), 500..=599 => SourceError::ServerError(status),
@@ -420,11 +345,6 @@ pub struct SourceHead {
pub storage_class: Option<String>, pub storage_class: Option<String>,
pub sse: Option<SourceSse>, pub sse: Option<SourceSse>,
pub is_multipart_etag: bool, pub is_multipart_etag: bool,
/// The provider's ETag is not derived from the object bytes (Azure
/// stamps an opaque concurrency token). Such an ETag is recorded for
/// provenance but must never be read as a content digest, so the
/// write-back path refuses to use it as the expected MD5.
pub etag_is_opaque: bool,
} }
/// Per-operation fields shared by HEAD and GET outputs. /// Per-operation fields shared by HEAD and GET outputs.
@@ -446,7 +366,7 @@ struct HeadParts {
sse_customer_algorithm: Option<String>, sse_customer_algorithm: Option<String>,
} }
pub(super) fn normalize_etag(etag: Option<String>) -> Option<String> { fn normalize_etag(etag: Option<String>) -> Option<String> {
etag.map(|etag| etag.trim().trim_matches('"').to_string()) etag.map(|etag| etag.trim().trim_matches('"').to_string())
.filter(|etag| !etag.is_empty()) .filter(|etag| !etag.is_empty())
} }
@@ -495,7 +415,6 @@ fn source_head(parts: HeadParts) -> Result<SourceHead, SourceError> {
storage_class: parts.storage_class, storage_class: parts.storage_class,
sse, sse,
is_multipart_etag, is_multipart_etag,
etag_is_opaque: false,
}) })
} }
@@ -706,55 +625,14 @@ impl fmt::Debug for SourceClient {
impl SourceClient { impl SourceClient {
pub async fn new(spec: &SourceClientSpec) -> Result<Self, RemoteS3ClientError> { pub async fn new(spec: &SourceClientSpec) -> Result<Self, RemoteS3ClientError> {
match &spec.backend { let endpoint = spec.endpoint_spec()?;
SourceBackendSpec::S3 => { let config = build_remote_s3_config(&endpoint).await?;
let endpoint = spec.endpoint_spec()?; Ok(Self::from_config_builder(config, endpoint.endpoint_url(), spec))
let config = build_remote_s3_config(&endpoint).await?;
Ok(Self::from_config_builder(config, endpoint.endpoint_url(), spec))
}
SourceBackendSpec::Azure(azure) => {
let backend = AzureSourceBackend::new(
&spec.endpoint,
&spec.bucket,
azure,
spec.timeouts,
spec.skip_tls_verify,
spec.ca_cert_pem.as_deref(),
)?;
Ok(Self::from_backend(Box::new(backend), spec))
}
#[cfg(not(feature = "gcs"))]
SourceBackendSpec::Gcs(_) => Err(RemoteS3ClientError::BackendNotCompiled("gcs_native")),
#[cfg(feature = "gcs")]
SourceBackendSpec::Gcs(gcs) => {
let backend = GcsNativeSourceBackend::new(
&spec.endpoint,
&spec.bucket,
gcs,
spec.timeouts,
spec.skip_tls_verify,
spec.ca_cert_pem.as_deref(),
)?;
Ok(Self::from_backend(Box::new(backend), spec))
}
}
}
/// Wraps a ready backend in the prefix-mapping client. The endpoint is
/// kept only for `Debug` and admin status.
fn from_backend(backend: Box<dyn SourceBackend>, spec: &SourceClientSpec) -> Self {
Self {
backend,
endpoint: spec.endpoint.clone(),
bucket: spec.bucket.clone(),
source_prefix: spec.source_prefix.clone().filter(|prefix| !prefix.is_empty()),
timeouts: spec.timeouts,
bandwidth_limit: spec.bandwidth_limit,
}
} }
/// `config` must come from [`SourceClientSpec::endpoint_spec`], which is /// `config` must come from [`SourceClientSpec::endpoint_spec`], which is
/// where the policy disabling SDK-level retries is declared. /// where the retry policy that keeps one logical call equal to one wire
/// request is declared.
fn from_config_builder(config: aws_sdk_s3::config::Builder, endpoint: String, spec: &SourceClientSpec) -> Self { fn from_config_builder(config: aws_sdk_s3::config::Builder, endpoint: String, spec: &SourceClientSpec) -> Self {
let client = S3Client::from_conf(config.interceptor(SourceProxyMarkerInterceptor::new()).build()); let client = S3Client::from_conf(config.interceptor(SourceProxyMarkerInterceptor::new()).build());
Self { Self {
@@ -876,16 +754,15 @@ impl SourceClient {
#[async_trait::async_trait] #[async_trait::async_trait]
impl SourceBackend for S3SourceBackend { impl SourceBackend for S3SourceBackend {
async fn head(&self, key: &str) -> Result<SourceHead, SourceError> { async fn head(&self, key: &str) -> Result<SourceHead, SourceError> {
match self.client.head_object().bucket(&self.bucket).key(key).send().await { let output = self
Ok(output) => source_head_from_head_output(output), .client
Err(err) if err.raw_response().is_some_and(|response| response.status().as_u16() == 404) => { .head_object()
// HEAD has no error body: a missing bucket must not poison .bucket(&self.bucket)
// the per-key negative cache as though only the key was absent. .key(key)
self.probe().await?; .send()
Err(SourceError::NotFound) .await
} .map_err(classify_sdk_error)?;
Err(err) => Err(classify_sdk_error(err)), source_head_from_head_output(output)
}
} }
/// Streams the object; `range` is passed through as an HTTP `Range` /// Streams the object; `range` is passed through as an HTTP `Range`
@@ -932,8 +809,8 @@ impl SourceBackend for S3SourceBackend {
.contents .contents
.unwrap_or_default() .unwrap_or_default()
.into_iter() .into_iter()
.map(s3_source_object) .filter_map(s3_source_object)
.collect::<Result<Vec<_>, _>>()?; .collect();
let common_prefixes = output let common_prefixes = output
.common_prefixes .common_prefixes
.unwrap_or_default() .unwrap_or_default()
@@ -972,20 +849,14 @@ impl SourceBackend for S3SourceBackend {
} }
} }
fn s3_source_object(object: SdkObject) -> Result<SourceObject, SourceError> { fn s3_source_object(object: SdkObject) -> Option<SourceObject> {
let key = object let key = object.key?;
.key
.ok_or_else(|| SourceError::Other("source listing object has no key".to_string()))?;
let size = object
.size
.and_then(|size| u64::try_from(size).ok())
.ok_or_else(|| SourceError::Other("source listing object has no valid size".to_string()))?;
let etag = normalize_etag(object.e_tag); let etag = normalize_etag(object.e_tag);
let is_multipart_etag = etag.as_deref().is_some_and(is_multipart_etag); let is_multipart_etag = etag.as_deref().is_some_and(is_multipart_etag);
Ok(SourceObject { Some(SourceObject {
key, key,
etag, etag,
size, size: object.size.and_then(|size| u64::try_from(size).ok()).unwrap_or(0),
last_modified: system_time(object.last_modified), last_modified: system_time(object.last_modified),
storage_class: object.storage_class.map(|class| class.as_str().to_string()), storage_class: object.storage_class.map(|class| class.as_str().to_string()),
is_multipart_etag, is_multipart_etag,
@@ -995,7 +866,6 @@ fn s3_source_object(object: SdkObject) -> Result<SourceObject, SourceError> {
#[cfg(test)] #[cfg(test)]
mod tests { mod tests {
use super::*; use super::*;
use crate::on_demand_migration::backend_contract::{BackendCapabilities, OBJECT_MD5, assert_backend_contract};
use aws_smithy_runtime_api::client::http::{HttpConnector, HttpConnectorFuture, SharedHttpConnector, http_client_fn}; use aws_smithy_runtime_api::client::http::{HttpConnector, HttpConnectorFuture, SharedHttpConnector, http_client_fn};
use aws_smithy_runtime_api::client::orchestrator::HttpRequest; use aws_smithy_runtime_api::client::orchestrator::HttpRequest;
use aws_smithy_runtime_api::client::result::ConnectorError; use aws_smithy_runtime_api::client::result::ConnectorError;
@@ -1113,32 +983,9 @@ mod tests {
retry: RemoteS3RetryPolicy::Disabled, retry: RemoteS3RetryPolicy::Disabled,
timeouts: SourceTimeouts::default(), timeouts: SourceTimeouts::default(),
bandwidth_limit: NonZeroU64::new(1_000_000), bandwidth_limit: NonZeroU64::new(1_000_000),
backend: SourceBackendSpec::S3,
} }
} }
#[cfg(not(feature = "gcs"))]
#[tokio::test]
async fn gcs_backend_not_compiled_keeps_hmac_s3_available() {
let mut native = spec(None);
native.provider = SourceProvider::GcsNative;
native.credentials = None;
native.backend = SourceBackendSpec::Gcs(GcsSourceSpec {
service_account_json: "{}".to_string(),
});
assert!(matches!(
SourceClient::new(&native).await,
Err(RemoteS3ClientError::BackendNotCompiled("gcs_native"))
));
let mut hmac = spec(None);
hmac.provider = SourceProvider::Gcs;
hmac.endpoint = "https://storage.googleapis.com".to_string();
SourceClient::new(&hmac)
.await
.expect("GCS HMAC uses the always-available S3 backend");
}
async fn scripted_client(spec: &SourceClientSpec, responses: Vec<Scripted>) -> (SourceClient, Recorded) { async fn scripted_client(spec: &SourceClientSpec, responses: Vec<Scripted>) -> (SourceClient, Recorded) {
let requests: Recorded = Arc::new(Mutex::new(Vec::new())); let requests: Recorded = Arc::new(Mutex::new(Vec::new()));
let connector = SharedHttpConnector::new(ScriptedConnector { let connector = SharedHttpConnector::new(ScriptedConnector {
@@ -1642,10 +1489,7 @@ mod tests {
#[tokio::test] #[tokio::test]
async fn source_error_classification_covers_every_class() { async fn source_error_classification_covers_every_class() {
let cases: Vec<(Scripted, &str, bool)> = vec![ let cases: Vec<(Scripted, &str, bool)> = vec![
(status(404, ""), "other", false), (status(404, ""), "not_found", false),
(status(404, "<Error><Code>NoSuchKey</Code></Error>"), "not_found", false),
(status(404, "<Error><Code>NoSuchBucket</Code></Error>"), "other", false),
(status(404, "<Error><Code>NoSuchVersion</Code></Error>"), "other", false),
(status(403, ACCESS_DENIED_BODY), "access_denied", false), (status(403, ACCESS_DENIED_BODY), "access_denied", false),
(status(401, ""), "access_denied", false), (status(401, ""), "access_denied", false),
(status(429, ""), "throttled", true), (status(429, ""), "throttled", true),
@@ -1668,35 +1512,14 @@ mod tests {
} }
} }
let (client, requests) = scripted_client(&spec(None), vec![status(404, ""), status(200, "")]).await; // HEAD carries no error body, so the classification must work from the
// status alone as well.
let (client, _) = scripted_client(&spec(None), vec![status(404, "")]).await;
assert!(matches!(client.head_object("missing").await, Err(SourceError::NotFound))); assert!(matches!(client.head_object("missing").await, Err(SourceError::NotFound)));
assert_eq!(recorded(&requests).len(), 2, "ambiguous HEAD 404 must check the bucket");
let (client, _) = scripted_client(&spec(None), vec![status(404, ""), status(404, "")]).await;
assert!(matches!(client.head_object("missing").await, Err(SourceError::Other(_))));
let (client, _) = scripted_client(&spec(None), vec![status(404, ""), status(403, "")]).await;
assert!(matches!(client.head_object("missing").await, Err(SourceError::AccessDenied)));
let (client, _) = scripted_client(&spec(None), vec![status(403, "")]).await; let (client, _) = scripted_client(&spec(None), vec![status(403, "")]).await;
assert!(matches!(client.head_object("secret").await, Err(SourceError::AccessDenied))); assert!(matches!(client.head_object("secret").await, Err(SourceError::AccessDenied)));
} }
#[test]
fn source_listing_rejects_missing_and_negative_sizes() {
for size in [None, Some(-1)] {
let object = SdkObject::builder().key("key").set_size(size).build();
assert!(matches!(s3_source_object(object), Err(SourceError::Other(_))));
}
assert!(matches!(
s3_source_object(SdkObject::builder().size(0).build()),
Err(SourceError::Other(_))
));
assert_eq!(
s3_source_object(SdkObject::builder().key("empty").size(0).build())
.expect("empty object")
.size,
0
);
}
#[tokio::test] #[tokio::test]
async fn source_client_debug_redacts_credentials() { async fn source_client_debug_redacts_credentials() {
let (client, _) = scripted_client(&spec(Some("data/")), Vec::new()).await; let (client, _) = scripted_client(&spec(Some("data/")), Vec::new()).await;
@@ -1763,102 +1586,7 @@ mod tests {
assert_eq!(resolve_path_style(PathStyle::VirtualHost, Minio, "10.0.0.1"), PathStyle::VirtualHost); assert_eq!(resolve_path_style(PathStyle::VirtualHost, Minio, "10.0.0.1"), PathStyle::VirtualHost);
assert_eq!(resolve_path_style(PathStyle::Path, Aws, "s3.amazonaws.com"), PathStyle::Path); assert_eq!(resolve_path_style(PathStyle::Path, Aws, "s3.amazonaws.com"), PathStyle::Path);
assert_eq!(SourceProvider::from_label(" AWS "), Some(Aws)); assert_eq!(SourceProvider::from_label(" AWS "), Some(Aws));
assert_eq!(SourceProvider::from_label(" Azure "), Some(Azure)); assert_eq!(SourceProvider::from_label("azure"), None);
assert_eq!(SourceProvider::from_label("gcs_native"), Some(GcsNative));
assert_eq!(SourceProvider::from_label("swift"), None);
}
const CONTRACT_LIST_PAGE_ONE: &str = r#"<?xml version="1.0" encoding="UTF-8"?>
<ListBucketResult xmlns="http://s3.amazonaws.com/doc/2006-03-01/">
<Name>source-bucket</Name>
<IsTruncated>true</IsTruncated>
<NextContinuationToken>cursor-1</NextContinuationToken>
<Contents>
<Key>dir/a.txt</Key>
<LastModified>2015-10-21T07:28:00.000Z</LastModified>
<ETag>&quot;5d41402abc4b2a76b9719d911017c592&quot;</ETag>
<Size>5</Size>
<StorageClass>STANDARD</StorageClass>
</Contents>
<CommonPrefixes><Prefix>dir/sub/</Prefix></CommonPrefixes>
</ListBucketResult>"#;
const CONTRACT_LIST_PAGE_TWO: &str = r#"<?xml version="1.0" encoding="UTF-8"?>
<ListBucketResult xmlns="http://s3.amazonaws.com/doc/2006-03-01/">
<Name>source-bucket</Name>
<IsTruncated>false</IsTruncated>
<Contents>
<Key>dir/b.txt</Key>
<LastModified>2015-10-21T07:28:00.000Z</LastModified>
<ETag>&quot;7d41402abc4b2a76b9719d911017c592&quot;</ETag>
<Size>7</Size>
</Contents>
</ListBucketResult>"#;
const CONTRACT_TAGGING: &str = r#"<?xml version="1.0" encoding="UTF-8"?>
<Tagging xmlns="http://s3.amazonaws.com/doc/2006-03-01/"><TagSet>
<Tag><Key>env</Key><Value>prod</Value></Tag>
</TagSet></Tagging>"#;
fn contract_object_headers(content_length: u64) -> Vec<(&'static str, String)> {
vec![
("etag", format!("\"{OBJECT_MD5}\"")),
("content-length", content_length.to_string()),
("content-type", "text/plain".to_string()),
("last-modified", "Wed, 21 Oct 2015 07:28:00 GMT".to_string()),
("x-amz-meta-owner", "alice".to_string()),
("x-amz-storage-class", "STANDARD".to_string()),
]
}
/// The S3 backend behind the scripted connector, without the prefix-mapping
/// client on top: the contract is a property of the backend itself.
async fn scripted_s3_backend(responses: Vec<Scripted>) -> S3SourceBackend {
let spec = spec(None);
let connector = SharedHttpConnector::new(ScriptedConnector {
requests: Arc::new(Mutex::new(Vec::new())),
responses: Arc::new(Mutex::new(responses.into_iter().collect())),
});
let http_client = http_client_fn(move |_settings, _components| connector.clone());
let endpoint = spec.endpoint_spec().expect("test spec endpoint should parse");
let config = build_remote_s3_config(&endpoint)
.await
.expect("test spec should build")
.http_client(http_client)
.interceptor(SourceProxyMarkerInterceptor::new());
S3SourceBackend {
client: S3Client::from_conf(config.build()),
bucket: spec.bucket.clone(),
}
}
#[tokio::test]
async fn s3_backend_satisfies_the_shared_backend_contract() {
let mut ranged = contract_object_headers(3);
ranged.push(("content-range", "bytes 1-3/5".to_string()));
let backend = scripted_s3_backend(vec![
ok(contract_object_headers(5), ""),
ok(contract_object_headers(5), "hello"),
ok(ranged, "ell"),
ok(Vec::new(), CONTRACT_LIST_PAGE_ONE),
ok(Vec::new(), CONTRACT_LIST_PAGE_TWO),
ok(Vec::new(), CONTRACT_TAGGING),
ok(Vec::new(), ""),
status(404, ""),
ok(Vec::new(), ""),
status(403, ACCESS_DENIED_BODY),
])
.await;
assert_backend_contract(
&backend,
BackendCapabilities {
etag_is_opaque: false,
supports_start_after: true,
supports_tagging: true,
},
)
.await;
} }
fn prefix_client(prefix: Option<String>) -> SourceClient { fn prefix_client(prefix: Option<String>) -> SourceClient {
@@ -19,12 +19,13 @@
//! [`SourceClient`], a circuit breaker, a negative cache, a per-key //! [`SourceClient`], a circuit breaker, a negative cache, a per-key
//! singleflight table, a pull concurrency limit and counters. Its lifecycle //! singleflight table, a pull concurrency limit and counters. Its lifecycle
//! follows the bucket metadata cache through the publish hook registered in //! follows the bucket metadata cache through the publish hook registered in
//! [`BUCKET_CONFIG_PUBLISH_HOOK`]; the hook fires on every cache install //! [`ON_DEMAND_MIGRATION_CONFIG_HOOK`]; the hook fires on every cache install
//! path (initial load, admin update, peer reload, refresh loop, lazy load). //! path (initial load, admin update, peer reload, refresh loop, lazy load).
//! //!
//! Change detection compares the config by value (`PartialEq`) rather than //! Change detection compares the config by value (`PartialEq`) rather than
//! by `updated_at`. The bucket incarnation is part of this comparison: //! by `updated_at`: the hook does not carry the timestamp, fetching it would
//! recreating a bucket must cancel old work even with identical configuration. //! re-enter the metadata system from inside its own publish path, and a
//! byte-identical config never needs a new client anyway.
//! //!
//! Client construction is async (TLS material may be read from disk), so //! Client construction is async (TLS material may be read from disk), so
//! the hook does not build inline: `publish` removes state synchronously and //! the hook does not build inline: `publish` removes state synchronously and
@@ -40,19 +41,17 @@
use super::backfill::{PriorityPullPermits, PullPermit, PullPriority}; use super::backfill::{PriorityPullPermits, PullPermit, PullPriority};
use super::breaker::{Breaker, BreakerState, BreakerTransition, BreakerVerdict}; use super::breaker::{Breaker, BreakerState, BreakerTransition, BreakerVerdict};
use super::config::{OnDemandMigrationConfig, PathStyle as ConfigPathStyle, Provider, SourceConfig}; use super::config::{
ON_DEMAND_MIGRATION_CONFIG_HOOK, OnDemandMigrationConfig, PathStyle as ConfigPathStyle, Provider, SourceConfig,
};
use super::list_through::{SOURCE_LIST_RATE_PER_SEC, SourceListRateLimiter}; use super::list_through::{SOURCE_LIST_RATE_PER_SEC, SourceListRateLimiter};
use super::negative_cache::NegativeCache; use super::negative_cache::NegativeCache;
use super::pull::{OdmWriteBack, PullQueue}; use super::pull::{OdmWriteBack, PullQueue};
use super::source_client::{ use super::source_client::{SourceClient, SourceClientSpec, SourceError, SourceProvider, SourceTimeouts};
AzureAuth, AzureSourceSpec, GcsSourceSpec, SourceBackendSpec, SourceClient, SourceClientSpec, SourceError, SourceProvider,
SourceTimeouts,
};
use super::stats::{GaugeGuard, OdmStats, OdmStatsSnapshot, PullFailureReason}; use super::stats::{GaugeGuard, OdmStats, OdmStatsSnapshot, PullFailureReason};
use super::storage_api::remote_s3_client::{ use crate::bucket::remote_s3_client::{
PathStyle as ClientPathStyle, RemoteCredentials, RemoteS3ClientError, RemoteS3RetryPolicy, PathStyle as ClientPathStyle, RemoteCredentials, RemoteS3ClientError, RemoteS3RetryPolicy,
}; };
use super::storage_api::{BUCKET_CONFIG_PUBLISH_HOOK, BUCKET_ON_DEMAND_MIGRATION_CONFIG};
use parking_lot::{Mutex, RwLock}; use parking_lot::{Mutex, RwLock};
use serde::{Deserialize, Serialize}; use serde::{Deserialize, Serialize};
use std::collections::HashMap; use std::collections::HashMap;
@@ -83,8 +82,6 @@ pub static GLOBAL_ON_DEMAND_MIGRATION_SYS: OnceLock<OnDemandMigrationSys> = Once
/// `resolve` as [`OdmLookup::Unavailable`] and through status snapshots. /// `resolve` as [`OdmLookup::Unavailable`] and through status snapshots.
#[derive(Clone, Debug, PartialEq, Eq, thiserror::Error)] #[derive(Clone, Debug, PartialEq, Eq, thiserror::Error)]
pub enum OdmStateError { pub enum OdmStateError {
#[error("the {0} backend is not included in this build")]
BackendNotCompiled(&'static str),
/// `source.credentials` is `null`; the shared client builder has no /// `source.credentials` is `null`; the shared client builder has no
/// anonymous mode yet (rustfs/backlog#2149 follow-up). /// anonymous mode yet (rustfs/backlog#2149 follow-up).
#[error("anonymous source access is not supported yet; configure source credentials")] #[error("anonymous source access is not supported yet; configure source credentials")]
@@ -269,7 +266,6 @@ impl Drop for InflightEntryGuard<'_> {
/// config change (counters excepted), removed when the config goes away. /// config change (counters excepted), removed when the config goes away.
pub struct BucketOdmState { pub struct BucketOdmState {
bucket: String, bucket: String,
incarnation_id: uuid::Uuid,
config: OnDemandMigrationConfig, config: OnDemandMigrationConfig,
applied_at: OffsetDateTime, applied_at: OffsetDateTime,
endpoint_host: String, endpoint_host: String,
@@ -307,24 +303,21 @@ impl BucketOdmState {
async fn build( async fn build(
bucket: &str, bucket: &str,
config: &OnDemandMigrationConfig, config: &OnDemandMigrationConfig,
incarnation_id: uuid::Uuid,
stats: Arc<OdmStats>, stats: Arc<OdmStats>,
write_back: Option<Arc<dyn OdmWriteBack>>, write_back: Option<Arc<dyn OdmWriteBack>>,
) -> Arc<Self> { ) -> Arc<Self> {
let spec = source_client_spec(config); let spec = source_client_spec(config);
let client = if config.source.credentials.is_none() && !config.source.provider.is_native() { let client = if config.source.credentials.is_none() {
Err(OdmStateError::AnonymousUnsupported) Err(OdmStateError::AnonymousUnsupported)
} else { } else {
SourceClient::new(&spec).await.map(Arc::new).map_err(|err| match err { SourceClient::new(&spec).await.map(Arc::new).map_err(|err| match err {
RemoteS3ClientError::MissingCredentials => OdmStateError::AnonymousUnsupported, RemoteS3ClientError::MissingCredentials => OdmStateError::AnonymousUnsupported,
RemoteS3ClientError::BackendNotCompiled(provider) => OdmStateError::BackendNotCompiled(provider),
other => OdmStateError::ClientBuild(other.to_string()), other => OdmStateError::ClientBuild(other.to_string()),
}) })
}; };
let policy = &config.policy; let policy = &config.policy;
Arc::new(Self { Arc::new(Self {
bucket: bucket.to_string(), bucket: bucket.to_string(),
incarnation_id,
endpoint_host: endpoint_host(&config.source), endpoint_host: endpoint_host(&config.source),
config: config.clone(), config: config.clone(),
applied_at: OffsetDateTime::now_utc(), applied_at: OffsetDateTime::now_utc(),
@@ -342,18 +335,10 @@ impl BucketOdmState {
}) })
} }
pub fn filter_incarnation(self: Arc<Self>, incarnation_id: uuid::Uuid) -> Option<Arc<Self>> {
(self.incarnation_id == incarnation_id && !self.is_cancelled()).then_some(self)
}
pub fn bucket(&self) -> &str { pub fn bucket(&self) -> &str {
&self.bucket &self.bucket
} }
pub fn incarnation_id(&self) -> uuid::Uuid {
self.incarnation_id
}
pub fn config(&self) -> &OnDemandMigrationConfig { pub fn config(&self) -> &OnDemandMigrationConfig {
&self.config &self.config
} }
@@ -634,7 +619,6 @@ pub fn source_client_spec(config: &OnDemandMigrationConfig) -> SourceClientSpec
// load on a source that is already failing. // load on a source that is already failing.
retry: RemoteS3RetryPolicy::Disabled, retry: RemoteS3RetryPolicy::Disabled,
bandwidth_limit: policy.bandwidth_limit_bytes_per_sec.and_then(NonZeroU64::new), bandwidth_limit: policy.bandwidth_limit_bytes_per_sec.and_then(NonZeroU64::new),
backend: source_backend_spec(source),
} }
} }
@@ -646,31 +630,6 @@ fn source_provider(provider: Provider) -> SourceProvider {
Provider::Rustfs => SourceProvider::Rustfs, Provider::Rustfs => SourceProvider::Rustfs,
Provider::R2 => SourceProvider::R2, Provider::R2 => SourceProvider::R2,
Provider::Gcs => SourceProvider::Gcs, Provider::Gcs => SourceProvider::Gcs,
Provider::Azure => SourceProvider::Azure,
Provider::GcsNative => SourceProvider::GcsNative,
}
}
/// Which backend the client builds. A native provider whose block is missing
/// falls back to the S3 spec, where the builder reports the missing
/// credentials: the config layer already refuses to store that shape, so this
/// only covers a config written by an older or hand-edited build.
pub fn source_backend_spec(source: &SourceConfig) -> SourceBackendSpec {
match (source.provider, source.azure.as_ref(), source.gcs.as_ref()) {
(Provider::Azure, Some(azure), _) => SourceBackendSpec::Azure(AzureSourceSpec {
account: azure.account.clone(),
auth: match (&azure.account_key, &azure.sas_token) {
(Some(key), _) => AzureAuth::SharedKey(key.clone()),
(None, Some(sas)) => AzureAuth::Sas(sas.clone()),
// Refused by `SourceConfig::validate`; an empty shared key
// fails closed at the builder rather than signing with none.
(None, None) => AzureAuth::SharedKey(String::new()),
},
}),
(Provider::GcsNative, _, Some(gcs)) => SourceBackendSpec::Gcs(GcsSourceSpec {
service_account_json: gcs.service_account_json.clone(),
}),
_ => SourceBackendSpec::S3,
} }
} }
@@ -749,51 +708,21 @@ impl OnDemandMigrationSys {
/// Registers `publish` as the bucket-metadata publish hook. Returns /// Registers `publish` as the bucket-metadata publish hook. Returns
/// `false` when a hook was already registered. /// `false` when a hook was already registered.
pub fn register_config_hook(&'static self) -> bool { pub fn register_config_hook(&'static self) -> bool {
BUCKET_CONFIG_PUBLISH_HOOK ON_DEMAND_MIGRATION_CONFIG_HOOK
.set(Box::new(move |bucket, config_file, stored| { .set(Box::new(move |bucket, config| self.publish(bucket, config)))
if config_file == BUCKET_ON_DEMAND_MIGRATION_CONFIG {
self.publish_stored(bucket, stored.map(|(bytes, _, incarnation)| (bytes, incarnation)));
}
}))
.is_ok() .is_ok()
} }
/// Corrupt persisted bytes withdraw state synchronously, just like deletion.
fn publish_stored(&'static self, bucket: &str, stored: Option<(&[u8], uuid::Uuid)>) {
let incarnation_id = stored.map(|(_, id)| id).unwrap_or_default();
match stored.map(|(bytes, _)| OnDemandMigrationConfig::from_json(bytes)).transpose() {
Ok(config) => self.publish_for_incarnation(bucket, incarnation_id, config.as_ref()),
Err(err) => {
warn!(
event = EVENT_ODM_BUCKET_STATE_APPLIED,
component = LOG_COMPONENT_ECSTORE,
subsystem = LOG_SUBSYSTEM_ON_DEMAND_MIGRATION,
result = "invalid",
bucket = %bucket,
error = %err,
"Failed to parse on-demand migration config"
);
self.publish_for_incarnation(bucket, incarnation_id, None);
}
}
}
/// Hook entry point: removals apply immediately, installs are spawned /// Hook entry point: removals apply immediately, installs are spawned
/// (client construction is async). Requires a Tokio runtime for the /// (client construction is async). Requires a Tokio runtime for the
/// install path; without one the config is logged and skipped. /// install path; without one the config is logged and skipped.
pub fn publish_for_incarnation( pub fn publish(&'static self, bucket: &str, config: Option<&OnDemandMigrationConfig>) {
&'static self, let generation = self.next_generation();
bucket: &str, let Some(config) = self.desired(config) else {
incarnation_id: uuid::Uuid,
config: Option<&OnDemandMigrationConfig>,
) {
let config = self.desired(config).filter(|_| !incarnation_id.is_nil());
let generation = self.reserve_generation(bucket, config.is_some());
let Some(config) = config else {
self.remove_with_generation(bucket, generation); self.remove_with_generation(bucket, generation);
return; return;
}; };
if self.is_unchanged(bucket, incarnation_id, config, generation) { if self.is_unchanged(bucket, config, generation) {
return; return;
} }
let Ok(handle) = tokio::runtime::Handle::try_current() else { let Ok(handle) = tokio::runtime::Handle::try_current() else {
@@ -812,49 +741,31 @@ impl OnDemandMigrationSys {
let bucket = bucket.to_string(); let bucket = bucket.to_string();
let config = config.clone(); let config = config.clone();
handle.spawn(async move { handle.spawn(async move {
self.apply_with_generation(&bucket, incarnation_id, Some(&config), generation) self.apply_with_generation(&bucket, Some(&config), generation).await;
.await;
}); });
} }
/// Installs, rebuilds, or removes the bucket state for `config`. /// Installs, rebuilds, or removes the bucket state for `config`.
/// Idempotent: the same config on an installed bucket is a no-op. /// Idempotent: the same config on an installed bucket is a no-op.
#[cfg(test)]
pub async fn apply(&self, bucket: &str, config: Option<&OnDemandMigrationConfig>) -> ApplyOutcome { pub async fn apply(&self, bucket: &str, config: Option<&OnDemandMigrationConfig>) -> ApplyOutcome {
self.apply_for_incarnation(bucket, uuid::Uuid::from_u128(1), config).await let generation = self.next_generation();
} self.apply_with_generation(bucket, config, generation).await
#[cfg(test)]
pub fn publish(&'static self, bucket: &str, config: Option<&OnDemandMigrationConfig>) {
self.publish_for_incarnation(bucket, uuid::Uuid::from_u128(1), config);
}
pub async fn apply_for_incarnation(
&self,
bucket: &str,
incarnation_id: uuid::Uuid,
config: Option<&OnDemandMigrationConfig>,
) -> ApplyOutcome {
let config = self.desired(config).filter(|_| !incarnation_id.is_nil());
let generation = self.reserve_generation(bucket, config.is_some());
self.apply_with_generation(bucket, incarnation_id, config, generation).await
} }
async fn apply_with_generation( async fn apply_with_generation(
&self, &self,
bucket: &str, bucket: &str,
incarnation_id: uuid::Uuid,
config: Option<&OnDemandMigrationConfig>, config: Option<&OnDemandMigrationConfig>,
generation: u64, generation: u64,
) -> ApplyOutcome { ) -> ApplyOutcome {
let Some(config) = self.desired(config) else { let Some(config) = self.desired(config) else {
return self.remove_with_generation(bucket, generation); return self.remove_with_generation(bucket, generation);
}; };
if self.is_unchanged(bucket, incarnation_id, config, generation) { if self.is_unchanged(bucket, config, generation) {
return ApplyOutcome::Unchanged; return ApplyOutcome::Unchanged;
} }
let stats = self.state(bucket).map(|state| Arc::clone(&state.stats)).unwrap_or_default(); let stats = self.state(bucket).map(|state| Arc::clone(&state.stats)).unwrap_or_default();
let state = BucketOdmState::build(bucket, config, incarnation_id, stats, self.write_back()).await; let state = BucketOdmState::build(bucket, config, stats, self.write_back()).await;
let (outcome, previous) = { let (outcome, previous) = {
let mut buckets = self.buckets.write(); let mut buckets = self.buckets.write();
@@ -898,13 +809,12 @@ impl OnDemandMigrationSys {
/// Removes a bucket's state (idempotent), cancelling its token. /// Removes a bucket's state (idempotent), cancelling its token.
pub fn remove(&self, bucket: &str) -> ApplyOutcome { pub fn remove(&self, bucket: &str) -> ApplyOutcome {
let generation = self.reserve_generation(bucket, false); let generation = self.next_generation();
self.remove_with_generation(bucket, generation) self.remove_with_generation(bucket, generation)
} }
/// One-shot lookup: module switch, bucket state, prefix filter, /// One-shot lookup: module switch, bucket state, prefix filter,
/// client availability, negative cache, breaker, in that order. /// client availability, negative cache, breaker, in that order.
#[cfg(test)]
pub fn resolve(&self, bucket: &str, key: &str) -> Option<OdmLookup> { pub fn resolve(&self, bucket: &str, key: &str) -> Option<OdmLookup> {
if !self.is_module_enabled() { if !self.is_module_enabled() {
return None; return None;
@@ -912,13 +822,6 @@ impl OnDemandMigrationSys {
self.state(bucket)?.resolve_key(key) self.state(bucket)?.resolve_key(key)
} }
pub fn resolve_for_incarnation(&self, bucket: &str, key: &str, incarnation_id: uuid::Uuid) -> Option<OdmLookup> {
if !self.is_module_enabled() {
return None;
}
self.state(bucket)?.filter_incarnation(incarnation_id)?.resolve_key(key)
}
pub fn state(&self, bucket: &str) -> Option<Arc<BucketOdmState>> { pub fn state(&self, bucket: &str) -> Option<Arc<BucketOdmState>> {
self.buckets.read().get(bucket).and_then(|slot| slot.state.clone()) self.buckets.read().get(bucket).and_then(|slot| slot.state.clone())
} }
@@ -947,17 +850,8 @@ impl OnDemandMigrationSys {
snapshots snapshots
} }
fn reserve_generation(&self, bucket: &str, installing: bool) -> u64 { fn next_generation(&self) -> u64 {
// Reserve a desired install before its async client build, under the self.generation.fetch_add(1, Ordering::Relaxed) + 1
// same lock that orders removals. Unconfigured buckets need no slot.
let mut buckets = self.buckets.write();
let generation = self.generation.fetch_add(1, Ordering::Relaxed) + 1;
if installing {
buckets.entry(bucket.to_string()).or_default().generation = generation;
} else if let Some(slot) = buckets.get_mut(bucket) {
slot.generation = generation;
}
generation
} }
fn desired<'c>(&self, config: Option<&'c OnDemandMigrationConfig>) -> Option<&'c OnDemandMigrationConfig> { fn desired<'c>(&self, config: Option<&'c OnDemandMigrationConfig>) -> Option<&'c OnDemandMigrationConfig> {
@@ -966,26 +860,15 @@ impl OnDemandMigrationSys {
/// Claims `generation` for the bucket when the installed state already /// Claims `generation` for the bucket when the installed state already
/// matches `config` and has a usable client. /// matches `config` and has a usable client.
fn is_unchanged(&self, bucket: &str, incarnation_id: uuid::Uuid, config: &OnDemandMigrationConfig, generation: u64) -> bool { fn is_unchanged(&self, bucket: &str, config: &OnDemandMigrationConfig, generation: u64) -> bool {
let mut buckets = self.buckets.write(); let mut buckets = self.buckets.write();
let Some(slot) = buckets.get_mut(bucket) else { let Some(slot) = buckets.get_mut(bucket) else {
return false; return false;
}; };
if slot.generation > generation {
return false;
}
if slot
.state
.as_ref()
.is_some_and(|state| state.incarnation_id != incarnation_id)
&& let Some(previous) = slot.state.take()
{
previous.cancel.cancel();
}
let unchanged = slot let unchanged = slot
.state .state
.as_ref() .as_ref()
.is_some_and(|state| state.client.is_ok() && state.incarnation_id == incarnation_id && state.config == *config); .is_some_and(|state| state.client.is_ok() && state.config == *config);
if unchanged && slot.generation < generation { if unchanged && slot.generation < generation {
slot.generation = generation; slot.generation = generation;
} }
@@ -1025,8 +908,8 @@ impl OnDemandMigrationSys {
#[cfg(test)] #[cfg(test)]
mod tests { mod tests {
use super::*; use super::*;
use crate::on_demand_migration::breaker::BREAKER_FAILURE_THRESHOLD; use crate::bucket::on_demand_migration::breaker::BREAKER_FAILURE_THRESHOLD;
use crate::on_demand_migration::config::{FilterConfig, PolicyConfig, SourceCredentials, SourceTimeout, TlsConfig}; use crate::bucket::on_demand_migration::config::{FilterConfig, PolicyConfig, SourceCredentials, SourceTimeout, TlsConfig};
use std::sync::atomic::AtomicUsize; use std::sync::atomic::AtomicUsize;
use tokio::sync::Barrier; use tokio::sync::Barrier;
@@ -1046,8 +929,6 @@ mod tests {
session_token: None, session_token: None,
}), }),
tls: TlsConfig::default(), tls: TlsConfig::default(),
azure: None,
gcs: None,
}, },
filter: FilterConfig { filter: FilterConfig {
prefix: prefix.map(str::to_string), prefix: prefix.map(str::to_string),
@@ -1164,45 +1045,6 @@ mod tests {
assert_eq!(sys.apply("b", Some(&cfg)).await, ApplyOutcome::Rebuilt); assert_eq!(sys.apply("b", Some(&cfg)).await, ApplyOutcome::Rebuilt);
} }
#[tokio::test]
async fn native_azure_uses_provider_credentials_without_s3_credentials() {
let sys = enabled_sys();
let mut cfg = config(None);
cfg.source.provider = Provider::Azure;
cfg.source.endpoint = None;
cfg.source.credentials = None;
cfg.source.azure = Some(super::super::config::AzureSourceConfig {
account: "legacyaccount".to_string(),
account_key: Some("c2VjcmV0LWtleQ==".to_string()),
sas_token: None,
});
assert_eq!(sys.apply("b", Some(&cfg)).await, ApplyOutcome::Installed);
let state = ready_state(sys.resolve("b", "k"));
assert!(state.client().is_ok(), "native credentials must not be classified as anonymous S3");
}
#[cfg(not(feature = "gcs"))]
#[tokio::test]
async fn gcs_backend_not_compiled_is_unavailable_not_anonymous() {
let sys = enabled_sys();
let mut cfg = config(None);
cfg.source.provider = Provider::GcsNative;
cfg.source.credentials = None;
cfg.source.gcs = Some(super::super::config::GcsSourceConfig {
service_account_json: "{}".to_string(),
});
let encoded = cfg.to_json().expect("GCS config is serializable without the backend");
let restored: OnDemandMigrationConfig = serde_json::from_slice(&encoded).expect("GCS config stays readable");
assert_eq!(restored, cfg);
assert_eq!(sys.apply("b", Some(&cfg)).await, ApplyOutcome::Installed);
match sys.resolve("b", "k") {
Some(OdmLookup::Unavailable { error, .. }) => {
assert_eq!(error, OdmStateError::BackendNotCompiled("gcs_native"));
}
other => panic!("expected unavailable backend, got {other:?}"),
}
}
#[tokio::test(flavor = "multi_thread", worker_threads = 4)] #[tokio::test(flavor = "multi_thread", worker_threads = 4)]
async fn singleflight_admits_one_leader_per_key() { async fn singleflight_admits_one_leader_per_key() {
let sys = enabled_sys(); let sys = enabled_sys();
@@ -1418,128 +1260,24 @@ mod tests {
assert!(state.is_cancelled()); assert!(state.is_cancelled());
} }
#[tokio::test]
async fn identical_config_on_recreated_bucket_cancels_old_state() {
let sys = enabled_sys();
let cfg = config(None);
let old_id = uuid::Uuid::new_v4();
let new_id = uuid::Uuid::new_v4();
sys.apply_for_incarnation("recreated", old_id, Some(&cfg)).await;
let old = sys.state("recreated").expect("old state installed");
assert!(sys.resolve_for_incarnation("recreated", "key", new_id).is_none());
sys.apply_for_incarnation("recreated", new_id, Some(&cfg)).await;
let replacement = sys.state("recreated").expect("replacement state installed");
assert!(old.is_cancelled());
assert!(!Arc::ptr_eq(&old, &replacement));
assert_eq!(replacement.incarnation_id(), new_id);
assert!(sys.resolve_for_incarnation("recreated", "key", old_id).is_none());
assert!(sys.resolve_for_incarnation("recreated", "key", new_id).is_some());
}
#[tokio::test]
async fn changed_delete_marker_policy_withdraws_the_captured_lookup() {
let sys = enabled_sys();
let incarnation = uuid::Uuid::new_v4();
let mut cfg = config(None);
cfg.policy.respect_local_delete_marker = false;
sys.apply_for_incarnation("policy-snapshot", incarnation, Some(&cfg)).await;
let captured = sys.state("policy-snapshot").expect("policy A installed");
assert!(!captured.config().policy.respect_local_delete_marker);
cfg.policy.respect_local_delete_marker = true;
sys.apply_for_incarnation("policy-snapshot", incarnation, Some(&cfg)).await;
let replacement = sys.state("policy-snapshot").expect("policy B installed");
assert!(replacement.config().policy.respect_local_delete_marker);
assert!(captured.is_cancelled());
assert!(
captured
.filter_incarnation(incarnation)
.and_then(|state| state.resolve_key("key"))
.is_none(),
"a request that evaluated policy A cannot continue through policy B"
);
assert!(
replacement
.clone()
.filter_incarnation(incarnation)
.and_then(|state| state.resolve_key("key"))
.is_some()
);
assert_eq!(
replacement
.stats()
.snapshot(replacement.breaker().state())
.source_latency
.count,
0
);
}
#[tokio::test]
async fn missing_incarnation_cannot_install_or_retain_source_state() {
let sys: &'static OnDemandMigrationSys = Box::leak(Box::new(enabled_sys()));
let cfg = config(None);
assert_eq!(
sys.apply_for_incarnation("missing", uuid::Uuid::nil(), Some(&cfg)).await,
ApplyOutcome::NotDesired
);
sys.publish_for_incarnation("missing", uuid::Uuid::nil(), Some(&cfg));
assert!(sys.state("missing").is_none());
sys.apply_for_incarnation("missing", uuid::Uuid::new_v4(), Some(&cfg)).await;
let state = sys.state("missing").expect("valid identity installed");
sys.publish_for_incarnation("missing", uuid::Uuid::nil(), Some(&cfg));
assert!(sys.state("missing").is_none());
assert!(state.is_cancelled());
}
#[tokio::test]
async fn corrupt_stored_config_withdraws_runtime_state() {
let sys: &'static OnDemandMigrationSys = Box::leak(Box::new(enabled_sys()));
let cfg = config(None);
assert_eq!(sys.apply("corrupt", Some(&cfg)).await, ApplyOutcome::Installed);
let state = sys.state("corrupt").expect("state installed");
sys.publish_stored("corrupt", Some((b"not-json", uuid::Uuid::from_u128(1))));
assert!(sys.state("corrupt").is_none(), "corruption cannot keep an older source active");
assert!(state.is_cancelled(), "corruption cancels in-flight work");
}
#[tokio::test]
async fn absent_config_updates_do_not_allocate_bucket_slots() {
let sys = enabled_sys();
for index in 0..1000 {
let bucket = format!("unconfigured-{index}");
assert_eq!(sys.apply(&bucket, None).await, ApplyOutcome::NotDesired);
assert_eq!(sys.remove(&bucket), ApplyOutcome::NotDesired);
}
assert!(sys.buckets.read().is_empty(), "unconfigured buckets must not accumulate tombstones");
}
#[tokio::test] #[tokio::test]
async fn stale_install_cannot_overwrite_a_later_removal() { async fn stale_install_cannot_overwrite_a_later_removal() {
let sys = enabled_sys(); let sys = enabled_sys();
let cfg = config(None); let cfg = config(None);
let older = sys.reserve_generation("b", true); let older = sys.next_generation();
let newer = sys.reserve_generation("b", false); let newer = sys.next_generation();
assert_eq!(sys.remove_with_generation("b", newer), ApplyOutcome::NotDesired); assert_eq!(sys.remove_with_generation("b", newer), ApplyOutcome::NotDesired);
assert_eq!( // The removal above did not create a slot; simulate an install that
sys.apply_with_generation("b", uuid::Uuid::from_u128(1), Some(&cfg), older) // started before it and finishes after.
.await, sys.apply_with_generation("b", Some(&cfg), older).await;
ApplyOutcome::Superseded assert!(sys.state("b").is_some(), "no slot yet, so the older install lands");
);
assert!(sys.state("b").is_none(), "removal must supersede an in-flight first install");
assert_eq!(sys.apply("b", Some(&cfg)).await, ApplyOutcome::Installed);
let installed = sys.state("b").unwrap(); let installed = sys.state("b").unwrap();
let older = sys.reserve_generation("b", true); let older = sys.next_generation();
let newer = sys.reserve_generation("b", false); let newer = sys.next_generation();
assert_eq!(sys.remove_with_generation("b", newer), ApplyOutcome::Removed); assert_eq!(sys.remove_with_generation("b", newer), ApplyOutcome::Removed);
assert!(installed.is_cancelled()); assert!(installed.is_cancelled());
assert_eq!( assert_eq!(sys.apply_with_generation("b", Some(&cfg), older).await, ApplyOutcome::Superseded);
sys.apply_with_generation("b", uuid::Uuid::from_u128(1), Some(&cfg), older)
.await,
ApplyOutcome::Superseded
);
assert!(sys.state("b").is_none(), "the stale install is discarded"); assert!(sys.state("b").is_none(), "the stale install is discarded");
} }
@@ -180,8 +180,6 @@ impl RemoteS3EndpointSpec {
#[derive(Debug, thiserror::Error)] #[derive(Debug, thiserror::Error)]
pub enum RemoteS3ClientError { pub enum RemoteS3ClientError {
#[error("the {0} backend is not included in this build")]
BackendNotCompiled(&'static str),
#[error("remote endpoint requires credentials")] #[error("remote endpoint requires credentials")]
MissingCredentials, MissingCredentials,
#[error("{0}")] #[error("{0}")]
@@ -283,7 +281,9 @@ impl Intercept for UserAgentSuffixInterceptor {
/// Builds the SDK config for `spec` without finalizing it, so callers can add /// Builds the SDK config for `spec` without finalizing it, so callers can add
/// interceptors or (in tests) swap the HTTP client before `build()`. /// interceptors or (in tests) swap the HTTP client before `build()`.
pub async fn build_remote_s3_config(spec: &RemoteS3EndpointSpec) -> Result<aws_sdk_s3::config::Builder, RemoteS3ClientError> { pub(crate) async fn build_remote_s3_config(
spec: &RemoteS3EndpointSpec,
) -> Result<aws_sdk_s3::config::Builder, RemoteS3ClientError> {
let Some(credentials) = &spec.credentials else { let Some(credentials) = &spec.credentials else {
return Err(RemoteS3ClientError::MissingCredentials); return Err(RemoteS3ClientError::MissingCredentials);
}; };
@@ -523,7 +523,7 @@ fn validate_ca_pem_bundle(ca_cert_pem: &[u8]) -> Result<(), String> {
Ok(()) Ok(())
} }
pub fn validate_target_ca_pem(ca_cert_pem: &str) -> Result<(), RemoteS3ClientError> { pub(crate) fn validate_target_ca_pem(ca_cert_pem: &str) -> Result<(), RemoteS3ClientError> {
validate_ca_pem_bundle(ca_cert_pem.as_bytes()).map_err(RemoteS3ClientError::InvalidCaPem) validate_ca_pem_bundle(ca_cert_pem.as_bytes()).map_err(RemoteS3ClientError::InvalidCaPem)
} }
-3
View File
@@ -956,9 +956,6 @@ pub struct ObjectOptions {
pub preserve_etag: Option<String>, pub preserve_etag: Option<String>,
pub metadata_chg: bool, pub metadata_chg: bool,
pub http_preconditions: Option<HTTPPreconditions>, pub http_preconditions: Option<HTTPPreconditions>,
/// Internal create-only writes may also preserve an acknowledged deletion.
/// Evaluated with `http_preconditions` under the namespace commit lock.
pub preserve_delete_marker: bool,
pub delete_replication: Option<ReplicationState>, pub delete_replication: Option<ReplicationState>,
pub delete_replication_config_snapshot: Option<Arc<DeleteReplicationConfigSnapshot>>, pub delete_replication_config_snapshot: Option<Arc<DeleteReplicationConfigSnapshot>>,
-112
View File
@@ -78,21 +78,6 @@ pub(crate) struct ScannerPublicationLeaseEntry {
pub(crate) _operation_guard: OwnedRwLockReadGuard<()>, pub(crate) _operation_guard: OwnedRwLockReadGuard<()>,
} }
pub(crate) struct NamespaceCommitGuard {
ctx: Arc<InstanceContext>,
counted: bool,
}
impl Drop for NamespaceCommitGuard {
fn drop(&mut self) {
if self.counted {
// Publish the new generation before a zero-pending publication probe.
self.ctx.advance_namespace_commit_generation();
self.ctx.namespace_commits.fetch_sub(1, Ordering::AcqRel);
}
}
}
/// Runtime state owned by a single `ECStore` instance. /// Runtime state owned by a single `ECStore` instance.
/// ///
/// This is intentionally minimal in the first migration slice; subsequent /// This is intentionally minimal in the first migration slice; subsequent
@@ -224,13 +209,9 @@ pub struct InstanceContext {
/// Last storage-owned movement snapshot observed under the operation /// Last storage-owned movement snapshot observed under the operation
/// gate. SetDisks cache writers fail closed until ECStore refreshes it. /// gate. SetDisks cache writers fail closed until ECStore refreshes it.
scanner_publication_state: AtomicU8, scanner_publication_state: AtomicU8,
namespace_commits: AtomicU64,
namespace_commit_generation: AtomicU64,
/// Resolves object-encryption material at the application boundary. /// Resolves object-encryption material at the application boundary.
object_encryption_resolver: OnceLock<Arc<dyn ObjectEncryptionResolver>>, object_encryption_resolver: OnceLock<Arc<dyn ObjectEncryptionResolver>>,
tier_delete_journal_recovery_stores: std::sync::Mutex<HashSet<Uuid>>, tier_delete_journal_recovery_stores: std::sync::Mutex<HashSet<Uuid>>,
#[cfg(test)]
suppress_tier_delete_journal_recovery: bool,
transition_transaction_recovery_stores: std::sync::Mutex<HashSet<Uuid>>, transition_transaction_recovery_stores: std::sync::Mutex<HashSet<Uuid>>,
tier_delete_journal_recovery_wakeup: tokio::sync::Notify, tier_delete_journal_recovery_wakeup: tokio::sync::Notify,
} }
@@ -275,12 +256,8 @@ impl InstanceContext {
data_movement_generation_exhausted: AtomicBool::new(false), data_movement_generation_exhausted: AtomicBool::new(false),
data_movement_generation_notify: Arc::new(Notify::new()), data_movement_generation_notify: Arc::new(Notify::new()),
scanner_publication_state: AtomicU8::new(SCANNER_PUBLICATION_STATE_UNKNOWN), scanner_publication_state: AtomicU8::new(SCANNER_PUBLICATION_STATE_UNKNOWN),
namespace_commits: AtomicU64::new(0),
namespace_commit_generation: AtomicU64::new(0),
object_encryption_resolver: OnceLock::new(), object_encryption_resolver: OnceLock::new(),
tier_delete_journal_recovery_stores: std::sync::Mutex::new(HashSet::new()), tier_delete_journal_recovery_stores: std::sync::Mutex::new(HashSet::new()),
#[cfg(test)]
suppress_tier_delete_journal_recovery: false,
transition_transaction_recovery_stores: std::sync::Mutex::new(HashSet::new()), transition_transaction_recovery_stores: std::sync::Mutex::new(HashSet::new()),
tier_delete_journal_recovery_wakeup: tokio::sync::Notify::new(), tier_delete_journal_recovery_wakeup: tokio::sync::Notify::new(),
} }
@@ -408,36 +385,6 @@ impl InstanceContext {
&& self.scanner_publication_state.load(Ordering::Acquire) == SCANNER_PUBLICATION_STATE_ALLOWED && self.scanner_publication_state.load(Ordering::Acquire) == SCANNER_PUBLICATION_STATE_ALLOWED
} }
pub(crate) fn begin_namespace_commit(self: &Arc<Self>) -> Arc<NamespaceCommitGuard> {
let counted = self
.namespace_commits
.fetch_update(Ordering::AcqRel, Ordering::Acquire, |count| count.checked_add(1))
.is_ok();
if counted {
self.advance_namespace_commit_generation();
} else {
self.namespace_commit_generation.store(u64::MAX, Ordering::Release);
}
Arc::new(NamespaceCommitGuard {
ctx: Arc::clone(self),
counted,
})
}
fn advance_namespace_commit_generation(&self) {
let _ = self
.namespace_commit_generation
.fetch_update(Ordering::AcqRel, Ordering::Acquire, |generation| Some(generation.saturating_add(1)));
}
pub(crate) fn namespace_commit_generation(&self) -> u64 {
self.namespace_commit_generation.load(Ordering::Acquire)
}
pub(crate) fn namespace_commits_pending(&self) -> bool {
self.namespace_commits.load(Ordering::Acquire) != 0 || self.namespace_commit_generation() == u64::MAX
}
pub(crate) fn set_scanner_publication_state(&self, blocked: bool) { pub(crate) fn set_scanner_publication_state(&self, blocked: bool) {
self.scanner_publication_state.store( self.scanner_publication_state.store(
if blocked { if blocked {
@@ -693,21 +640,12 @@ impl InstanceContext {
} }
pub(crate) fn mark_tier_delete_journal_recovery_started(&self, store_id: Uuid) -> bool { pub(crate) fn mark_tier_delete_journal_recovery_started(&self, store_id: Uuid) -> bool {
#[cfg(test)]
if self.suppress_tier_delete_journal_recovery {
return false;
}
self.tier_delete_journal_recovery_stores self.tier_delete_journal_recovery_stores
.lock() .lock()
.unwrap_or_else(std::sync::PoisonError::into_inner) .unwrap_or_else(std::sync::PoisonError::into_inner)
.insert(store_id) .insert(store_id)
} }
#[cfg(test)]
pub(crate) fn suppress_tier_delete_journal_recovery_for_test(&mut self) {
self.suppress_tier_delete_journal_recovery = true;
}
pub(crate) fn mark_transition_transaction_recovery_started(&self, store_id: Uuid) -> bool { pub(crate) fn mark_transition_transaction_recovery_started(&self, store_id: Uuid) -> bool {
self.transition_transaction_recovery_stores self.transition_transaction_recovery_stores
.lock() .lock()
@@ -818,50 +756,6 @@ pub fn bootstrap_ctx() -> Arc<InstanceContext> {
mod tests { mod tests {
use super::*; use super::*;
#[test]
fn namespace_commit_guards_are_instance_local_and_count_until_last_owner() {
let first = Arc::new(InstanceContext::new());
let other = Arc::new(InstanceContext::new());
first.set_scanner_publication_state(false);
other.set_scanner_publication_state(false);
assert!(first.scanner_publication_state_allowed());
let one = first.begin_namespace_commit();
let shared_owner = Arc::clone(&one);
let two = first.begin_namespace_commit();
assert!(first.namespace_commits_pending());
assert!(first.scanner_publication_state_allowed(), "pending writes must not block scan admission");
assert_eq!(first.namespace_commit_generation(), 2);
assert!(!other.namespace_commits_pending());
assert_eq!(other.namespace_commit_generation(), 0);
assert!(other.scanner_publication_state_allowed());
drop(one);
assert_eq!(first.namespace_commit_generation(), 2);
drop(shared_owner);
assert!(first.namespace_commits_pending());
assert_eq!(first.namespace_commit_generation(), 3);
drop(two);
assert!(!first.namespace_commits_pending());
assert_eq!(first.namespace_commit_generation(), 4);
assert!(first.scanner_publication_state_allowed());
}
#[test]
fn namespace_commit_counter_exhaustion_keeps_publication_blocked() {
for (count, generation) in [(0, u64::MAX - 1), (u64::MAX, 0)] {
let ctx = Arc::new(InstanceContext::new());
ctx.set_scanner_publication_state(false);
ctx.namespace_commits.store(count, Ordering::Release);
ctx.namespace_commit_generation.store(generation, Ordering::Release);
let guard = ctx.begin_namespace_commit();
assert!(ctx.namespace_commits_pending());
assert_eq!(ctx.namespace_commit_generation(), u64::MAX);
drop(guard);
assert!(ctx.namespace_commits_pending());
assert_eq!(ctx.namespace_commit_generation(), u64::MAX);
assert_eq!(ctx.namespace_commits.load(Ordering::Acquire), count);
}
}
// The SetupType inputs must derive the exact (is_erasure, // The SetupType inputs must derive the exact (is_erasure,
// is_dist_erasure, is_erasure_sd) triples that the original three // is_dist_erasure, is_erasure_sd) triples that the original three
// process-global erasure bools produced via update_erasure_type(). // process-global erasure bools produced via update_erasure_type().
@@ -1179,12 +1073,6 @@ mod tests {
assert!(!ctx_a.mark_tier_delete_journal_recovery_started(store_a)); assert!(!ctx_a.mark_tier_delete_journal_recovery_started(store_a));
assert!(ctx_a.mark_tier_delete_journal_recovery_started(store_b)); assert!(ctx_a.mark_tier_delete_journal_recovery_started(store_b));
assert!(ctx_b.mark_tier_delete_journal_recovery_started(store_a)); assert!(ctx_b.mark_tier_delete_journal_recovery_started(store_a));
let mut manual_ctx = InstanceContext::new();
manual_ctx.suppress_tier_delete_journal_recovery_for_test();
assert!(!manual_ctx.mark_tier_delete_journal_recovery_started(store_a));
assert!(!manual_ctx.mark_tier_delete_journal_recovery_started(store_b));
assert!(ctx_b.mark_tier_delete_journal_recovery_started(store_b));
} }
#[test] #[test]
-1
View File
@@ -25,7 +25,6 @@ pub(crate) mod tier_probe_intent;
pub mod warm_backend; pub mod warm_backend;
pub mod warm_backend_aliyun; pub mod warm_backend_aliyun;
pub mod warm_backend_azure; pub mod warm_backend_azure;
#[cfg(feature = "gcs")]
pub mod warm_backend_gcs; pub mod warm_backend_gcs;
pub mod warm_backend_huaweicloud; pub mod warm_backend_huaweicloud;
pub mod warm_backend_minio; pub mod warm_backend_minio;
+19 -168
View File
@@ -3541,7 +3541,7 @@ impl TierConfigMgr {
// Get tier configuration and create new driver // Get tier configuration and create new driver
let tier_config = self.tiers.get(tier_name).ok_or_else(|| ERR_TIER_NOT_FOUND.clone())?; let tier_config = self.tiers.get(tier_name).ok_or_else(|| ERR_TIER_NOT_FOUND.clone())?;
let driver = construct_warm_backend(tier_config).await?; let driver = new_warm_backend(tier_config, false).await?;
self.replace_driver(tier_name, driver)?; self.replace_driver(tier_name, driver)?;
Ok(self Ok(self
@@ -4486,11 +4486,6 @@ impl TierConfigMgr {
let committed_coordinator_intent = let committed_coordinator_intent =
committed_tier_mutation_intent(coordinator_intent.as_ref(), &committed_config_etag) committed_tier_mutation_intent(coordinator_intent.as_ref(), &committed_config_etag)
.map_err(TierConfigUpdateError::Save)?; .map_err(TierConfigUpdateError::Save)?;
// Persist Committed before notifying refresh; a Prepared disk record
// would restore the prepared block and invalidate our publish allowance.
let coordinator_commit =
commit_coordinator_tier_mutation_intent(api.clone(), coordinator_intent.as_ref(), &committed_config_etag)
.await;
if let Some(intent) = committed_coordinator_intent.as_ref() { if let Some(intent) = committed_coordinator_intent.as_ref() {
TierConfigMgr::apply_committed_mutation_intent_block(&handle, intent) TierConfigMgr::apply_committed_mutation_intent_block(&handle, intent)
.await .await
@@ -4501,9 +4496,9 @@ impl TierConfigMgr {
.map_err(TierConfigUpdateError::Publish)?, .map_err(TierConfigUpdateError::Publish)?,
); );
} }
// Config is already saved: retain the committed fence and wake recovery commit_coordinator_tier_mutation_intent(api.clone(), coordinator_intent.as_ref(), &committed_config_etag)
// even when the coordinator commit failed or its outcome is unknown. .await
coordinator_commit.map_err(TierConfigUpdateError::Save)?; .map_err(TierConfigUpdateError::Save)?;
if coordinated_config_update { if coordinated_config_update {
drop(update.take()); drop(update.take());
drop(config_lock.take()); drop(config_lock.take());
@@ -10608,11 +10603,6 @@ mod tests {
.expect_err("coordinator committed-state CAS failure must be observable"); .expect_err("coordinator committed-state CAS failure must be observable");
assert!(matches!(err, TierConfigUpdateError::Save(_))); assert!(matches!(err, TierConfigUpdateError::Save(_)));
assert!(manager.read().await.tiers.contains_key("COLD-A")); assert!(manager.read().await.tiers.contains_key("COLD-A"));
assert!(TierConfigMgr::has_committed_mutation_block(&manager).await);
let refresh = TierConfigMgr::mutation_refresh_notifier(&manager).await;
tokio::time::timeout(Duration::from_secs(1), refresh.notified())
.await
.expect("failed coordinator commit must notify recovery after saving config");
let blocked = match TierConfigMgr::acquire_operation_lease(&manager, "COLD-A").await { let blocked = match TierConfigMgr::acquire_operation_lease(&manager, "COLD-A").await {
Ok(_) => panic!("failed coordinator commit CAS must retain the local committed fence"), Ok(_) => panic!("failed coordinator commit CAS must retain the local committed fence"),
Err(err) => err, Err(err) => err,
@@ -14339,12 +14329,6 @@ mod tests {
after_commit: bool, after_commit: bool,
} }
#[derive(Debug, Default)]
struct CasCoordinatorCommitBarrier {
arrived: tokio::sync::Notify,
release: tokio::sync::Notify,
}
#[derive(Debug)] #[derive(Debug)]
struct CasConfigStore { struct CasConfigStore {
objects: tokio::sync::Mutex<HashMap<String, (Vec<u8>, String)>>, objects: tokio::sync::Mutex<HashMap<String, (Vec<u8>, String)>>,
@@ -14357,7 +14341,6 @@ mod tests {
fail_delete_prefix: tokio::sync::Mutex<Option<(String, usize)>>, fail_delete_prefix: tokio::sync::Mutex<Option<(String, usize)>>,
delete_log: tokio::sync::Mutex<Vec<String>>, delete_log: tokio::sync::Mutex<Vec<String>>,
list_barrier: tokio::sync::Mutex<Option<Arc<CasListBarrier>>>, list_barrier: tokio::sync::Mutex<Option<Arc<CasListBarrier>>>,
coordinator_commit_barrier: tokio::sync::Mutex<Option<Arc<CasCoordinatorCommitBarrier>>>,
intent_list_calls: AtomicUsize, intent_list_calls: AtomicUsize,
fail_reference_walk: AtomicBool, fail_reference_walk: AtomicBool,
reference_walk_send_count: AtomicUsize, reference_walk_send_count: AtomicUsize,
@@ -14380,7 +14363,6 @@ mod tests {
fail_delete_prefix: tokio::sync::Mutex::new(None), fail_delete_prefix: tokio::sync::Mutex::new(None),
delete_log: tokio::sync::Mutex::new(Vec::new()), delete_log: tokio::sync::Mutex::new(Vec::new()),
list_barrier: tokio::sync::Mutex::new(None), list_barrier: tokio::sync::Mutex::new(None),
coordinator_commit_barrier: tokio::sync::Mutex::new(None),
intent_list_calls: AtomicUsize::new(0), intent_list_calls: AtomicUsize::new(0),
fail_reference_walk: AtomicBool::new(false), fail_reference_walk: AtomicBool::new(false),
reference_walk_send_count: AtomicUsize::new(0), reference_walk_send_count: AtomicUsize::new(0),
@@ -14572,19 +14554,6 @@ mod tests {
} }
let mut payload = Vec::new(); let mut payload = Vec::new();
tokio::io::AsyncReadExt::read_to_end(&mut data.stream, &mut payload).await?; tokio::io::AsyncReadExt::read_to_end(&mut data.stream, &mut payload).await?;
if object.starts_with(crate::services::tier::tier_mutation_intent::TIER_COORDINATOR_MUTATION_INTENT_RECORD_PREFIX)
&& opts
.http_preconditions
.as_ref()
.and_then(HTTPPreconditions::if_match_value)
.is_some()
{
let barrier = self.coordinator_commit_barrier.lock().await.take();
if let Some(barrier) = barrier {
barrier.arrived.notify_one();
barrier.release.notified().await;
}
}
let race_rewrite = if opts let race_rewrite = if opts
.http_preconditions .http_preconditions
.as_ref() .as_ref()
@@ -15682,7 +15651,14 @@ mod tests {
); );
} }
async fn assert_lifecycle_only_reference_obeys_force(clear: bool, force: bool) { #[tokio::test]
async fn force_remove_and_save_bypasses_lifecycle_only_reference() {
// rustfs/rustfs#6832: reproduces the admin RemoveTier path (not just the lower-level
// reference-proof function) for a tier with zero transitioned objects but a lifecycle
// rule still pointing at it — the exact shape of
// `test_manual_transition_async_tier_failure_reports_terminal_partial` in e2e_test,
// which force-removes a tier a lifecycle rule still references to simulate a
// decommissioned backend.
let store = Arc::new(CasConfigStore::default()); let store = Arc::new(CasConfigStore::default());
let tier = build_rustfs_tier("COLD-A"); let tier = build_rustfs_tier("COLD-A");
let mut persisted = empty_mgr(); let mut persisted = empty_mgr();
@@ -15723,55 +15699,22 @@ mod tests {
let manager = TierConfigMgr::new(); let manager = TierConfigMgr::new();
manager.write().await.tiers.insert("COLD-A".to_string(), tier); manager.write().await.tiers.insert("COLD-A".to_string(), tier);
let mutation = if clear { TierConfigMgr::remove_and_save_with(&manager, store.clone(), "COLD-A", true)
TierCandidateMutation::Clear(force) .await
} else { .expect("force remove must bypass a lifecycle-config-only reference");
TierCandidateMutation::Remove("COLD-A".to_string(), force)
};
let result = TIER_DRIVER_TEST_FACTORY
.scope(
healthy_driver_factory(),
TierConfigMgr::update_candidate_with_config_lock(&manager, store.clone(), mutation),
)
.await;
if force {
result.expect("force mutation must bypass a lifecycle-config-only reference");
} else {
let err = result.expect_err("non-force mutation must reject a lifecycle-only reference");
let TierConfigUpdateError::Publish(err) = err else {
panic!("non-force mutation must fail during reference proof: {err:?}");
};
assert_eq!(err.code, ERR_TIER_BACKEND_IN_USE.code);
assert!(err.message.contains("move-current"), "{err}");
}
assert_eq!(manager.read().await.tiers.contains_key("COLD-A"), !force); assert!(!manager.read().await.tiers.contains_key("COLD-A"));
assert_eq!( assert!(
load_tier_config_for_update(store) !load_tier_config_for_update(store)
.await .await
.expect("config should still reload") .expect("config should still reload")
.0 .0
.tiers .tiers
.contains_key("COLD-A"), .contains_key("COLD-A"),
!force, "force removal must persist the empty candidate"
"persisted state must match the force mutation result"
); );
} }
#[tokio::test]
async fn remove_with_config_lock_obeys_force_for_lifecycle_only_reference() {
for force in [false, true] {
assert_lifecycle_only_reference_obeys_force(false, force).await;
}
}
#[tokio::test]
async fn clear_with_config_lock_obeys_force_for_lifecycle_only_reference() {
for force in [false, true] {
assert_lifecycle_only_reference_obeys_force(true, force).await;
}
}
#[tokio::test] #[tokio::test]
async fn zero_reference_proof_blocks_clear_before_config_save() { async fn zero_reference_proof_blocks_clear_before_config_save() {
let store = Arc::new(CasConfigStore::default()); let store = Arc::new(CasConfigStore::default());
@@ -17312,98 +17255,6 @@ mod tests {
assert_ne!(manager_a.read().await.empty(), manager_b.read().await.empty()); assert_ne!(manager_a.read().await.empty(), manager_b.read().await.empty());
} }
async fn assert_coordinator_commit_refresh_succeeds(mutation: TierCandidateMutation) {
let adding = matches!(mutation, TierCandidateMutation::Add(..));
let manager = TierConfigMgr::new();
let store = Arc::new(CasConfigStore::default());
if !adding {
let mut persisted = empty_mgr();
persisted.tiers.insert("COLD-A".to_string(), build_rustfs_tier("COLD-A"));
persisted
.save_tiering_config_if_current(store.clone(), None)
.await
.expect("existing tier fixture should persist");
let mut guard = manager.write().await;
install_lease_backend(&mut guard, "COLD-A", LeaseTestBackend::ready("old"));
}
let barrier = Arc::new(CasCoordinatorCommitBarrier::default());
*store.coordinator_commit_barrier.lock().await = Some(barrier.clone());
let update_manager = manager.clone();
let update_store = store.clone();
let update = tokio::spawn(async move {
TIER_DRIVER_TEST_FACTORY
.scope(
healthy_driver_factory(),
TIER_MUTATION_TEST_PEERS.scope(
Vec::new(),
TierConfigMgr::update_candidate_with_config_lock(&update_manager, update_store, mutation),
),
)
.await
});
tokio::time::timeout(Duration::from_secs(5), barrier.arrived.notified())
.await
.expect("mutation should reach coordinator commit after saving config");
assert_eq!(
load_tier_config_for_update(store.clone())
.await
.expect("saved config should be readable before coordinator commit")
.0
.tiers
.contains_key("COLD-A"),
adding
);
assert_eq!(
TierConfigMgr::load_coordinator_mutation_intents(store.clone())
.await
.expect("coordinator intent should remain readable")[0]
.state,
TierMutationIntentState::Prepared
);
let lock_requests = lock_unpoisoned(&store.lock_requests).len();
// Also exercise an independently scheduled refresh while the durable
// coordinator record is still Prepared, before its commit notification.
TierConfigMgr::request_committed_mutation_refresh(&manager).await;
TIER_MUTATION_TEST_PEERS
.scope(Vec::new(), async {
let worker = TierConfigMgr::refresh_tier_config_handle_with(manager.clone(), store.clone());
tokio::pin!(worker);
tokio::time::timeout(Duration::from_secs(5), async {
while lock_unpoisoned(&store.lock_requests).len() == lock_requests {
tokio::select! {
_ = &mut worker => panic!("refresh worker must remain available"),
_ = tokio::task::yield_now() => {}
}
}
})
.await
.expect("refresh should reconcile the Prepared record before waiting for the config lock");
barrier.release.notify_one();
let result = tokio::time::timeout(Duration::from_secs(5), async {
tokio::select! {
_ = &mut worker => panic!("refresh worker must remain available"),
result = update => result.expect("tier mutation task should join"),
}
})
.await
.expect("tier mutation should finish with refresh running");
result.expect("saved tier mutation must publish successfully on the first attempt");
})
.await;
assert_eq!(manager.read().await.tiers.contains_key("COLD-A"), adding);
}
#[tokio::test]
async fn tier_add_succeeds_with_refresh_during_coordinator_commit() {
assert_coordinator_commit_refresh_succeeds(TierCandidateMutation::Add(build_rustfs_tier("COLD-A"), true)).await;
}
#[tokio::test]
async fn tier_remove_succeeds_with_refresh_during_coordinator_commit() {
assert_coordinator_commit_refresh_succeeds(TierCandidateMutation::Remove("COLD-A".to_string(), true)).await;
}
async fn committed_refresh_fixture(fail_cleanup: bool) -> (Arc<RwLock<TierConfigMgr>>, Arc<CasConfigStore>, uuid::Uuid) { async fn committed_refresh_fixture(fail_cleanup: bool) -> (Arc<RwLock<TierConfigMgr>>, Arc<CasConfigStore>, uuid::Uuid) {
let manager = TierConfigMgr::new(); let manager = TierConfigMgr::new();
{ {
@@ -19,14 +19,13 @@
#![allow(clippy::all)] #![allow(clippy::all)]
use crate::error::is_err_bucket_not_found; use crate::error::is_err_bucket_not_found;
#[cfg(feature = "gcs")]
use crate::services::tier::warm_backend_gcs::WarmBackendGCS;
use crate::services::tier::{ use crate::services::tier::{
tier::{ERR_TIER_BACKEND_IN_USE, ERR_TIER_INVALID_CONFIG, ERR_TIER_TYPE_UNSUPPORTED}, tier::{ERR_TIER_BACKEND_IN_USE, ERR_TIER_INVALID_CONFIG, ERR_TIER_TYPE_UNSUPPORTED},
tier_config::{TierConfig, TierType}, tier_config::{TierConfig, TierType},
tier_handlers::{ERR_TIER_BUCKET_NOT_FOUND, ERR_TIER_NOT_FOUND, ERR_TIER_PERM_ERR}, tier_handlers::{ERR_TIER_BUCKET_NOT_FOUND, ERR_TIER_NOT_FOUND, ERR_TIER_PERM_ERR},
warm_backend_aliyun::WarmBackendAliyun, warm_backend_aliyun::WarmBackendAliyun,
warm_backend_azure::WarmBackendAzure, warm_backend_azure::WarmBackendAzure,
warm_backend_gcs::WarmBackendGCS,
warm_backend_huaweicloud::WarmBackendHuaweicloud, warm_backend_huaweicloud::WarmBackendHuaweicloud,
warm_backend_minio::WarmBackendMinIO, warm_backend_minio::WarmBackendMinIO,
warm_backend_r2::WarmBackendR2, warm_backend_r2::WarmBackendR2,
@@ -38,7 +37,7 @@ use crate::services::tier::{
use bytes::Bytes; use bytes::Bytes;
use http::StatusCode; use http::StatusCode;
use rustfs_s3_client::credentials::{Credentials, SignatureType, Static, Value}; use rustfs_s3_client::credentials::{Credentials, SignatureType, Static, Value};
use rustfs_s3_client::transition_api::{BucketLookupType, Options, TransitionClient, TransitionClientTimeouts, TransitionCore}; use rustfs_s3_client::transition_api::{BucketLookupType, Options, TransitionClient, TransitionCore};
use rustfs_s3_client::{ use rustfs_s3_client::{
admin_handler_utils::AdminError, admin_handler_utils::AdminError,
api_error_response::to_error_response, api_error_response::to_error_response,
@@ -321,27 +320,6 @@ pub(crate) fn endpoint_authority(url: &url::Url) -> Result<String, std::io::Erro
} }
} }
fn transition_timeout_from_env(env_key: &str, default_secs: u64) -> Duration {
Duration::from_secs(rustfs_utils::get_env_u64(env_key, default_secs))
}
pub(crate) fn transition_client_timeouts_from_env() -> TransitionClientTimeouts {
TransitionClientTimeouts::new(
transition_timeout_from_env(
rustfs_config::ENV_TIER_REMOTE_CONNECT_TIMEOUT_SECS,
rustfs_config::DEFAULT_TIER_REMOTE_CONNECT_TIMEOUT_SECS,
),
transition_timeout_from_env(
rustfs_config::ENV_TIER_REMOTE_REQUEST_TIMEOUT_SECS,
rustfs_config::DEFAULT_TIER_REMOTE_REQUEST_TIMEOUT_SECS,
),
transition_timeout_from_env(
rustfs_config::ENV_TIER_REMOTE_RESPONSE_BODY_IDLE_TIMEOUT_SECS,
rustfs_config::DEFAULT_TIER_REMOTE_RESPONSE_BODY_IDLE_TIMEOUT_SECS,
),
)
}
/// Build the [`WarmBackendS3`] shared by the S3-compatible warm backend providers. /// Build the [`WarmBackendS3`] shared by the S3-compatible warm backend providers.
/// ///
/// Credential, bucket, and endpoint validation run in this order because the /// Credential, bucket, and endpoint validation run in this order because the
@@ -372,7 +350,6 @@ pub(crate) async fn new_s3_compatible_warm_backend(
signer_type: SignatureType::SignatureV4, signer_type: SignatureType::SignatureV4,
..Default::default() ..Default::default()
})); }));
let timeouts = transition_client_timeouts_from_env();
let opts = Options { let opts = Options {
creds, creds,
secure: u.scheme() == "https", secure: u.scheme() == "https",
@@ -385,7 +362,7 @@ pub(crate) async fn new_s3_compatible_warm_backend(
// Run the SSRF guard after the host-presence check so a host-less endpoint // Run the SSRF guard after the host-presence check so a host-less endpoint
// keeps this constructor's stable error text. // keeps this constructor's stable error text.
(params.validate_endpoint)(&u).map_err(|err| std::io::Error::other(format!("tier endpoint is not allowed: {err}")))?; (params.validate_endpoint)(&u).map_err(|err| std::io::Error::other(format!("tier endpoint is not allowed: {err}")))?;
let client = TransitionClient::new_with_timeouts(&endpoint, opts, params.provider_tag, timeouts).await?; let client = TransitionClient::new(&endpoint, opts, params.provider_tag).await?;
let client = Arc::new(client); let client = Arc::new(client);
let core = TransitionCore(Arc::clone(&client)); let core = TransitionCore(Arc::clone(&client));
@@ -913,15 +890,6 @@ pub async fn new_warm_backend(tier: &TierConfig, probe: bool) -> Result<WarmBack
}); });
} }
} }
#[cfg(not(feature = "gcs"))]
TierType::GCS => {
return Err(AdminError {
code: ERR_TIER_TYPE_UNSUPPORTED.code.clone(),
message: "This build does not include the GCS backend; rebuild with the gcs feature".to_string(),
status_code: StatusCode::NOT_IMPLEMENTED,
});
}
#[cfg(feature = "gcs")]
TierType::GCS => { TierType::GCS => {
if let Some(gcs_config) = tier.gcs.as_ref() { if let Some(gcs_config) = tier.gcs.as_ref() {
let dd = WarmBackendGCS::new(gcs_config, &tier.name).await; let dd = WarmBackendGCS::new(gcs_config, &tier.name).await;
@@ -1038,27 +1006,6 @@ mod tests {
const PROBE_VERSION: &str = "remote-v2"; const PROBE_VERSION: &str = "remote-v2";
#[cfg(not(feature = "gcs"))]
#[tokio::test]
async fn gcs_backend_not_compiled_preserves_config() {
let json = r#"{"name":"ARCHIVE","type":"gcs","gcs":{"bucket":"archive","creds":"secret"}}"#;
let tier: TierConfig = serde_json::from_str(json).expect("GCS config remains readable without the backend");
assert_eq!(tier.tier_type, TierType::GCS);
let encoded = serde_json::to_vec(&tier).expect("GCS config remains writable");
let restored: TierConfig = serde_json::from_slice(&encoded).expect("GCS config round trips");
assert_eq!(restored.tier_type, TierType::GCS);
let restored_gcs = restored.gcs.as_ref().expect("GCS settings preserved");
assert_eq!(restored_gcs.bucket, "archive");
assert_eq!(restored_gcs.creds, "secret");
assert_eq!(tier.redacted().gcs.expect("redacted GCS settings").creds, "REDACTED");
let error = match new_warm_backend(&tier, false).await {
Ok(_) => panic!("an excluded GCS backend cannot be constructed"),
Err(error) => error,
};
assert_eq!(error.code, ERR_TIER_TYPE_UNSUPPORTED.code);
assert_eq!(error.status_code, StatusCode::NOT_IMPLEMENTED);
}
struct CountingBackend { struct CountingBackend {
put_result: fn() -> Result<String, std::io::Error>, put_result: fn() -> Result<String, std::io::Error>,
removes: Arc<AtomicUsize>, removes: Arc<AtomicUsize>,
@@ -26,7 +26,7 @@ use crate::services::tier::{
tier_config::TierS3, tier_config::TierS3,
warm_backend::{ warm_backend::{
TransitionCandidateIdentity, TransitionCandidateProbe, TransitionCandidateReconciler, WarmBackend, WarmBackendGetOpts, TransitionCandidateIdentity, TransitionCandidateProbe, TransitionCandidateReconciler, WarmBackend, WarmBackendGetOpts,
build_transition_put_options, endpoint_authority, transition_client_timeouts_from_env, build_transition_put_options, endpoint_authority,
}, },
}; };
use http::HeaderMap; use http::HeaderMap;
@@ -139,7 +139,6 @@ impl WarmBackendS3 {
} else { } else {
return Err(std::io::Error::other("insufficient parameters for S3 backend authentication")); return Err(std::io::Error::other("insufficient parameters for S3 backend authentication"));
} }
let timeouts = transition_client_timeouts_from_env();
let opts = Options { let opts = Options {
creds, creds,
secure: u.scheme() == "https", secure: u.scheme() == "https",
@@ -148,7 +147,7 @@ impl WarmBackendS3 {
..Default::default() ..Default::default()
}; };
let endpoint = endpoint_authority(&u)?; let endpoint = endpoint_authority(&u)?;
let client = TransitionClient::new_with_timeouts(&endpoint, opts, tier_type, timeouts).await?; let client = TransitionClient::new(&endpoint, opts, tier_type).await?;
let client = Arc::new(client); let client = Arc::new(client);
let core = TransitionCore(Arc::clone(&client)); let core = TransitionCore(Arc::clone(&client));
+71 -243
View File
@@ -3558,11 +3558,6 @@ impl RenameRollbackReceipt {
} }
} }
struct RenameRollbackOwnership {
receipt: Option<RenameRollbackReceipt>,
namespace_commit_guard: Option<Arc<crate::runtime::instance::NamespaceCommitGuard>>,
}
async fn inspect_incomplete_rename_rollback( async fn inspect_incomplete_rename_rollback(
disks: &[Option<DiskStore>], disks: &[Option<DiskStore>],
bucket: &str, bucket: &str,
@@ -3609,12 +3604,8 @@ async fn rollback_failed_rename(
dispatch_states: &[RenameDispatchState], dispatch_states: &[RenameDispatchState],
rollback_dirs: &[Option<Uuid>], rollback_dirs: &[Option<Uuid>],
dst: (&str, &str), dst: (&str, &str),
ownership: RenameRollbackOwnership, receipt: Option<RenameRollbackReceipt>,
) { ) {
let RenameRollbackOwnership {
receipt,
namespace_commit_guard,
} = ownership;
let owned_disks = disks.to_vec(); let owned_disks = disks.to_vec();
let owned_errs = errs.to_vec(); let owned_errs = errs.to_vec();
let owned_dispatch_states = dispatch_states.to_vec(); let owned_dispatch_states = dispatch_states.to_vec();
@@ -3660,9 +3651,7 @@ async fn rollback_failed_rename(
let fi = std::mem::take(&mut file_infos[disk_index]); let fi = std::mem::take(&mut file_infos[disk_index]);
let bucket = bucket.to_string(); let bucket = bucket.to_string();
let object = object.to_string(); let object = object.to_string();
let disk_namespace_commit_guard = namespace_commit_guard.clone();
let task = tokio::spawn(async move { let task = tokio::spawn(async move {
let _namespace_commit_guard = disk_namespace_commit_guard;
#[allow(clippy::let_unit_value)] #[allow(clippy::let_unit_value)]
let _task_guard = SetDisks::rename_fanout_task_guard(&object); let _task_guard = SetDisks::rename_fanout_task_guard(&object);
SetDisks::rename_fanout_barrier(&object, disk_index, rename_fanout_barrier_phase::ROLLBACK).await; SetDisks::rename_fanout_barrier(&object, disk_index, rename_fanout_barrier_phase::ROLLBACK).await;
@@ -3683,9 +3672,6 @@ async fn rollback_failed_rename(
}); });
tasks.push(async move { (disk_index, task.await) }); tasks.push(async move { (disk_index, task.await) });
} }
#[cfg(test)]
rollback_fault_injection::after_undo_dispatch(object);
let _namespace_commit_guard = namespace_commit_guard;
for (disk_index, result) in join_all(tasks).await { for (disk_index, result) in join_all(tasks).await {
outcomes[disk_index].outcome = rename_rollback_task_outcome(result); outcomes[disk_index].outcome = rename_rollback_task_outcome(result);
} }
@@ -3792,7 +3778,6 @@ pub(in crate::set_disk) struct RenameDataFenceOptions<'a> {
write_quorum: usize, write_quorum: usize,
scanner_publication_lease_tokens: Option<&'a HashMap<String, Uuid>>, scanner_publication_lease_tokens: Option<&'a HashMap<String, Uuid>>,
scanner_publication_commit_scope: Option<crate::object_api::ScannerPublicationCommitScope>, scanner_publication_commit_scope: Option<crate::object_api::ScannerPublicationCommitScope>,
namespace_commit_guard: Option<Arc<crate::runtime::instance::NamespaceCommitGuard>>,
rollback_receipt: Option<RenameRollbackReceipt>, rollback_receipt: Option<RenameRollbackReceipt>,
} }
@@ -3805,7 +3790,6 @@ impl<'a> RenameDataFenceOptions<'a> {
write_quorum, write_quorum,
scanner_publication_lease_tokens, scanner_publication_lease_tokens,
scanner_publication_commit_scope: None, scanner_publication_commit_scope: None,
namespace_commit_guard: None,
rollback_receipt: None, rollback_receipt: None,
} }
} }
@@ -3822,14 +3806,6 @@ impl<'a> RenameDataFenceOptions<'a> {
self.scanner_publication_commit_scope = scanner_publication_commit_scope; self.scanner_publication_commit_scope = scanner_publication_commit_scope;
self self
} }
pub(in crate::set_disk) fn with_namespace_commit_guard(
mut self,
namespace_commit_guard: Option<Arc<crate::runtime::instance::NamespaceCommitGuard>>,
) -> Self {
self.namespace_commit_guard = namespace_commit_guard;
self
}
} }
#[allow(dead_code, reason = "asserted by this file's tests (backlog#1823)")] #[allow(dead_code, reason = "asserted by this file's tests (backlog#1823)")]
@@ -4188,7 +4164,6 @@ impl SetDisks {
write_quorum, write_quorum,
scanner_publication_lease_tokens, scanner_publication_lease_tokens,
scanner_publication_commit_scope: _scanner_publication_commit_scope, scanner_publication_commit_scope: _scanner_publication_commit_scope,
namespace_commit_guard,
rollback_receipt, rollback_receipt,
} = fence_options; } = fence_options;
if let Some(file_info) = disks if let Some(file_info) = disks
@@ -4235,9 +4210,7 @@ impl SetDisks {
let dst_object = fanout_dst_object.clone(); let dst_object = fanout_dst_object.clone();
let file_info = file_info.clone(); let file_info = file_info.clone();
let successful_rename_completion_rank = successful_rename_completion_rank.clone(); let successful_rename_completion_rank = successful_rename_completion_rank.clone();
let namespace_commit_guard = namespace_commit_guard.clone();
tasks.spawn(async move { tasks.spawn(async move {
let _namespace_commit_guard = namespace_commit_guard;
let mut dispatch_state = RenameDispatchState::NotDispatched; let mut dispatch_state = RenameDispatchState::NotDispatched;
let result = std::panic::AssertUnwindSafe(async { let result = std::panic::AssertUnwindSafe(async {
#[allow(clippy::let_unit_value)] #[allow(clippy::let_unit_value)]
@@ -4399,10 +4372,7 @@ impl SetDisks {
&dispatch_states, &dispatch_states,
&data_dirs, &data_dirs,
(&fanout_dst_bucket, &fanout_dst_object), (&fanout_dst_bucket, &fanout_dst_object),
RenameRollbackOwnership { rollback_receipt,
receipt: rollback_receipt,
namespace_commit_guard,
},
) )
.await; .await;
if let Some(commit_tx) = commit_tx.take() { if let Some(commit_tx) = commit_tx.take() {
@@ -4558,7 +4528,6 @@ impl SetDisks {
write_quorum, write_quorum,
scanner_publication_lease_tokens, scanner_publication_lease_tokens,
scanner_publication_commit_scope, scanner_publication_commit_scope,
namespace_commit_guard,
rollback_receipt, rollback_receipt,
} = fence_options; } = fence_options;
if let Some(file_info) = disks if let Some(file_info) = disks
@@ -4592,7 +4561,6 @@ impl SetDisks {
let fanout_dst_bucket = dst_bucket.clone(); let fanout_dst_bucket = dst_bucket.clone();
let fanout_dst_object = dst_object.clone(); let fanout_dst_object = dst_object.clone();
let fanout_publication_scope = scanner_publication_commit_scope.clone(); let fanout_publication_scope = scanner_publication_commit_scope.clone();
let fanout_namespace_commit_guard = namespace_commit_guard.clone();
// Keep one coordinator task so a cancelled caller cannot drop partially // Keep one coordinator task so a cancelled caller cannot drop partially
// completed disk mutations. Per-disk futures stay ordered in `join_all`, // completed disk mutations. Per-disk futures stay ordered in `join_all`,
// preserving slot-indexed quorum and convergence accounting without a // preserving slot-indexed quorum and convergence accounting without a
@@ -4601,7 +4569,6 @@ impl SetDisks {
// Keep the storage-owned movement permit attached to the actual // Keep the storage-owned movement permit attached to the actual
// fan-out owner, even if the caller future is cancelled. // fan-out owner, even if the caller future is cancelled.
let _fanout_publication_scope = fanout_publication_scope; let _fanout_publication_scope = fanout_publication_scope;
let _namespace_commit_guard = fanout_namespace_commit_guard;
let successful_rename_completion_rank = let successful_rename_completion_rank =
rustfs_io_metrics::put_stage_metrics_enabled().then(|| Arc::new(AtomicUsize::new(0))); rustfs_io_metrics::put_stage_metrics_enabled().then(|| Arc::new(AtomicUsize::new(0)));
let futures = fanout_disks let futures = fanout_disks
@@ -4823,10 +4790,7 @@ impl SetDisks {
&dispatch_states, &dispatch_states,
&data_dirs, &data_dirs,
(&dst_bucket, &dst_object), (&dst_bucket, &dst_object),
RenameRollbackOwnership { rollback_receipt,
receipt: rollback_receipt,
namespace_commit_guard,
},
) )
.await; .await;
return Err(ret_err); return Err(ret_err);
@@ -6539,9 +6503,9 @@ impl SetDisks {
match oi { match oi {
Ok(oi) => { Ok(oi) => {
// Ordinary writes may proceed past a top-level delete marker; // Ordinary writes may proceed past a top-level delete marker;
// data movement and guarded internal writes must preserve it. // data movement must not replace an acknowledged deletion.
if oi.delete_marker { if oi.delete_marker {
return (opts.data_movement || opts.preserve_delete_marker).then_some(StorageError::PreconditionFailed); return opts.data_movement.then_some(StorageError::PreconditionFailed);
} }
let if_none_match = http_preconditions.if_none_match_value().map(str::to_owned); let if_none_match = http_preconditions.if_none_match_value().map(str::to_owned);
let if_match = http_preconditions.if_match_value().map(str::to_owned); let if_match = http_preconditions.if_match_value().map(str::to_owned);
@@ -6790,7 +6754,6 @@ pub(in crate::set_disk) mod rollback_fault_injection {
VolumeNotFoundAfterRename, VolumeNotFoundAfterRename,
PanicAfterRename, PanicAfterRename,
CoordinatorPanic, CoordinatorPanic,
RollbackCoordinatorPanic,
} }
fn registry() -> &'static Mutex<HashMap<String, (usize, Fault)>> { fn registry() -> &'static Mutex<HashMap<String, (usize, Fault)>> {
@@ -6853,17 +6816,6 @@ pub(in crate::set_disk) mod rollback_fault_injection {
panic!("injected rename coordinator panic"); panic!("injected rename coordinator panic");
} }
} }
pub(super) fn after_undo_dispatch(object: &str) {
let fault = registry()
.lock()
.expect("rollback registry should not poison")
.get(object)
.copied();
if matches!(fault, Some((_, Fault::RollbackCoordinatorPanic))) {
panic!("injected rollback coordinator panic");
}
}
} }
/// Test-only per-disk call counters for the metadata fan-out (backlog#1325, /// Test-only per-disk call counters for the metadata fan-out (backlog#1325,
@@ -7025,7 +6977,7 @@ pub(crate) mod rename_fanout_barrier {
use tokio::sync::Notify; use tokio::sync::Notify;
pub use super::rename_fanout_barrier_phase::{ pub use super::rename_fanout_barrier_phase::{
CLEANUP as PHASE_CLEANUP, READ_VERSION as PHASE_READ_VERSION, RENAME as PHASE_RENAME, ROLLBACK as PHASE_ROLLBACK, CLEANUP as PHASE_CLEANUP, READ_VERSION as PHASE_READ_VERSION, RENAME as PHASE_RENAME,
}; };
/// One armed barrier: the fan-out task matching `(disk_index, phase)` pauses. /// One armed barrier: the fan-out task matching `(disk_index, phase)` pauses.
@@ -10862,177 +10814,79 @@ mod tests {
#[tokio::test] #[tokio::test]
#[serial_test::serial(capacity_dirty_scope)] #[serial_test::serial(capacity_dirty_scope)]
async fn rename_rollback_incomplete_receipt_waits_for_undo_barrier() { async fn rename_rollback_incomplete_receipt_waits_for_undo_barrier() {
temp_env::async_with_vars([(ENV_RUSTFS_PUT_RENAME_EARLY_ACK_ENABLE, Some("true"))], async { for cancel_caller in [false, true] {
for (allow_early_ack, cancel_caller, object) in [ let bucket = "rename-rollback-barrier";
(false, false, "rollback-barrier-object"), let object = if cancel_caller {
(false, true, "rollback-barrier-cancelled"), "rollback-barrier-cancelled"
(true, false, "rollback-barrier-early-object"), } else {
(true, true, "rollback-barrier-early-cancelled"), "rollback-barrier-object"
] { };
let ctx = Arc::new(crate::runtime::instance::InstanceContext::new()); let (dirs, disks) = call_counter_local_disks(bucket, 4).await;
let bucket = "rename-rollback-barrier"; prepare_rename_source_dirs(&dirs, &disks, "source").await;
let (dirs, disks) = call_counter_local_disks(bucket, 4).await; let mut old = metadata_test_fileinfo(object);
prepare_rename_source_dirs(&dirs, &disks, "source").await; old.mod_time = Some(OffsetDateTime::now_utc());
let mut old = metadata_test_fileinfo(object); old.data = Some(Bytes::from_static(b"old-inline-body"));
old.mod_time = Some(OffsetDateTime::now_utc()); old.set_inline_data();
old.data = Some(Bytes::from_static(b"old-inline-body")); old.metadata.insert("etag".to_string(), "old-etag".to_string());
old.set_inline_data(); for disk in disks.iter().flatten() {
old.metadata.insert("etag".to_string(), "old-etag".to_string()); disk.write_metadata(bucket, bucket, object, old.clone())
for disk in disks.iter().flatten() {
disk.write_metadata(bucket, bucket, object, old.clone())
.await
.expect("old metadata should be staged");
}
let _rename_fault = rename_fault_injection::fail_rename_on(object, &[2, 3]);
let _undo_fault = rollback_fault_injection::arm(object, 0, rollback_fault_injection::Fault::Io);
let barrier = rename_fanout_barrier::arm(object, 0, rename_fanout_barrier_phase::ROLLBACK);
let receipt = RenameRollbackReceipt::default();
let mut rename = Box::pin(SetDisks::rename_data_owned_with_fence(
&disks,
(RUSTFS_META_TMP_BUCKET, "source"),
rename_commit_fileinfos(object, 4, "new-etag"),
(bucket, object),
allow_early_ack,
RenameDataFenceOptions::new(3, None)
.with_rollback_receipt(receipt.clone())
.with_namespace_commit_guard(Some(ctx.begin_namespace_commit())),
));
tokio::time::timeout(BARRIER_PAUSE_GUARD, async {
tokio::select! {
() = barrier.wait_until_paused() => {}
_ = rename.as_mut() => panic!("rename returned before the armed rollback barrier"),
}
})
.await
.expect("undo must reach its disk barrier");
assert!(receipt.0.get().is_none(), "pending undo must not be recorded as success");
assert!(ctx.namespace_commits_pending());
assert_eq!(ctx.namespace_commit_generation(), 1);
if cancel_caller {
drop(rename);
assert!(ctx.namespace_commits_pending(), "caller cancellation must not retire pending undo work");
assert_eq!(ctx.namespace_commit_generation(), 1);
barrier.release();
tokio::time::timeout(BARRIER_PAUSE_GUARD, async {
while receipt.0.get().is_none() || ctx.namespace_commits_pending() {
tokio::task::yield_now().await;
}
})
.await .await
.expect("cancelled caller must not cancel rollback accounting"); .expect("old metadata should be staged");
} else {
barrier.release();
assert!(rename.await.is_err());
}
assert!(
!ctx.namespace_commits_pending(),
"the completed rollback must release its namespace ownership"
);
assert_eq!(ctx.namespace_commit_generation(), 2);
assert!(receipt.is_incomplete(), "drained undo failure must survive in the receipt");
for dir in dirs.iter().skip(1) {
let reopened = reopen_local_disk(dir).await;
let restored = reopened
.read_version(
"",
bucket,
object,
"",
&ReadOptions {
read_data: true,
..Default::default()
},
)
.await
.expect("old version must remain readable after caller cancellation");
assert_eq!(restored.data.as_deref(), Some(b"old-inline-body".as_slice()));
}
} }
}) let _rename_fault = rename_fault_injection::fail_rename_on(object, &[2, 3]);
.await; let _undo_fault = rollback_fault_injection::arm(object, 0, rollback_fault_injection::Fault::Io);
} let barrier = rename_fanout_barrier::arm(object, 0, rename_fanout_barrier_phase::ROLLBACK);
let receipt = RenameRollbackReceipt::default();
#[tokio::test] let mut rename = Box::pin(SetDisks::rename_data_owned_with_fence(
#[serial_test::serial(capacity_dirty_scope)] &disks,
async fn rename_rollback_children_keep_namespace_ownership_after_coordinator_panic() { (RUSTFS_META_TMP_BUCKET, "source"),
temp_env::async_with_vars([(ENV_RUSTFS_PUT_RENAME_EARLY_ACK_ENABLE, Some("true"))], async { rename_commit_fileinfos(object, 4, "new-etag"),
for (allow_early_ack, object) in [ (bucket, object),
(false, "rollback-coordinator-panic"), false,
(true, "rollback-coordinator-panic-early"), RenameDataFenceOptions::new(3, None).with_rollback_receipt(receipt.clone()),
] { ));
let ctx = Arc::new(crate::runtime::instance::InstanceContext::new()); tokio::time::timeout(BARRIER_PAUSE_GUARD, async {
let bucket = "rename-rollback-coordinator-panic"; tokio::select! {
let (dirs, disks) = call_counter_local_disks(bucket, 4).await; () = barrier.wait_until_paused() => {}
prepare_rename_source_dirs(&dirs, &disks, "source").await; _ = rename.as_mut() => panic!("rename returned before the armed rollback barrier"),
let mut old = metadata_test_fileinfo(object);
old.mod_time = Some(OffsetDateTime::now_utc());
old.data = Some(Bytes::from_static(b"old-inline-body"));
old.set_inline_data();
old.metadata.insert("etag".to_string(), "old-etag".to_string());
for disk in disks.iter().flatten() {
disk.write_metadata(bucket, bucket, object, old.clone())
.await
.expect("old metadata should be staged");
} }
let _rename_fault = rename_fault_injection::fail_rename_on(object, &[2, 3]); })
let _rollback_fault = .await
rollback_fault_injection::arm(object, 0, rollback_fault_injection::Fault::RollbackCoordinatorPanic); .expect("undo must reach its disk barrier");
let barrier = rename_fanout_barrier::arm(object, 0, rename_fanout_barrier_phase::ROLLBACK); assert!(receipt.0.get().is_none(), "pending undo must not be recorded as success");
let receipt = RenameRollbackReceipt::default(); if cancel_caller {
let result = tokio::time::timeout( drop(rename);
BARRIER_PAUSE_GUARD,
SetDisks::rename_data_owned_with_fence(
&disks,
(RUSTFS_META_TMP_BUCKET, "source"),
rename_commit_fileinfos(object, 4, "new-etag"),
(bucket, object),
allow_early_ack,
RenameDataFenceOptions::new(3, None)
.with_rollback_receipt(receipt.clone())
.with_namespace_commit_guard(Some(ctx.begin_namespace_commit())),
),
)
.await
.expect("coordinator failure must return without waiting for detached undo tasks");
assert!(result.is_err());
tokio::time::timeout(BARRIER_PAUSE_GUARD, barrier.wait_until_paused())
.await
.expect("detached undo must reach its disk barrier");
assert!(
receipt.is_incomplete(),
"coordinator failure must preserve indeterminate recovery evidence"
);
assert!(ctx.namespace_commits_pending(), "the paused child must retain namespace ownership");
assert_eq!(ctx.namespace_commit_generation(), 1);
barrier.release(); barrier.release();
tokio::time::timeout(BARRIER_PAUSE_GUARD, async { tokio::time::timeout(BARRIER_PAUSE_GUARD, async {
while ctx.namespace_commits_pending() { while receipt.0.get().is_none() {
tokio::task::yield_now().await; tokio::task::yield_now().await;
} }
}) })
.await .await
.expect("completed undo children must release their namespace ownership"); .expect("cancelled caller must not cancel rollback accounting");
assert_eq!(ctx.namespace_commit_generation(), 2); } else {
for dir in &dirs { barrier.release();
let reopened = reopen_local_disk(dir).await; assert!(rename.await.is_err());
let restored = reopened
.read_version(
"",
bucket,
object,
"",
&ReadOptions {
read_data: true,
..Default::default()
},
)
.await
.expect("old version must remain readable after rollback coordinator failure");
assert_eq!(restored.data.as_deref(), Some(b"old-inline-body".as_slice()));
}
} }
}) assert!(receipt.is_incomplete(), "drained undo failure must survive in the receipt");
.await; for dir in dirs.iter().skip(1) {
let reopened = reopen_local_disk(dir).await;
let restored = reopened
.read_version(
"",
bucket,
object,
"",
&ReadOptions {
read_data: true,
..Default::default()
},
)
.await
.expect("old version must remain readable after caller cancellation");
assert_eq!(restored.data.as_deref(), Some(b"old-inline-body".as_slice()));
}
}
} }
#[tokio::test] #[tokio::test]
@@ -11147,35 +11001,9 @@ mod tests {
let mut file_infos = rename_commit_fileinfos(object, DISKS, "fresh-rollback-etag"); let mut file_infos = rename_commit_fileinfos(object, DISKS, "fresh-rollback-etag");
file_infos[3] = FileInfo::default(); file_infos[3] = FileInfo::default();
let ctx = Arc::new(crate::runtime::instance::InstanceContext::new()); SetDisks::rename_data(&disks, RUSTFS_META_TMP_BUCKET, "source", &file_infos, bucket, object, 4)
ctx.set_scanner_publication_state(false);
let barrier = rename_fanout_barrier::arm(object, 0, rename_fanout_barrier::PHASE_ROLLBACK);
let rename = SetDisks::rename_data_owned_with_fence(
&disks,
(RUSTFS_META_TMP_BUCKET, "source"),
file_infos,
(bucket, object),
false,
RenameDataFenceOptions::new(4, None).with_namespace_commit_guard(Some(ctx.begin_namespace_commit())),
);
let control = async {
barrier.wait_until_paused().await;
assert!(ctx.namespace_commits_pending(), "rollback must retain namespace publication ownership");
assert!(ctx.scanner_publication_state_allowed(), "rollback must not disable namespace walks");
assert_eq!(ctx.namespace_commit_generation(), 1);
barrier.release();
};
let (result, ()) = tokio::time::timeout(BARRIER_PAUSE_GUARD, async { tokio::join!(rename, control) })
.await .await
.expect("rename rollback must reach its barrier and finish after release"); .expect_err("three successful disks must fail a strict write quorum of four");
assert_eq!(
result.err(),
Some(DiskError::ErasureWriteQuorum),
"three successful disks must fail a strict write quorum of four"
);
assert!(!ctx.namespace_commits_pending());
assert!(ctx.scanner_publication_state_allowed());
assert_eq!(ctx.namespace_commit_generation(), 2);
for (idx, dir) in dirs.iter().enumerate() { for (idx, dir) in dirs.iter().enumerate() {
let reopened = reopen_local_disk(dir).await; let reopened = reopen_local_disk(dir).await;
+3 -363
View File
@@ -2490,9 +2490,9 @@ impl crate::storage_api_contracts::heal::HealOperations for SetDisks {
return Ok((result, err.map(|e| e.into()))); return Ok((result, err.map(|e| e.into())));
} }
// The inner heal and missing-object report read the registry again; let disks = self.disks.read().await;
// release this snapshot guard before a topology writer can queue between reads.
let disks = self.get_disks_internal().await; let disks = disks.clone();
let (_, errs) = Self::read_all_fileinfo(&disks, "", bucket, object, version_id, false, false, false) let (_, errs) = Self::read_all_fileinfo(&disks, "", bucket, object, version_id, false, false, false)
.await .await
.map_err(|e| to_object_err(e.into(), vec![bucket, object]))?; .map_err(|e| to_object_err(e.into(), vec![bucket, object]))?;
@@ -3419,366 +3419,6 @@ mod heal_result_report_tests {
assert_eq!(unformatted, DiskError::UnformattedDisk); assert_eq!(unformatted, DiskError::UnformattedDisk);
} }
#[derive(Clone, Copy)]
enum InventoryWriterHealCase {
Existing,
Missing,
MissingVersion,
}
async fn assert_heal_object_inventory_writer(case: InventoryWriterHealCase) {
use crate::set_disk::core::io_primitives::disk_call_counters;
use std::time::Duration;
use tokio::io::AsyncReadExt;
let (_temp_dirs, disks, set) = hermetic_set_disks_isolated(4).await;
let bucket = "heal-inventory-writer-bucket";
let object = match case {
InventoryWriterHealCase::Existing => "heal-inventory-writer-existing",
InventoryWriterHealCase::Missing => "heal-inventory-writer-missing",
InventoryWriterHealCase::MissingVersion => "heal-inventory-writer-missing-version",
};
set.make_bucket(
bucket,
&MakeBucketOptions {
versioning_enabled: true,
..Default::default()
},
)
.await
.expect("heal fixture bucket should be created");
let body = vec![0x67; 64 * 1024];
let stored_version = Uuid::new_v4();
let stored_version_string = stored_version.to_string();
let published = if matches!(case, InventoryWriterHealCase::Missing) {
None
} else {
let mut reader = PutObjReader::from_vec(body.clone());
let info = set
.put_object(
bucket,
object,
&mut reader,
&ObjectOptions {
no_lock: true,
versioned: true,
version_id: Some(stored_version_string.clone()),
..Default::default()
},
)
.await
.expect("full-fanout PUT should seed the heal fixture");
for disk in &disks {
let metadata = disk
.read_version("", bucket, object, &stored_version_string, &ReadOptions::default())
.await
.expect("the seeded version must be present on every disk");
assert_eq!(metadata.version_id, Some(stored_version));
assert_eq!(metadata.size, i64::try_from(body.len()).expect("fixture size should fit i64"));
}
Some(info)
};
let requested_version = match case {
InventoryWriterHealCase::Existing => stored_version_string.clone(),
InventoryWriterHealCase::Missing => String::new(),
InventoryWriterHealCase::MissingVersion => Uuid::new_v4().to_string(),
};
let opts = HealOpts {
no_lock: true,
..Default::default()
};
let calls = disk_call_counters::observe(object);
let read_gate = set.disks.read().await;
// UFCS selects the trait's outer precheck, not the same-named inherent heal.
let heal = <SetDisks as crate::storage_api_contracts::heal::HealOperations>::heal_object(
set.as_ref(),
bucket,
object,
&requested_version,
&opts,
);
tokio::pin!(heal);
assert!(matches!(
futures::poll!(tokio::task::unconstrained(heal.as_mut())),
std::task::Poll::Pending
));
// These tests use the current-thread runtime: full-wait metadata tasks
// have been spawned, but cannot run during the single unconstrained poll.
assert_eq!(calls.total(disk_call_counters::KIND_READ_VERSION), 0);
let writer = set.disks.write();
tokio::pin!(writer);
assert!(matches!(
futures::poll!(tokio::task::unconstrained(writer.as_mut())),
std::task::Poll::Pending
));
assert!(set.disks.try_read().is_err(), "the writer must already block new inventory readers");
tokio::time::timeout(Duration::from_secs(5), async {
while calls.total(disk_call_counters::KIND_READ_VERSION) < 4 {
tokio::task::yield_now().await;
}
})
.await
.expect("the suspended trait heal must have started the real metadata fanout");
for disk_index in 0..4 {
assert_eq!(calls.for_disk(disk_call_counters::KIND_READ_VERSION, disk_index), 1);
}
drop(read_gate);
let (_, outcome) =
tokio::time::timeout(Duration::from_secs(5), async { tokio::join!(async { drop(writer.await) }, heal) })
.await
.expect("trait heal must not deadlock its nested inventory read with the queued writer");
let (result, error) = outcome.expect("heal should report the object's outcome");
match case {
InventoryWriterHealCase::Existing => assert!(error.is_none(), "existing object heal failed: {error:?}"),
InventoryWriterHealCase::Missing => assert!(matches!(error, Some(Error::FileNotFound))),
InventoryWriterHealCase::MissingVersion => assert!(matches!(error, Some(Error::FileVersionNotFound))),
}
assert_eq!(result.bucket, bucket);
assert_eq!(result.object, object);
assert_eq!(result.version_id, requested_version);
assert_eq!(result.disk_count, 4);
assert_eq!(result.before.drives.len(), 4);
assert_eq!(result.after.drives.len(), 4);
for disk_index in 0..4 {
let endpoint = set.set_endpoints[disk_index].to_string();
assert_eq!(result.before.drives[disk_index].endpoint, endpoint);
assert_eq!(result.after.drives[disk_index].endpoint, endpoint);
}
if let Some(published) = published {
tokio::time::timeout(Duration::from_secs(10), async {
let mut reader = set
.get_object_reader(
bucket,
object,
None,
Default::default(),
&ObjectOptions {
versioned: true,
version_id: Some(stored_version_string),
..Default::default()
},
)
.await
.expect("the stored version must remain readable after heal");
assert_eq!(reader.object_info.etag, published.etag);
assert_eq!(reader.object_info.version_id, Some(stored_version));
let mut observed_body = Vec::new();
reader
.stream
.read_to_end(&mut observed_body)
.await
.expect("stored body should stream");
assert_eq!(observed_body, body);
})
.await
.expect("GET must finish after the inventory writer and heal");
}
}
#[tokio::test]
async fn heal_object_inventory_writer_existing() {
assert_heal_object_inventory_writer(InventoryWriterHealCase::Existing).await;
}
#[tokio::test]
async fn heal_object_inventory_writer_missing() {
assert_heal_object_inventory_writer(InventoryWriterHealCase::Missing).await;
}
#[tokio::test]
async fn heal_object_inventory_writer_missing_version() {
assert_heal_object_inventory_writer(InventoryWriterHealCase::MissingVersion).await;
}
#[tokio::test]
#[serial_test::serial]
async fn heal_object_with_queued_disk_renewal() {
use crate::layout::endpoints::SetupType;
use crate::runtime::instance::InstanceContext;
use crate::set_disk::core::io_primitives::disk_call_counters;
use std::collections::HashMap;
use std::future::Future;
use std::task::Poll;
use std::time::Duration;
use tokio::io::AsyncReadExt;
// renew_disk still registers local disks on the ambient context. Match
// the default serial group used by its other setup/registry fixtures,
// and restore only this temporary endpoint, including on a failed join.
struct RenewDiskTestState {
ctx: Arc<InstanceContext>,
was_dist_erasure: bool,
map: Arc<RwLock<HashMap<String, Option<DiskStore>>>>,
endpoint: String,
previous_disk: Option<Option<DiskStore>>,
}
impl Drop for RenewDiskTestState {
fn drop(&mut self) {
let ctx = self.ctx.clone();
let was_dist_erasure = self.was_dist_erasure;
let map = self.map.clone();
let endpoint = self.endpoint.clone();
let previous_disk = self.previous_disk.take();
let handle = tokio::runtime::Handle::current();
std::thread::spawn(move || {
handle.block_on(async move {
let mut map = map.write().await;
match previous_disk {
Some(disk) => {
map.insert(endpoint, disk);
}
None => {
map.remove(&endpoint);
}
}
drop(map);
if was_dist_erasure {
ctx.update_erasure_type(SetupType::DistErasure).await;
}
});
})
.join()
.expect("renew fixture state restoration should finish");
}
}
let (_temp_dirs, disks, set) = hermetic_set_disks_isolated(4).await;
let endpoint = set.set_endpoints[0].clone();
let ctx = crate::runtime::global::current_ctx();
let map = ctx.local_disk_map();
let restore = RenewDiskTestState {
ctx: ctx.clone(),
was_dist_erasure: ctx.is_dist_erasure().await,
map: map.clone(),
endpoint: endpoint.to_string(),
previous_disk: map.read().await.get(&endpoint.to_string()).cloned(),
};
// Only distributed erasure needs an override to avoid the ambient slot array.
if restore.was_dist_erasure {
ctx.update_erasure_type(SetupType::Erasure).await;
}
let bucket = "heal-disk-renewal-bucket";
let object = "heal-disk-renewal-object";
set.make_bucket(bucket, &MakeBucketOptions::default())
.await
.expect("renew fixture bucket should be created");
let body = vec![0x73; 64 * 1024];
let mut reader = PutObjReader::from_vec(body.clone());
let published = set
.put_object(
bucket,
object,
&mut reader,
&ObjectOptions {
no_lock: true,
..Default::default()
},
)
.await
.expect("full-fanout PUT should seed the renewal fixture");
for disk in &disks {
let metadata = disk
.read_version("", bucket, object, "", &ReadOptions::default())
.await
.expect("the seeded object must be present on every disk");
assert_eq!(metadata.size, i64::try_from(body.len()).expect("fixture size should fit i64"));
}
let opts = HealOpts {
no_lock: true,
..Default::default()
};
let calls = disk_call_counters::observe(object);
let read_gate = set.disks.read().await;
let heal = <SetDisks as crate::storage_api_contracts::heal::HealOperations>::heal_object(
set.as_ref(),
bucket,
object,
"",
&opts,
);
tokio::pin!(heal);
assert!(matches!(futures::poll!(tokio::task::unconstrained(heal.as_mut())), Poll::Pending));
assert_eq!(calls.total(disk_call_counters::KIND_READ_VERSION), 0);
let renew = set.renew_disk(&endpoint);
tokio::pin!(renew);
tokio::time::timeout(
Duration::from_secs(5),
futures::future::poll_fn(|cx| {
assert!(
std::pin::pin!(tokio::task::unconstrained(renew.as_mut()))
.poll(cx)
.is_pending(),
"renewal must reach its inventory write before returning"
);
if set.disks.try_read().is_err() {
Poll::Ready(())
} else {
Poll::Pending
}
}),
)
.await
.expect("real renewal must queue its topology writer behind the read gate");
let registered = map
.read()
.await
.get(&endpoint.to_string())
.cloned()
.flatten()
.expect("renewal must register the connected disk before its inventory write");
assert!(!Arc::ptr_eq(&registered, &disks[0]), "renewal must construct a new disk handle");
tokio::time::timeout(Duration::from_secs(5), async {
while calls.total(disk_call_counters::KIND_READ_VERSION) < 4 {
tokio::task::yield_now().await;
}
})
.await
.expect("the suspended trait heal must have started the real metadata fanout");
for disk_index in 0..4 {
assert_eq!(calls.for_disk(disk_call_counters::KIND_READ_VERSION, disk_index), 1);
}
drop(read_gate);
let (_, outcome) = tokio::time::timeout(Duration::from_secs(5), async { tokio::join!(renew, heal) })
.await
.expect("trait heal and real disk renewal must finish without a nested inventory read deadlock");
let (report, error) = outcome.expect("heal should report the existing object");
assert!(error.is_none(), "existing object heal failed after renewal: {error:?}");
assert_eq!(report.bucket, bucket);
assert_eq!(report.object, object);
assert_eq!(report.disk_count, 4);
let renewed = set.get_disks_internal().await[0]
.clone()
.expect("the renewed slot must remain online");
assert!(Arc::ptr_eq(&renewed, &registered), "the set must publish the newly connected handle");
assert_eq!(renewed.endpoint(), endpoint);
let format = load_format_erasure(&renewed, false)
.await
.expect("renewed disk format should remain readable");
assert_eq!(format.erasure.this, set.format.erasure.sets[0][0]);
tokio::time::timeout(Duration::from_secs(10), async {
let mut reader = set
.get_object_reader(bucket, object, None, Default::default(), &ObjectOptions::default())
.await
.expect("the object must remain readable after renewal and heal");
assert_eq!(reader.object_info.etag, published.etag);
let mut observed_body = Vec::new();
reader
.stream
.read_to_end(&mut observed_body)
.await
.expect("stored body should stream");
assert_eq!(observed_body, body);
})
.await
.expect("GET must finish after renewal and heal");
}
// Regression for #955: an offline disk must contribute exactly one drive // Regression for #955: an offline disk must contribute exactly one drive
// record. Before the fix the offline branch fell through and pushed a second // record. Before the fix the offline branch fell through and pushed a second
// (Corrupt) record for the same disk, so `before/after.drives` grew to // (Corrupt) record for the same disk, so `before/after.drives` grew to
+23 -119
View File
@@ -2452,9 +2452,10 @@ impl crate::storage_api_contracts::multipart::MultipartOperations for SetDisks {
let write_quorum = fi.write_quorum(self.default_write_quorum()); let write_quorum = fi.write_quorum(self.default_write_quorum());
let read_quorum = fi.read_quorum(self.default_read_quorum()); let read_quorum = fi.read_quorum(self.default_read_quorum());
// Release the registry guard before recovery and cleanup read it again: let disks = self.disks.read().await;
// a queued topology writer would otherwise deadlock those nested reads.
let disks = self.get_disks_internal().await; let disks = disks.clone();
// let disks = Self::shuffle_disks(&disks, &fi.erasure.distribution);
let part_path = format!("{}/{}/", upload_id_path, fi.data_dir.unwrap_or(Uuid::nil())); let part_path = format!("{}/{}/", upload_id_path, fi.data_dir.unwrap_or(Uuid::nil()));
self.recover_part_transactions(&part_path, read_quorum, write_quorum) self.recover_part_transactions(&part_path, read_quorum, write_quorum)
@@ -4050,7 +4051,6 @@ mod tests {
let _ = drain_global_dirty_scopes(); let _ = drain_global_dirty_scopes();
let rename_barrier = rename_fanout_barrier::arm(object, 0, rename_fanout_barrier::PHASE_RENAME); let rename_barrier = rename_fanout_barrier::arm(object, 0, rename_fanout_barrier::PHASE_RENAME);
let rename_tasks = rename_fanout_barrier::observe_tasks(object);
let complete_store = Arc::clone(&set_disks); let complete_store = Arc::clone(&set_disks);
let mut complete = tokio::spawn(async move { let mut complete = tokio::spawn(async move {
let mut opts = ObjectOptions::default(); let mut opts = ObjectOptions::default();
@@ -4062,6 +4062,16 @@ mod tests {
tokio::time::timeout(Duration::from_secs(30), rename_barrier.wait_until_paused()) tokio::time::timeout(Duration::from_secs(30), rename_barrier.wait_until_paused())
.await .await
.expect("multipart completion should pause one tail disk during rename"); .expect("multipart completion should pause one tail disk during rename");
assert!(
tokio::time::timeout(Duration::from_millis(100), &mut complete).await.is_err(),
"multipart completion must not publish success while a tail rename is still paused"
);
let initial = drain_global_dirty_scopes().into_iter().collect::<HashSet<_>>();
assert!(
initial.is_empty(),
"capacity must not be marked as committed before the full multipart rename finishes"
);
let abort_store = Arc::clone(&set_disks); let abort_store = Arc::clone(&set_disks);
let abort = tokio::spawn(async move { let abort = tokio::spawn(async move {
@@ -4070,46 +4080,21 @@ mod tests {
.await .await
}); });
signaling.wait_for_attempts(2).await; signaling.wait_for_attempts(2).await;
assert!(!abort.is_finished(), "the in-flight completion must retain the multipart upload guard");
// A paused rename does not establish that the other disks reached quorum. let retained_staging = futures::future::join_all(
let retained_staging = tokio::time::timeout(Duration::from_secs(30), async { disk_stores
loop { .iter()
let mut retained = 0; .map(|disk| disk.read_all(RUSTFS_META_MULTIPART_BUCKET, &staged_part)),
for result in futures::future::join_all( )
disk_stores
.iter()
.map(|disk| disk.read_all(RUSTFS_META_MULTIPART_BUCKET, &staged_part)),
)
.await
{
match result {
Ok(_) => retained += 1,
Err(DiskError::FileNotFound) => {}
Err(error) => panic!("staged rename source lookup failed: {error}"),
}
}
if retained <= 1 && rename_tasks.running() == 1 {
break retained;
}
tokio::time::sleep(Duration::from_millis(10)).await;
}
})
.await .await
.expect("unpaused multipart renames should finish before the tail is released"); .into_iter()
.filter(|result| result.is_ok())
.count();
assert_eq!( assert_eq!(
retained_staging, 1, retained_staging, 1,
"only the paused tail disk should still retain the multipart rename source" "only the paused tail disk should still retain the multipart rename source"
); );
assert!(
tokio::time::timeout(Duration::from_millis(100), &mut complete).await.is_err(),
"multipart completion must not publish success while a tail rename is still paused"
);
let initial = drain_global_dirty_scopes().into_iter().collect::<HashSet<_>>();
assert!(
initial.is_empty(),
"capacity must not be marked as committed before the full multipart rename finishes"
);
assert!(!abort.is_finished(), "the in-flight completion must retain the multipart upload guard");
signaling.set_target(rustfs_lock::ObjectKey::new(bucket, object)); signaling.set_target(rustfs_lock::ObjectKey::new(bucket, object));
let object_attempt = signaling.attempts.load(Ordering::Acquire) + 1; let object_attempt = signaling.attempts.load(Ordering::Acquire) + 1;
@@ -6758,87 +6743,6 @@ mod tests {
.await; .await;
} }
#[tokio::test(flavor = "multi_thread")]
#[serial]
async fn complete_multipart_releases_disk_snapshot_before_cleanup() {
let (temp_dirs, disk_stores, set_disks) = hermetic_set_disks(4).await;
let bucket = "multipart-topology-lock-bucket";
let object = "object";
let body = vec![0x65; 4096];
make_bucket_on_all(&disk_stores, bucket).await;
let (upload_id, parts) =
stage_upload_with_create_opts(&set_disks, bucket, object, &body, &ObjectOptions::default()).await;
let upload_id_path = SetDisks::get_upload_id_dir(bucket, object, &upload_id);
for dir in &temp_dirs {
assert!(
dir.path().join(RUSTFS_META_MULTIPART_BUCKET).join(&upload_id_path).exists(),
"the test must create real upload staging on every disk"
);
}
let barrier = MultipartCommitBarrier::install(bucket, object, MultipartCommitPause::AfterObjectPublication);
let complete_store = set_disks.clone();
let complete_upload_id = upload_id.clone();
let complete = tokio::spawn(async move {
complete_store
.complete_multipart_upload(bucket, object, &complete_upload_id, parts, &ObjectOptions::default())
.await
});
barrier.wait_until_paused().await;
// Hold a separate read gate so the real writer queues even when completion
// correctly releases its snapshot guard. Polling Pending proves admission
// to Tokio's write-preferring queue before the cleanup attempts another read.
let read_gate = set_disks.disks.read().await;
let writer = set_disks.disks.write();
tokio::pin!(writer);
assert!(matches!(
futures::poll!(tokio::task::unconstrained(writer.as_mut())),
std::task::Poll::Pending
));
assert!(
set_disks.disks.try_read().is_err(),
"the pending writer must already block new readers before the cleanup resumes"
);
drop(read_gate);
barrier.release();
let writer_guard = tokio::time::timeout(Duration::from_secs(5), writer)
.await
.expect("a queued topology writer must not deadlock with multipart cleanup's disk snapshot");
// A reconnect can publish the same handles; this test isolates admission
// order without changing the disks that contain the committed object.
drop(writer_guard);
tokio::time::timeout(Duration::from_secs(10), complete)
.await
.expect("multipart cleanup must finish after the topology writer releases")
.expect("completion task should not panic")
.expect("completion should preserve the successful object commit");
let mut reader = tokio::time::timeout(
Duration::from_secs(10),
set_disks.get_object_reader(bucket, object, None, HeaderMap::new(), &ObjectOptions::default()),
)
.await
.expect("GET should finish after completion")
.expect("the completed object should remain readable");
let mut observed_body = Vec::new();
tokio::time::timeout(Duration::from_secs(10), reader.stream.read_to_end(&mut observed_body))
.await
.expect("the completed object body should finish streaming")
.expect("the completed object body should be readable");
assert_eq!(observed_body, body);
assert!(matches!(
set_disks.check_upload_id_exists(bucket, object, &upload_id, false).await,
Err(StorageError::InvalidUploadID(..))
));
for dir in &temp_dirs {
assert!(
!dir.path().join(RUSTFS_META_MULTIPART_BUCKET).join(&upload_id_path).exists(),
"successful completion must remove its upload staging from every disk"
);
}
}
#[tokio::test(flavor = "multi_thread")] #[tokio::test(flavor = "multi_thread")]
#[serial] #[serial]
async fn complete_releases_object_lock_before_cleanup_and_keeps_upload_lock() { async fn complete_releases_object_lock_before_cleanup_and_keeps_upload_lock() {
+1 -4
View File
@@ -4459,10 +4459,7 @@ impl SetDisks {
commit_scanner_publication_lease_tokens.as_ref(), commit_scanner_publication_lease_tokens.as_ref(),
) )
.with_publication_scope(commit_scanner_publication_scope.clone()) .with_publication_scope(commit_scanner_publication_scope.clone())
.with_rollback_receipt(commit_rollback_receipt.clone()) .with_rollback_receipt(commit_rollback_receipt.clone()),
.with_namespace_commit_guard(
(!is_meta_bucketname(&commit_bucket)).then(|| commit_set.ctx.begin_namespace_commit()),
),
) )
.await; .await;
if let Some(scope) = commit_scanner_publication_scope.as_ref() { if let Some(scope) = commit_scanner_publication_scope.as_ref() {
+4 -14
View File
@@ -329,11 +329,11 @@ impl ECStore {
/// reuse its result, which is sound because bucket deletion/recreation /// reuse its result, which is sound because bucket deletion/recreation
/// requires the lifecycle WRITE lock and therefore cannot have run while /// requires the lifecycle WRITE lock and therefore cannot have run while
/// any read guard was continuously held. /// any read guard was continuously held.
pub async fn acquire_bucket_incarnation_fence( pub(crate) async fn acquire_bucket_incarnation_fence(
&self, &self,
bucket: &str, bucket: &str,
expected: uuid::Uuid, expected: uuid::Uuid,
) -> Result<super::BucketIncarnationFenceGuard> { ) -> Result<super::bucket_fence::BucketIncarnationFenceGuard> {
let inner = self.acquire_bucket_lifecycle_read_lock(bucket).await?; let inner = self.acquire_bucket_lifecycle_read_lock(bucket).await?;
let pieces = super::bucket_fence::FencePieces { let pieces = super::bucket_fence::FencePieces {
registry: self.bucket_fence_registry.clone(), registry: self.bucket_fence_registry.clone(),
@@ -1059,7 +1059,6 @@ mod tests {
use crate::storage_api_contracts::{ use crate::storage_api_contracts::{
bucket::{BucketOperations as _, BucketOptions, DeleteBucketOptions, MakeBucketOptions, SRBucketDeleteOp}, bucket::{BucketOperations as _, BucketOptions, DeleteBucketOptions, MakeBucketOptions, SRBucketDeleteOp},
list::ListOperations as _, list::ListOperations as _,
namespace::NamespaceLocking as _,
object::{ObjectIO as _, ObjectOperations as _}, object::{ObjectIO as _, ObjectOperations as _},
}; };
use crate::store::{ECStore, init_local_disks_with_instance_ctx}; use crate::store::{ECStore, init_local_disks_with_instance_ctx};
@@ -1487,19 +1486,10 @@ mod tests {
.put_object(bucket, object, &mut reader, &ObjectOptions::default()) .put_object(bucket, object, &mut reader, &ObjectOptions::default())
.await .await
.expect("object should be written"); .expect("object should be written");
let lock = ecstore.pools[0].disk_set[0]
.new_ns_lock(bucket, object)
.await
.expect("fixture namespace lock should be created");
drop(
lock.get_write_lock(Duration::from_secs(30))
.await
.expect("fixture rename tail should finish before checking its generation"),
);
assert_eq!( assert_eq!(
ecstore.scanner_namespace_mutation_generation(), ecstore.scanner_namespace_mutation_generation(),
generation_before_put.saturating_add(3), generation_before_put.saturating_add(1),
"successful object creation must observe the logical mutation and both fanout boundaries" "successful object creation should advance scanner namespace activity"
); );
ecstore ecstore
.get_object_info(bucket, object, &ObjectOptions::default()) .get_object_info(bucket, object, &ObjectOptions::default())
+1 -39
View File
@@ -150,7 +150,7 @@ impl BucketFenceRegistry {
/// A held bucket lifecycle read lock plus its registration in the fence /// A held bucket lifecycle read lock plus its registration in the fence
/// registry. Dropping the guard deregisters it; the memo is cleared when the /// registry. Dropping the guard deregisters it; the memo is cleared when the
/// last guard for the bucket drops (or a lost lock is observed). /// last guard for the bucket drops (or a lost lock is observed).
pub struct BucketIncarnationFenceGuard { pub(crate) struct BucketIncarnationFenceGuard {
inner: Option<NamespaceLockGuard>, inner: Option<NamespaceLockGuard>,
registry: Arc<BucketFenceRegistry>, registry: Arc<BucketFenceRegistry>,
bucket: String, bucket: String,
@@ -158,14 +158,6 @@ pub struct BucketIncarnationFenceGuard {
} }
impl BucketIncarnationFenceGuard { impl BucketIncarnationFenceGuard {
/// Propagate lifecycle lock loss into the storage commit checks.
/// The caller still owns this guard until the complete write tail drains.
pub fn attach_to_object_options(&self, opts: &mut crate::object_api::ObjectOptions) {
if let Some(guard) = self.namespace_lock_guard() {
opts.add_bucket_lifecycle_lock_guard(guard);
}
}
pub(crate) fn is_lock_lost(&self) -> bool { pub(crate) fn is_lock_lost(&self) -> bool {
self.inner.as_ref().is_some_and(NamespaceLockGuard::is_lock_lost) self.inner.as_ref().is_some_and(NamespaceLockGuard::is_lock_lost)
} }
@@ -354,36 +346,6 @@ mod tests {
first_pieces.abandon("b", first.token); first_pieces.abandon("b", first.token);
} }
#[tokio::test]
async fn checkpoint_options_inherit_bucket_fence_lock_loss() {
let lock = NamespaceLock::new("bucket-fence-options".to_string(), Arc::new(LocalClient::new()));
let inner = lock
.acquire_guard(&lock_request("options"))
.await
.expect("acquire")
.expect("quorum");
let pieces = FencePieces {
registry: Arc::default(),
inner,
};
let registration = pieces.enter("b");
let fence = pieces.into_guard("b", registration.token);
let mut opts = crate::object_api::ObjectOptions::default();
fence.attach_to_object_options(&mut opts);
let inherited = opts
.bucket_lifecycle_lock_fence
.as_ref()
.expect("checkpoint inherits lifecycle guard");
assert!(!inherited.is_lock_lost());
tokio::time::timeout(
Duration::from_secs(2),
fence.namespace_lock_guard().expect("held guard").lock_lost_notified(),
)
.await
.expect("distributed guard expires");
assert!(inherited.is_lock_lost(), "the actual pre-rename options must observe lifecycle lock loss");
}
#[test] #[test]
fn buckets_are_isolated() { fn buckets_are_isolated() {
let reg = BucketFenceRegistry::default(); let reg = BucketFenceRegistry::default();
File diff suppressed because it is too large Load Diff
+7 -9
View File
@@ -417,7 +417,6 @@ const MAX_UPLOADS_LIST: usize = 10000;
mod bucket; mod bucket;
mod bucket_fence; mod bucket_fence;
pub(crate) use bucket::await_bucket_namespace_operation; pub(crate) use bucket::await_bucket_namespace_operation;
pub use bucket_fence::BucketIncarnationFenceGuard;
mod heal; mod heal;
mod heal_walk; mod heal_walk;
pub use heal_walk::HealWalkVersion; pub use heal_walk::HealWalkVersion;
@@ -849,7 +848,7 @@ impl ECStore {
} }
pub fn scanner_namespace_mutation_generation(&self) -> u64 { pub fn scanner_namespace_mutation_generation(&self) -> u64 {
list_objects::scanner_namespace_mutation_generation().saturating_add(self.ctx.namespace_commit_generation()) list_objects::scanner_namespace_mutation_generation()
} }
pub async fn scanner_data_movement_active(&self) -> bool { pub async fn scanner_data_movement_active(&self) -> bool {
@@ -858,7 +857,7 @@ impl ECStore {
} }
/// Return the storage-owned movement state and generation as one /// Return the storage-owned movement state and generation as one
/// authenticated activity snapshot. The read lock is acquired before /// authenticated activity snapshot. The read lock is acquired before
/// the state locks (cancelers, pool metadata, then rebalance metadata), /// the state locks (cancelers, pool metadata, then rebalance metadata),
/// matching the transition writer order and preventing a terminal state /// matching the transition writer order and preventing a terminal state
/// from being reported with the preceding generation. /// from being reported with the preceding generation.
@@ -887,12 +886,11 @@ impl ECStore {
/// Returns whether scanner metadata may still be hidden by a local /// Returns whether scanner metadata may still be hidden by a local
/// data-movement state. Terminal failed/canceled decommission entries /// data-movement state. Terminal failed/canceled decommission entries
/// remain suspended until an operator clears or retries them, so they are /// remain suspended until an operator clears or retries them, so they are
/// a publication barrier even after the worker has stopped. Active PUT /// a publication barrier even after the worker has stopped.
/// rename fanouts also defer publication, including post-ACK tails.
pub async fn scanner_data_usage_publication_blocked(&self) -> bool { pub async fn scanner_data_usage_publication_blocked(&self) -> bool {
let operation_gate = self.ctx.data_movement_operation_gate(); let operation_gate = self.ctx.data_movement_operation_gate();
let _operation_guard = operation_gate.read_owned().await; let _operation_guard = operation_gate.read_owned().await;
self.scanner_data_usage_publication_snapshot_blocked().await || self.ctx.namespace_commits_pending() self.scanner_data_usage_publication_snapshot_blocked().await
} }
pub async fn scanner_data_movement_pause_status(&self) -> ScannerDataMovementPauseStatus { pub async fn scanner_data_movement_pause_status(&self) -> ScannerDataMovementPauseStatus {
@@ -1072,7 +1070,7 @@ impl ECStore {
{ {
return Err(Error::other("scanner publication lease generation is stale")); return Err(Error::other("scanner publication lease generation is stale"));
} }
if self.scanner_data_movement_snapshot_locked().await.1 || self.ctx.namespace_commits_pending() { if self.scanner_data_movement_snapshot_locked().await.1 {
return Err(Error::other("scanner publication lease is blocked by data movement")); return Err(Error::other("scanner publication lease is blocked by data movement"));
} }
@@ -1111,7 +1109,7 @@ impl ECStore {
{ {
return Err(Error::other("scanner publication lease generation is stale")); return Err(Error::other("scanner publication lease generation is stale"));
} }
if self.scanner_data_movement_snapshot_locked().await.1 || self.ctx.namespace_commits_pending() { if self.scanner_data_movement_snapshot_locked().await.1 {
return Err(Error::other("scanner publication lease is blocked by data movement")); return Err(Error::other("scanner publication lease is blocked by data movement"));
} }
if !self.ctx.scanner_publication_lease_is_active(token).await { if !self.ctx.scanner_publication_lease_is_active(token).await {
@@ -1131,7 +1129,7 @@ impl ECStore {
if self.ctx.data_movement_generation_exhausted() || self.ctx.data_movement_operation_epoch_exhausted() { if self.ctx.data_movement_generation_exhausted() || self.ctx.data_movement_operation_epoch_exhausted() {
return Err(Error::other("scanner publication lease generation is exhausted")); return Err(Error::other("scanner publication lease generation is exhausted"));
} }
if self.scanner_data_movement_snapshot_locked().await.1 || self.ctx.namespace_commits_pending() { if self.scanner_data_movement_snapshot_locked().await.1 {
return Err(Error::other("scanner publication lease is blocked by data movement")); return Err(Error::other("scanner publication lease is blocked by data movement"));
} }
let Some(lease_generation) = self.ctx.scanner_publication_lease_generation(token).await else { let Some(lease_generation) = self.ctx.scanner_publication_lease_generation(token).await else {
-4
View File
@@ -45,10 +45,6 @@ use uuid::Uuid;
use crate::heal::task::{HealOptions, HealPriority, HealRequest, HealType}; use crate::heal::task::{HealOptions, HealPriority, HealRequest, HealType};
/// Read-only inspection of committed MRF checkpoints. The legacy consumer
/// remains unchanged until ownership-aware replay is deployed.
pub mod snapshot;
/// Journal location inside the metadata bucket, following the resume-state /// Journal location inside the metadata bucket, following the resume-state
/// layout. /// layout.
pub(crate) const MRF_JOURNAL_PATH: &str = "buckets/.heal/mrf/journal.bin"; pub(crate) const MRF_JOURNAL_PATH: &str = "buckets/.heal/mrf/journal.bin";
-681
View File
@@ -1,681 +0,0 @@
// Copyright 2026 RustFS Team
//
// Licensed under the Apache License, Version 2.0 (the "License");
// you may not use this file except in compliance with the License.
// You may obtain a copy of the License at
//
// http://www.apache.org/licenses/LICENSE-2.0
//
// Unless required by applicable law or agreed to in writing, software
// distributed under the License is distributed on an "AS IS" BASIS,
// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
// See the License for the specific language governing permissions and
// limitations under the License.
//! Reader-first support for owner-local MRF checkpoints.
//!
//! Each of two slots has a payload and a commit manifest. The manifest binds
//! the writer identity, persistent sequence, length and whole-payload digest.
//! Replacing the inactive slot must leave the previous committed slot intact.
//! Production publication and reclamation are deliberately not enabled here.
//! An unreadable commit path cannot prove that only legacy data exists. This
//! explicit inspection API fails closed and never mutates recovery anchors.
//! It is not wired into the legacy consumer: that transition requires the
//! ownership-aware replay and producer handoff before writer activation.
//! One surviving committed replica supports process restart recovery only;
//! this reader does not establish a replication quorum or a power-loss policy.
use super::{MRF_JOURNAL_PATH, MRF_SCOPED_JOURNAL_PATH, decode_journal};
use crate::heal::RUSTFS_META_BUCKET;
use crate::heal::storage_api::owner::{EcstoreDiskAPI, EcstoreDiskError, EcstoreDiskStore};
use sha2::{Digest, Sha256};
use std::collections::HashMap;
use tokio::io::AsyncReadExt;
use uuid::Uuid;
// Root-level control files avoid requiring a new directory before the first
// atomic commit. They remain inside the storage owner's metadata volume.
const PAYLOAD_PATHS: [&str; 2] = [".heal-mrf-snapshot.0.bin", ".heal-mrf-snapshot.1.bin"];
const MANIFEST_PATHS: [&str; 2] = [".heal-mrf-commit.0.bin", ".heal-mrf-commit.1.bin"];
const MAGIC: &[u8; 8] = b"RFMRFC01";
const MANIFEST_LEN: usize = 8 + 1 + 16 + 8 + 8 + 32 + 32;
const VERSION: u8 = 1;
#[derive(Debug, thiserror::Error)]
pub enum SnapshotError {
#[error("MRF checkpoint has an invalid or incomplete commit record")]
Corrupt,
#[error("MRF checkpoint format is unsupported")]
Unsupported,
#[error("MRF checkpoint exceeds the configured byte limit")]
TooLarge,
#[error("MRF checkpoint replicas disagree at the same sequence")]
Conflict,
#[error("MRF checkpoint storage is unavailable")]
Disk(#[source] EcstoreDiskError),
#[error("MRF checkpoint body could not be read")]
Read(#[source] std::io::Error),
}
#[derive(Debug, PartialEq, Eq)]
struct Manifest {
owner: Uuid,
sequence: u64,
payload_len: usize,
payload_digest: [u8; 32],
}
impl Manifest {
fn decode(bytes: &[u8], limit: usize) -> Result<Self, SnapshotError> {
if bytes.len() != MANIFEST_LEN || &bytes[..8] != MAGIC {
return Err(SnapshotError::Corrupt);
}
if bytes[8] != VERSION {
return Err(SnapshotError::Unsupported);
}
let signed = MANIFEST_LEN - 32;
let checksum: [u8; 32] = Sha256::digest(&bytes[..signed]).into();
if checksum != bytes[signed..] {
return Err(SnapshotError::Corrupt);
}
let owner = Uuid::from_slice(&bytes[9..25]).map_err(|_| SnapshotError::Corrupt)?;
let sequence = u64::from_le_bytes(bytes[25..33].try_into().map_err(|_| SnapshotError::Corrupt)?);
let payload_len = u64::from_le_bytes(bytes[33..41].try_into().map_err(|_| SnapshotError::Corrupt)?);
let payload_len = usize::try_from(payload_len).map_err(|_| SnapshotError::TooLarge)?;
if owner.is_nil() || sequence == 0 || sequence == u64::MAX {
return Err(SnapshotError::Corrupt);
}
if payload_len > limit {
return Err(SnapshotError::TooLarge);
}
Ok(Self {
owner,
sequence,
payload_len,
payload_digest: bytes[41..73].try_into().map_err(|_| SnapshotError::Corrupt)?,
})
}
}
#[derive(Debug)]
pub struct CommittedSnapshot {
manifest: Manifest,
payload: Vec<u8>,
}
impl CommittedSnapshot {
/// Persistent single-writer sequence, not a process UUID ordering.
pub fn sequence(&self) -> u64 {
self.manifest.sequence
}
/// Identity recorded by the committed checkpoint's writer.
pub fn owner(&self) -> Uuid {
self.manifest.owner
}
/// Complete, checksum-validated record bytes. Inspection does not consume
/// these records or acknowledge completion to any producer.
pub fn payload(&self) -> &[u8] {
&self.payload
}
fn decode(manifest: &[u8], payload: Vec<u8>, limit: usize) -> Result<Self, SnapshotError> {
let manifest = Manifest::decode(manifest, limit)?;
let checksum: [u8; 32] = Sha256::digest(&payload).into();
if payload.len() != manifest.payload_len || checksum != manifest.payload_digest {
return Err(SnapshotError::Corrupt);
}
if decode_journal(&payload).1 != 0 {
return Err(SnapshotError::Corrupt);
}
Ok(Self { manifest, payload })
}
}
#[derive(Debug)]
pub enum RecoverySnapshot {
/// An intact legacy snapshot, without a comparable commit sequence.
Legacy(Vec<u8>),
/// A committed checkpoint requiring ownership-aware replay before use.
Committed(CommittedSnapshot),
}
async fn read_bounded(disk: &EcstoreDiskStore, path: &str, limit: usize) -> Result<Option<Vec<u8>>, SnapshotError> {
let reader = match EcstoreDiskAPI::read_file(disk.as_ref(), RUSTFS_META_BUCKET, path).await {
Ok(reader) => reader,
Err(EcstoreDiskError::FileNotFound | EcstoreDiskError::VolumeNotFound) => return Ok(None),
Err(error) => return Err(SnapshotError::Disk(error)),
};
let maximum = limit.checked_add(1).ok_or(SnapshotError::TooLarge)?;
let maximum = u64::try_from(maximum).map_err(|_| SnapshotError::TooLarge)?;
let mut bytes = Vec::new();
reader
.take(maximum)
.read_to_end(&mut bytes)
.await
.map_err(SnapshotError::Read)?;
if bytes.len() > limit {
return Err(SnapshotError::TooLarge);
}
Ok(Some(bytes))
}
fn select_snapshot(selected: &mut Option<CommittedSnapshot>, candidate: CommittedSnapshot) -> Result<(), SnapshotError> {
if let Some(current) = selected {
if current.manifest.sequence == candidate.manifest.sequence
&& (current.manifest != candidate.manifest || current.payload != candidate.payload)
{
return Err(SnapshotError::Conflict);
}
if current.manifest.sequence >= candidate.manifest.sequence {
return Ok(());
}
}
*selected = Some(candidate);
Ok(())
}
async fn read_committed(disks: &[EcstoreDiskStore], limit: usize) -> Result<Option<CommittedSnapshot>, SnapshotError> {
let mut selected = None;
let mut damaged = None;
let mut identities = HashMap::new();
for disk in disks {
for (manifest_path, payload_path) in MANIFEST_PATHS.into_iter().zip(PAYLOAD_PATHS) {
let candidate = async {
let Some(manifest) = read_bounded(disk, manifest_path, MANIFEST_LEN).await? else {
return Ok(None);
};
let header = Manifest::decode(&manifest, limit)?;
let payload = read_bounded(disk, payload_path, header.payload_len)
.await?
.ok_or(SnapshotError::Corrupt)?;
CommittedSnapshot::decode(&manifest, payload, limit).map(Some)
}
.await;
match candidate {
Ok(Some(candidate)) => {
let identity = (
candidate.manifest.owner,
candidate.manifest.payload_len,
candidate.manifest.payload_digest,
);
if identities
.insert(candidate.manifest.sequence, identity)
.is_some_and(|previous| previous != identity)
{
return Err(SnapshotError::Conflict);
}
select_snapshot(&mut selected, candidate)?;
}
Ok(None) => {}
// A future committed format may supersede all readable slots.
Err(SnapshotError::Unsupported) => return Err(SnapshotError::Unsupported),
Err(error) => damaged = Some(error),
}
}
}
match (selected, damaged) {
(Some(snapshot), _) => Ok(Some(snapshot)),
(None, Some(error)) => Err(error),
(None, None) => Ok(None),
}
}
async fn read_legacy(disks: &[EcstoreDiskStore], path: &str, limit: usize) -> Result<Option<Vec<u8>>, SnapshotError> {
let mut selected = None;
let mut incomplete: Option<Vec<u8>> = None;
for disk in disks {
match read_bounded(disk, path, limit).await {
Ok(Some(payload)) if decode_journal(&payload).1 == 0 => {
if selected.as_ref().is_some_and(|current| *current != payload) {
// Legacy snapshots have no sequence. There is no evidence
// that the first, longest or nonempty replica is newest.
return Err(SnapshotError::Conflict);
}
selected = Some(payload);
}
Ok(Some(payload)) => {
if let Some(previous) = &incomplete {
if previous.starts_with(&payload) {
continue;
}
if !payload.starts_with(previous) {
return Err(SnapshotError::Corrupt);
}
}
incomplete = Some(payload);
}
Ok(None) => {}
Err(error) => return Err(error),
}
}
if let Some(prefix) = incomplete
&& !selected.as_ref().is_some_and(|payload| payload.starts_with(&prefix))
{
// In particular, an empty O_TRUNC replica cannot supersede another
// replica containing intact records followed by a torn tail.
return Err(SnapshotError::Corrupt);
}
Ok(selected)
}
/// Inspect local MRF checkpoints without replaying, acknowledging or deleting.
///
/// `max_bytes` bounds each payload read. Every local replica is examined and
/// ambiguous identities, unavailable proof or unsupported formats return a
/// typed error. This API must not authorize a writer without the separate
/// ownership and mixed-version activation checks.
pub async fn inspect_local_recovery_snapshot(max_bytes: usize) -> Result<Option<RecoverySnapshot>, SnapshotError> {
read_recovery_snapshot(&super::journal_disks().await, max_bytes).await
}
async fn read_recovery_snapshot(disks: &[EcstoreDiskStore], limit: usize) -> Result<Option<RecoverySnapshot>, SnapshotError> {
if let Some(snapshot) = read_committed(disks, limit).await? {
return Ok(Some(RecoverySnapshot::Committed(snapshot)));
}
// RUSTFS_COMPAT_TODO(backlog-2263): inspect retained legacy MRF journals. Remove after all supported upgrade and rollback readers understand committed snapshots and retained journals have migrated.
if let Some(payload) = read_legacy(disks, MRF_SCOPED_JOURNAL_PATH, limit).await? {
return Ok(Some(RecoverySnapshot::Legacy(payload)));
}
Ok(read_legacy(disks, MRF_JOURNAL_PATH, limit)
.await?
.map(RecoverySnapshot::Legacy))
}
#[cfg(test)]
mod tests {
use super::*;
use crate::heal::mrf_queue::encode_intent;
use crate::heal::storage_api::owner::{EcstoreConditionalFileUpdate, EcstoreDiskBytes};
use crate::heal::{DiskOption, Endpoint, new_disk};
use rustfs_common::mrf_channel::{MrfIntent, MrfKind, MrfScope};
use std::sync::Arc;
use tempfile::TempDir;
fn payload(object: &str) -> Vec<u8> {
let intent = MrfIntent {
bucket: Arc::from("bucket"),
object: Arc::from(object),
version_id: None,
kind: MrfKind::PartialWrite,
scope: None,
lease: None,
enqueued_at_ms: 1234,
attempts: 0,
};
let mut bytes = Vec::new();
assert!(encode_intent(&intent, &mut bytes), "fixture must encode a full record");
bytes
}
fn manifest(owner: Uuid, sequence: u64, payload: &[u8]) -> Vec<u8> {
let mut bytes = Vec::with_capacity(MANIFEST_LEN);
bytes.extend_from_slice(MAGIC);
bytes.push(VERSION);
bytes.extend_from_slice(owner.as_bytes());
bytes.extend_from_slice(&sequence.to_le_bytes());
bytes.extend_from_slice(&u64::try_from(payload.len()).expect("fixture length fits").to_le_bytes());
bytes.extend_from_slice(&Sha256::digest(payload));
bytes.extend_from_slice(&Sha256::digest(&bytes));
bytes
}
async fn disk(root: &TempDir, name: &str) -> EcstoreDiskStore {
let path = root.path().join(name);
std::fs::create_dir_all(&path).expect("create disk directory");
let endpoint = Endpoint::try_from(path.to_string_lossy().as_ref()).expect("valid disk endpoint");
let disk = new_disk(
&endpoint,
&DiskOption {
cleanup: false,
health_check: false,
},
)
.await
.expect("open disk");
let result = EcstoreDiskAPI::make_volume(disk.as_ref(), RUSTFS_META_BUCKET).await;
assert!(
matches!(result, Ok(()) | Err(EcstoreDiskError::VolumeExists)),
"metadata volume: {result:?}"
);
disk
}
// Exercise the existing storage owner's atomic CAS primitive. No production
// caller publishes this format until ownership-aware replay is available.
async fn install(disk: &EcstoreDiskStore, path: &str, bytes: &[u8]) {
let expected = EcstoreDiskAPI::read_all(disk.as_ref(), RUSTFS_META_BUCKET, path).await.ok();
let result = EcstoreDiskAPI::compare_and_update_file(
disk.as_ref(),
RUSTFS_META_BUCKET,
path,
expected,
Some(EcstoreDiskBytes::copy_from_slice(bytes)),
)
.await
.expect("atomic snapshot slot write");
assert_eq!(result, EcstoreConditionalFileUpdate::Updated);
}
async fn commit(disk: &EcstoreDiskStore, slot: usize, owner: Uuid, sequence: u64, bytes: &[u8]) {
install(disk, PAYLOAD_PATHS[slot], bytes).await;
install(disk, MANIFEST_PATHS[slot], &manifest(owner, sequence, bytes)).await;
}
#[test]
fn manifest_validates_identity_sequence_length_and_digest() {
let bytes = payload("object");
let owner = Uuid::new_v4();
assert!(CommittedSnapshot::decode(&manifest(owner, 1, &bytes), bytes.clone(), bytes.len()).is_ok());
for (owner, sequence) in [(Uuid::nil(), 1), (owner, 0), (owner, u64::MAX)] {
assert!(matches!(
Manifest::decode(&manifest(owner, sequence, &bytes), bytes.len()),
Err(SnapshotError::Corrupt)
));
}
assert!(matches!(
Manifest::decode(&manifest(owner, 1, &bytes), bytes.len() - 1),
Err(SnapshotError::TooLarge)
));
let mut corrupt = manifest(owner, 1, &bytes);
corrupt[25] ^= 1;
assert!(matches!(Manifest::decode(&corrupt, bytes.len()), Err(SnapshotError::Corrupt)));
let mut unsupported = manifest(owner, 1, &bytes);
unsupported[8] = 2;
assert!(matches!(Manifest::decode(&unsupported, bytes.len()), Err(SnapshotError::Unsupported)));
}
#[test]
fn whole_payload_integrity_is_required_even_with_a_valid_manifest() {
let bytes = payload("object");
let owner = Uuid::new_v4();
let header = manifest(owner, 1, &bytes);
assert!(matches!(
CommittedSnapshot::decode(&header, bytes[..bytes.len() - 1].to_vec(), bytes.len()),
Err(SnapshotError::Corrupt)
));
let invalid = b"not an MRF record".to_vec();
assert!(matches!(
CommittedSnapshot::decode(&manifest(owner, 2, &invalid), invalid, bytes.len()),
Err(SnapshotError::Corrupt)
));
}
#[tokio::test]
async fn newest_complete_replica_wins_in_both_disk_orders() {
let root = TempDir::new().expect("test directory");
let first = disk(&root, "first").await;
let second = disk(&root, "second").await;
let owner = Uuid::new_v4();
commit(&first, 0, owner, 1, &payload("old")).await;
commit(&second, 1, owner, 2, &payload("new")).await;
for disks in [vec![first.clone(), second.clone()], vec![second.clone(), first.clone()]] {
let recovered = read_committed(&disks, 4096)
.await
.expect("read replicas")
.expect("committed snapshot");
assert_eq!(recovered.manifest.sequence, 2);
assert_eq!(recovered.payload, payload("new"));
}
}
#[tokio::test]
async fn divergent_commits_at_same_sequence_fail_closed() {
let root = TempDir::new().expect("test directory");
let first = disk(&root, "first").await;
let second = disk(&root, "second").await;
let owner = Uuid::new_v4();
commit(&first, 0, owner, 7, &payload("a")).await;
commit(&second, 1, owner, 7, &payload("b")).await;
assert!(matches!(read_committed(&[first, second], 4096).await, Err(SnapshotError::Conflict)));
}
#[tokio::test]
async fn newer_slot_does_not_hide_a_conflicting_commit_history() {
let root = TempDir::new().expect("test directory");
let first = disk(&root, "first").await;
let second = disk(&root, "second").await;
let owner = Uuid::new_v4();
commit(&first, 0, owner, 8, &payload("newest")).await;
commit(&first, 1, owner, 7, &payload("a")).await;
commit(&second, 1, owner, 7, &payload("b")).await;
assert!(matches!(read_committed(&[first, second], 4096).await, Err(SnapshotError::Conflict)));
}
#[tokio::test]
async fn uncommitted_or_torn_successor_preserves_previous_slot() {
let root = TempDir::new().expect("test directory");
let disk = disk(&root, "disk").await;
let owner = Uuid::new_v4();
let old = payload("old");
let next = payload("next");
commit(&disk, 0, owner, 1, &old).await;
install(&disk, PAYLOAD_PATHS[1], &next).await;
let recovered = read_committed(std::slice::from_ref(&disk), 4096)
.await
.expect("staged payload is not a commit")
.expect("old snapshot");
assert_eq!(recovered.payload, old);
install(&disk, MANIFEST_PATHS[1], &manifest(owner, 2, &next)[..20]).await;
let recovered = read_committed(std::slice::from_ref(&disk), 4096)
.await
.expect("torn manifest preserves old slot")
.expect("old snapshot");
assert_eq!(recovered.manifest.sequence, 1);
install(&disk, MANIFEST_PATHS[1], &manifest(owner, 2, &next)).await;
install(&disk, PAYLOAD_PATHS[1], b"torn").await;
let recovered = read_committed(&[disk], 4096)
.await
.expect("torn payload preserves old slot")
.expect("old snapshot");
assert_eq!(recovered.manifest.sequence, 1);
}
#[tokio::test]
async fn stale_manifest_cas_cannot_replace_committed_anchor() {
let root = TempDir::new().expect("test directory");
let disk = disk(&root, "disk").await;
let owner = Uuid::new_v4();
let bytes = payload("object");
commit(&disk, 0, owner, 1, &bytes).await;
let result = EcstoreDiskAPI::compare_and_update_file(
disk.as_ref(),
RUSTFS_META_BUCKET,
MANIFEST_PATHS[0],
None,
Some(manifest(owner, 2, &bytes).into()),
)
.await
.expect("CAS call");
assert_eq!(result, EcstoreConditionalFileUpdate::Mismatch);
let recovered = read_committed(&[disk], 4096)
.await
.expect("read old anchor")
.expect("snapshot");
assert_eq!(recovered.manifest.sequence, 1);
}
#[tokio::test]
async fn legacy_import_requires_complete_consistent_replicas() {
let root = TempDir::new().expect("test directory");
let first = disk(&root, "first").await;
let second = disk(&root, "second").await;
let bytes = payload("object");
for (disk, data) in [(&first, &bytes[..bytes.len() - 1]), (&second, bytes.as_slice())] {
EcstoreDiskAPI::write_all(
disk.as_ref(),
RUSTFS_META_BUCKET,
MRF_SCOPED_JOURNAL_PATH,
EcstoreDiskBytes::copy_from_slice(data),
)
.await
.expect("legacy fixture");
}
let disks = [first.clone(), second];
assert!(
matches!(read_recovery_snapshot(&disks, 4096).await.expect("intact legacy replica"), Some(RecoverySnapshot::Legacy(data)) if data == bytes)
);
EcstoreDiskAPI::write_all(first.as_ref(), RUSTFS_META_BUCKET, MRF_SCOPED_JOURNAL_PATH, payload("different").into())
.await
.expect("divergent fixture");
assert!(matches!(read_recovery_snapshot(&disks, 4096).await, Err(SnapshotError::Conflict)));
}
#[tokio::test]
async fn committed_inspection_leaves_payload_and_manifest_unchanged() {
let root = TempDir::new().expect("test directory");
let disk = disk(&root, "disk").await;
let owner = Uuid::new_v4();
let bytes = payload("object");
commit(&disk, 0, owner, 3, &bytes).await;
assert!(matches!(
read_recovery_snapshot(std::slice::from_ref(&disk), 4096)
.await
.expect("new snapshot"),
Some(RecoverySnapshot::Committed(_))
));
assert_eq!(
EcstoreDiskAPI::read_all(disk.as_ref(), RUSTFS_META_BUCKET, MANIFEST_PATHS[0])
.await
.expect("manifest retained")
.as_ref(),
manifest(owner, 3, &bytes)
);
assert_eq!(
EcstoreDiskAPI::read_all(disk.as_ref(), RUSTFS_META_BUCKET, PAYLOAD_PATHS[0])
.await
.expect("payload retained")
.as_ref(),
bytes
);
}
#[tokio::test]
async fn legacy_inspection_rejects_complete_subsets_and_scope_ambiguity() {
let scoped = |set_index| {
let intent = MrfIntent {
bucket: Arc::from("bucket"),
object: Arc::from("a"),
version_id: None,
kind: MrfKind::PartialWrite,
scope: Some(MrfScope {
pool_index: 0,
set_index,
}),
lease: None,
enqueued_at_ms: 1234,
attempts: 0,
};
let mut bytes = Vec::new();
assert!(encode_intent(&intent, &mut bytes), "scoped fixture must encode");
bytes
};
let mut superset = payload("a");
superset.extend_from_slice(&payload("b"));
for (case, first_bytes, second_bytes) in [
("complete-subset", payload("a"), superset),
("different-set", scoped(1), scoped(2)),
("unknown-scope", payload("a"), scoped(1)),
] {
let root = TempDir::new().expect("test directory");
let first = disk(&root, "first").await;
let second = disk(&root, "second").await;
for (disk, bytes) in [(&first, &first_bytes), (&second, &second_bytes)] {
assert_eq!(decode_journal(bytes).1, 0, "{case}: complete fixture");
EcstoreDiskAPI::write_all(
disk.as_ref(),
RUSTFS_META_BUCKET,
MRF_SCOPED_JOURNAL_PATH,
EcstoreDiskBytes::copy_from_slice(bytes),
)
.await
.expect("write legacy replica");
}
for disks in [vec![first.clone(), second.clone()], vec![second.clone(), first.clone()]] {
assert!(
matches!(read_recovery_snapshot(&disks, 4096).await, Err(SnapshotError::Conflict)),
"{case}: neither replica order proves a latest snapshot"
);
}
for (disk, bytes) in [(&first, &first_bytes), (&second, &second_bytes)] {
assert_eq!(
EcstoreDiskAPI::read_all(disk.as_ref(), RUSTFS_META_BUCKET, MRF_SCOPED_JOURNAL_PATH)
.await
.expect("legacy evidence retained")
.as_ref(),
bytes.as_slice(),
"{case}: inspection must preserve both source replicas"
);
}
}
}
#[tokio::test]
async fn oversized_or_corrupt_scoped_snapshot_never_falls_back_to_legacy() {
let root = TempDir::new().expect("test directory");
let disk = disk(&root, "disk").await;
EcstoreDiskAPI::write_all(disk.as_ref(), RUSTFS_META_BUCKET, MRF_SCOPED_JOURNAL_PATH, vec![0; 1025].into())
.await
.expect("oversized fixture");
EcstoreDiskAPI::write_all(disk.as_ref(), RUSTFS_META_BUCKET, MRF_JOURNAL_PATH, payload("old").into())
.await
.expect("legacy fixture");
assert!(matches!(
read_recovery_snapshot(std::slice::from_ref(&disk), 1024).await,
Err(SnapshotError::TooLarge)
));
assert!(matches!(read_recovery_snapshot(&[disk], 2048).await, Err(SnapshotError::Corrupt)));
}
#[tokio::test]
async fn empty_legacy_replica_cannot_erase_records_in_a_torn_replica() {
let root = TempDir::new().expect("test directory");
let first = disk(&root, "first").await;
let second = disk(&root, "second").await;
let mut incomplete = payload("durable-object");
incomplete.extend_from_slice(b"torn");
EcstoreDiskAPI::write_all(first.as_ref(), RUSTFS_META_BUCKET, MRF_SCOPED_JOURNAL_PATH, Vec::new().into())
.await
.expect("empty truncated replica");
EcstoreDiskAPI::write_all(second.as_ref(), RUSTFS_META_BUCKET, MRF_SCOPED_JOURNAL_PATH, incomplete.clone().into())
.await
.expect("records and torn tail");
for disks in [vec![first.clone(), second.clone()], vec![second.clone(), first.clone()]] {
assert!(matches!(read_recovery_snapshot(&disks, 4096).await, Err(SnapshotError::Corrupt)));
}
assert_eq!(
EcstoreDiskAPI::read_all(second.as_ref(), RUSTFS_META_BUCKET, MRF_SCOPED_JOURNAL_PATH)
.await
.expect("recovery anchor preserved")
.as_ref(),
incomplete
);
}
#[tokio::test]
async fn unreadable_commit_record_never_implies_legacy_only() {
let root = TempDir::new().expect("test directory");
let disk = disk(&root, "disk").await;
let legacy = payload("old");
EcstoreDiskAPI::write_all(disk.as_ref(), RUSTFS_META_BUCKET, MRF_JOURNAL_PATH, legacy.clone().into())
.await
.expect("legacy fixture");
// Opening a directory as a record either fails at open or at read,
// depending on the platform. Neither outcome proves absence.
std::fs::create_dir(root.path().join("disk").join(RUSTFS_META_BUCKET).join(MANIFEST_PATHS[0]))
.expect("unreadable manifest fixture");
let recovered = read_recovery_snapshot(std::slice::from_ref(&disk), 4096).await;
assert!(
matches!(recovered, Err(SnapshotError::Disk(_) | SnapshotError::Read(_))),
"must preserve unavailable proof: {recovered:?}"
);
assert_eq!(
EcstoreDiskAPI::read_all(disk.as_ref(), RUSTFS_META_BUCKET, MRF_JOURNAL_PATH)
.await
.expect("legacy remains")
.as_ref(),
legacy
);
}
}
@@ -44,9 +44,7 @@ use walkdir::WalkDir;
mod storage_api; mod storage_api;
use storage_api::integration::{ use storage_api::integration::{BucketOperations, ECStore, MakeBucketOptions, ObjectIO as _, ObjectOperations as _};
BucketOperations, ECStore, MakeBucketOptions, NamespaceLocking as _, ObjectIO as _, ObjectOperations as _,
};
/// 256 KiB + change: large enough to be stored as non-inline erasure shards /// 256 KiB + change: large enough to be stored as non-inline erasure shards
/// (so each data version materializes as an on-disk `part.*` file we can assert /// (so each data version materializes as an on-disk `part.*` file we can assert
@@ -108,7 +106,6 @@ async fn put_versioned(ecstore: &Arc<ECStore>, bucket: &str, object: &str, data:
.put_object(bucket, object, &mut reader, &opts) .put_object(bucket, object, &mut reader, &opts)
.await .await
.expect("versioned put_object failed"); .expect("versioned put_object failed");
wait_for_put_tail(ecstore, bucket, object).await;
info.version_id info.version_id
.map(|u| u.to_string()) .map(|u| u.to_string())
.expect("versioned put must return a version id") .expect("versioned put must return a version id")
@@ -120,7 +117,6 @@ async fn put_unversioned(ecstore: &Arc<ECStore>, bucket: &str, object: &str, dat
.put_object(bucket, object, &mut reader, &ObjectOptions::default()) .put_object(bucket, object, &mut reader, &ObjectOptions::default())
.await .await
.expect("unversioned put_object failed"); .expect("unversioned put_object failed");
wait_for_put_tail(ecstore, bucket, object).await;
} }
/// Create a delete-marker as the latest version (versioned:true, no version_id) /// Create a delete-marker as the latest version (versioned:true, no version_id)
@@ -164,16 +160,20 @@ fn xl_meta_path(obj_dir: &Path) -> PathBuf {
obj_dir.join("xl.meta") obj_dir.join("xl.meta")
} }
async fn wait_for_put_tail(ecstore: &Arc<ECStore>, bucket: &str, object: &str) { async fn wait_for_two_version_copies(disks: &[PathBuf], bucket: &str, object: &str) {
// Shards and xl.meta can exist before the detached PUT owner finishes. tokio::time::timeout(Duration::from_secs(5), async {
let lock = ecstore loop {
.new_ns_lock(bucket, object) if disks.iter().all(|disk| {
.await let object_dir = object_dir(disk, bucket, object);
.expect("fixture namespace lock should be created"); xl_meta_path(&object_dir).exists() && count_part_files(&object_dir) >= 2
let _settled = lock }) {
.get_write_lock(Duration::from_secs(30)) break;
.await }
.expect("PUT rename tail must finish before inspecting or wiping the fixture"); tokio::time::sleep(Duration::from_millis(10)).await;
}
})
.await
.expect("PUT rename tails must converge before wiping the versioned fixture");
} }
fn recreate_heal_opts() -> HealOpts { fn recreate_heal_opts() -> HealOpts {
@@ -305,13 +305,7 @@ mod serial_tests {
let data_v2 = versioned_test_data(20); let data_v2 = versioned_test_data(20);
let v1 = put_versioned(&ecstore, bucket, object, &data_v1).await; // OLD, non-latest let v1 = put_versioned(&ecstore, bucket, object, &data_v1).await; // OLD, non-latest
let v2 = put_versioned(&ecstore, bucket, object, &data_v2).await; // latest let v2 = put_versioned(&ecstore, bucket, object, &data_v2).await; // latest
assert!( wait_for_two_version_copies(&disk_paths, bucket, object).await;
disk_paths.iter().all(|disk| {
let dir = object_dir(disk, bucket, object);
xl_meta_path(&dir).exists() && count_part_files(&dir) >= 2
}),
"both versions must exist on every disk before wiping the fixture"
);
// ── Pre-wipe: prove the fixture actually has 2 versions on disk[0] ── // ── Pre-wipe: prove the fixture actually has 2 versions on disk[0] ──
let obj_dir0 = object_dir(&disk_paths[0], bucket, object); let obj_dir0 = object_dir(&disk_paths[0], bucket, object);
-1
View File
@@ -23,7 +23,6 @@ pub(crate) mod integration {
pub(crate) use rustfs_ecstore::api::storage::ECStore; pub(crate) use rustfs_ecstore::api::storage::ECStore;
pub(crate) use rustfs_storage_api::BucketOperations; pub(crate) use rustfs_storage_api::BucketOperations;
pub(crate) use rustfs_storage_api::MakeBucketOptions; pub(crate) use rustfs_storage_api::MakeBucketOptions;
pub(crate) use rustfs_storage_api::NamespaceLocking;
pub(crate) use rustfs_storage_api::ObjectIO; pub(crate) use rustfs_storage_api::ObjectIO;
pub(crate) use rustfs_storage_api::ObjectOperations; pub(crate) use rustfs_storage_api::ObjectOperations;
} }
+44 -460
View File
@@ -43,10 +43,6 @@ const ERR_LIFECYCLE_BUCKET_LOCKED: &str =
"ExpiredObjectAllVersions element and DelMarkerExpiration action cannot be used on an object locked bucket"; "ExpiredObjectAllVersions element and DelMarkerExpiration action cannot be used on an object locked bucket";
const ERR_LIFECYCLE_TOO_MANY_RULES: &str = "Lifecycle configuration should have at most 1000 rules"; const ERR_LIFECYCLE_TOO_MANY_RULES: &str = "Lifecycle configuration should have at most 1000 rules";
const ERR_LIFECYCLE_INVALID_EXPIRATION_DAYS: &str = "'Days' for Expiration action must be a positive integer"; const ERR_LIFECYCLE_INVALID_EXPIRATION_DAYS: &str = "'Days' for Expiration action must be a positive integer";
const ERR_LIFECYCLE_EXPIRATION_DAYS_DATE_CONFLICT: &str = "Expiration cannot specify both Days and Date";
const ERR_LIFECYCLE_MULTIPLE_TRANSITIONS: &str = "Only one Transition action per lifecycle rule is supported";
const ERR_LIFECYCLE_MULTIPLE_NONCURRENT_TRANSITIONS: &str =
"Only one NoncurrentVersionTransition action per lifecycle rule is supported";
const ERR_LIFECYCLE_INVALID_NONCURRENT_EXPIRATION_DAYS: &str = const ERR_LIFECYCLE_INVALID_NONCURRENT_EXPIRATION_DAYS: &str =
"'NoncurrentDays' for NoncurrentVersionExpiration action must be a positive integer"; "'NoncurrentDays' for NoncurrentVersionExpiration action must be a positive integer";
const ERR_LIFECYCLE_INVALID_ABORT_INCOMPLETE_MPU_DAYS: &str = const ERR_LIFECYCLE_INVALID_ABORT_INCOMPLETE_MPU_DAYS: &str =
@@ -365,12 +361,6 @@ impl Lifecycle for BucketLifecycleConfiguration {
{ {
return Err(std::io::Error::other(ERR_LIFECYCLE_INVALID_EXPIRED_OBJECT_ALL_VERSIONS)); return Err(std::io::Error::other(ERR_LIFECYCLE_INVALID_EXPIRED_OBJECT_ALL_VERSIONS));
} }
if expiration.days.is_some() && expiration.date.is_some() {
return Err(std::io::Error::new(
std::io::ErrorKind::InvalidInput,
ERR_LIFECYCLE_EXPIRATION_DAYS_DATE_CONFLICT,
));
}
if let Some(expiration_date) = &expiration.date { if let Some(expiration_date) = &expiration.date {
let date = OffsetDateTime::from(expiration_date.clone()); let date = OffsetDateTime::from(expiration_date.clone());
if date.hour() != 0 || date.minute() != 0 || date.second() != 0 || date.nanosecond() != 0 { if date.hour() != 0 || date.minute() != 0 || date.second() != 0 || date.nanosecond() != 0 {
@@ -404,20 +394,11 @@ impl Lifecycle for BucketLifecycleConfiguration {
} }
} }
if let Some(transitions) = &r.transitions { if let Some(transitions) = &r.transitions {
if transitions.len() > 1 {
return Err(std::io::Error::new(std::io::ErrorKind::InvalidInput, ERR_LIFECYCLE_MULTIPLE_TRANSITIONS));
}
for transition in transitions { for transition in transitions {
TransitionOps::validate(transition)?; TransitionOps::validate(transition)?;
} }
} }
if let Some(noncurrent_transitions) = &r.noncurrent_version_transitions { if let Some(noncurrent_transitions) = &r.noncurrent_version_transitions {
if noncurrent_transitions.len() > 1 {
return Err(std::io::Error::new(
std::io::ErrorKind::InvalidInput,
ERR_LIFECYCLE_MULTIPLE_NONCURRENT_TRANSITIONS,
));
}
for transition in noncurrent_transitions { for transition in noncurrent_transitions {
NoncurrentVersionTransitionOps::validate(transition)?; NoncurrentVersionTransitionOps::validate(transition)?;
} }
@@ -492,8 +473,6 @@ impl Lifecycle for BucketLifecycleConfiguration {
} }
async fn eval(&self, obj: &ObjectOpts) -> Event { async fn eval(&self, obj: &ObjectOpts) -> Event {
// A single-object lookup cannot prove how many newer historical versions
// survive. Count-dependent actions wait for the complete-group evaluator.
self.eval_inner(obj, OffsetDateTime::now_utc(), 0).await self.eval_inner(obj, OffsetDateTime::now_utc(), 0).await
} }
@@ -557,8 +536,23 @@ impl Lifecycle for BucketLifecycleConfiguration {
return Event::default(); return Event::default();
}; };
if let Some(event) = obj.restored_copy_expiry(now) { if let Some(restore_expires) = obj.restore_expires
events.push(event); && restore_expires.unix_timestamp() != 0
&& now.unix_timestamp() > restore_expires.unix_timestamp()
{
let mut action = IlmAction::DeleteRestoredAction;
if !obj.is_latest {
action = IlmAction::DeleteRestoredVersionAction;
}
events.push(Event {
action,
due: Some(now),
rule_id: "".into(),
noncurrent_days: 0,
newer_noncurrent_versions: 0,
storage_class: "".into(),
});
} }
if let Some(ref lc_rules) = self.filter_rules(obj).await { if let Some(ref lc_rules) = self.filter_rules(obj).await {
@@ -617,12 +611,17 @@ impl Lifecycle for BucketLifecycleConfiguration {
continue; continue;
} }
if !obj.is_latest
&& let Some(ref noncurrent_version_expiration) = rule.noncurrent_version_expiration
&& let Some(retain_newer_noncurrent_versions) = noncurrent_version_expiration.newer_noncurrent_versions
&& newer_noncurrent_versions < usize::try_from(retain_newer_noncurrent_versions).unwrap_or(usize::MAX)
{
continue;
}
if !obj.is_latest if !obj.is_latest
&& let Some(ref noncurrent_version_expiration) = rule.noncurrent_version_expiration && let Some(ref noncurrent_version_expiration) = rule.noncurrent_version_expiration
&& let Some(noncurrent_days) = noncurrent_version_expiration.noncurrent_days && let Some(noncurrent_days) = noncurrent_version_expiration.noncurrent_days
&& noncurrent_version_expiration
.newer_noncurrent_versions
.is_none_or(|retain| usize::try_from(retain).is_ok_and(|retain| newer_noncurrent_versions >= retain))
{ {
if let Some(successor_mod_time) = obj.successor_mod_time { if let Some(successor_mod_time) = obj.successor_mod_time {
let expected_expiry = expected_expiry_time(successor_mod_time, noncurrent_days); let expected_expiry = expected_expiry_time(successor_mod_time, noncurrent_days);
@@ -652,11 +651,7 @@ impl Lifecycle for BucketLifecycleConfiguration {
&& let Some(noncurrent_version_transition) = rule && let Some(noncurrent_version_transition) = rule
.noncurrent_version_transitions .noncurrent_version_transitions
.as_ref() .as_ref()
.filter(|transitions| transitions.len() == 1)
.and_then(|transitions| transitions.first()) .and_then(|transitions| transitions.first())
&& noncurrent_version_transition
.newer_noncurrent_versions
.is_none_or(|retain| usize::try_from(retain).is_ok_and(|retain| newer_noncurrent_versions >= retain))
&& let Some(storage_class) = noncurrent_version_transition.storage_class.as_ref() && let Some(storage_class) = noncurrent_version_transition.storage_class.as_ref()
&& !storage_class.as_str().is_empty() && !storage_class.as_str().is_empty()
&& !obj.delete_marker && !obj.delete_marker
@@ -740,11 +735,7 @@ impl Lifecycle for BucketLifecycleConfiguration {
} }
if obj.transition_status != TRANSITION_COMPLETE if obj.transition_status != TRANSITION_COMPLETE
&& let Some(transition) = rule && let Some(transition) = rule.transitions.as_ref().and_then(|transitions| transitions.first())
.transitions
.as_ref()
.filter(|transitions| transitions.len() == 1)
.and_then(|transitions| transitions.first())
&& let Some(storage_class) = transition.storage_class.as_ref() && let Some(storage_class) = transition.storage_class.as_ref()
&& !storage_class.as_str().is_empty() && !storage_class.as_str().is_empty()
{ {
@@ -767,15 +758,18 @@ impl Lifecycle for BucketLifecycleConfiguration {
} }
if !events.is_empty() { if !events.is_empty() {
// Eligible expiration takes precedence over transition, even when a // Select the winning event using a strict total order (MinIO semantics):
// failed transition has an earlier deadline. Within each action class, // the earliest `due` wins, and ties break toward delete-type actions. A
// prefer the earliest deadline using a deterministic total order. // missing `due` is treated as UNIX_EPOCH. This replaces a hand-written
// `sort_by` comparator that was not a strict weak ordering (it could return
// `Ordering::Less` for both `(a, b)` and `(b, a)`), which panics on the
// repository toolchain and did not deterministically pick the earliest event.
let event = events let event = events
.iter() .iter()
.min_by_key(|event| { .min_by_key(|event| {
( (
ilm_action_priority_rank(&event.action),
event.due.unwrap_or(OffsetDateTime::UNIX_EPOCH).unix_timestamp(), event.due.unwrap_or(OffsetDateTime::UNIX_EPOCH).unix_timestamp(),
ilm_action_priority_rank(&event.action),
) )
}) })
.cloned() .cloned()
@@ -1048,27 +1042,6 @@ impl ObjectOpts {
pub fn expired_object_deletemarker(&self) -> bool { pub fn expired_object_deletemarker(&self) -> bool {
self.delete_marker && self.is_latest && self.num_versions == 1 self.delete_marker && self.is_latest && self.num_versions == 1
} }
pub(crate) fn restored_copy_expiry(&self, now: OffsetDateTime) -> Option<Event> {
let restore_expires = self.restore_expires?;
// Restore metadata alone does not prove that a durable remote copy exists.
if self.transition_status != TRANSITION_COMPLETE
|| restore_expires.unix_timestamp() == 0
|| now.unix_timestamp() <= restore_expires.unix_timestamp()
{
return None;
}
let action = if self.is_latest {
IlmAction::DeleteRestoredAction
} else {
IlmAction::DeleteRestoredVersionAction
};
expiration_action_has_valid_target(action, self.version_id, self.is_latest, self.delete_marker).then(|| Event {
action,
due: Some(now),
..Default::default()
})
}
} }
/// Returns whether an expiry action has enough identity to target the object /// Returns whether an expiry action has enough identity to target the object
@@ -1091,8 +1064,11 @@ pub fn expiration_action_has_valid_target(
} }
} }
/// Eligible logical expiration takes precedence over transition and restore-copy /// Total-order rank for lifecycle actions used to break `due` ties.
/// cleanup. Deadlines break ties within an action class. ///
/// Delete-type actions rank before every other action so that, when two events
/// share the same `due`, a delete wins (MinIO semantics). The concrete numeric
/// values only matter relative to each other.
fn ilm_action_priority_rank(action: &IlmAction) -> u8 { fn ilm_action_priority_rank(action: &IlmAction) -> u8 {
match action { match action {
IlmAction::DeleteAllVersionsAction IlmAction::DeleteAllVersionsAction
@@ -4183,392 +4159,6 @@ mod tests {
assert_eq!(event.action, IlmAction::NoneAction); assert_eq!(event.action, IlmAction::NoneAction);
} }
mod adversarial_regressions {
use super::*;
use s3s::dto::NoncurrentVersionExpiration;
fn run(test: impl std::future::Future<Output = ()>) {
with_default_ilm_process_time(|| {
tokio::runtime::Builder::new_current_thread()
.build()
.expect("lifecycle regression runtime should build")
.block_on(test);
});
}
fn noncurrent_object() -> ObjectOpts {
ObjectOpts {
name: "logs/object".to_string(),
mod_time: Some(datetime!(2020-01-01 00:00:00 UTC)),
successor_mod_time: Some(datetime!(2020-01-02 00:00:00 UTC)),
version_id: Some(Uuid::from_u128(1)),
size: 1024 * 1024,
..Default::default()
}
}
#[test]
#[serial]
fn noncurrent_transition_retains_the_requested_newer_versions() {
run(async {
let mut rule = enabled_rule(None, None, Some("retain-two-hot-versions"));
rule.filter = Some(LifecycleRuleFilter::default());
rule.noncurrent_version_transitions = Some(vec![NoncurrentVersionTransition {
noncurrent_days: Some(1),
newer_noncurrent_versions: Some(2),
storage_class: Some(TransitionStorageClass::from_static("WARM")),
}]);
let lc = Arc::new(BucketLifecycleConfiguration {
rules: vec![rule],
expiry_updated_at: None,
});
lc.validate(&ObjectLockConfiguration::default())
.await
.expect("valid noncurrent transition policy");
let objects = (0..4)
.map(|index| ObjectOpts {
mod_time: Some(datetime!(2020-01-05 00:00:00 UTC) - Duration::days(index)),
successor_mod_time: (index > 0).then_some(datetime!(2020-01-06 00:00:00 UTC) - Duration::days(index)),
version_id: Some(Uuid::from_u128(u128::try_from(index + 1).expect("small version index"))),
is_latest: index == 0,
num_versions: 4,
..noncurrent_object()
})
.collect::<Vec<_>>();
let actions = crate::Evaluator::new(lc)
.eval(&objects)
.await
.expect("complete version chain should evaluate")
.into_iter()
.map(|event| event.action)
.collect::<Vec<_>>();
assert_eq!(
actions,
[
IlmAction::NoneAction,
IlmAction::NoneAction,
IlmAction::NoneAction,
IlmAction::TransitionVersionAction
],
"the two newest noncurrent versions must remain in their current storage class"
);
});
}
#[test]
#[serial]
fn noncurrent_transition_checks_count_age_and_single_object_context() {
run(async {
let mut rule = enabled_rule(None, None, Some("retain-two"));
rule.filter = Some(LifecycleRuleFilter::default());
rule.noncurrent_version_transitions = Some(vec![NoncurrentVersionTransition {
noncurrent_days: Some(3),
newer_noncurrent_versions: Some(2),
storage_class: Some(TransitionStorageClass::from_static("WARM")),
}]);
let mut lc = BucketLifecycleConfiguration {
rules: vec![rule],
expiry_updated_at: None,
};
lc.validate(&ObjectLockConfiguration::default())
.await
.expect("valid counted transition");
let object = noncurrent_object();
let now = datetime!(2020-01-10 00:00:00 UTC);
for (newer, expected) in [
(0, IlmAction::NoneAction),
(1, IlmAction::NoneAction),
(2, IlmAction::TransitionVersionAction),
(3, IlmAction::TransitionVersionAction),
] {
assert_eq!(lc.eval_inner(&object, now, newer).await.action, expected, "newer count: {newer}");
}
assert_eq!(
lc.eval_inner(&object, datetime!(2020-01-04 00:00:00 UTC), 2).await.action,
IlmAction::NoneAction,
"the retention count does not replace the age condition"
);
assert_eq!(
lc.eval(&object).await.action,
IlmAction::NoneAction,
"a single-object lookup must not assume a complete version history"
);
for retain in [None, Some(0), Some(-1), Some(i32::MAX)] {
lc.rules[0]
.noncurrent_version_transitions
.as_mut()
.expect("transition exists")[0]
.newer_noncurrent_versions = retain;
let expected = if matches!(retain, None | Some(0)) {
IlmAction::TransitionVersionAction
} else {
IlmAction::NoneAction
};
assert_eq!(lc.eval_inner(&object, now, 2).await.action, expected, "retention: {retain:?}");
}
});
}
#[test]
#[serial]
fn noncurrent_expiration_and_transition_have_independent_retention_counts() {
run(async {
let mut rule = enabled_rule(None, None, Some("independent-counts"));
rule.filter = Some(LifecycleRuleFilter::default());
rule.noncurrent_version_expiration = Some(NoncurrentVersionExpiration {
noncurrent_days: Some(90),
newer_noncurrent_versions: Some(4),
});
rule.noncurrent_version_transitions = Some(vec![NoncurrentVersionTransition {
noncurrent_days: Some(30),
newer_noncurrent_versions: Some(2),
storage_class: Some(TransitionStorageClass::from_static("WARM")),
}]);
let lc = BucketLifecycleConfiguration {
rules: vec![rule],
expiry_updated_at: None,
};
lc.validate(&ObjectLockConfiguration::default())
.await
.expect("valid independent retention limits");
let object = noncurrent_object();
let now = datetime!(2020-05-01 00:00:00 UTC);
for (newer, expected) in [
(1, IlmAction::NoneAction),
(2, IlmAction::TransitionVersionAction),
(3, IlmAction::TransitionVersionAction),
(4, IlmAction::DeleteVersionAction),
] {
assert_eq!(lc.eval_inner(&object, now, newer).await.action, expected, "newer count: {newer}");
}
});
}
#[test]
#[serial]
fn expiration_retention_does_not_skip_an_independent_transition() {
run(async {
let mut rule = enabled_rule(None, None, Some("transition-then-expire"));
rule.filter = Some(LifecycleRuleFilter::default());
rule.noncurrent_version_transitions = Some(vec![NoncurrentVersionTransition {
noncurrent_days: Some(1),
newer_noncurrent_versions: None,
storage_class: Some(TransitionStorageClass::from_static("WARM")),
}]);
let mut lc = BucketLifecycleConfiguration {
rules: vec![rule],
expiry_updated_at: None,
};
let object = noncurrent_object();
let now = datetime!(2020-01-10 00:00:00 UTC);
let transition_only = lc.eval_inner(&object, now, 0).await;
assert_eq!(transition_only.action, IlmAction::TransitionVersionAction);
lc.rules[0].noncurrent_version_expiration = Some(NoncurrentVersionExpiration {
noncurrent_days: Some(90),
newer_noncurrent_versions: Some(2),
});
lc.validate(&ObjectLockConfiguration::default())
.await
.expect("valid combined policy");
let combined = lc.eval_inner(&object, now, 0).await;
assert_eq!(combined.action, transition_only.action, "retention limits expiration, not transition");
assert_eq!(combined.storage_class, transition_only.storage_class);
});
}
#[test]
#[serial]
fn current_transition_rejects_multiple_stages_in_any_order() {
run(async {
let mut rule = enabled_rule(None, None, Some("two-current-transitions"));
rule.transitions = Some(vec![
Transition {
date: Some(datetime!(2020-03-01 00:00:00 UTC).into()),
days: None,
storage_class: Some(TransitionStorageClass::from_static("COLD")),
},
Transition {
date: Some(datetime!(2020-01-03 00:00:00 UTC).into()),
days: None,
storage_class: Some(TransitionStorageClass::from_static("WARM")),
},
]);
let mut lc = BucketLifecycleConfiguration {
rules: vec![rule],
expiry_updated_at: None,
};
let object = ObjectOpts {
is_latest: true,
..noncurrent_object()
};
let now = datetime!(2020-01-10 00:00:00 UTC);
for status in [ExpirationStatus::ENABLED, ExpirationStatus::DISABLED] {
lc.rules[0].status = ExpirationStatus::from_static(status);
for _ in 0..2 {
let err = lc
.validate(&ObjectLockConfiguration::default())
.await
.expect_err("multiple transition stages must be rejected");
assert_eq!(err.kind(), std::io::ErrorKind::InvalidInput);
assert_eq!(err.to_string(), ERR_LIFECYCLE_MULTIPLE_TRANSITIONS);
assert_eq!(
lc.eval_inner(&object, now, 0).await.action,
IlmAction::NoneAction,
"legacy multi-stage configurations must not silently execute their first stage"
);
lc.rules[0]
.transitions
.as_mut()
.expect("transition array is present")
.reverse();
}
}
lc.rules[0]
.transitions
.as_mut()
.expect("transition array is present")
.remove(0);
lc.rules[0].status = ExpirationStatus::from_static(ExpirationStatus::ENABLED);
lc.validate(&ObjectLockConfiguration::default())
.await
.expect("one stage is supported");
let event = lc.eval_inner(&object, now, 0).await;
assert_eq!(event.action, IlmAction::TransitionAction);
assert_eq!(event.storage_class, "WARM");
});
}
#[test]
#[serial]
fn noncurrent_transition_rejects_multiple_stages_in_any_order() {
run(async {
let mut rule = enabled_rule(None, None, Some("two-noncurrent-transitions"));
rule.noncurrent_version_transitions = Some(vec![
NoncurrentVersionTransition {
noncurrent_days: Some(30),
newer_noncurrent_versions: None,
storage_class: Some(TransitionStorageClass::from_static("COLD")),
},
NoncurrentVersionTransition {
noncurrent_days: Some(1),
newer_noncurrent_versions: None,
storage_class: Some(TransitionStorageClass::from_static("WARM")),
},
]);
let mut lc = BucketLifecycleConfiguration {
rules: vec![rule],
expiry_updated_at: None,
};
let object = noncurrent_object();
let now = datetime!(2020-01-10 00:00:00 UTC);
for status in [ExpirationStatus::ENABLED, ExpirationStatus::DISABLED] {
lc.rules[0].status = ExpirationStatus::from_static(status);
for _ in 0..2 {
let err = lc
.validate(&ObjectLockConfiguration::default())
.await
.expect_err("multiple noncurrent transition stages must be rejected");
assert_eq!(err.kind(), std::io::ErrorKind::InvalidInput);
assert_eq!(err.to_string(), ERR_LIFECYCLE_MULTIPLE_NONCURRENT_TRANSITIONS);
assert_eq!(
lc.eval_inner(&object, now, 0).await.action,
IlmAction::NoneAction,
"legacy multi-stage configurations must not silently execute their first stage"
);
lc.rules[0]
.noncurrent_version_transitions
.as_mut()
.expect("transition array is present")
.reverse();
}
}
lc.rules[0]
.noncurrent_version_transitions
.as_mut()
.expect("transition array is present")
.remove(0);
lc.rules[0].status = ExpirationStatus::from_static(ExpirationStatus::ENABLED);
lc.validate(&ObjectLockConfiguration::default())
.await
.expect("one stage is supported");
let event = lc.eval_inner(&object, now, 0).await;
assert_eq!(event.action, IlmAction::TransitionVersionAction);
assert_eq!(event.storage_class, "WARM");
});
}
#[test]
#[serial]
fn expiration_rejects_simultaneous_days_and_date() {
run(async {
let mut lc = BucketLifecycleConfiguration {
rules: vec![enabled_rule(
Some(LifecycleExpiration {
days: Some(1),
..Default::default()
}),
None,
Some("ambiguous-expiry"),
)],
expiry_updated_at: None,
};
lc.validate(&ObjectLockConfiguration::default())
.await
.expect("a single Days expiration is valid");
lc.rules[0].expiration.as_mut().expect("expiration is present").date =
Some(datetime!(2099-01-01 00:00:00 UTC).into());
let err = lc
.validate(&ObjectLockConfiguration::default())
.await
.expect_err("Days and Date are mutually exclusive; accepting both silently overrides Days");
assert_eq!(err.kind(), std::io::ErrorKind::InvalidInput);
assert_eq!(err.to_string(), ERR_LIFECYCLE_EXPIRATION_DAYS_DATE_CONFLICT);
});
}
#[test]
#[serial]
fn overdue_transition_does_not_starve_permanent_expiration() {
run(async {
let mut rule = enabled_rule(
Some(LifecycleExpiration {
days: Some(90),
..Default::default()
}),
None,
Some("archive-then-delete"),
);
rule.transitions = Some(vec![Transition {
days: Some(30),
date: None,
storage_class: Some(TransitionStorageClass::from_static("WARM")),
}]);
let lc = BucketLifecycleConfiguration {
rules: vec![rule],
expiry_updated_at: None,
};
lc.validate(&ObjectLockConfiguration::default())
.await
.expect("valid transition and expiration policy");
let object = ObjectOpts {
is_latest: true,
version_id: None,
transition_status: TRANSITION_PENDING.to_string(),
..noncurrent_object()
};
let before_expiration = lc.eval_inner(&object, datetime!(2020-02-15 00:00:00 UTC), 0).await;
assert_eq!(before_expiration.action, IlmAction::TransitionAction);
let overdue = lc.eval_inner(&object, datetime!(2020-05-01 00:00:00 UTC), 0).await;
assert_eq!(
overdue.action,
IlmAction::DeleteAction,
"an unavailable tier must not prevent permanent expiration indefinitely"
);
});
}
}
/// Property-based tests for the rule evaluator (backlog#1148 ilm-14, /// Property-based tests for the rule evaluator (backlog#1148 ilm-14,
/// follow-up to backlog#1030 / rustfs#4455). /// follow-up to backlog#1030 / rustfs#4455).
/// ///
@@ -4579,7 +4169,7 @@ mod tests {
/// ///
/// * `eval_inner` never panics and is deterministic for a fixed input; /// * `eval_inner` never panics and is deterministic for a fixed input;
/// * the winning event matches an independently recomputed candidate set: /// * the winning event matches an independently recomputed candidate set:
/// eligible expiration wins over transition, then earliest `due` wins (the /// earliest `due` wins, ties break toward delete-class actions (the
/// `min_by_key` selection that replaced the rustfs#4455 comparator); /// `min_by_key` selection that replaced the rustfs#4455 comparator);
/// * `expected_expiry_time` is monotonically non-decreasing in `days` and /// * `expected_expiry_time` is monotonically non-decreasing in `days` and
/// always lands on the processing boundary, both at production defaults /// always lands on the processing boundary, both at production defaults
@@ -4868,8 +4458,8 @@ mod tests {
/// consider for a live current version under `selection`-shaped rules /// consider for a live current version under `selection`-shaped rules
/// (expiration and first-transition only, no filters): expiration /// (expiration and first-transition only, no filters): expiration
/// fires when `now >= due`, transition when `now > due` and the object /// fires when `now >= due`, transition when `now > due` and the object
/// has not already transitioned. Eligible expiration wins over transition; /// has not already transitioned. Selection semantics under test:
/// the earliest deadline wins within the selected action class. /// earliest due wins, ties prefer delete-class.
fn oracle_candidates(lc: &BucketLifecycleConfiguration, obj: &ObjectOpts, now: OffsetDateTime) -> Vec<Candidate> { fn oracle_candidates(lc: &BucketLifecycleConfiguration, obj: &ObjectOpts, now: OffsetDateTime) -> Vec<Candidate> {
let mod_time = obj.mod_time.expect("selection strategy always sets mod_time"); let mod_time = obj.mod_time.expect("selection strategy always sets mod_time");
let mut candidates = Vec::new(); let mut candidates = Vec::new();
@@ -4958,8 +4548,8 @@ mod tests {
/// Differential test of winner selection (the rustfs#4455 fix): /// Differential test of winner selection (the rustfs#4455 fix):
/// for a live current version under randomized expiration and /// for a live current version under randomized expiration and
/// transition rules, `eval_inner`'s winner must carry the /// transition rules, `eval_inner`'s winner must carry the
/// earliest expiration from the independently recomputed candidate /// minimum `(due, rank)` of the independently recomputed
/// set, or the earliest transition when no expiration is eligible, /// candidate set — earliest due wins, ties prefer delete-class —
/// and must be `NoneAction` exactly when that set is empty. /// and must be `NoneAction` exactly when that set is empty.
#[test] #[test]
#[serial] #[serial]
@@ -4988,13 +4578,7 @@ mod tests {
// Oracle and evaluator must observe the same (pinned) time env. // Oracle and evaluator must observe the same (pinned) time env.
let (event, expected) = with_production_time_env(|| { let (event, expected) = with_production_time_env(|| {
let candidates = oracle_candidates(&lc, &obj, now); let expected = oracle_candidates(&lc, &obj, now).into_iter().min();
let expected = candidates
.iter()
.filter(|(_, rank)| *rank == 0)
.min()
.copied()
.or_else(|| candidates.into_iter().min());
let rt = tokio::runtime::Builder::new_current_thread() let rt = tokio::runtime::Builder::new_current_thread()
.enable_all() .enable_all()
.build() .build()
+7 -93
View File
@@ -116,10 +116,13 @@ impl Evaluator {
break 'top_loop; break 'top_loop;
} }
} }
// Restore expiry removes only the temporary local copy; the IlmAction::DeleteAction
// retained logical version and its remote data remain intact. | IlmAction::DeleteRestoredAction
IlmAction::DeleteAction | IlmAction::DeleteVersionAction if self.is_object_locked(obj) => { | IlmAction::DeleteVersionAction
event = obj.restored_copy_expiry(now).unwrap_or_default(); | IlmAction::DeleteRestoredVersionAction
if self.is_object_locked(obj) =>
{
event = Event::default();
} }
_ => {} _ => {}
} }
@@ -203,95 +206,6 @@ mod tests {
use super::*; use super::*;
use rustfs_replication::{ReplicationStatusType, VersionPurgeStatusType}; use rustfs_replication::{ReplicationStatusType, VersionPurgeStatusType};
#[tokio::test]
async fn adversarial_restore_expiry_survives_legal_hold() {
let mut policy = (*latest_expiration_lifecycle()).clone();
policy.rules[0].status = ExpirationStatus::from_static(ExpirationStatus::DISABLED);
let policy = Arc::new(policy);
policy
.validate(&lock_enabled_without_default_retention())
.await
.expect("valid disabled lifecycle rule");
let mut objects = [true, false].map(|is_latest| ObjectOpts {
is_latest,
num_versions: 2,
mod_time: Some(
OffsetDateTime::from_unix_timestamp(if is_latest { 1_200_000 } else { 1_000_000 })
.expect("fixed version timestamp"),
),
successor_mod_time: (!is_latest)
.then(|| OffsetDateTime::from_unix_timestamp(1_200_000).expect("fixed successor timestamp")),
transition_status: crate::TRANSITION_COMPLETE.to_string(),
restore_expires: Some(OffsetDateTime::from_unix_timestamp(2_000_000).expect("fixed expired restore timestamp")),
..current_object_opts(ReplicationStatusType::Completed)
});
let evaluator = Evaluator::new(policy).with_lock_retention(Some(lock_enabled_without_default_retention()));
let expected = [IlmAction::DeleteRestoredAction, IlmAction::DeleteRestoredVersionAction];
let unlocked = evaluator
.eval(&objects)
.await
.expect("unlocked restored versions should evaluate");
assert_eq!(unlocked.iter().map(|event| event.action).collect::<Vec<_>>(), expected);
for object in &mut objects {
object
.user_defined
.insert(X_AMZ_OBJECT_LOCK_LEGAL_HOLD.as_str().to_string(), "ON".to_string());
}
let locked = evaluator
.eval(&objects)
.await
.expect("locked restored versions should evaluate");
assert_eq!(
locked.iter().map(|event| event.action).collect::<Vec<_>>(),
expected,
"expiring a restored local copy preserves the retained logical version and remote object"
);
let mut expiring_policy = (*latest_expiration_lifecycle()).clone();
expiring_policy.rules[0].noncurrent_version_expiration = Some(NoncurrentVersionExpiration {
noncurrent_days: Some(1),
newer_noncurrent_versions: None,
});
let expiring_evaluator =
Evaluator::new(Arc::new(expiring_policy)).with_lock_retention(Some(lock_enabled_without_default_retention()));
let locked = expiring_evaluator
.eval(&objects)
.await
.expect("locked expired versions should evaluate");
assert_eq!(
locked.iter().map(|event| event.action).collect::<Vec<_>>(),
expected,
"blocked logical expiration must still allow an eligible restore-copy cleanup"
);
for status in [ReplicationStatusType::Pending, ReplicationStatusType::Failed] {
for object in &mut objects {
object.replication_status = status.clone();
}
for evaluator in [&evaluator, &expiring_evaluator] {
let events = evaluator.eval(&objects).await.expect("pending replication should evaluate");
assert!(events.iter().all(|event| event.action == IlmAction::NoneAction));
}
}
for object in &mut objects {
object.replication_status = ReplicationStatusType::Completed;
}
for transition_status in ["", crate::TRANSITION_PENDING, "unknown"] {
for object in &mut objects {
object.transition_status = transition_status.to_string();
}
for evaluator in [&evaluator, &expiring_evaluator] {
let events = evaluator.eval(&objects).await.expect("incomplete transition should evaluate");
assert!(
events.iter().all(|event| event.action == IlmAction::NoneAction),
"restore metadata cannot authorize cleanup without a completed transition"
);
}
}
}
fn expired_marker_lifecycle() -> Arc<BucketLifecycleConfiguration> { fn expired_marker_lifecycle() -> Arc<BucketLifecycleConfiguration> {
Arc::new(BucketLifecycleConfiguration { Arc::new(BucketLifecycleConfiguration {
expiry_updated_at: None, expiry_updated_at: None,
@@ -1 +1 @@
{"bucket":"photos","config":{"version":1,"enabled":true,"source":{"provider":"minio","endpoint":"https://source.example.com:9000","region":"us-east-1","bucket":"legacy-photos","path_style":"auto","credentials":{"access_key":"AKIASOURCE","secret_key":"REDACTED","session_token":null},"tls":{"skip_verify":false,"ca_cert_pem":null},"azure":null,"gcs":null},"filter":{"prefix":null,"source_prefix":"photos/"},"policy":{"head":"proxy","range_get":"serve_and_backfill","source_error":"propagate","list_through":false,"respect_local_delete_marker":true,"preserve_etag":true,"copy_tags":false,"emit_events":true,"negative_cache_ttl_secs":30,"inline_max_bytes":16777216,"multipart_part_size_bytes":67108864,"max_concurrent_pulls":8,"pull_queue_capacity":1024,"source_timeout":{"connect_ms":5000,"first_byte_ms":15000,"idle_ms":30000},"bandwidth_limit_bytes_per_sec":null}},"updated_at":"2026-09-02T10:00:00Z"} {"bucket":"photos","config":{"version":1,"enabled":true,"source":{"provider":"minio","endpoint":"https://source.example.com:9000","region":"us-east-1","bucket":"legacy-photos","path_style":"auto","credentials":{"access_key":"AKIASOURCE","secret_key":"REDACTED","session_token":null},"tls":{"skip_verify":false,"ca_cert_pem":null}},"filter":{"prefix":null,"source_prefix":"photos/"},"policy":{"head":"proxy","range_get":"serve_and_backfill","source_error":"propagate","list_through":false,"respect_local_delete_marker":true,"preserve_etag":true,"copy_tags":false,"emit_events":true,"negative_cache_ttl_secs":30,"inline_max_bytes":16777216,"multipart_part_size_bytes":67108864,"max_concurrent_pulls":8,"pull_queue_capacity":1024,"source_timeout":{"connect_ms":5000,"first_byte_ms":15000,"idle_ms":30000},"bandwidth_limit_bytes_per_sec":null}},"updated_at":"2026-09-02T10:00:00Z"}
@@ -1 +1 @@
{"version":1,"enabled":true,"source":{"provider":"minio","endpoint":"https://source.example.com:9000","region":"us-east-1","bucket":"legacy-photos","path_style":"auto","credentials":{"access_key":"AKIASOURCE","secret_key":"sourceSecretKey123","session_token":null},"tls":{"skip_verify":false,"ca_cert_pem":null},"azure":null,"gcs":null},"filter":{"prefix":null,"source_prefix":"photos/"},"policy":{"head":"proxy","range_get":"serve_and_backfill","source_error":"propagate","list_through":false,"respect_local_delete_marker":true,"preserve_etag":true,"copy_tags":false,"emit_events":true,"negative_cache_ttl_secs":30,"inline_max_bytes":16777216,"multipart_part_size_bytes":67108864,"max_concurrent_pulls":8,"pull_queue_capacity":1024,"source_timeout":{"connect_ms":5000,"first_byte_ms":15000,"idle_ms":30000},"bandwidth_limit_bytes_per_sec":null}} {"version":1,"enabled":true,"source":{"provider":"minio","endpoint":"https://source.example.com:9000","region":"us-east-1","bucket":"legacy-photos","path_style":"auto","credentials":{"access_key":"AKIASOURCE","secret_key":"sourceSecretKey123","session_token":null},"tls":{"skip_verify":false,"ca_cert_pem":null}},"filter":{"prefix":null,"source_prefix":"photos/"},"policy":{"head":"proxy","range_get":"serve_and_backfill","source_error":"propagate","list_through":false,"respect_local_delete_marker":true,"preserve_etag":true,"copy_tags":false,"emit_events":true,"negative_cache_ttl_secs":30,"inline_max_bytes":16777216,"multipart_part_size_bytes":67108864,"max_concurrent_pulls":8,"pull_queue_capacity":1024,"source_timeout":{"connect_ms":5000,"first_byte_ms":15000,"idle_ms":30000},"bandwidth_limit_bytes_per_sec":null}}
@@ -1 +1 @@
{"bucket":"photos","dry_run":false,"config":{"version":1,"enabled":true,"source":{"provider":"minio","endpoint":"https://source.example.com:9000","region":"us-east-1","bucket":"legacy-photos","path_style":"auto","credentials":{"access_key":"AKIASOURCE","secret_key":"REDACTED","session_token":null},"tls":{"skip_verify":false,"ca_cert_pem":null},"azure":null,"gcs":null},"filter":{"prefix":null,"source_prefix":"photos/"},"policy":{"head":"proxy","range_get":"serve_and_backfill","source_error":"propagate","list_through":false,"respect_local_delete_marker":true,"preserve_etag":true,"copy_tags":false,"emit_events":true,"negative_cache_ttl_secs":30,"inline_max_bytes":16777216,"multipart_part_size_bytes":67108864,"max_concurrent_pulls":8,"pull_queue_capacity":1024,"source_timeout":{"connect_ms":5000,"first_byte_ms":15000,"idle_ms":30000},"bandwidth_limit_bytes_per_sec":null}},"updated_at":"2026-09-02T10:00:00Z","probe":{"reachable":true,"listable":true,"sample_key":"photos/2024/01.jpg"}} {"bucket":"photos","dry_run":false,"config":{"version":1,"enabled":true,"source":{"provider":"minio","endpoint":"https://source.example.com:9000","region":"us-east-1","bucket":"legacy-photos","path_style":"auto","credentials":{"access_key":"AKIASOURCE","secret_key":"REDACTED","session_token":null},"tls":{"skip_verify":false,"ca_cert_pem":null}},"filter":{"prefix":null,"source_prefix":"photos/"},"policy":{"head":"proxy","range_get":"serve_and_backfill","source_error":"propagate","list_through":false,"respect_local_delete_marker":true,"preserve_etag":true,"copy_tags":false,"emit_events":true,"negative_cache_ttl_secs":30,"inline_max_bytes":16777216,"multipart_part_size_bytes":67108864,"max_concurrent_pulls":8,"pull_queue_capacity":1024,"source_timeout":{"connect_ms":5000,"first_byte_ms":15000,"idle_ms":30000},"bandwidth_limit_bytes_per_sec":null}},"updated_at":"2026-09-02T10:00:00Z","probe":{"reachable":true,"listable":true,"sample_key":"photos/2024/01.jpg"}}
+1 -89
View File
@@ -17,7 +17,7 @@
//! Wire types for `PUT`/`GET`/`DELETE /v3/on-demand-migration/{bucket}`, //! Wire types for `PUT`/`GET`/`DELETE /v3/on-demand-migration/{bucket}`,
//! `GET .../status`, `POST .../backfill?op=start|cancel` and //! `GET .../status`, `POST .../backfill?op=start|cancel` and
//! `GET .../backfill` (ODM-12), mirroring the server's config model //! `GET .../backfill` (ODM-12), mirroring the server's config model
//! (`rustfs/src/on_demand_migration/config.rs`) and handler //! (`crates/ecstore/src/bucket/on_demand_migration/config.rs`) and handler
//! responses (`rustfs/src/admin/handlers/on_demand_migration.rs`). The SDK //! responses (`rustfs/src/admin/handlers/on_demand_migration.rs`). The SDK
//! owns its own copies, madmin-go style; the fixtures under //! owns its own copies, madmin-go style; the fixtures under
//! `fixtures/on_demand_migration/` are the contract both sides pin //! `fixtures/on_demand_migration/` are the contract both sides pin
@@ -78,18 +78,10 @@ pub struct OnDemandMigrationSource {
#[serde(default)] #[serde(default)]
pub path_style: OnDemandMigrationPathStyle, pub path_style: OnDemandMigrationPathStyle,
/// `None` means anonymous access to a public source bucket. /// `None` means anonymous access to a public source bucket.
/// `None` means anonymous access to a public source bucket. The native
/// providers carry their credentials in `azure` / `gcs` instead.
#[serde(default)] #[serde(default)]
pub credentials: Option<OnDemandMigrationCredentials>, pub credentials: Option<OnDemandMigrationCredentials>,
#[serde(default)] #[serde(default)]
pub tls: OnDemandMigrationTls, pub tls: OnDemandMigrationTls,
/// Required for `azure` and rejected for every other provider.
#[serde(default)]
pub azure: Option<OnDemandMigrationAzure>,
/// Required for `gcs_native` and rejected for every other provider.
#[serde(default)]
pub gcs: Option<OnDemandMigrationGcs>,
} }
#[derive(Debug, Clone, Copy, PartialEq, Eq, Serialize, Deserialize)] #[derive(Debug, Clone, Copy, PartialEq, Eq, Serialize, Deserialize)]
@@ -100,49 +92,7 @@ pub enum OnDemandMigrationProvider {
Minio, Minio,
Rustfs, Rustfs,
R2, R2,
/// GCS XML interoperability API with HMAC keys.
Gcs, Gcs,
/// Native Azure Blob service.
Azure,
/// Native GCS JSON API with a service-account key.
#[serde(rename = "gcs_native")]
GcsNative,
}
/// Native Azure Blob parameters. The container is `source.bucket`; exactly one
/// of `account_key` and `sas_token` is set. Responses carry both as `REDACTED`.
#[derive(Clone, PartialEq, Eq, Serialize, Deserialize)]
pub struct OnDemandMigrationAzure {
pub account: String,
#[serde(default)]
pub account_key: Option<String>,
#[serde(default)]
pub sas_token: Option<String>,
}
impl fmt::Debug for OnDemandMigrationAzure {
fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result {
f.debug_struct("OnDemandMigrationAzure")
.field("account", &self.account)
.field("account_key", &self.account_key.as_ref().map(|_| "REDACTED"))
.field("sas_token", &self.sas_token.as_ref().map(|_| "REDACTED"))
.finish()
}
}
/// Native GCS parameters. The bucket is `source.bucket`; the key JSON embeds a
/// private key, so responses carry it as `REDACTED`.
#[derive(Clone, PartialEq, Eq, Serialize, Deserialize)]
pub struct OnDemandMigrationGcs {
pub service_account_json: String,
}
impl fmt::Debug for OnDemandMigrationGcs {
fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result {
f.debug_struct("OnDemandMigrationGcs")
.field("service_account_json", &"REDACTED")
.finish()
}
} }
#[derive(Debug, Clone, Copy, PartialEq, Eq, Default, Serialize, Deserialize)] #[derive(Debug, Clone, Copy, PartialEq, Eq, Default, Serialize, Deserialize)]
@@ -856,8 +806,6 @@ mod tests {
session_token: None, session_token: None,
}), }),
tls: OnDemandMigrationTls::default(), tls: OnDemandMigrationTls::default(),
azure: None,
gcs: None,
}); });
let mut expected: OnDemandMigrationConfig = serde_json::from_str(SET_REQUEST_FIXTURE.trim()).expect("fixture"); let mut expected: OnDemandMigrationConfig = serde_json::from_str(SET_REQUEST_FIXTURE.trim()).expect("fixture");
expected.filter.source_prefix = None; expected.filter.source_prefix = None;
@@ -873,42 +821,6 @@ mod tests {
assert!(minimal.source.credentials.is_none()); assert!(minimal.source.credentials.is_none());
} }
#[test]
fn native_provider_documents_round_trip_and_hide_their_secrets() {
for (label, json) in [
(
"azure",
r#"{"provider":"azure","endpoint":null,"region":"auto","bucket":"legacy-photos","path_style":"auto","credentials":null,"tls":{"skip_verify":false,"ca_cert_pem":null},"azure":{"account":"legacyaccount","account_key":null,"sas_token":"sv=2021-08-06&sig=topsecret"},"gcs":null}"#,
),
(
"gcs_native",
r#"{"provider":"gcs_native","endpoint":null,"region":"auto","bucket":"legacy-photos","path_style":"auto","credentials":null,"tls":{"skip_verify":false,"ca_cert_pem":null},"azure":null,"gcs":{"service_account_json":"{\"type\":\"service_account\"}"}}"#,
),
] {
let source: OnDemandMigrationSource = serde_json::from_str(json).unwrap_or_else(|err| panic!("{label}: {err}"));
assert_eq!(
serde_json::to_string(&source).expect("re-encodes"),
json,
"{label} must reproduce the server wire shape byte for byte"
);
}
let azure = OnDemandMigrationAzure {
account: "legacyaccount".to_string(),
account_key: Some("c2VjcmV0".to_string()),
sas_token: Some("sig=topsecret".to_string()),
};
let rendered = format!("{azure:?}");
assert!(rendered.contains("legacyaccount"));
assert!(!rendered.contains("c2VjcmV0"), "{rendered}");
assert!(!rendered.contains("topsecret"), "{rendered}");
let gcs = OnDemandMigrationGcs {
service_account_json: r#"{"private_key":"-----BEGIN PRIVATE KEY-----"}"#.to_string(),
};
assert!(!format!("{gcs:?}").contains("PRIVATE KEY"), "{gcs:?}");
}
#[test] #[test]
fn credentials_debug_never_prints_secrets() { fn credentials_debug_never_prints_secrets() {
let credentials = OnDemandMigrationCredentials { let credentials = OnDemandMigrationCredentials {
-1
View File
@@ -38,4 +38,3 @@ pub(crate) use storage_api::metrics::{
obs_on_demand_migration_snapshot, obs_replication_site_stats_snapshot, obs_resolve_object_store_handle, obs_on_demand_migration_snapshot, obs_replication_site_stats_snapshot, obs_resolve_object_store_handle,
obs_transition_state_handle, obs_transition_state_handle,
}; };
pub use storage_api::register_on_demand_migration_metrics_source;
+113 -51
View File
@@ -17,6 +17,13 @@ use std::time::Duration;
pub(crate) use rustfs_ecstore::api::bucket::bandwidth::monitor::Monitor as ObsBucketBandwidthMonitor; pub(crate) use rustfs_ecstore::api::bucket::bandwidth::monitor::Monitor as ObsBucketBandwidthMonitor;
pub(crate) use rustfs_ecstore::api::bucket::metadata_sys::get_quota_config as obs_get_quota_config; pub(crate) use rustfs_ecstore::api::bucket::metadata_sys::get_quota_config as obs_get_quota_config;
use rustfs_ecstore::api::bucket::on_demand_migration::backfill::{
BackfillCheckpoint as SourceBackfillCheckpoint, global_backfill_runner as source_global_backfill_runner,
};
use rustfs_ecstore::api::bucket::on_demand_migration::{
BreakerState as SourceOdmBreakerState, OdmBucketSnapshot as SourceOdmBucketSnapshot,
OnDemandMigrationSys as SourceOnDemandMigrationSys,
};
use rustfs_ecstore::api::bucket::replication::{ use rustfs_ecstore::api::bucket::replication::{
BucketReplicationStats as SourceBucketReplicationStats, DurableMrfBucketBacklog, DurableMrfTargetBacklog, BucketReplicationStats as SourceBucketReplicationStats, DurableMrfBucketBacklog, DurableMrfTargetBacklog,
MrfBucketBacklogObservability, RuntimeReplicationTargetBacklog, durable_mrf_backlog_summary_snapshot, MrfBucketBacklogObservability, RuntimeReplicationTargetBacklog, durable_mrf_backlog_summary_snapshot,
@@ -37,7 +44,9 @@ pub(crate) use rustfs_ecstore::api::runtime::{
pub(crate) use rustfs_ecstore::api::storage::ECStore as ObsStore; pub(crate) use rustfs_ecstore::api::storage::ECStore as ObsStore;
use rustfs_storage_api as storage_contracts; use rustfs_storage_api as storage_contracts;
use crate::metrics::collectors::{OdmBackfillBucketStats, OdmBackfillRuntimeStats, OnDemandMigrationBucketStats}; use crate::metrics::collectors::{
OdmBackfillBucketStats, OdmBackfillRuntimeStats, OnDemandMigrationBreakerState, OnDemandMigrationBucketStats,
};
#[derive(Debug, Clone, PartialEq)] #[derive(Debug, Clone, PartialEq)]
pub(crate) struct ObsBucketReplicationTargetStatsSnapshot { pub(crate) struct ObsBucketReplicationTargetStatsSnapshot {
@@ -456,37 +465,70 @@ pub(crate) async fn obs_bucket_replication_stats_snapshot() -> Vec<ObsBucketRepl
buckets buckets
} }
struct OnDemandMigrationMetricsSource { fn on_demand_migration_stats_from_snapshot(snapshot: SourceOdmBucketSnapshot) -> OnDemandMigrationBucketStats {
snapshot: fn() -> Vec<OnDemandMigrationBucketStats>, let stats = snapshot.stats;
backfill_snapshot: fn() -> Vec<OdmBackfillBucketStats>, OnDemandMigrationBucketStats {
} bucket: snapshot.bucket,
requests_total: stats.requests_total,
static ON_DEMAND_MIGRATION_METRICS_SOURCE: std::sync::OnceLock<OnDemandMigrationMetricsSource> = std::sync::OnceLock::new(); pulled_bytes_total: stats.pulled_bytes_total,
pulled_objects_total: stats.pulled_objects_total,
/// Register the application-owned ODM snapshots before starting the collector. pull_failures_total: stats.pull_failures_total,
pub fn register_on_demand_migration_metrics_source( inflight_pulls: stats.inflight_pulls,
snapshot: fn() -> Vec<OnDemandMigrationBucketStats>, queue_depth: stats.queue_depth,
backfill_snapshot: fn() -> Vec<OdmBackfillBucketStats>, source_latency_buckets: stats
) -> bool { .source_latency
ON_DEMAND_MIGRATION_METRICS_SOURCE .buckets
.set(OnDemandMigrationMetricsSource { .into_iter()
snapshot, .map(|bucket| (bucket.le_ms, bucket.count))
backfill_snapshot, .collect(),
}) source_latency_count: stats.source_latency.count,
.is_ok() source_latency_sum_ms: stats.source_latency.sum_ms,
breaker_state: match stats.breaker_state {
SourceOdmBreakerState::Closed => OnDemandMigrationBreakerState::Closed,
SourceOdmBreakerState::HalfOpen => OnDemandMigrationBreakerState::HalfOpen,
SourceOdmBreakerState::Open => OnDemandMigrationBreakerState::Open,
},
}
} }
/// Every bucket with live on-demand migration state on this node, sorted by
/// name. Empty while the module switch is off.
pub(crate) fn obs_on_demand_migration_snapshot() -> Vec<OnDemandMigrationBucketStats> { pub(crate) fn obs_on_demand_migration_snapshot() -> Vec<OnDemandMigrationBucketStats> {
ON_DEMAND_MIGRATION_METRICS_SOURCE SourceOnDemandMigrationSys::get()
.get() .snapshot()
.map(|source| (source.snapshot)()) .into_iter()
.unwrap_or_default() .map(on_demand_migration_stats_from_snapshot)
.collect()
} }
fn on_demand_migration_backfill_stats_from_checkpoint(
bucket: String,
checkpoint: SourceBackfillCheckpoint,
) -> OdmBackfillBucketStats {
OdmBackfillBucketStats {
bucket,
state: checkpoint.state.as_str().to_string(),
listed: checkpoint.listed,
enqueued: checkpoint.enqueued,
pulled: checkpoint.pulled,
skipped_existing: checkpoint.skipped_existing,
failed: checkpoint.failed,
bytes: checkpoint.bytes,
}
}
/// Backfill jobs running on this node, sorted by bucket. Empty until the
/// runner is installed, and empty again once a job finishes: the series are
/// per-node job progress, not a cluster-wide history.
pub(crate) fn obs_on_demand_migration_backfill_snapshot(server: String) -> OdmBackfillRuntimeStats { pub(crate) fn obs_on_demand_migration_backfill_snapshot(server: String) -> OdmBackfillRuntimeStats {
let buckets = ON_DEMAND_MIGRATION_METRICS_SOURCE let buckets = source_global_backfill_runner()
.get() .map(|runner| {
.map(|source| (source.backfill_snapshot)()) runner
.local_job_snapshots()
.into_iter()
.map(|(bucket, checkpoint)| on_demand_migration_backfill_stats_from_checkpoint(bucket, checkpoint))
.collect()
})
.unwrap_or_default(); .unwrap_or_default();
OdmBackfillRuntimeStats { server, buckets } OdmBackfillRuntimeStats { server, buckets }
} }
@@ -538,31 +580,6 @@ pub(crate) async fn obs_replication_site_stats_snapshot(current_data_transfer_ra
mod tests { mod tests {
use super::*; use super::*;
#[test]
fn on_demand_migration_callbacks_supply_runtime_snapshots() {
assert!(register_on_demand_migration_metrics_source(
|| vec![OnDemandMigrationBucketStats {
bucket: "configured".into(),
pulled_bytes_total: 4096,
..Default::default()
}],
|| vec![OdmBackfillBucketStats {
bucket: "backfill".into(),
pulled: 3,
..Default::default()
}],
));
let snapshot = obs_on_demand_migration_snapshot();
assert_eq!(snapshot.len(), 1);
assert_eq!(snapshot[0].bucket, "configured");
assert_eq!(snapshot[0].pulled_bytes_total, 4096);
let backfill = obs_on_demand_migration_backfill_snapshot("node-a".into());
assert_eq!(backfill.server, "node-a");
assert_eq!(backfill.buckets.len(), 1);
assert_eq!(backfill.buckets[0].bucket, "backfill");
assert_eq!(backfill.buckets[0].pulled, 3);
}
#[test] #[test]
fn obs_replication_numeric_conversions_floor_negative_values() { fn obs_replication_numeric_conversions_floor_negative_values() {
assert_eq!(i64_to_u64_floor_zero(-1), 0); assert_eq!(i64_to_u64_floor_zero(-1), 0);
@@ -755,6 +772,51 @@ mod tests {
assert_eq!(snapshot.mrf_last_flush_duration_millis, 4); assert_eq!(snapshot.mrf_last_flush_duration_millis, 4);
} }
#[test]
fn on_demand_migration_snapshot_projects_counters_and_breaker_state() {
// Built from JSON: the snapshot's timestamps use `time`, which obs does not depend on.
let snapshot: SourceOdmBucketSnapshot = serde_json::from_value(serde_json::json!({
"bucket": "photos",
"provider": "minio",
"endpoint_host": "source.example.com",
"applied_at": "2026-09-02T10:00:00Z",
"client_error": null,
"negative_cache_entries": 0,
"inflight_keys": 1,
"max_concurrent_pulls": 8,
"stats": {
"requests_total": {"get": {"source_hit": 2}},
"pulled_bytes_total": 4096,
"pulled_objects_total": {"inline": 1},
"pull_failures_total": {"source_timeout": 1},
"inflight_pulls": 1,
"queue_depth": 2,
"source_latency": {
"buckets": [{"le_ms": 5, "count": 1}, {"le_ms": 10, "count": 2}],
"count": 3,
"sum_ms": 90753
},
"last_source_error": {"class": "server_error", "at": "2026-09-02T10:00:00Z"},
"breaker_state": "open"
}
}))
.expect("runtime snapshot decodes");
let stats = on_demand_migration_stats_from_snapshot(snapshot);
assert_eq!(stats.bucket, "photos");
assert_eq!(stats.requests_total["get"]["source_hit"], 2);
assert_eq!(stats.pulled_bytes_total, 4096);
assert_eq!(stats.pulled_objects_total["inline"], 1);
assert_eq!(stats.pull_failures_total["source_timeout"], 1);
assert_eq!(stats.inflight_pulls, 1);
assert_eq!(stats.queue_depth, 2);
assert_eq!(stats.source_latency_buckets, vec![(5, 1), (10, 2)]);
assert_eq!(stats.source_latency_count, 3);
assert_eq!(stats.source_latency_sum_ms, 90_753);
assert_eq!(stats.breaker_state, OnDemandMigrationBreakerState::Open);
}
#[test] #[test]
fn bucket_replication_snapshot_preserves_durable_mrf_unavailable_state() { fn bucket_replication_snapshot_preserves_durable_mrf_unavailable_state() {
let snapshot = bucket_replication_stats_snapshot_from_parts( let snapshot = bucket_replication_stats_snapshot_from_parts(
+48 -204
View File
@@ -120,10 +120,18 @@ impl TransitionClient {
let h = resp.headers().clone(); let h = resp.headers().clone();
let mut body = resp.into_body();
let body_vec = if let Some(limit) = max_response_bytes { let body_vec = if let Some(limit) = max_response_bytes {
self.collect_response_body(resp.into_body(), limit).await? collect_response_body(body, limit).await?
} else { } else {
self.collect_response_body_unbounded(resp.into_body()).await? let mut body_vec = Vec::new();
while let Some(frame) = body.frame().await {
let frame = frame.map_err(|e| std::io::Error::new(std::io::ErrorKind::Other, e.to_string()))?;
if let Some(data) = frame.data_ref() {
body_vec.extend_from_slice(data);
}
}
body_vec
}; };
Ok((object_stat, h, BufReader::new(Cursor::new(body_vec)))) Ok((object_stat, h, BufReader::new(Cursor::new(body_vec))))
} }
@@ -135,7 +143,7 @@ mod bounded_response_tests {
use crate::{ use crate::{
api_get_options::GetObjectOptions, api_get_options::GetObjectOptions,
credentials::{Credentials, SignatureType, Static, Value}, credentials::{Credentials, SignatureType, Static, Value},
transition_api::{BucketLookupType, Options, TransitionClient, TransitionClientTimeouts, collect_response_body}, transition_api::{BucketLookupType, Options, TransitionClient, collect_response_body},
}; };
use http_body_util::Full; use http_body_util::Full;
use hyper::body::Bytes; use hyper::body::Bytes;
@@ -167,31 +175,7 @@ mod bounded_response_tests {
assert_eq!(err.kind(), std::io::ErrorKind::InvalidData); assert_eq!(err.kind(), std::io::ErrorKind::InvalidData);
} }
fn test_options() -> Options { async fn bounded_get_fixture(body: &'static [u8]) -> Option<(TransitionClient, tokio::task::JoinHandle<String>)> {
Options {
creds: Credentials::new(Static(Value {
access_key_id: "access-key".to_string(),
secret_access_key: "secret-key".to_string(),
signer_type: SignatureType::SignatureV4,
..Default::default()
})),
region: "us-east-1".to_string(),
bucket_lookup: BucketLookupType::BucketLookupPath,
max_retries: 1,
..Default::default()
}
}
async fn client_for_endpoint(endpoint: &str, timeouts: TransitionClientTimeouts) -> TransitionClient {
TransitionClient::new_with_timeouts(endpoint, test_options(), "", timeouts)
.await
.expect("fixture client should build")
}
async fn bounded_get_fixture_with_timeouts(
body: &'static [u8],
timeouts: TransitionClientTimeouts,
) -> Option<(TransitionClient, tokio::task::JoinHandle<String>)> {
let listener = match TcpListener::bind("127.0.0.1:0").await { let listener = match TcpListener::bind("127.0.0.1:0").await {
Ok(listener) => listener, Ok(listener) => listener,
Err(err) if err.kind() == std::io::ErrorKind::PermissionDenied => return None, Err(err) if err.kind() == std::io::ErrorKind::PermissionDenied => return None,
@@ -225,14 +209,27 @@ mod bounded_response_tests {
stream.write_all(body).await.expect("fixture should write response body"); stream.write_all(body).await.expect("fixture should write response body");
request request
}); });
let client = client_for_endpoint(&endpoint, timeouts).await; let client = TransitionClient::new(
&endpoint,
Options {
creds: Credentials::new(Static(Value {
access_key_id: "access-key".to_string(),
secret_access_key: "secret-key".to_string(),
signer_type: SignatureType::SignatureV4,
..Default::default()
})),
region: "us-east-1".to_string(),
bucket_lookup: BucketLookupType::BucketLookupPath,
max_retries: 1,
..Default::default()
},
"",
)
.await
.expect("fixture client should build");
Some((client, request)) Some((client, request))
} }
async fn bounded_get_fixture(body: &'static [u8]) -> Option<(TransitionClient, tokio::task::JoinHandle<String>)> {
bounded_get_fixture_with_timeouts(body, TransitionClientTimeouts::default()).await
}
#[tokio::test] #[tokio::test]
async fn real_transport_accepts_the_exact_closed_range_length() { async fn real_transport_accepts_the_exact_closed_range_length() {
let Some((client, request)) = bounded_get_fixture(b"RustFS!").await else { let Some((client, request)) = bounded_get_fixture(b"RustFS!").await else {
@@ -295,7 +292,24 @@ mod bounded_response_tests {
.local_addr() .local_addr()
.expect("listener local address should be available") .expect("listener local address should be available")
.to_string(); .to_string();
let client = client_for_endpoint(&endpoint, TransitionClientTimeouts::default()).await; let client = TransitionClient::new(
&endpoint,
Options {
creds: Credentials::new(Static(Value {
access_key_id: "access-key".to_string(),
secret_access_key: "secret-key".to_string(),
signer_type: SignatureType::SignatureV4,
..Default::default()
})),
region: "us-east-1".to_string(),
bucket_lookup: BucketLookupType::BucketLookupPath,
max_retries: 1,
..Default::default()
},
"",
)
.await
.expect("fixture client should build");
let mut opts = GetObjectOptions::default(); let mut opts = GetObjectOptions::default();
opts.headers opts.headers
.insert("range".to_string(), "bytes=0-18446744073709551615".to_string()); .insert("range".to_string(), "bytes=0-18446744073709551615".to_string());
@@ -312,176 +326,6 @@ mod bounded_response_tests {
.is_err() .is_err()
); );
} }
#[tokio::test]
async fn connection_refused_returns_without_waiting_for_the_request_timeout() {
let listener = match TcpListener::bind("127.0.0.1:0").await {
Ok(listener) => listener,
Err(err) if err.kind() == std::io::ErrorKind::PermissionDenied => return,
Err(err) => panic!("test listener should bind: {err}"),
};
let endpoint = listener
.local_addr()
.expect("listener local address should be available")
.to_string();
drop(listener);
let client = client_for_endpoint(
&endpoint,
TransitionClientTimeouts::new(Duration::from_secs(1), Duration::from_secs(5), Duration::from_secs(1)),
)
.await;
let mut opts = GetObjectOptions::default();
opts.set_range(0, 6).expect("the probe range should be valid");
let result = tokio::time::timeout(Duration::from_secs(2), client.get_object_inner("bucket", "probe", &opts))
.await
.expect("connection refused should return before the broader request timeout");
assert!(result.is_err(), "connection refused must fail instead of hanging");
}
#[tokio::test]
async fn response_header_stall_returns_timed_out() {
let listener = match TcpListener::bind("127.0.0.1:0").await {
Ok(listener) => listener,
Err(err) if err.kind() == std::io::ErrorKind::PermissionDenied => return,
Err(err) => panic!("test listener should bind: {err}"),
};
let endpoint = listener
.local_addr()
.expect("listener local address should be available")
.to_string();
let fixture = tokio::spawn(async move {
let (mut stream, _) = listener.accept().await.expect("fixture should accept one GET");
let mut request = Vec::new();
let mut buffer = [0; 1024];
loop {
let read = stream.read(&mut buffer).await.expect("fixture should read request headers");
assert_ne!(read, 0, "connection closed before request headers were received");
request.extend_from_slice(&buffer[..read]);
if request.windows(4).any(|window| window == b"\r\n\r\n") {
break;
}
}
tokio::time::sleep(Duration::from_millis(200)).await;
});
let client = client_for_endpoint(
&endpoint,
TransitionClientTimeouts::new(Duration::from_secs(1), Duration::from_millis(50), Duration::from_secs(1)),
)
.await;
let mut opts = GetObjectOptions::default();
opts.set_range(0, 6).expect("the probe range should be valid");
let err = client
.get_object_inner("bucket", "probe", &opts)
.await
.expect_err("response header stalls must be bounded");
assert_eq!(err.kind(), std::io::ErrorKind::TimedOut);
fixture.await.expect("fixture should join");
}
#[tokio::test]
async fn response_body_idle_stall_returns_timed_out() {
let listener = match TcpListener::bind("127.0.0.1:0").await {
Ok(listener) => listener,
Err(err) if err.kind() == std::io::ErrorKind::PermissionDenied => return,
Err(err) => panic!("test listener should bind: {err}"),
};
let endpoint = listener
.local_addr()
.expect("listener local address should be available")
.to_string();
let fixture = tokio::spawn(async move {
let (mut stream, _) = listener.accept().await.expect("fixture should accept one GET");
let mut request = Vec::new();
let mut buffer = [0; 1024];
loop {
let read = stream.read(&mut buffer).await.expect("fixture should read request headers");
assert_ne!(read, 0, "connection closed before request headers were received");
request.extend_from_slice(&buffer[..read]);
if request.windows(4).any(|window| window == b"\r\n\r\n") {
break;
}
}
stream
.write_all(b"HTTP/1.1 206 Partial Content\r\nContent-Length: 7\r\nConnection: close\r\n\r\nRu")
.await
.expect("fixture should write the first body chunk");
tokio::time::sleep(Duration::from_millis(200)).await;
});
let client = client_for_endpoint(
&endpoint,
TransitionClientTimeouts::new(Duration::from_secs(1), Duration::from_secs(1), Duration::from_millis(50)),
)
.await;
let mut opts = GetObjectOptions::default();
opts.set_range(0, 6).expect("the probe range should be valid");
let err = client
.get_object_inner("bucket", "probe", &opts)
.await
.expect_err("body stalls after partial progress must be bounded");
assert_eq!(err.kind(), std::io::ErrorKind::TimedOut);
fixture.await.expect("fixture should join");
}
#[tokio::test]
async fn response_body_idle_timer_resets_on_progress() {
let listener = match TcpListener::bind("127.0.0.1:0").await {
Ok(listener) => listener,
Err(err) if err.kind() == std::io::ErrorKind::PermissionDenied => return,
Err(err) => panic!("test listener should bind: {err}"),
};
let endpoint = listener
.local_addr()
.expect("listener local address should be available")
.to_string();
let fixture = tokio::spawn(async move {
let (mut stream, _) = listener.accept().await.expect("fixture should accept one GET");
let mut request = Vec::new();
let mut buffer = [0; 1024];
loop {
let read = stream.read(&mut buffer).await.expect("fixture should read request headers");
assert_ne!(read, 0, "connection closed before request headers were received");
request.extend_from_slice(&buffer[..read]);
if request.windows(4).any(|window| window == b"\r\n\r\n") {
break;
}
}
stream
.write_all(b"HTTP/1.1 206 Partial Content\r\nContent-Length: 7\r\nConnection: close\r\n\r\n")
.await
.expect("fixture should write response headers");
for byte in b"RustFS!" {
stream.write_all(&[*byte]).await.expect("fixture should write body progress");
tokio::time::sleep(Duration::from_millis(20)).await;
}
});
let client = client_for_endpoint(
&endpoint,
TransitionClientTimeouts::new(Duration::from_millis(10), Duration::from_secs(1), Duration::from_millis(100)),
)
.await;
let mut opts = GetObjectOptions::default();
opts.set_range(0, 6).expect("the probe range should be valid");
let (_, _, mut reader) = client
.get_object_inner("bucket", "probe", &opts)
.await
.expect("continuous body progress must not be killed by the idle timer");
let mut body = Vec::new();
reader
.read_to_end(&mut body)
.await
.expect("bounded response should be readable");
assert_eq!(body, b"RustFS!");
fixture.await.expect("fixture should join");
}
} }
#[derive(Default)] #[derive(Default)]
+10 -82
View File
@@ -27,6 +27,7 @@ use crate::{
transition_api::{ReaderImpl, RequestMetadata, TransitionClient, collect_response_body}, transition_api::{ReaderImpl, RequestMetadata, TransitionClient, collect_response_body},
}; };
use http::{HeaderMap, StatusCode}; use http::{HeaderMap, StatusCode};
use http_body_util::BodyExt;
use hyper::body::Body; use hyper::body::Body;
use hyper::body::Bytes; use hyper::body::Bytes;
use rustfs_config::MAX_S3_CLIENT_RESPONSE_SIZE; use rustfs_config::MAX_S3_CLIENT_RESPONSE_SIZE;
@@ -123,9 +124,14 @@ impl TransitionClient {
} }
//let mut list_bucket_result = ListBucketV2Result::default(); //let mut list_bucket_result = ListBucketV2Result::default();
let body_vec = self let mut body_vec = Vec::new();
.collect_response_body(resp.into_body(), MAX_S3_CLIENT_RESPONSE_SIZE) let mut body = resp.into_body();
.await?; while let Some(frame) = body.frame().await {
let frame = frame.map_err(|e| std::io::Error::new(std::io::ErrorKind::Other, e.to_string()))?;
if let Some(data) = frame.data_ref() {
body_vec.extend_from_slice(data);
}
}
let mut list_bucket_result = match quick_xml::de::from_str::<ListBucketV2Result>(&String::from_utf8_lossy(&body_vec)) { let mut list_bucket_result = match quick_xml::de::from_str::<ListBucketV2Result>(&String::from_utf8_lossy(&body_vec)) {
Ok(result) => result, Ok(result) => result,
Err(err) => { Err(err) => {
@@ -208,9 +214,7 @@ impl TransitionClient {
let resp_status = resp.status(); let resp_status = resp.status();
let headers = resp.headers().clone(); let headers = resp.headers().clone();
let body = self let body = collect_response_body(resp.into_body(), MAX_S3_CLIENT_RESPONSE_SIZE).await?;
.collect_response_body(resp.into_body(), MAX_S3_CLIENT_RESPONSE_SIZE)
.await?;
if resp_status != StatusCode::OK { if resp_status != StatusCode::OK {
return Err(std::io::Error::other(http_resp_to_error_response( return Err(std::io::Error::other(http_resp_to_error_response(
resp_status, resp_status,
@@ -424,30 +428,6 @@ fn decode_s3_name(name: &str, encoding_type: &str) -> Result<String, std::io::Er
#[cfg(test)] #[cfg(test)]
mod tests { mod tests {
use super::*; use super::*;
use crate::{
credentials::{Credentials, SignatureType, Static, Value},
transition_api::{BucketLookupType, Options, TransitionClientTimeouts},
};
use std::time::Duration;
use tokio::{
io::{AsyncReadExt, AsyncWriteExt},
net::TcpListener,
};
fn timeout_test_options() -> Options {
Options {
creds: Credentials::new(Static(Value {
access_key_id: "access-key".to_string(),
secret_access_key: "secret-key".to_string(),
signer_type: SignatureType::SignatureV4,
..Default::default()
})),
region: "us-east-1".to_string(),
bucket_lookup: BucketLookupType::BucketLookupPath,
max_retries: 1,
..Default::default()
}
}
#[test] #[test]
fn list_versions_xml_preserves_versions_and_delete_markers() { fn list_versions_xml_preserves_versions_and_delete_markers() {
@@ -545,56 +525,4 @@ mod tests {
assert_eq!(parsed.common_prefixes.len(), 1); assert_eq!(parsed.common_prefixes.len(), 1);
assert_eq!(parsed.common_prefixes[0].prefix, "subdir/"); assert_eq!(parsed.common_prefixes[0].prefix, "subdir/");
} }
#[tokio::test]
async fn list_objects_v2_body_stall_returns_timed_out() {
let listener = match TcpListener::bind("127.0.0.1:0").await {
Ok(listener) => listener,
Err(err) if err.kind() == std::io::ErrorKind::PermissionDenied => return,
Err(err) => panic!("test listener should bind: {err}"),
};
let endpoint = listener
.local_addr()
.expect("listener local address should be available")
.to_string();
let fixture = tokio::spawn(async move {
let (mut stream, _) = listener.accept().await.expect("fixture should accept one list request");
let mut request = Vec::new();
let mut buffer = [0; 1024];
loop {
let read = stream.read(&mut buffer).await.expect("fixture should read request headers");
assert_ne!(read, 0, "connection closed before request headers were received");
request.extend_from_slice(&buffer[..read]);
if request.windows(4).any(|window| window == b"\r\n\r\n") {
break;
}
}
stream
.write_all(b"HTTP/1.1 200 OK\r\nContent-Length: 512\r\nConnection: close\r\n\r\n<ListBucketResult><Name>warm")
.await
.expect("fixture should write a partial list response");
tokio::time::sleep(Duration::from_millis(200)).await;
});
let client = TransitionClient::new_with_timeouts(
&endpoint,
timeout_test_options(),
"",
TransitionClientTimeouts::new(Duration::from_secs(1), Duration::from_secs(1), Duration::from_millis(50)),
)
.await
.expect("fixture client should build");
client
.bucket_loc_cache
.lock()
.expect("location cache should lock")
.set("bucket", "us-east-1");
let err = client
.list_objects_v2_query("bucket", "", "", false, false, "", "", 1, HeaderMap::new())
.await
.expect_err("a stalled ListObjectsV2 body must be bounded");
assert_eq!(err.kind(), std::io::ErrorKind::TimedOut);
fixture.await.expect("fixture should join");
}
} }
@@ -18,6 +18,7 @@
#![allow(clippy::all)] #![allow(clippy::all)]
use http::{HeaderMap, HeaderName, StatusCode}; use http::{HeaderMap, HeaderName, StatusCode};
use http_body_util::BodyExt;
use hyper::body::Bytes; use hyper::body::Bytes;
use s3s::S3ErrorCode; use s3s::S3ErrorCode;
use std::collections::HashMap; use std::collections::HashMap;
@@ -246,9 +247,14 @@ impl TransitionClient {
// Parse the CreateMultipartUpload response for the UploadId. Returning a // Parse the CreateMultipartUpload response for the UploadId. Returning a
// default (empty) result here made every multipart transition fail at the // default (empty) result here made every multipart transition fail at the
// first UploadPart with "UploadID cannot be empty" (rustfs/rustfs#4811). // first UploadPart with "UploadID cannot be empty" (rustfs/rustfs#4811).
let body_vec = self let mut body_vec = Vec::new();
.collect_response_body(resp.into_body(), rustfs_config::MAX_S3_CLIENT_RESPONSE_SIZE) let mut body = resp.into_body();
.await?; while let Some(frame) = body.frame().await {
let frame = frame.map_err(|e| std::io::Error::other(e.to_string()))?;
if let Some(data) = frame.data_ref() {
body_vec.extend_from_slice(data);
}
}
let initiate_multipart_upload_result = let initiate_multipart_upload_result =
quick_xml::de::from_str::<InitiateMultipartUploadResult>(&String::from_utf8_lossy(&body_vec)) quick_xml::de::from_str::<InitiateMultipartUploadResult>(&String::from_utf8_lossy(&body_vec))
.map_err(|e| std::io::Error::other(format!("failed to parse CreateMultipartUpload response: {e}")))?; .map_err(|e| std::io::Error::other(format!("failed to parse CreateMultipartUpload response: {e}")))?;
+9 -3
View File
@@ -19,6 +19,7 @@
#![allow(clippy::all)] #![allow(clippy::all)]
use http::{HeaderMap, HeaderValue, Method, StatusCode}; use http::{HeaderMap, HeaderValue, Method, StatusCode};
use http_body_util::BodyExt;
use hyper::body::Body; use hyper::body::Body;
use hyper::body::Bytes; use hyper::body::Bytes;
use rustfs_utils::HashAlgorithm; use rustfs_utils::HashAlgorithm;
@@ -350,9 +351,14 @@ impl TransitionClient {
) )
.await?; .await?;
let body_vec = self let mut body_vec = Vec::new();
.collect_response_body(resp.into_body(), rustfs_config::MAX_S3_CLIENT_RESPONSE_SIZE) let mut body = resp.into_body();
.await?; while let Some(frame) = body.frame().await {
let frame = frame.map_err(|e| std::io::Error::new(std::io::ErrorKind::Other, e.to_string()))?;
if let Some(data) = frame.data_ref() {
body_vec.extend_from_slice(data);
}
}
process_remove_multi_objects_response( process_remove_multi_objects_response(
ReaderImpl::Body(Bytes::from(body_vec)), ReaderImpl::Body(Bytes::from(body_vec)),
bucket_name, bucket_name,
+11 -72
View File
@@ -19,6 +19,7 @@
#![allow(clippy::all)] #![allow(clippy::all)]
use http::{HeaderMap, HeaderValue, StatusCode}; use http::{HeaderMap, HeaderValue, StatusCode};
use http_body_util::BodyExt;
use hyper::body::Body; use hyper::body::Body;
use hyper::body::Bytes; use hyper::body::Bytes;
use rustfs_utils::EMPTY_STRING_SHA256_HASH; use rustfs_utils::EMPTY_STRING_SHA256_HASH;
@@ -118,9 +119,14 @@ impl TransitionClient {
let resp_status = resp.status(); let resp_status = resp.status();
let h = resp.headers().clone(); let h = resp.headers().clone();
let body_vec = self let mut body_vec = Vec::new();
.collect_response_body(resp.into_body(), rustfs_config::MAX_S3_CLIENT_RESPONSE_SIZE) let mut body = resp.into_body();
.await?; while let Some(frame) = body.frame().await {
let frame = frame.map_err(|e| std::io::Error::new(std::io::ErrorKind::Other, e.to_string()))?;
if let Some(data) = frame.data_ref() {
body_vec.extend_from_slice(data);
}
}
let resperr = http_resp_to_error_response(resp_status, &h, body_vec, bucket_name, ""); let resperr = http_resp_to_error_response(resp_status, &h, body_vec, bucket_name, "");
warn!("bucket exists, resperr: {:?}", resperr); warn!("bucket exists, resperr: {:?}", resperr);
@@ -164,13 +170,11 @@ impl TransitionClient {
let resp_status = resp.status(); let resp_status = resp.status();
let h = resp.headers().clone(); let h = resp.headers().clone();
let body_vec = self let body_vec = collect_response_body(resp.into_body(), rustfs_config::MAX_S3_CLIENT_RESPONSE_SIZE).await?;
.collect_response_body(resp.into_body(), rustfs_config::MAX_S3_CLIENT_RESPONSE_SIZE)
.await?;
parse_bucket_versioning_response(resp_status, &h, body_vec, bucket_name) parse_bucket_versioning_response(resp_status, &h, body_vec, bucket_name)
} }
Err(err) => Err(err), Err(err) => Err(std::io::Error::other(err)),
} }
} }
@@ -270,14 +274,8 @@ impl TransitionClient {
#[cfg(test)] #[cfg(test)]
mod tests { mod tests {
use super::parse_bucket_versioning_response; use super::parse_bucket_versioning_response;
use crate::{
credentials::{Credentials, SignatureType, Static, Value},
transition_api::{BucketLookupType, Options, TransitionClient, TransitionClientTimeouts},
};
use http::{HeaderMap, StatusCode}; use http::{HeaderMap, StatusCode};
use s3s::dto::BucketVersioningStatus; use s3s::dto::BucketVersioningStatus;
use std::time::Duration;
use tokio::{io::AsyncReadExt, net::TcpListener};
#[test] #[test]
fn parses_bucket_versioning_statuses_mfa_delete_and_unversioned_state() { fn parses_bucket_versioning_statuses_mfa_delete_and_unversioned_state() {
@@ -340,63 +338,4 @@ mod tests {
assert_eq!(strict_err.kind(), std::io::ErrorKind::InvalidData); assert_eq!(strict_err.kind(), std::io::ErrorKind::InvalidData);
} }
} }
#[tokio::test]
async fn get_bucket_versioning_preserves_request_timeout_kind() {
let listener = match TcpListener::bind("127.0.0.1:0").await {
Ok(listener) => listener,
Err(err) if err.kind() == std::io::ErrorKind::PermissionDenied => return,
Err(err) => panic!("test listener should bind: {err}"),
};
let endpoint = listener
.local_addr()
.expect("listener local address should be available")
.to_string();
let fixture = tokio::spawn(async move {
let (mut stream, _) = listener.accept().await.expect("fixture should accept one versioning request");
let mut request = Vec::new();
let mut buffer = [0; 1024];
loop {
let read = stream.read(&mut buffer).await.expect("fixture should read request headers");
assert_ne!(read, 0, "connection closed before request headers were received");
request.extend_from_slice(&buffer[..read]);
if request.windows(4).any(|window| window == b"\r\n\r\n") {
break;
}
}
tokio::time::sleep(Duration::from_millis(200)).await;
});
let client = TransitionClient::new_with_timeouts(
&endpoint,
Options {
creds: Credentials::new(Static(Value {
access_key_id: "access-key".to_string(),
secret_access_key: "secret-key".to_string(),
signer_type: SignatureType::SignatureV4,
..Default::default()
})),
region: "us-east-1".to_string(),
bucket_lookup: BucketLookupType::BucketLookupPath,
max_retries: 1,
..Default::default()
},
"",
TransitionClientTimeouts::new(Duration::from_secs(1), Duration::from_millis(50), Duration::from_secs(1)),
)
.await
.expect("fixture client should build");
client
.bucket_loc_cache
.lock()
.expect("location cache should lock")
.set("bucket", "us-east-1");
let err = client
.get_bucket_versioning("bucket")
.await
.expect_err("a stalled versioning request must time out");
assert_eq!(err.kind(), std::io::ErrorKind::TimedOut);
fixture.await.expect("fixture should join");
}
} }
+10 -5
View File
@@ -26,6 +26,7 @@ use crate::{
transition_api::{CreateBucketConfiguration, LocationConstraint, TransitionClient}, transition_api::{CreateBucketConfiguration, LocationConstraint, TransitionClient},
}; };
use http::Request; use http::Request;
use http_body_util::BodyExt;
use hyper::StatusCode; use hyper::StatusCode;
use hyper::body::Body; use hyper::body::Body;
use hyper::body::Bytes; use hyper::body::Bytes;
@@ -85,7 +86,7 @@ impl TransitionClient {
let req = self.get_bucket_location_request(bucket_name)?; let req = self.get_bucket_location_request(bucket_name)?;
let mut resp = self.doit(req).await?; let mut resp = self.doit(req).await?;
location = process_bucket_location_response(self, resp, bucket_name, &self.tier_type).await?; location = process_bucket_location_response(resp, bucket_name, &self.tier_type).await?;
{ {
if let Ok(mut bucket_loc_cache) = self.bucket_loc_cache.lock() { if let Ok(mut bucket_loc_cache) = self.bucket_loc_cache.lock() {
bucket_loc_cache.set(bucket_name, &location); bucket_loc_cache.set(bucket_name, &location);
@@ -197,7 +198,6 @@ impl TransitionClient {
} }
async fn process_bucket_location_response( async fn process_bucket_location_response(
client: &TransitionClient,
mut resp: http::Response<Incoming>, mut resp: http::Response<Incoming>,
bucket_name: &str, bucket_name: &str,
tier_type: &str, tier_type: &str,
@@ -237,9 +237,14 @@ async fn process_bucket_location_response(
} }
//} //}
let body_vec = client let mut body_vec = Vec::new();
.collect_response_body(resp.into_body(), MAX_S3_CLIENT_RESPONSE_SIZE) let mut body = resp.into_body();
.await?; while let Some(frame) = body.frame().await {
let frame = frame.map_err(|e| std::io::Error::new(std::io::ErrorKind::Other, e.to_string()))?;
if let Some(data) = frame.data_ref() {
body_vec.extend_from_slice(data);
}
}
let mut location = "".to_string(); let mut location = "".to_string();
if tier_type == "huaweicloud" { if tier_type == "huaweicloud" {
if let Ok(body_str) = String::from_utf8(body_vec) { if let Ok(body_str) = String::from_utf8(body_vec) {
+41 -328
View File
@@ -41,7 +41,7 @@ use http::{
request::{Builder, Request}, request::{Builder, Request},
}; };
use http_body::Body; use http_body::Body;
use http_body_util::BodyExt; use http_body_util::{BodyExt, LengthLimitError, Limited};
use hyper::body::Bytes; use hyper::body::Bytes;
use hyper::body::Incoming; use hyper::body::Incoming;
use hyper_rustls::{ConfigBuilderExt, HttpsConnector}; use hyper_rustls::{ConfigBuilderExt, HttpsConnector};
@@ -67,12 +67,10 @@ use s3s::dto::Owner;
use s3s::dto::ReplicationStatus; use s3s::dto::ReplicationStatus;
use serde::{Deserialize, Serialize}; use serde::{Deserialize, Serialize};
use sha2::Sha256; use sha2::Sha256;
use std::error::Error as StdError;
use std::io::Cursor; use std::io::Cursor;
use std::pin::Pin; use std::pin::Pin;
use std::sync::atomic::{AtomicI32, Ordering}; use std::sync::atomic::{AtomicI32, Ordering};
use std::task::{Context, Poll}; use std::task::{Context, Poll};
use std::time::Duration as StdDuration;
use std::{ use std::{
collections::HashMap, collections::HashMap,
sync::{Arc, Mutex}, sync::{Arc, Mutex},
@@ -81,108 +79,28 @@ use time::Duration;
use time::OffsetDateTime; use time::OffsetDateTime;
use tokio::io::BufReader; use tokio::io::BufReader;
use tokio::io::{AsyncRead, AsyncReadExt}; use tokio::io::{AsyncRead, AsyncReadExt};
use tracing::{debug, error, trace, warn}; use tracing::{debug, error, warn};
use url::{Url, form_urlencoded}; use url::{Url, form_urlencoded};
use uuid::Uuid; use uuid::Uuid;
const C_USER_AGENT: &str = "RustFS (linux; x86)"; const C_USER_AGENT: &str = "RustFS (linux; x86)";
pub const MAX_S3_ERROR_RESPONSE_SIZE: usize = 64 * 1024; pub const MAX_S3_ERROR_RESPONSE_SIZE: usize = 64 * 1024;
const EVENT_TIER_REMOTE_TRANSPORT: &str = "tier_remote_transport";
const LOG_COMPONENT_S3_CLIENT: &str = "s3_client";
const LOG_SUBSYSTEM_TIER: &str = "tier";
const SUCCESS_STATUS: [StatusCode; 3] = [StatusCode::OK, StatusCode::NO_CONTENT, StatusCode::PARTIAL_CONTENT]; const SUCCESS_STATUS: [StatusCode; 3] = [StatusCode::OK, StatusCode::NO_CONTENT, StatusCode::PARTIAL_CONTENT];
fn response_body_exceeds_limit_error() -> std::io::Error {
std::io::Error::new(std::io::ErrorKind::InvalidData, "remote tier response body exceeds limit")
}
fn remote_tier_timeout_error(message: &'static str) -> std::io::Error {
std::io::Error::new(std::io::ErrorKind::TimedOut, message)
}
fn source_chain_has_io_kind(error: &(dyn StdError + 'static), kind: std::io::ErrorKind) -> bool {
let mut current = Some(error);
while let Some(error) = current {
if error
.downcast_ref::<std::io::Error>()
.is_some_and(|io_error| io_error.kind() == kind)
{
return true;
}
current = error.source();
}
false
}
fn transition_transport_error(err: hyper_util::client::legacy::Error) -> std::io::Error {
if source_chain_has_io_kind(&err, std::io::ErrorKind::TimedOut) {
return remote_tier_timeout_error("remote tier connection timed out");
}
std::io::Error::other(err)
}
async fn next_response_body_data<B>(
mut body: Pin<&mut B>,
idle_timeout: Option<StdDuration>,
) -> Result<Option<Bytes>, std::io::Error>
where
B: Body<Data = Bytes>,
B::Error: Into<Box<dyn StdError + Send + Sync>>,
{
let next_nonempty_data = async {
loop {
let Some(frame) = std::future::poll_fn(|cx| body.as_mut().poll_frame(cx)).await else {
return Ok(None);
};
let frame = frame.map_err(std::io::Error::other)?;
let Ok(data) = frame.into_data() else {
continue;
};
if !data.is_empty() {
return Ok(Some(data));
}
}
};
if let Some(idle_timeout) = idle_timeout {
tokio::time::timeout(idle_timeout, next_nonempty_data)
.await
.map_err(|_| remote_tier_timeout_error("remote tier response body stalled"))?
} else {
next_nonempty_data.await
}
}
async fn collect_response_body_inner<B>(
body: B,
limit: Option<usize>,
idle_timeout: Option<StdDuration>,
) -> Result<Vec<u8>, std::io::Error>
where
B: Body<Data = Bytes>,
B::Error: Into<Box<dyn StdError + Send + Sync>>,
{
let mut body_vec = Vec::new();
let mut body = std::pin::pin!(body);
while let Some(data) = next_response_body_data(body.as_mut(), idle_timeout).await? {
let Some(new_len) = body_vec.len().checked_add(data.len()) else {
return Err(response_body_exceeds_limit_error());
};
if limit.is_some_and(|limit| new_len > limit) {
return Err(response_body_exceeds_limit_error());
}
body_vec.extend_from_slice(&data);
}
Ok(body_vec)
}
pub async fn collect_response_body<B>(body: B, limit: usize) -> Result<Vec<u8>, std::io::Error> pub async fn collect_response_body<B>(body: B, limit: usize) -> Result<Vec<u8>, std::io::Error>
where where
B: Body<Data = Bytes>, B: Body<Data = Bytes>,
B::Error: Into<Box<dyn StdError + Send + Sync>>, B::Error: Into<Box<dyn std::error::Error + Send + Sync>>,
{ {
collect_response_body_inner(body, Some(limit), None).await let body = Limited::new(body, limit).collect().await.map_err(|err| {
if err.is::<LengthLimitError>() {
std::io::Error::new(std::io::ErrorKind::InvalidData, "remote tier response body exceeds limit")
} else {
std::io::Error::other(err)
}
})?;
Ok(body.to_bytes().to_vec())
} }
const C_UNKNOWN: i32 = -1; const C_UNKNOWN: i32 = -1;
@@ -278,62 +196,6 @@ pub struct TransitionClient {
pub trailing_header_support: bool, pub trailing_header_support: bool,
pub max_retries: i64, pub max_retries: i64,
pub tier_type: String, pub tier_type: String,
pub timeouts: TransitionClientTimeouts,
}
#[derive(Debug, Clone, Copy, PartialEq, Eq)]
pub struct TransitionClientTimeouts {
pub connect_timeout: StdDuration,
pub request_timeout: StdDuration,
pub response_body_idle_timeout: StdDuration,
}
impl TransitionClientTimeouts {
pub const fn new(
connect_timeout: StdDuration,
request_timeout: StdDuration,
response_body_idle_timeout: StdDuration,
) -> Self {
Self {
connect_timeout,
request_timeout,
response_body_idle_timeout,
}
}
fn validate(self) -> Result<Self, std::io::Error> {
if self.connect_timeout.is_zero() {
return Err(std::io::Error::new(
std::io::ErrorKind::InvalidInput,
"remote tier connect timeout must be greater than zero",
));
}
if self.request_timeout.is_zero() {
return Err(std::io::Error::new(
std::io::ErrorKind::InvalidInput,
"remote tier request timeout must be greater than zero",
));
}
if self.response_body_idle_timeout.is_zero() {
return Err(std::io::Error::new(
std::io::ErrorKind::InvalidInput,
"remote tier response body idle timeout must be greater than zero",
));
}
Ok(self)
}
}
impl Default for TransitionClientTimeouts {
fn default() -> Self {
Self {
connect_timeout: StdDuration::from_secs(rustfs_config::DEFAULT_TIER_REMOTE_CONNECT_TIMEOUT_SECS),
request_timeout: StdDuration::from_secs(rustfs_config::DEFAULT_TIER_REMOTE_REQUEST_TIMEOUT_SECS),
response_body_idle_timeout: StdDuration::from_secs(
rustfs_config::DEFAULT_TIER_REMOTE_RESPONSE_BODY_IDLE_TIMEOUT_SECS,
),
}
}
} }
#[derive(Debug, Default)] #[derive(Debug, Default)]
@@ -426,28 +288,12 @@ async fn build_tls_config() -> Result<rustls::ClientConfig, std::io::Error> {
impl TransitionClient { impl TransitionClient {
pub async fn new(endpoint: &str, opts: Options, tier_type: &str) -> Result<TransitionClient, std::io::Error> { pub async fn new(endpoint: &str, opts: Options, tier_type: &str) -> Result<TransitionClient, std::io::Error> {
Self::private_new(endpoint, opts, tier_type, TransitionClientTimeouts::default()).await let client = Self::private_new(endpoint, opts, tier_type).await?;
Ok(client)
} }
/// Builds a transition client with explicit transport timeout budgets. async fn private_new(endpoint: &str, opts: Options, tier_type: &str) -> Result<TransitionClient, std::io::Error> {
///
/// [`Self::new`] keeps the historical constructor surface and uses the
/// production defaults from [`TransitionClientTimeouts::default`].
pub async fn new_with_timeouts(
endpoint: &str,
opts: Options,
tier_type: &str,
timeouts: TransitionClientTimeouts,
) -> Result<TransitionClient, std::io::Error> {
Self::private_new(endpoint, opts, tier_type, timeouts).await
}
async fn private_new(
endpoint: &str,
opts: Options,
tier_type: &str,
timeouts: TransitionClientTimeouts,
) -> Result<TransitionClient, std::io::Error> {
if rustls::crypto::CryptoProvider::get_default().is_none() { if rustls::crypto::CryptoProvider::get_default().is_none() {
// No default provider is set yet; try to install aws-lc-rs. // No default provider is set yet; try to install aws-lc-rs.
// `install_default` can only fail if another thread races us and installs a provider // `install_default` can only fail if another thread races us and installs a provider
@@ -460,19 +306,15 @@ impl TransitionClient {
} }
let endpoint_url = get_endpoint_url(endpoint, opts.secure)?; let endpoint_url = get_endpoint_url(endpoint, opts.secure)?;
let timeouts = timeouts.validate()?;
let tls = build_tls_config().await?; let tls = build_tls_config().await?;
let mut http = HttpConnector::new();
http.enforce_http(false);
http.set_connect_timeout(Some(timeouts.connect_timeout));
let https = hyper_rustls::HttpsConnectorBuilder::new() let https = hyper_rustls::HttpsConnectorBuilder::new()
.with_tls_config(tls) .with_tls_config(tls)
.https_or_http() .https_or_http()
.enable_http1() .enable_http1()
.enable_http2() .enable_http2()
.wrap_connector(http); .build();
let http_client = Client::builder(TokioExecutor::new()).build(https); let http_client = Client::builder(TokioExecutor::new()).build(https);
let mut client = TransitionClient { let mut client = TransitionClient {
@@ -495,7 +337,6 @@ impl TransitionClient {
trailing_header_support: opts.trailing_headers, trailing_header_support: opts.trailing_headers,
max_retries: opts.max_retries, max_retries: opts.max_retries,
tier_type: tier_type.to_string(), tier_type: tier_type.to_string(),
timeouts,
}; };
{ {
@@ -660,43 +501,29 @@ impl TransitionClient {
} }
pub async fn doit(&self, req: Request<s3s::Body>) -> Result<Response<Incoming>, std::io::Error> { pub async fn doit(&self, req: Request<s3s::Body>) -> Result<Response<Incoming>, std::io::Error> {
let req_method;
let req_uri;
let resp;
let http_client = self.http_client.clone(); let http_client = self.http_client.clone();
let req_method = req.method().clone(); {
let resp = tokio::time::timeout(self.timeouts.request_timeout, http_client.request(req)).await; req_method = req.method().clone();
req_uri = req.uri().clone();
debug!("endpoint_url: {}", self.endpoint_url.as_str().to_string());
resp = http_client.request(req);
}
let resp = resp.await;
debug!("http_client url: {} {}", req_method, req_uri);
if let Err(err) = resp {
error!("http_client call error: {:?}", err);
return Err(std::io::Error::other(err));
}
let resp = match resp { let resp = match resp {
Ok(Ok(resp)) => resp, Ok(r) => r,
Ok(Err(err)) => { Err(_) => return Err(std::io::Error::other("Unexpected error in response")),
let err = transition_transport_error(err);
error!(
event = EVENT_TIER_REMOTE_TRANSPORT,
component = LOG_COMPONENT_S3_CLIENT,
subsystem = LOG_SUBSYSTEM_TIER,
method = %req_method,
error_kind = ?err.kind(),
"remote tier request failed"
);
return Err(err);
}
Err(_) => {
warn!(
event = EVENT_TIER_REMOTE_TRANSPORT,
component = LOG_COMPONENT_S3_CLIENT,
subsystem = LOG_SUBSYSTEM_TIER,
method = %req_method,
timeout_ms = self.timeouts.request_timeout.as_millis(),
"remote tier request timed out before response headers"
);
return Err(remote_tier_timeout_error("remote tier request timed out before response headers"));
}
}; };
trace!( debug!(status = %resp.status(), "remote tier response received");
event = EVENT_TIER_REMOTE_TRANSPORT,
component = LOG_COMPONENT_S3_CLIENT,
subsystem = LOG_SUBSYSTEM_TIER,
method = %req_method,
status = %resp.status(),
"remote tier response received"
);
//let b = resp.body_mut().store_all_unlimited().await.unwrap().to_vec(); //let b = resp.body_mut().store_all_unlimited().await.unwrap().to_vec();
//debug!("http_resp_body: {}", String::from_utf8(b).unwrap()); //debug!("http_resp_body: {}", String::from_utf8(b).unwrap());
@@ -710,15 +537,7 @@ impl TransitionClient {
.and_then(|value| value.to_str().ok()) .and_then(|value| value.to_str().ok())
.unwrap_or_default() .unwrap_or_default()
.to_string(); .to_string();
warn!( warn!(status = %status, request_id, "remote tier request rejected");
event = EVENT_TIER_REMOTE_TRANSPORT,
component = LOG_COMPONENT_S3_CLIENT,
subsystem = LOG_SUBSYSTEM_TIER,
method = %req_method,
status = %status,
request_id,
"remote tier request rejected"
);
} }
Ok(resp) Ok(resp)
} }
@@ -762,9 +581,7 @@ impl TransitionClient {
let resp_status = resp.status(); let resp_status = resp.status();
let h = resp.headers().clone(); let h = resp.headers().clone();
let body_vec = self let body_vec = collect_response_body(resp.into_body(), MAX_S3_ERROR_RESPONSE_SIZE).await?;
.collect_response_body(resp.into_body(), MAX_S3_ERROR_RESPONSE_SIZE)
.await?;
let parsed_error = let parsed_error =
http_resp_to_error_response(resp_status, &h, body_vec, &metadata.bucket_name, &metadata.object_name); http_resp_to_error_response(resp_status, &h, body_vec, &metadata.bucket_name, &metadata.object_name);
let routing_region = parsed_error.region; let routing_region = parsed_error.region;
@@ -818,22 +635,6 @@ impl TransitionClient {
Err(std::io::Error::other("remote tier request did not produce a response")) Err(std::io::Error::other("remote tier request did not produce a response"))
} }
pub async fn collect_response_body<B>(&self, body: B, limit: usize) -> Result<Vec<u8>, std::io::Error>
where
B: Body<Data = Bytes>,
B::Error: Into<Box<dyn StdError + Send + Sync>>,
{
collect_response_body_inner(body, Some(limit), Some(self.timeouts.response_body_idle_timeout)).await
}
pub async fn collect_response_body_unbounded<B>(&self, body: B) -> Result<Vec<u8>, std::io::Error>
where
B: Body<Data = Bytes>,
B::Error: Into<Box<dyn StdError + Send + Sync>>,
{
collect_response_body_inner(body, None, Some(self.timeouts.response_body_idle_timeout)).await
}
async fn new_request( async fn new_request(
&self, &self,
method: &http::Method, method: &http::Method,
@@ -1703,17 +1504,12 @@ pub struct CreateBucketConfiguration {
mod tests { mod tests {
use super::{ use super::{
MAX_S3_CLIENT_RESPONSE_SIZE, MAX_S3_ERROR_RESPONSE_SIZE, SignatureType, build_tls_config, collect_response_body, MAX_S3_CLIENT_RESPONSE_SIZE, MAX_S3_ERROR_RESPONSE_SIZE, SignatureType, build_tls_config, collect_response_body,
collect_response_body_inner, signer_error_to_io_error, to_object_info_for_provider, validate_header_values, signer_error_to_io_error, to_object_info_for_provider, validate_header_values, with_rustls_init_guard,
with_rustls_init_guard,
}; };
use crate::provider_versions::{BucketVersioningState, ProviderVersionCapabilities, RemoteVersion}; use crate::provider_versions::{BucketVersioningState, ProviderVersionCapabilities, RemoteVersion};
use futures::stream; use http::{HeaderMap, HeaderValue};
use http::{HeaderMap, HeaderValue, Request}; use http_body_util::Full;
use http_body::Frame;
use http_body_util::{Full, StreamBody};
use hyper::body::Bytes; use hyper::body::Bytes;
use std::time::Duration as StdDuration;
use tokio::net::TcpListener;
use uuid::Uuid; use uuid::Uuid;
#[tokio::test] #[tokio::test]
@@ -1744,77 +1540,6 @@ mod tests {
assert_eq!(err.kind(), std::io::ErrorKind::InvalidData); assert_eq!(err.kind(), std::io::ErrorKind::InvalidData);
} }
#[tokio::test]
async fn empty_data_frames_do_not_reset_the_body_idle_timeout() {
let frames = stream::unfold((), |_| async {
tokio::time::sleep(StdDuration::from_millis(10)).await;
Some((Ok::<_, std::io::Error>(Frame::data(Bytes::new())), ()))
});
let body = StreamBody::new(Box::pin(frames));
let err = tokio::time::timeout(
StdDuration::from_millis(200),
collect_response_body_inner(body, Some(1), Some(StdDuration::from_millis(50))),
)
.await
.expect("the collector should enforce its own body idle timeout")
.expect_err("empty frames must not count as body progress");
assert_eq!(err.kind(), std::io::ErrorKind::TimedOut);
}
#[tokio::test]
async fn public_body_collector_accepts_non_unpin_bodies() {
let body = StreamBody::new(stream::once(async { Ok::<_, std::io::Error>(Frame::data(Bytes::from_static(b"ok"))) }));
let collected = collect_response_body(body, 2)
.await
.expect("the public collector should pin non-Unpin bodies internally");
assert_eq!(collected, b"ok");
}
#[tokio::test]
async fn https_endpoints_reach_the_transport_connector() {
let listener = match TcpListener::bind("127.0.0.1:0").await {
Ok(listener) => listener,
Err(err) if err.kind() == std::io::ErrorKind::PermissionDenied => return,
Err(err) => panic!("test listener should bind: {err}"),
};
let endpoint = listener
.local_addr()
.expect("listener local address should be available")
.to_string();
let accepted = tokio::spawn(async move {
let (stream, _) = tokio::time::timeout(StdDuration::from_secs(1), listener.accept())
.await
.expect("HTTPS connector should reach the TCP listener")
.expect("fixture should accept the HTTPS connection");
drop(stream);
});
let client = super::TransitionClient::new_with_timeouts(
&endpoint,
super::Options {
secure: true,
..Default::default()
},
"",
super::TransitionClientTimeouts::new(StdDuration::from_secs(1), StdDuration::from_secs(1), StdDuration::from_secs(1)),
)
.await
.expect("fixture client should build");
let request = Request::builder()
.uri(format!("https://{endpoint}/"))
.body(s3s::Body::empty())
.expect("fixture request should build");
client
.doit(request)
.await
.expect_err("the fixture closes before completing the TLS handshake");
accepted.await.expect("fixture should join");
}
#[test] #[test]
fn rustls_guard_converts_panics_to_io_errors() { fn rustls_guard_converts_panics_to_io_errors() {
let err = with_rustls_init_guard(|| -> Result<(), std::io::Error> { panic!("missing provider") }) let err = with_rustls_init_guard(|| -> Result<(), std::io::Error> { panic!("missing provider") })
@@ -1848,18 +1573,6 @@ mod tests {
assert!(outcome.is_ok(), "provider install guard must not panic when a provider is already set"); assert!(outcome.is_ok(), "provider install guard must not panic when a provider is already set");
} }
#[test]
fn transition_timeouts_reject_zero_budgets() {
for timeouts in [
super::TransitionClientTimeouts::new(StdDuration::ZERO, StdDuration::from_secs(1), StdDuration::from_secs(1)),
super::TransitionClientTimeouts::new(StdDuration::from_secs(1), StdDuration::ZERO, StdDuration::from_secs(1)),
super::TransitionClientTimeouts::new(StdDuration::from_secs(1), StdDuration::from_secs(1), StdDuration::ZERO),
] {
let err = timeouts.validate().expect_err("zero timeout budgets must fail closed");
assert_eq!(err.kind(), std::io::ErrorKind::InvalidInput);
}
}
#[test] #[test]
fn validate_header_values_returns_header_name_for_non_utf8_values() { fn validate_header_values_returns_header_name_for_non_utf8_values() {
let mut headers = HeaderMap::new(); let mut headers = HeaderMap::new();
+2 -10
View File
@@ -196,7 +196,7 @@ pub(crate) async fn read_config_revision<S: ScannerObjectIO>(store: Arc<S>, path
} }
} }
#[derive(Clone, Debug, PartialEq, Eq)] #[derive(Clone, Debug)]
pub(crate) struct DataUsageCacheRevisions { pub(crate) struct DataUsageCacheRevisions {
main: DataUsageCacheRevision, main: DataUsageCacheRevision,
backup: Option<DataUsageCacheRevision>, backup: Option<DataUsageCacheRevision>,
@@ -503,10 +503,6 @@ pub struct DataUsageCacheInfo {
pub lkg_leader_epoch: Option<u64>, pub lkg_leader_epoch: Option<u64>,
#[serde(default)] #[serde(default)]
pub lkg_scan_plan_digest: Option<DataUsageScanPlanDigest>, pub lkg_scan_plan_digest: Option<DataUsageScanPlanDigest>,
/// Activity-sensitive identity for same-cycle set snapshot reuse. The
/// structural plan remains reusable across ordinary bucket writes.
#[serde(default)]
pub scan_execution_digest: Option<DataUsageScanPlanDigest>,
} }
impl Serialize for DataUsageCacheInfo { impl Serialize for DataUsageCacheInfo {
@@ -523,8 +519,7 @@ impl Serialize for DataUsageCacheInfo {
+ usize::from(self.lkg_next_cycle.is_some()) + usize::from(self.lkg_next_cycle.is_some())
+ usize::from(self.lkg_last_update.is_some()) + usize::from(self.lkg_last_update.is_some())
+ usize::from(self.lkg_leader_epoch.is_some()) + usize::from(self.lkg_leader_epoch.is_some())
+ usize::from(self.lkg_scan_plan_digest.is_some()) + usize::from(self.lkg_scan_plan_digest.is_some());
+ usize::from(self.scan_execution_digest.is_some());
let mut state = serializer.serialize_map(Some(field_count))?; let mut state = serializer.serialize_map(Some(field_count))?;
state.serialize_entry("name", &self.name)?; state.serialize_entry("name", &self.name)?;
state.serialize_entry("next_cycle", &self.next_cycle)?; state.serialize_entry("next_cycle", &self.next_cycle)?;
@@ -563,9 +558,6 @@ impl Serialize for DataUsageCacheInfo {
if let Some(scan_plan_digest) = self.lkg_scan_plan_digest { if let Some(scan_plan_digest) = self.lkg_scan_plan_digest {
state.serialize_entry("lkg_scan_plan_digest", &scan_plan_digest)?; state.serialize_entry("lkg_scan_plan_digest", &scan_plan_digest)?;
} }
if let Some(scan_execution_digest) = self.scan_execution_digest {
state.serialize_entry("scan_execution_digest", &scan_execution_digest)?;
}
state.end() state.end()
} }
} }
@@ -1067,7 +1067,6 @@ fn test_data_usage_cache_info_deserialize_defaults_scan_resume_after() {
assert!(decoded.source.is_none()); assert!(decoded.source.is_none());
assert!(!decoded.snapshot_complete); assert!(!decoded.snapshot_complete);
assert!(decoded.scan_plan_digest.is_none()); assert!(decoded.scan_plan_digest.is_none());
assert!(decoded.scan_execution_digest.is_none());
assert_eq!(decoded.cache_key_format, 0); assert_eq!(decoded.cache_key_format, 0);
} }
@@ -1110,7 +1109,6 @@ fn test_data_usage_cache_info_unmarshal_old_msgpack_defaults_scan_resume_after()
assert!(decoded.source.is_none()); assert!(decoded.source.is_none());
assert!(!decoded.snapshot_complete); assert!(!decoded.snapshot_complete);
assert!(decoded.scan_plan_digest.is_none()); assert!(decoded.scan_plan_digest.is_none());
assert!(decoded.scan_execution_digest.is_none());
assert_eq!(decoded.cache_key_format, 0); assert_eq!(decoded.cache_key_format, 0);
} }
@@ -1147,7 +1145,6 @@ fn test_new_data_usage_cache_msgpack_round_trips_and_supports_old_reader() {
source: Some(DataUsageCacheSource::new(1, 2)), source: Some(DataUsageCacheSource::new(1, 2)),
snapshot_complete: true, snapshot_complete: true,
scan_plan_digest: Some(TEST_PLAN_DIGEST), scan_plan_digest: Some(TEST_PLAN_DIGEST),
scan_execution_digest: Some(DataUsageScanPlanDigest([42; 32])),
cache_key_format: DATA_USAGE_CACHE_KEY_FORMAT, cache_key_format: DATA_USAGE_CACHE_KEY_FORMAT,
..Default::default() ..Default::default()
}, },
@@ -1167,7 +1164,6 @@ fn test_new_data_usage_cache_msgpack_round_trips_and_supports_old_reader() {
assert_eq!(current.info.source, Some(DataUsageCacheSource::new(1, 2))); assert_eq!(current.info.source, Some(DataUsageCacheSource::new(1, 2)));
assert!(current.info.snapshot_complete); assert!(current.info.snapshot_complete);
assert_eq!(current.info.scan_plan_digest, Some(TEST_PLAN_DIGEST)); assert_eq!(current.info.scan_plan_digest, Some(TEST_PLAN_DIGEST));
assert_eq!(current.info.scan_execution_digest, Some(DataUsageScanPlanDigest([42; 32])));
assert_eq!(current.info.cache_key_format, DATA_USAGE_CACHE_KEY_FORMAT); assert_eq!(current.info.cache_key_format, DATA_USAGE_CACHE_KEY_FORMAT);
assert_eq!(current.find("bucket").map(|entry| entry.objects), Some(3)); assert_eq!(current.find("bucket").map(|entry| entry.objects), Some(3));
+5 -31
View File
@@ -1616,7 +1616,7 @@ where
// Refresh the storage-owned movement snapshot before reading background // Refresh the storage-owned movement snapshot before reading background
// heal state. A missing heal object yields an in-memory default; do not // heal state. A missing heal object yields an in-memory default; do not
// let that default influence a cycle while publication is blocked. // let that default influence a cycle while publication is blocked.
if storeapi.scanner_data_movement_pause_status().await.paused { if storeapi.scanner_data_usage_publication_blocked().await {
mark_scan_cycle_idle(cycle_info, &mut cycle_metrics_guard).await; mark_scan_cycle_idle(cycle_info, &mut cycle_metrics_guard).await;
return ScannerCycleOutcome::Deferred(ScannerCycleDeferReason::DataMovement); return ScannerCycleOutcome::Deferred(ScannerCycleDeferReason::DataMovement);
} }
@@ -1816,19 +1816,6 @@ where
let publication_defer_reason = publication_defer_reason let publication_defer_reason = publication_defer_reason
.or(remote_lease_defer_reason) .or(remote_lease_defer_reason)
.or(remote_lease_fence_defer_reason); .or(remote_lease_fence_defer_reason);
// A PUT tail can finish between the walk and lease acquisition without
// changing the movement epoch accepted by those leases. Re-prove the
// namespace baseline only after every peer has granted publication.
let post_lease_activity_defer_reason = if publication_defer_reason.is_none()
&& remote_publication_leases.is_some()
&& let Ok(result) = &scan_result
&& result.status == ScannerCycleStatus::Complete
{
scanner_post_lease_activity_defer_reason(result.activity_digest(), probe_scanner_activity(storeapi.as_ref(), true).await)
} else {
None
};
let publication_defer_reason = publication_defer_reason.or(post_lease_activity_defer_reason);
// Include reasons discovered while acquiring or validating remote leases. // Include reasons discovered while acquiring or validating remote leases.
let publication_deferred = publication_defer_reason.is_some(); let publication_deferred = publication_defer_reason.is_some();
let budget_elapsed = cycle_budget.budget_elapsed() && !ctx.is_cancelled(); let budget_elapsed = cycle_budget.budget_elapsed() && !ctx.is_cancelled();
@@ -3253,21 +3240,6 @@ where
} }
} }
fn scanner_post_lease_activity_defer_reason(
expected_digest: Option<[u8; 32]>,
activity: Result<ScannerActivitySnapshot, String>,
) -> Option<ScannerCycleDeferReason> {
match activity {
Ok(snapshot)
if scanner_activity_allows_usage_publication(&snapshot)
&& expected_digest == Some(scanner_activity_snapshot_digest(&snapshot)) =>
{
None
}
Ok(_) | Err(_) => Some(ScannerCycleDeferReason::ActivityBaselineUnavailable),
}
}
#[derive(Clone, Copy, Debug, PartialEq, Eq)] #[derive(Clone, Copy, Debug, PartialEq, Eq)]
enum ScannerCyclePreCommitOutcome { enum ScannerCyclePreCommitOutcome {
RecoverCacheCycle(u64), RecoverCacheCycle(u64),
@@ -3456,11 +3428,13 @@ use cycle_state::*;
use leadership::*; use leadership::*;
use usage_store::*; use usage_store::*;
#[cfg(test)]
pub(crate) use activity::scanner_activity_snapshot_digest;
pub use activity::scanner_topology_digest; pub use activity::scanner_topology_digest;
pub(crate) use activity::{ pub(crate) use activity::{
ScannerActivitySnapshot, ScannerDirtyUsageAcknowledgement, probe_scanner_activity, scanner_activity_allows_usage_publication, ScannerActivitySnapshot, ScannerDirtyUsageAcknowledgement, probe_scanner_activity, scanner_activity_allows_usage_publication,
scanner_activity_dirty_usage_state_for_host, scanner_activity_publication_lease_targets, scanner_activity_snapshot_digest, scanner_activity_dirty_usage_state_for_host, scanner_activity_publication_lease_targets, scanner_activity_structural_digest,
scanner_activity_structural_digest, scanner_dirty_usage_acknowledgements, scanner_dirty_usage_acknowledgements,
}; };
pub(crate) use activity::{ScannerCycleOutcome, scanner_cycle_outcome_with_pending_maintenance}; pub(crate) use activity::{ScannerCycleOutcome, scanner_cycle_outcome_with_pending_maintenance};
pub use backlog::{ pub use backlog::{
+1
View File
@@ -902,6 +902,7 @@ where
observation observation
} }
#[cfg(test)]
pub(crate) fn scanner_activity_snapshot_digest(snapshot: &ScannerActivitySnapshot) -> [u8; 32] { pub(crate) fn scanner_activity_snapshot_digest(snapshot: &ScannerActivitySnapshot) -> [u8; 32] {
let mut hasher = Sha256::new(); let mut hasher = Sha256::new();
hasher.update(u64::try_from(snapshot.len()).unwrap_or(u64::MAX).to_be_bytes()); hasher.update(u64::try_from(snapshot.len()).unwrap_or(u64::MAX).to_be_bytes());
+1 -172
View File
@@ -15,8 +15,7 @@
use super::heal_info::{classify_background_heal_read_error, decode_background_heal_info}; use super::heal_info::{classify_background_heal_read_error, decode_background_heal_info};
use super::*; use super::*;
use crate::EcstoreResult; use crate::EcstoreResult;
use crate::storage_api::owner::ecstore_hold_namespace_commit; use crate::storage_api::scan::BucketOperations as _;
use crate::storage_api::scan::{BucketOperations as _, ObjectIO as _};
use crate::{ use crate::{
DATA_USAGE_BLOOM_RECOVERY_PATH, DATA_USAGE_CACHE_KEY_FORMAT, DATA_USAGE_CACHE_NAME, DATA_USAGE_ROOT, DATA_USAGE_BLOOM_RECOVERY_PATH, DATA_USAGE_CACHE_KEY_FORMAT, DATA_USAGE_CACHE_NAME, DATA_USAGE_ROOT,
DataUsageCachePrepareOutcome, DataUsageCacheSource, DataUsageEntry, DataUsageScanPlanDigest, Endpoint, EndpointServerPools, DataUsageCachePrepareOutcome, DataUsageCacheSource, DataUsageEntry, DataUsageScanPlanDigest, Endpoint, EndpointServerPools,
@@ -1166,116 +1165,6 @@ async fn run_data_scanner_cycle_publishes_activity_for_owner_lifetime() {
global_metrics().set_cycle(None).await; global_metrics().set_cycle(None).await;
} }
#[tokio::test]
#[serial]
async fn coordinator_walks_during_pending_put_without_persisting_or_acknowledging_usage() {
crate::scanner_io::clear_dirty_usage_buckets_for_tests();
let (_temp_dir, store) = setup_scanner_cycle_store().await;
let bucket = format!("scanner-coordinator-pending-{}", Uuid::new_v4().simple());
store
.make_bucket(&bucket, &crate::storage_api::scan::MakeBucketOptions::default())
.await
.expect("fixture bucket should be created");
let mut reader = PutObjReader::from_vec(b"first".to_vec());
store.pools[0].disk_set[0]
.put_object(
&bucket,
"object",
&mut reader,
&ObjectOptions {
no_lock: true,
..Default::default()
},
)
.await
.expect("fixture object should finish its rename fanout");
crate::scanner_io::record_dirty_usage_bucket(&bucket);
let dirty_before = crate::scanner_io::dirty_usage_buckets_for_tests();
let baseline = read_config(store.clone(), DATA_USAGE_OBJ_NAME_PATH.as_str())
.await
.expect("fixture usage baseline should be readable");
let pending = ecstore_hold_namespace_commit(store.as_ref());
let ctx = CancellationToken::new();
let budget = ScannerCycleBudget::new_with_progress_tracking(&ctx, ScannerCycleBudgetConfig::default());
let mut cycle_info = CurrentCycle {
next: 1,
..Default::default()
};
let mut revision = DataUsageCacheRevision::Missing;
let outcome = tokio::time::timeout(
Duration::from_secs(30),
run_data_scanner_cycle_with_budget(&ctx, &store, &mut cycle_info, &mut revision, 1, Arc::clone(&budget)),
)
.await
.expect("the coordinator must finish its namespace walk while a PUT is pending");
assert_eq!(budget.progress().0, 1, "the coordinator must reach actual object traversal");
assert_eq!(outcome, ScannerCycleOutcome::Deferred(ScannerCycleDeferReason::DataMovement));
assert_eq!(cycle_info.next, 1, "a rejected publication must not advance the cycle");
assert_eq!(revision, DataUsageCacheRevision::Missing);
assert_eq!(crate::scanner_io::dirty_usage_buckets_for_tests(), dirty_before);
assert_eq!(
read_config(store.clone(), DATA_USAGE_OBJ_NAME_PATH.as_str())
.await
.expect("the prior authoritative usage must remain readable"),
baseline,
"the pending candidate must not replace the authoritative baseline"
);
let committed_body = b"committed-after-walk";
let mut reader = PutObjReader::from_vec(committed_body.to_vec());
store.pools[0].disk_set[0]
.put_object(
&bucket,
"object",
&mut reader,
&ObjectOptions {
no_lock: true,
..Default::default()
},
)
.await
.expect("the pending tail must change the physical object before it drains");
assert_eq!(crate::scanner_io::dirty_usage_buckets_for_tests(), dirty_before);
drop(pending);
let retry_budget = ScannerCycleBudget::new_with_progress_tracking(&ctx, ScannerCycleBudgetConfig::default());
let outcome = tokio::time::timeout(
Duration::from_secs(30),
run_data_scanner_cycle_with_budget(&ctx, &store, &mut cycle_info, &mut revision, 1, Arc::clone(&retry_budget)),
)
.await
.expect("the same cycle must converge after the pending PUT drains");
assert_eq!(
retry_budget.progress().0,
1,
"the same-cycle retry must not reuse the pre-tail bucket cache"
);
assert!(matches!(
outcome,
ScannerCycleOutcome::Completed | ScannerCycleOutcome::CompletedWithPendingMaintenance
));
assert_eq!(cycle_info.next, 2);
assert!(!crate::scanner_io::dirty_usage_buckets_for_tests().contains_key(&bucket));
let usage = read_config(store.clone(), DATA_USAGE_OBJ_NAME_PATH.as_str())
.await
.expect("the converged usage should be persisted");
let usage: DataUsageInfo = serde_json::from_slice(&usage).expect("the persisted usage should decode");
assert_eq!(usage.usage_snapshot_converged, Some(true));
assert_eq!(usage.scanner_cycle, Some(1));
assert_eq!(usage.objects_total_count, 1);
assert_eq!(
usage.objects_total_size,
u64::try_from(committed_body.len()).expect("fixture body length")
);
let bucket_usage = usage
.buckets_usage
.get(&bucket)
.expect("the scanned bucket should be published");
assert_eq!(bucket_usage.objects_count, 1);
assert_eq!(bucket_usage.size, u64::try_from(committed_body.len()).expect("fixture body length"));
global_metrics().set_cycle(None).await;
crate::scanner_io::clear_dirty_usage_buckets_for_tests();
}
#[tokio::test] #[tokio::test]
#[serial] #[serial]
async fn test_finalize_partial_scan_cycle_advances_and_persists_counter() { async fn test_finalize_partial_scan_cycle_advances_and_persists_counter() {
@@ -8596,66 +8485,6 @@ fn scanner_node_activity(epoch: &str, namespace_generation: u64, maintenance_gen
} }
} }
#[test]
fn post_lease_activity_proof_rejects_a_put_tail_that_finished_before_lease_acquisition() {
let before = BTreeMap::from([("node-2".to_string(), scanner_node_activity("epoch-a", 7, 3))]);
let expected_digest = Some(scanner_activity_snapshot_digest(&before));
assert_eq!(scanner_post_lease_activity_defer_reason(expected_digest, Ok(before.clone())), None);
let mut after = before.clone();
after
.get_mut("node-2")
.expect("writer should be present")
.namespace_generation += 1;
assert_eq!(
before["node-2"].movement_generation, after["node-2"].movement_generation,
"the existing movement-only lease remains valid after a PUT tail drains"
);
assert!(scanner_activity_allows_usage_publication(&after));
let reason = scanner_post_lease_activity_defer_reason(expected_digest, Ok(after));
assert_eq!(reason, Some(ScannerCycleDeferReason::ActivityBaselineUnavailable));
let result = ScannerCycleResult::new(ScannerCycleStatus::Complete, None).with_remote_dirty_usage_acknowledgements(vec![
ScannerDirtyUsageAcknowledgement {
host: "node-2".to_string(),
instance_id: "epoch-a".to_string(),
generation: 5,
},
]);
let (outcome, _, acknowledgements) = finalize_scanner_cycle_result(
result,
DataUsagePersistOutcome::Deferred(reason.expect("changed namespace should defer publication")),
);
assert_eq!(
outcome,
ScannerCycleOutcome::Deferred(ScannerCycleDeferReason::ActivityBaselineUnavailable)
);
assert!(
acknowledgements.is_empty(),
"a rejected publication must not acknowledge the peer's dirty usage"
);
}
#[test]
fn post_lease_activity_proof_requires_a_complete_matching_baseline() {
let before = BTreeMap::from([("node-2".to_string(), scanner_node_activity("epoch-a", 7, 3))]);
let digest = scanner_activity_snapshot_digest(&before);
let mut blocked = before.clone();
blocked.get_mut("node-2").expect("peer should be present").publication_blocked = true;
let blocked_digest = scanner_activity_snapshot_digest(&blocked);
for (expected, observed) in [
(None, Ok(before)),
(Some(digest), Err("peer is unavailable".to_string())),
(Some(digest), Ok(BTreeMap::new())),
(Some(blocked_digest), Ok(blocked)),
] {
assert_eq!(
scanner_post_lease_activity_defer_reason(expected, observed),
Some(ScannerCycleDeferReason::ActivityBaselineUnavailable)
);
}
}
#[test] #[test]
fn scanner_activity_snapshot_digest_fences_storage_topology() { fn scanner_activity_snapshot_digest_fences_storage_topology() {
let first = BTreeMap::from([("node-2".to_string(), scanner_node_activity("epoch-a", 7, 3))]); let first = BTreeMap::from([("node-2".to_string(), scanner_node_activity("epoch-a", 7, 3))]);
+2 -18
View File
@@ -12,7 +12,7 @@
// See the License for the specific language governing permissions and // See the License for the specific language governing permissions and
// limitations under the License. // limitations under the License.
use crate::data_usage_define::{DATA_USAGE_CACHE_KEY_FORMAT, DataUsageCacheRevisions}; use crate::data_usage_define::DATA_USAGE_CACHE_KEY_FORMAT;
use crate::scanner_budget::ScannerCycleBudget; use crate::scanner_budget::ScannerCycleBudget;
use crate::scanner_folder::{ScannerItem, scan_data_folder}; use crate::scanner_folder::{ScannerItem, scan_data_folder};
use crate::sleeper::SCANNER_SLEEPER; use crate::sleeper::SCANNER_SLEEPER;
@@ -271,8 +271,6 @@ pub struct ScannerBucketScanPlan {
all_buckets: Arc<Vec<BucketInfo>>, all_buckets: Arc<Vec<BucketInfo>>,
scope: ScannerBucketScanScope, scope: ScannerBucketScanScope,
digest: DataUsageScanPlanDigest, digest: DataUsageScanPlanDigest,
// Cache work must invalidate on namespace completion even when its scoped baseline remains reusable.
execution_digest: DataUsageScanPlanDigest,
leader_epoch: u64, leader_epoch: u64,
tier_registry_generation: u64, tier_registry_generation: u64,
/// Epoch captured once for the whole scanner cycle. `None` is retained /// Epoch captured once for the whole scanner cycle. `None` is retained
@@ -458,12 +456,9 @@ async fn scanner_cycle_activity_status<S>(
where where
S: ScannerStorage, S: ScannerStorage,
{ {
// Read the pending-commit barrier before sampling its completion generation.
// A tail that drains during this await must invalidate the earlier baseline.
let publication_blocked = store.scanner_data_usage_publication_blocked().await;
match crate::scanner::probe_scanner_activity(store, distributed).await { match crate::scanner::probe_scanner_activity(store, distributed).await {
Ok(after) => { Ok(after) => {
let status = if !publication_blocked && after == *before { let status = if after == *before {
ScannerCycleActivityStatus::Unchanged ScannerCycleActivityStatus::Unchanged
} else { } else {
ScannerCycleActivityStatus::Changed ScannerCycleActivityStatus::Changed
@@ -765,7 +760,6 @@ fn scanner_activity_preflight(
pub(crate) struct ScannerCycleResult { pub(crate) struct ScannerCycleResult {
pub(crate) status: ScannerCycleStatus, pub(crate) status: ScannerCycleStatus,
publication_epoch: Option<u64>, publication_epoch: Option<u64>,
activity_digest: Option<[u8; 32]>,
observational_snapshot_published: bool, observational_snapshot_published: bool,
dirty_usage_clear: Option<DirtyUsageBuckets>, dirty_usage_clear: Option<DirtyUsageBuckets>,
remote_dirty_usage_acknowledgements: Vec<crate::scanner::ScannerDirtyUsageAcknowledgement>, remote_dirty_usage_acknowledgements: Vec<crate::scanner::ScannerDirtyUsageAcknowledgement>,
@@ -780,7 +774,6 @@ impl ScannerCycleResult {
Self { Self {
status, status,
publication_epoch: None, publication_epoch: None,
activity_digest: None,
observational_snapshot_published: false, observational_snapshot_published: false,
dirty_usage_clear, dirty_usage_clear,
remote_dirty_usage_acknowledgements: Vec::new(), remote_dirty_usage_acknowledgements: Vec::new(),
@@ -800,15 +793,6 @@ impl ScannerCycleResult {
self.publication_epoch self.publication_epoch
} }
fn with_activity_digest(mut self, activity_digest: [u8; 32]) -> Self {
self.activity_digest = Some(activity_digest);
self
}
pub(crate) fn activity_digest(&self) -> Option<[u8; 32]> {
self.activity_digest
}
pub(crate) fn with_observational_snapshot_published(mut self, published: bool) -> Self { pub(crate) fn with_observational_snapshot_published(mut self, published: bool) -> Self {
self.observational_snapshot_published = published; self.observational_snapshot_published = published;
self self
+12 -30
View File
@@ -604,12 +604,10 @@ pub(super) async fn persist_and_publish_cache_snapshot(
store: Arc<SetDisks>, store: Arc<SetDisks>,
updates: &mpsc::Sender<DataUsageCache>, updates: &mpsc::Sender<DataUsageCache>,
mut cache_snapshot: DataUsageCache, mut cache_snapshot: DataUsageCache,
initial_revisions: Option<&DataUsageCacheRevisions>,
cache_cycle_floor: &AtomicU64, cache_cycle_floor: &AtomicU64,
expected_publication_epoch: u64, expected_publication_epoch: u64,
) -> Option<SystemTime> { ) -> Option<SystemTime> {
let source = cache_snapshot.info.source?; let source = cache_snapshot.info.source?;
let execution_digest = cache_snapshot.info.scan_execution_digest?;
let guard = match acquire_scanner_cache_locks(store.as_ref(), DATA_USAGE_CACHE_NAME, source).await { let guard = match acquire_scanner_cache_locks(store.as_ref(), DATA_USAGE_CACHE_NAME, source).await {
Ok(guard) => guard, Ok(guard) => guard,
Err(err) => { Err(err) => {
@@ -674,36 +672,20 @@ pub(super) async fn persist_and_publish_cache_snapshot(
); );
return None; return None;
} }
if persisted.info.scan_execution_digest == Some(execution_digest) if matches!(
&& matches!( current_cache_root_entry_with_generation(
current_cache_root_entry_with_generation( &persisted,
&persisted, DATA_USAGE_ROOT,
DATA_USAGE_ROOT, source,
source, cache_snapshot.info.next_cycle,
cache_snapshot.info.next_cycle, cache_snapshot.info.leader_epoch,
cache_snapshot.info.leader_epoch, scan_plan_digest,
scan_plan_digest, cache_snapshot.info.tier_registry_generation,
cache_snapshot.info.tier_registry_generation, ),
), Ok(Some(_))
Ok(Some(_)) ) {
)
{
cache_snapshot = persisted; cache_snapshot = persisted;
} else { } else {
// A later execution may have completed while this scan was walking.
// Only replace the cache revision from which this scan started.
if initial_revisions != Some(&revisions) {
warn!(
target: "rustfs::scanner::io",
event = EVENT_SCANNER_CACHE_PERSIST_STATE,
component = LOG_COMPONENT_SCANNER,
subsystem = LOG_SUBSYSTEM_IO,
state = "scan_baseline_revision_changed",
cache_name = DATA_USAGE_CACHE_NAME,
"Scanner skipped set snapshot without an unchanged baseline revision"
);
return None;
}
if guard.is_lock_lost() { if guard.is_lock_lost() {
error!( error!(
target: "rustfs::scanner::io", target: "rustfs::scanner::io",
+15 -24
View File
@@ -118,7 +118,6 @@ impl ScannerIOCache for SetDisks {
all_buckets, all_buckets,
scope, scope,
digest: scan_plan_digest, digest: scan_plan_digest,
execution_digest,
leader_epoch, leader_epoch,
tier_registry_generation, tier_registry_generation,
publication_epoch, publication_epoch,
@@ -138,24 +137,20 @@ impl ScannerIOCache for SetDisks {
.ok_or_else(|| StorageError::other("scanner cache publication is blocked by data movement"))?, .ok_or_else(|| StorageError::other("scanner cache publication is blocked by data movement"))?,
}; };
let mut old_cache = DataUsageCache::default(); let mut old_cache = DataUsageCache::default();
let initial_revisions = match old_cache.load_with_revisions(self.clone(), DATA_USAGE_CACHE_NAME).await { if let Err(e) = old_cache.load(self.clone(), DATA_USAGE_CACHE_NAME).await {
Ok(revisions) => Some(revisions), warn!(
Err(e) => { target: "rustfs::scanner::io",
warn!( event = EVENT_SCANNER_CACHE_PERSIST_STATE,
target: "rustfs::scanner::io", component = LOG_COMPONENT_SCANNER,
event = EVENT_SCANNER_CACHE_PERSIST_STATE, subsystem = LOG_SUBSYSTEM_IO,
component = LOG_COMPONENT_SCANNER, pool = self.pool_index,
subsystem = LOG_SUBSYSTEM_IO, set = self.set_index,
pool = self.pool_index, cache_name = DATA_USAGE_CACHE_NAME,
set = self.set_index, state = "old_cache_load_failed",
cache_name = DATA_USAGE_CACHE_NAME, error = %e,
state = "old_cache_load_failed", "Scanner old data usage cache load failed; rebuilding from bucket caches"
error = %e, );
"Scanner old data usage cache load failed; rebuilding from bucket caches" }
);
None
}
};
let scoped_scan = prepare_scoped_set_scan( let scoped_scan = prepare_scoped_set_scan(
&old_cache, &old_cache,
&buckets, &buckets,
@@ -200,7 +195,6 @@ impl ScannerIOCache for SetDisks {
}; };
cache.info.last_update = Some(now); cache.info.last_update = Some(now);
cache.info.snapshot_complete = true; cache.info.snapshot_complete = true;
cache.info.scan_execution_digest = Some(execution_digest);
cache.info.lkg_snapshot_complete = false; cache.info.lkg_snapshot_complete = false;
cache.info.lkg_next_cycle = None; cache.info.lkg_next_cycle = None;
cache.info.lkg_last_update = None; cache.info.lkg_last_update = None;
@@ -214,7 +208,6 @@ impl ScannerIOCache for SetDisks {
self, self,
&updates, &updates,
cache, cache,
initial_revisions.as_ref(),
cache_cycle_floor.as_ref(), cache_cycle_floor.as_ref(),
expected_publication_epoch, expected_publication_epoch,
) )
@@ -644,7 +637,7 @@ impl ScannerIOCache for SetDisks {
let cache_name = path_join_buf(&[&bucket.name, DATA_USAGE_CACHE_NAME]); let cache_name = path_join_buf(&[&bucket.name, DATA_USAGE_CACHE_NAME]);
let bucket_scan_plan_digest = let bucket_scan_plan_digest =
scanner_bucket_cache_digest(execution_digest, dirty_usage_buckets_clone.get(&bucket.name).copied()); scanner_bucket_cache_digest(scan_plan_digest, dirty_usage_buckets_clone.get(&bucket.name).copied());
if let Some(server_epoch) = remote_server_epoch { if let Some(server_epoch) = remote_server_epoch {
let request_sequence = remote_session_sequence; let request_sequence = remote_session_sequence;
@@ -1367,7 +1360,6 @@ impl ScannerIOCache for SetDisks {
cache.info.next_cycle = want_cycle; cache.info.next_cycle = want_cycle;
cache.info.last_update.get_or_insert_with(SystemTime::now); cache.info.last_update.get_or_insert_with(SystemTime::now);
cache.info.snapshot_complete = true; cache.info.snapshot_complete = true;
cache.info.scan_execution_digest = Some(execution_digest);
cache.info.lkg_snapshot_complete = false; cache.info.lkg_snapshot_complete = false;
cache.info.lkg_next_cycle = None; cache.info.lkg_next_cycle = None;
cache.info.lkg_last_update = None; cache.info.lkg_last_update = None;
@@ -1379,7 +1371,6 @@ impl ScannerIOCache for SetDisks {
self.clone(), self.clone(),
&updates, &updates,
cache_snapshot, cache_snapshot,
initial_revisions.as_ref(),
cache_cycle_floor.as_ref(), cache_cycle_floor.as_ref(),
expected_publication_epoch, expected_publication_epoch,
) )
+1 -9
View File
@@ -180,7 +180,7 @@ where
// canceled decommission remains suspended after its worker exits, so // canceled decommission remains suspended after its worker exits, so
// starting a scan in that state could build a snapshot that cannot be // starting a scan in that state could build a snapshot that cannot be
// routed to the authoritative metadata object. // routed to the authoritative metadata object.
if store.scanner_data_movement_pause_status().await.paused { if store.scanner_data_usage_publication_blocked().await {
debug!( debug!(
target: "rustfs::scanner::io", target: "rustfs::scanner::io",
event = EVENT_SCANNER_SET_STATE, event = EVENT_SCANNER_SET_STATE,
@@ -260,13 +260,8 @@ where
} }
} }
bucket_plan_complete &= buckets_by_source.keys().copied().collect::<HashSet<_>>() == *expected_sources; bucket_plan_complete &= buckets_by_source.keys().copied().collect::<HashSet<_>>() == *expected_sources;
let activity_digest = crate::scanner::scanner_activity_snapshot_digest(&activity_before);
let scan_plan_digest = let scan_plan_digest =
scanner_bucket_plan_digest(&all_buckets, crate::scanner::scanner_activity_structural_digest(&activity_before)); scanner_bucket_plan_digest(&all_buckets, crate::scanner::scanner_activity_structural_digest(&activity_before));
let mut execution_hasher = Sha256::new();
execution_hasher.update(scan_plan_digest.0);
execution_hasher.update(activity_digest);
let execution_digest = DataUsageScanPlanDigest(execution_hasher.finalize().into());
let dirty_usage_snapshot = Arc::new(snapshot_dirty_usage_buckets(&all_buckets, dirty_generation_before_bucket_list)); let dirty_usage_snapshot = Arc::new(snapshot_dirty_usage_buckets(&all_buckets, dirty_generation_before_bucket_list));
let scan_scope = resolve_scanner_bucket_scan_scope( let scan_scope = resolve_scanner_bucket_scan_scope(
store, store,
@@ -331,7 +326,6 @@ where
}; };
return Ok(ScannerCycleResult::new(status, dirty_usage_clear) return Ok(ScannerCycleResult::new(status, dirty_usage_clear)
.with_publication_epoch(publication_epoch) .with_publication_epoch(publication_epoch)
.with_activity_digest(activity_digest)
.with_observational_snapshot_published(observational_snapshot_published) .with_observational_snapshot_published(observational_snapshot_published)
.with_remote_publication_lease_targets(remote_publication_lease_targets) .with_remote_publication_lease_targets(remote_publication_lease_targets)
.with_remote_dirty_usage_acknowledgements(remote_dirty_usage_acknowledgements)); .with_remote_dirty_usage_acknowledgements(remote_dirty_usage_acknowledgements));
@@ -416,7 +410,6 @@ where
all_buckets: Arc::clone(&all_buckets), all_buckets: Arc::clone(&all_buckets),
scope: scan_scope.clone(), scope: scan_scope.clone(),
digest: scan_plan_digest, digest: scan_plan_digest,
execution_digest,
leader_epoch, leader_epoch,
tier_registry_generation, tier_registry_generation,
publication_epoch, publication_epoch,
@@ -605,7 +598,6 @@ where
}; };
Ok(ScannerCycleResult::new(cycle_status, dirty_usage_clear) Ok(ScannerCycleResult::new(cycle_status, dirty_usage_clear)
.with_publication_epoch(publication_epoch) .with_publication_epoch(publication_epoch)
.with_activity_digest(activity_digest)
.with_observational_snapshot_published(observational_snapshot_published) .with_observational_snapshot_published(observational_snapshot_published)
.with_remote_publication_lease_targets(remote_publication_lease_targets) .with_remote_publication_lease_targets(remote_publication_lease_targets)
.with_remote_dirty_usage_acknowledgements(remote_dirty_usage_acknowledgements) .with_remote_dirty_usage_acknowledgements(remote_dirty_usage_acknowledgements)
+1 -236
View File
@@ -20,7 +20,6 @@ use crate::scanner_folder::ScannerItem;
use crate::storage_api::EcstoreScannerPeerDirtyUsageSnapshot; use crate::storage_api::EcstoreScannerPeerDirtyUsageSnapshot;
use crate::storage_api::owner::{ use crate::storage_api::owner::{
EcstorePoolDecommissionInfo, EcstoreRebalStatus, EcstoreRebalanceInfo, EcstoreRebalanceMeta, EcstoreRebalanceStats, EcstorePoolDecommissionInfo, EcstoreRebalStatus, EcstoreRebalanceInfo, EcstoreRebalanceMeta, EcstoreRebalanceStats,
ecstore_hold_namespace_commit,
}; };
use crate::storage_api::scan::{BucketOperations as _, DeleteBucketOptions, MakeBucketOptions, ObjectIO as _}; use crate::storage_api::scan::{BucketOperations as _, DeleteBucketOptions, MakeBucketOptions, ObjectIO as _};
use crate::{ use crate::{
@@ -344,16 +343,6 @@ async fn multi_pool_scanner_cycle_publishes_combined_usage() {
.put_object(&bucket, object, &mut reader, &ScannerObjectOptions::default()) .put_object(&bucket, object, &mut reader, &ScannerObjectOptions::default())
.await .await
.expect("object should be written to its selected pool"); .expect("object should be written to its selected pool");
// Quorum ACK can precede tail publication on the disk chosen to scan.
let lock = store.pools[pool_index].disk_set[0]
.new_ns_lock(&bucket, object)
.await
.expect("fixture namespace lock should be created");
let _settled = lock
.get_write_lock(Duration::from_secs(30))
.await
.expect("fixture rename tail should finish before the usage scan");
} }
let ctx = CancellationToken::new(); let ctx = CancellationToken::new();
@@ -373,7 +362,7 @@ async fn multi_pool_scanner_cycle_publishes_combined_usage() {
.buckets_usage .buckets_usage
.get(&bucket) .get(&bucket)
.expect("combined bucket usage should be present"); .expect("combined bucket usage should be present");
assert_eq!(bucket_usage.objects_count, 2, "{usage:?}"); assert_eq!(bucket_usage.objects_count, 2);
assert_eq!(bucket_usage.size, 11); assert_eq!(bucket_usage.size, 11);
assert_eq!(usage.objects_total_count, 2); assert_eq!(usage.objects_total_count, 2);
assert_eq!(usage.objects_total_size, 11); assert_eq!(usage.objects_total_size, 11);
@@ -383,102 +372,6 @@ async fn multi_pool_scanner_cycle_publishes_combined_usage() {
); );
} }
#[tokio::test]
#[serial]
async fn pending_put_commit_keeps_scanner_walk_live_without_authoritative_usage() {
let (_temp_dir, store) = setup_two_pool_scanner_store().await;
let bucket = format!("scanner-pending-put-{}", Uuid::new_v4().simple());
store
.make_bucket(&bucket, &MakeBucketOptions::default())
.await
.expect("bucket should be created across both pools");
for (pool_index, (object, body)) in [("pool-a", b"first".as_slice()), ("pool-b", b"second".as_slice())]
.into_iter()
.enumerate()
{
let mut reader = ScannerPutObjReader::from_vec(body.to_vec());
store.pools[pool_index].disk_set[0]
.put_object(
&bucket,
object,
&mut reader,
&ScannerObjectOptions {
no_lock: true,
..Default::default()
},
)
.await
.expect("fixture objects must finish their rename fanouts before scanning");
}
let mut pending = Some(ecstore_hold_namespace_commit(store.as_ref()));
let mut previous_activity_digest = None;
let mut structural_plan_digest = None;
for (cycle, converged) in [(1, false), (2, true)] {
if converged {
drop(pending.take());
}
assert_eq!(store.scanner_data_usage_publication_blocked().await, !converged);
assert!(!store.scanner_data_movement_pause_status().await.paused);
let activity = crate::scanner::probe_scanner_activity(store.as_ref(), false)
.await
.expect("the fixture activity should be observable");
let activity_digest = crate::scanner::scanner_activity_snapshot_digest(&activity);
if let Some(previous) = previous_activity_digest.replace(activity_digest) {
assert_ne!(previous, activity_digest, "draining a namespace commit must change the publication proof");
}
let ctx = CancellationToken::new();
let budget = ScannerCycleBudget::new_with_progress_tracking(&ctx, ScannerCycleBudgetConfig::default());
let (updates, mut receiver) = mpsc::channel(1);
let result = tokio::time::timeout(
Duration::from_secs(30),
ScannerIOCycle::nsscanner_with_status(
store.as_ref(),
ctx,
Arc::clone(&budget),
updates,
cycle,
1,
HealScanMode::Normal,
),
)
.await
.expect("namespace scanning must finish while a PUT commit is pending")
.expect("namespace scanning must remain available during a pending PUT commit");
assert_eq!(result.activity_digest(), Some(activity_digest));
if !converged {
assert_eq!(budget.progress().0, 2, "the pending commit must not suppress actual object traversal");
}
assert_eq!(
result.status,
if converged {
ScannerCycleStatus::Complete
} else {
ScannerCycleStatus::Superseded
}
);
let usage = receiver
.recv()
.await
.expect("the completed walk should produce a usage candidate");
assert_eq!(usage.usage_snapshot_converged, Some(converged));
assert_eq!(usage.scanner_cycle, Some(cycle));
assert_eq!(usage.objects_total_count, 2);
assert_eq!(usage.objects_total_size, 11);
assert_eq!(usage.usage_snapshot_set_states.len(), 2);
for state in &usage.usage_snapshot_set_states {
let digest = state
.scan_plan_digest
.expect("each set must retain its structural cache identity");
assert_eq!(*structural_plan_digest.get_or_insert(digest), digest);
}
let bucket_usage = usage.buckets_usage.get(&bucket).expect("the walked bucket must be present");
assert_eq!(bucket_usage.objects_count, 2);
assert_eq!(bucket_usage.size, 11);
assert!(receiver.recv().await.is_none(), "each walk must emit exactly one terminal candidate");
}
}
#[tokio::test] #[tokio::test]
#[serial] #[serial]
async fn multi_pool_scanner_cycle_zero_fills_bucket_absent_from_first_pool() { async fn multi_pool_scanner_cycle_zero_fills_bucket_absent_from_first_pool() {
@@ -494,16 +387,6 @@ async fn multi_pool_scanner_cycle_zero_fills_bucket_absent_from_first_pool() {
.put_object(&bucket, "pool-b", &mut reader, &ScannerObjectOptions::default()) .put_object(&bucket, "pool-b", &mut reader, &ScannerObjectOptions::default())
.await .await
.expect("object should be written only to the second pool"); .expect("object should be written only to the second pool");
{
let lock = store.pools[1].disk_set[0]
.new_ns_lock(&bucket, "pool-b")
.await
.expect("fixture namespace lock should be created");
let _settled = lock
.get_write_lock(Duration::from_secs(30))
.await
.expect("fixture rename tail should finish before the usage scan");
}
store.pools[0] store.pools[0]
.delete_bucket(&bucket, &DeleteBucketOptions::default()) .delete_bucket(&bucket, &DeleteBucketOptions::default())
.await .await
@@ -914,124 +797,6 @@ fn complete_set_usage_cache(buckets: &[(&str, usize)], scan_plan_digest: DataUsa
cache cache
} }
#[tokio::test]
#[serial]
async fn set_snapshot_reuse_requires_execution_identity_and_fences_stale_writers() {
let (_temp_dir, store) = setup_two_pool_scanner_store().await;
let set = Arc::clone(&store.pools[0].disk_set[0]);
let epoch = scanner_publication_epoch(Arc::clone(&set)).await.expect("idle set admission");
let mut legacy = complete_set_usage_cache(&[("photos", 5)], DataUsageScanPlanDigest([1; 32]));
legacy.info.source = Some(DataUsageCacheSource::new(0, 0));
legacy
.save(Arc::clone(&set), DATA_USAGE_CACHE_NAME)
.await
.expect("seed legacy set cache");
let mut persisted = DataUsageCache::default();
let initial = persisted
.load_with_revisions(Arc::clone(&set), DATA_USAGE_CACHE_NAME)
.await
.expect("capture the shared starting revision");
let mut fresh = legacy.clone();
fresh.info.scan_execution_digest = Some(DataUsageScanPlanDigest([2; 32]));
fresh.replace(
"photos",
DATA_USAGE_ROOT,
DataUsageEntry {
size: 20,
objects: 1,
..Default::default()
},
);
let cycle_floor = AtomicU64::new(fresh.info.next_cycle);
let (tx, mut rx) = mpsc::channel(1);
assert!(
persist_and_publish_cache_snapshot(Arc::clone(&set), &tx, fresh.clone(), Some(&initial), &cycle_floor, epoch)
.await
.is_some(),
"a legacy cache without execution identity must be refreshed"
);
let published = rx.try_recv().expect("fresh snapshot should be forwarded");
assert_eq!(published.find("photos").expect("published bucket").size, 20);
assert_eq!(published.info.scan_execution_digest, fresh.info.scan_execution_digest);
let current = persisted
.load_with_revisions(Arc::clone(&set), DATA_USAGE_CACHE_NAME)
.await
.expect("capture the current revision for the unidentified execution");
let mut stale = legacy.clone();
stale.info.scan_execution_digest = Some(DataUsageScanPlanDigest([3; 32]));
for (candidate, revisions) in [(stale, &initial), (legacy, &current)] {
assert!(
persist_and_publish_cache_snapshot(Arc::clone(&set), &tx, candidate, Some(revisions), &cycle_floor, epoch)
.await
.is_none(),
"a stale or unidentified execution must not replace the newer snapshot"
);
assert!(matches!(rx.try_recv(), Err(mpsc::error::TryRecvError::Empty)));
}
fresh.info.scan_execution_digest = Some(DataUsageScanPlanDigest([4; 32]));
assert!(
persist_and_publish_cache_snapshot(Arc::clone(&set), &tx, fresh.clone(), None, &cycle_floor, epoch)
.await
.is_none(),
"an unreadable starting revision must not authorize an overwrite"
);
fresh.info.scan_execution_digest = published.info.scan_execution_digest;
fresh.replace("photos", DATA_USAGE_ROOT, DataUsageEntry::default());
assert!(
persist_and_publish_cache_snapshot(Arc::clone(&set), &tx, fresh, Some(&initial), &cycle_floor, epoch)
.await
.is_some(),
"an overlapping identical execution must reuse the completed snapshot"
);
assert_eq!(
rx.try_recv()
.expect("reused snapshot")
.find("photos")
.expect("reused bucket")
.size,
20
);
persisted
.load(Arc::clone(&set), DATA_USAGE_CACHE_NAME)
.await
.expect("read the final durable set cache");
assert_eq!(persisted.find("photos").expect("durable bucket").size, 20);
assert_eq!(persisted.info.scan_execution_digest, published.info.scan_execution_digest);
let ctx = CancellationToken::new();
let empty_execution = DataUsageScanPlanDigest([5; 32]);
set.nsscanner_cache(
ctx.clone(),
ScannerCycleBudget::new(&ctx, ScannerCycleBudgetConfig::default()),
ScannerBucketScanPlan {
buckets: Vec::new(),
all_buckets: Arc::new(Vec::new()),
scope: ScannerBucketScanScope::default(),
digest: DataUsageScanPlanDigest([6; 32]),
execution_digest: empty_execution,
leader_epoch: 11,
tier_registry_generation: 13,
publication_epoch: Some(epoch),
dirty_usage_buckets: Arc::new(HashMap::new()),
bucket_failures: ScannerBucketFailureState::default(),
pending_maintenance_work: Arc::new(AtomicBool::new(false)),
cache_cycle_floor: Arc::new(AtomicU64::new(8)),
},
tx,
8,
HealScanMode::Normal,
)
.await
.expect("empty set scope should replace its prior nonempty cache");
let empty = rx.try_recv().expect("empty set snapshot should be published");
assert_eq!(empty.info.scan_execution_digest, Some(empty_execution));
assert!(empty.info.snapshot_complete);
let root = empty.checked_flatten(DATA_USAGE_ROOT).expect("complete empty root");
assert_eq!((root.size, root.objects), (0, 0));
}
fn complete_usage_baseline( fn complete_usage_baseline(
source: DataUsageCacheSource, source: DataUsageCacheSource,
scan_plan_digest: DataUsageScanPlanDigest, scan_plan_digest: DataUsageScanPlanDigest,
-3
View File
@@ -127,9 +127,6 @@ pub(crate) use rustfs_lifecycle::{
use rustfs_storage_api as storage_contracts; use rustfs_storage_api as storage_contracts;
pub(crate) mod owner { pub(crate) mod owner {
#[cfg(test)]
pub(crate) use rustfs_ecstore::api::set_disk::test_util::hold_namespace_commit as ecstore_hold_namespace_commit;
pub(crate) use super::storage_contracts::{ pub(crate) use super::storage_contracts::{
HTTPPreconditions, HTTPRangeSpec, NS_SCANNER_PROTOCOL_VERSION, ObjectIO, ObjectOperations, ObjectToDelete, HTTPPreconditions, HTTPRangeSpec, NS_SCANNER_PROTOCOL_VERSION, ObjectIO, ObjectOperations, ObjectToDelete,
}; };
@@ -1020,7 +1020,6 @@ mod serial_tests {
} }
let (_disk_paths, ecstore) = setup_isolated_test_env(false).await; let (_disk_paths, ecstore) = setup_isolated_test_env(false).await;
let expired_recovery_time = i128::from(i64::MAX / 2);
for case in [ for case in [
CleanupCase::Persisted, CleanupCase::Persisted,
@@ -1118,7 +1117,7 @@ mod serial_tests {
.await .await
.expect("active unknown ownership must remain fenced after the transaction store was offline"); .expect("active unknown ownership must remain fenced after the transaction store was offline");
assert_eq!((retained.scanned, retained.recovered, retained.retained, retained.failed), (1, 0, 1, 0)); assert_eq!((retained.scanned, retained.recovered, retained.retained, retained.failed), (1, 0, 1, 0));
let recovered = recover_transition_transaction_records_at(ecstore.clone(), 100, None, expired_recovery_time) let recovered = recover_transition_transaction_records_at(ecstore.clone(), 100, None, i128::MAX)
.await .await
.expect("expired unknown ownership may use the provider's missing proof"); .expect("expired unknown ownership may use the provider's missing proof");
assert_eq!( assert_eq!(
@@ -1167,7 +1166,7 @@ mod serial_tests {
assert_eq!(retained.recovered, 0); assert_eq!(retained.recovered, 0);
assert_eq!(retained.retained + retained.failed, 1); assert_eq!(retained.retained + retained.failed, 1);
backend.set_remove_failure(false); backend.set_remove_failure(false);
let recovered = recover_transition_transaction_records_at(ecstore.clone(), 100, None, expired_recovery_time) let recovered = recover_transition_transaction_records_at(ecstore.clone(), 100, None, i128::MAX)
.await .await
.expect("expired recovery should delete the candidate after the backend becomes available"); .expect("expired recovery should delete the candidate after the backend becomes available");
assert_eq!( assert_eq!(
@@ -13,8 +13,8 @@ Operator-facing behaviour, configuration, and troubleshooting for these services
| Service | Desired source | Current-status inputs | Status surface | Side effects | | Service | Desired source | Current-status inputs | Status surface | Side effects |
|---|---|---|---|---| |---|---|---|---|---|
| Write-back pull pipeline (`rustfs/src/on_demand_migration/pull.rs`; the local write is delegated to the app layer in `rustfs/src/app/object/on_demand_migration_put.rs`) | The bucket's `on-demand-migration.json` (`enabled`, `policy.max_concurrent_pulls`, `pull_queue_capacity`, `multipart_part_size_bytes`, `bandwidth_limit_bytes_per_sec`) together with the process switch `RUSTFS_ON_DEMAND_MIGRATION_ENABLED` | Per-bucket runtime state in `rustfs/src/on_demand_migration/sys.rs`: whether a state is installed, whether its source client built, its cancellation token, queue depth, in-flight pull permits | `GET /rustfs/admin/v3/on-demand-migration/{bucket}/status` (`inflight_pulls`, `queue_depth`, `counters.pulled_*`, `counters.pull_failures_total`) and the `rustfs_on_demand_migration_*` series | Source GET/HEAD/GetObjectTagging traffic; local object writes through the internal put path, hence quota consumption, bucket default SSE, versioning, Object Lock defaults, `ObjectCreated` notifications, and outbound replication scheduling | | Write-back pull pipeline (`crates/ecstore/src/bucket/on_demand_migration/pull.rs`; the local write is delegated to the app layer in `rustfs/src/app/object/on_demand_migration_put.rs`) | The bucket's `on-demand-migration.json` (`enabled`, `policy.max_concurrent_pulls`, `pull_queue_capacity`, `multipart_part_size_bytes`, `bandwidth_limit_bytes_per_sec`) together with the process switch `RUSTFS_ON_DEMAND_MIGRATION_ENABLED` | Per-bucket runtime state in `crates/ecstore/src/bucket/on_demand_migration/sys.rs`: whether a state is installed, whether its source client built, its cancellation token, queue depth, in-flight pull permits | `GET /rustfs/admin/v3/on-demand-migration/{bucket}/status` (`inflight_pulls`, `queue_depth`, `counters.pulled_*`, `counters.pull_failures_total`) and the `rustfs_on_demand_migration_*` series | Source GET/HEAD/GetObjectTagging traffic; local object writes through the internal put path, hence quota consumption, bucket default SSE, versioning, Object Lock defaults, `ObjectCreated` notifications, and outbound replication scheduling |
| Backfill job (module under `rustfs/src/on_demand_migration/`, rustfs/backlog#2159) | An admin `start` request plus the bucket config; invalidated when the config's `updated_at` changes or the config is deleted | The persisted checkpoint under the bucket's metadata prefix, its `state` field, and the owner lease | The backfill section of the bucket status endpoint and the `rustfs_on_demand_migration_backfill_*` series | Source `ListObjectsV2` paging; queue admission into the write-back pipeline (and therefore all of its side effects); checkpoint writes | | Backfill job (module under `crates/ecstore/src/bucket/on_demand_migration/`, rustfs/backlog#2159 — not yet in the tree) | An admin `start` request plus the bucket config; invalidated when the config's `updated_at` changes or the config is deleted | The persisted checkpoint under the bucket's metadata prefix, its `state` field, and the owner lease | The backfill section of the bucket status endpoint and the `rustfs_on_demand_migration_backfill_*` series | Source `ListObjectsV2` paging; queue admission into the write-back pipeline (and therefore all of its side effects); checkpoint writes |
| Backfill recovery loop (registered from `rustfs/src/startup_background.rs`, rustfs/backlog#2159) | The set of persisted checkpoints in `state = running`; runs on every node | Checkpoint owner lease expiry | Takeover is reported through the same backfill status; a takeover emits a warn-level lease event | Claims the lease and resumes the backfill job, inheriting its side effects. Scanning checkpoints is read-only | | Backfill recovery loop (registered from `rustfs/src/startup_background.rs`, rustfs/backlog#2159 — not yet in the tree) | The set of persisted checkpoints in `state = running`; runs on every node | Checkpoint owner lease expiry | Takeover is reported through the same backfill status; a takeover emits a warn-level lease event | Claims the lease and resumes the backfill job, inheriting its side effects. Scanning checkpoints is read-only |
The pull pipeline has no separate loop of its own: a bucket's queue dispatcher starts lazily on the first background pull and is cancelled when the bucket's state is rebuilt or removed, and each inline pull commits in a task that outlives its request so a client disconnect cannot truncate the stored object. Neither the switch nor the config is re-read by the workers: the bucket-metadata publish hook rebuilds the state, which is the only desired-state path. The pull pipeline has no separate loop of its own: a bucket's queue dispatcher starts lazily on the first background pull and is cancelled when the bucket's state is rebuilt or removed, and each inline pull commits in a task that outlives its request so a client disconnect cannot truncate the stored object. Neither the switch nor the config is re-read by the workers: the bucket-metadata publish hook rebuilds the state, which is the only desired-state path.
@@ -11,7 +11,6 @@
## Open Items ## Open Items
- `backlog-2263` legacy heal MRF inspection: retained per-record journals remain readable while committed-snapshot ownership and writer activation are staged. Remove legacy import only after all supported direct-upgrade and rollback readers understand committed snapshots and migration tooling confirms that no retained or restorable legacy journal requires it. This does not enable a new writer or change the automatic legacy consumer.
- `backlog-1337` legacy restore orphan recovery: releases that predate the restore worker-lock marker can leave a valid operation-id and `ongoing-request="true"` after cancellation or process failure, with no durable liveness proof. New servers allow an exact, non-nil legacy generation to be superseded only when its consistently parsed request date is at least 24 hours old. Remove the clock-based legacy fallback after the minimum supported direct-upgrade release writes the v1 worker-lock marker on every restore and operators have resolved every retained pre-v1 ongoing generation. - `backlog-1337` legacy restore orphan recovery: releases that predate the restore worker-lock marker can leave a valid operation-id and `ongoing-request="true"` after cancellation or process failure, with no durable liveness proof. New servers allow an exact, non-nil legacy generation to be superseded only when its consistently parsed request date is at least 24 hours old. Remove the clock-based legacy fallback after the minimum supported direct-upgrade release writes the v1 worker-lock marker on every restore and operators have resolved every retained pre-v1 ongoing generation.
- `backlog-2133-tier-delete-chunk-parent` bounded tier-delete dispatch compatibility: prefixes at or below the legacy manifest limit keep the byte-compatible v1 single-manifest protocol, while larger prefixes place a chunk-parent sentinel at the original deterministic root path and use operation-scoped child manifests. Older binaries reject the sentinel and child paths, preserving the v6 sole-owner downgrade fence instead of starting a competing local delete. Remove the v1 reader and fail-closed mixed-version sentinel only after every supported rollback release validates the parent/child protocol and migration tooling confirms that no retained v1 dispatch manifest remains. - `backlog-2133-tier-delete-chunk-parent` bounded tier-delete dispatch compatibility: prefixes at or below the legacy manifest limit keep the byte-compatible v1 single-manifest protocol, while larger prefixes place a chunk-parent sentinel at the original deterministic root path and use operation-scoped child manifests. Older binaries reject the sentinel and child paths, preserving the v6 sole-owner downgrade fence instead of starting a competing local delete. Remove the v1 reader and fail-closed mixed-version sentinel only after every supported rollback release validates the parent/child protocol and migration tooling confirms that no retained v1 dispatch manifest remains.
- `tokio-tar-extension-limits` bounded archive parser hardening: Snowball extraction depends on precedence-resolved MinIO PAX metadata; per-entry and cumulative extension limits; a physical-entry limit; cancellation-safe parsing and ownership of large streamed members; fused streams after errors; and compatibility with minio-go streams that omit the two-block terminator. Swift bulk extraction also uses the same fork. Keep the reviewed pin while the Snowball path is prototyped against tar-codec/tar-framing. Remove it only after a released API exposes the effective allowed vendor records, RustFS provides a cancellation-safe handoff for borrowed member payloads, footerless input is accepted solely when authenticated request framing proves EOF immediately after a complete member, the existing resource-limit, cancellation, error-fuse, and real minio-go fixtures pass against the replacement, and Swift no longer depends on the fork. - `tokio-tar-extension-limits` bounded archive parser hardening: Snowball extraction depends on precedence-resolved MinIO PAX metadata; per-entry and cumulative extension limits; a physical-entry limit; cancellation-safe parsing and ownership of large streamed members; fused streams after errors; and compatibility with minio-go streams that omit the two-block terminator. Swift bulk extraction also uses the same fork. Keep the reviewed pin while the Snowball path is prototyped against tar-codec/tar-framing. Remove it only after a released API exposes the effective allowed vendor records, RustFS provides a cancellation-safe handoff for borrowed member payloads, footerless input is accepted solely when authenticated request framing proves EOF immediately after a complete member, the existing resource-limit, cancellation, error-fuse, and real minio-go fixtures pass against the replacement, and Swift no longer depends on the fork.

Some files were not shown because too many files have changed in this diff Show More