Compare commits

..

1 Commits

Author SHA1 Message Date
loverustfs 43a7f94f14 fix(scanner): preserve failed usage and reliable heal sampling 2026-09-05 09:57:10 +08:00
228 changed files with 5565 additions and 29084 deletions
+1 -1
View File
@@ -1,2 +1,2 @@
sha256-darwin=a881fd7d3f5cb94654221ca85b8b30cce1b95e608824a55a15339cbc294e6d34 sha256-darwin=a881fd7d3f5cb94654221ca85b8b30cce1b95e608824a55a15339cbc294e6d34
sha256-linux=a2933d83dfe74ffa03410a0959333a1c48288b8469ca9f17273d449d7510c24b sha256-linux=e9a8d64e73f627c4d26c236dbbba690c9ee03a9e26d42a4244515b4439365535
-72
View File
@@ -1,72 +0,0 @@
{
"lane": "ci/test-and-lint",
"tests": [
{
"invariant": "write-quorum",
"suite": "rustfs-ecstore",
"name": "set_disk::ops::object::inline_put_commit_path_tests::inline_put_direct_commit_accepts_exact_quorum_and_rejects_quorum_minus_one"
},
{
"invariant": "metadata-rollback",
"suite": "rustfs-ecstore",
"name": "set_disk::core::io_primitives::tests::write_unique_file_info_reverts_metadata_when_write_quorum_fails"
},
{
"invariant": "stale-writer",
"suite": "rustfs-ecstore",
"name": "set_disk::ops::object::put_object_tmp_cleanup_tests::put_object_no_lock_aborts_after_outer_namespace_lock_loss"
},
{
"invariant": "range-body",
"suite": "rustfs-ecstore",
"name": "set_disk::ops::object::transition_upload_integrity_tests::transitioned_compressed_object_range_get_returns_plaintext_slice"
},
{
"invariant": "multipart-cancellation",
"suite": "rustfs-ecstore",
"name": "set_disk::ops::multipart::tests::cancelled_complete_keeps_upload_lock_through_tail_cleanup"
},
{
"invariant": "list-uncommitted-version",
"suite": "rustfs-filemeta",
"name": "metacache::tests::resolve_with_write_quorum_slack_keeps_partial_latest_hidden_during_merge"
},
{
"invariant": "minio-object-fixture",
"suite": "rustfs-filemeta",
"name": "filemeta::test::parses_real_minio_object_xlmeta"
},
{
"invariant": "corrupt-part-arrays",
"suite": "rustfs-filemeta",
"name": "filemeta::test::crc_valid_but_part_arrays_corrupt_into_fileinfo_errors_not_panics"
}
],
"fixtures": [
{
"path": "crates/filemeta/tests/fixtures/minio/object_large_bin.xlmeta.hex",
"sha256": "e8093767806d701e639b48d023190e858fbc4cde69bcfd83c22af8cba8452ce5",
"source": "MinIO RELEASE.2025-07-23T15-54-02Z; crates/ecstore/tests/fixtures/minio/README.md"
},
{
"path": "crates/filemeta/tests/fixtures/minio/object_small_txt.xlmeta.hex",
"sha256": "2a415ad3a3be5a9440035d4026ff880e0e8c1ec1701be9f4e077734e8dce03da",
"source": "MinIO RELEASE.2025-07-23T15-54-02Z; crates/ecstore/tests/fixtures/minio/README.md"
},
{
"path": "crates/filemeta/tests/fixtures/minio/object_versioned_txt.xlmeta.hex",
"sha256": "7f21f50c326dd8b0228deb6dbdb7052b3d0a3f8ee6c85d43486f0e6bb7a97261",
"source": "MinIO RELEASE.2025-07-23T15-54-02Z; crates/ecstore/tests/fixtures/minio/README.md"
},
{
"path": "crates/ecstore/tests/fixtures/minio/bucket_metadata.blob.hex",
"sha256": "f2b6e260aff106adf6039feb1c645686e84e75404ff725491fb18668be5db203",
"source": "MinIO RELEASE.2025-07-23T15-54-02Z; crates/ecstore/tests/fixtures/minio/README.md"
},
{
"path": "crates/ecstore/tests/fixtures/minio/bucket_metadata_full.xlmeta.hex",
"sha256": "3b6de589519c08a1614c8bd409bb8199c17d42043861b07bce513075e6fbfc12",
"source": "MinIO RELEASE.2025-07-23T15-54-02Z; crates/ecstore/tests/fixtures/minio/README.md"
}
]
}
+2 -3
View File
@@ -3,10 +3,9 @@
.NOTPARALLEL: pre-commit pre-pr dev-check .NOTPARALLEL: pre-commit pre-pr dev-check
.PHONY: setup-hooks .PHONY: setup-hooks
setup-hooks: ## Install the configured pre-commit hooks setup-hooks: ## Set up git hooks
@echo "🔧 Setting up git hooks..." @echo "🔧 Setting up git hooks..."
pre-commit validate-config chmod +x .git/hooks/pre-commit
pre-commit install
@echo "✅ Git hooks setup complete!" @echo "✅ Git hooks setup complete!"
.PHONY: doc-paths-check .PHONY: doc-paths-check
-2
View File
@@ -40,8 +40,6 @@ script-tests: ## Run shell script tests
$(RUSTFS_PYTHON_BIN) ./scripts/check_test_wiring.py --self-test $(RUSTFS_PYTHON_BIN) ./scripts/check_test_wiring.py --self-test
$(RUSTFS_PYTHON_BIN) ./scripts/check_security_coverage.py --self-test $(RUSTFS_PYTHON_BIN) ./scripts/check_security_coverage.py --self-test
$(RUSTFS_PYTHON_BIN) ./scripts/check_scheduled_validation_freshness.py --self-test $(RUSTFS_PYTHON_BIN) ./scripts/check_scheduled_validation_freshness.py --self-test
$(RUSTFS_PYTHON_BIN) ./scripts/test_security_workflow.py
$(RUSTFS_PYTHON_BIN) ./scripts/test_nightly_candidate.py
$(RUSTFS_PYTHON_BIN) ./scripts/s3-tests/test_report_compat.py $(RUSTFS_PYTHON_BIN) ./scripts/s3-tests/test_report_compat.py
bash -n ./scripts/validate_object_data_cache_cold_stampede.sh bash -n ./scripts/validate_object_data_cache_cold_stampede.sh
$(RUSTFS_PYTHON_BIN) ./scripts/check_object_data_cache_follower_samples.py --self-test $(RUSTFS_PYTHON_BIN) ./scripts/check_object_data_cache_follower_samples.py --self-test
-120
View File
@@ -1,120 +0,0 @@
# Copyright 2024 RustFS Team
#
# Licensed under the Apache License, Version 2.0 (the "License");
# you may not use this file except in compliance with the License.
# You may obtain a copy of the License at
#
# http://www.apache.org/licenses/LICENSE-2.0
#
# Unless required by applicable law or agreed to in writing, software
# distributed under the License is distributed on an "AS IS" BASIS,
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
# See the License for the specific language governing permissions and
# limitations under the License.
name: Quick Checks
description: Run the shared compile-free RustFS quality checks.
runs:
using: composite
steps:
- name: Install quality tools
uses: taiki-e/install-action@bffeee26d4db9be238a4ea78d8826604ebcb594d # v2
with:
tool: |
ripgrep@15.2.0
shellcheck@0.11.0
- name: Install actionlint
shell: bash
run: |
actionlint_dir="$(mktemp -d "${RUNNER_TEMP}/actionlint.XXXXXX")"
curl --fail --location --silent --show-error \
--output "$actionlint_dir/actionlint.tar.gz" \
https://github.com/rhysd/actionlint/releases/download/v1.7.12/actionlint_1.7.12_linux_amd64.tar.gz
echo "8aca8db96f1b94770f1b0d72b6dddcb1ebb8123cb3712530b08cc387b349a3d8 $actionlint_dir/actionlint.tar.gz" | sha256sum --check --status
tar -xzf "$actionlint_dir/actionlint.tar.gz" -C "$actionlint_dir" actionlint
rm "$actionlint_dir/actionlint.tar.gz"
echo "$actionlint_dir" >> "$GITHUB_PATH"
- name: Install Rust toolchain
uses: dtolnay/rust-toolchain@29eef336d9b2848a0b548edc03f92a220660cdb8 # stable
with:
components: rustfmt
- name: Check workflow syntax and shell scripts
shell: bash
run: shellcheck --version && actionlint
- name: Check code formatting
shell: bash
run: cargo fmt --all --check
- name: Check unsafe code allowances
shell: bash
run: ./scripts/check_unsafe_code_allowances.sh
- name: Check layered dependencies
shell: bash
run: ./scripts/check_layer_dependencies.sh
- name: Check architecture migration rules
shell: bash
run: ./scripts/check_architecture_migration_rules.sh
- name: Check logging guardrails
shell: bash
run: ./scripts/check_logging_guardrails.sh
- name: Check error other(format!) ratchet
shell: bash
run: ./scripts/check_error_other_format_ratchet.sh
- name: Check tokio io-uring feature guard
shell: bash
run: ./scripts/check_no_tokio_io_uring.sh
- name: Check extension schema boundaries
shell: bash
run: ./scripts/check_extension_schema_boundaries.sh
- name: Check body-cache whitelist guard
shell: bash
run: ./scripts/check_body_cache_whitelist.sh
- name: Check s3s footprint ratchet
shell: bash
run: ./scripts/check_s3s_footprint.sh
- name: Check cryptographic capability wording
shell: bash
run: ./scripts/check_fips_wording.sh
- name: Check no embedded secret material
shell: bash
run: ./scripts/check_embedded_secrets.sh
- name: Run script contract tests
shell: bash
run: make script-tests
- name: Check test wiring
shell: bash
run: |
python3 ./scripts/check_test_wiring.py --self-test
python3 ./scripts/check_scheduled_validation_freshness.py --self-test
python3 ./scripts/test_security_workflow.py
python3 ./scripts/test_nightly_candidate.py
python3 ./scripts/check_test_wiring.py
- name: Check no planning docs committed
shell: bash
run: ./scripts/check_no_planning_docs.sh
- name: Check CI paths stay in sync
shell: bash
run: ./scripts/check_ci_paths_sync.sh
- name: Check io_uring lane --lib precondition
shell: bash
run: ./scripts/check_uring_lane_lib_only.sh
+5 -5
View File
@@ -10,16 +10,16 @@ Use N/A when there is no related issue.
## Summary of Changes ## Summary of Changes
<!-- <!--
Describe the concrete problem and resulting behavior. For a behavior change, name the input or state that triggers it and the expected outcome. Explain any new dependency or abstraction that the change needs. Briefly explain what changed and why reviewers should accept it.
Focus on behavior, compatibility, and review-relevant context.
--> -->
## Verification ## Verification
<!-- <!--
Give 13 concrete pieces of evidence for the changed behavior: the test or command, its observed result, and the regression it catches. For a bug fix, record a failing-before/passing-after check or explain why it was unavailable. List the commands or checks you ran, for example:
- `make pre-commit`
Identify the tested commit and any local changes. When testing a prebuilt binary or external service, include its source/version and artifact identity; a successful run against a different build is not evidence for this change. Use N/A only when verification is not applicable.
List relevant checks not run and the remaining risk. Use the validation tier in AGENTS.md; do not run broader checks solely to fill this section. For documentation-only changes, list the applicable documentation checks. Use N/A only when verification is not applicable.
--> -->
## Impact ## Impact
+88 -6
View File
@@ -12,10 +12,24 @@
# See the License for the specific language governing permissions and # See the License for the specific language governing permissions and
# limitations under the License. # limitations under the License.
# Reports the existing required checks for paths excluded by ci.yml. # Companion to ci.yml for required status checks.
# Mixed PRs can trigger both workflows; their Quick Checks jobs use one shared #
# action to keep validation coverage aligned. Keep this paths list in sync with # ci.yml skips docs-only pull requests via paths-ignore, but the branch ruleset
# ci.yml's pull_request.paths-ignore via scripts/check_ci_paths_sync.sh. # requires a check named "Test and Lint" — without this workflow a docs-only PR
# would wait on it forever. This workflow triggers on exactly the paths ci.yml
# ignores and reports success under the same job name. Mixed PRs trigger both
# workflows and the real check still gates: a required check with any failing
# run blocks the merge.
# https://docs.github.com/en/repositories/configuring-branches-and-merges-in-your-repository/defining-the-mergeability-of-pull-requests/troubleshooting-required-status-checks#handling-skipped-but-required-checks
#
# "Quick Checks" is mirrored here ahead of the ruleset change that will make it
# required too (rustfs/backlog#1599). Until that change lands this job is
# inert; mirroring it first is what lets the ruleset change happen without
# stranding docs-only PRs on a check nobody reports.
#
# Keep the paths list below in sync with the pull_request paths-ignore list
# in ci.yml, and keep the quick-checks steps below byte-identical to the
# quick-checks job in ci.yml.
name: Continuous Integration (docs only) name: Continuous Integration (docs only)
@@ -45,6 +59,19 @@ permissions:
contents: read contents: read
jobs: jobs:
# Deliberately NOT a bare `echo`. Once "Quick Checks" becomes a required
# check, ci.yml gates every expensive job behind it, so a mixed PR reports
# two check runs with this name: the real one (45-51s) and this companion.
# GitHub has no written contract for how it picks between same-named
# required check runs ("latest wins" vs "any failure blocks"), so instead of
# relying on ordering we make both runs execute the same commands against
# the same merge ref — their conclusions are then necessarily identical and
# the choice does not matter. Keep these steps byte-identical to the
# quick-checks job in ci.yml (a guard script that asserts this, and the paths
# sync below, is tracked in rustfs/backlog#1603).
#
# For a genuinely docs-only PR this adds no strictness (no code changed, so
# fmt and the guards always pass) and costs ~50s of ubuntu-latest.
quick-checks: quick-checks:
name: Quick Checks name: Quick Checks
runs-on: ubuntu-latest runs-on: ubuntu-latest
@@ -55,8 +82,63 @@ jobs:
with: with:
persist-credentials: false persist-credentials: false
- name: Run shared quick checks - name: Install ripgrep
uses: ./.github/actions/quick-checks uses: taiki-e/install-action@bffeee26d4db9be238a4ea78d8826604ebcb594d # v2
with:
tool: ripgrep@15.2.0
- name: Install Rust toolchain
uses: dtolnay/rust-toolchain@29eef336d9b2848a0b548edc03f92a220660cdb8 # stable
with:
components: rustfmt
- name: Check code formatting
run: cargo fmt --all --check
- name: Check unsafe code allowances
run: ./scripts/check_unsafe_code_allowances.sh
- name: Check layered dependencies
run: ./scripts/check_layer_dependencies.sh
- name: Check architecture migration rules
run: ./scripts/check_architecture_migration_rules.sh
- name: Check logging guardrails
run: ./scripts/check_logging_guardrails.sh
- name: Check tokio io-uring feature guard
run: ./scripts/check_no_tokio_io_uring.sh
- name: Check extension schema boundaries
run: ./scripts/check_extension_schema_boundaries.sh
- name: Check body-cache whitelist guard
run: ./scripts/check_body_cache_whitelist.sh
- name: Check s3s footprint ratchet
run: ./scripts/check_s3s_footprint.sh
- name: Check cryptographic capability wording
run: ./scripts/check_fips_wording.sh
- name: Check no embedded secret material
run: ./scripts/check_embedded_secrets.sh
- name: Check test wiring
run: |
python3 ./scripts/check_test_wiring.py --self-test
python3 ./scripts/check_scheduled_validation_freshness.py --self-test
python3 ./scripts/check_test_wiring.py
- name: Check no planning docs committed
run: ./scripts/check_no_planning_docs.sh
- name: Check CI paths stay in sync
run: ./scripts/check_ci_paths_sync.sh
- name: Check io_uring lane --lib precondition
run: ./scripts/check_uring_lane_lib_only.sh
test-and-lint: test-and-lint:
name: Test and Lint name: Test and Lint
+66 -10
View File
@@ -100,7 +100,12 @@ jobs:
- name: Typos check with custom config file - name: Typos check with custom config file
uses: crate-ci/typos@37bb98842b0d8c4ffebdb75301a13db0267cef89 # master uses: crate-ci/typos@37bb98842b0d8c4ffebdb75301a13db0267cef89 # master
# Fail early with compile-free checks shared with docs-only CI. # Fast, compile-free checks that fail early so contributors get feedback in
# ~1 minute instead of waiting for the full test job.
#
# These steps are mirrored byte-for-byte in ci-docs-only.yml so that a mixed
# PR, which reports two check runs named "Quick Checks", cannot get one red
# and one green. Edit both jobs together.
quick-checks: quick-checks:
name: Quick Checks name: Quick Checks
if: github.event_name != 'pull_request' || github.event.action != 'closed' if: github.event_name != 'pull_request' || github.event.action != 'closed'
@@ -112,8 +117,66 @@ jobs:
with: with:
persist-credentials: false persist-credentials: false
- name: Run shared quick checks - name: Install ripgrep
uses: ./.github/actions/quick-checks uses: taiki-e/install-action@bffeee26d4db9be238a4ea78d8826604ebcb594d # v2
with:
tool: ripgrep@15.2.0
- name: Install Rust toolchain
uses: dtolnay/rust-toolchain@29eef336d9b2848a0b548edc03f92a220660cdb8 # stable
with:
components: rustfmt
- name: Check code formatting
run: cargo fmt --all --check
- name: Check unsafe code allowances
run: ./scripts/check_unsafe_code_allowances.sh
- name: Check layered dependencies
run: ./scripts/check_layer_dependencies.sh
- name: Check architecture migration rules
run: ./scripts/check_architecture_migration_rules.sh
- name: Check logging guardrails
run: ./scripts/check_logging_guardrails.sh
- name: Check error other(format!) ratchet
run: ./scripts/check_error_other_format_ratchet.sh
- name: Check tokio io-uring feature guard
run: ./scripts/check_no_tokio_io_uring.sh
- name: Check extension schema boundaries
run: ./scripts/check_extension_schema_boundaries.sh
- name: Check body-cache whitelist guard
run: ./scripts/check_body_cache_whitelist.sh
- name: Check s3s footprint ratchet
run: ./scripts/check_s3s_footprint.sh
- name: Check cryptographic capability wording
run: ./scripts/check_fips_wording.sh
- name: Check no embedded secret material
run: ./scripts/check_embedded_secrets.sh
- name: Check test wiring
run: |
python3 ./scripts/check_test_wiring.py --self-test
python3 ./scripts/check_scheduled_validation_freshness.py --self-test
python3 ./scripts/check_test_wiring.py
- name: Check no planning docs committed
run: ./scripts/check_no_planning_docs.sh
- name: Check CI paths stay in sync
run: ./scripts/check_ci_paths_sync.sh
- name: Check io_uring lane --lib precondition
run: ./scripts/check_uring_lane_lib_only.sh
test-and-lint: test-and-lint:
name: Test and Lint name: Test and Lint
@@ -206,7 +269,6 @@ jobs:
CARGO_BUILD_JOBS: ${{ (github.event_name == 'push' || github.event_name == 'workflow_dispatch') && '3' || '2' }} CARGO_BUILD_JOBS: ${{ (github.event_name == 'push' || github.event_name == 'workflow_dispatch') && '3' || '2' }}
run: | run: |
mkdir -p artifacts/test-and-lint mkdir -p artifacts/test-and-lint
rm -f target/nextest/ci/junit.xml
./scripts/ci/resource_sampler.sh start nextest ./scripts/ci/resource_sampler.sh start nextest
trap './scripts/ci/resource_sampler.sh stop' EXIT trap './scripts/ci/resource_sampler.sh stop' EXIT
set +e set +e
@@ -215,12 +277,6 @@ jobs:
--status-level all --final-status-level all \ --status-level all --final-status-level all \
2>&1 | tee artifacts/test-and-lint/nextest.log 2>&1 | tee artifacts/test-and-lint/nextest.log
status=${PIPESTATUS[0]} status=${PIPESTATUS[0]}
if [[ "${status}" -eq 0 ]]; then
cargo nextest list --profile ci --all --exclude e2e_test --message-format json \
> artifacts/test-and-lint/core-test-listing.json \
&& python3 scripts/check_test_wiring.py --check-core artifacts/test-and-lint/core-test-listing.json \
&& test -s target/nextest/ci/junit.xml || status=$?
fi
{ {
echo "command=cargo nextest run --profile ci --all --exclude e2e_test" echo "command=cargo nextest run --profile ci --all --exclude e2e_test"
echo "exit_status=${status}" echo "exit_status=${status}"
+5 -24
View File
@@ -19,9 +19,7 @@ on:
paths: paths:
- ".github/workflows/e2e-upgrade.yml" - ".github/workflows/e2e-upgrade.yml"
- "crates/e2e_test/src/common.rs" - "crates/e2e_test/src/common.rs"
- "crates/e2e_test/src/fake_s3_target/**"
- "crates/e2e_test/src/lib.rs" - "crates/e2e_test/src/lib.rs"
- "crates/e2e_test/src/replication_extension_test.rs"
- "crates/e2e_test/src/upgrade_compatibility_test.rs" - "crates/e2e_test/src/upgrade_compatibility_test.rs"
- "crates/ecstore/**" - "crates/ecstore/**"
- "crates/filemeta/**" - "crates/filemeta/**"
@@ -46,9 +44,9 @@ concurrency:
env: env:
CARGO_TERM_COLOR: always CARGO_TERM_COLOR: always
RUST_BACKTRACE: 1 RUST_BACKTRACE: 1
UPGRADE_SOURCE_VERSION: 1.0.0-rc.5 UPGRADE_SOURCE_VERSION: 1.0.0-rc.2
UPGRADE_SOURCE_ASSET: rustfs-linux-x86_64-gnu-v1.0.0-rc.5.zip UPGRADE_SOURCE_ASSET: rustfs-linux-x86_64-gnu-v1.0.0-rc.2.zip
UPGRADE_SOURCE_SHA256: 3ee8df71e8edcfada533be452c4135868f697bc515460ae97b027313eade7a3d UPGRADE_SOURCE_SHA256: 7c789386bf85278f865b8e0d359bf4edb84d5aa408cc3fa54a18c25ca74cd6e7
jobs: jobs:
upgrade: upgrade:
@@ -57,31 +55,14 @@ jobs:
fail-fast: false fail-fast: false
matrix: matrix:
include: include:
# The two `_from_rc2_` tests keep their names: they assert - name: Direct upgrade from rc.2
# release-independent object contracts and pass unchanged against the
# newer pinned source, so renaming them would only churn history and
# the CI required-check names. UPGRADE_SOURCE_VERSION above is the
# single source of truth for which release they actually run against.
- name: Direct upgrade from the previous release
cache_key: e2e-direct-upgrade cache_key: e2e-direct-upgrade
test: direct_upgrade_from_rc2_preserves_object_contracts test: direct_upgrade_from_rc2_preserves_object_contracts
artifact: direct-upgrade artifact: direct-upgrade
- name: Mixed-version rolling upgrade from the previous release - name: Mixed-version rolling upgrade from rc.2
cache_key: e2e-mixed-version-upgrade cache_key: e2e-mixed-version-upgrade
test: rolling_upgrade_from_rc2_preserves_mixed_version_contracts test: rolling_upgrade_from_rc2_preserves_mixed_version_contracts
artifact: mixed-version-upgrade artifact: mixed-version-upgrade
- name: Bucket configuration survives the upgrade
cache_key: e2e-bucket-config-upgrade
test: direct_upgrade_from_previous_release_preserves_bucket_configuration
artifact: bucket-config-upgrade
- name: Rollback reads current bucket metadata
cache_key: e2e-bucket-config-rollback
test: rollback_to_previous_release_reads_current_bucket_metadata
artifact: bucket-config-rollback
- name: ODM configuration recovery after rc.5 rollback
cache_key: e2e-odm-config-rollback
test: rc5_rollback_requires_restoring_odm_configuration
artifact: odm-config-rollback
runs-on: ubuntu-latest runs-on: ubuntu-latest
timeout-minutes: 60 timeout-minutes: 60
env: env:
+8 -51
View File
@@ -166,9 +166,8 @@ jobs:
# e.g. https://dl.rustfs.com/artifacts/rustfs/packages/nightly/... . # e.g. https://dl.rustfs.com/artifacts/rustfs/packages/nightly/... .
# Skipped when the R2 secrets are not configured (artifact-only mode). # Skipped when the R2 secrets are not configured (artifact-only mode).
- name: Upload DEB to Cloudflare R2 - name: Upload DEB to Cloudflare R2
id: publish if: env.R2_ACCESS_KEY_ID != ''
env: env:
DEB_FILE: ${{ steps.deb.outputs.deb_file }}
R2_ACCESS_KEY_ID: ${{ secrets.R2_ACCESS_KEY_ID }} R2_ACCESS_KEY_ID: ${{ secrets.R2_ACCESS_KEY_ID }}
R2_SECRET_ACCESS_KEY: ${{ secrets.R2_SECRET_ACCESS_KEY }} R2_SECRET_ACCESS_KEY: ${{ secrets.R2_SECRET_ACCESS_KEY }}
R2_ENDPOINT: ${{ secrets.R2_ENDPOINT }} R2_ENDPOINT: ${{ secrets.R2_ENDPOINT }}
@@ -183,70 +182,28 @@ jobs:
exit 0 exit 0
fi fi
if ! command -v aws >/dev/null 2>&1; then
sudo apt-get update && sudo apt-get install -y -qq awscli
fi
export AWS_ACCESS_KEY_ID="$R2_ACCESS_KEY_ID" export AWS_ACCESS_KEY_ID="$R2_ACCESS_KEY_ID"
export AWS_SECRET_ACCESS_KEY="$R2_SECRET_ACCESS_KEY" export AWS_SECRET_ACCESS_KEY="$R2_SECRET_ACCESS_KEY"
export AWS_DEFAULT_REGION="auto" export AWS_DEFAULT_REGION="auto"
SOURCE_SHA="$(git rev-parse HEAD)" DEB_FILE="${{ steps.deb.outputs.deb_file }}"
if [[ "${SOURCE_SHA}" != "${GITHUB_SHA}" ]]; then
echo "Checkout SHA does not match the nightly build run" >&2
exit 1
fi
DEB_SHA256="$(sha256sum "${DEB_FILE}" | cut -d ' ' -f 1)"
CANDIDATE_KEY="artifacts/rustfs/packages/nightly/runs/${GITHUB_RUN_ID}/${GITHUB_RUN_ATTEMPT}/${DEB_SHA256}/rustfs.deb"
CANDIDATE_URL="https://dl.rustfs.com/${CANDIDATE_KEY}"
# Old AWS CLI models lack conditional PutObject support. Never fall
# back to an overwriting upload for a candidate.
AWS_CLI=aws
if ! "${AWS_CLI}" s3api put-object --generate-cli-skeleton input | jq -e 'has("IfNoneMatch")' >/dev/null; then
sudo apt-get update
sudo apt-get install -y -qq python3-venv
AWS_CLI_DIR="$(mktemp -d "${RUNNER_TEMP}/nightly-awscli.XXXXXX")"
trap 'rm -rf "${AWS_CLI_DIR}"' EXIT
python3 -m venv "${AWS_CLI_DIR}"
"${AWS_CLI_DIR}/bin/python" -m pip install --disable-pip-version-check 'awscli==1.44.79'
AWS_CLI="${AWS_CLI_DIR}/bin/aws"
fi
"${AWS_CLI}" s3api put-object --generate-cli-skeleton input | jq -e 'has("IfNoneMatch")' >/dev/null
"${AWS_CLI}" --version
"${AWS_CLI}" s3api put-object --bucket "${R2_BUCKET}" --key "${CANDIDATE_KEY}" \
--body "${DEB_FILE}" --if-none-match '*' --endpoint-url "${R2_ENDPOINT}"
PUBLISHED_SHA256="$(curl -fsSL --retry 3 --connect-timeout 15 --max-time 300 "${CANDIDATE_URL}" | sha256sum | cut -d ' ' -f 1)"
if [[ "${PUBLISHED_SHA256}" != "${DEB_SHA256}" ]]; then
echo "Published candidate checksum does not match the built package" >&2
exit 1
fi
R2_PREFIX="s3://${R2_BUCKET}/artifacts/rustfs/packages/nightly/" R2_PREFIX="s3://${R2_BUCKET}/artifacts/rustfs/packages/nightly/"
echo "📤 Uploading ${DEB_FILE} to ${R2_PREFIX}" echo "📤 Uploading ${DEB_FILE} to ${R2_PREFIX}"
"${AWS_CLI}" s3 cp "${DEB_FILE}" "${R2_PREFIX}" --endpoint-url "$R2_ENDPOINT" --only-show-errors aws s3 cp "${DEB_FILE}" "${R2_PREFIX}" --endpoint-url "$R2_ENDPOINT" --only-show-errors
# Stable "latest" alias so tests can fetch the newest nightly # Stable "latest" alias so tests can fetch the newest nightly
# without knowing today's date. # without knowing today's date.
echo "📤 Uploading latest alias" echo "📤 Uploading latest alias"
"${AWS_CLI}" s3 cp "${DEB_FILE}" "${R2_PREFIX}rustfs-nightly-latest.deb" \ aws s3 cp "${DEB_FILE}" "${R2_PREFIX}rustfs-nightly-latest.deb" \
--endpoint-url "$R2_ENDPOINT" --only-show-errors --endpoint-url "$R2_ENDPOINT" --only-show-errors
echo "✅ R2 upload complete" echo "✅ R2 upload complete"
CANDIDATE_FILE="${RUNNER_TEMP}/nightly-candidate-${GITHUB_RUN_ID}-${GITHUB_RUN_ATTEMPT}.json"
jq -n --arg source_sha "${SOURCE_SHA}" \
--argjson build_run_id "${GITHUB_RUN_ID}" --argjson build_run_attempt "${GITHUB_RUN_ATTEMPT}" \
--arg package_url "${CANDIDATE_URL}" --arg package_sha256 "${DEB_SHA256}" \
'{schema: 1, source_sha: $source_sha, build_run_id: $build_run_id, build_run_attempt: $build_run_attempt, package_url: $package_url, package_sha256: $package_sha256}' \
> "${CANDIDATE_FILE}"
echo "candidate_file=${CANDIDATE_FILE}" >> "${GITHUB_OUTPUT}"
- name: Upload nightly candidate manifest
if: ${{ steps.publish.outputs.candidate_file != '' }}
uses: actions/upload-artifact@b7c566a772e6b6bfb58ed0dc250532a479d7789f # v6
with:
name: nightly-candidate-${{ github.run_id }}-${{ github.run_attempt }}
path: ${{ steps.publish.outputs.candidate_file }}
if-no-files-found: error
# Live-Vault lane for the rustfs-kms suite (rustfs/backlog#1774). # Live-Vault lane for the rustfs-kms suite (rustfs/backlog#1774).
# #
# RUSTFS_KMS_VAULT_TOKEN is the single switch that adds the Vault KV2 and # RUSTFS_KMS_VAULT_TOKEN is the single switch that adds the Vault KV2 and
+15 -2
View File
@@ -14,8 +14,8 @@
# Functional chain driver: runs the ten functional suites in a fixed order # Functional chain driver: runs the ten functional suites in a fixed order
# (upgrade -> s3 -> kms -> tier -> storage -> heal -> pool -> security -> # (upgrade -> s3 -> kms -> tier -> storage -> heal -> pool -> security ->
# replication -> performance). Each suite attempts the next handoff even # replication, with performance on its own runner in parallel) and guarantees
# when its tests fail. # the chain keeps moving even when individual suites fail.
# #
# Each suite workflow can still be dispatched standalone (workflow_dispatch); # Each suite workflow can still be dispatched standalone (workflow_dispatch);
# only chain-triggered runs forward to the next suite via repository_dispatch, # only chain-triggered runs forward to the next suite via repository_dispatch,
@@ -59,3 +59,16 @@ jobs:
gh api --method POST repos/rustfs/rustfs/dispatches \ gh api --method POST repos/rustfs/rustfs/dispatches \
-f event_type='rustfs-chain-upgrade' \ -f event_type='rustfs-chain-upgrade' \
-F 'client_payload[from_suite]=nightly-build' -F 'client_payload[from_suite]=nightly-build'
- name: Dispatch performance suite (parallel, own runner)
env:
GH_TOKEN: ${{ secrets.PF_TESTING_GH_TOKEN }}
run: |
set -euo pipefail
if [ -z "${GH_TOKEN:-}" ]; then
echo "PF_TESTING_GH_TOKEN is not configured; cannot dispatch performance" >&2
exit 1
fi
gh api --method POST repos/rustfs/rustfs/dispatches \
-f event_type='rustfs-chain-performance' \
-F 'client_payload[from_suite]=nightly-build'
+37 -70
View File
@@ -54,26 +54,14 @@ env:
jobs: jobs:
heal-test: heal-test:
runs-on: smoke-testing runs-on: smoke-testing
# Requirement: a failing suite must not fail the workflow; failures
# are filed to rustfs/backlog and the chain continues.
continue-on-error: true
timeout-minutes: 480 timeout-minutes: 480
# Standalone manual run, or one link of the nightly functional chain # Standalone manual run, or one link of the nightly functional chain
# (storage -> heal -> pool). Pool expansion no longer re-runs heal. # (storage -> heal -> pool). Pool expansion no longer re-runs heal.
if: ${{ github.event_name == 'workflow_dispatch' || github.event_name == 'repository_dispatch' }} if: ${{ github.event_name == 'workflow_dispatch' || github.event_name == 'repository_dispatch' }}
steps: steps:
- name: Initialize functional evidence
id: evidence
run: |
set -euo pipefail
umask 077
FUNCTIONAL_ARTIFACTS_DIR="${RUNNER_TEMP}/rustfs-heal-${GITHUB_RUN_ID}-${GITHUB_RUN_ATTEMPT}"
mkdir -- "${FUNCTIONAL_ARTIFACTS_DIR}" "${FUNCTIONAL_ARTIFACTS_DIR}-scratch"
{
printf 'FUNCTIONAL_ARTIFACTS_DIR=%s\n' "${FUNCTIONAL_ARTIFACTS_DIR}"
printf 'LOG_FILE=%s/suite.log\n' "${FUNCTIONAL_ARTIFACTS_DIR}"
printf 'RUSTFS_WARP_LOG_FILE=%s/warp.log\n' "${FUNCTIONAL_ARTIFACTS_DIR}"
printf 'REPORT_FILE=%s/report.md\n' "${FUNCTIONAL_ARTIFACTS_DIR}"
printf 'TMPDIR=%s-scratch\n' "${FUNCTIONAL_ARTIFACTS_DIR}"
} >> "${GITHUB_ENV}"
# auto-testing is private: clone it with the dedicated PF token (not # auto-testing is private: clone it with the dedicated PF token (not
# GITHUB_TOKEN) and retry transient GitHub/network failures. # GITHUB_TOKEN) and retry transient GitHub/network failures.
- name: Checkout auto-testing scripts (with retry) - name: Checkout auto-testing scripts (with retry)
@@ -129,7 +117,7 @@ jobs:
else else
ARGS+=(--package-url "${{ env.RUSTFS_NIGHTLY_PACKAGE_URL }}") ARGS+=(--package-url "${{ env.RUSTFS_NIGHTLY_PACKAGE_URL }}")
fi fi
./auto-testing/rustfs_heal_test.sh "${ARGS[@]}" --log-file "${LOG_FILE}" ./auto-testing/rustfs_heal_test.sh "${ARGS[@]}"
- name: Preflight checks - name: Preflight checks
run: | run: |
@@ -139,7 +127,7 @@ jobs:
else else
ARGS+=(--package-url "${{ env.RUSTFS_NIGHTLY_PACKAGE_URL }}") ARGS+=(--package-url "${{ env.RUSTFS_NIGHTLY_PACKAGE_URL }}")
fi fi
./auto-testing/rustfs_heal_test.sh "${ARGS[@]}" --log-file "${LOG_FILE}" ./auto-testing/rustfs_heal_test.sh "${ARGS[@]}"
- name: Run heal test (write -> outage -> heal -> verify) - name: Run heal test (write -> outage -> heal -> verify)
id: test id: test
@@ -149,10 +137,13 @@ jobs:
--endpoint "${{ env.RUSTFS_API_ENDPOINT }}" \ --endpoint "${{ env.RUSTFS_API_ENDPOINT }}" \
--stop-node-gb "${{ inputs.stop_node_gb || '15' }}" \ --stop-node-gb "${{ inputs.stop_node_gb || '15' }}" \
--warp-stop-gb "${{ inputs.warp_stop_gb || '40' }}" \ --warp-stop-gb "${{ inputs.warp_stop_gb || '40' }}" \
--log-file "${LOG_FILE}" --log-file /tmp/rustfs-heal-test.log
- name: Generate report - name: Generate report
if: ${{ always() && steps.evidence.outcome == 'success' }} if: always()
env:
LOG_FILE: /tmp/rustfs-heal-test.log
REPORT_FILE: /tmp/rustfs-heal-report.md
run: | run: |
set -euo pipefail set -euo pipefail
PACKAGE_URL='${{ inputs.package_url }}' PACKAGE_URL='${{ inputs.package_url }}'
@@ -161,9 +152,8 @@ jobs:
else else
PACKAGE_SOURCE="${RUSTFS_NIGHTLY_PACKAGE_URL}" PACKAGE_SOURCE="${RUSTFS_NIGHTLY_PACKAGE_URL}"
fi fi
STEPS_TABLE="${FUNCTIONAL_ARTIFACTS_DIR}/steps.md" STEPS_TABLE="/tmp/rustfs-heal-steps.md"
CASE_RESULT=success python3 - "${LOG_FILE}" "${STEPS_TABLE}" <<'PY'
python3 - "${LOG_FILE}" "${STEPS_TABLE}" <<'PY' || CASE_RESULT=failure
import re import re
import sys import sys
@@ -175,7 +165,6 @@ jobs:
steps = {} steps = {}
order = [] order = []
status_rank = {'SKIP': 0, 'PASS': 1, 'FAIL': 2}
version = None version = None
version_node = None version_node = None
verdict = None verdict = None
@@ -189,15 +178,14 @@ jobs:
n, desc, status = m.group(1), m.group(2), m.group(3) n, desc, status = m.group(1), m.group(2), m.group(3)
if n not in steps: if n not in steps:
order.append(n) order.append(n)
if n not in steps or status_rank[status] > status_rank[steps[n][1]]: steps[n] = (desc, status) # later lines win (fail after pass)
steps[n] = (desc, status)
continue continue
m = ver_re.match(line) m = ver_re.match(line)
if m: if m:
version, version_node = m.group(1), m.group(2) version, version_node = m.group(1), m.group(2)
continue continue
m = result_re.match(line) m = result_re.match(line)
if m and verdict != 'FAIL': if m:
verdict, verdict_detail = m.group(1), m.group(2) verdict, verdict_detail = m.group(1), m.group(2)
except FileNotFoundError: except FileNotFoundError:
pass pass
@@ -217,43 +205,30 @@ jobs:
out.write(f'| {n} | {desc} | {status} |\n') out.write(f'| {n} | {desc} | {status} |\n')
if not order: if not order:
out.write('| - | - | NOT RUN (no step result lines found) |\n') out.write('| - | - | NOT RUN (no step result lines found) |\n')
complete = set(steps) == {str(n) for n in range(1, 8)}
sys.exit(0 if complete and verdict != 'FAIL' and all(status == 'PASS' for _, status in steps.values()) else 1)
PY PY
RESULT=failure
if [ '${{ steps.test.outcome }}' = 'success' ] && [ "${CASE_RESULT}" = 'success' ]; then
RESULT=success
fi
{ {
echo "# RustFS heal test report" echo "# RustFS heal test report"
echo "" echo ""
echo "- Run: ${{ github.server_url }}/${{ github.repository }}/actions/runs/${{ github.run_id }}" echo "- Run: ${{ github.server_url }}/${{ github.repository }}/actions/runs/${{ github.run_id }}"
echo "- Attempt: ${GITHUB_RUN_ATTEMPT}"
echo "- Workflow Commit: ${GITHUB_SHA}"
echo "- Trigger: ${{ github.event_name }}" echo "- Trigger: ${{ github.event_name }}"
echo "- Package: ${PACKAGE_SOURCE}" echo "- Package: ${PACKAGE_SOURCE}"
echo "- Test Step Outcome: ${RESULT}" echo "- Test Step Outcome: ${{ steps.test.outcome }}"
echo "- Suite Step Outcome: ${{ steps.test.outcome }}"
echo "" echo ""
if [ "${RESULT}" = "success" ]; then cat "${STEPS_TABLE}" || true
cat "${STEPS_TABLE}" echo ""
echo "" echo "## Log tail"
echo "## Log tail" echo '```text'
echo '```text' tail -n 200 "${LOG_FILE}" || true
tail -n 200 "${LOG_FILE}" echo '```'
echo '```'
else
echo "The suite or evidence validation failed. See this run's artifact for partial step results and suite.log."
fi
} | tee "${REPORT_FILE}" } | tee "${REPORT_FILE}"
cat "${REPORT_FILE}" >> "${GITHUB_STEP_SUMMARY}" cat "${REPORT_FILE}" >> "${GITHUB_STEP_SUMMARY}"
[ "${RESULT}" = "success" ]
- name: Upload functional report to dashboard - name: Upload functional report to dashboard
if: ${{ always() && steps.evidence.outcome == 'success' }} if: always()
continue-on-error: true continue-on-error: true
env: env:
GH_TOKEN: ${{ env.PF_TESTING_GH_TOKEN }} GH_TOKEN: ${{ env.PF_TESTING_GH_TOKEN }}
REPORT_FILE: /tmp/rustfs-heal-report.md
SUITE: heal SUITE: heal
run: | run: |
set -euo pipefail set -euo pipefail
@@ -263,32 +238,28 @@ jobs:
fi fi
DATE="$(date -u +%Y-%m-%d)" DATE="$(date -u +%Y-%m-%d)"
REPORT_PATH="functional-reports/${SUITE}/${DATE}.md" REPORT_PATH="functional-reports/${SUITE}/${DATE}.md"
# Base64-encode the report into a temp file and feed it to jq via CONTENT="$(python3 -c 'import base64,sys;print(base64.b64encode(open(sys.argv[1],"rb").read()).decode())' "${REPORT_FILE}")"
# --rawfile: large reports (e.g. pool) exceed the OS argv limit and
# make `jq --arg content "${CONTENT}"` fail with "Argument list too long".
B64_FILE="$(mktemp)"
python3 -c 'import base64,sys;print(base64.b64encode(open(sys.argv[1],"rb").read()).decode())' "${REPORT_FILE}" > "${B64_FILE}"
SHA="$(gh api "repos/rustfs/dashboard/contents/${REPORT_PATH}" -q '.sha' 2>/dev/null || true)" SHA="$(gh api "repos/rustfs/dashboard/contents/${REPORT_PATH}" -q '.sha' 2>/dev/null || true)"
if [ -n "${SHA}" ]; then if [ -n "${SHA}" ]; then
jq -n --arg msg "report(${SUITE}): ${DATE}" --rawfile content "${B64_FILE}" --arg sha "${SHA}" \ jq -n --arg msg "report(${SUITE}): ${DATE}" --arg content "${CONTENT}" --arg sha "${SHA}" \
'{message:$msg, content:($content|rtrimstr("\n")), sha:$sha}' \ '{message:$msg, content:$content, sha:$sha}' \
| gh api --method PUT "repos/rustfs/dashboard/contents/${REPORT_PATH}" --input - >/dev/null | gh api --method PUT "repos/rustfs/dashboard/contents/${REPORT_PATH}" --input - >/dev/null
else else
jq -n --arg msg "report(${SUITE}): ${DATE}" --rawfile content "${B64_FILE}" \ jq -n --arg msg "report(${SUITE}): ${DATE}" --arg content "${CONTENT}" \
'{message:$msg, content:($content|rtrimstr("\n"))}' \ '{message:$msg, content:$content}' \
| gh api --method PUT "repos/rustfs/dashboard/contents/${REPORT_PATH}" --input - >/dev/null | gh api --method PUT "repos/rustfs/dashboard/contents/${REPORT_PATH}" --input - >/dev/null
fi fi
rm -f "${B64_FILE}"
- name: File failure issue in rustfs/backlog - name: File failure issue in rustfs/backlog
if: ${{ always() && (failure() || steps.test.outcome == 'failure' || steps.test.outcome == 'cancelled') }} if: ${{ always() && (failure() || steps.test.outcome == 'failure' || steps.test.outcome == 'cancelled') }}
continue-on-error: true continue-on-error: true
env: env:
GH_TOKEN: ${{ secrets.PF_TESTING_GH_TOKEN }} GH_TOKEN: ${{ secrets.PF_TESTING_GH_TOKEN }}
EVIDENCE_OUTCOME: ${{ steps.evidence.outcome }}
SUITE: 'heal' SUITE: 'heal'
SUITE_LABEL: 'Heal' SUITE_LABEL: 'Heal'
RUN_URL: ${{ github.server_url }}/${{ github.repository }}/actions/runs/${{ github.run_id }} RUN_URL: ${{ github.server_url }}/${{ github.repository }}/actions/runs/${{ github.run_id }}
REPORT_FILE: '/tmp/rustfs-heal-report.md'
LOG_FILE: '/tmp/rustfs-heal-test.log'
run: | run: |
set -euo pipefail set -euo pipefail
if [ -z "${GH_TOKEN:-}" ]; then if [ -z "${GH_TOKEN:-}" ]; then
@@ -316,16 +287,14 @@ jobs:
echo "" echo ""
echo "- Suite: \`${SUITE}\`" echo "- Suite: \`${SUITE}\`"
echo "- Run: ${RUN_URL}" echo "- Run: ${RUN_URL}"
echo "- Attempt: ${GITHUB_RUN_ATTEMPT}"
echo "- Workflow Commit: ${GITHUB_SHA}"
echo "- Trigger: ${GITHUB_EVENT_NAME}" echo "- Trigger: ${GITHUB_EVENT_NAME}"
echo "- Date: $(date -u +%Y-%m-%d)" echo "- Date: $(date -u +%Y-%m-%d)"
echo "" echo ""
echo "## Report (errors and symptoms)" echo "## Report (errors and symptoms)"
echo "" echo ""
if [ "${EVIDENCE_OUTCOME}" = "success" ] && [ -s "${REPORT_FILE}" ]; then if [ -s "${REPORT_FILE}" ]; then
redact < "${REPORT_FILE}" redact < "${REPORT_FILE}"
elif [ "${EVIDENCE_OUTCOME}" = "success" ] && [ -s "${LOG_FILE:-}" ]; then elif [ -s "${LOG_FILE:-}" ]; then
echo "(report file missing; log tail below)" echo "(report file missing; log tail below)"
echo "" echo ""
tail -n 200 "${LOG_FILE}" | redact tail -n 200 "${LOG_FILE}" | redact
@@ -341,16 +310,14 @@ jobs:
echo "filed backlog issue for suite ${SUITE}" echo "filed backlog issue for suite ${SUITE}"
- name: Upload test logs - name: Upload test logs
if: ${{ always() && steps.evidence.outcome == 'success' }} if: always()
uses: actions/upload-artifact@b7c566a772e6b6bfb58ed0dc250532a479d7789f # v6 uses: actions/upload-artifact@b7c566a772e6b6bfb58ed0dc250532a479d7789f # v6
with: with:
name: rustfs-heal-test-${{ github.run_id }}-${{ github.run_attempt }} name: rustfs-heal-test-${{ github.run_id }}
path: | path: |
${{ env.FUNCTIONAL_ARTIFACTS_DIR }}/report.md /tmp/rustfs-heal-test*.log
${{ env.FUNCTIONAL_ARTIFACTS_DIR }}/suite.log /tmp/rustfs-warp.*.log
${{ env.FUNCTIONAL_ARTIFACTS_DIR }}/warp.log if-no-files-found: warn
${{ env.FUNCTIONAL_ARTIFACTS_DIR }}/steps.md
if-no-files-found: error
- name: Cleanup environment (after) - name: Cleanup environment (after)
if: ${{ always() && inputs.cleanup_after != 'false' }} if: ${{ always() && inputs.cleanup_after != 'false' }}
+81 -64
View File
@@ -49,28 +49,10 @@ env:
jobs: jobs:
kms-test: kms-test:
runs-on: smoke-testing runs-on: smoke-testing
continue-on-error: true
timeout-minutes: 420 timeout-minutes: 420
if: ${{ github.event_name == 'workflow_dispatch' || github.event_name == 'repository_dispatch' }} if: ${{ github.event_name == 'workflow_dispatch' || github.event_name == 'repository_dispatch' }}
steps: steps:
- name: Checkout repository (for report parser)
uses: actions/checkout@9c091bb21b7c1c1d1991bb908d89e4e9dddfe3e0 # v7
with:
persist-credentials: false
- name: Initialize functional evidence
id: evidence
run: |
set -euo pipefail
umask 077
FUNCTIONAL_ARTIFACTS_DIR="${RUNNER_TEMP}/rustfs-kms-${GITHUB_RUN_ID}-${GITHUB_RUN_ATTEMPT}"
mkdir -- "${FUNCTIONAL_ARTIFACTS_DIR}" "${FUNCTIONAL_ARTIFACTS_DIR}-scratch"
{
printf 'FUNCTIONAL_ARTIFACTS_DIR=%s\n' "${FUNCTIONAL_ARTIFACTS_DIR}"
printf 'LOG_FILE=%s/suite.log\n' "${FUNCTIONAL_ARTIFACTS_DIR}"
printf 'REPORT_FILE=%s/report.md\n' "${FUNCTIONAL_ARTIFACTS_DIR}"
printf 'TMPDIR=%s-scratch\n' "${FUNCTIONAL_ARTIFACTS_DIR}"
} >> "${GITHUB_ENV}"
# auto-testing is private: clone it with the dedicated PF token (not # auto-testing is private: clone it with the dedicated PF token (not
# GITHUB_TOKEN) and retry transient GitHub/network failures. # GITHUB_TOKEN) and retry transient GitHub/network failures.
- name: Checkout auto-testing scripts (with retry) - name: Checkout auto-testing scripts (with retry)
@@ -127,6 +109,9 @@ jobs:
- name: Run KMS suite - name: Run KMS suite
id: test id: test
continue-on-error: true
env:
LOG_FILE: /tmp/rustfs-kms.log
run: | run: |
set -euo pipefail set -euo pipefail
chmod +x auto-testing/rustfs-kms-test.sh chmod +x auto-testing/rustfs-kms-test.sh
@@ -156,7 +141,10 @@ jobs:
./auto-testing/rustfs-kms-test.sh "${ARGS[@]}" ./auto-testing/rustfs-kms-test.sh "${ARGS[@]}"
- name: Generate report - name: Generate report
if: ${{ always() && steps.evidence.outcome == 'success' }} if: always()
env:
LOG_FILE: /tmp/rustfs-kms.log
REPORT_FILE: /tmp/rustfs-kms-report.md
run: | run: |
set -euo pipefail set -euo pipefail
PACKAGE_URL='${{ inputs.package_url }}' PACKAGE_URL='${{ inputs.package_url }}'
@@ -168,43 +156,79 @@ jobs:
else else
PACKAGE_SOURCE="${RUSTFS_NIGHTLY_PACKAGE_URL}" PACKAGE_SOURCE="${RUSTFS_NIGHTLY_PACKAGE_URL}"
fi fi
CASE_TABLE="${FUNCTIONAL_ARTIFACTS_DIR}/cases.md" CASE_TABLE="/tmp/rustfs-kms-cases.md"
CASE_RESULT=success python3 - "${LOG_FILE}" "${CASE_TABLE}" <<'PY'
python3 scripts/functional_case_report.py "${LOG_FILE}" "${CASE_TABLE}" || CASE_RESULT=failure import re
RESULT=failure import sys
if [ '${{ steps.test.outcome }}' = 'success' ] && [ "${CASE_RESULT}" = 'success' ]; then
RESULT=success log_file, out_file = sys.argv[1], sys.argv[2]
fi ansi = re.compile(r'\x1b\[[0-9;]*m')
start_re = re.compile(r'^---\s+([A-Z]+-[0-9]+)\s+(.+?)\s+---$')
done_re = re.compile(r'^\[(PASS|FAIL|UNSUPPORTED)\]\s+([A-Z]+-[0-9]+)\b')
rows = []
index = {}
try:
with open(log_file, 'r', encoding='utf-8', errors='replace') as fh:
for raw in fh:
line = ansi.sub('', raw).strip()
m = start_re.match(line)
if m:
case_id, name = m.group(1), m.group(2)
if case_id not in index:
index[case_id] = len(rows)
rows.append([case_id, name, 'RUNNING'])
continue
m = done_re.match(line)
if m:
status, case_id = m.group(1), m.group(2)
if case_id in index:
rows[index[case_id]][2] = status
else:
rows.append([case_id, case_id, status])
index[case_id] = len(rows) - 1
except FileNotFoundError:
rows = []
counts = {'PASS': 0, 'FAIL': 0, 'UNSUPPORTED': 0, 'RUNNING': 0}
for _, _, status in rows:
counts[status] = counts.get(status, 0) + 1
with open(out_file, 'w', encoding='utf-8') as out:
out.write('## Case Summary\n\n')
out.write(f"- Total: {len(rows)}\\n")
out.write(f"- PASS: {counts.get('PASS', 0)}\\n")
out.write(f"- FAIL: {counts.get('FAIL', 0)}\\n")
out.write(f"- UNSUPPORTED: {counts.get('UNSUPPORTED', 0)}\\n")
out.write('\\n')
out.write('| Case | Name | Status |\\n')
out.write('| --- | --- | --- |\\n')
for case_id, name, status in rows:
out.write(f'| {case_id} | {name} | {status} |\\n')
PY
{ {
echo "# RustFS KMS test report" echo "# RustFS KMS test report"
echo "" echo ""
echo "- Run: ${{ github.server_url }}/${{ github.repository }}/actions/runs/${{ github.run_id }}" echo "- Run: ${{ github.server_url }}/${{ github.repository }}/actions/runs/${{ github.run_id }}"
echo "- Attempt: ${GITHUB_RUN_ATTEMPT}"
echo "- Workflow Commit: ${GITHUB_SHA}"
echo "- Trigger: ${{ github.event_name }}" echo "- Trigger: ${{ github.event_name }}"
echo "- Package: ${PACKAGE_SOURCE}" echo "- Package: ${PACKAGE_SOURCE}"
echo "- Test Step Outcome: ${RESULT}" echo "- Test Step Outcome: ${{ steps.test.outcome }}"
echo "- Suite Step Outcome: ${{ steps.test.outcome }}"
echo "" echo ""
if [ "${RESULT}" = "success" ]; then cat "${CASE_TABLE}" || true
cat "${CASE_TABLE}" echo ""
echo "" echo "## Log tail"
echo "## Log tail" echo '```text'
echo '```text' tail -n 200 "${LOG_FILE}" || true
tail -n 200 "${LOG_FILE}" echo '```'
echo '```'
else
echo "The suite or evidence validation failed. See this run's artifact for partial case results and suite.log."
fi
} | tee "${REPORT_FILE}" } | tee "${REPORT_FILE}"
cat "${REPORT_FILE}" >> "${GITHUB_STEP_SUMMARY}" cat "${REPORT_FILE}" >> "${GITHUB_STEP_SUMMARY}"
[ "${RESULT}" = "success" ]
- name: Upload functional report to dashboard - name: Upload functional report to dashboard
if: ${{ always() && steps.evidence.outcome == 'success' }} if: always()
continue-on-error: true continue-on-error: true
env: env:
GH_TOKEN: ${{ env.PF_TESTING_GH_TOKEN }} GH_TOKEN: ${{ env.PF_TESTING_GH_TOKEN }}
REPORT_FILE: /tmp/rustfs-kms-report.md
SUITE: kms SUITE: kms
run: | run: |
set -euo pipefail set -euo pipefail
@@ -214,32 +238,28 @@ jobs:
fi fi
DATE="$(date -u +%Y-%m-%d)" DATE="$(date -u +%Y-%m-%d)"
REPORT_PATH="functional-reports/${SUITE}/${DATE}.md" REPORT_PATH="functional-reports/${SUITE}/${DATE}.md"
# Base64-encode the report into a temp file and feed it to jq via CONTENT="$(python3 -c 'import base64,sys;print(base64.b64encode(open(sys.argv[1],"rb").read()).decode())' "${REPORT_FILE}")"
# --rawfile: large reports (e.g. pool) exceed the OS argv limit and
# make `jq --arg content "${CONTENT}"` fail with "Argument list too long".
B64_FILE="$(mktemp)"
python3 -c 'import base64,sys;print(base64.b64encode(open(sys.argv[1],"rb").read()).decode())' "${REPORT_FILE}" > "${B64_FILE}"
SHA="$(gh api "repos/rustfs/dashboard/contents/${REPORT_PATH}" -q '.sha' 2>/dev/null || true)" SHA="$(gh api "repos/rustfs/dashboard/contents/${REPORT_PATH}" -q '.sha' 2>/dev/null || true)"
if [ -n "${SHA}" ]; then if [ -n "${SHA}" ]; then
jq -n --arg msg "report(${SUITE}): ${DATE}" --rawfile content "${B64_FILE}" --arg sha "${SHA}" \ jq -n --arg msg "report(${SUITE}): ${DATE}" --arg content "${CONTENT}" --arg sha "${SHA}" \
'{message:$msg, content:($content|rtrimstr("\n")), sha:$sha}' \ '{message:$msg, content:$content, sha:$sha}' \
| gh api --method PUT "repos/rustfs/dashboard/contents/${REPORT_PATH}" --input - >/dev/null | gh api --method PUT "repos/rustfs/dashboard/contents/${REPORT_PATH}" --input - >/dev/null
else else
jq -n --arg msg "report(${SUITE}): ${DATE}" --rawfile content "${B64_FILE}" \ jq -n --arg msg "report(${SUITE}): ${DATE}" --arg content "${CONTENT}" \
'{message:$msg, content:($content|rtrimstr("\n"))}' \ '{message:$msg, content:$content}' \
| gh api --method PUT "repos/rustfs/dashboard/contents/${REPORT_PATH}" --input - >/dev/null | gh api --method PUT "repos/rustfs/dashboard/contents/${REPORT_PATH}" --input - >/dev/null
fi fi
rm -f "${B64_FILE}"
- name: File failure issue in rustfs/backlog - name: File failure issue in rustfs/backlog
if: ${{ always() && (failure() || steps.test.outcome == 'failure' || steps.test.outcome == 'cancelled') }} if: ${{ always() && (failure() || steps.test.outcome == 'failure' || steps.test.outcome == 'cancelled') }}
continue-on-error: true continue-on-error: true
env: env:
GH_TOKEN: ${{ secrets.PF_TESTING_GH_TOKEN }} GH_TOKEN: ${{ secrets.PF_TESTING_GH_TOKEN }}
EVIDENCE_OUTCOME: ${{ steps.evidence.outcome }}
SUITE: 'kms' SUITE: 'kms'
SUITE_LABEL: 'KMS' SUITE_LABEL: 'KMS'
RUN_URL: ${{ github.server_url }}/${{ github.repository }}/actions/runs/${{ github.run_id }} RUN_URL: ${{ github.server_url }}/${{ github.repository }}/actions/runs/${{ github.run_id }}
REPORT_FILE: '/tmp/rustfs-kms-report.md'
LOG_FILE: '/tmp/rustfs-kms.log'
run: | run: |
set -euo pipefail set -euo pipefail
if [ -z "${GH_TOKEN:-}" ]; then if [ -z "${GH_TOKEN:-}" ]; then
@@ -267,16 +287,14 @@ jobs:
echo "" echo ""
echo "- Suite: \`${SUITE}\`" echo "- Suite: \`${SUITE}\`"
echo "- Run: ${RUN_URL}" echo "- Run: ${RUN_URL}"
echo "- Attempt: ${GITHUB_RUN_ATTEMPT}"
echo "- Workflow Commit: ${GITHUB_SHA}"
echo "- Trigger: ${GITHUB_EVENT_NAME}" echo "- Trigger: ${GITHUB_EVENT_NAME}"
echo "- Date: $(date -u +%Y-%m-%d)" echo "- Date: $(date -u +%Y-%m-%d)"
echo "" echo ""
echo "## Report (errors and symptoms)" echo "## Report (errors and symptoms)"
echo "" echo ""
if [ "${EVIDENCE_OUTCOME}" = "success" ] && [ -s "${REPORT_FILE}" ]; then if [ -s "${REPORT_FILE}" ]; then
redact < "${REPORT_FILE}" redact < "${REPORT_FILE}"
elif [ "${EVIDENCE_OUTCOME}" = "success" ] && [ -s "${LOG_FILE:-}" ]; then elif [ -s "${LOG_FILE:-}" ]; then
echo "(report file missing; log tail below)" echo "(report file missing; log tail below)"
echo "" echo ""
tail -n 200 "${LOG_FILE}" | redact tail -n 200 "${LOG_FILE}" | redact
@@ -292,15 +310,14 @@ jobs:
echo "filed backlog issue for suite ${SUITE}" echo "filed backlog issue for suite ${SUITE}"
- name: Upload report and logs - name: Upload report and logs
if: ${{ always() && steps.evidence.outcome == 'success' }} if: always()
uses: actions/upload-artifact@b7c566a772e6b6bfb58ed0dc250532a479d7789f # v6 uses: actions/upload-artifact@b7c566a772e6b6bfb58ed0dc250532a479d7789f # v6
with: with:
name: rustfs-kms-test-${{ github.run_id }}-${{ github.run_attempt }} name: rustfs-kms-test-${{ github.run_id }}
path: | path: |
${{ env.FUNCTIONAL_ARTIFACTS_DIR }}/report.md /tmp/rustfs-kms.log
${{ env.FUNCTIONAL_ARTIFACTS_DIR }}/suite.log /tmp/rustfs-kms-report.md
${{ env.FUNCTIONAL_ARTIFACTS_DIR }}/cases.md if-no-files-found: warn
if-no-files-found: error
- name: Cleanup environment (after) - name: Cleanup environment (after)
if: always() if: always()
+32 -48
View File
@@ -49,16 +49,17 @@ on:
type: boolean type: boolean
default: true default: true
repository_dispatch: repository_dispatch:
# Chain handoff: dispatched when the replication suite finishes. # Chain entry: dispatched by rustfs-functional-chain.yml (runs on its own
# pf-testing runner, in parallel with the shared-VM chain).
types: [rustfs-chain-performance] types: [rustfs-chain-performance]
permissions: permissions:
contents: read contents: read
# The default performance nodes overlap the other suites' remote VMs, even # Dedicated pf-testing runner/environment: own concurrency group so perf runs
# though the runner differs. Hold the shared lock through cleanup as well. # never block (or are blocked by) the pool-expansion / heal tests.
concurrency: concurrency:
group: rustfs-shared-functional-tests group: rustfs-performance-test
cancel-in-progress: false cancel-in-progress: false
defaults: defaults:
@@ -75,33 +76,22 @@ env:
# Package used by the nightly run (workflow_dispatch inputs are empty for # Package used by the nightly run (workflow_dispatch inputs are empty for
# workflow_run events), i.e. the latest nightly deb published by nightly-gnu.yml. # workflow_run events), i.e. the latest nightly deb published by nightly-gnu.yml.
RUSTFS_NIGHTLY_PACKAGE_URL: ${{ vars.RUSTFS_NIGHTLY_PACKAGE_URL || 'https://dl.rustfs.com/artifacts/rustfs/packages/nightly/rustfs-nightly-latest.deb' }} RUSTFS_NIGHTLY_PACKAGE_URL: ${{ vars.RUSTFS_NIGHTLY_PACKAGE_URL || 'https://dl.rustfs.com/artifacts/rustfs/packages/nightly/rustfs-nightly-latest.deb' }}
# Fixed benchmark result directory so later steps can read summary.md
RUSTFS_RESULT_DIR: /tmp/rustfs-perf-results
# Cross-repo token for uploading reports to rustfs/dashboard (set in repo settings) # Cross-repo token for uploading reports to rustfs/dashboard (set in repo settings)
PF_TESTING_GH_TOKEN: ${{ secrets.PF_TESTING_GH_TOKEN }} PF_TESTING_GH_TOKEN: ${{ secrets.PF_TESTING_GH_TOKEN }}
jobs: jobs:
performance-test: performance-test:
runs-on: pf-testing runs-on: pf-testing
# Requirement: a failing benchmark must not fail the workflow;
# failures are filed to rustfs/backlog.
continue-on-error: true
timeout-minutes: 900 timeout-minutes: 900
# Run on manual dispatch, or when the nightly build completed successfully. # Run on manual dispatch, or when the nightly build completed successfully.
# Skipped when nightly failed. # Skipped when nightly failed.
if: ${{ github.event_name == 'workflow_dispatch' || github.event_name == 'repository_dispatch' }} if: ${{ github.event_name == 'workflow_dispatch' || github.event_name == 'repository_dispatch' }}
steps: steps:
- name: Initialize functional evidence
id: evidence
run: |
set -euo pipefail
umask 077
FUNCTIONAL_ARTIFACTS_DIR="${RUNNER_TEMP}/rustfs-performance-${GITHUB_RUN_ID}-${GITHUB_RUN_ATTEMPT}"
mkdir -- "${FUNCTIONAL_ARTIFACTS_DIR}" "${FUNCTIONAL_ARTIFACTS_DIR}-scratch"
{
printf 'FUNCTIONAL_ARTIFACTS_DIR=%s\n' "${FUNCTIONAL_ARTIFACTS_DIR}"
printf 'LOG_FILE=%s/suite.log\n' "${FUNCTIONAL_ARTIFACTS_DIR}"
printf 'REPORT_FILE=%s/report.md\n' "${FUNCTIONAL_ARTIFACTS_DIR}"
printf 'TMPDIR=%s-scratch\n' "${FUNCTIONAL_ARTIFACTS_DIR}"
printf 'RUSTFS_RESULT_DIR=%s/results\n' "${FUNCTIONAL_ARTIFACTS_DIR}"
printf 'VERSION_FILE=%s/version.txt\n' "${FUNCTIONAL_ARTIFACTS_DIR}"
} >> "${GITHUB_ENV}"
# auto-testing is private: clone it with the dedicated PF token (not # auto-testing is private: clone it with the dedicated PF token (not
# GITHUB_TOKEN) and retry transient GitHub/network failures. # GITHUB_TOKEN) and retry transient GitHub/network failures.
- name: Checkout auto-testing scripts (with retry) - name: Checkout auto-testing scripts (with retry)
@@ -133,7 +123,7 @@ jobs:
if: ${{ inputs.cleanup_before != 'false' }} if: ${{ inputs.cleanup_before != 'false' }}
run: | run: |
chmod +x auto-testing/rustfs_performance_test.sh chmod +x auto-testing/rustfs_performance_test.sh
./auto-testing/rustfs_performance_test.sh --step 1 -y --log-file "${LOG_FILE:-/dev/null}" ./auto-testing/rustfs_performance_test.sh --step 1 -y
- name: Install RustFS package & start cluster (4x4) - name: Install RustFS package & start cluster (4x4)
run: | run: |
@@ -143,7 +133,7 @@ jobs:
else else
ARGS+=(--package-url "${{ env.RUSTFS_NIGHTLY_PACKAGE_URL }}") ARGS+=(--package-url "${{ env.RUSTFS_NIGHTLY_PACKAGE_URL }}")
fi fi
./auto-testing/rustfs_performance_test.sh "${ARGS[@]}" --log-file "${LOG_FILE}" ./auto-testing/rustfs_performance_test.sh "${ARGS[@]}"
- name: Preflight checks - name: Preflight checks
run: | run: |
@@ -153,7 +143,7 @@ jobs:
else else
ARGS+=(--package-url "${{ env.RUSTFS_NIGHTLY_PACKAGE_URL }}") ARGS+=(--package-url "${{ env.RUSTFS_NIGHTLY_PACKAGE_URL }}")
fi fi
./auto-testing/rustfs_performance_test.sh "${ARGS[@]}" --log-file "${LOG_FILE}" ./auto-testing/rustfs_performance_test.sh "${ARGS[@]}"
- name: Run benchmark (GET/PUT/MIXED) - name: Run benchmark (GET/PUT/MIXED)
id: benchmark id: benchmark
@@ -166,15 +156,17 @@ jobs:
--step 5 -y \ --step 5 -y \
--warp-duration "${{ inputs.warp_duration || '5m' }}" \ --warp-duration "${{ inputs.warp_duration || '5m' }}" \
--warp-concurrency "${{ inputs.warp_concurrency || '64' }}" \ --warp-concurrency "${{ inputs.warp_concurrency || '64' }}" \
--log-file "${LOG_FILE}" --log-file /tmp/rustfs-perf-test.log
- name: Analyze results - name: Analyze results
if: ${{ steps.benchmark.conclusion == 'success' }} if: ${{ steps.benchmark.conclusion == 'success' }}
run: | run: |
./auto-testing/rustfs_performance_test.sh --step 6 -y --log-file "${LOG_FILE:-/dev/null}" ./auto-testing/rustfs_performance_test.sh --step 6 -y
- name: Collect RustFS version info - name: Collect RustFS version info
if: ${{ steps.benchmark.conclusion == 'success' }} if: ${{ steps.benchmark.conclusion == 'success' }}
env:
VERSION_FILE: /tmp/rustfs-version.txt
run: | run: |
set -euo pipefail set -euo pipefail
read -r -a NODES <<< "${RUSTFS_NODES}" read -r -a NODES <<< "${RUSTFS_NODES}"
@@ -194,6 +186,7 @@ jobs:
env: env:
GH_TOKEN: ${{ env.PF_TESTING_GH_TOKEN }} GH_TOKEN: ${{ env.PF_TESTING_GH_TOKEN }}
RESULT_DIR: ${{ env.RUSTFS_RESULT_DIR }} RESULT_DIR: ${{ env.RUSTFS_RESULT_DIR }}
VERSION_FILE: /tmp/rustfs-version.txt
run: | run: |
set -euo pipefail set -euo pipefail
if [ -z "${GH_TOKEN:-}" ]; then if [ -z "${GH_TOKEN:-}" ]; then
@@ -201,7 +194,7 @@ jobs:
exit 0 exit 0
fi fi
SUMMARY="${RESULT_DIR}/summary.md" SUMMARY="${RESULT_DIR}/summary.md"
[ -s "${SUMMARY}" ] || { echo "summary.md not found at ${SUMMARY}"; exit 1; } [ -f "${SUMMARY}" ] || { echo "summary.md not found at ${SUMMARY}"; exit 1; }
DATE="$(date -u +%Y-%m-%d)" DATE="$(date -u +%Y-%m-%d)"
REPORT_PATH="reports/${DATE}.md" REPORT_PATH="reports/${DATE}.md"
{ {
@@ -209,8 +202,6 @@ jobs:
echo "" echo ""
echo "- **Date**: ${DATE}" echo "- **Date**: ${DATE}"
echo "- **Run**: ${{ github.server_url }}/${{ github.repository }}/actions/runs/${{ github.run_id }}" echo "- **Run**: ${{ github.server_url }}/${{ github.repository }}/actions/runs/${{ github.run_id }}"
echo "- **Attempt**: ${GITHUB_RUN_ATTEMPT}"
echo "- **Workflow Commit**: ${GITHUB_SHA}"
echo "- **Trigger**: ${{ github.event_name }}" echo "- **Trigger**: ${{ github.event_name }}"
echo "- **Package**: ${{ inputs.package_url || 'nightly (R2 latest)' }}" echo "- **Package**: ${{ inputs.package_url || 'nightly (R2 latest)' }}"
echo "" echo ""
@@ -220,8 +211,8 @@ jobs:
echo '```text' echo '```text'
cat "${VERSION_FILE}" cat "${VERSION_FILE}"
echo '```' echo '```'
} > "${REPORT_FILE}" } > /tmp/rustfs-perf-report.md
CONTENT="$(python3 -c 'import base64,sys; print(base64.b64encode(open(sys.argv[1],"rb").read()).decode())' "${REPORT_FILE}")" CONTENT="$(python3 -c 'import base64; print(base64.b64encode(open("/tmp/rustfs-perf-report.md","rb").read()).decode())')"
SHA="$(gh api "repos/rustfs/dashboard/contents/${REPORT_PATH}" -q '.sha' 2>/dev/null || true)" SHA="$(gh api "repos/rustfs/dashboard/contents/${REPORT_PATH}" -q '.sha' 2>/dev/null || true)"
if [ -n "${SHA}" ]; then if [ -n "${SHA}" ]; then
jq -n --arg msg "report: ${DATE}" --arg content "${CONTENT}" --arg sha "${SHA}" \ jq -n --arg msg "report: ${DATE}" --arg content "${CONTENT}" --arg sha "${SHA}" \
@@ -240,10 +231,11 @@ jobs:
continue-on-error: true continue-on-error: true
env: env:
GH_TOKEN: ${{ secrets.PF_TESTING_GH_TOKEN }} GH_TOKEN: ${{ secrets.PF_TESTING_GH_TOKEN }}
EVIDENCE_OUTCOME: ${{ steps.evidence.outcome }}
SUITE: 'performance' SUITE: 'performance'
SUITE_LABEL: 'Performance' SUITE_LABEL: 'Performance'
RUN_URL: ${{ github.server_url }}/${{ github.repository }}/actions/runs/${{ github.run_id }} RUN_URL: ${{ github.server_url }}/${{ github.repository }}/actions/runs/${{ github.run_id }}
REPORT_FILE: '/tmp/rustfs-perf-report.md'
LOG_FILE: '/tmp/rustfs-perf-test.log'
run: | run: |
set -euo pipefail set -euo pipefail
if [ -z "${GH_TOKEN:-}" ]; then if [ -z "${GH_TOKEN:-}" ]; then
@@ -271,16 +263,14 @@ jobs:
echo "" echo ""
echo "- Suite: \`${SUITE}\`" echo "- Suite: \`${SUITE}\`"
echo "- Run: ${RUN_URL}" echo "- Run: ${RUN_URL}"
echo "- Attempt: ${GITHUB_RUN_ATTEMPT}"
echo "- Workflow Commit: ${GITHUB_SHA}"
echo "- Trigger: ${GITHUB_EVENT_NAME}" echo "- Trigger: ${GITHUB_EVENT_NAME}"
echo "- Date: $(date -u +%Y-%m-%d)" echo "- Date: $(date -u +%Y-%m-%d)"
echo "" echo ""
echo "## Report (errors and symptoms)" echo "## Report (errors and symptoms)"
echo "" echo ""
if [ "${EVIDENCE_OUTCOME}" = "success" ] && [ -s "${REPORT_FILE}" ]; then if [ -s "${REPORT_FILE}" ]; then
redact < "${REPORT_FILE}" redact < "${REPORT_FILE}"
elif [ "${EVIDENCE_OUTCOME}" = "success" ] && [ -s "${LOG_FILE:-}" ]; then elif [ -s "${LOG_FILE:-}" ]; then
echo "(report file missing; log tail below)" echo "(report file missing; log tail below)"
echo "" echo ""
tail -n 200 "${LOG_FILE}" | redact tail -n 200 "${LOG_FILE}" | redact
@@ -296,26 +286,20 @@ jobs:
echo "filed backlog issue for suite ${SUITE}" echo "filed backlog issue for suite ${SUITE}"
- name: Upload test logs & results - name: Upload test logs & results
if: ${{ always() && steps.evidence.outcome == 'success' }} if: always()
uses: actions/upload-artifact@b7c566a772e6b6bfb58ed0dc250532a479d7789f # v6 uses: actions/upload-artifact@b7c566a772e6b6bfb58ed0dc250532a479d7789f # v6
with: with:
name: rustfs-perf-test-${{ github.run_id }}-${{ github.run_attempt }} name: rustfs-perf-test-${{ github.run_id }}
path: | path: |
${{ env.FUNCTIONAL_ARTIFACTS_DIR }}/report.md /tmp/rustfs-perf-test*.log
${{ env.FUNCTIONAL_ARTIFACTS_DIR }}/suite.log /tmp/rustfs-perf-results/**
${{ env.FUNCTIONAL_ARTIFACTS_DIR }}/version.txt /tmp/rustfs-version.txt
${{ env.FUNCTIONAL_ARTIFACTS_DIR }}/results/master.log if-no-files-found: warn
${{ env.FUNCTIONAL_ARTIFACTS_DIR }}/results/summary.md
${{ env.FUNCTIONAL_ARTIFACTS_DIR }}/results/summary.tsv
${{ env.FUNCTIONAL_ARTIFACTS_DIR }}/results/get_*.txt
${{ env.FUNCTIONAL_ARTIFACTS_DIR }}/results/put_*.txt
${{ env.FUNCTIONAL_ARTIFACTS_DIR }}/results/mixed_*.txt
if-no-files-found: error
- name: Reset test environment (after) - name: Reset test environment (after)
if: ${{ always() && inputs.cleanup_after != 'false' }} if: ${{ always() && inputs.cleanup_after != 'false' }}
run: | run: |
./auto-testing/rustfs_performance_test.sh --step 7 -y --log-file "${LOG_FILE:-/dev/null}" ./auto-testing/rustfs_performance_test.sh --step 7 -y
- name: Notify on failure - name: Notify on failure
if: failure() if: failure()
+8 -10
View File
@@ -76,6 +76,9 @@ jobs:
pool-expansion-test: pool-expansion-test:
name: Pool expansion / decommission test name: Pool expansion / decommission test
runs-on: smoke-testing runs-on: smoke-testing
# Requirement: a failing suite must not fail the workflow; failures
# are filed to rustfs/backlog and the chain continues.
continue-on-error: true
timeout-minutes: 360 timeout-minutes: 360
if: ${{ github.event_name == 'workflow_dispatch' || github.event_name == 'repository_dispatch' }} if: ${{ github.event_name == 'workflow_dispatch' || github.event_name == 'repository_dispatch' }}
env: env:
@@ -539,22 +542,17 @@ jobs:
fi fi
DATE="$(date -u +%Y-%m-%d)" DATE="$(date -u +%Y-%m-%d)"
REPORT_PATH="functional-reports/${SUITE}/${DATE}.md" REPORT_PATH="functional-reports/${SUITE}/${DATE}.md"
# Base64-encode the report into a temp file and feed it to jq via CONTENT="$(python3 -c 'import base64,sys;print(base64.b64encode(open(sys.argv[1],"rb").read()).decode())' "${REPORT_FILE}")"
# --rawfile: large reports (e.g. pool) exceed the OS argv limit and
# make `jq --arg content "${CONTENT}"` fail with "Argument list too long".
B64_FILE="$(mktemp)"
python3 -c 'import base64,sys;print(base64.b64encode(open(sys.argv[1],"rb").read()).decode())' "${REPORT_FILE}" > "${B64_FILE}"
SHA="$(gh api "repos/rustfs/dashboard/contents/${REPORT_PATH}" -q '.sha' 2>/dev/null || true)" SHA="$(gh api "repos/rustfs/dashboard/contents/${REPORT_PATH}" -q '.sha' 2>/dev/null || true)"
if [ -n "${SHA}" ]; then if [ -n "${SHA}" ]; then
jq -n --arg msg "report(${SUITE}): ${DATE}" --rawfile content "${B64_FILE}" --arg sha "${SHA}" \ jq -n --arg msg "report(${SUITE}): ${DATE}" --arg content "${CONTENT}" --arg sha "${SHA}" \
'{message:$msg, content:($content|rtrimstr("\n")), sha:$sha}' \ '{message:$msg, content:$content, sha:$sha}' \
| gh api --method PUT "repos/rustfs/dashboard/contents/${REPORT_PATH}" --input - >/dev/null | gh api --method PUT "repos/rustfs/dashboard/contents/${REPORT_PATH}" --input - >/dev/null
else else
jq -n --arg msg "report(${SUITE}): ${DATE}" --rawfile content "${B64_FILE}" \ jq -n --arg msg "report(${SUITE}): ${DATE}" --arg content "${CONTENT}" \
'{message:$msg, content:($content|rtrimstr("\n"))}' \ '{message:$msg, content:$content}' \
| gh api --method PUT "repos/rustfs/dashboard/contents/${REPORT_PATH}" --input - >/dev/null | gh api --method PUT "repos/rustfs/dashboard/contents/${REPORT_PATH}" --input - >/dev/null
fi fi
rm -f "${B64_FILE}"
- name: File failure issue in rustfs/backlog - name: File failure issue in rustfs/backlog
if: ${{ always() && (failure() || steps.pool_test.outcome == 'failure' || steps.pool_test.outcome == 'cancelled') }} if: ${{ always() && (failure() || steps.pool_test.outcome == 'failure' || steps.pool_test.outcome == 'cancelled') }}
+90 -107
View File
@@ -34,7 +34,8 @@ on:
- site - site
default: all default: all
repository_dispatch: repository_dispatch:
# Chain handoff: dispatched when the security suite finishes. # Chain handoff: dispatched when the security suite finishes. This is the
# last link of the functional chain.
types: [rustfs-chain-replication] types: [rustfs-chain-replication]
permissions: permissions:
@@ -61,28 +62,12 @@ env:
jobs: jobs:
replication-test: replication-test:
runs-on: smoke-testing runs-on: smoke-testing
# A failed replication run must not break the chain or the workflow: the
# failure is reported to rustfs/backlog instead (see the issue step).
continue-on-error: true
timeout-minutes: 360 timeout-minutes: 360
if: ${{ github.event_name == 'workflow_dispatch' || github.event_name == 'repository_dispatch' }} if: ${{ github.event_name == 'workflow_dispatch' || github.event_name == 'repository_dispatch' }}
steps: steps:
- name: Checkout repository (for report parser)
uses: actions/checkout@9c091bb21b7c1c1d1991bb908d89e4e9dddfe3e0 # v7
with:
persist-credentials: false
- name: Initialize functional evidence
id: evidence
run: |
set -euo pipefail
umask 077
FUNCTIONAL_ARTIFACTS_DIR="${RUNNER_TEMP}/rustfs-replication-${GITHUB_RUN_ID}-${GITHUB_RUN_ATTEMPT}"
mkdir -- "${FUNCTIONAL_ARTIFACTS_DIR}" "${FUNCTIONAL_ARTIFACTS_DIR}-scratch"
{
printf 'FUNCTIONAL_ARTIFACTS_DIR=%s\n' "${FUNCTIONAL_ARTIFACTS_DIR}"
printf 'LOG_FILE=%s/suite.log\n' "${FUNCTIONAL_ARTIFACTS_DIR}"
printf 'REPORT_FILE=%s/report.md\n' "${FUNCTIONAL_ARTIFACTS_DIR}"
printf 'TMPDIR=%s-scratch\n' "${FUNCTIONAL_ARTIFACTS_DIR}"
} >> "${GITHUB_ENV}"
# auto-testing is private: clone it with the dedicated PF token (not # auto-testing is private: clone it with the dedicated PF token (not
# GITHUB_TOKEN) and retry transient GitHub/network failures. # GITHUB_TOKEN) and retry transient GitHub/network failures.
- name: Checkout auto-testing scripts (with retry) - name: Checkout auto-testing scripts (with retry)
@@ -131,6 +116,9 @@ jobs:
- name: Run replication suite - name: Run replication suite
id: test id: test
continue-on-error: true
env:
LOG_FILE: /tmp/rustfs-replication.log
run: | run: |
set -euo pipefail set -euo pipefail
chmod +x auto-testing/rustfs-replication-test.sh chmod +x auto-testing/rustfs-replication-test.sh
@@ -153,7 +141,10 @@ jobs:
./auto-testing/rustfs-replication-test.sh "${ARGS[@]}" ./auto-testing/rustfs-replication-test.sh "${ARGS[@]}"
- name: Generate report - name: Generate report
if: ${{ always() && steps.evidence.outcome == 'success' }} if: always()
env:
LOG_FILE: /tmp/rustfs-replication.log
REPORT_FILE: /tmp/rustfs-replication-report.md
run: | run: |
set -euo pipefail set -euo pipefail
PACKAGE_URL='${{ inputs.package_url }}' PACKAGE_URL='${{ inputs.package_url }}'
@@ -175,44 +166,80 @@ jobs:
RUSTFS_VERSION_INFO="${DETECTED_VERSION}" RUSTFS_VERSION_INFO="${DETECTED_VERSION}"
fi fi
fi fi
CASE_TABLE="${FUNCTIONAL_ARTIFACTS_DIR}/cases.md" CASE_TABLE="/tmp/rustfs-replication-cases.md"
CASE_RESULT=success python3 - "${LOG_FILE}" "${CASE_TABLE}" <<'PY'
python3 scripts/functional_case_report.py "${LOG_FILE}" "${CASE_TABLE}" || CASE_RESULT=failure import re
RESULT=failure import sys
if [ '${{ steps.test.outcome }}' = 'success' ] && [ "${CASE_RESULT}" = 'success' ]; then
RESULT=success log_file, out_file = sys.argv[1], sys.argv[2]
fi ansi = re.compile(r'\x1b\[[0-9;]*m')
start_re = re.compile(r'^---\s+([A-Z0-9]+-[0-9]+)\s+(.+?)\s+---$')
done_re = re.compile(r'^\[(PASS|FAIL|UNSUPPORTED)\]\s+([A-Z0-9]+-[0-9]+)\b')
rows = []
index = {}
try:
with open(log_file, 'r', encoding='utf-8', errors='replace') as fh:
for raw in fh:
line = ansi.sub('', raw).strip()
m = start_re.match(line)
if m:
case_id, name = m.group(1), m.group(2)
if case_id not in index:
index[case_id] = len(rows)
rows.append([case_id, name, 'RUNNING'])
continue
m = done_re.match(line)
if m:
status, case_id = m.group(1), m.group(2)
if case_id in index:
rows[index[case_id]][2] = status
else:
rows.append([case_id, case_id, status])
index[case_id] = len(rows) - 1
except FileNotFoundError:
rows = []
counts = {'PASS': 0, 'FAIL': 0, 'UNSUPPORTED': 0, 'RUNNING': 0}
for _, _, status in rows:
counts[status] = counts.get(status, 0) + 1
with open(out_file, 'w', encoding='utf-8') as out:
out.write('## Case Summary\n\n')
out.write(f"- Total: {len(rows)}\\n")
out.write(f"- PASS: {counts.get('PASS', 0)}\\n")
out.write(f"- FAIL: {counts.get('FAIL', 0)}\\n")
out.write(f"- UNSUPPORTED: {counts.get('UNSUPPORTED', 0)}\\n")
out.write('\\n')
out.write('| Case | Name | Status |\\n')
out.write('| --- | --- | --- |\\n')
for case_id, name, status in rows:
out.write(f'| {case_id} | {name} | {status} |\\n')
PY
{ {
echo "# RustFS replication test report" echo "# RustFS replication test report"
echo "" echo ""
echo "- Run: ${{ github.server_url }}/${{ github.repository }}/actions/runs/${{ github.run_id }}" echo "- Run: ${{ github.server_url }}/${{ github.repository }}/actions/runs/${{ github.run_id }}"
echo "- Attempt: ${GITHUB_RUN_ATTEMPT}"
echo "- Workflow Commit: ${GITHUB_SHA}"
echo "- Trigger: ${{ github.event_name }}" echo "- Trigger: ${{ github.event_name }}"
echo "- Package: ${PACKAGE_SOURCE}" echo "- Package: ${PACKAGE_SOURCE}"
echo "- RustFS Version: ${RUSTFS_VERSION_INFO}" echo "- RustFS Version: ${RUSTFS_VERSION_INFO}"
echo "- Test Step Outcome: ${RESULT}" echo "- Test Step Outcome: ${{ steps.test.outcome }}"
echo "- Suite Step Outcome: ${{ steps.test.outcome }}"
echo "" echo ""
if [ "${RESULT}" = "success" ]; then cat "${CASE_TABLE}" || true
cat "${CASE_TABLE}" echo ""
echo "" echo "## Log tail"
echo "## Log tail" echo '```text'
echo '```text' tail -n 200 "${LOG_FILE}" || true
tail -n 200 "${LOG_FILE}" echo '```'
echo '```'
else
echo "The suite or evidence validation failed. See this run's artifact for partial case results and suite.log."
fi
} | tee "${REPORT_FILE}" } | tee "${REPORT_FILE}"
cat "${REPORT_FILE}" >> "${GITHUB_STEP_SUMMARY}" cat "${REPORT_FILE}" >> "${GITHUB_STEP_SUMMARY}"
[ "${RESULT}" = "success" ]
- name: Upload functional report to dashboard - name: Upload functional report to dashboard
if: ${{ always() && steps.evidence.outcome == 'success' }} if: always()
continue-on-error: true continue-on-error: true
env: env:
GH_TOKEN: ${{ env.PF_TESTING_GH_TOKEN }} GH_TOKEN: ${{ env.PF_TESTING_GH_TOKEN }}
REPORT_FILE: /tmp/rustfs-replication-report.md
SUITE: replication SUITE: replication
run: | run: |
set -euo pipefail set -euo pipefail
@@ -222,32 +249,28 @@ jobs:
fi fi
DATE="$(date -u +%Y-%m-%d)" DATE="$(date -u +%Y-%m-%d)"
REPORT_PATH="functional-reports/${SUITE}/${DATE}.md" REPORT_PATH="functional-reports/${SUITE}/${DATE}.md"
# Base64-encode the report into a temp file and feed it to jq via CONTENT="$(python3 -c 'import base64,sys;print(base64.b64encode(open(sys.argv[1],"rb").read()).decode())' "${REPORT_FILE}")"
# --rawfile: large reports (e.g. pool) exceed the OS argv limit and
# make `jq --arg content "${CONTENT}"` fail with "Argument list too long".
B64_FILE="$(mktemp)"
python3 -c 'import base64,sys;print(base64.b64encode(open(sys.argv[1],"rb").read()).decode())' "${REPORT_FILE}" > "${B64_FILE}"
SHA="$(gh api "repos/rustfs/dashboard/contents/${REPORT_PATH}" -q '.sha' 2>/dev/null || true)" SHA="$(gh api "repos/rustfs/dashboard/contents/${REPORT_PATH}" -q '.sha' 2>/dev/null || true)"
if [ -n "${SHA}" ]; then if [ -n "${SHA}" ]; then
jq -n --arg msg "report(${SUITE}): ${DATE}" --rawfile content "${B64_FILE}" --arg sha "${SHA}" \ jq -n --arg msg "report(${SUITE}): ${DATE}" --arg content "${CONTENT}" --arg sha "${SHA}" \
'{message:$msg, content:($content|rtrimstr("\n")), sha:$sha}' \ '{message:$msg, content:$content, sha:$sha}' \
| gh api --method PUT "repos/rustfs/dashboard/contents/${REPORT_PATH}" --input - >/dev/null | gh api --method PUT "repos/rustfs/dashboard/contents/${REPORT_PATH}" --input - >/dev/null
else else
jq -n --arg msg "report(${SUITE}): ${DATE}" --rawfile content "${B64_FILE}" \ jq -n --arg msg "report(${SUITE}): ${DATE}" --arg content "${CONTENT}" \
'{message:$msg, content:($content|rtrimstr("\n"))}' \ '{message:$msg, content:$content}' \
| gh api --method PUT "repos/rustfs/dashboard/contents/${REPORT_PATH}" --input - >/dev/null | gh api --method PUT "repos/rustfs/dashboard/contents/${REPORT_PATH}" --input - >/dev/null
fi fi
rm -f "${B64_FILE}"
- name: File failure issue in rustfs/backlog - name: File failure issue in rustfs/backlog
if: ${{ always() && (failure() || steps.test.outcome == 'failure' || steps.test.outcome == 'cancelled') }} if: ${{ always() && (failure() || steps.test.outcome == 'failure' || steps.test.outcome == 'cancelled') }}
continue-on-error: true continue-on-error: true
env: env:
GH_TOKEN: ${{ secrets.PF_TESTING_GH_TOKEN }} GH_TOKEN: ${{ secrets.PF_TESTING_GH_TOKEN }}
EVIDENCE_OUTCOME: ${{ steps.evidence.outcome }}
SUITE: 'replication' SUITE: 'replication'
SUITE_LABEL: 'Replication (bucket + site)' SUITE_LABEL: 'Replication (bucket + site)'
RUN_URL: ${{ github.server_url }}/${{ github.repository }}/actions/runs/${{ github.run_id }} RUN_URL: ${{ github.server_url }}/${{ github.repository }}/actions/runs/${{ github.run_id }}
REPORT_FILE: '/tmp/rustfs-replication-report.md'
LOG_FILE: '/tmp/rustfs-replication.log'
run: | run: |
set -euo pipefail set -euo pipefail
if [ -z "${GH_TOKEN:-}" ]; then if [ -z "${GH_TOKEN:-}" ]; then
@@ -275,16 +298,14 @@ jobs:
echo "" echo ""
echo "- Suite: \`${SUITE}\`" echo "- Suite: \`${SUITE}\`"
echo "- Run: ${RUN_URL}" echo "- Run: ${RUN_URL}"
echo "- Attempt: ${GITHUB_RUN_ATTEMPT}"
echo "- Workflow Commit: ${GITHUB_SHA}"
echo "- Trigger: ${GITHUB_EVENT_NAME}" echo "- Trigger: ${GITHUB_EVENT_NAME}"
echo "- Date: $(date -u +%Y-%m-%d)" echo "- Date: $(date -u +%Y-%m-%d)"
echo "" echo ""
echo "## Report (errors and symptoms)" echo "## Report (errors and symptoms)"
echo "" echo ""
if [ "${EVIDENCE_OUTCOME}" = "success" ] && [ -s "${REPORT_FILE}" ]; then if [ -s "${REPORT_FILE}" ]; then
redact < "${REPORT_FILE}" redact < "${REPORT_FILE}"
elif [ "${EVIDENCE_OUTCOME}" = "success" ] && [ -s "${LOG_FILE:-}" ]; then elif [ -s "${LOG_FILE:-}" ]; then
echo "(report file missing; log tail below)" echo "(report file missing; log tail below)"
echo "" echo ""
tail -n 200 "${LOG_FILE}" | redact tail -n 200 "${LOG_FILE}" | redact
@@ -300,15 +321,14 @@ jobs:
echo "filed backlog issue for suite ${SUITE}" echo "filed backlog issue for suite ${SUITE}"
- name: Upload report and logs - name: Upload report and logs
if: ${{ always() && steps.evidence.outcome == 'success' }} if: always()
uses: actions/upload-artifact@b7c566a772e6b6bfb58ed0dc250532a479d7789f # v6 uses: actions/upload-artifact@b7c566a772e6b6bfb58ed0dc250532a479d7789f # v6
with: with:
name: rustfs-replication-${{ github.run_id }}-${{ github.run_attempt }} name: rustfs-replication-${{ github.run_id }}
path: | path: |
${{ env.FUNCTIONAL_ARTIFACTS_DIR }}/report.md /tmp/rustfs-replication.log
${{ env.FUNCTIONAL_ARTIFACTS_DIR }}/suite.log /tmp/rustfs-replication-report.md
${{ env.FUNCTIONAL_ARTIFACTS_DIR }}/cases.md if-no-files-found: warn
if-no-files-found: error
- name: Cleanup environment (after) - name: Cleanup environment (after)
if: always() if: always()
@@ -329,50 +349,13 @@ jobs:
' '
done done
- name: "Continue functional chain (next: Performance)" - name: Chain complete
# Replication is the last link of the functional chain: nothing to
# dispatch after it. This step just records that the chain finished.
if: ${{ always() && github.event_name == 'repository_dispatch' }} if: ${{ always() && github.event_name == 'repository_dispatch' }}
env:
GH_TOKEN: ${{ secrets.PF_TESTING_GH_TOKEN }}
run: | run: |
set -uo pipefail echo "Functional chain complete: replication (final suite) finished."
if [ -z "${GH_TOKEN:-}" ]; then echo "from_suite=security trigger=${{ github.event_name }} outcome=${{ steps.test.outcome }}"
echo "PF_TESTING_GH_TOKEN is not configured; cannot dispatch the next suite" >&2
exit 1
fi
DISPATCHED=0
for attempt in 1 2 3; do
if gh api --method POST repos/rustfs/rustfs/dispatches \
-f event_type='rustfs-chain-performance' \
-F 'client_payload[from_suite]=replication'; then
echo "dispatched next suite Performance (attempt ${attempt})"
DISPATCHED=1
break
fi
echo "dispatch attempt ${attempt} failed; retrying in ${attempt}0s" >&2
sleep "${attempt}0"
done
if [ "${DISPATCHED:-0}" -ne 1 ]; then
echo "ERROR: functional chain stalled: could not dispatch Performance after 3 attempts" >&2
TITLE="[functional][chain] stalled after replication (run ${GITHUB_RUN_ID})"
BODY_FILE="$(mktemp)"
trap 'rm -f "${BODY_FILE}"' EXIT
{
echo "The functional chain could not hand off from **replication** to **Performance** after 3 attempts."
echo ""
echo "- Failed suite job: ${GITHUB_SERVER_URL}/${GITHUB_REPOSITORY}/actions/runs/${GITHUB_RUN_ID}"
echo "- Expected next event: 'rustfs-chain-performance'"
echo "- Likely cause: PF_TESTING_GH_TOKEN lacks contents:write on rustfs/rustfs, or the GitHub API was unavailable."
echo "- Recovery: re-dispatch manually with"
FENCE="$(printf "\x60\x60\x60")"; echo " ${FENCE}"
echo " gh api --method POST repos/rustfs/rustfs/dispatches -f event_type='rustfs-chain-performance'"
FENCE="$(printf "\x60\x60\x60")"; echo " ${FENCE}"
} > "${BODY_FILE}"
gh issue create -R rustfs/backlog --title "${TITLE}" \
--body-file "${BODY_FILE}" --label functional-test \
|| gh issue create -R rustfs/backlog --title "${TITLE}" --body-file "${BODY_FILE}" \
|| echo "could not file the stall alert issue either; check the token" >&2
exit 1
fi
- name: Notify on failure - name: Notify on failure
if: failure() if: failure()
+84 -64
View File
@@ -37,28 +37,10 @@ env:
jobs: jobs:
s3-compat-test: s3-compat-test:
runs-on: smoke-testing runs-on: smoke-testing
continue-on-error: true
timeout-minutes: 360 timeout-minutes: 360
if: ${{ github.event_name == 'workflow_dispatch' || github.event_name == 'repository_dispatch' }} if: ${{ github.event_name == 'workflow_dispatch' || github.event_name == 'repository_dispatch' }}
steps: steps:
- name: Checkout repository (for report parser)
uses: actions/checkout@9c091bb21b7c1c1d1991bb908d89e4e9dddfe3e0 # v7
with:
persist-credentials: false
- name: Initialize functional evidence
id: evidence
run: |
set -euo pipefail
umask 077
FUNCTIONAL_ARTIFACTS_DIR="${RUNNER_TEMP}/rustfs-s3-compat-${GITHUB_RUN_ID}-${GITHUB_RUN_ATTEMPT}"
mkdir -- "${FUNCTIONAL_ARTIFACTS_DIR}" "${FUNCTIONAL_ARTIFACTS_DIR}-scratch"
{
printf 'FUNCTIONAL_ARTIFACTS_DIR=%s\n' "${FUNCTIONAL_ARTIFACTS_DIR}"
printf 'LOG_FILE=%s/suite.log\n' "${FUNCTIONAL_ARTIFACTS_DIR}"
printf 'REPORT_FILE=%s/report.md\n' "${FUNCTIONAL_ARTIFACTS_DIR}"
printf 'TMPDIR=%s-scratch\n' "${FUNCTIONAL_ARTIFACTS_DIR}"
} >> "${GITHUB_ENV}"
# auto-testing is private: clone it with the dedicated PF token (not # auto-testing is private: clone it with the dedicated PF token (not
# GITHUB_TOKEN) and retry transient GitHub/network failures. # GITHUB_TOKEN) and retry transient GitHub/network failures.
- name: Checkout auto-testing scripts (with retry) - name: Checkout auto-testing scripts (with retry)
@@ -106,6 +88,9 @@ jobs:
- name: Run S3 compatibility suite - name: Run S3 compatibility suite
id: test id: test
continue-on-error: true
env:
LOG_FILE: /tmp/rustfs-s3-compat.log
run: | run: |
set -euo pipefail set -euo pipefail
chmod +x auto-testing/rustfs-s3-compat-test.sh chmod +x auto-testing/rustfs-s3-compat-test.sh
@@ -122,7 +107,10 @@ jobs:
./auto-testing/rustfs-s3-compat-test.sh "${ARGS[@]}" ./auto-testing/rustfs-s3-compat-test.sh "${ARGS[@]}"
- name: Generate report - name: Generate report
if: ${{ always() && steps.evidence.outcome == 'success' }} if: always()
env:
LOG_FILE: /tmp/rustfs-s3-compat.log
REPORT_FILE: /tmp/rustfs-s3-compat-report.md
run: | run: |
set -euo pipefail set -euo pipefail
PACKAGE_URL='${{ inputs.package_url }}' PACKAGE_URL='${{ inputs.package_url }}'
@@ -144,44 +132,83 @@ jobs:
RUSTFS_VERSION_INFO="${DETECTED_VERSION}" RUSTFS_VERSION_INFO="${DETECTED_VERSION}"
fi fi
fi fi
CASE_TABLE="${FUNCTIONAL_ARTIFACTS_DIR}/cases.md" CASE_TABLE="/tmp/rustfs-s3-compat-cases.md"
CASE_RESULT=success python3 - "${LOG_FILE}" "${CASE_TABLE}" <<'PY'
python3 scripts/functional_case_report.py "${LOG_FILE}" "${CASE_TABLE}" || CASE_RESULT=failure import re
RESULT=failure import sys
if [ '${{ steps.test.outcome }}' = 'success' ] && [ "${CASE_RESULT}" = 'success' ]; then
RESULT=success log_file, out_file = sys.argv[1], sys.argv[2]
fi ansi = re.compile(r'\x1b\[[0-9;]*m')
start_re = re.compile(r'^---\s+([A-Z0-9]+-[0-9]+)\s+(.+?)\s+---$')
done_re = re.compile(r'^\[(PASS|FAIL|UNSUPPORTED)\]\s+([A-Z0-9]+-[0-9]+)\b')
rows = []
index = {}
current = None
try:
with open(log_file, 'r', encoding='utf-8', errors='replace') as fh:
for raw in fh:
line = ansi.sub('', raw).strip()
m = start_re.match(line)
if m:
case_id, name = m.group(1), m.group(2)
current = case_id
if case_id not in index:
index[case_id] = len(rows)
rows.append([case_id, name, 'RUNNING'])
continue
m = done_re.match(line)
if m:
status, case_id = m.group(1), m.group(2)
if case_id in index:
rows[index[case_id]][2] = status
else:
rows.append([case_id, case_id, status])
index[case_id] = len(rows) - 1
current = None
except FileNotFoundError:
rows = []
counts = {'PASS': 0, 'FAIL': 0, 'UNSUPPORTED': 0, 'RUNNING': 0}
for _, _, status in rows:
counts[status] = counts.get(status, 0) + 1
with open(out_file, 'w', encoding='utf-8') as out:
out.write('## Case Summary\n\n')
out.write(f"- Total: {len(rows)}\\n")
out.write(f"- PASS: {counts.get('PASS', 0)}\\n")
out.write(f"- FAIL: {counts.get('FAIL', 0)}\\n")
out.write(f"- UNSUPPORTED: {counts.get('UNSUPPORTED', 0)}\\n")
out.write('\\n')
out.write('| Case | Name | Status |\\n')
out.write('| --- | --- | --- |\\n')
for case_id, name, status in rows:
out.write(f'| {case_id} | {name} | {status} |\\n')
PY
{ {
echo "# RustFS S3 compatibility test report" echo "# RustFS S3 compatibility test report"
echo "" echo ""
echo "- Run: ${{ github.server_url }}/${{ github.repository }}/actions/runs/${{ github.run_id }}" echo "- Run: ${{ github.server_url }}/${{ github.repository }}/actions/runs/${{ github.run_id }}"
echo "- Attempt: ${GITHUB_RUN_ATTEMPT}"
echo "- Workflow Commit: ${GITHUB_SHA}"
echo "- Trigger: ${{ github.event_name }}" echo "- Trigger: ${{ github.event_name }}"
echo "- Package: ${PACKAGE_SOURCE}" echo "- Package: ${PACKAGE_SOURCE}"
echo "- RustFS Version: ${RUSTFS_VERSION_INFO}" echo "- RustFS Version: ${RUSTFS_VERSION_INFO}"
echo "- Test Step Outcome: ${RESULT}" echo "- Test Step Outcome: ${{ steps.test.outcome }}"
echo "- Suite Step Outcome: ${{ steps.test.outcome }}"
echo "" echo ""
if [ "${RESULT}" = "success" ]; then cat "${CASE_TABLE}" || true
cat "${CASE_TABLE}" echo ""
echo "" echo "## Log tail"
echo "## Log tail" echo '```text'
echo '```text' tail -n 200 "${LOG_FILE}" || true
tail -n 200 "${LOG_FILE}" echo '```'
echo '```'
else
echo "The suite or evidence validation failed. See this run's artifact for partial case results and suite.log."
fi
} | tee "${REPORT_FILE}" } | tee "${REPORT_FILE}"
cat "${REPORT_FILE}" >> "${GITHUB_STEP_SUMMARY}" cat "${REPORT_FILE}" >> "${GITHUB_STEP_SUMMARY}"
[ "${RESULT}" = "success" ]
- name: Upload functional report to dashboard - name: Upload functional report to dashboard
if: ${{ always() && steps.evidence.outcome == 'success' }} if: always()
continue-on-error: true continue-on-error: true
env: env:
GH_TOKEN: ${{ env.PF_TESTING_GH_TOKEN }} GH_TOKEN: ${{ env.PF_TESTING_GH_TOKEN }}
REPORT_FILE: /tmp/rustfs-s3-compat-report.md
SUITE: s3 SUITE: s3
run: | run: |
set -euo pipefail set -euo pipefail
@@ -191,32 +218,28 @@ jobs:
fi fi
DATE="$(date -u +%Y-%m-%d)" DATE="$(date -u +%Y-%m-%d)"
REPORT_PATH="functional-reports/${SUITE}/${DATE}.md" REPORT_PATH="functional-reports/${SUITE}/${DATE}.md"
# Base64-encode the report into a temp file and feed it to jq via CONTENT="$(python3 -c 'import base64,sys;print(base64.b64encode(open(sys.argv[1],"rb").read()).decode())' "${REPORT_FILE}")"
# --rawfile: large reports (e.g. pool) exceed the OS argv limit and
# make `jq --arg content "${CONTENT}"` fail with "Argument list too long".
B64_FILE="$(mktemp)"
python3 -c 'import base64,sys;print(base64.b64encode(open(sys.argv[1],"rb").read()).decode())' "${REPORT_FILE}" > "${B64_FILE}"
SHA="$(gh api "repos/rustfs/dashboard/contents/${REPORT_PATH}" -q '.sha' 2>/dev/null || true)" SHA="$(gh api "repos/rustfs/dashboard/contents/${REPORT_PATH}" -q '.sha' 2>/dev/null || true)"
if [ -n "${SHA}" ]; then if [ -n "${SHA}" ]; then
jq -n --arg msg "report(${SUITE}): ${DATE}" --rawfile content "${B64_FILE}" --arg sha "${SHA}" \ jq -n --arg msg "report(${SUITE}): ${DATE}" --arg content "${CONTENT}" --arg sha "${SHA}" \
'{message:$msg, content:($content|rtrimstr("\n")), sha:$sha}' \ '{message:$msg, content:$content, sha:$sha}' \
| gh api --method PUT "repos/rustfs/dashboard/contents/${REPORT_PATH}" --input - >/dev/null | gh api --method PUT "repos/rustfs/dashboard/contents/${REPORT_PATH}" --input - >/dev/null
else else
jq -n --arg msg "report(${SUITE}): ${DATE}" --rawfile content "${B64_FILE}" \ jq -n --arg msg "report(${SUITE}): ${DATE}" --arg content "${CONTENT}" \
'{message:$msg, content:($content|rtrimstr("\n"))}' \ '{message:$msg, content:$content}' \
| gh api --method PUT "repos/rustfs/dashboard/contents/${REPORT_PATH}" --input - >/dev/null | gh api --method PUT "repos/rustfs/dashboard/contents/${REPORT_PATH}" --input - >/dev/null
fi fi
rm -f "${B64_FILE}"
- name: File failure issue in rustfs/backlog - name: File failure issue in rustfs/backlog
if: ${{ always() && (failure() || steps.test.outcome == 'failure' || steps.test.outcome == 'cancelled') }} if: ${{ always() && (failure() || steps.test.outcome == 'failure' || steps.test.outcome == 'cancelled') }}
continue-on-error: true continue-on-error: true
env: env:
GH_TOKEN: ${{ secrets.PF_TESTING_GH_TOKEN }} GH_TOKEN: ${{ secrets.PF_TESTING_GH_TOKEN }}
EVIDENCE_OUTCOME: ${{ steps.evidence.outcome }}
SUITE: 's3' SUITE: 's3'
SUITE_LABEL: 'S3 compatibility' SUITE_LABEL: 'S3 compatibility'
RUN_URL: ${{ github.server_url }}/${{ github.repository }}/actions/runs/${{ github.run_id }} RUN_URL: ${{ github.server_url }}/${{ github.repository }}/actions/runs/${{ github.run_id }}
REPORT_FILE: '/tmp/rustfs-s3-compat-report.md'
LOG_FILE: '/tmp/rustfs-s3-compat.log'
run: | run: |
set -euo pipefail set -euo pipefail
if [ -z "${GH_TOKEN:-}" ]; then if [ -z "${GH_TOKEN:-}" ]; then
@@ -244,16 +267,14 @@ jobs:
echo "" echo ""
echo "- Suite: \`${SUITE}\`" echo "- Suite: \`${SUITE}\`"
echo "- Run: ${RUN_URL}" echo "- Run: ${RUN_URL}"
echo "- Attempt: ${GITHUB_RUN_ATTEMPT}"
echo "- Workflow Commit: ${GITHUB_SHA}"
echo "- Trigger: ${GITHUB_EVENT_NAME}" echo "- Trigger: ${GITHUB_EVENT_NAME}"
echo "- Date: $(date -u +%Y-%m-%d)" echo "- Date: $(date -u +%Y-%m-%d)"
echo "" echo ""
echo "## Report (errors and symptoms)" echo "## Report (errors and symptoms)"
echo "" echo ""
if [ "${EVIDENCE_OUTCOME}" = "success" ] && [ -s "${REPORT_FILE}" ]; then if [ -s "${REPORT_FILE}" ]; then
redact < "${REPORT_FILE}" redact < "${REPORT_FILE}"
elif [ "${EVIDENCE_OUTCOME}" = "success" ] && [ -s "${LOG_FILE:-}" ]; then elif [ -s "${LOG_FILE:-}" ]; then
echo "(report file missing; log tail below)" echo "(report file missing; log tail below)"
echo "" echo ""
tail -n 200 "${LOG_FILE}" | redact tail -n 200 "${LOG_FILE}" | redact
@@ -269,15 +290,14 @@ jobs:
echo "filed backlog issue for suite ${SUITE}" echo "filed backlog issue for suite ${SUITE}"
- name: Upload report and logs - name: Upload report and logs
if: ${{ always() && steps.evidence.outcome == 'success' }} if: always()
uses: actions/upload-artifact@b7c566a772e6b6bfb58ed0dc250532a479d7789f # v6 uses: actions/upload-artifact@b7c566a772e6b6bfb58ed0dc250532a479d7789f # v6
with: with:
name: rustfs-s3-compat-${{ github.run_id }}-${{ github.run_attempt }} name: rustfs-s3-compat-${{ github.run_id }}
path: | path: |
${{ env.FUNCTIONAL_ARTIFACTS_DIR }}/report.md /tmp/rustfs-s3-compat.log
${{ env.FUNCTIONAL_ARTIFACTS_DIR }}/suite.log /tmp/rustfs-s3-compat-report.md
${{ env.FUNCTIONAL_ARTIFACTS_DIR }}/cases.md if-no-files-found: warn
if-no-files-found: error
- name: Cleanup environment (after) - name: Cleanup environment (after)
if: always() if: always()
+33 -72
View File
@@ -74,27 +74,10 @@ env:
jobs: jobs:
security-test: security-test:
runs-on: smoke-testing runs-on: smoke-testing
continue-on-error: true
timeout-minutes: 360 timeout-minutes: 360
if: ${{ github.event_name == 'workflow_dispatch' || github.event_name == 'repository_dispatch' }} if: ${{ github.event_name == 'workflow_dispatch' || github.event_name == 'repository_dispatch' }}
steps: steps:
# Checkout the repository into its own subdirectory. Checking out at
# the workspace root would wipe the auto-testing clone above (that is
# exactly how run 33934141181 lost rustfs-security-test.sh).
- name: Checkout repository (for the OIDC live gate script)
uses: actions/checkout@9c091bb21b7c1c1d1991bb908d89e4e9dddfe3e0 # v7
with:
persist-credentials: false
path: rustfs-repo
- name: Initialize security evidence
id: evidence
run: |
set -euo pipefail
umask 077
SECURITY_ARTIFACTS_DIR="${RUNNER_TEMP}/rustfs-security-${GITHUB_RUN_ID}-${GITHUB_RUN_ATTEMPT}"
mkdir -- "${SECURITY_ARTIFACTS_DIR}" "${SECURITY_ARTIFACTS_DIR}-scratch"
printf 'SECURITY_ARTIFACTS_DIR=%s\n' "${SECURITY_ARTIFACTS_DIR}" >> "${GITHUB_ENV}"
# auto-testing is private: clone it with the dedicated PF token (not # auto-testing is private: clone it with the dedicated PF token (not
# GITHUB_TOKEN) and retry transient GitHub/network failures. # GITHUB_TOKEN) and retry transient GitHub/network failures.
- name: Checkout auto-testing scripts (with retry) - name: Checkout auto-testing scripts (with retry)
@@ -115,6 +98,11 @@ jobs:
echo "ERROR: unable to clone rustfs/auto-testing after 5 attempts" >&2 echo "ERROR: unable to clone rustfs/auto-testing after 5 attempts" >&2
exit 1 exit 1
- name: Checkout repository (for the OIDC live gate script)
uses: actions/checkout@9c091bb21b7c1c1d1991bb908d89e4e9dddfe3e0 # v7
with:
persist-credentials: false
- name: Show environment - name: Show environment
run: | run: |
uname -a uname -a
@@ -147,9 +135,8 @@ jobs:
id: test id: test
continue-on-error: true continue-on-error: true
env: env:
REPORT_FILE: ${{ env.SECURITY_ARTIFACTS_DIR }}/suite-report.md REPORT_FILE: /tmp/rustfs-security-report.md
TMPDIR: ${{ env.SECURITY_ARTIFACTS_DIR }}-scratch RUSTFS_SECURITY_OIDC_LIVE_SCRIPT: ${{ github.workspace }}/scripts/test/oidc_keycloak_live.sh
RUSTFS_SECURITY_OIDC_LIVE_SCRIPT: ${{ github.workspace }}/rustfs-repo/scripts/test/oidc_keycloak_live.sh
run: | run: |
set -euo pipefail set -euo pipefail
chmod +x auto-testing/rustfs-security-test.sh chmod +x auto-testing/rustfs-security-test.sh
@@ -172,48 +159,29 @@ jobs:
else else
ARGS+=(--package-url "${RUSTFS_NIGHTLY_PACKAGE_URL}") ARGS+=(--package-url "${RUSTFS_NIGHTLY_PACKAGE_URL}")
fi fi
GITHUB_STEP_SUMMARY=/dev/null ./auto-testing/rustfs-security-test.sh "${ARGS[@]}" 2>&1 | tee "${SECURITY_ARTIFACTS_DIR}/suite.log" ./auto-testing/rustfs-security-test.sh "${ARGS[@]}"
- name: Generate report - name: Generate report
id: report if: always()
if: ${{ always() && steps.evidence.outcome == 'success' }}
env:
TEST_OUTCOME: ${{ steps.test.outcome }}
run: | run: |
set -euo pipefail set -euo pipefail
RESULT=failure if [ ! -f /tmp/rustfs-security-report.md ]; then
if [ "${TEST_OUTCOME}" = "success" ] && [ -s "${SECURITY_ARTIFACTS_DIR}/suite-report.md" ]; then {
RESULT=success echo "# RustFS security test report"
echo ""
echo "- Run: ${{ github.server_url }}/${{ github.repository }}/actions/runs/${{ github.run_id }}"
echo "- Trigger: ${{ github.event_name }}"
echo "- Test Step Outcome: failure (suite did not produce a report)"
} > /tmp/rustfs-security-report.md
fi fi
{ cat /tmp/rustfs-security-report.md >> "${GITHUB_STEP_SUMMARY}"
echo "# RustFS security test report"
echo ""
echo "- Run: ${GITHUB_SERVER_URL}/${GITHUB_REPOSITORY}/actions/runs/${GITHUB_RUN_ID}"
echo "- Attempt: ${GITHUB_RUN_ATTEMPT}"
echo "- Workflow Commit: ${GITHUB_SHA}"
echo "- Trigger: ${GITHUB_EVENT_NAME}"
echo "- Test Step Outcome: ${RESULT}"
echo "- Suite Step Outcome: ${TEST_OUTCOME}"
echo ""
# The dashboard prioritizes case rows over the step outcome.
# Keep partial case results in the artifact when the suite fails.
if [ "${RESULT}" = "success" ]; then
cat "${SECURITY_ARTIFACTS_DIR}/suite-report.md"
elif [ -s "${SECURITY_ARTIFACTS_DIR}/suite-report.md" ]; then
echo "The suite did not complete successfully. See suite-report.md in this run's artifact for diagnostics."
else
echo "The suite did not produce a non-empty report."
fi
} > "${SECURITY_ARTIFACTS_DIR}/report.md"
cat "${SECURITY_ARTIFACTS_DIR}/report.md" >> "${GITHUB_STEP_SUMMARY}"
[ "${RESULT}" = "success" ]
- name: Upload functional report to dashboard - name: Upload functional report to dashboard
if: ${{ always() && steps.evidence.outcome == 'success' }} if: always()
continue-on-error: true continue-on-error: true
env: env:
GH_TOKEN: ${{ env.PF_TESTING_GH_TOKEN }} GH_TOKEN: ${{ env.PF_TESTING_GH_TOKEN }}
REPORT_FILE: ${{ env.SECURITY_ARTIFACTS_DIR }}/report.md REPORT_FILE: /tmp/rustfs-security-report.md
SUITE: security SUITE: security
run: | run: |
set -euo pipefail set -euo pipefail
@@ -223,22 +191,17 @@ jobs:
fi fi
DATE="$(date -u +%Y-%m-%d)" DATE="$(date -u +%Y-%m-%d)"
REPORT_PATH="functional-reports/${SUITE}/${DATE}.md" REPORT_PATH="functional-reports/${SUITE}/${DATE}.md"
# Base64-encode the report into a temp file and feed it to jq via CONTENT="$(python3 -c 'import base64,sys;print(base64.b64encode(open(sys.argv[1],"rb").read()).decode())' "${REPORT_FILE}")"
# --rawfile: large reports (e.g. pool) exceed the OS argv limit and
# make `jq --arg content "${CONTENT}"` fail with "Argument list too long".
B64_FILE="$(mktemp)"
python3 -c 'import base64,sys;print(base64.b64encode(open(sys.argv[1],"rb").read()).decode())' "${REPORT_FILE}" > "${B64_FILE}"
SHA="$(gh api "repos/rustfs/dashboard/contents/${REPORT_PATH}" -q '.sha' 2>/dev/null || true)" SHA="$(gh api "repos/rustfs/dashboard/contents/${REPORT_PATH}" -q '.sha' 2>/dev/null || true)"
if [ -n "${SHA}" ]; then if [ -n "${SHA}" ]; then
jq -n --arg msg "report(${SUITE}): ${DATE}" --rawfile content "${B64_FILE}" --arg sha "${SHA}" \ jq -n --arg msg "report(${SUITE}): ${DATE}" --arg content "${CONTENT}" --arg sha "${SHA}" \
'{message:$msg, content:($content|rtrimstr("\n")), sha:$sha}' \ '{message:$msg, content:$content, sha:$sha}' \
| gh api --method PUT "repos/rustfs/dashboard/contents/${REPORT_PATH}" --input - >/dev/null | gh api --method PUT "repos/rustfs/dashboard/contents/${REPORT_PATH}" --input - >/dev/null
else else
jq -n --arg msg "report(${SUITE}): ${DATE}" --rawfile content "${B64_FILE}" \ jq -n --arg msg "report(${SUITE}): ${DATE}" --arg content "${CONTENT}" \
'{message:$msg, content:($content|rtrimstr("\n"))}' \ '{message:$msg, content:$content}' \
| gh api --method PUT "repos/rustfs/dashboard/contents/${REPORT_PATH}" --input - >/dev/null | gh api --method PUT "repos/rustfs/dashboard/contents/${REPORT_PATH}" --input - >/dev/null
fi fi
rm -f "${B64_FILE}"
- name: File failure issue in rustfs/backlog - name: File failure issue in rustfs/backlog
if: ${{ always() && (failure() || steps.test.outcome == 'failure' || steps.test.outcome == 'cancelled') }} if: ${{ always() && (failure() || steps.test.outcome == 'failure' || steps.test.outcome == 'cancelled') }}
@@ -247,9 +210,8 @@ jobs:
GH_TOKEN: ${{ secrets.PF_TESTING_GH_TOKEN }} GH_TOKEN: ${{ secrets.PF_TESTING_GH_TOKEN }}
SUITE: 'security' SUITE: 'security'
SUITE_LABEL: 'Security' SUITE_LABEL: 'Security'
EVIDENCE_OUTCOME: ${{ steps.evidence.outcome }}
RUN_URL: ${{ github.server_url }}/${{ github.repository }}/actions/runs/${{ github.run_id }} RUN_URL: ${{ github.server_url }}/${{ github.repository }}/actions/runs/${{ github.run_id }}
REPORT_FILE: ${{ env.SECURITY_ARTIFACTS_DIR }}/report.md REPORT_FILE: '/tmp/rustfs-security-report.md'
LOG_FILE: '' LOG_FILE: ''
run: | run: |
set -euo pipefail set -euo pipefail
@@ -283,7 +245,7 @@ jobs:
echo "" echo ""
echo "## Report (errors and symptoms)" echo "## Report (errors and symptoms)"
echo "" echo ""
if [ "${EVIDENCE_OUTCOME}" = "success" ] && [ -s "${REPORT_FILE}" ]; then if [ -s "${REPORT_FILE}" ]; then
redact < "${REPORT_FILE}" redact < "${REPORT_FILE}"
elif [ -s "${LOG_FILE:-}" ]; then elif [ -s "${LOG_FILE:-}" ]; then
echo "(report file missing; log tail below)" echo "(report file missing; log tail below)"
@@ -301,15 +263,14 @@ jobs:
echo "filed backlog issue for suite ${SUITE}" echo "filed backlog issue for suite ${SUITE}"
- name: Upload report and logs - name: Upload report and logs
if: ${{ always() && steps.evidence.outcome == 'success' }} if: always()
uses: actions/upload-artifact@b7c566a772e6b6bfb58ed0dc250532a479d7789f # v6 uses: actions/upload-artifact@b7c566a772e6b6bfb58ed0dc250532a479d7789f # v6
with: with:
name: rustfs-security-test-${{ github.run_id }}-${{ github.run_attempt }} name: rustfs-security-test-${{ github.run_id }}
path: | path: |
${{ env.SECURITY_ARTIFACTS_DIR }}/report.md /tmp/rustfs-security-report.md
${{ env.SECURITY_ARTIFACTS_DIR }}/suite.log /tmp/rustfs-security.*/*
${{ env.SECURITY_ARTIFACTS_DIR }}/suite-report.md if-no-files-found: ignore
if-no-files-found: error
retention-days: 3 retention-days: 3
- name: Cleanup environment (after) - name: Cleanup environment (after)
+84 -64
View File
@@ -46,28 +46,10 @@ env:
jobs: jobs:
storage-test: storage-test:
runs-on: smoke-testing runs-on: smoke-testing
continue-on-error: true
timeout-minutes: 360 timeout-minutes: 360
if: ${{ github.event_name == 'workflow_dispatch' || github.event_name == 'repository_dispatch' }} if: ${{ github.event_name == 'workflow_dispatch' || github.event_name == 'repository_dispatch' }}
steps: steps:
- name: Checkout repository (for report parser)
uses: actions/checkout@9c091bb21b7c1c1d1991bb908d89e4e9dddfe3e0 # v7
with:
persist-credentials: false
- name: Initialize functional evidence
id: evidence
run: |
set -euo pipefail
umask 077
FUNCTIONAL_ARTIFACTS_DIR="${RUNNER_TEMP}/rustfs-storage-${GITHUB_RUN_ID}-${GITHUB_RUN_ATTEMPT}"
mkdir -- "${FUNCTIONAL_ARTIFACTS_DIR}" "${FUNCTIONAL_ARTIFACTS_DIR}-scratch"
{
printf 'FUNCTIONAL_ARTIFACTS_DIR=%s\n' "${FUNCTIONAL_ARTIFACTS_DIR}"
printf 'LOG_FILE=%s/suite.log\n' "${FUNCTIONAL_ARTIFACTS_DIR}"
printf 'REPORT_FILE=%s/report.md\n' "${FUNCTIONAL_ARTIFACTS_DIR}"
printf 'TMPDIR=%s-scratch\n' "${FUNCTIONAL_ARTIFACTS_DIR}"
} >> "${GITHUB_ENV}"
# auto-testing is private: clone it with the dedicated PF token (not # auto-testing is private: clone it with the dedicated PF token (not
# GITHUB_TOKEN) and retry transient GitHub/network failures. # GITHUB_TOKEN) and retry transient GitHub/network failures.
- name: Checkout auto-testing scripts (with retry) - name: Checkout auto-testing scripts (with retry)
@@ -115,6 +97,9 @@ jobs:
- name: Run storage engine suite - name: Run storage engine suite
id: test id: test
continue-on-error: true
env:
LOG_FILE: /tmp/rustfs-storage.log
run: | run: |
set -euo pipefail set -euo pipefail
chmod +x auto-testing/rustfs-storage-test.sh chmod +x auto-testing/rustfs-storage-test.sh
@@ -137,7 +122,10 @@ jobs:
./auto-testing/rustfs-storage-test.sh "${ARGS[@]}" ./auto-testing/rustfs-storage-test.sh "${ARGS[@]}"
- name: Generate report - name: Generate report
if: ${{ always() && steps.evidence.outcome == 'success' }} if: always()
env:
LOG_FILE: /tmp/rustfs-storage.log
REPORT_FILE: /tmp/rustfs-storage-report.md
run: | run: |
set -euo pipefail set -euo pipefail
PACKAGE_URL='${{ inputs.package_url }}' PACKAGE_URL='${{ inputs.package_url }}'
@@ -159,44 +147,83 @@ jobs:
RUSTFS_VERSION_INFO="${DETECTED_VERSION}" RUSTFS_VERSION_INFO="${DETECTED_VERSION}"
fi fi
fi fi
CASE_TABLE="${FUNCTIONAL_ARTIFACTS_DIR}/cases.md" CASE_TABLE="/tmp/rustfs-storage-cases.md"
CASE_RESULT=success python3 - "${LOG_FILE}" "${CASE_TABLE}" <<'PY'
python3 scripts/functional_case_report.py "${LOG_FILE}" "${CASE_TABLE}" || CASE_RESULT=failure import re
RESULT=failure import sys
if [ '${{ steps.test.outcome }}' = 'success' ] && [ "${CASE_RESULT}" = 'success' ]; then
RESULT=success log_file, out_file = sys.argv[1], sys.argv[2]
fi ansi = re.compile(r'\x1b\[[0-9;]*m')
start_re = re.compile(r'^---\s+([A-Z0-9]+-[0-9]+)\s+(.+?)\s+---$')
done_re = re.compile(r'^\[(PASS|FAIL|UNSUPPORTED)\]\s+([A-Z0-9]+-[0-9]+)\b')
rows = []
index = {}
current = None
try:
with open(log_file, 'r', encoding='utf-8', errors='replace') as fh:
for raw in fh:
line = ansi.sub('', raw).strip()
m = start_re.match(line)
if m:
case_id, name = m.group(1), m.group(2)
current = case_id
if case_id not in index:
index[case_id] = len(rows)
rows.append([case_id, name, 'RUNNING'])
continue
m = done_re.match(line)
if m:
status, case_id = m.group(1), m.group(2)
if case_id in index:
rows[index[case_id]][2] = status
else:
rows.append([case_id, case_id, status])
index[case_id] = len(rows) - 1
current = None
except FileNotFoundError:
rows = []
counts = {'PASS': 0, 'FAIL': 0, 'UNSUPPORTED': 0, 'RUNNING': 0}
for _, _, status in rows:
counts[status] = counts.get(status, 0) + 1
with open(out_file, 'w', encoding='utf-8') as out:
out.write('## Case Summary\n\n')
out.write(f"- Total: {len(rows)}\\n")
out.write(f"- PASS: {counts.get('PASS', 0)}\\n")
out.write(f"- FAIL: {counts.get('FAIL', 0)}\\n")
out.write(f"- UNSUPPORTED: {counts.get('UNSUPPORTED', 0)}\\n")
out.write('\\n')
out.write('| Case | Name | Status |\\n')
out.write('| --- | --- | --- |\\n')
for case_id, name, status in rows:
out.write(f'| {case_id} | {name} | {status} |\\n')
PY
{ {
echo "# RustFS storage engine test report" echo "# RustFS storage engine test report"
echo "" echo ""
echo "- Run: ${{ github.server_url }}/${{ github.repository }}/actions/runs/${{ github.run_id }}" echo "- Run: ${{ github.server_url }}/${{ github.repository }}/actions/runs/${{ github.run_id }}"
echo "- Attempt: ${GITHUB_RUN_ATTEMPT}"
echo "- Workflow Commit: ${GITHUB_SHA}"
echo "- Trigger: ${{ github.event_name }}" echo "- Trigger: ${{ github.event_name }}"
echo "- Package: ${PACKAGE_SOURCE}" echo "- Package: ${PACKAGE_SOURCE}"
echo "- RustFS Version: ${RUSTFS_VERSION_INFO}" echo "- RustFS Version: ${RUSTFS_VERSION_INFO}"
echo "- Test Step Outcome: ${RESULT}" echo "- Test Step Outcome: ${{ steps.test.outcome }}"
echo "- Suite Step Outcome: ${{ steps.test.outcome }}"
echo "" echo ""
if [ "${RESULT}" = "success" ]; then cat "${CASE_TABLE}" || true
cat "${CASE_TABLE}" echo ""
echo "" echo "## Log tail"
echo "## Log tail" echo '```text'
echo '```text' tail -n 200 "${LOG_FILE}" || true
tail -n 200 "${LOG_FILE}" echo '```'
echo '```'
else
echo "The suite or evidence validation failed. See this run's artifact for partial case results and suite.log."
fi
} | tee "${REPORT_FILE}" } | tee "${REPORT_FILE}"
cat "${REPORT_FILE}" >> "${GITHUB_STEP_SUMMARY}" cat "${REPORT_FILE}" >> "${GITHUB_STEP_SUMMARY}"
[ "${RESULT}" = "success" ]
- name: Upload functional report to dashboard - name: Upload functional report to dashboard
if: ${{ always() && steps.evidence.outcome == 'success' }} if: always()
continue-on-error: true continue-on-error: true
env: env:
GH_TOKEN: ${{ env.PF_TESTING_GH_TOKEN }} GH_TOKEN: ${{ env.PF_TESTING_GH_TOKEN }}
REPORT_FILE: /tmp/rustfs-storage-report.md
SUITE: storage SUITE: storage
run: | run: |
set -euo pipefail set -euo pipefail
@@ -206,32 +233,28 @@ jobs:
fi fi
DATE="$(date -u +%Y-%m-%d)" DATE="$(date -u +%Y-%m-%d)"
REPORT_PATH="functional-reports/${SUITE}/${DATE}.md" REPORT_PATH="functional-reports/${SUITE}/${DATE}.md"
# Base64-encode the report into a temp file and feed it to jq via CONTENT="$(python3 -c 'import base64,sys;print(base64.b64encode(open(sys.argv[1],"rb").read()).decode())' "${REPORT_FILE}")"
# --rawfile: large reports (e.g. pool) exceed the OS argv limit and
# make `jq --arg content "${CONTENT}"` fail with "Argument list too long".
B64_FILE="$(mktemp)"
python3 -c 'import base64,sys;print(base64.b64encode(open(sys.argv[1],"rb").read()).decode())' "${REPORT_FILE}" > "${B64_FILE}"
SHA="$(gh api "repos/rustfs/dashboard/contents/${REPORT_PATH}" -q '.sha' 2>/dev/null || true)" SHA="$(gh api "repos/rustfs/dashboard/contents/${REPORT_PATH}" -q '.sha' 2>/dev/null || true)"
if [ -n "${SHA}" ]; then if [ -n "${SHA}" ]; then
jq -n --arg msg "report(${SUITE}): ${DATE}" --rawfile content "${B64_FILE}" --arg sha "${SHA}" \ jq -n --arg msg "report(${SUITE}): ${DATE}" --arg content "${CONTENT}" --arg sha "${SHA}" \
'{message:$msg, content:($content|rtrimstr("\n")), sha:$sha}' \ '{message:$msg, content:$content, sha:$sha}' \
| gh api --method PUT "repos/rustfs/dashboard/contents/${REPORT_PATH}" --input - >/dev/null | gh api --method PUT "repos/rustfs/dashboard/contents/${REPORT_PATH}" --input - >/dev/null
else else
jq -n --arg msg "report(${SUITE}): ${DATE}" --rawfile content "${B64_FILE}" \ jq -n --arg msg "report(${SUITE}): ${DATE}" --arg content "${CONTENT}" \
'{message:$msg, content:($content|rtrimstr("\n"))}' \ '{message:$msg, content:$content}' \
| gh api --method PUT "repos/rustfs/dashboard/contents/${REPORT_PATH}" --input - >/dev/null | gh api --method PUT "repos/rustfs/dashboard/contents/${REPORT_PATH}" --input - >/dev/null
fi fi
rm -f "${B64_FILE}"
- name: File failure issue in rustfs/backlog - name: File failure issue in rustfs/backlog
if: ${{ always() && (failure() || steps.test.outcome == 'failure' || steps.test.outcome == 'cancelled') }} if: ${{ always() && (failure() || steps.test.outcome == 'failure' || steps.test.outcome == 'cancelled') }}
continue-on-error: true continue-on-error: true
env: env:
GH_TOKEN: ${{ secrets.PF_TESTING_GH_TOKEN }} GH_TOKEN: ${{ secrets.PF_TESTING_GH_TOKEN }}
EVIDENCE_OUTCOME: ${{ steps.evidence.outcome }}
SUITE: 'storage' SUITE: 'storage'
SUITE_LABEL: 'Storage engine' SUITE_LABEL: 'Storage engine'
RUN_URL: ${{ github.server_url }}/${{ github.repository }}/actions/runs/${{ github.run_id }} RUN_URL: ${{ github.server_url }}/${{ github.repository }}/actions/runs/${{ github.run_id }}
REPORT_FILE: '/tmp/rustfs-storage-report.md'
LOG_FILE: '/tmp/rustfs-storage.log'
run: | run: |
set -euo pipefail set -euo pipefail
if [ -z "${GH_TOKEN:-}" ]; then if [ -z "${GH_TOKEN:-}" ]; then
@@ -259,16 +282,14 @@ jobs:
echo "" echo ""
echo "- Suite: \`${SUITE}\`" echo "- Suite: \`${SUITE}\`"
echo "- Run: ${RUN_URL}" echo "- Run: ${RUN_URL}"
echo "- Attempt: ${GITHUB_RUN_ATTEMPT}"
echo "- Workflow Commit: ${GITHUB_SHA}"
echo "- Trigger: ${GITHUB_EVENT_NAME}" echo "- Trigger: ${GITHUB_EVENT_NAME}"
echo "- Date: $(date -u +%Y-%m-%d)" echo "- Date: $(date -u +%Y-%m-%d)"
echo "" echo ""
echo "## Report (errors and symptoms)" echo "## Report (errors and symptoms)"
echo "" echo ""
if [ "${EVIDENCE_OUTCOME}" = "success" ] && [ -s "${REPORT_FILE}" ]; then if [ -s "${REPORT_FILE}" ]; then
redact < "${REPORT_FILE}" redact < "${REPORT_FILE}"
elif [ "${EVIDENCE_OUTCOME}" = "success" ] && [ -s "${LOG_FILE:-}" ]; then elif [ -s "${LOG_FILE:-}" ]; then
echo "(report file missing; log tail below)" echo "(report file missing; log tail below)"
echo "" echo ""
tail -n 200 "${LOG_FILE}" | redact tail -n 200 "${LOG_FILE}" | redact
@@ -284,15 +305,14 @@ jobs:
echo "filed backlog issue for suite ${SUITE}" echo "filed backlog issue for suite ${SUITE}"
- name: Upload report and logs - name: Upload report and logs
if: ${{ always() && steps.evidence.outcome == 'success' }} if: always()
uses: actions/upload-artifact@b7c566a772e6b6bfb58ed0dc250532a479d7789f # v6 uses: actions/upload-artifact@b7c566a772e6b6bfb58ed0dc250532a479d7789f # v6
with: with:
name: rustfs-storage-${{ github.run_id }}-${{ github.run_attempt }} name: rustfs-storage-${{ github.run_id }}
path: | path: |
${{ env.FUNCTIONAL_ARTIFACTS_DIR }}/report.md /tmp/rustfs-storage.log
${{ env.FUNCTIONAL_ARTIFACTS_DIR }}/suite.log /tmp/rustfs-storage-report.md
${{ env.FUNCTIONAL_ARTIFACTS_DIR }}/cases.md if-no-files-found: warn
if-no-files-found: error
- name: Cleanup environment (after) - name: Cleanup environment (after)
if: always() if: always()
+8 -10
View File
@@ -61,6 +61,9 @@ env:
jobs: jobs:
tier-test: tier-test:
runs-on: smoke-testing runs-on: smoke-testing
# Requirement: a failing suite must not fail the workflow; failures
# are filed to rustfs/backlog and the chain continues.
continue-on-error: true
timeout-minutes: 420 timeout-minutes: 420
if: ${{ github.event_name == 'workflow_dispatch' || github.event_name == 'repository_dispatch' }} if: ${{ github.event_name == 'workflow_dispatch' || github.event_name == 'repository_dispatch' }}
steps: steps:
@@ -377,22 +380,17 @@ jobs:
fi fi
DATE="$(date -u +%Y-%m-%d)" DATE="$(date -u +%Y-%m-%d)"
REPORT_PATH="functional-reports/${SUITE}/${DATE}.md" REPORT_PATH="functional-reports/${SUITE}/${DATE}.md"
# Base64-encode the report into a temp file and feed it to jq via CONTENT="$(python3 -c 'import base64,sys;print(base64.b64encode(open(sys.argv[1],"rb").read()).decode())' "${REPORT_FILE}")"
# --rawfile: large reports (e.g. pool) exceed the OS argv limit and
# make `jq --arg content "${CONTENT}"` fail with "Argument list too long".
B64_FILE="$(mktemp)"
python3 -c 'import base64,sys;print(base64.b64encode(open(sys.argv[1],"rb").read()).decode())' "${REPORT_FILE}" > "${B64_FILE}"
SHA="$(gh api "repos/rustfs/dashboard/contents/${REPORT_PATH}" -q '.sha' 2>/dev/null || true)" SHA="$(gh api "repos/rustfs/dashboard/contents/${REPORT_PATH}" -q '.sha' 2>/dev/null || true)"
if [ -n "${SHA}" ]; then if [ -n "${SHA}" ]; then
jq -n --arg msg "report(${SUITE}): ${DATE}" --rawfile content "${B64_FILE}" --arg sha "${SHA}" \ jq -n --arg msg "report(${SUITE}): ${DATE}" --arg content "${CONTENT}" --arg sha "${SHA}" \
'{message:$msg, content:($content|rtrimstr("\n")), sha:$sha}' \ '{message:$msg, content:$content, sha:$sha}' \
| gh api --method PUT "repos/rustfs/dashboard/contents/${REPORT_PATH}" --input - >/dev/null | gh api --method PUT "repos/rustfs/dashboard/contents/${REPORT_PATH}" --input - >/dev/null
else else
jq -n --arg msg "report(${SUITE}): ${DATE}" --rawfile content "${B64_FILE}" \ jq -n --arg msg "report(${SUITE}): ${DATE}" --arg content "${CONTENT}" \
'{message:$msg, content:($content|rtrimstr("\n"))}' \ '{message:$msg, content:$content}' \
| gh api --method PUT "repos/rustfs/dashboard/contents/${REPORT_PATH}" --input - >/dev/null | gh api --method PUT "repos/rustfs/dashboard/contents/${REPORT_PATH}" --input - >/dev/null
fi fi
rm -f "${B64_FILE}"
- name: Verify required tier evidence - name: Verify required tier evidence
id: evidence_verify id: evidence_verify
+105 -94
View File
@@ -18,7 +18,7 @@ on:
workflow_dispatch: workflow_dispatch:
inputs: inputs:
from_version: from_version:
description: 'OLD RustFS release tag, e.g. 1.0.0-rc.3 (its release must ship a .deb asset). Leave empty for the default.' description: 'OLD RustFS release tag (must ship a .deb asset, e.g. 1.0.0-rc.3)'
required: false required: false
default: '1.0.0-rc.3' default: '1.0.0-rc.3'
from_url: from_url:
@@ -26,7 +26,7 @@ on:
required: false required: false
type: string type: string
to_version: to_version:
description: 'NEW RustFS release tag, e.g. 1.0.0-rc.5 (any version with a .deb asset). Leave empty for latest nightly.' description: 'NEW RustFS release tag (leave empty for latest nightly)'
required: false required: false
to_url: to_url:
description: 'NEW .deb URL. Overrides to_version / nightly default.' description: 'NEW .deb URL. Overrides to_version / nightly default.'
@@ -79,28 +79,10 @@ env:
jobs: jobs:
upgrade-test: upgrade-test:
runs-on: smoke-testing runs-on: smoke-testing
continue-on-error: true
timeout-minutes: 420 timeout-minutes: 420
if: ${{ github.event_name == 'workflow_dispatch' || github.event_name == 'repository_dispatch' }} if: ${{ github.event_name == 'workflow_dispatch' || github.event_name == 'repository_dispatch' }}
steps: steps:
- name: Checkout repository (for report parser)
uses: actions/checkout@9c091bb21b7c1c1d1991bb908d89e4e9dddfe3e0 # v7
with:
persist-credentials: false
- name: Initialize functional evidence
id: evidence
run: |
set -euo pipefail
umask 077
FUNCTIONAL_ARTIFACTS_DIR="${RUNNER_TEMP}/rustfs-upgrade-${GITHUB_RUN_ID}-${GITHUB_RUN_ATTEMPT}"
mkdir -- "${FUNCTIONAL_ARTIFACTS_DIR}" "${FUNCTIONAL_ARTIFACTS_DIR}-scratch"
{
printf 'FUNCTIONAL_ARTIFACTS_DIR=%s\n' "${FUNCTIONAL_ARTIFACTS_DIR}"
printf 'LOG_FILE=%s/suite.log\n' "${FUNCTIONAL_ARTIFACTS_DIR}"
printf 'REPORT_FILE=%s/report.md\n' "${FUNCTIONAL_ARTIFACTS_DIR}"
printf 'TMPDIR=%s-scratch\n' "${FUNCTIONAL_ARTIFACTS_DIR}"
} >> "${GITHUB_ENV}"
# auto-testing is private: clone it with the dedicated PF token (not # auto-testing is private: clone it with the dedicated PF token (not
# GITHUB_TOKEN) and retry transient GitHub/network failures. # GITHUB_TOKEN) and retry transient GitHub/network failures.
- name: Checkout auto-testing scripts (with retry) - name: Checkout auto-testing scripts (with retry)
@@ -160,8 +142,9 @@ jobs:
- name: Run upgrade compatibility suite - name: Run upgrade compatibility suite
id: test id: test
continue-on-error: true
env: env:
GH_TOKEN: ${{ secrets.PF_TESTING_GH_TOKEN }} LOG_FILE: /tmp/rustfs-upgrade.log
run: | run: |
set -euo pipefail set -euo pipefail
chmod +x auto-testing/rustfs-upgrade-test.sh chmod +x auto-testing/rustfs-upgrade-test.sh
@@ -192,33 +175,13 @@ jobs:
else else
ARGS+=(--to-url "${RUSTFS_NIGHTLY_PACKAGE_URL}") ARGS+=(--to-url "${RUSTFS_NIGHTLY_PACKAGE_URL}")
fi fi
# Fail fast with a clear message when a requested release tag has
# no .deb asset (e.g. 1.0.0-rc.4 ships only zips), instead of
# letting the suite die mid-run on a 404.
check_release_asset() {
local version="$1" tag asset url
[ -n "${version}" ] && [ "${version}" != "null" ] || return 0
tag="${version#v}"
asset="rustfs_${tag//-/.}_amd64.deb"
url="https://github.com/rustfs/rustfs/releases/download/${tag}/${asset}"
if ! gh api "repos/rustfs/rustfs/releases/tags/${tag}" --jq '.assets[].name' 2>/dev/null | grep -qxF "${asset}"; then
echo "ERROR: release ${tag} has no downloadable asset ${asset}:" >&2
echo " ${url}" >&2
echo "Pick a tag whose release ships a .deb (check its release assets)." >&2
exit 1
fi
echo "resolved ${tag} -> ${url}"
}
if [ -z "${FROM_URL}" ]; then
check_release_asset "${FROM_VERSION}"
fi
if [ -z "${TO_URL}" ]; then
check_release_asset "${TO_VERSION}"
fi
./auto-testing/rustfs-upgrade-test.sh "${ARGS[@]}" ./auto-testing/rustfs-upgrade-test.sh "${ARGS[@]}"
- name: Generate report - name: Generate report
if: ${{ always() && steps.evidence.outcome == 'success' }} if: always()
env:
LOG_FILE: /tmp/rustfs-upgrade.log
REPORT_FILE: /tmp/rustfs-upgrade-report.md
run: | run: |
set -euo pipefail set -euo pipefail
FROM_URL='${{ inputs.from_url }}' FROM_URL='${{ inputs.from_url }}'
@@ -239,47 +202,103 @@ jobs:
else else
TO_SOURCE="${RUSTFS_NIGHTLY_PACKAGE_URL}" TO_SOURCE="${RUSTFS_NIGHTLY_PACKAGE_URL}"
fi fi
CASE_TABLE="${FUNCTIONAL_ARTIFACTS_DIR}/cases.md" CASE_TABLE="/tmp/rustfs-upgrade-cases.md"
MATRIX_TABLE="${FUNCTIONAL_ARTIFACTS_DIR}/matrix.md" MATRIX_TABLE="/tmp/rustfs-upgrade-matrix.md"
CASE_RESULT=success python3 - "${LOG_FILE}" "${CASE_TABLE}" "${MATRIX_TABLE}" <<'PY'
python3 scripts/functional_case_report.py "${LOG_FILE}" "${CASE_TABLE}" "${MATRIX_TABLE}" || CASE_RESULT=failure import re
RESULT=failure import sys
if [ '${{ steps.test.outcome }}' = 'success' ] && [ "${CASE_RESULT}" = 'success' ]; then
RESULT=success log_file, out_file, matrix_file = sys.argv[1], sys.argv[2], sys.argv[3]
fi ansi = re.compile(r'\x1b\[[0-9;]*m')
start_re = re.compile(r'^---\s+([A-Z]+-[0-9]+)\s+(.+?)\s+---$')
done_re = re.compile(r'^\[(PASS|FAIL|UNSUPPORTED)\]\s+([A-Z]+-[0-9]+)\b')
topo_re = re.compile(
r'^\[UPG-TOPO\]\s+(\S+)\s+(\S+)\s+(\S+)\s+(\S+)\s+PASS=(\d+)\s+FAIL=(\d+)\s*$')
rows = []
index = {}
topo_rows = []
try:
with open(log_file, 'r', encoding='utf-8', errors='replace') as fh:
for raw in fh:
line = ansi.sub('', raw).strip()
m = topo_re.match(line)
if m:
topo_rows.append(m.groups())
continue
m = start_re.match(line)
if m:
case_id, name = m.group(1), m.group(2)
if case_id not in index:
index[case_id] = len(rows)
rows.append([case_id, name, 'RUNNING'])
continue
m = done_re.match(line)
if m:
status, case_id = m.group(1), m.group(2)
if case_id in index:
rows[index[case_id]][2] = status
else:
rows.append([case_id, case_id, status])
index[case_id] = len(rows) - 1
except FileNotFoundError:
rows = []
counts = {'PASS': 0, 'FAIL': 0, 'UNSUPPORTED': 0, 'RUNNING': 0}
for _, _, status in rows:
counts[status] = counts.get(status, 0) + 1
with open(out_file, 'w', encoding='utf-8') as out:
out.write('## Case Summary\n\n')
out.write(f"- Total: {len(rows)}\\n")
out.write(f"- PASS: {counts.get('PASS', 0)}\\n")
out.write(f"- FAIL: {counts.get('FAIL', 0)}\\n")
out.write(f"- UNSUPPORTED: {counts.get('UNSUPPORTED', 0)}\\n")
out.write('\\n')
out.write('| Case | Name | Status |\\n')
out.write('| --- | --- | --- |\\n')
for case_id, name, status in rows:
out.write(f'| {case_id} | {name} | {status} |\\n')
# Upgrade matrix: one row per topology/backend with the versions
# captured on the nodes (rustfs --version) and the aggregated
# result. The dashboard renders this table directly.
with open(matrix_file, 'w', encoding='utf-8') as out:
out.write('## Upgrade Matrix\n\n')
out.write('| Topology | KMS Backend | From Version | To Version | Result |\n')
out.write('| --- | --- | --- | --- | --- |\n')
for topo, backend, old_v, new_v, npass, nfail in topo_rows:
result = 'PASS' if nfail == '0' else 'FAIL'
out.write(f'| {topo} | {backend} | {old_v} | {new_v} | {result} (PASS={npass} FAIL={nfail}) |\n')
if not topo_rows:
out.write('| - | - | - | - | NOT RUN (suite failed before upgrade) |\n')
PY
{ {
echo "# RustFS upgrade compatibility report" echo "# RustFS upgrade compatibility report"
echo "" echo ""
echo "- Run: ${{ github.server_url }}/${{ github.repository }}/actions/runs/${{ github.run_id }}" echo "- Run: ${{ github.server_url }}/${{ github.repository }}/actions/runs/${{ github.run_id }}"
echo "- Attempt: ${GITHUB_RUN_ATTEMPT}"
echo "- Workflow Commit: ${GITHUB_SHA}"
echo "- Trigger: ${{ github.event_name }}" echo "- Trigger: ${{ github.event_name }}"
echo "- From: ${FROM_SOURCE}" echo "- From: ${FROM_SOURCE}"
echo "- To: ${TO_SOURCE}" echo "- To: ${TO_SOURCE}"
echo "- Test Step Outcome: ${RESULT}" echo "- Test Step Outcome: ${{ steps.test.outcome }}"
echo "- Suite Step Outcome: ${{ steps.test.outcome }}"
echo "" echo ""
if [ "${RESULT}" = "success" ]; then cat "${MATRIX_TABLE}" || true
cat "${MATRIX_TABLE}" echo ""
echo "" cat "${CASE_TABLE}" || true
cat "${CASE_TABLE}" echo ""
echo "" echo "## Log tail"
echo "## Log tail" echo '```text'
echo '```text' tail -n 200 "${LOG_FILE}" || true
tail -n 200 "${LOG_FILE}" echo '```'
echo '```'
else
echo "The suite or evidence validation failed. See this run's artifact for partial case results and suite.log."
fi
} | tee "${REPORT_FILE}" } | tee "${REPORT_FILE}"
cat "${REPORT_FILE}" >> "${GITHUB_STEP_SUMMARY}" cat "${REPORT_FILE}" >> "${GITHUB_STEP_SUMMARY}"
[ "${RESULT}" = "success" ]
- name: Upload functional report to dashboard - name: Upload functional report to dashboard
if: ${{ always() && steps.evidence.outcome == 'success' }} if: always()
continue-on-error: true continue-on-error: true
env: env:
GH_TOKEN: ${{ env.PF_TESTING_GH_TOKEN }} GH_TOKEN: ${{ env.PF_TESTING_GH_TOKEN }}
REPORT_FILE: /tmp/rustfs-upgrade-report.md
SUITE: upgrade SUITE: upgrade
run: | run: |
set -euo pipefail set -euo pipefail
@@ -289,32 +308,28 @@ jobs:
fi fi
DATE="$(date -u +%Y-%m-%d)" DATE="$(date -u +%Y-%m-%d)"
REPORT_PATH="functional-reports/${SUITE}/${DATE}.md" REPORT_PATH="functional-reports/${SUITE}/${DATE}.md"
# Base64-encode the report into a temp file and feed it to jq via CONTENT="$(python3 -c 'import base64,sys;print(base64.b64encode(open(sys.argv[1],"rb").read()).decode())' "${REPORT_FILE}")"
# --rawfile: large reports (e.g. pool) exceed the OS argv limit and
# make `jq --arg content "${CONTENT}"` fail with "Argument list too long".
B64_FILE="$(mktemp)"
python3 -c 'import base64,sys;print(base64.b64encode(open(sys.argv[1],"rb").read()).decode())' "${REPORT_FILE}" > "${B64_FILE}"
SHA="$(gh api "repos/rustfs/dashboard/contents/${REPORT_PATH}" -q '.sha' 2>/dev/null || true)" SHA="$(gh api "repos/rustfs/dashboard/contents/${REPORT_PATH}" -q '.sha' 2>/dev/null || true)"
if [ -n "${SHA}" ]; then if [ -n "${SHA}" ]; then
jq -n --arg msg "report(${SUITE}): ${DATE}" --rawfile content "${B64_FILE}" --arg sha "${SHA}" \ jq -n --arg msg "report(${SUITE}): ${DATE}" --arg content "${CONTENT}" --arg sha "${SHA}" \
'{message:$msg, content:($content|rtrimstr("\n")), sha:$sha}' \ '{message:$msg, content:$content, sha:$sha}' \
| gh api --method PUT "repos/rustfs/dashboard/contents/${REPORT_PATH}" --input - >/dev/null | gh api --method PUT "repos/rustfs/dashboard/contents/${REPORT_PATH}" --input - >/dev/null
else else
jq -n --arg msg "report(${SUITE}): ${DATE}" --rawfile content "${B64_FILE}" \ jq -n --arg msg "report(${SUITE}): ${DATE}" --arg content "${CONTENT}" \
'{message:$msg, content:($content|rtrimstr("\n"))}' \ '{message:$msg, content:$content}' \
| gh api --method PUT "repos/rustfs/dashboard/contents/${REPORT_PATH}" --input - >/dev/null | gh api --method PUT "repos/rustfs/dashboard/contents/${REPORT_PATH}" --input - >/dev/null
fi fi
rm -f "${B64_FILE}"
- name: File failure issue in rustfs/backlog - name: File failure issue in rustfs/backlog
if: ${{ always() && (failure() || steps.test.outcome == 'failure' || steps.test.outcome == 'cancelled') }} if: ${{ always() && (failure() || steps.test.outcome == 'failure' || steps.test.outcome == 'cancelled') }}
continue-on-error: true continue-on-error: true
env: env:
GH_TOKEN: ${{ secrets.PF_TESTING_GH_TOKEN }} GH_TOKEN: ${{ secrets.PF_TESTING_GH_TOKEN }}
EVIDENCE_OUTCOME: ${{ steps.evidence.outcome }}
SUITE: 'upgrade' SUITE: 'upgrade'
SUITE_LABEL: 'Upgrade compatibility' SUITE_LABEL: 'Upgrade compatibility'
RUN_URL: ${{ github.server_url }}/${{ github.repository }}/actions/runs/${{ github.run_id }} RUN_URL: ${{ github.server_url }}/${{ github.repository }}/actions/runs/${{ github.run_id }}
REPORT_FILE: '/tmp/rustfs-upgrade-report.md'
LOG_FILE: '/tmp/rustfs-upgrade.log'
run: | run: |
set -euo pipefail set -euo pipefail
if [ -z "${GH_TOKEN:-}" ]; then if [ -z "${GH_TOKEN:-}" ]; then
@@ -342,16 +357,14 @@ jobs:
echo "" echo ""
echo "- Suite: \`${SUITE}\`" echo "- Suite: \`${SUITE}\`"
echo "- Run: ${RUN_URL}" echo "- Run: ${RUN_URL}"
echo "- Attempt: ${GITHUB_RUN_ATTEMPT}"
echo "- Workflow Commit: ${GITHUB_SHA}"
echo "- Trigger: ${GITHUB_EVENT_NAME}" echo "- Trigger: ${GITHUB_EVENT_NAME}"
echo "- Date: $(date -u +%Y-%m-%d)" echo "- Date: $(date -u +%Y-%m-%d)"
echo "" echo ""
echo "## Report (errors and symptoms)" echo "## Report (errors and symptoms)"
echo "" echo ""
if [ "${EVIDENCE_OUTCOME}" = "success" ] && [ -s "${REPORT_FILE}" ]; then if [ -s "${REPORT_FILE}" ]; then
redact < "${REPORT_FILE}" redact < "${REPORT_FILE}"
elif [ "${EVIDENCE_OUTCOME}" = "success" ] && [ -s "${LOG_FILE:-}" ]; then elif [ -s "${LOG_FILE:-}" ]; then
echo "(report file missing; log tail below)" echo "(report file missing; log tail below)"
echo "" echo ""
tail -n 200 "${LOG_FILE}" | redact tail -n 200 "${LOG_FILE}" | redact
@@ -367,16 +380,14 @@ jobs:
echo "filed backlog issue for suite ${SUITE}" echo "filed backlog issue for suite ${SUITE}"
- name: Upload report and logs - name: Upload report and logs
if: ${{ always() && steps.evidence.outcome == 'success' }} if: always()
uses: actions/upload-artifact@b7c566a772e6b6bfb58ed0dc250532a479d7789f # v6 uses: actions/upload-artifact@b7c566a772e6b6bfb58ed0dc250532a479d7789f # v6
with: with:
name: rustfs-upgrade-test-${{ github.run_id }}-${{ github.run_attempt }} name: rustfs-upgrade-test-${{ github.run_id }}
path: | path: |
${{ env.FUNCTIONAL_ARTIFACTS_DIR }}/report.md /tmp/rustfs-upgrade-report.md
${{ env.FUNCTIONAL_ARTIFACTS_DIR }}/suite.log /tmp/rustfs-upgrade.*/*
${{ env.FUNCTIONAL_ARTIFACTS_DIR }}/cases.md if-no-files-found: ignore
${{ env.FUNCTIONAL_ARTIFACTS_DIR }}/matrix.md
if-no-files-found: error
retention-days: 3 retention-days: 3
- name: Cleanup environment (after) - name: Cleanup environment (after)
@@ -42,7 +42,6 @@ jobs:
- name: Check latest scheduled runs - name: Check latest scheduled runs
env: env:
GH_TOKEN: ${{ secrets.GITHUB_TOKEN }} GH_TOKEN: ${{ secrets.GITHUB_TOKEN }}
RUSTFS_DEFAULT_BRANCH: ${{ github.event.repository.default_branch }}
run: | run: |
set +e set +e
python3 scripts/check_scheduled_validation_freshness.py \ python3 scripts/check_scheduled_validation_freshness.py \
-1
View File
@@ -33,7 +33,6 @@ profile.json
*.zst *.zst
.secrets .secrets
*.go *.go
!crates/zip/tests/fixtures/snowball/**/generate/*.go
*.pb *.pb
*.svg *.svg
deploy/logs/*.log.* deploy/logs/*.log.*
+3 -3
View File
@@ -3,9 +3,9 @@
repos: repos:
- repo: local - repo: local
hooks: hooks:
- id: rustfs-fmt-check - id: rustfs-dev-check
name: Rust formatting name: rustfs dev-check
entry: cargo fmt --all --check entry: make dev-check
language: system language: system
types: [rust] types: [rust]
pass_filenames: false pass_filenames: false
+1 -4
View File
@@ -18,10 +18,7 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0
- Read paths: an object at or below `policy.inline_max_bytes` (16 MiB by default) is teed to the client and to the local store in a single source read; a larger object or a Range read streams through and a background pull stores the whole object. A HEAD miss is proxied to the source and stores nothing (`policy.head = local_only` disables it). Every source-backed response carries `x-rustfs-on-demand-migration: source` - Read paths: an object at or below `policy.inline_max_bytes` (16 MiB by default) is teed to the client and to the local store in a single source read; a larger object or a Range read streams through and a background pull stores the whole object. A HEAD miss is proxied to the source and stores nothing (`policy.head = local_only` disables it). Every source-backed response carries `x-rustfs-on-demand-migration: source`
- Protections: a per-source circuit breaker, a per-key negative cache, singleflight per key, a concurrency limit and a bounded pull queue shared by the inline and background paths, an optional bandwidth limit, an anti-loop request marker, and the shared outbound-endpoint (SSRF) policy - Protections: a per-source circuit breaker, a per-key negative cache, singleflight per key, a concurrency limit and a bounded pull queue shared by the inline and background paths, an optional bandwidth limit, an anti-loop request marker, and the shared outbound-endpoint (SSRF) policy
- Metrics under `rustfs_on_demand_migration_*` (`requests_total`, `pulled_bytes_total`, `pulled_objects_total`, `pull_failures_total`, `inflight_pulls`, `queue_depth`, `source_latency_seconds_*`, `breaker_state`), mirrored per node by the admin status route - Metrics under `rustfs_on_demand_migration_*` (`requests_total`, `pulled_bytes_total`, `pulled_objects_total`, `pull_failures_total`, `inflight_pulls`, `queue_depth`, `source_latency_seconds_*`, `breaker_state`), mirrored per node by the admin status route
- Listings: `ListObjects` v1 remains local with ordinary key markers. `ListObjectsV2` can merge source objects when `policy.list_through = true`; this is off by default - Limitations: listings show only local objects (the source is not merged into `ListObjectsV2`); PUT and DELETE never reach the source; a source object updated after it was pulled is not re-fetched; SSE-C source objects are unsupported and answer 424; `Last-Modified` on a pulled object is the local write time, with the source timestamp kept in metadata
- Upgrade and rollback: finish upgrading every node before enabling ODM. An rc.5 node that writes bucket configuration drops the ODM fields from metadata; neither a later restart nor moving the service out of ECStore recovers them. Before rollback, disable ODM and securely retain the original full configuration and credentials. After every node returns to a compatible version, restore and validate that configuration. Redacted exports cannot replace the credential backup; source-only objects are unavailable through RustFS while ODM is disabled. See the upgrade and rollback section of `docs/operations/on-demand-migration.md`
- Optional Google dependencies: default and `full` server builds retain native GCS support. `cargo build -p rustfs --no-default-features --features ftps,webdav` excludes Google SDKs while preserving configuration decoding and redaction; native GCS ODM and tier operations require the `gcs` feature. Do not use that build with existing GCS-tiered data
- Limitations: PUT and DELETE never reach the source; a source object updated after it was pulled is not re-fetched; SSE-C source objects are unsupported and answer 424; `Last-Modified` on a pulled object is the local write time, with the source timestamp kept in metadata
- **NATS JetStream Publish Path**: Opt-in at-least-once delivery for the NATS notify and audit targets. A NATS Core publish flushes to the connection without awaiting a broker acknowledgement, so an event can be lost across a broker restart or a reconnect after the send queue has already cleared it. A queued event now clears only after the JetStream `PublishAck`, so bucket notifications survive those interruptions. Off by default and byte-identical to the NATS Core path when disabled. - **NATS JetStream Publish Path**: Opt-in at-least-once delivery for the NATS notify and audit targets. A NATS Core publish flushes to the connection without awaiting a broker acknowledgement, so an event can be lost across a broker restart or a reconnect after the send queue has already cleared it. A queued event now clears only after the JetStream `PublishAck`, so bucket notifications survive those interruptions. Off by default and byte-identical to the NATS Core path when disabled.
- Three configuration keys per target: `JETSTREAM_ENABLE`, `JETSTREAM_STREAM_NAME`, and `JETSTREAM_ACK_TIMEOUT_SECS`, under the `RUSTFS_NOTIFY_NATS_` and `RUSTFS_AUDIT_NATS_` prefixes - Three configuration keys per target: `JETSTREAM_ENABLE`, `JETSTREAM_STREAM_NAME`, and `JETSTREAM_ACK_TIMEOUT_SECS`, under the `RUSTFS_NOTIFY_NATS_` and `RUSTFS_AUDIT_NATS_` prefixes
- Durable store-and-forward with a stable dedup id sent as the `Nats-Msg-Id` header, so a replay after a crash is collapsed by the server duplicate window - Durable store-and-forward with a stable dedup id sent as the `Nats-Msg-Id` header, so a replay after a crash is collapsed by the server duplicate window
+37 -11
View File
@@ -109,17 +109,24 @@ affected boundaries and risks. CI still runs its configured repository gates.
### 🔒 Git Pre-commit Hooks (optional) ### 🔒 Git Pre-commit Hooks (optional)
The optional hook uses the checked-in `.pre-commit-config.yaml`. Install [pre-commit](https://pre-commit.com/#installation), then run this from the checkout or a linked worktree: Git hooks are **not** versioned in this repository, so a fresh clone has no
active pre-commit hook. If you add your own `.git/hooks/pre-commit` (a good
choice is a one-liner that runs `make pre-commit`), you can mark it executable
with:
```bash ```bash
make setup-hooks make setup-hooks
``` ```
The hook runs `cargo fmt --all --check` when staged files include Rust source. It does not compile the workspace or run tests. Fix formatting with `cargo fmt --all`, inspect and stage the result, then commit again. Or manually:
`pre-commit install` resolves Git's hook directory for linked worktrees and preserves an existing hook in migration mode. If you use `core.hooksPath`, keep that hook manager and integrate `pre-commit run` there; the installer refuses to silently replace that configuration. ```bash
chmod +x .git/hooks/pre-commit
```
A local hook provides early formatting feedback. With or without it, follow the verification tiers in `AGENTS.md`, run relevant behavioral tests, and satisfy the CI merge gates. `make pre-commit` and `make dev-check` remain explicit broader commands. With or without a hook, follow the verification tiers in `AGENTS.md`. Run the
applicable scoped checks, and reserve `make pre-pr` for broad cross-module
changes whose impact cannot be bounded by those checks.
### 📝 Formatting Configuration ### 📝 Formatting Configuration
@@ -131,11 +138,31 @@ fn_call_width = 90
single_line_let_else_max_width = 100 single_line_let_else_max_width = 100
``` ```
### 🚫 Commit Prevention
If you set up a pre-commit hook and your code doesn't meet the formatting requirements, the hook will:
1. **Block the commit** and show clear error messages
2. **Provide exact commands** to fix the issues
3. **Guide you through** the resolution process
Example output when formatting fails:
```
❌ Code formatting check failed!
💡 Please run 'cargo fmt --all' to format your code before committing.
🔧 Quick fix:
cargo fmt --all
git add .
git commit
```
### 🔄 Development Workflow ### 🔄 Development Workflow
1. **Make your changes** 1. **Make your changes**
2. **Format your code**: `make fmt` or `cargo fmt --all` 2. **Format your code**: `make fmt` or `cargo fmt --all`
3. **Select relevant checks** using the validation tier in `AGENTS.md`; use `make pre-commit` when its broader fast gate adds useful coverage 3. **Run the fast gate**: `make pre-commit` (no clippy, no tests)
4. **Commit your changes**: `git commit -m "your message"` 4. **Commit your changes**: `git commit -m "your message"`
5. **Complete the applicable multi-role adversarial review** for non-exempt changes (see `AGENTS.md`) 5. **Complete the applicable multi-role adversarial review** for non-exempt changes (see `AGENTS.md`)
6. **Run applicable scoped checks before opening/updating a PR**; consider 6. **Run applicable scoped checks before opening/updating a PR**; consider
@@ -179,12 +206,11 @@ Configure your IDE to:
#### Pre-commit hook not running? #### Pre-commit hook not running?
```bash ```bash
pre-commit validate-config # Check if hook is executable
pre-commit run --all-files ls -la .git/hooks/pre-commit
# Inspect any configured hook manager; do not overwrite it.
git config --get core.hooksPath # Make it executable if needed
# Install if no separate hook manager is configured. chmod +x .git/hooks/pre-commit
make setup-hooks
``` ```
#### Formatting issues? #### Formatting issues?
Generated
+111 -356
View File
File diff suppressed because it is too large Load Diff
+12 -17
View File
@@ -168,7 +168,7 @@ reqwest = "0.13.4"
rustfs-kafka-async = { version = "1.3.1" } rustfs-kafka-async = { version = "1.3.1" }
socket2 = { version = "0.6.5" } socket2 = { version = "0.6.5" }
tokio = { version = "1.53.1" } tokio = { version = "1.53.1" }
tokio-rustls = { default-features = false, version = "0.26.5" } tokio-rustls = { default-features = false, version = "0.26.4" }
tokio-stream = { version = "0.1.19" } tokio-stream = { version = "0.1.19" }
tokio-test = "0.4.5" tokio-test = "0.4.5"
tokio-util = { version = "0.7.19" } tokio-util = { version = "0.7.19" }
@@ -199,10 +199,10 @@ serde_urlencoded = "0.7.1"
# matching stable releases are not available yet, while previous stable lines # matching stable releases are not available yet, while previous stable lines
# have incompatible APIs. Keep them exact-pinned and monitor upstream for stable # have incompatible APIs. Keep them exact-pinned and monitor upstream for stable
# releases. # releases.
aes-gcm = { version = "0.11.1" } aes-gcm = { version = "=0.11.1" }
argon2 = { version = "0.6.0" } argon2 = { version = "=0.6.0" }
blake2 = "0.11.0" blake2 = "=0.11.0"
chacha20poly1305 = { version = "0.11.0" } chacha20poly1305 = { version = "=0.11.0" }
crc-fast = "1.10.0" crc-fast = "1.10.0"
hmac = { version = "0.13.0" } hmac = { version = "0.13.0" }
jsonwebtoken = { version = "11.0.0" } jsonwebtoken = { version = "11.0.0" }
@@ -234,19 +234,15 @@ tokio-postgres-rustls = "0.14.0"
# Utilities and Tools # Utilities and Tools
anyhow = "1.0.104" anyhow = "1.0.104"
arc-swap = "1.9.2" arc-swap = "1.9.2"
# RUSTFS_COMPAT_TODO(tokio-tar-extension-limits): keep the fork pin while Snowball and Swift still depend on it. Remove after Snowball uses a released tar-codec/tar-framing API that exposes precedence-resolved MinIO vendor records, RustFS preserves cancellation-safe ownership of large streamed members, footerless minio-go input is accepted only at an authenticated complete request boundary, the existing resource-limit, cancellation, and error-fuse regressions pass, and Swift no longer needs this fork. # RUSTFS_COMPAT_TODO(tokio-tar-extension-limits): keep the fork pin until every parser hardening used by Snowball is released upstream. Remove after astral-sh/tokio-tar#118 is merged and a published release includes extension, physical-entry, and sparse limits, cancellation-safe sparse parsing, and error-fused entry streams.
astral-tokio-tar = { git = "https://github.com/cxymds/tokio-tar.git", rev = "603756478b7668436e464519c77ccac22a99ba96" } astral-tokio-tar = { git = "https://github.com/cxymds/tokio-tar.git", rev = "603756478b7668436e464519c77ccac22a99ba96" }
# Candidate Snowball parser versions exercised by rustfs-zip compatibility fixtures.
tar-codec = "0.0.14"
tar-framing = "0.0.14"
atoi = "3.1.0" atoi = "3.1.0"
atomic_enum = "0.3.0" atomic_enum = "0.3.0"
aws-config = { version = "1.12.0" } aws-config = { version = "1.11.0" }
aws-credential-types = { version = "1.3.0" } aws-credential-types = { version = "1.3.0" }
aws-sdk-kms = { default-features = false, version = "1.118.0" } aws-sdk-kms = { default-features = false, version = "1.117.0" }
aws-sdk-s3 = { default-features = false, version = "1.145.0" } aws-sdk-s3 = { default-features = false, version = "1.144.0" }
aws-sdk-sts = { default-features = false, version = "1.114.0" } aws-sdk-sts = { default-features = false, version = "1.113.0" }
aws-smithy-async = { version = "1.3.0" }
aws-smithy-http-client = { default-features = false, version = "1.4.0" } aws-smithy-http-client = { default-features = false, version = "1.4.0" }
aws-smithy-runtime-api = { version = "1.16.0" } aws-smithy-runtime-api = { version = "1.16.0" }
aws-smithy-types = { version = "1.6.3" } aws-smithy-types = { version = "1.6.3" }
@@ -343,7 +339,7 @@ windows = { version = "0.62.2" }
windows-sys = "0.61.2" windows-sys = "0.61.2"
xxhash-rust = { version = "0.8.18" } xxhash-rust = { version = "0.8.18" }
zip = "8.6.0" zip = "8.6.0"
zstd = "0.14.0" zstd = "0.13.3"
# Observability and Metrics # Observability and Metrics
metrics = "0.24.6" metrics = "0.24.6"
@@ -371,8 +367,7 @@ dav-server = "0.11.0"
# Performance Analysis and Memory Profiling # Performance Analysis and Memory Profiling
rustfs-mimalloc = { version = "0.5.3" } rustfs-mimalloc = { version = "0.5.3" }
# Preserve Unicode focus filters until rustfs/backlog#2302 is resolved. hotpath = { version = "0.25.0", default-features = false }
hotpath = { version = "=0.25.0", default-features = false }
# Snapshot testing for output format regression detection # Snapshot testing for output format regression detection
insta = { version = "1.48" } insta = { version = "1.48" }
-15
View File
@@ -130,21 +130,6 @@ Scanner cycle budget controls:
- timeout returns S3 `SlowDown`, so clients should use normal SDK retry handling. - timeout returns S3 `SlowDown`, so clients should use normal SDK retry handling.
- this is not a fdatasync or group-commit switch. Track fdatasync batching separately with `rustfs_s3_put_object_rename_fdatasync_batch_files`. - this is not a fdatasync or group-commit switch. Track fdatasync batching separately with `rustfs_s3_put_object_rename_fdatasync_batch_files`.
## Remote tier timeout environment variables
- `RUSTFS_TIER_REMOTE_CONNECT_TIMEOUT_SECS`
- remote tier TCP connect timeout.
- default is `10`.
- must be positive; zero fails tier client initialization, while an invalid integer is logged and falls back to the default.
- `RUSTFS_TIER_REMOTE_REQUEST_TIMEOUT_SECS`
- remote tier request timeout through response headers.
- default is `86400` so large transition uploads keep a production-safe budget.
- must be positive; zero fails tier client initialization, while an invalid integer is logged and falls back to the default. Very large values are accepted and act as a correspondingly long budget.
- `RUSTFS_TIER_REMOTE_RESPONSE_BODY_IDLE_TIMEOUT_SECS`
- maximum idle time between remote tier response-body chunks.
- default is `60`; the timer resets only when non-empty body data keeps progressing.
- must be positive; zero fails tier client initialization, while an invalid integer is logged and falls back to the default.
## Drive timeout environment variables ## Drive timeout environment variables
- `RUSTFS_DRIVE_METADATA_TIMEOUT_SECS` - `RUSTFS_DRIVE_METADATA_TIMEOUT_SECS`
-32
View File
@@ -137,28 +137,6 @@ pub const DEFAULT_TIER_REMOTE_VERSION_STATE_FLEET_CONFIRMED: bool = false;
const _: () = assert!(!DEFAULT_TIER_REMOTE_VERSION_STATE_WRITE); const _: () = assert!(!DEFAULT_TIER_REMOTE_VERSION_STATE_WRITE);
const _: () = assert!(!DEFAULT_TIER_REMOTE_VERSION_STATE_FLEET_CONFIRMED); const _: () = assert!(!DEFAULT_TIER_REMOTE_VERSION_STATE_FLEET_CONFIRMED);
/// Environment variable for remote tier TCP connect timeout in seconds.
pub const ENV_TIER_REMOTE_CONNECT_TIMEOUT_SECS: &str = "RUSTFS_TIER_REMOTE_CONNECT_TIMEOUT_SECS";
/// Default remote tier TCP connect timeout in seconds.
pub const DEFAULT_TIER_REMOTE_CONNECT_TIMEOUT_SECS: u64 = 10;
/// Environment variable for the remote tier request timeout in seconds.
///
/// This bounds upload/download request progress through response headers. The
/// default is intentionally large so multi-TiB transition uploads keep their
/// previous production budget while black-hole remotes no longer wait forever.
pub const ENV_TIER_REMOTE_REQUEST_TIMEOUT_SECS: &str = "RUSTFS_TIER_REMOTE_REQUEST_TIMEOUT_SECS";
/// Default remote tier request timeout in seconds.
pub const DEFAULT_TIER_REMOTE_REQUEST_TIMEOUT_SECS: u64 = 24 * 60 * 60;
/// Environment variable for remote tier response-body idle timeout in seconds.
///
/// The timer is re-armed on every non-empty response-body chunk, so slow but
/// progressing remotes can continue while silent response bodies are cancelled.
pub const ENV_TIER_REMOTE_RESPONSE_BODY_IDLE_TIMEOUT_SECS: &str = "RUSTFS_TIER_REMOTE_RESPONSE_BODY_IDLE_TIMEOUT_SECS";
/// Default remote tier response-body idle timeout in seconds.
pub const DEFAULT_TIER_REMOTE_RESPONSE_BODY_IDLE_TIMEOUT_SECS: u64 = 60;
/// Request the object-transaction fencing contract used by storage-owned /// Request the object-transaction fencing contract used by storage-owned
/// cleanup receipts and lock-window optimizations. /// cleanup receipts and lock-window optimizations.
/// ///
@@ -834,16 +812,6 @@ mod remote_version_state_tests {
); );
} }
#[test]
fn remote_tier_timeout_env_names_are_stable() {
assert_eq!(super::ENV_TIER_REMOTE_CONNECT_TIMEOUT_SECS, "RUSTFS_TIER_REMOTE_CONNECT_TIMEOUT_SECS");
assert_eq!(super::ENV_TIER_REMOTE_REQUEST_TIMEOUT_SECS, "RUSTFS_TIER_REMOTE_REQUEST_TIMEOUT_SECS");
assert_eq!(
super::ENV_TIER_REMOTE_RESPONSE_BODY_IDLE_TIMEOUT_SECS,
"RUSTFS_TIER_REMOTE_RESPONSE_BODY_IDLE_TIMEOUT_SECS"
);
}
#[test] #[test]
fn data_movement_part_checksum_gate_uses_stable_environment_names() { fn data_movement_part_checksum_gate_uses_stable_environment_names() {
assert_eq!(super::ENV_DATA_MOVEMENT_PART_CHECKSUMS_WRITE, "RUSTFS_DATA_MOVEMENT_PART_CHECKSUMS_WRITE"); assert_eq!(super::ENV_DATA_MOVEMENT_PART_CHECKSUMS_WRITE, "RUSTFS_DATA_MOVEMENT_PART_CHECKSUMS_WRITE");
@@ -20,10 +20,9 @@
//! journal (`count_requests`) carries the assertion in every one of them. //! journal (`count_requests`) carries the assertion in every one of them.
use super::common::{BoxError, OdmTestEnv, RawResponse, SeedObject, start_configured_env}; use super::common::{BoxError, OdmTestEnv, RawResponse, SeedObject, start_configured_env};
use crate::fake_s3_target::{FaultAction, Operation}; use crate::fake_s3_target::Operation;
use aws_sdk_s3::types::{BucketVersioningStatus, VersioningConfiguration}; use aws_sdk_s3::types::{BucketVersioningStatus, VersioningConfiguration};
use bytes::Bytes; use bytes::Bytes;
use futures::{StreamExt, TryStreamExt};
use std::time::Duration; use std::time::Duration;
type TestResult = Result<(), BoxError>; type TestResult = Result<(), BoxError>;
@@ -146,38 +145,14 @@ async fn test_odm_range_burst_overflows_the_pull_queue_without_failing_clients()
.await?; .await?;
let body = payload(128 * 1024); let body = payload(128 * 1024);
let blocker = "queue/blocker.bin";
env.seed_source(SOURCE_BUCKET, &[SeedObject::new(blocker, body.clone())]);
// The one-chunk range completes immediately; its full background pull
// occupies the only slot while the remaining requests fill the queue.
env.source.inject_for_key(
Operation::GetObject,
blocker,
FaultAction::SlowSendBody {
chunk_bytes: 1024,
delay: Duration::from_millis(100),
},
2,
);
let response = env
.raw_object_request(http::Method::GET, bucket, blocker, &[("range", "bytes=0-1023")])
.await?;
assert_eq!(response.status, 206);
assert_eq!(response.body, body.slice(0..1024));
env.wait_for_status_counter(bucket, "/inflight_pulls", 1, SETTLE).await?;
let keys: Vec<String> = (0..REQUESTS).map(|index| format!("queue/object-{index:03}.bin")).collect(); let keys: Vec<String> = (0..REQUESTS).map(|index| format!("queue/object-{index:03}.bin")).collect();
let seeds: Vec<SeedObject> = keys.iter().map(|key| SeedObject::new(key.clone(), body.clone())).collect(); let seeds: Vec<SeedObject> = keys.iter().map(|key| SeedObject::new(key.clone(), body.clone())).collect();
env.seed_source(SOURCE_BUCKET, &seeds); env.seed_source(SOURCE_BUCKET, &seeds);
// Bound source connections below the fixture's limit while still let responses: Vec<RawResponse> = futures::future::try_join_all(
// submitting all 100 requests to the eight-slot background queue.
let responses: Vec<RawResponse> = futures::stream::iter(
keys.iter() keys.iter()
.map(|key| env.raw_object_request(http::Method::GET, bucket, key, &[("range", "bytes=0-1023")])), .map(|key| env.raw_object_request(http::Method::GET, bucket, key, &[("range", "bytes=0-1023")])),
) )
.buffered(16)
.try_collect()
.await?; .await?;
for (key, response) in keys.iter().zip(&responses) { for (key, response) in keys.iter().zip(&responses) {
assert_eq!(response.status, 206, "{key}: {}", String::from_utf8_lossy(&response.body)); assert_eq!(response.status, 206, "{key}: {}", String::from_utf8_lossy(&response.body));
@@ -193,15 +168,6 @@ async fn test_odm_range_burst_overflows_the_pull_queue_without_failing_clients()
.wait_for_status_counter(bucket, "/counters/pull_failures_total/queue_full", 1, SETTLE) .wait_for_status_counter(bucket, "/counters/pull_failures_total/queue_full", 1, SETTLE)
.await?; .await?;
assert!(queue_full > 0, "a 100-deep burst must overflow an 8-slot queue"); assert!(queue_full > 0, "a 100-deep burst must overflow an 8-slot queue");
let queue_full = usize::try_from(queue_full)?;
assert!(queue_full <= REQUESTS);
env.wait_for_status_counter(
bucket,
"/counters/pulled_objects_total/background",
u64::try_from(REQUESTS + 1 - queue_full)?,
SETTLE,
)
.await?;
let ranged_reads: usize = keys.iter().map(|key| source_get_count(&env, key)).sum(); let ranged_reads: usize = keys.iter().map(|key| source_get_count(&env, key)).sum();
assert!( assert!(
@@ -209,6 +175,9 @@ async fn test_odm_range_burst_overflows_the_pull_queue_without_failing_clients()
"every reader is served from the source: {ranged_reads} GETs for {REQUESTS} readers" "every reader is served from the source: {ranged_reads} GETs for {REQUESTS} readers"
); );
let dropped = keys.iter().filter(|key| source_get_count(&env, key) == 1).count(); let dropped = keys.iter().filter(|key| source_get_count(&env, key) == 1).count();
assert_eq!(dropped, queue_full, "only overflowed keys remain without a background GET"); assert!(
dropped > 0,
"the overflowed keys are the ones with no backfill GET, but every key got one"
);
Ok(()) Ok(())
} }
@@ -265,13 +265,16 @@ async fn list_through_rejects_a_tampered_continuation_token() -> TestResult {
let decoded = String::from_utf8(base64_simd::STANDARD.decode_to_vec(token.as_bytes())?)?; let decoded = String::from_utf8(base64_simd::STANDARD.decode_to_vec(token.as_bytes())?)?;
assert!(decoded.contains("\"t\":\"odm-list\""), "the merged token is an envelope: {decoded}"); assert!(decoded.contains("\"t\":\"odm-list\""), "the merged token is an envelope: {decoded}");
let tampered = base64_simd::STANDARD.encode_to_string(decoded.replace("\"v\":1", "\"v\":3").as_bytes()); let tampered = base64_simd::STANDARD.encode_to_string(decoded.replace("\"v\":1", "\"v\":2").as_bytes());
assert_ne!(tampered, token, "the test must change the token version"); let rejected = env
let query = serde_urlencoded::to_string([("continuation-token", tampered.as_str())])?; .raw_list_objects_v2(bucket, &format!("continuation-token={tampered}"))
let rejected = env.raw_list_objects_v2(bucket, &query).await?; .await?;
let error_body = String::from_utf8_lossy(&rejected.body); assert_eq!(
assert_eq!(rejected.status, 400, "a bumped token version is a client error: {}", error_body); rejected.status,
assert!(error_body.contains("<Code>InvalidArgument</Code>"), "{error_body}"); 400,
"a bumped token version is a client error: {}",
String::from_utf8_lossy(&rejected.body)
);
Ok(()) Ok(())
} }
@@ -6743,99 +6743,6 @@ async fn test_site_replication_replicates_object_with_bucket_versioning_real_dua
Ok(()) Ok(())
} }
#[tokio::test]
async fn test_site_replication_replays_bucket_created_during_peer_outage_real_dual_node() -> TestResult {
init_logging();
// Keep compilation outside the scenario timeout. Recovery itself waits
// for the production 30-second lightweight retry tick.
let _rustfs_binary = rustfs_binary_path();
match timeout(Duration::from_secs(150), async {
let mut site_env = replication_fast_env();
site_env.extend_from_slice(LOOPBACK_REPLICATION_TARGET_ENV);
let mut site_a_env = RustFSTestEnvironment::new().await?;
site_a_env.start_rustfs_server_with_env(vec![], &site_env).await?;
let mut site_b_env = RustFSTestEnvironment::new().await?;
site_b_env.start_rustfs_server_without_cleanup_with_env(&site_env).await?;
let site_a_client = site_a_env.create_s3_client();
let site_b_client = site_b_env.create_s3_client();
let bucket = "site-repl-peer-outage";
let key = "after-recovery.txt";
let payload = b"site replication recovered the missed bucket".to_vec();
let add_status = site_replication_add(
&site_a_env,
&[
PeerSite {
name: "outage-site-a".to_string(),
endpoint: site_a_env.url.clone(),
access_key: site_a_env.access_key.clone(),
secret_key: site_a_env.secret_key.clone(),
..Default::default()
},
PeerSite {
name: "outage-site-b".to_string(),
endpoint: site_b_env.url.clone(),
access_key: site_b_env.access_key.clone(),
secret_key: site_b_env.secret_key.clone(),
..Default::default()
},
],
)
.await?;
assert!(add_status.success, "unexpected site add result: {add_status:?}");
wait_for_site_replication_enabled(&site_a_env, 2).await?;
wait_for_site_replication_enabled(&site_b_env, 2).await?;
site_b_env.stop_server();
site_a_client.create_bucket().bucket(bucket).send().await?;
site_a_client.head_bucket().bucket(bucket).send().await?;
let queued = site_replication_info(&site_a_env)
.await?
.retry_stats
.ok_or("peer outage did not persist a site replication retry event")?;
assert!(queued.pending + queued.failed > 0, "peer outage retry queue was unexpectedly empty");
site_b_env.restart_server_preserving_data(vec![], &site_env).await?;
let recovery_deadline = tokio::time::Instant::now() + Duration::from_secs(75);
loop {
let bucket_recovered = site_b_client.head_bucket().bucket(bucket).send().await.is_ok();
let queue_empty = site_replication_info(&site_a_env).await?.retry_stats.is_none();
if bucket_recovered && queue_empty {
break;
}
if tokio::time::Instant::now() >= recovery_deadline {
return Err(format!(
"site replication retry did not settle after peer recovery; bucket_recovered={bucket_recovered}, queue_empty={queue_empty}"
)
.into());
}
sleep(Duration::from_millis(250)).await;
}
site_a_client
.put_object()
.bucket(bucket)
.key(key)
.body(ByteStream::from(payload.clone()))
.send()
.await?;
assert_eq!(wait_for_object_on_target(&site_b_client, bucket, key).await?, payload);
Ok(())
})
.await
{
Ok(result) => result,
Err(_) => Err("site replication peer-outage recovery timed out after 150 seconds".into()),
}
}
/// Re-applying a site's own replication config must not disable the peer's reverse direction. /// Re-applying a site's own replication config must not disable the peer's reverse direction.
/// ///
/// `PutBucketReplication` broadcasts the config to every peer — the console's replication /// `PutBucketReplication` broadcasts the config to every peer — the console's replication
@@ -12,34 +12,21 @@
// See the License for the specific language governing permissions and // See the License for the specific language governing permissions and
// limitations under the License. // limitations under the License.
use crate::common::{ use crate::common::{RustFSTestClusterEnvironment, RustFSTestEnvironment, init_logging, rustfs_binary_path};
RustFSTestClusterEnvironment, RustFSTestEnvironment, admin_request, init_logging, replication_fast_env, rustfs_binary_path,
};
use crate::fake_s3_target::{BucketMode, FAKE_ACCESS_KEY, FAKE_SECRET_KEY, FakeS3Target};
use crate::on_demand_migration::common::{ODM_SERVER_ENV, OdmTestEnv, SeedObject};
use crate::replication_extension_test::{
LOOPBACK_REPLICATION_TARGET_ENV, ReplicationTargetOptions, put_bucket_replication, set_replication_target_with_options,
};
use aws_sdk_s3::Client; use aws_sdk_s3::Client;
use aws_sdk_s3::error::ProvideErrorMetadata; use aws_sdk_s3::error::ProvideErrorMetadata;
use aws_sdk_s3::primitives::ByteStream; use aws_sdk_s3::primitives::ByteStream;
use aws_sdk_s3::types::{ use aws_sdk_s3::types::{
BucketLifecycleConfiguration, BucketVersioningStatus, CompletedMultipartUpload, CompletedPart, DefaultRetention, BucketVersioningStatus, CompletedMultipartUpload, CompletedPart, ServerSideEncryption, VersioningConfiguration,
ExpirationStatus, LifecycleExpiration, LifecycleRule, LifecycleRuleFilter, ObjectLockConfiguration, ObjectLockEnabled,
ObjectLockRetentionMode, ObjectLockRule, PublicAccessBlockConfiguration, ServerSideEncryption, ServerSideEncryptionByDefault,
ServerSideEncryptionConfiguration, ServerSideEncryptionRule, Tag, Tagging, VersioningConfiguration,
}; };
use http::{Method, StatusCode};
use std::path::{Path, PathBuf}; use std::path::{Path, PathBuf};
use std::time::Duration; use std::time::Duration;
use tokio::task::JoinSet; use tokio::task::JoinSet;
use tokio::time::{Instant, sleep}; use tokio::time::{Instant, sleep};
type TestResult = Result<(), Box<dyn std::error::Error + Send + Sync>>; type TestResult = Result<(), Box<dyn std::error::Error + Send + Sync>>;
type BoxError = Box<dyn std::error::Error + Send + Sync>;
const SOURCE_BINARY_ENV: &str = "RUSTFS_UPGRADE_SOURCE_BINARY"; const SOURCE_BINARY_ENV: &str = "RUSTFS_UPGRADE_SOURCE_BINARY";
const RC5_COMMIT: &str = "40a2470feb567201165a5b809b7598bb4b1f68f5";
const SSE_MASTER_KEY_ENV: &str = "RUSTFS_SSE_S3_MASTER_KEY"; const SSE_MASTER_KEY_ENV: &str = "RUSTFS_SSE_S3_MASTER_KEY";
const SSE_MASTER_KEY: &str = "QkJCQkJCQkJCQkJCQkJCQkJCQkJCQkJCQkJCQkJCQkI="; const SSE_MASTER_KEY: &str = "QkJCQkJCQkJCQkJCQkJCQkJCQkJCQkJCQkJCQkJCQkI=";
const PLAIN_BUCKET: &str = "upgrade-plain-data"; const PLAIN_BUCKET: &str = "upgrade-plain-data";
@@ -53,32 +40,6 @@ const MULTIPART_UPLOADS_PER_WORKER: usize = 16;
// comfortably covers that window plus CI scheduling jitter. // comfortably covers that window plus CI scheduling jitter.
const LISTING_CONVERGENCE_TIMEOUT: Duration = Duration::from_secs(30); const LISTING_CONVERGENCE_TIMEOUT: Duration = Duration::from_secs(30);
// Bucket-configuration upgrade/rollback scenarios (rustfs#7172, #7183, #7089).
const CONFIG_PLAIN_BUCKET: &str = "upgrade-config-plain";
const CONFIG_ENCRYPTED_BUCKET: &str = "upgrade-config-encrypted";
const CONFIG_REPLICATED_BUCKET: &str = "upgrade-config-replicated";
const CONFIG_LOCKED_BUCKET: &str = "upgrade-config-locked";
const CONFIG_REPLICA_BUCKET: &str = "upgrade-config-replica";
const ROLLBACK_BUCKET: &str = "rollback-config-data";
const ROLLBACK_REPLICA_BUCKET: &str = "rollback-config-replica";
const BUCKET_QUOTA_BYTES: u64 = 64 * 1024 * 1024;
const LIFECYCLE_RULE_ID: &str = "upgrade-expire-logs";
const LIFECYCLE_PREFIX: &str = "logs/";
const LIFECYCLE_DAYS: i32 = 30;
const BUCKET_TAG_KEY: &str = "owner";
const BUCKET_TAG_VALUE: &str = "upgrade-compatibility";
const OBJECT_LOCK_DAYS: i32 = 1;
// `set-bucket-quota` answers 503 until the scanner has made the bucket's usage
// authoritative; the quota test uses the same 30s budget.
const QUOTA_READINESS_TIMEOUT: Duration = Duration::from_secs(30);
// Quota admission fails closed while a freshly started server has neither
// authoritative usage nor a persisted degraded baseline for the bucket
// (rustfs#5716), so a write to a quota-enabled bucket is retryable-503 for that
// window. It is a restart property, not an upgrade property — the same window
// opens on the very first start — so the write assertions ride it out instead
// of treating it as an upgrade failure.
const QUOTA_ADMISSION_WARMUP_TIMEOUT: Duration = Duration::from_secs(90);
fn source_binary() -> Result<PathBuf, Box<dyn std::error::Error + Send + Sync>> { fn source_binary() -> Result<PathBuf, Box<dyn std::error::Error + Send + Sync>> {
let path = std::env::var_os(SOURCE_BINARY_ENV) let path = std::env::var_os(SOURCE_BINARY_ENV)
.map(PathBuf::from) .map(PathBuf::from)
@@ -279,93 +240,6 @@ async fn exercise_mixed_cluster(
Ok(()) Ok(())
} }
/// Pins the published old writer's limitation and the supported recovery
/// procedure. This is not a promise that mixed-version ODM is supported.
/// Replace the loss assertion when ODM gains independent persistence;
/// preserving configuration across rc.5 writes is then an improvement.
#[tokio::test]
#[ignore = "requires the pinned 1.0.0-rc.5 release binary"]
async fn rc5_rollback_requires_restoring_odm_configuration() -> TestResult {
init_logging();
let previous_binary = source_binary()?;
let version = tokio::process::Command::new(&previous_binary)
.arg("--version")
.output()
.await?;
assert!(version.status.success(), "previous binary must report its version");
assert!(
String::from_utf8(version.stdout)?.contains(RC5_COMMIT),
"this compatibility scenario requires the published rc.5 writer"
);
let mut env = OdmTestEnv::start().await?;
let bucket = "odm-rc5-rollback";
let source_bucket = "odm-rc5-source";
env.source.create_bucket_with_mode(source_bucket, BucketMode::Unversioned);
env.seed_source(
source_bucket,
&[SeedObject::new(
"source-only",
bytes::Bytes::from_static(b"source read after recovery"),
)],
);
env.rustfs.create_test_bucket(bucket).await?;
let saved_config = env.fake_source_spec(source_bucket);
assert_eq!(env.configure_source(bucket, &saved_config).await?.status, 200);
let before = env.get_config(bucket).await?;
assert_eq!(before.status, 200);
let expected_config = before
.json()?
.get("config")
.cloned()
.ok_or("configuration response omitted config")?;
env.client
.put_object()
.bucket(bucket)
.key("local")
.body(ByteStream::from_static(b"local data survives rollback"))
.send()
.await?;
env.rustfs.restart_server_preserving_data(vec![], ODM_SERVER_ENV).await?;
let restarted = env.get_config(bucket).await?;
assert_eq!(restarted.status, 200, "a current writer preserves ODM across restart");
assert_eq!(restarted.json()?.get("config"), Some(&expected_config));
restart_from_binary(&mut env.rustfs, &previous_binary, &[]).await?;
env.client
.put_bucket_tagging()
.bucket(bucket)
.tagging(
Tagging::builder()
.tag_set(Tag::builder().key("writer").value("rc5").build()?)
.build()?,
)
.send()
.await?;
env.rustfs.restart_server_preserving_data(vec![], ODM_SERVER_ENV).await?;
let missing = env.get_config(bucket).await?;
assert_eq!(missing.status, 404, "rc.5 rewrites metadata without ODM keys");
assert!(missing.body.contains("NoSuchConfiguration"));
assert_eq!(read_object(&env.client, bucket, "local", None).await?.1, b"local data survives rollback");
let tags = env.client.get_bucket_tagging().bucket(bucket).send().await?;
assert!(tags.tag_set().iter().any(|tag| tag.key() == "writer" && tag.value() == "rc5"));
assert_eq!(
env.configure_source(bucket, &saved_config).await?.status,
200,
"restore from saved full configuration"
);
env.rustfs.restart_server_preserving_data(vec![], ODM_SERVER_ENV).await?;
let restored = env.get_config(bucket).await?;
assert_eq!(restored.status, 200, "restored ODM configuration persists");
assert_eq!(restored.json()?.get("config"), Some(&expected_config));
env.wait_until_source_consulted(bucket).await?;
assert_eq!(
read_object(&env.client, bucket, "source-only", None).await?.1,
b"source read after recovery"
);
Ok(())
}
#[tokio::test] #[tokio::test]
#[ignore = "requires a pinned previous RustFS release binary"] #[ignore = "requires a pinned previous RustFS release binary"]
async fn direct_upgrade_from_rc2_preserves_object_contracts() -> TestResult { async fn direct_upgrade_from_rc2_preserves_object_contracts() -> TestResult {
@@ -555,653 +429,3 @@ async fn rolling_upgrade_from_rc2_preserves_mixed_version_contracts() -> TestRes
Ok(()) Ok(())
} }
/// Child-process environment shared by both bucket-configuration scenarios.
///
/// The replication target is an in-process fake bound to `127.0.0.1`, which
/// `set-remote-target` rejects as an SSRF risk without the loopback opt-in, and
/// the proxy bypass keeps a developer's `HTTP_PROXY` from intercepting the
/// server's outbound health check.
fn bucket_config_server_env() -> Vec<(&'static str, &'static str)> {
let mut env = vec![
(SSE_MASTER_KEY_ENV, SSE_MASTER_KEY),
("NO_PROXY", "127.0.0.1,localhost"),
("HTTP_PROXY", ""),
("HTTPS_PROXY", ""),
// Shorten the scanner cycle so the bucket's usage becomes authoritative
// in seconds; both `set-bucket-quota` and quota admission block on it.
("RUSTFS_SCANNER_CYCLE", "1"),
("RUSTFS_SCANNER_START_DELAY_SECS", "0"),
];
env.extend_from_slice(LOOPBACK_REPLICATION_TARGET_ENV);
env.extend(replication_fast_env());
env
}
/// Restart `env` in place on the same data directory using an explicit binary.
///
/// [`RustFSTestEnvironment::restart_server_preserving_data`] always relaunches
/// the workspace build, which is the upgrade direction only. The rollback
/// scenario needs the reverse: stop the current build and bring the pinned
/// previous release up on the metadata that build just wrote.
async fn restart_from_binary(env: &mut RustFSTestEnvironment, binary: &Path, server_env: &[(&str, &str)]) -> TestResult {
env.stop_server();
env.start_rustfs_server_from_binary(binary, vec![], server_env).await
}
async fn set_bucket_quota(env: &RustFSTestEnvironment, bucket: &str, quota_bytes: u64) -> TestResult {
let path = format!("/rustfs/admin/v3/quota/{bucket}");
let body = serde_json::json!({ "quota": quota_bytes, "quota_type": "HARD" }).to_string();
let deadline = Instant::now() + QUOTA_READINESS_TIMEOUT;
loop {
let (status, response) =
admin_request(&env.url, Method::PUT, &path, Some(body.clone()), &env.access_key, &env.secret_key).await?;
if status.is_success() {
return Ok(());
}
if status != StatusCode::SERVICE_UNAVAILABLE || Instant::now() >= deadline {
return Err(format!("setting the quota of {bucket} failed: {status} {response}").into());
}
sleep(Duration::from_millis(500)).await;
}
}
/// PUT into a quota-enabled bucket, riding out the post-start quota-admission
/// warm-up described on [`QUOTA_ADMISSION_WARMUP_TIMEOUT`].
///
/// Only `ServiceUnavailable` is retried: any other failure, and a warm-up that
/// never ends, is a genuine regression and surfaces as an error.
async fn put_object_through_quota_warmup(client: &Client, bucket: &str, key: &str, body: &'static [u8]) -> TestResult {
let deadline = Instant::now() + QUOTA_ADMISSION_WARMUP_TIMEOUT;
loop {
let result = client
.put_object()
.bucket(bucket)
.key(key)
.body(ByteStream::from_static(body))
.send()
.await;
let error = match result {
Ok(_) => return Ok(()),
Err(error) => error,
};
let retryable = error.as_service_error().and_then(ProvideErrorMetadata::code) == Some("ServiceUnavailable");
if !retryable || Instant::now() >= deadline {
return Err(format!("PUT {bucket}/{key} failed after the quota warm-up window: {error}").into());
}
sleep(Duration::from_millis(500)).await;
}
}
async fn get_bucket_quota(env: &RustFSTestEnvironment, bucket: &str) -> Result<Option<u64>, BoxError> {
let path = format!("/rustfs/admin/v3/quota/{bucket}");
let (status, response) = admin_request(&env.url, Method::GET, &path, None, &env.access_key, &env.secret_key).await?;
if status != StatusCode::OK {
return Err(format!("reading the quota of {bucket} failed: {status} {response}").into());
}
let quota: serde_json::Value = serde_json::from_str(&response)?;
Ok(quota.get("quota").and_then(serde_json::Value::as_u64))
}
/// `GET /rustfs/admin/v3/list-remote-targets?bucket=...`.
///
/// Returns an error for any non-200, because rustfs#7172 made this endpoint
/// fail closed on a `bucket-targets.json` blob the running build cannot parse.
/// An upgrade that misreads a blob written by the previous release therefore
/// shows up here as an error, and a silently dropped target shows up as an
/// empty list — the caller must distinguish the two.
async fn list_remote_targets(env: &RustFSTestEnvironment, bucket: &str) -> Result<Vec<serde_json::Value>, BoxError> {
let path = format!("/rustfs/admin/v3/list-remote-targets?bucket={}", urlencoding::encode(bucket));
let (status, response) = admin_request(&env.url, Method::GET, &path, None, &env.access_key, &env.secret_key).await?;
if status != StatusCode::OK {
return Err(format!("list-remote-targets for {bucket} failed: {status} {response}").into());
}
Ok(serde_json::from_str(&response)?)
}
/// Assert that `bucket` still carries exactly the replication target `arn`.
async fn assert_remote_target_preserved(env: &RustFSTestEnvironment, bucket: &str, arn: &str, context: &str) -> TestResult {
let targets = list_remote_targets(env, bucket).await?;
assert_eq!(
targets.len(),
1,
"{context}: list-remote-targets must still report the single configured target, got {targets:?}"
);
assert_eq!(
targets[0].get("arn").and_then(serde_json::Value::as_str),
Some(arn),
"{context}: the target ARN changed across the restart: {targets:?}"
);
Ok(())
}
/// Configure a replication target on `bucket` pointing at the in-process fake,
/// then attach an enabled replication rule for it. Returns the target ARN.
async fn configure_replication(
env: &RustFSTestEnvironment,
bucket: &str,
target: &FakeS3Target,
target_bucket: &str,
) -> Result<String, BoxError> {
let arn = set_replication_target_with_options(
env,
bucket,
ReplicationTargetOptions {
endpoint: &target.address(),
access_key: FAKE_ACCESS_KEY,
secret_key: FAKE_SECRET_KEY,
target_bucket,
secure: false,
skip_tls_verify: false,
ca_cert_pem: None,
},
)
.await?;
put_bucket_replication(env, bucket, &arn).await?;
Ok(arn)
}
async fn put_default_sse_s3_encryption(client: &Client, bucket: &str) -> TestResult {
let configuration = ServerSideEncryptionConfiguration::builder()
.rules(
ServerSideEncryptionRule::builder()
.apply_server_side_encryption_by_default(
ServerSideEncryptionByDefault::builder()
.sse_algorithm(ServerSideEncryption::Aes256)
.build()?,
)
.build(),
)
.build()?;
client
.put_bucket_encryption()
.bucket(bucket)
.server_side_encryption_configuration(configuration)
.send()
.await?;
Ok(())
}
async fn assert_default_sse_s3_encryption(client: &Client, bucket: &str, context: &str) -> TestResult {
let response = client.get_bucket_encryption().bucket(bucket).send().await?;
let rules = response
.server_side_encryption_configuration()
.ok_or("GetBucketEncryption omitted the configuration")?
.rules();
assert_eq!(rules.len(), 1, "{context}: expected exactly one encryption rule, got {rules:?}");
assert_eq!(
rules[0]
.apply_server_side_encryption_by_default()
.map(ServerSideEncryptionByDefault::sse_algorithm),
Some(&ServerSideEncryption::Aes256),
"{context}: the default encryption algorithm changed"
);
Ok(())
}
async fn put_bucket_tag(client: &Client, bucket: &str) -> TestResult {
let tagging = Tagging::builder()
.tag_set(Tag::builder().key(BUCKET_TAG_KEY).value(BUCKET_TAG_VALUE).build()?)
.build()?;
client.put_bucket_tagging().bucket(bucket).tagging(tagging).send().await?;
Ok(())
}
async fn assert_bucket_tag(client: &Client, bucket: &str, context: &str) -> TestResult {
let tags = client.get_bucket_tagging().bucket(bucket).send().await?;
let tag_set = tags.tag_set();
assert_eq!(tag_set.len(), 1, "{context}: expected exactly one bucket tag, got {tag_set:?}");
assert_eq!(tag_set[0].key(), BUCKET_TAG_KEY, "{context}: bucket tag key changed");
assert_eq!(tag_set[0].value(), BUCKET_TAG_VALUE, "{context}: bucket tag value changed");
Ok(())
}
async fn assert_versioning_enabled(client: &Client, bucket: &str, context: &str) -> TestResult {
let versioning = client.get_bucket_versioning().bucket(bucket).send().await?;
assert_eq!(
versioning.status(),
Some(&BucketVersioningStatus::Enabled),
"{context}: versioning is no longer Enabled on {bucket}"
);
Ok(())
}
fn bucket_policy_document(bucket: &str) -> serde_json::Value {
serde_json::json!({
"Version": "2012-10-17",
"Statement": [{
"Sid": "UpgradePublicRead",
"Effect": "Allow",
"Principal": { "AWS": ["*"] },
"Action": ["s3:GetObject"],
"Resource": [format!("arn:aws:s3:::{bucket}/public/*")]
}]
})
}
/// `GET .../on-demand-migration/{bucket}/status`.
///
/// The migration module defaults on from rustfs#7089, so a bucket that never
/// configured a source must still answer `configured: false` rather than
/// engaging the migration path.
async fn assert_migration_not_configured(env: &RustFSTestEnvironment, bucket: &str) -> TestResult {
let path = format!("/rustfs/admin/v3/on-demand-migration/{bucket}/status");
let (status, response) = admin_request(&env.url, Method::GET, &path, None, &env.access_key, &env.secret_key).await?;
assert_eq!(
status,
StatusCode::OK,
"the migration status endpoint must answer for an unconfigured bucket: {status} {response}"
);
let body: serde_json::Value = serde_json::from_str(&response)?;
assert_eq!(
body.get("configured"),
Some(&serde_json::Value::Bool(false)),
"a bucket upgraded from the previous release must not look migration-configured: {body}"
);
Ok(())
}
/// A GET for a key that was never written must be a plain `NoSuchKey`.
///
/// With the migration module on by default this is the cheap proof that an
/// unconfigured bucket never consults a source: any migration engagement would
/// surface as a different status or error code here.
async fn assert_missing_key_is_no_such_key(client: &Client, bucket: &str, key: &str) -> TestResult {
let error = client
.get_object()
.bucket(bucket)
.key(key)
.send()
.await
.expect_err("a key that was never written must not be readable");
assert_eq!(
error.raw_response().map(|response| response.status().as_u16()),
Some(404),
"a missing key must stay a 404 on a bucket with no migration configuration"
);
assert_eq!(
error.as_service_error().and_then(ProvideErrorMetadata::code),
Some("NoSuchKey"),
"a missing key must stay NoSuchKey on a bucket with no migration configuration"
);
Ok(())
}
/// Bucket configuration written by the pinned previous release must survive an
/// upgrade to the current build unchanged, and must keep working.
///
/// This pins the three on-disk surfaces the on-demand-migration series moved:
///
/// * `BucketMetadata` grew two msgpack keys (encoded map length 44 -> 46), so
/// every configuration read below decodes a 44-key blob on 46-key code.
/// * rustfs#7172 made an unreadable `bucket-targets.json` / encryption /
/// public-access-block / quota blob "present but unreadable" instead of
/// silently defaulting, and made `list-remote-targets` fail closed on it. A
/// replication target configured by the old release must therefore still be
/// *listed*, not dropped and not an error.
/// * rustfs#7183 made the object write path refuse a PUT when the bucket's
/// encryption configuration cannot be read, so a misparsed SSE config would
/// turn every PUT to that bucket into a 500.
///
/// Not covered on purpose: on-demand-migration configuration itself, which the
/// previous release has no public API for — the reverse direction is asserted
/// instead (an upgraded bucket reports `configured: false`).
#[tokio::test]
#[ignore = "requires a pinned previous RustFS release binary"]
async fn direct_upgrade_from_previous_release_preserves_bucket_configuration() -> TestResult {
init_logging();
let previous_binary = source_binary()?;
// In-process: the fake target outlives both server processes, so the
// replication target stays reachable across the upgrade.
let replication_target = FakeS3Target::start().await?;
replication_target.create_bucket(CONFIG_REPLICA_BUCKET);
let mut env = RustFSTestEnvironment::new().await?;
let server_env = bucket_config_server_env();
env.start_rustfs_server_from_binary(&previous_binary, vec![], &server_env)
.await?;
let old_client = env.create_s3_client();
env.create_test_bucket(CONFIG_PLAIN_BUCKET).await?;
env.create_test_bucket(CONFIG_ENCRYPTED_BUCKET).await?;
env.create_test_bucket(CONFIG_REPLICATED_BUCKET).await?;
old_client
.create_bucket()
.bucket(CONFIG_LOCKED_BUCKET)
.object_lock_enabled_for_bucket(true)
.send()
.await?;
// Plain bucket: policy, tags, lifecycle, quota.
let policy = bucket_policy_document(CONFIG_PLAIN_BUCKET);
old_client
.put_bucket_policy()
.bucket(CONFIG_PLAIN_BUCKET)
.policy(policy.to_string())
.send()
.await?;
put_bucket_tag(&old_client, CONFIG_PLAIN_BUCKET).await?;
old_client
.put_bucket_lifecycle_configuration()
.bucket(CONFIG_PLAIN_BUCKET)
.lifecycle_configuration(
BucketLifecycleConfiguration::builder()
.rules(
LifecycleRule::builder()
.id(LIFECYCLE_RULE_ID)
.status(ExpirationStatus::Enabled)
.filter(LifecycleRuleFilter::builder().prefix(LIFECYCLE_PREFIX).build())
.expiration(LifecycleExpiration::builder().days(LIFECYCLE_DAYS).build())
.build()?,
)
.build()?,
)
.send()
.await?;
set_bucket_quota(&env, CONFIG_PLAIN_BUCKET, BUCKET_QUOTA_BYTES).await?;
// Encrypted bucket: SSE-S3 default encryption plus a fully restrictive
// public access block, both of which rustfs#7172 now fails closed on.
put_default_sse_s3_encryption(&old_client, CONFIG_ENCRYPTED_BUCKET).await?;
old_client
.put_public_access_block()
.bucket(CONFIG_ENCRYPTED_BUCKET)
.public_access_block_configuration(
PublicAccessBlockConfiguration::builder()
.block_public_acls(true)
.ignore_public_acls(true)
.block_public_policy(true)
.restrict_public_buckets(true)
.build(),
)
.send()
.await?;
// Replicated bucket: versioning, a validated remote target, a rule.
enable_versioning(&old_client, CONFIG_REPLICATED_BUCKET).await?;
let target_arn = configure_replication(&env, CONFIG_REPLICATED_BUCKET, &replication_target, CONFIG_REPLICA_BUCKET).await?;
assert_remote_target_preserved(&env, CONFIG_REPLICATED_BUCKET, &target_arn, "before the upgrade").await?;
// Object-lock bucket: a default GOVERNANCE retention on a fresh bucket.
old_client
.put_object_lock_configuration()
.bucket(CONFIG_LOCKED_BUCKET)
.object_lock_configuration(
ObjectLockConfiguration::builder()
.object_lock_enabled(ObjectLockEnabled::Enabled)
.rule(
ObjectLockRule::builder()
.default_retention(
DefaultRetention::builder()
.mode(ObjectLockRetentionMode::Governance)
.days(OBJECT_LOCK_DAYS)
.build(),
)
.build(),
)
.build(),
)
.send()
.await?;
let plain_key = "plain/written-by-previous";
let plain_bytes = b"plain object written by the previous RustFS release";
put_object_through_quota_warmup(&old_client, CONFIG_PLAIN_BUCKET, plain_key, plain_bytes).await?;
let encrypted_key = "encrypted/written-by-previous";
let encrypted_bytes = b"default-encrypted object written by the previous RustFS release";
old_client
.put_object()
.bucket(CONFIG_ENCRYPTED_BUCKET)
.key(encrypted_key)
.body(ByteStream::from_static(encrypted_bytes))
.send()
.await?;
assert_eq!(
read_object(&old_client, CONFIG_ENCRYPTED_BUCKET, encrypted_key, None)
.await?
.0,
Some(ServerSideEncryption::Aes256),
"the previous release must apply the bucket default encryption it just accepted"
);
// The multipart object lives in the default-encrypted bucket so the
// upgraded build has to reassemble parts *and* re-derive the object key.
let multipart_key = "encrypted/multipart-written-by-previous";
let multipart_parts = vec![vec![b'm'; 5 * 1024 * 1024], b"final multipart bytes".to_vec()];
let multipart_bytes = multipart_parts.concat();
write_multipart(&old_client, CONFIG_ENCRYPTED_BUCKET, multipart_key, &multipart_parts).await?;
let versioned_key = "versioned/written-by-previous";
let versioned_bytes = b"versioned object written by the previous RustFS release";
let versioned_id = old_client
.put_object()
.bucket(CONFIG_REPLICATED_BUCKET)
.key(versioned_key)
.body(ByteStream::from_static(versioned_bytes))
.send()
.await?
.version_id()
.ok_or("versioned PUT omitted version ID")?
.to_string();
env.restart_server_preserving_data(vec![], &server_env).await?;
let new_client = env.create_s3_client();
// Every configuration must read back unchanged on the upgraded build.
let upgraded_policy = new_client.get_bucket_policy().bucket(CONFIG_PLAIN_BUCKET).send().await?;
let upgraded_policy: serde_json::Value =
serde_json::from_str(upgraded_policy.policy().ok_or("GetBucketPolicy omitted the document")?)?;
assert_eq!(upgraded_policy, policy, "the bucket policy changed across the upgrade");
assert_bucket_tag(&new_client, CONFIG_PLAIN_BUCKET, "after the upgrade").await?;
let lifecycle = new_client
.get_bucket_lifecycle_configuration()
.bucket(CONFIG_PLAIN_BUCKET)
.send()
.await?;
let rules = lifecycle.rules();
assert_eq!(rules.len(), 1, "the lifecycle rule count changed across the upgrade: {rules:?}");
assert_eq!(rules[0].id(), Some(LIFECYCLE_RULE_ID));
assert_eq!(rules[0].status(), &ExpirationStatus::Enabled);
assert_eq!(
rules[0].expiration().and_then(LifecycleExpiration::days),
Some(LIFECYCLE_DAYS),
"the lifecycle expiration changed across the upgrade"
);
assert_eq!(
get_bucket_quota(&env, CONFIG_PLAIN_BUCKET).await?,
Some(BUCKET_QUOTA_BYTES),
"the bucket quota changed across the upgrade"
);
assert_default_sse_s3_encryption(&new_client, CONFIG_ENCRYPTED_BUCKET, "after the upgrade").await?;
let public_access_block = new_client
.get_public_access_block()
.bucket(CONFIG_ENCRYPTED_BUCKET)
.send()
.await?;
let public_access_block = public_access_block
.public_access_block_configuration()
.ok_or("GetPublicAccessBlock omitted the configuration")?;
assert_eq!(public_access_block.block_public_acls(), Some(true));
assert_eq!(public_access_block.ignore_public_acls(), Some(true));
assert_eq!(public_access_block.block_public_policy(), Some(true));
assert_eq!(public_access_block.restrict_public_buckets(), Some(true));
assert_versioning_enabled(&new_client, CONFIG_REPLICATED_BUCKET, "after the upgrade").await?;
// rustfs#7172: neither an empty list nor an error is acceptable here.
assert_remote_target_preserved(&env, CONFIG_REPLICATED_BUCKET, &target_arn, "after the upgrade").await?;
let replication = new_client
.get_bucket_replication()
.bucket(CONFIG_REPLICATED_BUCKET)
.send()
.await?;
let replication_rules = replication
.replication_configuration()
.ok_or("GetBucketReplication omitted the configuration")?
.rules();
assert_eq!(
replication_rules.len(),
1,
"the replication rule count changed across the upgrade: {replication_rules:?}"
);
assert_eq!(
replication_rules[0].destination().map(|destination| destination.bucket()),
Some(target_arn.as_str()),
"the replication rule no longer points at the configured target"
);
let object_lock = new_client
.get_object_lock_configuration()
.bucket(CONFIG_LOCKED_BUCKET)
.send()
.await?;
let object_lock = object_lock
.object_lock_configuration()
.ok_or("GetObjectLockConfiguration omitted the configuration")?;
assert_eq!(object_lock.object_lock_enabled(), Some(&ObjectLockEnabled::Enabled));
let retention = object_lock
.rule()
.and_then(ObjectLockRule::default_retention)
.ok_or("the object lock configuration lost its default retention")?;
assert_eq!(retention.mode(), Some(&ObjectLockRetentionMode::Governance));
assert_eq!(retention.days(), Some(OBJECT_LOCK_DAYS));
// rustfs#7183: a PUT into the default-encrypted bucket must still succeed
// and still come back encrypted.
let post_upgrade_encrypted_key = "encrypted/written-after-upgrade";
let post_upgrade_encrypted_bytes = b"default-encrypted object written by the current RustFS build";
new_client
.put_object()
.bucket(CONFIG_ENCRYPTED_BUCKET)
.key(post_upgrade_encrypted_key)
.body(ByteStream::from_static(post_upgrade_encrypted_bytes))
.send()
.await?;
let (encryption, body) = read_object(&new_client, CONFIG_ENCRYPTED_BUCKET, post_upgrade_encrypted_key, None).await?;
assert_eq!(
encryption,
Some(ServerSideEncryption::Aes256),
"a PUT after the upgrade lost the bucket default encryption"
);
assert_eq!(body, post_upgrade_encrypted_bytes);
let post_upgrade_plain_key = "plain/written-after-upgrade";
let post_upgrade_plain_bytes = b"plain object written by the current RustFS build";
put_object_through_quota_warmup(&new_client, CONFIG_PLAIN_BUCKET, post_upgrade_plain_key, post_upgrade_plain_bytes).await?;
let (encryption, body) = read_object(&new_client, CONFIG_PLAIN_BUCKET, post_upgrade_plain_key, None).await?;
assert_eq!(encryption, None, "a bucket without default encryption must not encrypt a PUT");
assert_eq!(body, post_upgrade_plain_bytes);
// Every object written by the previous release reads back byte-identical.
assert_eq!(read_object(&new_client, CONFIG_PLAIN_BUCKET, plain_key, None).await?.1, plain_bytes);
let (encryption, body) = read_object(&new_client, CONFIG_ENCRYPTED_BUCKET, encrypted_key, None).await?;
assert_eq!(encryption, Some(ServerSideEncryption::Aes256));
assert_eq!(body, encrypted_bytes);
let (encryption, body) = read_object(&new_client, CONFIG_ENCRYPTED_BUCKET, multipart_key, None).await?;
assert_eq!(encryption, Some(ServerSideEncryption::Aes256));
assert_eq!(body, multipart_bytes, "the multipart object did not survive the upgrade");
assert_eq!(
read_object(&new_client, CONFIG_REPLICATED_BUCKET, versioned_key, Some(&versioned_id))
.await?
.1,
versioned_bytes
);
// rustfs#7089: the migration module is on by default, but a bucket that
// never configured a source behaves exactly as before.
assert_migration_not_configured(&env, CONFIG_PLAIN_BUCKET).await?;
assert_missing_key_is_no_such_key(&new_client, CONFIG_PLAIN_BUCKET, "plain/never-written").await?;
replication_target.shutdown().await;
Ok(())
}
/// Rolling back to the pinned previous release must still read the bucket
/// metadata the current build wrote.
///
/// This is the other half of the `BucketMetadata` 44 -> 46 key change: the
/// current build writes a 46-key msgpack map with `OnDemandMigrationConfigJSON`
/// and `OnDemandMigrationConfigUpdatedAt`, and the previous release's decoder
/// has to skip those two unknown keys instead of failing the whole blob. If it
/// did not, every configuration read below would come back empty or error and
/// the rollback would silently discard the bucket's configuration.
#[tokio::test]
#[ignore = "requires a pinned previous RustFS release binary"]
async fn rollback_to_previous_release_reads_current_bucket_metadata() -> TestResult {
init_logging();
let previous_binary = source_binary()?;
let replication_target = FakeS3Target::start().await?;
replication_target.create_bucket(ROLLBACK_REPLICA_BUCKET);
let mut env = RustFSTestEnvironment::new().await?;
let server_env = bucket_config_server_env();
env.start_rustfs_server_with_env(vec![], &server_env).await?;
let new_client = env.create_s3_client();
env.create_test_bucket(ROLLBACK_BUCKET).await?;
enable_versioning(&new_client, ROLLBACK_BUCKET).await?;
put_default_sse_s3_encryption(&new_client, ROLLBACK_BUCKET).await?;
put_bucket_tag(&new_client, ROLLBACK_BUCKET).await?;
let target_arn = configure_replication(&env, ROLLBACK_BUCKET, &replication_target, ROLLBACK_REPLICA_BUCKET).await?;
assert_remote_target_preserved(&env, ROLLBACK_BUCKET, &target_arn, "before the rollback").await?;
let single_key = "rollback/single";
let single_bytes = b"single-part object written by the current RustFS build";
let single_version = new_client
.put_object()
.bucket(ROLLBACK_BUCKET)
.key(single_key)
.body(ByteStream::from_static(single_bytes))
.send()
.await?
.version_id()
.ok_or("versioned PUT omitted version ID")?
.to_string();
let multipart_key = "rollback/multipart";
let multipart_parts = vec![vec![b'r'; 5 * 1024 * 1024], b"final rollback bytes".to_vec()];
let multipart_bytes = multipart_parts.concat();
write_multipart(&new_client, ROLLBACK_BUCKET, multipart_key, &multipart_parts).await?;
restart_from_binary(&mut env, &previous_binary, &server_env).await?;
let old_client = env.create_s3_client();
assert_versioning_enabled(&old_client, ROLLBACK_BUCKET, "after the rollback").await?;
assert_default_sse_s3_encryption(&old_client, ROLLBACK_BUCKET, "after the rollback").await?;
assert_bucket_tag(&old_client, ROLLBACK_BUCKET, "after the rollback").await?;
assert_remote_target_preserved(&env, ROLLBACK_BUCKET, &target_arn, "after the rollback").await?;
let (encryption, body) = read_object(&old_client, ROLLBACK_BUCKET, single_key, Some(&single_version)).await?;
assert_eq!(encryption, Some(ServerSideEncryption::Aes256));
assert_eq!(body, single_bytes);
let (encryption, body) = read_object(&old_client, ROLLBACK_BUCKET, multipart_key, None).await?;
assert_eq!(encryption, Some(ServerSideEncryption::Aes256));
assert_eq!(body, multipart_bytes, "the multipart object did not survive the rollback");
// A PUT on the rolled-back release must still honour the encryption
// configuration it decoded out of the current build's metadata blob.
let post_rollback_key = "rollback/written-after-rollback";
let post_rollback_bytes = b"object written by the previous RustFS release after the rollback";
old_client
.put_object()
.bucket(ROLLBACK_BUCKET)
.key(post_rollback_key)
.body(ByteStream::from_static(post_rollback_bytes))
.send()
.await?;
let (encryption, body) = read_object(&old_client, ROLLBACK_BUCKET, post_rollback_key, None).await?;
assert_eq!(
encryption,
Some(ServerSideEncryption::Aes256),
"the rolled-back release lost the bucket default encryption"
);
assert_eq!(body, post_rollback_bytes);
replication_target.shutdown().await;
Ok(())
}
+2 -4
View File
@@ -31,7 +31,6 @@ workspace = true
[features] [features]
default = [] default = []
gcs = ["dep:google-cloud-storage", "dep:google-cloud-auth"]
# Compiles the controlled list-objects namespace-journal chaos injector into a # Compiles the controlled list-objects namespace-journal chaos injector into a
# production binary (it is always available to tests). Off by default so the # production binary (it is always available to tests). Off by default so the
# RUSTFS_LIST_OBJECTS_NAMESPACE_JOURNAL_CHAOS_* env vars cannot rewrite journal # RUSTFS_LIST_OBJECTS_NAMESPACE_JOURNAL_CHAOS_* env vars cannot rewrite journal
@@ -213,8 +212,8 @@ aws-smithy-runtime-api = { workspace = true, features = ["http-1x"] }
parking_lot = { workspace = true } parking_lot = { workspace = true }
base64-simd.workspace = true base64-simd.workspace = true
serde_urlencoded.workspace = true serde_urlencoded.workspace = true
google-cloud-storage = { workspace = true, optional = true } google-cloud-storage = { workspace = true }
google-cloud-auth = { workspace = true, optional = true } google-cloud-auth = { workspace = true }
faster-hex = { workspace = true } faster-hex = { workspace = true }
ratelimit = { workspace = true } ratelimit = { workspace = true }
aws-smithy-http-client = { workspace = true, default-features = false, features = ["rustls-aws-lc"] } aws-smithy-http-client = { workspace = true, default-features = false, features = ["rustls-aws-lc"] }
@@ -245,7 +244,6 @@ windows-sys = { workspace = true, features = [
windows-sys = { workspace = true, features = ["Win32_System_Ioctl"] } windows-sys = { workspace = true, features = ["Win32_System_Ioctl"] }
[dev-dependencies] [dev-dependencies]
aws-smithy-async.workspace = true
tokio = { workspace = true, features = ["rt-multi-thread", "macros", "test-util", "fs"] } tokio = { workspace = true, features = ["rt-multi-thread", "macros", "test-util", "fs"] }
criterion = { workspace = true, features = ["html_reports"] } criterion = { workspace = true, features = ["html_reports"] }
temp-env = { workspace = true, features = ["async_closure"] } temp-env = { workspace = true, features = ["async_closure"] }
+65 -30
View File
@@ -146,23 +146,66 @@ pub mod bucket {
}; };
} }
pub mod metadata_sys { pub mod on_demand_migration {
pub use crate::bucket::metadata_sys::{ pub use crate::bucket::on_demand_migration::{
BUCKET_CONFIG_PUBLISH_HOOK, BucketConfigPublishHook, BucketMetadataMutationGuard, BucketMetadataSys, ApplyOutcome, BREAKER_FAILURE_THRESHOLD, BREAKER_FAILURE_WINDOW, BREAKER_HALF_OPEN_MAX_PROBES, BREAKER_OPEN_DURATION,
ObjectLockConfigState, acquire_bucket_metadata_transaction_lock, Breaker, BreakerState, BreakerTransition, BreakerVerdict, BucketOdmState, GLOBAL_ON_DEMAND_MIGRATION_SYS, GaugeGuard,
acquire_bucket_metadata_transaction_lock_for_incarnation, acquire_scanner_bucket_incarnation_fence, LastSourceError, LatencyBucketSnapshot, NEGATIVE_CACHE_MAX_ENTRIES, NegativeCache, OdmBucketSnapshot, OdmLookup,
capture_bucket_metadata_incarnation, delete, delete_if_incarnation, delete_under_transaction_lock, get, OdmOp, OdmOutcome, OdmStateError, OdmStats, OdmStatsSnapshot, OnDemandMigrationSys, PullError, PullFailureReason,
get_accelerate_config, get_bucket_policy, get_bucket_policy_raw, get_bucket_targets_config, get_config_from_disk, PullFollower, PullLeader, PullOutcome, PullPath, PullResult, PullSlot, SOURCE_LATENCY_BUCKET_BOUNDS_MS,
get_cors_config, get_durability_config, get_global_bucket_metadata_sys, get_lifecycle_config, get_logging_config, SourceLatencySnapshot, source_client_spec,
get_notification_config, get_object_lock_config, get_object_lock_config_state, get_on_demand_migration_config,
get_on_demand_migration_config_in, get_public_access_block_config, get_quota_config, get_replication_config,
get_request_payment_config, get_sse_config, get_tagging_config, get_versioning_config, get_website_config,
init_bucket_metadata_sys, list_bucket_targets, reload_bucket_metadata, remove_bucket_metadata, set_bucket_metadata,
update, update_bucket_targets_under_transaction_lock, update_config_with, update_if_incarnation,
update_quota_if_incarnation, update_under_transaction_lock,
}; };
pub use crate::bucket::on_demand_migration::{
ConfigPublishHook, FilterConfig, HeadPolicy, ON_DEMAND_MIGRATION_CONFIG_HOOK, ON_DEMAND_MIGRATION_CONFIG_VERSION,
OnDemandMigrationConfig, OnDemandMigrationConfigError, PathStyle, PolicyConfig, Provider, RangeGetPolicy,
SourceConfig, SourceCredentials, SourceErrorPolicy, SourceTimeout, TlsConfig, ValidationContext,
};
pub use crate::bucket::on_demand_migration::{
EnqueueOutcome, LocalObject, MAX_MULTIPART_PARTS, OdmWriteBack, PULL_MAX_RETRIES, PULL_RETRY_BASE_DELAYS,
PullCompletion, PullQueue, PullReason, PullSource, QueuedPullOutcome, SourceBody, SourceIdleGuard, WriteBackBody,
WriteBackError, WriteBackOutcome, WriteBackPart, WriteBackRequest, commit_inline, commit_inline_with,
idle_guarded_body,
};
pub use crate::bucket::on_demand_migration::{
FetchRequest, LIST_THROUGH_TOKEN_VERSION, ListEntryKey, ListThroughCursor, ListThroughMerger, ListThroughToken,
ListThroughTokenError, MAX_LIST_FETCHES_PER_SIDE, MergeOutcome, MergePick, MergeSide, SOURCE_LIST_MAX_RATE_WAIT,
SOURCE_LIST_RATE_PER_SEC, SourceListPlan, SourceListRateLimiter, decode_continuation_token, source_list_plan,
};
pub mod backfill {
pub use crate::bucket::on_demand_migration::backfill::{
BACKFILL_CHECKPOINT_FILE, BACKFILL_CHECKPOINT_FORMAT_VERSION, BACKFILL_FAILED_KEYS_CAPACITY, BACKFILL_LEASE,
BACKFILL_LEASE_LOCK_PREFIX, BACKFILL_LIST_PAGE_SIZE, BACKFILL_RECOVERY_INTERVAL, BACKFILL_SAVE_EVERY_KEYS,
BACKFILL_SAVE_INTERVAL, BackfillCheckpoint, BackfillContext, BackfillContextFactory, BackfillError,
BackfillLastError, BackfillOwner, BackfillRecoveryStats, BackfillRequest, BackfillRunner, BackfillState,
BucketBackfillContext, LocalBackfillObject, PriorityPullPermits, PullPermit, PullPriority, SkipExisting,
StoredCheckpoint, SysBackfillContexts, global_backfill_runner, install_global_backfill_runner, key_hash,
read_checkpoint, run_backfill_recovery_loop, spawn_backfill_recovery_loop,
};
}
pub mod source_client {
pub use crate::bucket::on_demand_migration::source_client::{
SourceClient, SourceClientSpec, SourceError, SourceGet, SourceHead, SourceListRequest, SourceObject, SourcePage,
SourceProbe, SourceProvider, SourceSse, SourceTimeouts, USER_AGENT_SUFFIX, is_multipart_etag, range_header_value,
resolve_path_style,
};
}
}
pub mod metadata_sys {
#[cfg(feature = "test-util")] #[cfg(feature = "test-util")]
pub use crate::bucket::metadata_sys::{ConfigWriteLockProbe, test_support}; pub use crate::bucket::metadata_sys::ConfigWriteLockProbe;
pub use crate::bucket::metadata_sys::{
BucketMetadataMutationGuard, BucketMetadataSys, ObjectLockConfigState, acquire_bucket_metadata_transaction_lock,
acquire_bucket_metadata_transaction_lock_for_incarnation, capture_bucket_metadata_incarnation, delete,
delete_if_incarnation, delete_under_transaction_lock, get, get_accelerate_config, get_bucket_policy,
get_bucket_policy_raw, get_bucket_targets_config, get_config_from_disk, get_cors_config, get_durability_config,
get_global_bucket_metadata_sys, get_lifecycle_config, get_logging_config, get_notification_config,
get_object_lock_config, get_object_lock_config_state, get_on_demand_migration_config, get_public_access_block_config,
get_quota_config, get_replication_config, get_request_payment_config, get_sse_config, get_tagging_config,
get_versioning_config, get_website_config, init_bucket_metadata_sys, list_bucket_targets, reload_bucket_metadata,
remove_bucket_metadata, set_bucket_metadata, update, update_bucket_targets_under_transaction_lock,
update_config_with, update_if_incarnation, update_quota_if_incarnation, update_under_transaction_lock,
};
} }
pub mod migration { pub mod migration {
@@ -205,7 +248,7 @@ pub mod bucket {
pub mod remote_s3_client { pub mod remote_s3_client {
pub use crate::bucket::remote_s3_client::{ pub use crate::bucket::remote_s3_client::{
PathStyle, RemoteCredentials, RemoteS3ClientError, RemoteS3EndpointSpec, RemoteS3RetryPolicy, build_remote_s3_client, PathStyle, RemoteCredentials, RemoteS3ClientError, RemoteS3EndpointSpec, RemoteS3RetryPolicy, build_remote_s3_client,
build_remote_s3_config, validate_remote_endpoint, validate_target_ca_pem, validate_remote_endpoint,
}; };
} }
@@ -436,11 +479,9 @@ pub mod notification {
#[cfg(any(test, feature = "test-util"))] #[cfg(any(test, feature = "test-util"))]
pub use crate::services::notification_sys::rotate_cross_pool_fence_fleet_proof_for_test; pub use crate::services::notification_sys::rotate_cross_pool_fence_fleet_proof_for_test;
pub use crate::services::notification_sys::{ pub use crate::services::notification_sys::{
ClusterTierDailyStats, CrossPoolFenceFleetProofToken, LegacyTransitionStateReconcileFleetProofToken, NotificationPeerErr, ClusterTierDailyStats, CrossPoolFenceFleetProofToken, NotificationPeerErr, NotificationSys, ScannerPublicationLeaseGrant,
NotificationSys, ScannerPublicationLeaseGrant, acquire_cross_pool_fence_fleet_proof, acquire_cross_pool_fence_fleet_proof, cross_pool_fence_fleet_proof_matches, get_global_notification_sys,
acquire_legacy_transition_state_reconcile_fleet_proof, cross_pool_fence_fleet_proof_matches, get_global_notification_sys, new_global_notification_sys, scanner_peer_transport_error_message_is_retryable, start_remote_version_state_fleet_probe,
legacy_transition_state_reconcile_fleet_proof_matches, new_global_notification_sys,
scanner_peer_transport_error_message_is_retryable, start_remote_version_state_fleet_probe,
}; };
} }
@@ -451,9 +492,9 @@ pub mod object {
ObjectInfo, ObjectLockConfigSnapshot, ObjectMutationHook, ObjectOptions, PutObjReader, QuotaAdmission, ObjectInfo, ObjectLockConfigSnapshot, ObjectMutationHook, ObjectOptions, PutObjReader, QuotaAdmission,
RangedDecompressReader, ReadEncryptionMaterial, ReadEncryptionMode, ReadEncryptionRequest, RangedDecompressReader, ReadEncryptionMaterial, ReadEncryptionMode, ReadEncryptionRequest,
SCANNER_PUBLICATION_LEASE_FENCE_METADATA_KEY, ScannerPublicationCommitScope, ScannerPublicationCommitStartError, SCANNER_PUBLICATION_LEASE_FENCE_METADATA_KEY, ScannerPublicationCommitScope, ScannerPublicationCommitStartError,
ScannerPublicationCommitState, StreamConsumer, WriteCompletion, get_object_body_cache_plaintext_len, ScannerPublicationCommitState, StreamConsumer, get_object_body_cache_plaintext_len, lookup_get_object_body_cache_hook,
lookup_get_object_body_cache_hook, register_get_object_body_cache_hook, register_object_mutation_hook, register_get_object_body_cache_hook, register_object_mutation_hook, unregister_get_object_body_cache_hook,
unregister_get_object_body_cache_hook, unregister_object_mutation_hook, unregister_object_mutation_hook,
}; };
pub use crate::store::{ pub use crate::store::{
PrepareSelectObjectSnapshotError, PreparedGetObjectReader, SelectObjectSnapshot, SelectObjectSnapshotReadError, PrepareSelectObjectSnapshotError, PreparedGetObjectReader, SelectObjectSnapshot, SelectObjectSnapshotReadError,
@@ -517,12 +558,6 @@ pub mod set_disk {
pub mod test_util { pub mod test_util {
pub use crate::bucket::quota::reservation::fail_next_quota_ledger_save_for_test; pub use crate::bucket::quota::reservation::fail_next_quota_ledger_save_for_test;
pub use crate::set_disk::{MultipartCommitBarrier, MultipartCommitPause, PutObjectCommitBarrier, PutObjectCommitPause}; pub use crate::set_disk::{MultipartCommitBarrier, MultipartCommitPause, PutObjectCommitBarrier, PutObjectCommitPause};
/// Keep a namespace commit pending until the returned owner is dropped.
#[must_use]
pub fn hold_namespace_commit(store: &crate::store::ECStore) -> impl Send + Sync {
store.ctx.begin_namespace_commit()
}
} }
} }
+25 -84
View File
@@ -59,7 +59,7 @@ use rustfs_utils::http::{
insert_header, insert_header,
}; };
use serde::{Deserialize, Serialize}; use serde::{Deserialize, Serialize};
use std::collections::{HashMap, HashSet}; use std::collections::HashMap;
use std::error::Error; use std::error::Error;
use std::fmt; use std::fmt;
use std::str::FromStr as _; use std::str::FromStr as _;
@@ -376,11 +376,6 @@ pub struct BucketTargetSys {
/// [`SsecPassthroughCapability`]; reset alongside `arn_remotes_map`. /// [`SsecPassthroughCapability`]; reset alongside `arn_remotes_map`.
ssec_passthrough_map: Arc<RwLock<HashMap<String, SsecPassthroughRecord>>>, ssec_passthrough_map: Arc<RwLock<HashMap<String, SsecPassthroughRecord>>>,
pub targets_map: Arc<RwLock<HashMap<String, Vec<BucketTarget>>>>, pub targets_map: Arc<RwLock<HashMap<String, Vec<BucketTarget>>>>,
/// Buckets whose persisted `bucket-targets.json` exists but cannot be
/// decoded (rustfs/backlog#2282). Written under the bucket's update mutex
/// alongside `targets_map`, and read before it so an unreadable
/// configuration surfaces as a typed error instead of an empty target set.
unreadable_targets: Arc<RwLock<HashSet<String>>>,
pub h_mutex: Arc<RwLock<HashMap<String, EpHealth>>>, pub h_mutex: Arc<RwLock<HashMap<String, EpHealth>>>,
target_h_mutex: Arc<RwLock<HashMap<String, EpHealth>>>, target_h_mutex: Arc<RwLock<HashMap<String, EpHealth>>>,
pub hc_client: Arc<HttpClient>, pub hc_client: Arc<HttpClient>,
@@ -424,7 +419,6 @@ impl BucketTargetSys {
arn_remotes_map: Arc::new(RwLock::new(HashMap::new())), arn_remotes_map: Arc::new(RwLock::new(HashMap::new())),
ssec_passthrough_map: Arc::new(RwLock::new(HashMap::new())), ssec_passthrough_map: Arc::new(RwLock::new(HashMap::new())),
targets_map: Arc::new(RwLock::new(HashMap::new())), targets_map: Arc::new(RwLock::new(HashMap::new())),
unreadable_targets: Arc::new(RwLock::new(HashSet::new())),
h_mutex: Arc::new(RwLock::new(HashMap::new())), h_mutex: Arc::new(RwLock::new(HashMap::new())),
target_h_mutex: Arc::new(RwLock::new(HashMap::new())), target_h_mutex: Arc::new(RwLock::new(HashMap::new())),
hc_client: Arc::new(build_health_check_client()), hc_client: Arc::new(build_health_check_client()),
@@ -634,40 +628,30 @@ impl BucketTargetSys {
health_map.clone() health_map.clone()
} }
/// Targets of one bucket, or of every bucket when `bucket` is empty. pub async fn list_targets(&self, bucket: &str, arn_type: &str) -> Vec<BucketTarget> {
///
/// A bucket that simply has no targets yields an empty list; a bucket
/// whose persisted configuration cannot be decoded is an error, so an
/// admin listing reports the fault instead of an empty list that reads as
/// "replication is not configured" (rustfs/backlog#2282).
pub async fn list_targets(&self, bucket: &str, arn_type: &str) -> Result<Vec<BucketTarget>, BucketTargetError> {
let health_stats = self.target_health_stats().await; let health_stats = self.target_health_stats().await;
let mut targets = Vec::new(); let mut targets = Vec::new();
if !bucket.is_empty() { if !bucket.is_empty() {
match self.list_bucket_targets(bucket).await { if let Ok(bucket_targets) = self.list_bucket_targets(bucket).await {
Ok(bucket_targets) => { for mut target in bucket_targets.targets {
for mut target in bucket_targets.targets { if arn_type.is_empty() || target.target_type.to_string() == arn_type {
if arn_type.is_empty() || target.target_type.to_string() == arn_type { if let Some(health) = health_stats.get(&target.arn) {
if let Some(health) = health_stats.get(&target.arn) { target.total_downtime = health.offline_duration;
target.total_downtime = health.offline_duration; target.online = health.online;
target.online = health.online; target.last_online = health.last_online;
target.last_online = health.last_online; target.latency = target::LatencyStat {
target.latency = target::LatencyStat { curr: health.latency.curr,
curr: health.latency.curr, avg: health.latency.avg,
avg: health.latency.avg, max: health.latency.peak,
max: health.latency.peak, };
}; target.offline_count = health.offline_count;
target.offline_count = health.offline_count;
}
targets.push(target);
} }
targets.push(target);
} }
} }
Err(BucketTargetError::BucketRemoteTargetNotFound { .. }) => {}
Err(err) => return Err(err),
} }
return Ok(targets); return targets;
} }
let targets_map = self.targets_map.read().await; let targets_map = self.targets_map.read().await;
@@ -690,16 +674,10 @@ impl BucketTargetSys {
} }
} }
Ok(targets) targets
} }
pub async fn list_bucket_targets(&self, bucket: &str) -> Result<BucketTargets, BucketTargetError> { pub async fn list_bucket_targets(&self, bucket: &str) -> Result<BucketTargets, BucketTargetError> {
if self.unreadable_targets.read().await.contains(bucket) {
return Err(BucketTargetError::BucketRemoteTargetsUnreadable {
bucket: bucket.to_string(),
});
}
let targets_map = self.targets_map.read().await; let targets_map = self.targets_map.read().await;
if let Some(targets) = targets_map.get(bucket) { if let Some(targets) = targets_map.get(bucket) {
Ok(BucketTargets { Ok(BucketTargets {
@@ -712,30 +690,13 @@ impl BucketTargetSys {
} }
} }
/// Record that this bucket's persisted targets configuration exists but
/// cannot be decoded (rustfs/backlog#2282).
///
/// Any snapshot published from an earlier readable load is deliberately
/// left in place: withdrawing it would produce exactly the silent "no
/// targets configured" state this marker exists to prevent. The marker is
/// cleared by the next successful publish, which is what makes a repaired
/// configuration take effect without a restart.
pub async fn mark_targets_unreadable(&self, bucket: &str) {
let update_mutex = self.target_update_mutex(bucket).await;
let _update_guard = update_mutex.lock().await;
self.unreadable_targets.write().await.insert(bucket.to_string());
}
pub async fn delete(&self, bucket: &str) { pub async fn delete(&self, bucket: &str) {
let update_mutex = self.target_update_mutex(bucket).await; let update_mutex = self.target_update_mutex(bucket).await;
let _update_guard = update_mutex.lock().await; let _update_guard = update_mutex.lock().await;
// Lock order: unreadable_targets, then targets_map, then // Lock order: targets_map, then arn_remotes_map, then target_h_mutex,
// arn_remotes_map, then target_h_mutex, then ssec_passthrough_map // then ssec_passthrough_map (always last; also taken standalone by the
// (always last; also taken standalone by the capability accessors). // capability accessors).
self.unreadable_targets.write().await.remove(bucket);
let mut targets_map = self.targets_map.write().await; let mut targets_map = self.targets_map.write().await;
let mut arn_remotes_map = self.arn_remotes_map.write().await; let mut arn_remotes_map = self.arn_remotes_map.write().await;
let mut health_map = self.target_h_mutex.write().await; let mut health_map = self.target_h_mutex.write().await;
@@ -1132,11 +1093,6 @@ impl BucketTargetSys {
/// Keeping persisted-config reads under the same mutex prevents a stale /// Keeping persisted-config reads under the same mutex prevents a stale
/// reload from overwriting a concurrent credential rotation. /// reload from overwriting a concurrent credential rotation.
async fn update_all_targets_locked(&self, bucket: &str, targets: Option<&BucketTargets>) { async fn update_all_targets_locked(&self, bucket: &str, targets: Option<&BucketTargets>) {
// Reaching here means the persisted configuration decoded, so the
// unreadable marker (if any) is stale. Cleared before the maps below
// so `unreadable_targets` stays the outermost of this module's locks.
self.unreadable_targets.write().await.remove(bucket);
let mut clients = Vec::new(); let mut clients = Vec::new();
if let Some(new_targets) = targets { if let Some(new_targets) = targets {
for target in &new_targets.targets { for target in &new_targets.targets {
@@ -1144,9 +1100,9 @@ impl BucketTargetSys {
} }
} }
// Lock order: unreadable_targets (above), then targets_map, then // Lock order: targets_map, then arn_remotes_map, then target_h_mutex,
// arn_remotes_map, then target_h_mutex, then ssec_passthrough_map // then ssec_passthrough_map (always last; also taken standalone by the
// (always last; also taken standalone by the capability accessors). // capability accessors).
let mut targets_map = self.targets_map.write().await; let mut targets_map = self.targets_map.write().await;
let mut arn_remotes_map = self.arn_remotes_map.write().await; let mut arn_remotes_map = self.arn_remotes_map.write().await;
let mut health_map = self.target_h_mutex.write().await; let mut health_map = self.target_h_mutex.write().await;
@@ -1205,11 +1161,6 @@ impl BucketTargetSys {
} }
pub async fn set(&self, bucket: &str, meta: &BucketMetadata) { pub async fn set(&self, bucket: &str, meta: &BucketMetadata) {
if meta.bucket_targets_unreadable() {
self.mark_targets_unreadable(bucket).await;
return;
}
let Some(config) = &meta.bucket_target_config else { let Some(config) = &meta.bucket_target_config else {
return; return;
}; };
@@ -2325,13 +2276,6 @@ pub enum BucketTargetError {
BucketRemoteTargetNotFound { BucketRemoteTargetNotFound {
bucket: String, bucket: String,
}, },
/// The bucket's persisted targets configuration exists but cannot be
/// decoded. Distinct from `BucketRemoteTargetNotFound`, which means the
/// bucket genuinely has no targets: callers must not degrade this one to
/// an empty target set (rustfs/backlog#2282).
BucketRemoteTargetsUnreadable {
bucket: String,
},
BucketRemoteArnTypeInvalid { BucketRemoteArnTypeInvalid {
bucket: String, bucket: String,
}, },
@@ -2365,9 +2309,6 @@ impl fmt::Display for BucketTargetError {
BucketTargetError::BucketRemoteTargetNotFound { bucket } => { BucketTargetError::BucketRemoteTargetNotFound { bucket } => {
write!(f, "Remote target not found for bucket: {bucket}") write!(f, "Remote target not found for bucket: {bucket}")
} }
BucketTargetError::BucketRemoteTargetsUnreadable { bucket } => {
write!(f, "Persisted replication target configuration is unreadable for bucket: {bucket}")
}
BucketTargetError::BucketRemoteArnTypeInvalid { bucket } => { BucketTargetError::BucketRemoteArnTypeInvalid { bucket } => {
write!(f, "Invalid ARN type for bucket: {bucket}") write!(f, "Invalid ARN type for bucket: {bucket}")
} }
@@ -3315,7 +3256,7 @@ mod tests {
}], }],
); );
let targets = sys.list_targets("", "").await.expect("listing every bucket's targets"); let targets = sys.list_targets("", "").await;
assert_eq!(targets.len(), 1); assert_eq!(targets.len(), 1);
assert!(!targets[0].online); assert!(!targets[0].online);
@@ -584,173 +584,33 @@ impl ExpiryOp for FreeVersionTask {
} }
} }
#[derive(Clone, Copy, Debug, Eq, PartialEq)]
enum TransitionDeleteVersionPlan {
Direct { version_id_exact: bool },
ProbeLegacyUnknown,
}
fn legacy_transition_version_state_missing(oi: &ObjectInfo) -> Result<bool, std::io::Error> {
use rustfs_utils::http::metadata_compat::{
SUFFIX_TRANSITIONED_VERSION_ID, SUFFIX_TRANSITIONED_VERSION_STATE, contains_key_str, get_consistent_str,
};
if !contains_key_str(&oi.user_defined, SUFFIX_TRANSITIONED_VERSION_STATE) {
let version_key_present = contains_key_str(&oi.user_defined, SUFFIX_TRANSITIONED_VERSION_ID);
if version_key_present {
if oi.transitioned_object.version_id.is_empty() {
let has_non_empty_version = oi.user_defined.iter().any(|(key, value)| {
rustfs_utils::http::metadata_compat::strip_internal_prefix_preserving_case(key)
.is_some_and(|suffix| suffix.eq_ignore_ascii_case(SUFFIX_TRANSITIONED_VERSION_ID))
&& !value.is_empty()
});
if !has_non_empty_version {
// MinIO writes the transitioned-versionID key with an empty value
// for unversioned tier objects. The backend probe remains the proof.
return Ok(true);
}
} else if get_consistent_str(&oi.user_defined, SUFFIX_TRANSITIONED_VERSION_ID)
== Some(oi.transitioned_object.version_id.as_str())
{
return Ok(true);
}
return Err(std::io::Error::new(
std::io::ErrorKind::InvalidData,
"legacy remote tier version metadata is conflicting or malformed",
));
}
if !oi.transitioned_object.version_id.is_empty() {
return Err(std::io::Error::new(
std::io::ErrorKind::InvalidData,
"legacy remote tier version metadata is missing or inconsistent",
));
}
return Ok(true);
}
let persisted = get_consistent_str(&oi.user_defined, SUFFIX_TRANSITIONED_VERSION_STATE).ok_or_else(|| {
std::io::Error::new(
std::io::ErrorKind::InvalidData,
"remote tier object has conflicting transition version state metadata",
)
})?;
if persisted != oi.transition_version_state.as_str() {
return Err(std::io::Error::new(
std::io::ErrorKind::InvalidData,
"remote tier object transition version state metadata changed during decoding",
));
}
Ok(false)
}
fn transition_remote_version_delete_plan(oi: &ObjectInfo) -> Result<TransitionDeleteVersionPlan, std::io::Error> {
match oi.transition_version_state {
rustfs_filemeta::TransitionVersionState::Unknown => {
if legacy_transition_version_state_missing(oi)? {
Ok(TransitionDeleteVersionPlan::ProbeLegacyUnknown)
} else {
validate_transition_remote_version(oi)
.map(|version_id_exact| TransitionDeleteVersionPlan::Direct { version_id_exact })
}
}
_ => validate_transition_remote_version(oi)
.map(|version_id_exact| TransitionDeleteVersionPlan::Direct { version_id_exact }),
}
}
#[derive(Clone, Copy, Debug, Eq, PartialEq)]
struct ResolvedTransitionDeleteVersion {
version_id_exact: bool,
remote_already_missing: bool,
}
async fn acquire_free_version_tier_lease( async fn acquire_free_version_tier_lease(
oi: &ObjectInfo, oi: &ObjectInfo,
tier_config_mgr: &Arc<RwLock<TierConfigMgr>>, tier_config_mgr: &Arc<RwLock<TierConfigMgr>>,
) -> Result<(TierOperationLease, TransitionDeleteVersionPlan), std::io::Error> { ) -> Result<(TierOperationLease, bool), std::io::Error> {
let delete_plan = transition_remote_version_delete_plan(oi)?; let version_id_exact = validate_transition_remote_version(oi)?;
let identity = tier_destination_id_from_metadata(&oi.user_defined)? let identity = tier_destination_id_from_metadata(&oi.user_defined)?
.ok_or_else(|| std::io::Error::other("tier free-version has no durable backend identity"))?; .ok_or_else(|| std::io::Error::other("tier free-version has no durable backend identity"))?;
let lease = let lease =
TierConfigMgr::acquire_operation_lease_for_backend_identity(tier_config_mgr, &oi.transitioned_object.tier, identity) TierConfigMgr::acquire_operation_lease_for_backend_identity(tier_config_mgr, &oi.transitioned_object.tier, identity)
.await .await
.map_err(std::io::Error::other)?; .map_err(std::io::Error::other)?;
Ok((lease, delete_plan)) Ok((lease, version_id_exact))
}
async fn resolve_transition_delete_version_plan(
oi: &ObjectInfo,
lease: &TierOperationLease,
delete_plan: TransitionDeleteVersionPlan,
) -> Result<ResolvedTransitionDeleteVersion, std::io::Error> {
match delete_plan {
TransitionDeleteVersionPlan::Direct { version_id_exact } => Ok(ResolvedTransitionDeleteVersion {
version_id_exact,
remote_already_missing: false,
}),
TransitionDeleteVersionPlan::ProbeLegacyUnknown => {
let expected_version = oi.transitioned_object.version_id.as_str();
if expected_version.is_empty() {
return Err(std::io::Error::new(
std::io::ErrorKind::WouldBlock,
"remote tier cannot safely delete a legacy object without an exact version ID",
));
}
let probe = lease
.probe_transition_version(&oi.transitioned_object.name, expected_version)
.await?;
match (expected_version, probe) {
(expected, crate::services::tier::warm_backend::TransitionCandidateProbe::VersionedPresent(actual))
if expected == actual =>
{
lease.validate_remote_version_id(expected)?;
Ok(ResolvedTransitionDeleteVersion {
version_id_exact: true,
remote_already_missing: false,
})
}
(_, crate::services::tier::warm_backend::TransitionCandidateProbe::Missing) => {
Ok(ResolvedTransitionDeleteVersion {
version_id_exact: false,
remote_already_missing: true,
})
}
(_, crate::services::tier::warm_backend::TransitionCandidateProbe::Unsupported) => Err(std::io::Error::new(
std::io::ErrorKind::Unsupported,
"remote tier cannot prove legacy transition delete state",
)),
_ => Err(std::io::Error::new(
std::io::ErrorKind::WouldBlock,
"remote tier object version state is unknown",
)),
}
}
}
}
async fn execute_resolved_transition_delete(
oi: &ObjectInfo,
lease: &TierOperationLease,
resolved: ResolvedTransitionDeleteVersion,
) -> Result<(), std::io::Error> {
if !resolved.remote_already_missing {
delete_object_from_remote_tier_with_lease_idempotent(
&oi.transitioned_object.name,
&oi.transitioned_object.version_id,
lease,
resolved.version_id_exact,
)
.await?;
}
Ok(())
} }
async fn delete_free_version_remote_object_with_lease( async fn delete_free_version_remote_object_with_lease(
oi: &ObjectInfo, oi: &ObjectInfo,
lease: &TierOperationLease, lease: &TierOperationLease,
delete_plan: TransitionDeleteVersionPlan, version_id_exact: bool,
) -> Result<(), std::io::Error> { ) -> Result<(), std::io::Error> {
let resolved = resolve_transition_delete_version_plan(oi, lease, delete_plan).await?; delete_object_from_remote_tier_with_lease_idempotent(
execute_resolved_transition_delete(oi, lease, resolved).await &oi.transitioned_object.name,
&oi.transitioned_object.version_id,
lease,
version_id_exact,
)
.await?;
Ok(())
} }
fn free_version_physical_topology_generation(api: &ECStore) -> String { fn free_version_physical_topology_generation(api: &ECStore) -> String {
@@ -781,16 +641,6 @@ fn free_version_remote_tuple_matches(candidate: &ObjectInfo, expected: &ObjectIn
if candidate.transition_version_state == rustfs_filemeta::TransitionVersionState::Unknown if candidate.transition_version_state == rustfs_filemeta::TransitionVersionState::Unknown
|| expected.transition_version_state == rustfs_filemeta::TransitionVersionState::Unknown || expected.transition_version_state == rustfs_filemeta::TransitionVersionState::Unknown
{ {
let candidate_legacy_missing = legacy_transition_version_state_missing(candidate)?;
let expected_legacy_missing = legacy_transition_version_state_missing(expected)?;
if candidate.transition_version_state == rustfs_filemeta::TransitionVersionState::Unknown
&& expected.transition_version_state == rustfs_filemeta::TransitionVersionState::Unknown
&& candidate_legacy_missing
&& expected_legacy_missing
&& candidate.transitioned_object.version_id == expected.transitioned_object.version_id
{
return Ok(true);
}
return Err(std::io::Error::new( return Err(std::io::Error::new(
std::io::ErrorKind::WouldBlock, std::io::ErrorKind::WouldBlock,
"tier free-version remote version state is unknown", "tier free-version remote version state is unknown",
@@ -866,7 +716,7 @@ async fn cleanup_free_version_exact(api: Arc<ECStore>, oi: &ObjectInfo, cancel:
.acquire_bucket_lifecycle_read_lock(&oi.bucket) .acquire_bucket_lifecycle_read_lock(&oi.bucket)
.await .await
.map_err(std::io::Error::other)?; .map_err(std::io::Error::other)?;
let (lease, delete_plan) = acquire_free_version_tier_lease(oi, &api.tier_config_mgr()).await?; let (lease, version_id_exact) = acquire_free_version_tier_lease(oi, &api.tier_config_mgr()).await?;
let local_object = encode_dir_object(&oi.name); let local_object = encode_dir_object(&oi.name);
let object_guards = api let object_guards = api
.acquire_all_physical_object_write_locks("tier_free_version_cleanup", &oi.bucket, &local_object) .acquire_all_physical_object_write_locks("tier_free_version_cleanup", &oi.bucket, &local_object)
@@ -884,30 +734,16 @@ async fn cleanup_free_version_exact(api: Arc<ECStore>, oi: &ObjectInfo, cancel:
"tier free-version cleanup fence is invalid before remote delete", "tier free-version cleanup fence is invalid before remote delete",
)); ));
} }
let resolved = tokio::select! {
_ = cancel.cancelled() => {
return Err(std::io::Error::new(std::io::ErrorKind::Interrupted, "tier free-version cleanup was cancelled"));
}
result = tokio::time::timeout_at(deadline, resolve_transition_delete_version_plan(oi, &lease, delete_plan)) => {
result.map_err(|_| {
std::io::Error::new(std::io::ErrorKind::TimedOut, "tier free-version remote probe timed out")
})??
}
};
if !free_version_cleanup_fences_current(&topology_generation, &api, &bucket_guard, &object_guards, &lease, cancel, deadline) {
return Err(std::io::Error::new(
std::io::ErrorKind::WouldBlock,
"tier free-version cleanup fence changed after remote probe",
));
}
tokio::select! { tokio::select! {
_ = cancel.cancelled() => { _ = cancel.cancelled() => {
return Err(std::io::Error::new(std::io::ErrorKind::Interrupted, "tier free-version cleanup was cancelled")); return Err(std::io::Error::new(std::io::ErrorKind::Interrupted, "tier free-version cleanup was cancelled"));
} }
result = tokio::time::timeout_at(deadline, execute_resolved_transition_delete(oi, &lease, resolved)) => { result = tokio::time::timeout_at(
result.map_err(|_| { deadline,
std::io::Error::new(std::io::ErrorKind::TimedOut, "tier free-version remote delete timed out") delete_free_version_remote_object_with_lease(oi, &lease, version_id_exact),
})??; ) => {
result
.map_err(|_| std::io::Error::new(std::io::ErrorKind::TimedOut, "tier free-version remote delete timed out"))??;
} }
} }
if !free_version_cleanup_fences_current(&topology_generation, &api, &bucket_guard, &object_guards, &lease, cancel, deadline) { if !free_version_cleanup_fences_current(&topology_generation, &api, &bucket_guard, &object_guards, &lease, cancel, deadline) {
@@ -955,8 +791,8 @@ async fn delete_free_version_remote_object(
oi: &ObjectInfo, oi: &ObjectInfo,
tier_config_mgr: &Arc<RwLock<TierConfigMgr>>, tier_config_mgr: &Arc<RwLock<TierConfigMgr>>,
) -> Result<(), std::io::Error> { ) -> Result<(), std::io::Error> {
let (lease, delete_plan) = acquire_free_version_tier_lease(oi, tier_config_mgr).await?; let (lease, version_id_exact) = acquire_free_version_tier_lease(oi, tier_config_mgr).await?;
delete_free_version_remote_object_with_lease(oi, &lease, delete_plan).await delete_free_version_remote_object_with_lease(oi, &lease, version_id_exact).await
} }
#[allow( #[allow(
@@ -972,8 +808,8 @@ where
F: FnOnce() -> Fut, F: FnOnce() -> Fut,
Fut: std::future::Future<Output = T>, Fut: std::future::Future<Output = T>,
{ {
let (lease, delete_plan) = acquire_free_version_tier_lease(oi, tier_config_mgr).await?; let (lease, version_id_exact) = acquire_free_version_tier_lease(oi, tier_config_mgr).await?;
delete_free_version_remote_object_with_lease(oi, &lease, delete_plan).await?; delete_free_version_remote_object_with_lease(oi, &lease, version_id_exact).await?;
let result = delete_local().await; let result = delete_local().await;
drop(lease); drop(lease);
Ok(result) Ok(result)
@@ -4852,39 +4688,6 @@ fn validate_transition_remote_version(oi: &ObjectInfo) -> Result<bool, std::io::
} }
} }
#[derive(Clone, Copy, Debug, Eq, PartialEq)]
enum TransitionReadVersionPlan {
Direct,
ProbeLegacyUnversioned,
}
const LEGACY_TRANSITION_READ_PROBE_TIMEOUT: StdDuration = StdDuration::from_secs(30);
fn transition_remote_version_read_plan(oi: &ObjectInfo) -> Result<TransitionReadVersionPlan, std::io::Error> {
let version = oi.transitioned_object.version_id.as_str();
match oi.transition_version_state {
rustfs_filemeta::TransitionVersionState::Unknown => {
if !legacy_transition_version_state_missing(oi)? {
return validate_transition_remote_version(oi).map(|_| TransitionReadVersionPlan::Direct);
}
if version.is_empty() {
Ok(TransitionReadVersionPlan::ProbeLegacyUnversioned)
} else {
Ok(TransitionReadVersionPlan::Direct)
}
}
rustfs_filemeta::TransitionVersionState::KnownDisabled if version.is_empty() => Ok(TransitionReadVersionPlan::Direct),
rustfs_filemeta::TransitionVersionState::SuspendedNull if version == "null" => Ok(TransitionReadVersionPlan::Direct),
rustfs_filemeta::TransitionVersionState::Exact if !version.is_empty() && version != "null" => {
Ok(TransitionReadVersionPlan::Direct)
}
_ => Err(std::io::Error::new(
std::io::ErrorKind::InvalidData,
"remote tier object version state conflicts with its version ID",
)),
}
}
// The resolver joins the tier manager as the second injected port this read // The resolver joins the tier manager as the second injected port this read
// needs; grouping the request half into a struct would churn every call site of // needs; grouping the request half into a struct would churn every call site of
// a bug fix. // a bug fix.
@@ -4899,12 +4702,7 @@ pub(crate) async fn get_transitioned_object_reader_with_tier_manager(
tier_config_mgr: &Arc<RwLock<TierConfigMgr>>, tier_config_mgr: &Arc<RwLock<TierConfigMgr>>,
resolver: Option<&dyn ObjectEncryptionResolver>, resolver: Option<&dyn ObjectEncryptionResolver>,
) -> Result<GetObjectReader, std::io::Error> { ) -> Result<GetObjectReader, std::io::Error> {
let read_plan = transition_remote_version_read_plan(oi)?; validate_transition_remote_version(oi)?;
// Reject invalid ranges and encryption requests before a compatibility
// probe can amplify them into remote listing work.
let plan = ReadPlan::build_for_request(rs.clone(), oi, opts, h, resolver)
.await
.map_err(|err| std::io::Error::other(format!("building the read plan for {bucket}/{object} failed: {err}")))?;
let expected_identity = tier_destination_id_from_metadata(&oi.user_defined)?; let expected_identity = tier_destination_id_from_metadata(&oi.user_defined)?;
let lease = match expected_identity { let lease = match expected_identity {
Some(identity) => { Some(identity) => {
@@ -4918,36 +4716,7 @@ pub(crate) async fn get_transitioned_object_reader_with_tier_manager(
Err(err) => return Err(std::io::Error::other(err)), Err(err) => return Err(std::io::Error::other(err)),
}; };
match read_plan { tgt_client.validate_remote_version_id(&oi.transitioned_object.version_id)?;
TransitionReadVersionPlan::Direct => {
tgt_client.validate_remote_version_id(&oi.transitioned_object.version_id)?;
}
TransitionReadVersionPlan::ProbeLegacyUnversioned => {
// RUSTFS_COMPAT_TODO(backlog#2203): remove operation-time probing
// after an admin reconcile can persist every proven legacy state.
let probe = tokio::time::timeout(
LEGACY_TRANSITION_READ_PROBE_TIMEOUT,
tgt_client.probe_transition_candidate(&oi.transitioned_object.name),
)
.await
.map_err(|_| std::io::Error::new(std::io::ErrorKind::TimedOut, "legacy remote tier version probe timed out"))??;
match probe {
crate::services::tier::warm_backend::TransitionCandidateProbe::UnversionedPresent => {}
crate::services::tier::warm_backend::TransitionCandidateProbe::Unsupported => {
return Err(std::io::Error::new(
std::io::ErrorKind::Unsupported,
"remote tier cannot prove legacy unversioned transition state",
));
}
_ => {
return Err(std::io::Error::new(
std::io::ErrorKind::InvalidData,
"remote tier object version state is unknown",
));
}
}
}
}
// The same read plan the local path uses, so the tier fetch is positioned in // The same read plan the local path uses, so the tier fetch is positioned in
// the object's *stored* coordinate system and the stream is handed the same // the object's *stored* coordinate system and the stream is handed the same
@@ -4955,6 +4724,9 @@ pub(crate) async fn get_transitioned_object_reader_with_tier_manager(
// through a plaintext-coordinate range and skipping the transform is how a // through a plaintext-coordinate range and skipping the transform is how a
// transitioned SSE object used to come back as silently corrupt bytes of the // transitioned SSE object used to come back as silently corrupt bytes of the
// right length (rustfs/rustfs#6025). // right length (rustfs/rustfs#6025).
let plan = ReadPlan::build_for_request(rs.clone(), oi, opts, h, resolver)
.await
.map_err(|err| std::io::Error::other(format!("building the read plan for {bucket}/{object} failed: {err}")))?;
let (off, length) = (plan.storage_offset() as i64, plan.storage_length()); let (off, length) = (plan.storage_offset() as i64, plan.storage_length());
let mut gopts = WarmBackendGetOpts::default(); let mut gopts = WarmBackendGetOpts::default();
@@ -5827,13 +5599,11 @@ mod tests {
use crate::layout::endpoints::{EndpointServerPools, Endpoints, PoolEndpoints}; use crate::layout::endpoints::{EndpointServerPools, Endpoints, PoolEndpoints};
use crate::object_api::{ObjectInfo, ObjectOptions, PutObjReader}; use crate::object_api::{ObjectInfo, ObjectOptions, PutObjReader};
#[cfg(feature = "test-util")] #[cfg(feature = "test-util")]
use crate::services::tier::test_util::MockWarmOp;
#[cfg(feature = "test-util")]
use crate::services::tier::test_util::register_mock_tier; use crate::services::tier::test_util::register_mock_tier;
#[cfg(feature = "test-util")] #[cfg(feature = "test-util")]
use crate::services::tier::tier::TierConfigMgr; use crate::services::tier::tier::TierConfigMgr;
#[cfg(feature = "test-util")] #[cfg(feature = "test-util")]
use crate::services::tier::warm_backend::{TransitionCandidateProbe, WarmBackend as _}; use crate::services::tier::warm_backend::WarmBackend as _;
use crate::set_disk::{MultipartCommitBarrier, MultipartCommitPause}; use crate::set_disk::{MultipartCommitBarrier, MultipartCommitPause};
use crate::set_disk::{RUSTFS_MULTIPART_BUCKET_KEY, RUSTFS_MULTIPART_OBJECT_KEY}; use crate::set_disk::{RUSTFS_MULTIPART_BUCKET_KEY, RUSTFS_MULTIPART_OBJECT_KEY};
use crate::storage_api_contracts::namespace::NamespaceLocking as _; use crate::storage_api_contracts::namespace::NamespaceLocking as _;
@@ -6529,75 +6299,7 @@ mod tests {
#[cfg(feature = "test-util")] #[cfg(feature = "test-util")]
#[tokio::test] #[tokio::test]
async fn transitioned_get_allows_legacy_unknown_exact_version_for_non_destructive_read() { async fn transitioned_get_rejects_unknown_version_state_before_backend_io() {
let manager = TierConfigMgr::new();
let tier = format!("COLDTIER{}", &Uuid::new_v4().simple().to_string()[..8]).to_uppercase();
let backend = register_mock_tier(&manager, &tier).await;
let remote_object = format!("remote/{}", Uuid::new_v4());
let body = Bytes::from_static(b"legacy transitioned object body");
let remote_version = backend
.put(
&remote_object,
ReaderImpl::Body(body.clone()),
i64::try_from(body.len()).expect("body length should fit"),
)
.await
.expect("mock remote object should be stored");
let mut user_defined = HashMap::new();
insert_legacy_transition_version_id(&mut user_defined, &remote_version);
let object_info = ObjectInfo {
bucket: "bucket".to_string(),
name: "object".to_string(),
size: i64::try_from(body.len()).expect("body length should fit"),
transitioned_object: TransitionedObject {
name: remote_object,
version_id: remote_version,
status: crate::bucket::lifecycle::lifecycle::TRANSITION_COMPLETE.to_string(),
tier: tier.clone(),
..Default::default()
},
transition_version_state: rustfs_filemeta::TransitionVersionState::Unknown,
user_defined: user_defined.into(),
..Default::default()
};
let range = Some(crate::storage_api_contracts::range::HTTPRangeSpec {
is_suffix_length: false,
start: 7,
end: 18,
});
let mut reader = get_transitioned_object_reader_with_tier_manager(
&object_info.bucket,
&object_info.name,
&range,
&HeaderMap::new(),
&object_info,
&ObjectOptions::default(),
&manager,
None,
)
.await
.expect("legacy unknown state should still allow a non-destructive read");
let mut got = Vec::new();
reader
.stream
.read_to_end(&mut got)
.await
.expect("transitioned reader should drain");
assert_eq!(got, &body.as_ref()[7..=18]);
assert_eq!(backend.get_count().await, 1);
assert_eq!(backend.remove_count().await, 0);
assert_eq!(
TierConfigMgr::active_operation_lease_count(&manager, &tier).await,
0,
"tier generation lease should release after EOF"
);
}
#[cfg(feature = "test-util")]
#[tokio::test]
async fn transitioned_get_rejects_explicit_unknown_version_state_before_backend_io() {
let manager = TierConfigMgr::new(); let manager = TierConfigMgr::new();
let tier = format!("COLDTIER{}", &Uuid::new_v4().simple().to_string()[..8]).to_uppercase(); let tier = format!("COLDTIER{}", &Uuid::new_v4().simple().to_string()[..8]).to_uppercase();
let backend = register_mock_tier(&manager, &tier).await; let backend = register_mock_tier(&manager, &tier).await;
@@ -6613,7 +6315,6 @@ mod tests {
..Default::default() ..Default::default()
}, },
transition_version_state: rustfs_filemeta::TransitionVersionState::Unknown, transition_version_state: rustfs_filemeta::TransitionVersionState::Unknown,
user_defined: user_defined_with_transition_version_state(rustfs_filemeta::TransitionVersionState::Unknown).into(),
..Default::default() ..Default::default()
}; };
@@ -6629,202 +6330,19 @@ mod tests {
) )
.await .await
{ {
Ok(_) => panic!("explicit unknown remote version state must fail before backend IO"), Ok(_) => panic!("unknown remote version state must fail before backend IO"),
Err(err) => err, Err(err) => err,
}; };
assert_eq!(err.kind(), std::io::ErrorKind::InvalidData); assert_eq!(err.kind(), std::io::ErrorKind::InvalidData);
assert_eq!(backend.op_log().await, Vec::<MockWarmOp>::new());
assert_eq!(backend.get_count().await, 0); assert_eq!(backend.get_count().await, 0);
} }
#[cfg(feature = "test-util")] #[cfg(feature = "test-util")]
#[tokio::test] #[tokio::test]
async fn transitioned_get_rejects_present_but_invalid_legacy_version_metadata() { async fn free_version_delete_rejects_unknown_version_state_before_backend_io() {
let manager = TierConfigMgr::new();
let tier = format!("COLDTIER{}", &Uuid::new_v4().simple().to_string()[..8]).to_uppercase();
let backend = register_mock_tier(&manager, &tier).await;
for persisted_version in [
Uuid::nil().to_string(),
"\u{fffd}".to_string(),
"bad\u{0001}version".to_string(),
] {
let mut user_defined = HashMap::new();
insert_legacy_transition_version_id(&mut user_defined, &persisted_version);
let object_info = ObjectInfo {
bucket: "bucket".to_string(),
name: "object".to_string(),
size: 1,
transitioned_object: TransitionedObject {
name: "remote/object".to_string(),
version_id: String::new(),
status: crate::bucket::lifecycle::lifecycle::TRANSITION_COMPLETE.to_string(),
tier: tier.clone(),
..Default::default()
},
transition_version_state: rustfs_filemeta::TransitionVersionState::Unknown,
user_defined: user_defined.into(),
..Default::default()
};
let err = match get_transitioned_object_reader_with_tier_manager(
&object_info.bucket,
&object_info.name,
&None,
&HeaderMap::new(),
&object_info,
&ObjectOptions::default(),
&manager,
None,
)
.await
{
Ok(_) => panic!("present but invalid legacy version metadata must fail before backend IO"),
Err(err) => err,
};
assert_eq!(err.kind(), std::io::ErrorKind::InvalidData);
}
assert_eq!(backend.op_log().await, Vec::<MockWarmOp>::new());
}
#[cfg(feature = "test-util")]
#[tokio::test]
async fn transitioned_get_probes_legacy_empty_unknown_state_before_unversioned_read() {
let manager = TierConfigMgr::new();
let tier = format!("COLDTIER{}", &Uuid::new_v4().simple().to_string()[..8]).to_uppercase();
let backend = register_mock_tier(&manager, &tier).await;
backend.set_put_remote_version(Some(String::new())).await;
let remote_object = format!("remote/{}", Uuid::new_v4());
let body = Bytes::from_static(b"legacy unversioned transitioned object body");
let remote_version = backend
.put(
&remote_object,
ReaderImpl::Body(body.clone()),
i64::try_from(body.len()).expect("body length should fit"),
)
.await
.expect("mock remote object should be stored");
assert!(remote_version.is_empty());
let object_info = ObjectInfo {
bucket: "bucket".to_string(),
name: "object".to_string(),
size: i64::try_from(body.len()).expect("body length should fit"),
transitioned_object: TransitionedObject {
name: remote_object.clone(),
version_id: String::new(),
status: crate::bucket::lifecycle::lifecycle::TRANSITION_COMPLETE.to_string(),
tier: tier.clone(),
..Default::default()
},
transition_version_state: rustfs_filemeta::TransitionVersionState::Unknown,
user_defined: HashMap::from([("x-minio-internal-transitioned-versionID".to_string(), String::new())]).into(),
..Default::default()
};
let mut reader = get_transitioned_object_reader_with_tier_manager(
&object_info.bucket,
&object_info.name,
&None,
&HeaderMap::new(),
&object_info,
&ObjectOptions::default(),
&manager,
None,
)
.await
.expect("probe-proven legacy unversioned state should allow a non-destructive read");
let mut got = Vec::new();
reader
.stream
.read_to_end(&mut got)
.await
.expect("transitioned reader should drain");
assert_eq!(got, body.as_ref());
assert_eq!(backend.remove_count().await, 0);
assert_eq!(
backend.op_log().await,
vec![
MockWarmOp::Put {
object: remote_object.clone()
},
MockWarmOp::Probe {
object: remote_object.clone()
},
MockWarmOp::Get { object: remote_object },
]
);
assert_eq!(
TierConfigMgr::active_operation_lease_count(&manager, &tier).await,
0,
"tier generation lease should release after EOF"
);
}
#[cfg(feature = "test-util")]
#[tokio::test]
async fn transitioned_get_rejects_ambiguous_empty_unknown_state_without_backend_get() {
let manager = TierConfigMgr::new();
let tier = format!("COLDTIER{}", &Uuid::new_v4().simple().to_string()[..8]).to_uppercase();
let backend = register_mock_tier(&manager, &tier).await;
let remote_object = format!("remote/{}", Uuid::new_v4());
backend
.set_transition_candidate_probe_override(Some(TransitionCandidateProbe::VersionedPresent(
"versioned-candidate".to_string(),
)))
.await;
let object_info = ObjectInfo {
bucket: "bucket".to_string(),
name: "object".to_string(),
size: 1,
transitioned_object: TransitionedObject {
name: remote_object.clone(),
version_id: String::new(),
status: crate::bucket::lifecycle::lifecycle::TRANSITION_COMPLETE.to_string(),
tier,
..Default::default()
},
transition_version_state: rustfs_filemeta::TransitionVersionState::Unknown,
..Default::default()
};
let err = match get_transitioned_object_reader_with_tier_manager(
&object_info.bucket,
&object_info.name,
&None,
&HeaderMap::new(),
&object_info,
&ObjectOptions::default(),
&manager,
None,
)
.await
{
Ok(_) => panic!("versioned legacy unknown state without stored version must fail before backend GET"),
Err(err) => err,
};
assert_eq!(err.kind(), std::io::ErrorKind::InvalidData);
assert_eq!(backend.op_log().await, vec![MockWarmOp::Probe { object: remote_object }]);
assert_eq!(backend.get_count().await, 0);
assert_eq!(backend.remove_count().await, 0);
}
#[cfg(feature = "test-util")]
#[tokio::test]
async fn free_version_delete_rejects_explicit_unknown_before_backend_io() {
let manager = TierConfigMgr::new(); let manager = TierConfigMgr::new();
let backend = register_mock_tier(&manager, "WARM").await; let backend = register_mock_tier(&manager, "WARM").await;
let identity = test_tier_destination_identity(&manager, "WARM").await;
let mut user_defined = user_defined_with_tier_destination_identity(identity);
rustfs_utils::http::metadata_compat::insert_str(
&mut user_defined,
rustfs_utils::http::metadata_compat::SUFFIX_TRANSITIONED_VERSION_STATE,
rustfs_filemeta::TransitionVersionState::Unknown.as_str().to_string(),
);
let object_info = ObjectInfo { let object_info = ObjectInfo {
transitioned_object: TransitionedObject { transitioned_object: TransitionedObject {
name: "remote/object".to_string(), name: "remote/object".to_string(),
@@ -6833,251 +6351,17 @@ mod tests {
..Default::default() ..Default::default()
}, },
transition_version_state: rustfs_filemeta::TransitionVersionState::Unknown, transition_version_state: rustfs_filemeta::TransitionVersionState::Unknown,
user_defined: user_defined.into(),
..Default::default() ..Default::default()
}; };
let err = super::delete_free_version_remote_object(&object_info, &manager) let err = super::delete_free_version_remote_object(&object_info, &manager)
.await .await
.expect_err("explicit unknown cleanup must fail before backend IO"); .expect_err("unknown remote version state must fail before backend IO");
assert_eq!(err.kind(), std::io::ErrorKind::InvalidData); assert_eq!(err.kind(), std::io::ErrorKind::InvalidData);
assert!(err.to_string().contains("version state is unknown"));
assert_eq!(backend.op_log().await, Vec::<MockWarmOp>::new());
assert_eq!(backend.remove_count().await, 0); assert_eq!(backend.remove_count().await, 0);
} }
#[cfg(feature = "test-util")]
async fn test_tier_destination_identity(
manager: &Arc<tokio::sync::RwLock<TierConfigMgr>>,
tier: &str,
) -> crate::services::tier::tier::TierDestinationId {
TierConfigMgr::acquire_operation_lease(manager, tier)
.await
.expect("test tier lease should be available")
.backend_identity()
}
#[cfg(feature = "test-util")]
fn user_defined_with_tier_destination_identity(
identity: crate::services::tier::tier::TierDestinationId,
) -> HashMap<String, String> {
let mut user_defined = HashMap::new();
rustfs_utils::http::metadata_compat::insert_str(
&mut user_defined,
rustfs_utils::http::metadata_compat::SUFFIX_TRANSITION_TIER_DESTINATION_ID,
rustfs_utils::crypto::hex(identity),
);
user_defined
}
#[cfg(feature = "test-util")]
fn user_defined_with_transition_version_state(state: rustfs_filemeta::TransitionVersionState) -> HashMap<String, String> {
let mut user_defined = HashMap::new();
rustfs_utils::http::metadata_compat::insert_str(
&mut user_defined,
rustfs_utils::http::metadata_compat::SUFFIX_TRANSITIONED_VERSION_STATE,
state.as_str().to_string(),
);
user_defined
}
#[cfg(feature = "test-util")]
fn insert_legacy_transition_version_id(user_defined: &mut HashMap<String, String>, version_id: &str) {
rustfs_utils::http::metadata_compat::insert_str(
user_defined,
rustfs_utils::http::metadata_compat::SUFFIX_TRANSITIONED_VERSION_ID,
version_id.to_string(),
);
}
#[cfg(feature = "test-util")]
#[tokio::test]
async fn free_version_tuple_rejects_mixed_legacy_missing_and_explicit_unknown() {
let manager = TierConfigMgr::new();
register_mock_tier(&manager, "WARM").await;
let identity = test_tier_destination_identity(&manager, "WARM").await;
let mut legacy_metadata = user_defined_with_tier_destination_identity(identity);
insert_legacy_transition_version_id(&mut legacy_metadata, "legacy-version");
let mut explicit_metadata = legacy_metadata.clone();
rustfs_utils::http::metadata_compat::insert_str(
&mut explicit_metadata,
rustfs_utils::http::metadata_compat::SUFFIX_TRANSITIONED_VERSION_STATE,
rustfs_filemeta::TransitionVersionState::Unknown.as_str().to_string(),
);
let make_info = |user_defined: HashMap<String, String>| ObjectInfo {
transitioned_object: TransitionedObject {
name: "remote/object".to_string(),
version_id: "legacy-version".to_string(),
tier: "WARM".to_string(),
..Default::default()
},
transition_version_state: rustfs_filemeta::TransitionVersionState::Unknown,
user_defined: user_defined.into(),
..Default::default()
};
let err = super::free_version_remote_tuple_matches(&make_info(legacy_metadata), &make_info(explicit_metadata))
.expect_err("mixed legacy-missing and explicit unknown provenance must fail closed");
assert_eq!(err.kind(), std::io::ErrorKind::WouldBlock);
}
#[cfg(feature = "test-util")]
#[tokio::test]
async fn free_version_delete_probes_exact_version_hidden_by_current_delete_marker() {
let manager = TierConfigMgr::new();
let tier = "WARM";
let backend = register_mock_tier(&manager, tier).await;
let identity = test_tier_destination_identity(&manager, tier).await;
let remote_object = format!("remote/{}", Uuid::new_v4());
let body = Bytes::from_static(b"legacy exact cleanup body");
let remote_version = backend
.put(
&remote_object,
ReaderImpl::Body(body),
i64::try_from(b"legacy exact cleanup body".len()).expect("body length should fit"),
)
.await
.expect("mock remote object should be stored");
let mut user_defined = user_defined_with_tier_destination_identity(identity);
insert_legacy_transition_version_id(&mut user_defined, &remote_version);
backend
.set_transition_candidate_probe_override(Some(TransitionCandidateProbe::Missing))
.await;
assert_eq!(
backend
.probe_transition_candidate_state(&remote_object)
.await
.expect("current remote view should be readable"),
TransitionCandidateProbe::Missing,
"a current delete marker must hide the historical data version from an unversioned probe"
);
backend.clear_op_log().await;
let object_info = ObjectInfo {
transitioned_object: TransitionedObject {
name: remote_object.clone(),
version_id: remote_version,
tier: tier.to_string(),
..Default::default()
},
transition_version_state: rustfs_filemeta::TransitionVersionState::Unknown,
user_defined: user_defined.into(),
..Default::default()
};
super::delete_free_version_remote_object(&object_info, &manager)
.await
.expect("probe-proven legacy exact cleanup should delete the remote version");
super::delete_free_version_remote_object(&object_info, &manager)
.await
.expect("a retry after the exact remote version is already missing should be idempotent");
assert_eq!(
backend.op_log().await,
vec![
MockWarmOp::Get {
object: remote_object.clone()
},
MockWarmOp::Remove {
object: remote_object.clone()
},
MockWarmOp::Get {
object: remote_object.clone()
},
]
);
assert_eq!(
backend.remove_versions().await,
vec![(remote_object, object_info.transitioned_object.version_id)]
);
}
#[cfg(feature = "test-util")]
#[tokio::test]
async fn free_version_delete_retains_legacy_unknown_unversioned_object() {
let manager = TierConfigMgr::new();
let tier = "WARM";
let backend = register_mock_tier(&manager, tier).await;
backend.set_put_remote_version(Some(String::new())).await;
let identity = test_tier_destination_identity(&manager, tier).await;
let remote_object = format!("remote/{}", Uuid::new_v4());
let body = Bytes::from_static(b"legacy unversioned cleanup body");
let remote_version = backend
.put(
&remote_object,
ReaderImpl::Body(body),
i64::try_from(b"legacy unversioned cleanup body".len()).expect("body length should fit"),
)
.await
.expect("mock remote object should be stored");
assert!(remote_version.is_empty());
backend.clear_op_log().await;
let mut user_defined = user_defined_with_tier_destination_identity(identity);
user_defined.insert("x-minio-internal-transitioned-versionID".to_string(), String::new());
let object_info = ObjectInfo {
transitioned_object: TransitionedObject {
name: remote_object.clone(),
version_id: String::new(),
tier: tier.to_string(),
..Default::default()
},
transition_version_state: rustfs_filemeta::TransitionVersionState::Unknown,
user_defined: user_defined.into(),
..Default::default()
};
let err = super::delete_free_version_remote_object(&object_info, &manager)
.await
.expect_err("legacy unversioned cleanup cannot exclude a versioning-state race");
assert_eq!(err.kind(), std::io::ErrorKind::WouldBlock);
assert!(backend.op_log().await.is_empty());
assert_eq!(backend.remove_count().await, 0);
assert!(backend.remove_versions().await.is_empty());
}
#[cfg(feature = "test-util")]
#[tokio::test]
async fn free_version_delete_does_not_remove_a_different_remote_version() {
let manager = TierConfigMgr::new();
let tier = "WARM";
let backend = register_mock_tier(&manager, tier).await;
let identity = test_tier_destination_identity(&manager, tier).await;
let remote_object = format!("remote/{}", Uuid::new_v4());
backend.set_put_remote_version(Some("different-version".to_string())).await;
backend
.put(
&remote_object,
ReaderImpl::Body(Bytes::from_static(b"different remote version")),
i64::try_from(b"different remote version".len()).expect("body length should fit"),
)
.await
.expect("different remote version should be stored");
backend.clear_op_log().await;
let mut user_defined = user_defined_with_tier_destination_identity(identity);
insert_legacy_transition_version_id(&mut user_defined, "legacy-version");
let object_info = ObjectInfo {
transitioned_object: TransitionedObject {
name: remote_object.clone(),
version_id: "legacy-version".to_string(),
tier: tier.to_string(),
..Default::default()
},
transition_version_state: rustfs_filemeta::TransitionVersionState::Unknown,
user_defined: user_defined.into(),
..Default::default()
};
super::delete_free_version_remote_object(&object_info, &manager)
.await
.expect("a missing exact legacy version should be an idempotent cleanup success");
assert_eq!(backend.op_log().await, vec![MockWarmOp::Get { object: remote_object }]);
assert_eq!(backend.remove_count().await, 0);
assert!(backend.remove_versions().await.is_empty());
}
#[cfg(feature = "test-util")] #[cfg(feature = "test-util")]
#[tokio::test] #[tokio::test]
async fn free_version_remote_delete_requires_persisted_destination_identity() { async fn free_version_remote_delete_requires_persisted_destination_identity() {
@@ -1170,7 +1170,6 @@ pub async fn save_manual_transition_job_record_if_current(
data.clone(), data.clone(),
&ObjectOptions { &ObjectOptions {
max_parity: true, max_parity: true,
write_completion: crate::object_api::WriteCompletion::TailDrained,
http_preconditions: Some(HTTPPreconditions { http_preconditions: Some(HTTPPreconditions {
if_match: Some(current_etag.to_string()), if_match: Some(current_etag.to_string()),
..Default::default() ..Default::default()
@@ -1243,7 +1242,6 @@ pub(crate) async fn save_manual_transition_worker_result_if_absent(
data, data,
&ObjectOptions { &ObjectOptions {
max_parity: true, max_parity: true,
write_completion: crate::object_api::WriteCompletion::TailDrained,
http_preconditions: Some(HTTPPreconditions { http_preconditions: Some(HTTPPreconditions {
if_none_match: Some("*".to_string()), if_none_match: Some("*".to_string()),
..Default::default() ..Default::default()
@@ -1272,7 +1270,6 @@ pub(crate) async fn save_manual_transition_task_if_absent(
data, data,
&ObjectOptions { &ObjectOptions {
max_parity: true, max_parity: true,
write_completion: crate::object_api::WriteCompletion::TailDrained,
http_preconditions: Some(HTTPPreconditions { http_preconditions: Some(HTTPPreconditions {
if_none_match: Some("*".to_string()), if_none_match: Some("*".to_string()),
..Default::default() ..Default::default()
@@ -1624,7 +1621,6 @@ pub async fn save_manual_transition_scope_admission_if_absent(
data.clone(), data.clone(),
&ObjectOptions { &ObjectOptions {
max_parity: true, max_parity: true,
write_completion: crate::object_api::WriteCompletion::TailDrained,
http_preconditions: Some(HTTPPreconditions { http_preconditions: Some(HTTPPreconditions {
if_none_match: Some("*".to_string()), if_none_match: Some("*".to_string()),
..Default::default() ..Default::default()
@@ -1676,7 +1672,6 @@ pub async fn save_manual_transition_scope_admission_if_current(
data.clone(), data.clone(),
&ObjectOptions { &ObjectOptions {
max_parity: true, max_parity: true,
write_completion: crate::object_api::WriteCompletion::TailDrained,
http_preconditions: Some(HTTPPreconditions { http_preconditions: Some(HTTPPreconditions {
if_match: Some(current_etag.to_string()), if_match: Some(current_etag.to_string()),
..Default::default() ..Default::default()
@@ -1733,7 +1733,6 @@ async fn save_config_if_none_fenced(
data.clone(), data.clone(),
&ObjectOptions { &ObjectOptions {
max_parity: true, max_parity: true,
write_completion: crate::object_api::WriteCompletion::TailDrained,
http_preconditions: Some(HTTPPreconditions { http_preconditions: Some(HTTPPreconditions {
if_none_match: Some("*".to_string()), if_none_match: Some("*".to_string()),
..Default::default() ..Default::default()
@@ -1833,7 +1832,6 @@ async fn save_decommission_manifest_checkpoint_if_match(
let mut opts = ObjectOptions { let mut opts = ObjectOptions {
max_parity: true, max_parity: true,
write_completion: crate::object_api::WriteCompletion::TailDrained,
no_lock: true, no_lock: true,
http_preconditions: Some(HTTPPreconditions { http_preconditions: Some(HTTPPreconditions {
if_match: Some(observed_etag), if_match: Some(observed_etag),
@@ -1962,7 +1960,6 @@ async fn save_config_if_match_fenced(
data.clone(), data.clone(),
&ObjectOptions { &ObjectOptions {
max_parity: true, max_parity: true,
write_completion: crate::object_api::WriteCompletion::TailDrained,
http_preconditions: Some(HTTPPreconditions { http_preconditions: Some(HTTPPreconditions {
if_match: Some(etag.to_string()), if_match: Some(etag.to_string()),
..Default::default() ..Default::default()
@@ -3783,7 +3780,6 @@ where
data.clone(), data.clone(),
&ObjectOptions { &ObjectOptions {
max_parity: true, max_parity: true,
write_completion: crate::object_api::WriteCompletion::TailDrained,
http_preconditions: Some(HTTPPreconditions { http_preconditions: Some(HTTPPreconditions {
if_none_match: Some("*".to_string()), if_none_match: Some("*".to_string()),
..Default::default() ..Default::default()
@@ -3873,7 +3869,6 @@ where
data.clone(), data.clone(),
&ObjectOptions { &ObjectOptions {
max_parity: true, max_parity: true,
write_completion: crate::object_api::WriteCompletion::TailDrained,
http_preconditions: Some(HTTPPreconditions { http_preconditions: Some(HTTPPreconditions {
if_match: Some(etag), if_match: Some(etag),
..Default::default() ..Default::default()
@@ -3898,7 +3893,6 @@ where
data.clone(), data.clone(),
&ObjectOptions { &ObjectOptions {
max_parity: true, max_parity: true,
write_completion: crate::object_api::WriteCompletion::TailDrained,
http_preconditions: Some(HTTPPreconditions { http_preconditions: Some(HTTPPreconditions {
if_none_match: Some("*".to_string()), if_none_match: Some("*".to_string()),
..Default::default() ..Default::default()
@@ -15,6 +15,8 @@
#![allow(unused_variables)] #![allow(unused_variables)]
#![allow(unused_mut)] #![allow(unused_mut)]
#![allow(unused_assignments)] #![allow(unused_assignments)]
#![allow(unused_must_use)]
#![allow(clippy::all)]
use super::runtime_boundary as runtime_sources; use super::runtime_boundary as runtime_sources;
use crate::bucket::lifecycle::bucket_lifecycle_ops::ExpiryOp; use crate::bucket::lifecycle::bucket_lifecycle_ops::ExpiryOp;
@@ -70,11 +72,9 @@ static REMOTE_DELETE_BREAKER: LazyLock<Mutex<RemoteDeleteBreaker>> = LazyLock::n
}); });
#[cfg(test)] #[cfg(test)]
type RemoteTierDeleteTestHook = Box<dyn Fn(&str, &str, &str) -> std::io::Result<()> + Send + Sync>; static REMOTE_TIER_DELETE_TEST_HOOK: std::sync::LazyLock<
std::sync::Mutex<Option<Box<dyn Fn(&str, &str, &str) -> std::io::Result<()> + Send + Sync>>>,
#[cfg(test)] > = std::sync::LazyLock::new(|| std::sync::Mutex::new(None));
static REMOTE_TIER_DELETE_TEST_HOOK: std::sync::LazyLock<std::sync::Mutex<Option<RemoteTierDeleteTestHook>>> =
std::sync::LazyLock::new(|| std::sync::Mutex::new(None));
#[derive(Debug)] #[derive(Debug)]
struct RemoteDeleteBreaker { struct RemoteDeleteBreaker {
@@ -107,7 +107,7 @@ impl RemoteDeleteBreaker {
fn prune(&mut self, now: Instant) { fn prune(&mut self, now: Instant) {
while let Some(ts) = self.failures.front().copied() { while let Some(ts) = self.failures.front().copied() {
if now.duration_since(ts) > self.window { if now.duration_since(ts) > self.window {
let _ = self.failures.pop_front(); self.failures.pop_front();
} else { } else {
break; break;
} }
@@ -137,10 +137,10 @@ fn is_signer_header_error(err: &std::io::Error) -> bool {
return false; return false;
} }
if let Some(source) = err.get_ref() if let Some(source) = err.get_ref() {
&& error_chain_contains_signer_header_marker(source) if error_chain_contains_signer_header_marker(source) {
{ return true;
return true; }
} }
let message = err.to_string().to_ascii_lowercase(); let message = err.to_string().to_ascii_lowercase();
@@ -205,7 +205,7 @@ impl ObjSweeper {
#[allow(dead_code, reason = "MinIO-parity surface with no caller in this port (backlog#1823)")] #[allow(dead_code, reason = "MinIO-parity surface with no caller in this port (backlog#1823)")]
pub fn with_version(&mut self, vid: Option<Uuid>) -> &Self { pub fn with_version(&mut self, vid: Option<Uuid>) -> &Self {
self.version_id = vid; self.version_id = vid.clone();
self self
} }
@@ -219,7 +219,7 @@ impl ObjSweeper {
#[allow(dead_code, reason = "MinIO-parity surface with no caller in this port (backlog#1823)")] #[allow(dead_code, reason = "MinIO-parity surface with no caller in this port (backlog#1823)")]
pub fn get_opts(&self) -> lifecycle::ObjectOpts { pub fn get_opts(&self) -> lifecycle::ObjectOpts {
let mut opts = ObjectOpts { let mut opts = ObjectOpts {
version_id: self.version_id, version_id: self.version_id.clone(),
versioned: self.versioned, versioned: self.versioned,
version_suspended: self.suspended, version_suspended: self.suspended,
..Default::default() ..Default::default()
@@ -388,8 +388,8 @@ impl Jentry {
impl ExpiryOp for Jentry { impl ExpiryOp for Jentry {
fn op_hash(&self) -> u64 { fn op_hash(&self) -> u64 {
let mut hasher = Sha256::new(); let mut hasher = Sha256::new();
hasher.update(self.tier_name.as_bytes()); hasher.update(format!("{}", self.tier_name).as_bytes());
hasher.update(self.obj_name.as_bytes()); hasher.update(format!("{}", self.obj_name).as_bytes());
xxh64::xxh64(hasher.finalize().as_slice(), XXHASH_SEED) xxh64::xxh64(hasher.finalize().as_slice(), XXHASH_SEED)
} }
@@ -436,7 +436,7 @@ async fn delete_object_from_remote_tier_raw_with_manager(
tier_name: &str, tier_name: &str,
tier_config_mgr: &Arc<tokio::sync::RwLock<TierConfigMgr>>, tier_config_mgr: &Arc<tokio::sync::RwLock<TierConfigMgr>>,
) -> Result<(), std::io::Error> { ) -> Result<(), std::io::Error> {
let lease = TierConfigMgr::acquire_operation_lease(tier_config_mgr, tier_name) let lease = TierConfigMgr::acquire_operation_lease(&tier_config_mgr, tier_name)
.await .await
.map_err(std::io::Error::other)?; .map_err(std::io::Error::other)?;
delete_object_from_remote_tier_raw_with_lease(obj_name, rv_id, &lease, false, true).await delete_object_from_remote_tier_raw_with_lease(obj_name, rv_id, &lease, false, true).await
@@ -612,7 +612,6 @@ pub(crate) async fn save_transition_transaction_record(
data.clone(), data.clone(),
&ObjectOptions { &ObjectOptions {
max_parity: true, max_parity: true,
write_completion: crate::object_api::WriteCompletion::TailDrained,
http_preconditions: Some(HTTPPreconditions { http_preconditions: Some(HTTPPreconditions {
if_none_match: Some("*".to_string()), if_none_match: Some("*".to_string()),
..Default::default() ..Default::default()
@@ -659,7 +658,6 @@ pub(crate) async fn save_transition_transaction_record_if_current(
data.clone(), data.clone(),
&ObjectOptions { &ObjectOptions {
max_parity: true, max_parity: true,
write_completion: crate::object_api::WriteCompletion::TailDrained,
http_preconditions: Some(HTTPPreconditions { http_preconditions: Some(HTTPPreconditions {
if_match: Some(etag), if_match: Some(etag),
..Default::default() ..Default::default()
+73 -188
View File
@@ -477,29 +477,28 @@ impl BucketMetadata {
!self.table_bucket_config_json.is_empty() !self.table_bucket_config_json.is_empty()
} }
/// `bucket-targets.json` is stored for this bucket but this build cannot
/// decode it.
///
/// Keeps "no replication targets configured" and "the target
/// configuration cannot be read" apart, the same distinction the
/// `fabricated` marker draws for the bucket metadata as a whole. Only
/// meaningful after [`Self::parse_all_configs`] has run; readers must fail
/// closed on `true` instead of serving an empty target set.
pub fn bucket_targets_unreadable(&self) -> bool {
!self.bucket_targets_config_json.is_empty() && self.bucket_target_config.is_none()
}
/// Opaque application-owned configuration with its persisted update time.
/// Empty bytes mean absent or cleared; decoding belongs to the consumer.
pub fn on_demand_migration_config(&self) -> Option<(&[u8], OffsetDateTime)> {
(!self.on_demand_migration_config_json.is_empty()).then_some((
self.on_demand_migration_config_json.as_slice(),
self.on_demand_migration_config_updated_at,
))
}
/// Parsed per-bucket durability override, if a valid one is stored. /// Parsed per-bucket durability override, if a valid one is stored.
/// Invalid payloads follow the global mode after logging a parse failure. ///
/// Absent/empty/unparsable payloads all mean "no override" (the bucket
/// follows the global durability mode); a parse failure is logged so a
/// corrupted entry cannot silently change fsync behavior.
/// Parsed on-demand migration config, if one is stored.
///
/// `Ok(None)` means no config (absent or cleared). A stored payload that
/// does not parse is an error, never a default: the runtime must not
/// pull from a source it cannot describe.
pub fn on_demand_migration_config(
&self,
) -> std::result::Result<
Option<super::on_demand_migration::OnDemandMigrationConfig>,
super::on_demand_migration::OnDemandMigrationConfigError,
> {
if self.on_demand_migration_config_json.is_empty() {
return Ok(None);
}
super::on_demand_migration::OnDemandMigrationConfig::from_json(&self.on_demand_migration_config_json).map(Some)
}
pub fn durability_config(&self) -> Option<super::durability::BucketDurabilityConfig> { pub fn durability_config(&self) -> Option<super::durability::BucketDurabilityConfig> {
if self.durability_config_json.is_empty() { if self.durability_config_json.is_empty() {
return None; return None;
@@ -905,6 +904,13 @@ impl BucketMetadata {
self.durability_config_updated_at = updated; self.durability_config_updated_at = updated;
} }
BUCKET_ON_DEMAND_MIGRATION_CONFIG => { BUCKET_ON_DEMAND_MIGRATION_CONFIG => {
// Structural check only (shape, unknown fields); the
// deployment-relative rules run in the admin handler with a
// `ValidationContext`. A blob this build cannot read must not
// be persisted for every later reader to trip over.
if !data.is_empty() {
super::on_demand_migration::OnDemandMigrationConfig::from_json(&data).map_err(Error::other)?;
}
self.on_demand_migration_config_json = data; self.on_demand_migration_config_json = data;
self.on_demand_migration_config_updated_at = updated; self.on_demand_migration_config_updated_at = updated;
} }
@@ -958,32 +964,7 @@ impl BucketMetadata {
Ok(()) Ok(())
} }
/// Decode every stored sub-configuration into its typed field. fn parse_all_configs(&mut self) -> Result<()> {
///
/// A decode failure never fails the whole load: this runs on every bucket
/// metadata read, including startup and peer reload, so one bucket's
/// corrupt sub-configuration must not make the bucket — or the node —
/// unloadable. Instead the failure is *retained*: the raw bytes stay
/// untouched and the typed field stays `None`, so `!raw.is_empty() &&
/// typed.is_none()` is the durable "exists but cannot be read" signal that
/// each accessor keys off. Which accessors must fail closed on it:
///
/// | Config | Verdict |
/// |---|---|
/// | policy | Fails closed: `get_bucket_policy` re-parses the raw JSON and propagates the error; `get_bucket_policy_raw` returns the stored bytes. |
/// | object lock | Fails closed in `object_lock_config_state_from_authoritative_metadata`; a retention decision may never be taken on a guess. |
/// | versioning | Fails closed in `get_versioning_config`; guessing Unversioned would make delete markers and version ids diverge from what is on disk. |
/// | replication | Fails closed in `get_replication_config`. |
/// | bucket targets | Fails closed in `get_bucket_targets_config`, and `sync_bucket_target_sys` marks the bucket unreadable in `BucketTargetSys` instead of publishing an empty target set (rustfs/backlog#2282). |
/// | encryption | Fails closed in `get_sse_config`: degrading to "no default encryption" stores plaintext objects the operator required to be encrypted. |
/// | public access block | Fails closed in `get_public_access_block_config`: degrading grants the anonymous access the operator asked to block. |
/// | quota | Fails closed in `get_quota_config`; the enforcement path in `quota::checker` already re-parses the raw JSON and refuses on error. |
/// | lifecycle | Safe to degrade: no rules means no expiration and no transition, so nothing is deleted or moved on the strength of an unreadable rule set. The bucket keeps serving reads and writes. |
/// | notification | Safe to degrade: events are an outbound side channel; no consumer draws a durability or authorization conclusion from their absence. |
/// | tagging | Safe to degrade: bucket tags are cost-allocation labels here; object-level tag conditions come from object metadata, not this blob. |
/// | CORS | Safe to degrade: an absent CORS configuration rejects cross-origin browser requests, which is already the restrictive direction. |
/// | logging, website, accelerate, request payment, bucket ACL | Safe to degrade: each only shapes an optional response or an optional side channel, and none of them authorizes an action or decides whether data is retained. |
pub(super) fn parse_all_configs(&mut self) -> Result<()> {
if let Err(e) = self.parse_policy_config() { if let Err(e) = self.parse_policy_config() {
tracing::warn!( tracing::warn!(
event = "bucket_metadata_parse_failed", event = "bucket_metadata_parse_failed",
@@ -1107,26 +1088,20 @@ impl BucketMetadata {
"Failed to parse bucket metadata config" "Failed to parse bucket metadata config"
); );
} }
// A stored targets blob that cannot be decoded must not collapse into
// the empty target set: that is indistinguishable from "no replication
// configured", so replication stops and no caller ever sees an error
// (rustfs/backlog#2282). Leaving the typed field `None` while the raw
// bytes stay non-empty is the retained parse failure every targets
// reader keys off; the bytes are preserved so the configuration is
// still recoverable.
self.bucket_target_config = None;
if !self.bucket_targets_config_json.is_empty() { if !self.bucket_targets_config_json.is_empty() {
match serde_json::from_slice::<BucketTargets>(&self.bucket_targets_config_json) { if let Err(e) = serde_json::from_slice::<BucketTargets>(&self.bucket_targets_config_json)
Ok(targets) => self.bucket_target_config = Some(targets), .map(|t| self.bucket_target_config = Some(t))
Err(e) => tracing::error!( {
tracing::warn!(
event = "bucket_metadata_parse_failed", event = "bucket_metadata_parse_failed",
component = "ecstore", component = "ecstore",
subsystem = "bucket_metadata", subsystem = "bucket_metadata",
bucket = %self.name, bucket = %self.name,
config = "bucket_targets", config = "bucket_targets",
error = %e, error = %e,
"Bucket replication targets are unreadable; replication for this bucket fails closed" "Failed to parse bucket metadata config"
), );
self.bucket_target_config = Some(BucketTargets::default());
} }
} else { } else {
self.bucket_target_config = Some(BucketTargets::default()); self.bucket_target_config = Some(BucketTargets::default());
@@ -1560,117 +1535,6 @@ mod test {
assert_eq!(bucket_targets.targets[0].target_bucket, "target-bucket"); assert_eq!(bucket_targets.targets[0].target_bucket, "target-bucket");
} }
/// rustfs/backlog#2282: a stored targets blob this build cannot decode
/// must not become the empty target set, and must stay distinguishable
/// from a bucket that never configured a target.
#[test]
fn unreadable_bucket_targets_never_degrade_to_an_empty_target_set() {
let truncated = br#"{"targets":[{"endpoint":"s3.example.com","#.to_vec();
let mut corrupt = BucketMetadata::new("corrupt-targets");
corrupt.bucket_targets_config_json = truncated.clone();
corrupt
.parse_all_configs()
.expect("one unreadable sub-config must not fail the whole metadata load");
assert!(
corrupt.bucket_target_config.is_none(),
"an undecodable targets blob must not produce a target set at all"
);
assert!(corrupt.bucket_targets_unreadable());
assert_eq!(
corrupt.bucket_targets_config_json, truncated,
"the raw bytes must survive so the configuration stays recoverable"
);
// The genuinely-absent case is unchanged, and the two now diverge.
let mut absent = BucketMetadata::new("no-targets");
absent.parse_all_configs().expect("absent targets parse");
assert!(
absent.bucket_target_config.as_ref().is_some_and(BucketTargets::is_empty),
"a bucket that configured no target still reads as an empty target set"
);
assert!(!absent.bucket_targets_unreadable());
}
/// `Credentials` carries no struct-level `serde(default)`, so one target
/// missing `secretKey` is a hard parse error for the whole document. That
/// must surface as "unreadable", never as "no targets configured".
#[test]
fn bucket_targets_missing_secret_key_are_unreadable_not_empty() {
let mut bm = BucketMetadata::new("missing-secret-key");
bm.bucket_targets_config_json = br#"{"targets":[{"endpoint":"s3.example.com","targetbucket":"remote","arn":"arn:rustfs:replication:us-east-1:src:1","credentials":{"accessKey":"AKIAEXAMPLE"}}]}"#.to_vec();
bm.parse_all_configs()
.expect("a rejected targets document must not fail the whole metadata load");
assert!(
bm.bucket_targets_unreadable(),
"a targets document rejected for a missing secretKey is unreadable, not empty"
);
assert!(bm.bucket_target_config.is_none());
}
/// The invariant every branch of `parse_all_configs` shares: a stored but
/// undecodable payload keeps its raw bytes and leaves the typed field
/// `None`, so no branch fabricates a value. What a reader may then do with
/// that state is decided per config; see the table on `parse_all_configs`.
#[test]
fn every_config_branch_retains_its_parse_failure_instead_of_defaulting() {
let malformed_xml = b"<not-a-valid-document".to_vec();
let malformed_json = b"{not-json".to_vec();
let mut bm = BucketMetadata::new("all-configs-malformed");
bm.policy_config_json = malformed_json.clone();
bm.quota_config_json = malformed_json.clone();
bm.bucket_targets_config_json = malformed_json.clone();
bm.notification_config_xml = malformed_xml.clone();
bm.lifecycle_config_xml = malformed_xml.clone();
bm.object_lock_config_xml = malformed_xml.clone();
bm.versioning_config_xml = malformed_xml.clone();
bm.encryption_config_xml = malformed_xml.clone();
bm.tagging_config_xml = malformed_xml.clone();
bm.replication_config_xml = malformed_xml.clone();
bm.cors_config_xml = malformed_xml.clone();
bm.logging_config_xml = malformed_xml.clone();
bm.website_config_xml = malformed_xml.clone();
bm.accelerate_config_xml = malformed_xml.clone();
bm.request_payment_config_xml = malformed_xml.clone();
bm.public_access_block_config_xml = malformed_xml.clone();
// `bucket_acl_config_json` is only checked for UTF-8, so only invalid
// UTF-8 exercises its failure branch.
bm.bucket_acl_config_json = vec![0xff, 0xfe];
bm.parse_all_configs()
.expect("a bucket whose every config is corrupt must still load its metadata");
let cleared: [(&str, bool); 17] = [
("policy", bm.policy_config.is_none()),
("quota", bm.quota_config.is_none()),
("bucket_targets", bm.bucket_target_config.is_none()),
("notification", bm.notification_config.is_none()),
("lifecycle", bm.lifecycle_config.is_none()),
("object_lock", bm.object_lock_config.is_none()),
("versioning", bm.versioning_config.is_none()),
("encryption", bm.sse_config.is_none()),
("tagging", bm.tagging_config.is_none()),
("replication", bm.replication_config.is_none()),
("cors", bm.cors_config.is_none()),
("logging", bm.logging_config.is_none()),
("website", bm.website_config.is_none()),
("accelerate", bm.accelerate_config.is_none()),
("request_payment", bm.request_payment_config.is_none()),
("public_access_block", bm.public_access_block_config.is_none()),
("bucket_acl", bm.bucket_acl_config.is_none()),
];
for (config, is_cleared) in cleared {
assert!(is_cleared, "{config}: a corrupt payload must not be replaced by a default");
}
assert_eq!(bm.bucket_targets_config_json, malformed_json, "raw bytes are retained");
assert_eq!(bm.lifecycle_config_xml, malformed_xml, "raw bytes are retained");
}
#[test] #[test]
fn lifecycle_update_config_clears_parsed_config_on_delete() { fn lifecycle_update_config_clears_parsed_config_on_delete() {
let mut bm = BucketMetadata::new("test-bucket"); let mut bm = BucketMetadata::new("test-bucket");
@@ -1960,30 +1824,51 @@ mod test {
const ODM_JSON: &[u8] = br#"{"version":1,"enabled":true,"source":{"provider":"minio","endpoint":"https://legacy.example.com:9000","region":"auto","bucket":"legacy-bucket","credentials":{"access_key":"AK","secret_key":"SK"}}}"#; const ODM_JSON: &[u8] = br#"{"version":1,"enabled":true,"source":{"provider":"minio","endpoint":"https://legacy.example.com:9000","region":"auto","bucket":"legacy-bucket","credentials":{"access_key":"AK","secret_key":"SK"}}}"#;
/// The metadata codec preserves application-owned bytes and timestamps. /// rustfs/backlog#2148: the on-demand migration config is a RustFS
/// extension entry that round-trips through `update_config` and the
/// msgpack codec, clears on delete, and never parses corruption into a
/// default.
#[test] #[test]
fn on_demand_migration_config_round_trips_and_tracks_updates() { fn on_demand_migration_config_round_trips_and_tracks_updates() {
use crate::bucket::on_demand_migration::{OnDemandMigrationConfig, OnDemandMigrationConfigError};
let mut bm = BucketMetadata::new("odm-bucket"); let mut bm = BucketMetadata::new("odm-bucket");
assert_eq!(bm.on_demand_migration_config(), None, "fresh metadata carries no config"); assert_eq!(bm.on_demand_migration_config(), Ok(None), "fresh metadata carries no config");
let expected = OnDemandMigrationConfig::from_json(ODM_JSON).unwrap();
bm.update_config(BUCKET_ON_DEMAND_MIGRATION_CONFIG, ODM_JSON.to_vec()) bm.update_config(BUCKET_ON_DEMAND_MIGRATION_CONFIG, ODM_JSON.to_vec())
.expect("opaque config is accepted"); .expect("valid config is accepted");
let stamped = bm.on_demand_migration_config_updated_at; assert_ne!(bm.on_demand_migration_config_updated_at, OffsetDateTime::UNIX_EPOCH);
assert_ne!(stamped, OffsetDateTime::UNIX_EPOCH); assert_eq!(bm.on_demand_migration_config(), Ok(Some(expected.clone())));
assert_eq!(bm.on_demand_migration_config(), Some((ODM_JSON, stamped)));
let back = BucketMetadata::unmarshal(&bm.marshal_msg().unwrap()).unwrap(); let back = BucketMetadata::unmarshal(&bm.marshal_msg().unwrap()).unwrap();
assert_eq!(back.on_demand_migration_config_json, bm.on_demand_migration_config_json); assert_eq!(back.on_demand_migration_config_json, bm.on_demand_migration_config_json);
assert_eq!(back.on_demand_migration_config_updated_at.unix_timestamp(), stamped.unix_timestamp()); assert_eq!(
back.on_demand_migration_config_updated_at.unix_timestamp(),
bm.on_demand_migration_config_updated_at.unix_timestamp()
);
assert_eq!(back.on_demand_migration_config(), Ok(Some(expected)));
// A blob this build cannot read is rejected at the write boundary
// rather than persisted for every reader to trip over.
let before = bm.on_demand_migration_config_json.clone();
assert!(
bm.update_config(BUCKET_ON_DEMAND_MIGRATION_CONFIG, br#"{"source":{"provider":"s3"},"bogus":1}"#.to_vec())
.is_err()
);
assert_eq!(bm.on_demand_migration_config_json, before, "a rejected update leaves the blob untouched");
// Delete clears the entry.
let stamped = bm.on_demand_migration_config_updated_at;
bm.update_config(BUCKET_ON_DEMAND_MIGRATION_CONFIG, Vec::new()).unwrap(); bm.update_config(BUCKET_ON_DEMAND_MIGRATION_CONFIG, Vec::new()).unwrap();
assert!(bm.on_demand_migration_config_json.is_empty()); assert!(bm.on_demand_migration_config_json.is_empty());
assert_eq!(bm.on_demand_migration_config(), None); assert_eq!(bm.on_demand_migration_config(), Ok(None));
assert!(bm.on_demand_migration_config_updated_at >= stamped); assert!(bm.on_demand_migration_config_updated_at >= stamped);
bm.update_config(BUCKET_ON_DEMAND_MIGRATION_CONFIG, b"not-json".to_vec())
.unwrap(); // Corruption that bypassed `update_config` (disk, another writer)
let back = BucketMetadata::unmarshal(&bm.marshal_msg().unwrap()).unwrap(); // is a typed error, never a default.
assert_eq!( bm.on_demand_migration_config_json = b"not-json".to_vec();
back.on_demand_migration_config_json, b"not-json", assert!(matches!(bm.on_demand_migration_config(), Err(OnDemandMigrationConfigError::Malformed(_))));
"metadata must not reinterpret application bytes"
);
} }
/// rustfs/backlog#2148: a `.metadata.bin` written before the on-demand /// rustfs/backlog#2148: a `.metadata.bin` written before the on-demand
@@ -1995,7 +1880,7 @@ mod test {
let mut bm = BucketMetadata::unmarshal(&blob[4..]).expect("unmarshal MinIO bucket metadata"); let mut bm = BucketMetadata::unmarshal(&blob[4..]).expect("unmarshal MinIO bucket metadata");
assert!(bm.on_demand_migration_config_json.is_empty()); assert!(bm.on_demand_migration_config_json.is_empty());
assert_eq!(bm.on_demand_migration_config_updated_at, OffsetDateTime::UNIX_EPOCH); assert_eq!(bm.on_demand_migration_config_updated_at, OffsetDateTime::UNIX_EPOCH);
assert_eq!(bm.on_demand_migration_config(), None); assert_eq!(bm.on_demand_migration_config(), Ok(None));
bm.default_timestamps(); bm.default_timestamps();
assert_ne!(bm.created, OffsetDateTime::UNIX_EPOCH, "fixture must carry a real creation time"); assert_ne!(bm.created, OffsetDateTime::UNIX_EPOCH, "fixture must carry a real creation time");
+109 -333
View File
@@ -19,6 +19,7 @@ use super::quota::BucketQuota;
use super::target::BucketTargets; use super::target::BucketTargets;
use crate::bucket::bucket_target_sys::BucketTargetSys; use crate::bucket::bucket_target_sys::BucketTargetSys;
use crate::bucket::metadata::{load_bucket_metadata_parse, load_bucket_metadata_parse_with_presence}; use crate::bucket::metadata::{load_bucket_metadata_parse, load_bucket_metadata_parse_with_presence};
use crate::bucket::on_demand_migration::{ON_DEMAND_MIGRATION_CONFIG_HOOK, OnDemandMigrationConfig};
use crate::bucket::utils::is_meta_bucketname; use crate::bucket::utils::is_meta_bucketname;
use crate::disk::RUSTFS_META_BUCKET; use crate::disk::RUSTFS_META_BUCKET;
use crate::error::{Error, Result, is_err_bucket_not_found, is_err_strict_volume_not_found}; use crate::error::{Error, Result, is_err_bucket_not_found, is_err_strict_volume_not_found};
@@ -48,11 +49,6 @@ use tokio_util::sync::CancellationToken;
use tracing::{error, warn}; use tracing::{error, warn};
use uuid::Uuid; use uuid::Uuid;
/// Opaque bucket configuration notifications for application-owned services.
/// `None` withdraws a configuration; consumers validate nonempty bytes.
pub type BucketConfigPublishHook = Box<dyn Fn(&str, &str, Option<(&[u8], OffsetDateTime, Uuid)>) + Send + Sync>;
pub static BUCKET_CONFIG_PUBLISH_HOOK: std::sync::OnceLock<BucketConfigPublishHook> = std::sync::OnceLock::new();
const BUCKET_METADATA_REFRESH_INTERVAL: Duration = Duration::from_secs(15 * 60); const BUCKET_METADATA_REFRESH_INTERVAL: Duration = Duration::from_secs(15 * 60);
#[cfg(any(test, feature = "test-util"))] #[cfg(any(test, feature = "test-util"))]
@@ -364,16 +360,6 @@ async fn refresh_buckets_metadata_once(sys: Arc<RwLock<BucketMetadataSys>>) {
} }
async fn sync_bucket_target_sys(bucket: &str, bm: &BucketMetadata) { async fn sync_bucket_target_sys(bucket: &str, bm: &BucketMetadata) {
if bm.bucket_targets_unreadable() {
// "The configuration cannot be read" is not "no targets configured".
// Publishing an empty snapshot here is what silently stopped
// replication (rustfs/backlog#2282): mark the bucket instead, so every
// targets reader gets a typed error, and leave any snapshot from an
// earlier readable load in place rather than withdrawing it.
BucketTargetSys::get().mark_targets_unreadable(bucket).await;
return;
}
BucketTargetSys::get() BucketTargetSys::get()
.update_all_targets(bucket, bm.bucket_target_config.as_ref()) .update_all_targets(bucket, bm.bucket_target_config.as_ref())
.await; .await;
@@ -399,21 +385,39 @@ fn clear_bucket_durability(bucket: &str) {
crate::disk::local::bucket_durability::set(bucket, None); crate::disk::local::bucket_durability::set(bucket, None);
} }
/// Publish application-owned bytes on every cache install path. /// Publish the bucket's on-demand migration config (or its absence) to the
/// runtime registered in `ON_DEMAND_MIGRATION_CONFIG_HOOK`.
///
/// Called from the same five cache-install paths as
/// [`sync_bucket_durability`]. A stored payload this build cannot parse is
/// published as `None`: the runtime must stop pulling for that bucket rather
/// than keep an older config or guess.
fn sync_on_demand_migration(bucket: &str, bm: &BucketMetadata) { fn sync_on_demand_migration(bucket: &str, bm: &BucketMetadata) {
if let Some(hook) = BUCKET_CONFIG_PUBLISH_HOOK.get() { let Some(hook) = ON_DEMAND_MIGRATION_CONFIG_HOOK.get() else {
hook( return;
bucket, };
super::metadata::BUCKET_ON_DEMAND_MIGRATION_CONFIG, match bm.on_demand_migration_config() {
bm.on_demand_migration_config() Ok(config) => hook(bucket, config.as_ref()),
.map(|(bytes, stamp)| (bytes, stamp, bm.bucket_incarnation_id)), Err(err) => {
); warn!(
event = "bucket_metadata_parse_failed",
component = "ecstore",
subsystem = "bucket_metadata",
bucket = %bucket,
config = "on_demand_migration",
error = %err,
"Failed to parse bucket metadata config"
);
hook(bucket, None);
}
} }
} }
/// Withdraw a bucket's on-demand migration config when its metadata leaves
/// the cache.
fn clear_on_demand_migration(bucket: &str) { fn clear_on_demand_migration(bucket: &str) {
if let Some(hook) = BUCKET_CONFIG_PUBLISH_HOOK.get() { if let Some(hook) = ON_DEMAND_MIGRATION_CONFIG_HOOK.get() {
hook(bucket, super::metadata::BUCKET_ON_DEMAND_MIGRATION_CONFIG, None); hook(bucket, None);
} }
} }
@@ -641,12 +645,6 @@ pub struct BucketMetadataMutationGuard {
} }
impl BucketMetadataMutationGuard { impl BucketMetadataMutationGuard {
/// Returns the storage-verified identity while both incarnation fences remain valid.
pub fn checked_bucket_incarnation(&self) -> Result<(&str, Uuid)> {
self.ensure_valid(&self.bucket)?;
Ok((&self.bucket, self.incarnation_id))
}
fn ensure_valid(&self, bucket: &str) -> Result<()> { fn ensure_valid(&self, bucket: &str) -> Result<()> {
if self.bucket != bucket { if self.bucket != bucket {
return Err(Error::other("bucket metadata mutation guard does not match bucket")); return Err(Error::other("bucket metadata mutation guard does not match bucket"));
@@ -666,29 +664,6 @@ async fn acquire_config_write_guard_for_incarnation(
sys: Arc<RwLock<BucketMetadataSys>>, sys: Arc<RwLock<BucketMetadataSys>>,
bucket: &str, bucket: &str,
expected_incarnation_id: Option<Uuid>, expected_incarnation_id: Option<Uuid>,
) -> Result<BucketMetadataMutationGuard> {
acquire_config_write_guard_with_migration(sys, bucket, expected_incarnation_id, true).await
}
/// Scanner probes must not create an incarnation to make a capability available.
pub async fn acquire_scanner_bucket_incarnation_fence(
bucket: &str,
expected_incarnation_id: Uuid,
expected_owner_id: Uuid,
) -> Result<BucketMetadataMutationGuard> {
super::utils::check_valid_bucket_name(bucket)?;
let sys = get_bucket_metadata_sys()?;
if expected_owner_id.is_nil() || sys.read().await.api.id != expected_owner_id || expected_incarnation_id.is_nil() {
return Err(Error::other("scanner bucket incarnation owner does not match"));
}
acquire_config_write_guard_with_migration(sys, bucket, Some(expected_incarnation_id), false).await
}
async fn acquire_config_write_guard_with_migration(
sys: Arc<RwLock<BucketMetadataSys>>,
bucket: &str,
expected_incarnation_id: Option<Uuid>,
migrate: bool,
) -> Result<BucketMetadataMutationGuard> { ) -> Result<BucketMetadataMutationGuard> {
let metadata_sys = sys.read().await.clone(); let metadata_sys = sys.read().await.clone();
let lifecycle_guard = metadata_sys.api.acquire_bucket_lifecycle_read_lock(bucket).await?; let lifecycle_guard = metadata_sys.api.acquire_bucket_lifecycle_read_lock(bucket).await?;
@@ -696,15 +671,13 @@ async fn acquire_config_write_guard_with_migration(
// Legacy buckets are migrated while the lifecycle fence prevents a // Legacy buckets are migrated while the lifecycle fence prevents a
// same-name replacement. The second read under the write transaction is // same-name replacement. The second read under the write transaction is
// the CAS source of truth for the actual rewrite. // the CAS source of truth for the actual rewrite.
if migrate { await_bucket_namespace_operation(
await_bucket_namespace_operation( Some(&lifecycle_guard),
Some(&lifecycle_guard), bucket,
bucket, "bucket config incarnation migration",
"bucket config incarnation migration", metadata_sys.get_bucket_incarnation_id(bucket),
metadata_sys.get_bucket_incarnation_id(bucket), )
) .await?;
.await?;
}
let transaction_guard = await_bucket_namespace_operation( let transaction_guard = await_bucket_namespace_operation(
Some(&lifecycle_guard), Some(&lifecycle_guard),
bucket, bucket,
@@ -1035,21 +1008,15 @@ pub async fn get_durability_config(
} }
/// The bucket's on-demand migration config with its update time, or /// The bucket's on-demand migration config with its update time, or
/// `Ok(None)` when the bucket has none. Bytes are opaque to the metadata owner. /// `Ok(None)` when the bucket has none. A stored payload that does not parse
pub async fn get_on_demand_migration_config(bucket: &str) -> Result<Option<(Vec<u8>, OffsetDateTime)>> { /// is a typed error (`OnDemandMigrationConfigError` inside `Error::Io`).
pub async fn get_on_demand_migration_config(bucket: &str) -> Result<Option<(OnDemandMigrationConfig, OffsetDateTime)>> {
let bucket_meta_sys_lock = get_bucket_metadata_sys()?; let bucket_meta_sys_lock = get_bucket_metadata_sys()?;
let bucket_meta_sys = bucket_meta_sys_lock.read().await; let bucket_meta_sys = bucket_meta_sys_lock.read().await;
bucket_meta_sys.get_on_demand_migration_config(bucket).await bucket_meta_sys.get_on_demand_migration_config(bucket).await
} }
/// Resolve opaque configuration from the store's own metadata system.
pub async fn get_on_demand_migration_config_in(api: &ECStore, bucket: &str) -> Result<Option<(Vec<u8>, OffsetDateTime)>> {
let sys = bucket_metadata_sys_of(&api.ctx)?;
let lock = sys.read().await;
lock.get_on_demand_migration_config(bucket).await
}
pub async fn get_quota_config(bucket: &str) -> Result<(BucketQuota, OffsetDateTime)> { pub async fn get_quota_config(bucket: &str) -> Result<(BucketQuota, OffsetDateTime)> {
let bucket_meta_sys_lock = get_bucket_metadata_sys()?; let bucket_meta_sys_lock = get_bucket_metadata_sys()?;
let bucket_meta_sys = bucket_meta_sys_lock.read().await; let bucket_meta_sys = bucket_meta_sys_lock.read().await;
@@ -2151,9 +2118,7 @@ impl BucketMetadataSys {
pub async fn get_public_access_block_config(&self, bucket: &str) -> Result<(PublicAccessBlockConfiguration, OffsetDateTime)> { pub async fn get_public_access_block_config(&self, bucket: &str) -> Result<(PublicAccessBlockConfiguration, OffsetDateTime)> {
let (bm, _) = self.get_config(bucket).await?; let (bm, _) = self.get_config(bucket).await?;
if !bm.public_access_block_config_xml.is_empty() && bm.public_access_block_config.is_none() { if let Some(config) = &bm.public_access_block_config {
Err(Error::other("persisted bucket public access block configuration is invalid"))
} else if let Some(config) = &bm.public_access_block_config {
Ok((config.clone(), bm.public_access_block_config_updated_at)) Ok((config.clone(), bm.public_access_block_config_updated_at))
} else { } else {
Err(Error::ConfigNotFound) Err(Error::ConfigNotFound)
@@ -2464,9 +2429,7 @@ impl BucketMetadataSys {
pub async fn get_sse_config(&self, bucket: &str) -> Result<(ServerSideEncryptionConfiguration, OffsetDateTime)> { pub async fn get_sse_config(&self, bucket: &str) -> Result<(ServerSideEncryptionConfiguration, OffsetDateTime)> {
let (bm, _) = self.get_config(bucket).await?; let (bm, _) = self.get_config(bucket).await?;
if !bm.encryption_config_xml.is_empty() && bm.sse_config.is_none() { if let Some(config) = &bm.sse_config {
Err(Error::other("persisted bucket encryption configuration is invalid"))
} else if let Some(config) = &bm.sse_config {
Ok((config.clone(), bm.encryption_config_updated_at)) Ok((config.clone(), bm.encryption_config_updated_at))
} else { } else {
Err(Error::ConfigNotFound) Err(Error::ConfigNotFound)
@@ -2537,9 +2500,7 @@ impl BucketMetadataSys {
pub async fn get_quota_config(&self, bucket: &str) -> Result<(BucketQuota, OffsetDateTime)> { pub async fn get_quota_config(&self, bucket: &str) -> Result<(BucketQuota, OffsetDateTime)> {
let (bm, _) = self.get_config(bucket).await?; let (bm, _) = self.get_config(bucket).await?;
if !bm.quota_config_json.is_empty() && bm.quota_config.is_none() { if let Some(config) = &bm.quota_config {
Err(Error::other("persisted bucket quota configuration is invalid"))
} else if let Some(config) = &bm.quota_config {
Ok((config.clone(), bm.quota_config_updated_at)) Ok((config.clone(), bm.quota_config_updated_at))
} else { } else {
Err(Error::ConfigNotFound) Err(Error::ConfigNotFound)
@@ -2561,9 +2522,7 @@ impl BucketMetadataSys {
pub async fn get_bucket_targets_config(&self, bucket: &str) -> Result<BucketTargets> { pub async fn get_bucket_targets_config(&self, bucket: &str) -> Result<BucketTargets> {
let (bm, _) = self.get_config(bucket).await?; let (bm, _) = self.get_config(bucket).await?;
if bm.bucket_targets_unreadable() { if let Some(config) = &bm.bucket_target_config {
Err(Error::other("persisted bucket replication target configuration is invalid"))
} else if let Some(config) = &bm.bucket_target_config {
Ok(config.clone()) Ok(config.clone())
} else { } else {
Err(Error::ConfigNotFound) Err(Error::ConfigNotFound)
@@ -2571,27 +2530,29 @@ impl BucketMetadataSys {
} }
/// See [`get_on_demand_migration_config`]. /// See [`get_on_demand_migration_config`].
pub async fn get_on_demand_migration_config(&self, bucket: &str) -> Result<Option<(Vec<u8>, OffsetDateTime)>> { pub async fn get_on_demand_migration_config(
&self,
bucket: &str,
) -> Result<Option<(OnDemandMigrationConfig, OffsetDateTime)>> {
let (bm, _) = self.get_config(bucket).await?; let (bm, _) = self.get_config(bucket).await?;
Ok(bm let config = bm.on_demand_migration_config().map_err(Error::other)?;
.on_demand_migration_config() Ok(config.map(|config| (config, bm.on_demand_migration_config_updated_at)))
.map(|(bytes, updated_at)| (bytes.to_vec(), updated_at)))
} }
} }
/// Test-only fixture shared with sibling modules (e.g. the quota checker /// Test-only fixture shared with sibling modules (e.g. the quota checker
/// tests): a 4-disk `ECStore` on an isolated instance context, so tests /// tests): a 4-disk `ECStore` on an isolated instance context, so tests
/// exercising the metadata system never touch ambient process state. /// exercising the metadata system never touch ambient process state.
#[cfg(any(test, feature = "test-util"))] #[cfg(test)]
pub mod test_support { pub(crate) mod test_support {
use super::*; use super::*;
use crate::disk::endpoint::Endpoint; use crate::disk::endpoint::Endpoint;
use crate::layout::endpoints::{EndpointServerPools, Endpoints, PoolEndpoints}; use crate::layout::endpoints::{EndpointServerPools, Endpoints, PoolEndpoints};
use crate::runtime::instance::InstanceContext; use crate::runtime::instance::InstanceContext;
use crate::store::init_local_disks_with_instance_ctx; use crate::store::init_local_disks_with_instance_ctx;
pub async fn isolated_store_over_temp_disks() -> (Vec<tempfile::TempDir>, Arc<ECStore>) { pub(crate) async fn isolated_store_over_temp_disks() -> (Vec<tempfile::TempDir>, Arc<ECStore>) {
let mut dirs = Vec::with_capacity(4); let mut dirs = Vec::with_capacity(4);
let mut endpoints = Vec::with_capacity(4); let mut endpoints = Vec::with_capacity(4);
for disk_idx in 0..4 { for disk_idx in 0..4 {
@@ -2632,7 +2593,6 @@ pub mod test_support {
mod tests { mod tests {
use super::test_support::isolated_store_over_temp_disks; use super::test_support::isolated_store_over_temp_disks;
use super::*; use super::*;
use crate::bucket::bucket_target_sys::BucketTargetError;
use crate::bucket::metadata::{ use crate::bucket::metadata::{
BUCKET_ACCELERATE_CONFIG, BUCKET_CORS_CONFIG, BUCKET_LIFECYCLE_CONFIG, BUCKET_LOGGING_CONFIG, BUCKET_NOTIFICATION_CONFIG, BUCKET_ACCELERATE_CONFIG, BUCKET_CORS_CONFIG, BUCKET_LIFECYCLE_CONFIG, BUCKET_LOGGING_CONFIG, BUCKET_NOTIFICATION_CONFIG,
BUCKET_POLICY_CONFIG, BUCKET_PUBLIC_ACCESS_BLOCK_CONFIG, BUCKET_REPLICATION_CONFIG, BUCKET_REQUEST_PAYMENT_CONFIG, BUCKET_POLICY_CONFIG, BUCKET_PUBLIC_ACCESS_BLOCK_CONFIG, BUCKET_REPLICATION_CONFIG, BUCKET_REQUEST_PAYMENT_CONFIG,
@@ -2828,36 +2788,6 @@ mod tests {
); );
} }
/// The `parse_all_configs` audit (rustfs/backlog#2282): every accessor
/// whose configuration grants something — plaintext storage, anonymous
/// access, capacity, replication targets — reports a corrupt payload as
/// invalid rather than as absent, because "absent" is what grants it.
#[tokio::test]
async fn malformed_permissive_configs_are_not_reported_as_absent() {
let (_dirs, ecstore) = isolated_store_over_temp_disks().await;
let sys = BucketMetadataSys::new(ecstore);
let bucket = "malformed-permissive-config";
let mut metadata = BucketMetadata::new(bucket);
metadata.encryption_config_xml = b"<ServerSideEncryptionConfiguration".to_vec();
metadata.public_access_block_config_xml = b"<PublicAccessBlockConfiguration".to_vec();
metadata.quota_config_json = b"{not-json".to_vec();
metadata.bucket_targets_config_json = b"{not-json".to_vec();
metadata
.parse_all_configs()
.expect("a corrupt sub-config must not fail the load");
sys.set(bucket.to_string(), Arc::new(metadata)).await;
for (config, result) in [
("encryption", sys.get_sse_config(bucket).await.err()),
("public access block", sys.get_public_access_block_config(bucket).await.err()),
("quota", sys.get_quota_config(bucket).await.err()),
("bucket targets", sys.get_bucket_targets_config(bucket).await.err()),
] {
let err = result.unwrap_or_else(|| panic!("malformed {config} metadata must not read as a value"));
assert_ne!(err, Error::ConfigNotFound, "malformed {config} metadata must not be reported as absent");
}
}
#[tokio::test] #[tokio::test]
async fn config_states_distinguish_authoritative_absence_from_fabricated_metadata() { async fn config_states_distinguish_authoritative_absence_from_fabricated_metadata() {
use std::sync::atomic::Ordering; use std::sync::atomic::Ordering;
@@ -3197,82 +3127,6 @@ mod tests {
); );
} }
#[tokio::test]
async fn scoped_dirty_usage_incarnation_probe_does_not_migrate_legacy_metadata() {
let (dirs, store) = isolated_store_over_temp_disks().await;
let sys = Arc::new(RwLock::new(BucketMetadataSys::new(store.clone())));
let bucket = "scoped-ack-legacy";
for dir in &dirs {
std::fs::create_dir_all(dir.path().join(bucket)).expect("create legacy bucket");
}
let mut metadata = BucketMetadata::new(bucket);
metadata.bucket_incarnation_id = Uuid::nil();
sys.read()
.await
.persist_and_set(metadata)
.await
.expect("persist legacy metadata");
assert!(
acquire_config_write_guard_with_migration(sys.clone(), bucket, Some(Uuid::new_v4()), false)
.await
.is_err()
);
assert!(load_bucket_incarnation(store, bucket).await.expect("read sidecar").is_none());
assert!(
sys.read()
.await
.get_config_from_disk(bucket)
.await
.expect("read metadata")
.bucket_incarnation_id
.is_nil()
);
}
#[tokio::test(flavor = "multi_thread", worker_threads = 4)]
#[serial]
async fn scoped_dirty_usage_incarnation_rejects_deleted_and_recreated_bucket() {
let (_dirs, store) = isolated_store_over_temp_disks().await;
init_bucket_metadata_sys(store.clone(), Vec::new()).await;
let sys = bucket_metadata_sys_of(&store.ctx).expect("metadata owner");
let bucket = "scoped-ack-recreated";
store
.make_bucket(bucket, &MakeBucketOptions::default())
.await
.expect("create bucket");
let old = store.bucket_incarnation_id_from_disk(bucket).await.expect("old incarnation");
let guard = acquire_config_write_guard_with_migration(sys.clone(), bucket, Some(old), false)
.await
.expect("trusted incarnation fence");
assert_eq!(guard.checked_bucket_incarnation().expect("valid fences"), (bucket, old));
drop(guard);
store
.delete_bucket(bucket, &DeleteBucketOptions::default())
.await
.expect("delete bucket");
assert!(
acquire_config_write_guard_with_migration(sys.clone(), bucket, Some(old), false)
.await
.is_err()
);
store
.make_bucket(bucket, &MakeBucketOptions::default())
.await
.expect("recreate bucket");
let new = store.bucket_incarnation_id_from_disk(bucket).await.expect("new incarnation");
assert_ne!(old, new);
assert!(
acquire_config_write_guard_with_migration(sys.clone(), bucket, Some(old), false)
.await
.is_err()
);
assert!(
acquire_config_write_guard_with_migration(sys, bucket, Some(new), false)
.await
.is_ok()
);
}
#[tokio::test] #[tokio::test]
async fn old_node_metadata_rewrite_cannot_replace_bucket_incarnation_sidecar() { async fn old_node_metadata_rewrite_cannot_replace_bucket_incarnation_sidecar() {
let (dirs, ecstore) = isolated_store_over_temp_disks().await; let (dirs, ecstore) = isolated_store_over_temp_disks().await;
@@ -4212,114 +4066,6 @@ mod tests {
target_sys.delete(bucket).await; target_sys.delete(bucket).await;
} }
/// rustfs/backlog#2282: an unreadable `bucket-targets.json` reaches every
/// targets reader as a typed error; it neither withdraws a snapshot a
/// previous readable load published, nor collapses into the "no targets
/// configured" state that a bucket with an absent configuration reports.
#[tokio::test]
#[serial]
async fn unreadable_bucket_targets_fail_closed_and_stay_distinct_from_absent() {
let (_dirs, ecstore) = isolated_store_over_temp_disks().await;
let sys = BucketMetadataSys::new(ecstore);
let target_sys = BucketTargetSys::get();
let unreadable = "targets-unreadable";
let absent = "targets-absent";
target_sys.delete(unreadable).await;
target_sys.delete(absent).await;
// A readable load publishes this bucket's targets.
let mut readable = BucketMetadata::new(unreadable);
readable.bucket_target_config = Some(BucketTargets {
targets: vec![target(unreadable, "live")],
});
sync_bucket_target_sys(unreadable, &readable).await;
assert_eq!(
target_sys
.list_bucket_targets(unreadable)
.await
.expect("readable targets publish")
.targets
.len(),
1
);
// The same bucket reloaded with a blob that cannot be decoded.
let mut corrupt = BucketMetadata::new(unreadable);
corrupt.bucket_targets_config_json = br#"{"targets":[{"endpoint":"#.to_vec();
corrupt
.parse_all_configs()
.expect("an unreadable targets blob must not fail the metadata load");
sys.set(unreadable.to_string(), Arc::new(corrupt)).await;
assert!(
matches!(
target_sys.list_bucket_targets(unreadable).await,
Err(BucketTargetError::BucketRemoteTargetsUnreadable { .. })
),
"an unreadable configuration must not read as an empty or a missing target set"
);
assert!(
target_sys.list_targets(unreadable, "").await.is_err(),
"the admin listing must surface the fault instead of an empty list"
);
let err = sys
.get_bucket_targets_config(unreadable)
.await
.expect_err("an unreadable targets configuration must not read as a value");
assert_ne!(err, Error::ConfigNotFound, "unreadable must not be reported as absent");
// A bucket that never configured a target keeps its previous behavior.
let mut no_targets = BucketMetadata::new(absent);
no_targets.parse_all_configs().expect("absent targets parse");
sys.set(absent.to_string(), Arc::new(no_targets)).await;
assert!(
matches!(
target_sys.list_bucket_targets(absent).await,
Err(BucketTargetError::BucketRemoteTargetNotFound { .. })
),
"an absent configuration must still report as a missing target set"
);
assert!(
target_sys
.list_targets(absent, "")
.await
.expect("an absent configuration lists no targets")
.is_empty()
);
assert!(
sys.get_bucket_targets_config(absent)
.await
.expect("an absent targets configuration still reads as an empty set")
.is_empty(),
"the absent path must keep returning an empty target set, exactly as before"
);
// One bucket's unreadable configuration does not reach another bucket.
assert!(!matches!(
target_sys.list_bucket_targets(absent).await,
Err(BucketTargetError::BucketRemoteTargetsUnreadable { .. })
));
// A repaired configuration takes effect on the next load, no restart.
let mut repaired = BucketMetadata::new(unreadable);
repaired.bucket_target_config = Some(BucketTargets {
targets: vec![target(unreadable, "repaired")],
});
sync_bucket_target_sys(unreadable, &repaired).await;
assert_eq!(
target_sys
.list_bucket_targets(unreadable)
.await
.expect("a repaired configuration clears the unreadable marker")
.targets
.len(),
1
);
target_sys.delete(unreadable).await;
target_sys.delete(absent).await;
}
#[tokio::test] #[tokio::test]
#[serial] #[serial]
async fn metadata_reload_clears_stale_bucket_targets_when_config_is_removed() { async fn metadata_reload_clears_stale_bucket_targets_when_config_is_removed() {
@@ -4375,26 +4121,19 @@ mod tests {
const ODM_JSON: &[u8] = br#"{"source":{"provider":"minio","endpoint":"https://legacy.example.com:9000","region":"auto","bucket":"legacy-bucket","credentials":{"access_key":"AK","secret_key":"SK"}}}"#; const ODM_JSON: &[u8] = br#"{"source":{"provider":"minio","endpoint":"https://legacy.example.com:9000","region":"auto","bucket":"legacy-bucket","credentials":{"access_key":"AK","secret_key":"SK"}}}"#;
type RecordedOdmConfig = Option<(Vec<u8>, OffsetDateTime, Uuid)>;
type RecordedOdmHookCall = (String, RecordedOdmConfig);
/// Every `(bucket, config)` the recording hook has seen. Tests filter by /// Every `(bucket, config)` the recording hook has seen. Tests filter by
/// their own bucket name; the hook is process-wide and set once. /// their own bucket name; the hook is process-wide and set once.
static ODM_HOOK_CALLS: std::sync::Mutex<Vec<RecordedOdmHookCall>> = std::sync::Mutex::new(Vec::new()); static ODM_HOOK_CALLS: std::sync::Mutex<Vec<(String, Option<OnDemandMigrationConfig>)>> = std::sync::Mutex::new(Vec::new());
fn install_recording_odm_hook() { fn install_recording_odm_hook() {
BUCKET_CONFIG_PUBLISH_HOOK.get_or_init(|| { ON_DEMAND_MIGRATION_CONFIG_HOOK.get_or_init(|| {
Box::new(|bucket, config_file, config| { Box::new(|bucket, config| {
assert_eq!(config_file, super::super::metadata::BUCKET_ON_DEMAND_MIGRATION_CONFIG); ODM_HOOK_CALLS.lock().unwrap().push((bucket.to_string(), config.cloned()));
ODM_HOOK_CALLS.lock().unwrap().push((
bucket.to_string(),
config.map(|(bytes, stamp, incarnation)| (bytes.to_vec(), stamp, incarnation)),
));
}) })
}); });
} }
fn odm_hook_calls(bucket: &str) -> Vec<RecordedOdmConfig> { fn odm_hook_calls(bucket: &str) -> Vec<Option<OnDemandMigrationConfig>> {
ODM_HOOK_CALLS ODM_HOOK_CALLS
.lock() .lock()
.unwrap() .unwrap()
@@ -4404,6 +4143,54 @@ mod tests {
.collect() .collect()
} }
/// rustfs/backlog#2148: the accessor reports absence as `Ok(None)` and a
/// stored payload it cannot parse as a typed error, never as a default
/// and never as `ConfigNotFound`.
#[tokio::test]
async fn get_on_demand_migration_config_distinguishes_absent_from_corrupt() {
use crate::bucket::on_demand_migration::OnDemandMigrationConfigError;
let (_dirs, ecstore) = isolated_store_over_temp_disks().await;
let sys = BucketMetadataSys::new(ecstore);
let bucket = "odm-accessor";
sys.set(bucket.to_string(), Arc::new(BucketMetadata::new(bucket))).await;
assert_eq!(sys.get_on_demand_migration_config(bucket).await.unwrap(), None);
let mut corrupt = BucketMetadata::new(bucket);
corrupt.on_demand_migration_config_json = br#"{"source":{"provider":"s3"},"bogus":1}"#.to_vec();
sys.set(bucket.to_string(), Arc::new(corrupt)).await;
let err = sys
.get_on_demand_migration_config(bucket)
.await
.expect_err("corrupt config must not read as a default");
assert_ne!(err, Error::ConfigNotFound, "corruption must not be reported as absence");
let typed = match &err {
Error::Io(io) => io
.get_ref()
.and_then(|source| source.downcast_ref::<OnDemandMigrationConfigError>()),
_ => None,
};
assert!(
matches!(typed, Some(OnDemandMigrationConfigError::Malformed(_))),
"typed parse error must survive the Result boundary, got: {err:?}"
);
let mut valid = BucketMetadata::new(bucket);
valid
.update_config(crate::bucket::metadata::BUCKET_ON_DEMAND_MIGRATION_CONFIG, ODM_JSON.to_vec())
.unwrap();
let stamped = valid.on_demand_migration_config_updated_at;
sys.set(bucket.to_string(), Arc::new(valid)).await;
let (config, updated_at) = sys
.get_on_demand_migration_config(bucket)
.await
.unwrap()
.expect("stored config is returned");
assert_eq!(config, OnDemandMigrationConfig::from_json(ODM_JSON).unwrap());
assert_eq!(updated_at, stamped);
}
/// rustfs/backlog#2148: the publish hook fires on every path that /// rustfs/backlog#2148: the publish hook fires on every path that
/// installs bucket metadata into the cache (set, initial load, peer /// installs bucket metadata into the cache (set, initial load, peer
/// reload, refresh loop, lazy load) and withdraws on removal, mirroring /// reload, refresh loop, lazy load) and withdraws on removal, mirroring
@@ -4417,22 +4204,15 @@ mod tests {
for dir in &dirs { for dir in &dirs {
std::fs::create_dir_all(dir.path().join(bucket)).expect("physical bucket should exist"); std::fs::create_dir_all(dir.path().join(bucket)).expect("physical bucket should exist");
} }
let expected = OnDemandMigrationConfig::from_json(ODM_JSON).unwrap();
let incarnation = Uuid::new_v4();
let expect_publish = |before: usize, label: &str| { let expect_publish = |before: usize, label: &str| {
let calls = odm_hook_calls(bucket); let calls = odm_hook_calls(bucket);
assert_eq!(calls.len(), before + 1, "{label} must publish exactly once"); assert_eq!(calls.len(), before + 1, "{label} must publish exactly once");
assert_eq!( assert_eq!(calls.last().unwrap().as_ref(), Some(&expected), "{label} must publish the stored config");
calls.last().unwrap().as_ref().map(|(bytes, _, _)| bytes.as_slice()),
Some(ODM_JSON),
"{label} must publish the stored bytes"
);
assert_eq!(calls.last().unwrap().as_ref().map(|(_, _, id)| *id), Some(incarnation));
}; };
// set (via persist_new_and_set, which installs through `set`). // set (via persist_new_and_set, which installs through `set`).
let mut bm = BucketMetadata::new(bucket); let mut bm = BucketMetadata::new(bucket);
bm.bucket_incarnation_id = incarnation;
bm.update_config(crate::bucket::metadata::BUCKET_ON_DEMAND_MIGRATION_CONFIG, ODM_JSON.to_vec()) bm.update_config(crate::bucket::metadata::BUCKET_ON_DEMAND_MIGRATION_CONFIG, ODM_JSON.to_vec())
.unwrap(); .unwrap();
let writer = BucketMetadataSys::new(ecstore.clone()); let writer = BucketMetadataSys::new(ecstore.clone());
@@ -4474,18 +4254,14 @@ mod tests {
assert_eq!(calls.len(), before + 1, "remove must withdraw exactly once"); assert_eq!(calls.len(), before + 1, "remove must withdraw exactly once");
assert_eq!(calls.last().unwrap(), &None); assert_eq!(calls.last().unwrap(), &None);
// Opaque bytes reach the application even if they are not valid JSON. // A corrupt payload is withdrawn, never published as a config.
let mut corrupt = BucketMetadata::new(bucket); let mut corrupt = BucketMetadata::new(bucket);
corrupt.on_demand_migration_config_json = b"not-json".to_vec(); corrupt.on_demand_migration_config_json = b"not-json".to_vec();
let before = odm_hook_calls(bucket).len(); let before = odm_hook_calls(bucket).len();
lazy.set(bucket.to_string(), Arc::new(corrupt)).await; lazy.set(bucket.to_string(), Arc::new(corrupt)).await;
let calls = odm_hook_calls(bucket); let calls = odm_hook_calls(bucket);
assert_eq!(calls.len(), before + 1); assert_eq!(calls.len(), before + 1);
assert_eq!( assert_eq!(calls.last().unwrap(), &None, "unreadable config must publish absence");
calls.last().unwrap().as_ref().map(|(bytes, _, _)| bytes.as_slice()),
Some(b"not-json".as_slice()),
"the application validates opaque config bytes"
);
} }
#[tokio::test] #[tokio::test]
+1
View File
@@ -26,6 +26,7 @@ mod metadata_test;
pub mod migration; pub mod migration;
mod msgp_decode; mod msgp_decode;
pub mod object_lock; pub mod object_lock;
pub mod on_demand_migration;
pub mod policy_sys; pub mod policy_sys;
pub mod quota; pub mod quota;
pub mod remote_s3_client; pub mod remote_s3_client;
@@ -25,8 +25,8 @@
//! [`BACKFILL_SAVE_INTERVAL`], and at every page end, with an `If-Match` //! [`BACKFILL_SAVE_INTERVAL`], and at every page end, with an `If-Match`
//! compare-and-set so a concurrent cancel or takeover is never overwritten. //! compare-and-set so a concurrent cancel or takeover is never overwritten.
//! - The `continuation_token` only advances once every pull queued from the //! - The `continuation_token` only advances once every pull queued from the
//! page before it has succeeded. After a failure it stays at that page, //! page before it has reported back, so a crash re-lists at most one page
//! so crash recovery cannot skip failed pulls (existing keys are skipped). //! (already-present keys are then skipped, never re-pulled).
//! - The owner holds a lease of [`BACKFILL_LEASE`] renewed by every save. The //! - The owner holds a lease of [`BACKFILL_LEASE`] renewed by every save. The
//! recovery loop ([`run_backfill_recovery_loop`]) scans the buckets this //! recovery loop ([`run_backfill_recovery_loop`]) scans the buckets this
//! node has an ODM state for every [`BACKFILL_RECOVERY_INTERVAL`] and takes //! node has an ODM state for every [`BACKFILL_RECOVERY_INTERVAL`] and takes
@@ -45,12 +45,16 @@
use super::pull::{EnqueueOutcome, PullReason, QueuedPullOutcome}; use super::pull::{EnqueueOutcome, PullReason, QueuedPullOutcome};
use super::source_client::{SourceError, SourcePage}; use super::source_client::{SourceError, SourcePage};
use super::storage_api::{
BUCKET_META_PREFIX, ECStore, HTTPPreconditions, NamespaceLocking as _, ObjectOperations as _, ObjectOptions,
RUSTFS_META_BUCKET, StorageError, WriteCompletion, get_lock_acquire_timeout, get_on_demand_migration_config_in,
local_node_name, read_config_with_metadata, save_config_with_opts,
};
use super::sys::{BucketOdmState, OnDemandMigrationSys}; use super::sys::{BucketOdmState, OnDemandMigrationSys};
use crate::bucket::metadata_sys::bucket_metadata_sys_of;
use crate::config::com::{read_config_with_metadata, save_config_with_opts};
use crate::disk::{BUCKET_META_PREFIX, RUSTFS_META_BUCKET};
use crate::error::Error as StorageError;
use crate::object_api::ObjectOptions;
use crate::runtime::sources::local_node_name;
use crate::set_disk::get_lock_acquire_timeout;
use crate::storage_api_contracts::{namespace::NamespaceLocking as _, object::HTTPPreconditions, object::ObjectOperations as _};
use crate::store::ECStore;
use async_trait::async_trait; use async_trait::async_trait;
use futures::StreamExt; use futures::StreamExt;
use futures::stream::FuturesUnordered; use futures::stream::FuturesUnordered;
@@ -363,15 +367,14 @@ pub struct LocalBackfillObject {
pub source_etag: Option<String>, pub source_etag: Option<String>,
} }
/// Shared report of a new or coalesced pull; absent only when not admitted. /// Receiver of one queued pull's report; `None` when the pull was coalesced
pub type PullReport = Option<super::pull::QueuedPullReport>; /// into one already running.
pub type PullReport = Option<oneshot::Receiver<QueuedPullOutcome>>;
/// Everything the job needs from its bucket, so the loop can run against a /// Everything the job needs from its bucket, so the loop can run against a
/// mock in unit tests. Production: [`BucketBackfillContext`]. /// mock in unit tests. Production: [`BucketBackfillContext`].
#[async_trait] #[async_trait]
pub trait BackfillContext: Send + Sync { pub trait BackfillContext: Send + Sync {
/// The bucket incarnation captured by this context.
fn incarnation_id(&self) -> Uuid;
/// One source page in the local key namespace. /// One source page in the local key namespace.
async fn list_page(&self, prefix: Option<&str>, token: Option<&str>, max_keys: i32) -> Result<SourcePage, SourceError>; async fn list_page(&self, prefix: Option<&str>, token: Option<&str>, max_keys: i32) -> Result<SourcePage, SourceError>;
/// Whether the breaker admits source traffic right now. /// Whether the breaker admits source traffic right now.
@@ -413,10 +416,6 @@ impl BucketBackfillContext {
#[async_trait] #[async_trait]
impl BackfillContext for BucketBackfillContext { impl BackfillContext for BucketBackfillContext {
fn incarnation_id(&self) -> Uuid {
self.state.incarnation_id()
}
async fn list_page(&self, prefix: Option<&str>, token: Option<&str>, max_keys: i32) -> Result<SourcePage, SourceError> { async fn list_page(&self, prefix: Option<&str>, token: Option<&str>, max_keys: i32) -> Result<SourcePage, SourceError> {
let client = self.state.client().map_err(|err| SourceError::Unsupported(err.to_string()))?; let client = self.state.client().map_err(|err| SourceError::Unsupported(err.to_string()))?;
let started = Instant::now(); let started = Instant::now();
@@ -476,10 +475,12 @@ impl BackfillContext for BucketBackfillContext {
} }
async fn config_updated_at(&self) -> Result<Option<OffsetDateTime>, StorageError> { async fn config_updated_at(&self) -> Result<Option<OffsetDateTime>, StorageError> {
Ok( let sys = bucket_metadata_sys_of(&self.api.ctx)?;
super::config::decode_stored_config(get_on_demand_migration_config_in(&self.api, self.state.bucket()).await?)? let guard = sys.read().await;
.map(|(_, updated_at)| updated_at), Ok(guard
) .get_on_demand_migration_config(self.state.bucket())
.await?
.map(|(_, updated_at)| updated_at))
} }
} }
@@ -667,36 +668,8 @@ pub async fn read_checkpoint(api: &Arc<ECStore>, bucket: &str) -> Result<Option<
async fn write_checkpoint( async fn write_checkpoint(
api: &Arc<ECStore>, api: &Arc<ECStore>,
bucket: &str, bucket: &str,
incarnation_id: Uuid,
checkpoint: &BackfillCheckpoint, checkpoint: &BackfillCheckpoint,
expected_etag: Option<&str>, expected_etag: Option<&str>,
) -> Result<String, BackfillError> {
let api = Arc::clone(api);
let bucket = bucket.to_string();
let checkpoint = checkpoint.clone();
let expected_etag = expected_etag.map(str::to_string);
// The storage commit owns detached work. Keep its user-bucket fence alive
// even when a caller aborts its waiter before the erasure tail has drained.
tokio::spawn(async move {
let fence = api.acquire_bucket_incarnation_fence(&bucket, incarnation_id).await?;
let mut opts = ObjectOptions::default();
fence.attach_to_object_options(&mut opts);
let result = write_checkpoint_while_fenced(&api, &bucket, &checkpoint, expected_etag.as_deref(), opts).await;
drop(fence);
result
})
.await
.map_err(|err| StorageError::other(format!("backfill checkpoint task failed: {err}")))?
}
/// The caller holds the destination bucket's lifecycle fence through the CAS
/// write and its read-back, including the drained erasure write tail.
async fn write_checkpoint_while_fenced(
api: &Arc<ECStore>,
bucket: &str,
checkpoint: &BackfillCheckpoint,
expected_etag: Option<&str>,
mut opts: ObjectOptions,
) -> Result<String, BackfillError> { ) -> Result<String, BackfillError> {
let data = checkpoint.to_json()?; let data = checkpoint.to_json()?;
let preconditions = match expected_etag { let preconditions = match expected_etag {
@@ -709,9 +682,11 @@ async fn write_checkpoint_while_fenced(
..Default::default() ..Default::default()
}, },
}; };
opts.max_parity = true; let opts = ObjectOptions {
opts.write_completion = WriteCompletion::TailDrained; max_parity: true,
opts.http_preconditions = Some(preconditions); http_preconditions: Some(preconditions),
..Default::default()
};
match save_config_with_opts(Arc::clone(api), &checkpoint_path(bucket), data, &opts).await { match save_config_with_opts(Arc::clone(api), &checkpoint_path(bucket), data, &opts).await {
Ok(()) => {} Ok(()) => {}
Err(StorageError::PreconditionFailed) => return Err(BackfillError::Conflict(bucket.to_string())), Err(StorageError::PreconditionFailed) => return Err(BackfillError::Conflict(bucket.to_string())),
@@ -882,14 +857,7 @@ impl BackfillRunner {
}); });
} }
let checkpoint = BackfillCheckpoint::new(&request, config_updated_at, &self.node, now); let checkpoint = BackfillCheckpoint::new(&request, config_updated_at, &self.node, now);
let etag = write_checkpoint( let etag = write_checkpoint(&self.api, bucket, &checkpoint, stored.as_ref().map(|s| s.etag.as_str())).await?;
&self.api,
bucket,
context.incarnation_id(),
&checkpoint,
stored.as_ref().map(|s| s.etag.as_str()),
)
.await?;
info!( info!(
event = EVENT_ODM_BACKFILL_STATE, event = EVENT_ODM_BACKFILL_STATE,
component = LOG_COMPONENT_ECSTORE, component = LOG_COMPONENT_ECSTORE,
@@ -924,42 +892,30 @@ impl BackfillRunner {
} }
return Ok(handle.snapshot.lock().clone()); return Ok(handle.snapshot.lock().clone());
} }
let incarnation_id = self.api.bucket_incarnation_id_from_disk(bucket).await?; let _lock = self.lease_lock(bucket, get_lock_acquire_timeout()).await?;
let lock = self.lease_lock(bucket, get_lock_acquire_timeout()).await?; let Some(stored) = read_checkpoint(&self.api, bucket).await? else {
let api = Arc::clone(&self.api); return Err(BackfillError::NotFound(bucket.to_string()));
let bucket = bucket.to_string(); };
tokio::spawn(async move { if !stored.checkpoint.state.is_active() {
let _lock = lock; return Ok(stored.checkpoint);
let fence = api.acquire_bucket_incarnation_fence(&bucket, incarnation_id).await?; }
let mut opts = ObjectOptions::default(); let mut checkpoint = stored.checkpoint;
fence.attach_to_object_options(&mut opts); let now = OffsetDateTime::now_utc();
let Some(stored) = read_checkpoint(&api, &bucket).await? else { checkpoint.state = BackfillState::Cancelled;
return Err(BackfillError::NotFound(bucket.to_string())); checkpoint.updated_at = now;
}; write_checkpoint(&self.api, bucket, &checkpoint, Some(&stored.etag)).await?;
if !stored.checkpoint.state.is_active() { info!(
return Ok(stored.checkpoint); event = EVENT_ODM_BACKFILL_STATE,
} component = LOG_COMPONENT_ECSTORE,
let mut checkpoint = stored.checkpoint; subsystem = LOG_SUBSYSTEM_ON_DEMAND_MIGRATION,
let now = OffsetDateTime::now_utc(); state = checkpoint.state.as_str(),
checkpoint.state = BackfillState::Cancelled; result = "cancelled",
checkpoint.updated_at = now; bucket = %bucket,
write_checkpoint_while_fenced(&api, &bucket, &checkpoint, Some(&stored.etag), opts).await?; job_id = %checkpoint.job_id,
info!( owner = %checkpoint.owner.as_ref().map(|o| o.node.as_str()).unwrap_or_default(),
event = EVENT_ODM_BACKFILL_STATE, "On-demand migration backfill job cancelled remotely"
component = LOG_COMPONENT_ECSTORE, );
subsystem = LOG_SUBSYSTEM_ON_DEMAND_MIGRATION, Ok(checkpoint)
state = checkpoint.state.as_str(),
result = "cancelled",
bucket = %bucket,
job_id = %checkpoint.job_id,
owner = %checkpoint.owner.as_ref().map(|o| o.node.as_str()).unwrap_or_default(),
"On-demand migration backfill job cancelled remotely"
);
drop(fence);
Ok(checkpoint)
})
.await
.map_err(|err| StorageError::other(format!("backfill cancellation task failed: {err}")))?
} }
/// Latest checkpoint: the in-memory progress of a local job, else the /// Latest checkpoint: the in-memory progress of a local job, else the
@@ -1052,7 +1008,7 @@ impl BackfillRunner {
checkpoint.state = BackfillState::Cancelled; checkpoint.state = BackfillState::Cancelled;
checkpoint.updated_at = now; checkpoint.updated_at = now;
checkpoint.record_failure("config_changed", None, now); checkpoint.record_failure("config_changed", None, now);
write_checkpoint(&self.api, bucket, context.incarnation_id(), &checkpoint, Some(&stored.etag)).await?; write_checkpoint(&self.api, bucket, &checkpoint, Some(&stored.etag)).await?;
info!( info!(
event = EVENT_ODM_BACKFILL_STATE, event = EVENT_ODM_BACKFILL_STATE,
component = LOG_COMPONENT_ECSTORE, component = LOG_COMPONENT_ECSTORE,
@@ -1071,7 +1027,7 @@ impl BackfillRunner {
node: self.node.clone(), node: self.node.clone(),
lease_until: now + BACKFILL_LEASE, lease_until: now + BACKFILL_LEASE,
}); });
let etag = write_checkpoint(&self.api, bucket, context.incarnation_id(), &checkpoint, Some(&stored.etag)).await?; let etag = write_checkpoint(&self.api, bucket, &checkpoint, Some(&stored.etag)).await?;
warn!( warn!(
event = EVENT_ODM_BACKFILL_LEASE_TAKEOVER, event = EVENT_ODM_BACKFILL_LEASE_TAKEOVER,
component = LOG_COMPONENT_ECSTORE, component = LOG_COMPONENT_ECSTORE,
@@ -1234,11 +1190,9 @@ impl Job {
} }
async fn main_loop(&mut self) -> Result<(), Stop> { async fn main_loop(&mut self) -> Result<(), Stop> {
let mut cursor = self.checkpoint.continuation_token.clone();
let failed_at_resume = self.checkpoint.failed;
loop { loop {
self.check_cancel()?; self.check_cancel()?;
let page = self.list_page(cursor.as_deref()).await?; let page = self.list_page().await?;
for object in &page.objects { for object in &page.objects {
self.check_cancel()?; self.check_cancel()?;
self.checkpoint.listed += 1; self.checkpoint.listed += 1;
@@ -1250,13 +1204,10 @@ impl Job {
self.drain_ready(); self.drain_ready();
self.tick(false).await?; self.tick(false).await?;
} }
// A persisted cursor certifies successful work, not just listing // Only advance the cursor once every pull of this page reported
// progress. Keep it at the first failed page for crash recovery. // back, so a takeover re-lists at most this page.
self.drain_all().await?; self.drain_all().await?;
cursor = page.next_continuation_token; self.checkpoint.continuation_token = page.next_continuation_token.clone();
if self.checkpoint.failed == failed_at_resume {
self.checkpoint.continuation_token = cursor.clone();
}
self.tick(true).await?; self.tick(true).await?;
if !page.is_truncated { if !page.is_truncated {
return Ok(()); return Ok(());
@@ -1271,7 +1222,7 @@ impl Job {
} }
} }
async fn list_page(&mut self, cursor: Option<&str>) -> Result<SourcePage, Stop> { async fn list_page(&mut self) -> Result<SourcePage, Stop> {
let mut attempt = 0; let mut attempt = 0;
loop { loop {
while !self.context.source_available() { while !self.context.source_available() {
@@ -1279,7 +1230,7 @@ impl Job {
self.tick(false).await?; self.tick(false).await?;
} }
let prefix = self.checkpoint.prefix.clone(); let prefix = self.checkpoint.prefix.clone();
let token = cursor.map(str::to_string); let token = self.checkpoint.continuation_token.clone();
match self match self
.context .context
.list_page(prefix.as_deref(), token.as_deref(), BACKFILL_LIST_PAGE_SIZE) .list_page(prefix.as_deref(), token.as_deref(), BACKFILL_LIST_PAGE_SIZE)
@@ -1353,10 +1304,9 @@ impl Job {
} }
loop { loop {
match self.context.enqueue(key) { match self.context.enqueue(key) {
(EnqueueOutcome::Enqueued | EnqueueOutcome::Coalesced, report) => { (EnqueueOutcome::Enqueued, report) => {
self.checkpoint.enqueued += 1; self.checkpoint.enqueued += 1;
let rx = report.ok_or(Stop::Unavailable)?; if let Some(rx) = report {
{
let key = key.to_string(); let key = key.to_string();
self.outstanding.push(Box::pin(async move { (key, rx.await) })); self.outstanding.push(Box::pin(async move { (key, rx.await) }));
} }
@@ -1371,6 +1321,11 @@ impl Job {
); );
return Ok(()); return Ok(());
} }
(EnqueueOutcome::Coalesced, _) => {
// Someone else pulls it; its result is not ours to count.
self.checkpoint.enqueued += 1;
return Ok(());
}
(EnqueueOutcome::QueueFull, _) => { (EnqueueOutcome::QueueFull, _) => {
// Wait, never drop: one completion frees a slot. // Wait, never drop: one completion frees a slot.
if self.outstanding.is_empty() { if self.outstanding.is_empty() {
@@ -1510,8 +1465,7 @@ impl Job {
lease_until: now + BACKFILL_LEASE, lease_until: now + BACKFILL_LEASE,
}); });
} }
let etag = let etag = write_checkpoint(&self.api, &self.bucket, &self.checkpoint, Some(&self.etag)).await?;
write_checkpoint(&self.api, &self.bucket, self.context.incarnation_id(), &self.checkpoint, Some(&self.etag)).await?;
self.etag = etag; self.etag = etag;
self.keys_since_save = 0; self.keys_since_save = 0;
self.last_save = Instant::now(); self.last_save = Instant::now();
@@ -1523,7 +1477,7 @@ impl Job {
/// Spawns [`run_backfill_recovery_loop`] on the store's shutdown token; /// Spawns [`run_backfill_recovery_loop`] on the store's shutdown token;
/// `false` (nothing spawned) when the store has no background token. /// `false` (nothing spawned) when the store has no background token.
pub fn spawn_backfill_recovery_loop(runner: Arc<BackfillRunner>) -> bool { pub fn spawn_backfill_recovery_loop(runner: Arc<BackfillRunner>) -> bool {
let Some(cancel) = runner.api.background_cancel_token() else { let Some(cancel) = runner.api.ctx.background_cancel_token() else {
return false; return false;
}; };
tokio::spawn(run_backfill_recovery_loop(runner, cancel)); tokio::spawn(run_backfill_recovery_loop(runner, cancel));
@@ -1551,13 +1505,10 @@ pub async fn run_backfill_recovery_loop(runner: Arc<BackfillRunner>, cancel: Can
#[cfg(test)] #[cfg(test)]
mod tests { mod tests {
use super::super::storage_api::test_support::{
BUCKET_LIFECYCLE_LOCK_OBJECT, BucketOperations as _, PutObjectCommitBarrier, PutObjectCommitPause,
isolated_store_over_temp_disks,
};
use super::*; use super::*;
use crate::on_demand_migration::source_client::SourceObject; use crate::bucket::metadata_sys::test_support::isolated_store_over_temp_disks;
use crate::on_demand_migration::sys::PullError; use crate::bucket::on_demand_migration::source_client::SourceObject;
use crate::bucket::on_demand_migration::sys::PullError;
use std::collections::{BTreeSet, HashSet}; use std::collections::{BTreeSet, HashSet};
use std::sync::atomic::AtomicBool; use std::sync::atomic::AtomicBool;
@@ -1680,7 +1631,6 @@ mod tests {
/// Scripted source + local store + queue with a controllable report path. /// Scripted source + local store + queue with a controllable report path.
struct MockContext { struct MockContext {
incarnation_id: Mutex<Option<Uuid>>,
objects: Vec<SourceObject>, objects: Vec<SourceObject>,
page_size: usize, page_size: usize,
local: Mutex<HashMap<String, LocalBackfillObject>>, local: Mutex<HashMap<String, LocalBackfillObject>>,
@@ -1689,7 +1639,6 @@ mod tests {
queue_capacity: usize, queue_capacity: usize,
pending: Mutex<Vec<(String, oneshot::Sender<QueuedPullOutcome>)>>, pending: Mutex<Vec<(String, oneshot::Sender<QueuedPullOutcome>)>>,
fail_keys: HashSet<String>, fail_keys: HashSet<String>,
coalesced: bool,
auto_complete: AtomicBool, auto_complete: AtomicBool,
cancel: CancellationToken, cancel: CancellationToken,
config_updated_at: Mutex<Option<OffsetDateTime>>, config_updated_at: Mutex<Option<OffsetDateTime>>,
@@ -1709,7 +1658,6 @@ mod tests {
}) })
.collect(); .collect();
Arc::new(Self { Arc::new(Self {
incarnation_id: Mutex::new(None),
objects, objects,
page_size, page_size,
local: Mutex::new(HashMap::new()), local: Mutex::new(HashMap::new()),
@@ -1718,7 +1666,6 @@ mod tests {
queue_capacity: usize::MAX, queue_capacity: usize::MAX,
pending: Mutex::new(Vec::new()), pending: Mutex::new(Vec::new()),
fail_keys: HashSet::new(), fail_keys: HashSet::new(),
coalesced: false,
auto_complete: AtomicBool::new(true), auto_complete: AtomicBool::new(true),
cancel: CancellationToken::new(), cancel: CancellationToken::new(),
config_updated_at: Mutex::new(Some(ts(1_700_000_000))), config_updated_at: Mutex::new(Some(ts(1_700_000_000))),
@@ -1744,10 +1691,6 @@ mod tests {
#[async_trait] #[async_trait]
impl BackfillContext for MockContext { impl BackfillContext for MockContext {
fn incarnation_id(&self) -> Uuid {
self.incarnation_id.lock().expect("test bucket initialized")
}
async fn list_page(&self, prefix: Option<&str>, token: Option<&str>, max_keys: i32) -> Result<SourcePage, SourceError> { async fn list_page(&self, prefix: Option<&str>, token: Option<&str>, max_keys: i32) -> Result<SourcePage, SourceError> {
if let Some(err) = self.list_error.lock().take() { if let Some(err) = self.list_error.lock().take() {
return Err(err); return Err(err);
@@ -1802,12 +1745,7 @@ mod tests {
} else { } else {
self.pending.lock().push((key.to_string(), tx)); self.pending.lock().push((key.to_string(), tx));
} }
let outcome = if self.coalesced { (EnqueueOutcome::Enqueued, Some(rx))
EnqueueOutcome::Coalesced
} else {
EnqueueOutcome::Enqueued
};
(outcome, Some(futures::FutureExt::shared(rx)))
} }
fn cancel_token(&self) -> CancellationToken { fn cancel_token(&self) -> CancellationToken {
@@ -1840,17 +1778,12 @@ mod tests {
context: Arc<MockContext>, context: Arc<MockContext>,
) -> (Vec<tempfile::TempDir>, Arc<ECStore>, Arc<BackfillRunner>) { ) -> (Vec<tempfile::TempDir>, Arc<ECStore>, Arc<BackfillRunner>) {
let (dirs, store) = isolated_store_over_temp_disks().await; let (dirs, store) = isolated_store_over_temp_disks().await;
super::super::storage_api::test_support::init_bucket_metadata_sys(Arc::clone(&store), Vec::new()).await; // The isolated store has no bucket metadata system; the checkpoint
store // only needs the bucket's directory under the metadata volume.
.make_bucket(bucket, &Default::default()) for dir in &dirs {
.await std::fs::create_dir_all(dir.path().join(RUSTFS_META_BUCKET).join(BUCKET_META_PREFIX).join(bucket))
.expect("create test bucket"); .expect("test bucket metadata directory");
*context.incarnation_id.lock() = Some( }
store
.bucket_incarnation_id_from_disk(bucket)
.await
.expect("test bucket identity"),
);
let runner = runner_on(node, bucket, context, Arc::clone(&store)); let runner = runner_on(node, bucket, context, Arc::clone(&store));
(dirs, store, runner) (dirs, store, runner)
} }
@@ -1860,129 +1793,6 @@ mod tests {
BackfillRunner::new(store, node, Arc::new(contexts)) BackfillRunner::new(store, node, Arc::new(contexts))
} }
#[tokio::test]
async fn cancelled_checkpoint_waiter_keeps_bucket_fenced_until_commit_finishes() {
for (suffix, pause) in [
("before", PutObjectCommitPause::BeforeQuotaRename),
("after", PutObjectCommitPause::AfterRenameQuorum),
] {
let bucket = format!("backfill-cancel-tail-{suffix}");
let context = MockContext::new(0, 1);
let (_dirs, store, _runner) = runner_with("node-a", &bucket, Arc::clone(&context)).await;
let original_incarnation = context.incarnation_id();
let checkpoint = BackfillCheckpoint::new(&BackfillRequest::default(), ts(1_700_000_000), "node-a", ts(1_700_000_001));
let barrier = PutObjectCommitBarrier::install(RUSTFS_META_BUCKET, &checkpoint_path(&bucket), pause);
let writer_api = Arc::clone(&store);
let writer_bucket = bucket.clone();
let waiter = tokio::spawn(async move {
write_checkpoint(&writer_api, &writer_bucket, original_incarnation, &checkpoint, None).await
});
barrier.wait_until_paused().await;
waiter.abort();
assert!(waiter.await.expect_err("caller aborted").is_cancelled());
let lifecycle_lock = store
.new_ns_lock(&bucket, BUCKET_LIFECYCLE_LOCK_OBJECT)
.await
.expect("lifecycle lock");
{
let mut probe = Box::pin(lifecycle_lock.get_write_lock(Duration::from_secs(1)));
assert!(
futures::poll!(probe.as_mut()).is_pending(),
"lifecycle writer must first try to acquire the lock"
);
assert!(
tokio::time::timeout(Duration::from_millis(100), probe.as_mut())
.await
.is_err(),
"the checkpoint owner must retain the user bucket lifecycle read lock after caller cancellation"
);
}
let delete_api = Arc::clone(&store);
let delete_bucket = bucket.clone();
let mut deletion = tokio::spawn(async move { delete_api.delete_bucket(&delete_bucket, &Default::default()).await });
assert!(
tokio::time::timeout(Duration::from_millis(100), &mut deletion).await.is_err(),
"DeleteBucket must wait for the checkpoint owner after its caller aborts"
);
barrier.release();
tokio::time::timeout(Duration::from_secs(10), deletion)
.await
.expect("commit must drain and release its lifecycle guard")
.expect("delete task")
.expect("delete original bucket");
store
.make_bucket(&bucket, &Default::default())
.await
.expect("recreate bucket");
assert_ne!(
original_incarnation,
store.bucket_incarnation_id_from_disk(&bucket).await.expect("new identity")
);
assert!(
read_checkpoint(&store, &bucket)
.await
.expect("read recreated bucket")
.is_none(),
"no old checkpoint may outlive bucket deletion"
);
}
}
#[tokio::test]
async fn stale_checkpoint_writer_cannot_resurrect_or_overwrite_a_recreated_bucket() {
let bucket = "backfill-incarnation";
let context = MockContext::new(0, 1);
let (_dirs, store, _runner) = runner_with("node-a", bucket, Arc::clone(&context)).await;
let old_incarnation = context.incarnation_id();
let old = BackfillCheckpoint::new(&BackfillRequest::default(), ts(1_700_000_000), "node-a", ts(1_700_000_001));
let old_etag = write_checkpoint(&store, bucket, old_incarnation, &old, None)
.await
.expect("old checkpoint");
store
.delete_bucket(bucket, &Default::default())
.await
.expect("delete original bucket");
store.make_bucket(bucket, &Default::default()).await.expect("recreate bucket");
let current_incarnation = store.bucket_incarnation_id_from_disk(bucket).await.expect("new identity");
assert_ne!(old_incarnation, current_incarnation);
assert!(
read_checkpoint(&store, bucket)
.await
.expect("read after recreation")
.is_none()
);
for expected_etag in [None, Some(old_etag.as_str())] {
let error = write_checkpoint(&store, bucket, old_incarnation, &old, expected_etag)
.await
.expect_err("stale writer rejected");
assert!(matches!(error, BackfillError::Storage(StorageError::BucketNotFound(_))));
}
assert!(
read_checkpoint(&store, bucket)
.await
.expect("stale writer left no checkpoint")
.is_none()
);
let current = BackfillCheckpoint::new(&BackfillRequest::default(), ts(1_700_000_000), "node-b", ts(1_700_000_002));
let current_etag = write_checkpoint(&store, bucket, current_incarnation, &current, None)
.await
.expect("current checkpoint");
let error = write_checkpoint(&store, bucket, old_incarnation, &old, Some(&current_etag))
.await
.expect_err("old identity cannot overwrite a matching ETag");
assert!(matches!(error, BackfillError::Storage(StorageError::BucketNotFound(_))));
let stored = read_checkpoint(&store, bucket)
.await
.expect("read current checkpoint")
.expect("current checkpoint remains");
assert_eq!(stored.etag, current_etag);
assert_eq!(stored.checkpoint, current);
}
#[tokio::test] #[tokio::test]
async fn full_backfill_lists_pages_and_counts_every_key() { async fn full_backfill_lists_pages_and_counts_every_key() {
let bucket = "backfill-full"; let bucket = "backfill-full";
@@ -2101,7 +1911,7 @@ mod tests {
#[tokio::test] #[tokio::test]
async fn failed_pulls_are_counted_hashed_and_finish_with_failures() { async fn failed_pulls_are_counted_hashed_and_finish_with_failures() {
let bucket = "backfill-failed"; let bucket = "backfill-failed";
let mut context = MockContext::new(5, 2); let mut context = MockContext::new(5, 1000);
Arc::get_mut(&mut context) Arc::get_mut(&mut context)
.expect("unshared") .expect("unshared")
.fail_keys .fail_keys
@@ -2116,52 +1926,12 @@ mod tests {
.checkpoint; .checkpoint;
assert_eq!(cp.state, BackfillState::CompletedWithFailures); assert_eq!(cp.state, BackfillState::CompletedWithFailures);
assert_eq!((cp.pulled, cp.failed), (4, 1)); assert_eq!((cp.pulled, cp.failed), (4, 1));
assert_eq!(cp.continuation_token.as_deref(), Some("2"), "retain the first failed page for recovery");
assert_eq!(cp.failed_keys, vec![key_hash("k/00002")]); assert_eq!(cp.failed_keys, vec![key_hash("k/00002")]);
let last = cp.last_error.expect("last error"); let last = cp.last_error.expect("last error");
assert_eq!(last.class, "local_write"); assert_eq!(last.class, "local_write");
assert_eq!(last.key_hash.as_deref(), Some(key_hash("k/00002").as_str())); assert_eq!(last.key_hash.as_deref(), Some(key_hash("k/00002").as_str()));
} }
#[tokio::test]
async fn coalesced_pulls_block_the_checkpoint_and_report_failures() {
let bucket = "backfill-coalesced";
let mut context = MockContext::new(1, 1);
{
let ctx = Arc::get_mut(&mut context).expect("unshared");
ctx.coalesced = true;
ctx.auto_complete = AtomicBool::new(false);
ctx.fail_keys.insert("k/00000".to_string());
}
let (_dirs, store, runner) = runner_with("node-a", bucket, Arc::clone(&context)).await;
runner.start(bucket, BackfillRequest::default()).await.expect("start");
tokio::time::timeout(Duration::from_secs(10), async {
while context.pending.lock().is_empty() {
tokio::task::yield_now().await;
}
})
.await
.expect("job enqueued");
assert!(runner.is_running_locally(bucket), "coalescing is not completion");
let cp = read_checkpoint(&store, bucket)
.await
.expect("read")
.expect("checkpoint")
.checkpoint;
assert!(cp.state.is_active());
assert!(cp.continuation_token.is_none());
context.complete_pending();
runner.wait_until_idle(bucket).await;
let cp = read_checkpoint(&store, bucket)
.await
.expect("read")
.expect("checkpoint")
.checkpoint;
assert_eq!(cp.state, BackfillState::CompletedWithFailures);
assert_eq!((cp.enqueued, cp.pulled, cp.failed), (1, 0, 1));
assert_eq!(cp.failed_keys, vec![key_hash("k/00000")]);
}
#[tokio::test] #[tokio::test]
async fn listing_failure_marks_the_job_failed_with_the_error_class() { async fn listing_failure_marks_the_job_failed_with_the_error_class() {
let bucket = "backfill-list-error"; let bucket = "backfill-list-error";
@@ -2322,7 +2092,7 @@ mod tests {
node: "node-a".to_string(), node: "node-a".to_string(),
lease_until: now - Duration::from_secs(120), lease_until: now - Duration::from_secs(120),
}); });
let etag = write_checkpoint(&store, bucket, context.incarnation_id(), &crashed, None) let etag = write_checkpoint(&store, bucket, &crashed, None)
.await .await
.expect("seed checkpoint"); .expect("seed checkpoint");
@@ -2333,7 +2103,7 @@ mod tests {
lease_until: now + Duration::from_secs(60), lease_until: now + Duration::from_secs(60),
}); });
live.updated_at = now; live.updated_at = now;
let etag = write_checkpoint(&store, bucket, context.incarnation_id(), &live, Some(&etag)) let etag = write_checkpoint(&store, bucket, &live, Some(&etag))
.await .await
.expect("live lease"); .expect("live lease");
assert_eq!(runner.recover_once().await.taken_over, 0, "unexpired lease must not be taken over"); assert_eq!(runner.recover_once().await.taken_over, 0, "unexpired lease must not be taken over");
@@ -2350,7 +2120,7 @@ mod tests {
lease_until: now - Duration::from_secs(1), lease_until: now - Duration::from_secs(1),
}); });
expired.updated_at = now + Duration::from_millis(1); expired.updated_at = now + Duration::from_millis(1);
write_checkpoint(&store, bucket, context.incarnation_id(), &expired, Some(&etag)) write_checkpoint(&store, bucket, &expired, Some(&etag))
.await .await
.expect("expire lease"); .expect("expire lease");
let stats = runner.recover_once().await; let stats = runner.recover_once().await;
@@ -2374,68 +2144,6 @@ mod tests {
assert_eq!(runner.recover_once().await.taken_over, 0, "a finished job is not recovered"); assert_eq!(runner.recover_once().await.taken_over, 0, "a finished job is not recovered");
} }
#[tokio::test]
async fn recovery_advances_past_historical_failures_but_pins_new_failures() {
let bucket = "backfill-takeover-failed";
let mut context = MockContext::new(8, 2);
{
let ctx = Arc::get_mut(&mut context).expect("unshared");
ctx.auto_complete = AtomicBool::new(false);
ctx.fail_keys.insert("k/00004".to_string());
}
let (_dirs, store, runner) = runner_with("node-b", bucket, Arc::clone(&context)).await;
let crashed_at = OffsetDateTime::now_utc() - Duration::from_secs(300);
let mut crashed = BackfillCheckpoint::new(&BackfillRequest::default(), ts(1_700_000_000), "node-a", crashed_at);
crashed.continuation_token = Some("2".to_string());
crashed.failed = 1;
crashed.record_failure("local_write", Some("k/00002"), crashed_at);
write_checkpoint(&store, bucket, context.incarnation_id(), &crashed, None)
.await
.expect("seed failed page with an expired lease");
assert_eq!(runner.recover_once().await.taken_over, 1);
for (page_start, durable_token, failures) in [(2, "2", 1), (4, "4", 1), (6, "4", 2)] {
tokio::time::timeout(Duration::from_secs(10), async {
loop {
if context.pending.lock().len() == 2 {
break;
}
tokio::task::yield_now().await;
}
})
.await
.expect("resumed page enqueued before its reports complete");
assert_eq!(
context.pending.lock().iter().map(|(key, _)| key.clone()).collect::<Vec<_>>(),
vec![format!("k/{page_start:05}"), format!("k/{:05}", page_start + 1)]
);
let cp = read_checkpoint(&store, bucket)
.await
.expect("read persisted page boundary")
.expect("checkpoint")
.checkpoint;
assert_eq!(cp.job_id, crashed.job_id);
assert_eq!(cp.owner.as_ref().map(|owner| owner.node.as_str()), Some("node-b"));
assert_eq!(cp.continuation_token.as_deref(), Some(durable_token));
assert_eq!(cp.failed, failures);
context.complete_pending();
}
runner.wait_until_idle(bucket).await;
let cp = read_checkpoint(&store, bucket)
.await
.expect("read completed checkpoint")
.expect("checkpoint")
.checkpoint;
assert_eq!(cp.state, BackfillState::CompletedWithFailures);
assert_eq!((cp.pulled, cp.failed), (5, 2));
assert_eq!(cp.continuation_token.as_deref(), Some("4"));
assert_eq!(cp.failed_keys, vec![key_hash("k/00002"), key_hash("k/00004")]);
assert_eq!(
context.list_requests.lock().as_slice(),
&[Some("2".to_string()), Some("4".to_string()), Some("6".to_string())]
);
}
#[tokio::test] #[tokio::test]
async fn recovery_cancels_a_job_whose_config_changed_and_reclaims_own_node_jobs() { async fn recovery_cancels_a_job_whose_config_changed_and_reclaims_own_node_jobs() {
let bucket = "backfill-recovery-config"; let bucket = "backfill-recovery-config";
@@ -2445,9 +2153,7 @@ mod tests {
// Same node name, unexpired lease: only a restart can produce this. // Same node name, unexpired lease: only a restart can produce this.
let own = BackfillCheckpoint::new(&BackfillRequest::default(), ts(1_700_000_000), "node-a", now); let own = BackfillCheckpoint::new(&BackfillRequest::default(), ts(1_700_000_000), "node-a", now);
let etag = write_checkpoint(&store, bucket, context.incarnation_id(), &own, None) let etag = write_checkpoint(&store, bucket, &own, None).await.expect("seed");
.await
.expect("seed");
assert_eq!(runner.recover_once().await.taken_over, 1, "own-node running job is reclaimed at once"); assert_eq!(runner.recover_once().await.taken_over, 1, "own-node running job is reclaimed at once");
runner.wait_until_idle(bucket).await; runner.wait_until_idle(bucket).await;
let cp = read_checkpoint(&store, bucket) let cp = read_checkpoint(&store, bucket)
@@ -2465,7 +2171,7 @@ mod tests {
node: "node-z".to_string(), node: "node-z".to_string(),
lease_until: now - Duration::from_secs(1), lease_until: now - Duration::from_secs(1),
}); });
write_checkpoint(&store, bucket, context.incarnation_id(), &stale, Some(&stored.etag)) write_checkpoint(&store, bucket, &stale, Some(&stored.etag))
.await .await
.expect("seed stale"); .expect("seed stale");
let stats = runner.recover_once().await; let stats = runner.recover_once().await;
@@ -86,12 +86,7 @@ impl BreakerVerdict {
Some(SourceError::Throttled | SourceError::Timeout | SourceError::Connect(_) | SourceError::ServerError(_)) => { Some(SourceError::Throttled | SourceError::Timeout | SourceError::Connect(_) | SourceError::ServerError(_)) => {
BreakerVerdict::Failure BreakerVerdict::Failure
} }
Some( Some(SourceError::AccessDenied | SourceError::Unsupported(_) | SourceError::Other(_)) => BreakerVerdict::Neutral,
SourceError::AccessDenied
| SourceError::Unsupported(_)
| SourceError::InvalidPagination(_)
| SourceError::Other(_),
) => BreakerVerdict::Neutral,
} }
} }
} }
@@ -14,44 +14,22 @@
//! Bucket-level On-Demand Migration configuration: wire model (JSON stored //! Bucket-level On-Demand Migration configuration: wire model (JSON stored
//! under `on-demand-migration.json`), pure validation, credential redaction, //! under `on-demand-migration.json`), pure validation, credential redaction,
//! and persisted-config decoding (rustfs/backlog#2148). //! and the publish hook the runtime registers into (rustfs/backlog#2148).
//! //!
//! The persisted blob is not encrypted; it shares the trust boundary of //! The persisted blob is not encrypted; it shares the trust boundary of
//! `bucket-targets.json` and `tier-config.bin`. //! `bucket-targets.json` and `tier-config.bin`.
use serde::{Deserialize, Serialize}; use serde::{Deserialize, Serialize};
use std::fmt; use std::fmt;
use std::sync::OnceLock;
use url::Url; use url::Url;
/// Decode bytes only at the service boundary, preserving typed corruption errors.
pub(super) fn decode_stored_config(
stored: Option<(Vec<u8>, time::OffsetDateTime)>,
) -> Result<Option<(OnDemandMigrationConfig, time::OffsetDateTime)>, super::storage_api::StorageError> {
stored
.map(|(bytes, updated_at)| {
OnDemandMigrationConfig::from_json(&bytes)
.map(|config| (config, updated_at))
.map_err(super::storage_api::StorageError::other)
})
.transpose()
}
pub(crate) async fn get_config(
bucket: &str,
) -> Result<Option<(OnDemandMigrationConfig, time::OffsetDateTime)>, super::storage_api::StorageError> {
decode_stored_config(super::storage_api::get_on_demand_migration_config(bucket).await?)
}
/// The only wire version this build reads and writes. /// The only wire version this build reads and writes.
pub const ON_DEMAND_MIGRATION_CONFIG_VERSION: u32 = 1; pub const ON_DEMAND_MIGRATION_CONFIG_VERSION: u32 = 1;
const REDACTED: &str = "REDACTED"; const REDACTED: &str = "REDACTED";
const AUTO_REGION: &str = "auto"; const AUTO_REGION: &str = "auto";
const AUTO_REGION_FALLBACK: &str = "us-east-1"; const AUTO_REGION_FALLBACK: &str = "us-east-1";
/// Public Azure Blob host suffix; the account name is the first label.
pub const AZURE_BLOB_SUFFIX: &str = "blob.core.windows.net";
/// Public Google Cloud Storage endpoint for the native provider.
pub const GCS_DEFAULT_ENDPOINT: &str = "https://storage.googleapis.com";
const KIB: u64 = 1024; const KIB: u64 = 1024;
const MIB: u64 = 1024 * KIB; const MIB: u64 = 1024 * KIB;
@@ -97,25 +75,14 @@ pub struct SourceConfig {
pub bucket: String, pub bucket: String,
#[serde(default)] #[serde(default)]
pub path_style: PathStyle, pub path_style: PathStyle,
/// `None` means anonymous access to a public source bucket. Only the /// `None` means anonymous access to a public source bucket.
/// SigV4 providers read it; `azure` and `gcs_native` carry their own
/// credentials in `azure` / `gcs`.
#[serde(default)] #[serde(default)]
pub credentials: Option<SourceCredentials>, pub credentials: Option<SourceCredentials>,
#[serde(default)] #[serde(default)]
pub tls: TlsConfig, pub tls: TlsConfig,
/// Required for [`Provider::Azure`] and rejected for every other
/// provider.
#[serde(default)]
pub azure: Option<AzureSourceConfig>,
/// Required for [`Provider::GcsNative`] and rejected for every other
/// provider. [`Provider::Gcs`] keeps using `credentials` because it
/// speaks the S3 interoperability API.
#[serde(default)]
pub gcs: Option<GcsSourceConfig>,
} }
/// Source vendor family. /// Source vendor family. `azure` is deliberately absent from this version.
#[derive(Debug, Clone, Copy, PartialEq, Eq, Serialize, Deserialize)] #[derive(Debug, Clone, Copy, PartialEq, Eq, Serialize, Deserialize)]
#[serde(rename_all = "lowercase")] #[serde(rename_all = "lowercase")]
pub enum Provider { pub enum Provider {
@@ -127,12 +94,6 @@ pub enum Provider {
R2, R2,
/// GCS XML interoperability API with HMAC keys. /// GCS XML interoperability API with HMAC keys.
Gcs, Gcs,
/// Native Azure Blob service; parameters in `source.azure`.
Azure,
/// Native GCS JSON API with a service-account key; parameters in
/// `source.gcs`.
#[serde(rename = "gcs_native")]
GcsNative,
} }
impl Provider { impl Provider {
@@ -144,22 +105,13 @@ impl Provider {
Provider::Rustfs => "rustfs", Provider::Rustfs => "rustfs",
Provider::R2 => "r2", Provider::R2 => "r2",
Provider::Gcs => "gcs", Provider::Gcs => "gcs",
Provider::Azure => "azure",
Provider::GcsNative => "gcs_native",
} }
} }
/// Providers that do not speak S3 and therefore ignore `region`,
/// `path_style` and `credentials`.
pub fn is_native(&self) -> bool {
matches!(self, Provider::Azure | Provider::GcsNative)
}
/// Providers whose SDKs accept `region = "auto"`; RustFS maps it to /// Providers whose SDKs accept `region = "auto"`; RustFS maps it to
/// `us-east-1` for signing. The native providers never sign with a /// `us-east-1` for signing.
/// region, so they accept it as well.
fn accepts_auto_region(&self) -> bool { fn accepts_auto_region(&self) -> bool {
matches!(self, Provider::R2 | Provider::Minio | Provider::Rustfs) || self.is_native() matches!(self, Provider::R2 | Provider::Minio | Provider::Rustfs)
} }
} }
@@ -212,73 +164,6 @@ impl fmt::Debug for SourceCredentials {
} }
} }
/// Native Azure Blob source parameters. The container is `source.bucket`,
/// so a config never carries two names for the same container. Exactly one
/// of `account_key` and `sas_token` must be set: the account key signs with
/// Shared Key, the SAS token is appended to every request URL.
#[derive(Clone, PartialEq, Eq, Serialize, Deserialize)]
#[serde(deny_unknown_fields)]
pub struct AzureSourceConfig {
/// Storage account name; also derives the default `blob.core.windows.net`
/// endpoint when `source.endpoint` is absent.
pub account: String,
/// Base64 shared key of the storage account.
#[serde(default)]
pub account_key: Option<String>,
/// SAS query string without the leading `?`.
#[serde(default)]
pub sas_token: Option<String>,
}
impl AzureSourceConfig {
/// A copy safe to return to admin clients or log: both secrets are
/// replaced by `REDACTED`, and whether each is set stays visible.
pub fn redacted(&self) -> Self {
Self {
account: self.account.clone(),
account_key: self.account_key.as_ref().map(|_| REDACTED.to_string()),
sas_token: self.sas_token.as_ref().map(|_| REDACTED.to_string()),
}
}
}
impl fmt::Debug for AzureSourceConfig {
fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result {
f.debug_struct("AzureSourceConfig")
.field("account", &self.account)
.field("account_key", &self.account_key.as_ref().map(|_| REDACTED))
.field("sas_token", &self.sas_token.as_ref().map(|_| REDACTED))
.finish()
}
}
/// Native Google Cloud Storage source parameters. The bucket is
/// `source.bucket`; only the service-account key lives here.
#[derive(Clone, PartialEq, Eq, Serialize, Deserialize)]
#[serde(deny_unknown_fields)]
pub struct GcsSourceConfig {
/// Service-account key JSON, verbatim as downloaded from Google Cloud.
pub service_account_json: String,
}
impl GcsSourceConfig {
/// A copy safe to return to admin clients or log: the whole key JSON is
/// a secret (it embeds the private key), so it is replaced wholesale.
pub fn redacted(&self) -> Self {
Self {
service_account_json: REDACTED.to_string(),
}
}
}
impl fmt::Debug for GcsSourceConfig {
fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result {
f.debug_struct("GcsSourceConfig")
.field("service_account_json", &REDACTED)
.finish()
}
}
#[derive(Debug, Clone, PartialEq, Eq, Default, Serialize, Deserialize)] #[derive(Debug, Clone, PartialEq, Eq, Default, Serialize, Deserialize)]
#[serde(deny_unknown_fields)] #[serde(deny_unknown_fields)]
pub struct TlsConfig { pub struct TlsConfig {
@@ -469,14 +354,6 @@ pub enum OnDemandMigrationConfigError {
InvalidBucket(&'static str), InvalidBucket(&'static str),
#[error("source credentials field {0} must not be empty")] #[error("source credentials field {0} must not be empty")]
EmptyCredential(&'static str), EmptyCredential(&'static str),
#[error("source.{0} is required for provider {1}")]
MissingProviderBlock(&'static str, Provider),
#[error("source.{0} is not valid for provider {1}")]
UnexpectedProviderBlock(&'static str, Provider),
/// Carries only the reason: the block holds account keys, SAS tokens and
/// service-account JSON, so no value of it is ever echoed.
#[error("source.{0} is invalid: {1}")]
InvalidProviderBlock(&'static str, &'static str),
#[error("source tls.ca_cert_pem is not a PEM certificate")] #[error("source tls.ca_cert_pem is not a PEM certificate")]
InvalidCaCert, InvalidCaCert,
#[error("filter.{0} must be null or a non-empty string")] #[error("filter.{0} must be null or a non-empty string")]
@@ -511,8 +388,6 @@ impl OnDemandMigrationConfig {
pub fn redacted(&self) -> Self { pub fn redacted(&self) -> Self {
let mut copy = self.clone(); let mut copy = self.clone();
copy.source.credentials = self.source.credentials.as_ref().map(SourceCredentials::redacted); copy.source.credentials = self.source.credentials.as_ref().map(SourceCredentials::redacted);
copy.source.azure = self.source.azure.as_ref().map(AzureSourceConfig::redacted);
copy.source.gcs = self.source.gcs.as_ref().map(GcsSourceConfig::redacted);
copy copy
} }
@@ -558,12 +433,6 @@ impl SourceConfig {
match (&self.endpoint, self.provider) { match (&self.endpoint, self.provider) {
(Some(endpoint), _) => endpoint.clone(), (Some(endpoint), _) => endpoint.clone(),
(None, Provider::Aws) => format!("https://s3.{}.amazonaws.com", self.region), (None, Provider::Aws) => format!("https://s3.{}.amazonaws.com", self.region),
(None, Provider::Azure) => self
.azure
.as_ref()
.map(|azure| format!("https://{}.{AZURE_BLOB_SUFFIX}", azure.account))
.unwrap_or_default(),
(None, Provider::GcsNative) => GCS_DEFAULT_ENDPOINT.to_string(),
(None, _) => String::new(), (None, _) => String::new(),
} }
} }
@@ -579,8 +448,6 @@ impl SourceConfig {
} }
fn validate(&self) -> Result<(), OnDemandMigrationConfigError> { fn validate(&self) -> Result<(), OnDemandMigrationConfigError> {
self.validate_provider_block()?;
if self.region.is_empty() { if self.region.is_empty() {
return Err(OnDemandMigrationConfigError::EmptyRegion); return Err(OnDemandMigrationConfigError::EmptyRegion);
} }
@@ -599,9 +466,6 @@ impl SourceConfig {
)); ));
} }
} }
// Both native providers derive a fixed endpoint; Azure's is built
// from the account name, already checked by `validate_provider_block`.
None if self.provider.is_native() => {}
None => return Err(OnDemandMigrationConfigError::MissingEndpoint(self.provider)), None => return Err(OnDemandMigrationConfigError::MissingEndpoint(self.provider)),
} }
@@ -632,84 +496,6 @@ impl SourceConfig {
Ok(()) Ok(())
} }
/// The provider-specific block must be present for exactly its own
/// provider: a stray `azure` block on an `s3` source would otherwise be
/// accepted, stored, and silently ignored by the client builder.
fn validate_provider_block(&self) -> Result<(), OnDemandMigrationConfigError> {
let missing = OnDemandMigrationConfigError::MissingProviderBlock;
let unexpected = OnDemandMigrationConfigError::UnexpectedProviderBlock;
let invalid = OnDemandMigrationConfigError::InvalidProviderBlock;
if self.provider != Provider::Azure && self.azure.is_some() {
return Err(unexpected("azure", self.provider));
}
if self.provider != Provider::GcsNative && self.gcs.is_some() {
return Err(unexpected("gcs", self.provider));
}
match self.provider {
Provider::Azure => {
let azure = self.azure.as_ref().ok_or(missing("azure", self.provider))?;
if azure.account.is_empty() {
return Err(invalid("azure", "account must not be empty"));
}
// The account feeds a hostname when the endpoint is derived:
// keep it to label characters so it cannot rewrite the host.
if !azure.account.bytes().all(|b| b.is_ascii_alphanumeric() || b == b'-') {
return Err(invalid("azure", "account contains characters outside [A-Za-z0-9-]"));
}
match (azure.account_key.as_deref(), azure.sas_token.as_deref()) {
(Some(_), Some(_)) => return Err(invalid("azure", "account_key and sas_token are mutually exclusive")),
(None, None) => return Err(invalid("azure", "one of account_key and sas_token is required")),
(Some(key), None) => {
if key.is_empty() {
return Err(invalid("azure", "account_key must not be empty"));
}
// Decoded here so a mistyped key fails at the admin
// boundary instead of on the first source request.
if base64_simd::STANDARD.decode_to_vec(key.as_bytes()).is_err() {
return Err(invalid("azure", "account_key is not base64"));
}
}
(None, Some(sas)) => {
if sas.is_empty() {
return Err(invalid("azure", "sas_token must not be empty"));
}
if sas.starts_with('?') {
return Err(invalid("azure", "sas_token must not start with '?'"));
}
if sas.chars().any(char::is_whitespace) {
return Err(invalid("azure", "sas_token must not contain whitespace"));
}
}
}
}
Provider::GcsNative => {
let gcs = self.gcs.as_ref().ok_or(missing("gcs", self.provider))?;
let key: serde_json::Value = serde_json::from_str(&gcs.service_account_json)
.map_err(|_| invalid("gcs", "service_account_json is not valid JSON"))?;
let Some(object) = key.as_object() else {
return Err(invalid("gcs", "service_account_json is not a JSON object"));
};
if object.get("type").and_then(serde_json::Value::as_str) != Some("service_account") {
return Err(invalid("gcs", "service_account_json is not a service_account key"));
}
for field in ["client_email", "private_key"] {
if object
.get(field)
.and_then(serde_json::Value::as_str)
.is_none_or(str::is_empty)
{
return Err(invalid("gcs", "service_account_json is missing client_email or private_key"));
}
}
}
Provider::S3 | Provider::Aws | Provider::Minio | Provider::Rustfs | Provider::R2 | Provider::Gcs => {}
}
Ok(())
}
} }
fn validate_endpoint(endpoint: &str) -> Result<(), OnDemandMigrationConfigError> { fn validate_endpoint(endpoint: &str) -> Result<(), OnDemandMigrationConfigError> {
@@ -804,6 +590,16 @@ impl EndpointKey {
} }
} }
/// Signature of the runtime publish hook: called with the bucket name and
/// its parsed config (`None` when absent, cleared, or unreadable) every time
/// the bucket's metadata is installed into or removed from the cache.
pub type ConfigPublishHook = Box<dyn Fn(&str, Option<&OnDemandMigrationConfig>) + Send + Sync>;
/// Registration point for the runtime (`OnDemandMigrationSys`). Until it is
/// set, metadata publishes are no-ops for ODM, so this crate carries no
/// runtime dependency and the config layer stays inert.
pub static ON_DEMAND_MIGRATION_CONFIG_HOOK: OnceLock<ConfigPublishHook> = OnceLock::new();
#[cfg(test)] #[cfg(test)]
mod tests { mod tests {
use super::*; use super::*;
@@ -903,15 +699,7 @@ mod tests {
), ),
( (
"provider enum", "provider enum",
r#"{"source":{"provider":"swift","endpoint":"https://h","region":"r","bucket":"b"}}"#, r#"{"source":{"provider":"azure","endpoint":"https://h","region":"r","bucket":"b"}}"#,
),
(
"azure block",
r#"{"source":{"provider":"azure","region":"auto","bucket":"b","azure":{"account":"acct","account_key":"a2V5","extra":1}}}"#,
),
(
"gcs block",
r#"{"source":{"provider":"gcs_native","region":"auto","bucket":"b","gcs":{"service_account_json":"{}","extra":1}}}"#,
), ),
] { ] {
let err = OnDemandMigrationConfig::from_json(json.as_bytes()).expect_err(label); let err = OnDemandMigrationConfig::from_json(json.as_bytes()).expect_err(label);
@@ -1032,201 +820,9 @@ mod tests {
"{provider}" "{provider}"
); );
} }
// The native providers never sign with a region, so "auto" is the
// honest value to write for them.
for cfg in [azure_cfg(), gcs_native_cfg()] {
assert_eq!(cfg.source.region, "auto");
cfg.validate(empty_ctx())
.unwrap_or_else(|err| panic!("{}: {err}", cfg.source.provider));
}
assert_eq!(sample().source.effective_region(), "us-west-1"); assert_eq!(sample().source.effective_region(), "us-west-1");
} }
const SERVICE_ACCOUNT_JSON: &str = r#"{"type":"service_account","project_id":"p","client_email":"a@b.iam.gserviceaccount.com","private_key":"-----BEGIN PRIVATE KEY-----\nsecret\n-----END PRIVATE KEY-----"}"#;
fn azure_cfg() -> OnDemandMigrationConfig {
let mut cfg = sample();
cfg.source.provider = Provider::Azure;
cfg.source.endpoint = None;
cfg.source.region = "auto".to_string();
cfg.source.credentials = None;
cfg.source.azure = Some(AzureSourceConfig {
account: "legacyaccount".to_string(),
account_key: Some("c2VjcmV0LWtleQ==".to_string()),
sas_token: None,
});
cfg
}
fn gcs_native_cfg() -> OnDemandMigrationConfig {
let mut cfg = sample();
cfg.source.provider = Provider::GcsNative;
cfg.source.endpoint = None;
cfg.source.region = "auto".to_string();
cfg.source.credentials = None;
cfg.source.gcs = Some(GcsSourceConfig {
service_account_json: SERVICE_ACCOUNT_JSON.to_string(),
});
cfg
}
#[test]
fn native_providers_derive_their_endpoint_and_round_trip_on_the_wire() {
let azure = azure_cfg();
assert_eq!(azure.source.effective_endpoint(), "https://legacyaccount.blob.core.windows.net");
let gcs = gcs_native_cfg();
assert_eq!(gcs.source.effective_endpoint(), "https://storage.googleapis.com");
for cfg in [azure_cfg(), gcs_native_cfg()] {
let json = cfg.to_json().expect("config must serialize");
assert_eq!(OnDemandMigrationConfig::from_json(&json).expect("config must parse"), cfg);
}
// The wire labels are part of the admin contract.
assert!(
String::from_utf8(azure_cfg().to_json().expect("json"))
.expect("utf8")
.contains(r#""provider":"azure""#)
);
assert!(
String::from_utf8(gcs_native_cfg().to_json().expect("json"))
.expect("utf8")
.contains(r#""provider":"gcs_native""#)
);
}
#[test]
fn an_explicit_endpoint_overrides_the_derived_native_one() {
// Azurite and fake-gcs-server are addressed this way.
let mut cfg = azure_cfg();
cfg.source.endpoint = Some("http://azurite.example.com:10000".to_string());
cfg.validate(empty_ctx()).expect("an explicit native endpoint is allowed");
assert_eq!(cfg.source.effective_endpoint(), "http://azurite.example.com:10000");
cfg.source.endpoint = Some("http://azurite.example.com:10000/devstoreaccount1".to_string());
assert!(
matches!(cfg.validate(empty_ctx()), Err(OnDemandMigrationConfigError::InvalidEndpoint(_))),
"a native endpoint is still an origin"
);
}
#[test]
fn a_provider_block_belongs_to_exactly_its_own_provider() {
let mut cfg = sample();
cfg.source.azure = azure_cfg().source.azure;
assert_eq!(
cfg.validate(empty_ctx()),
Err(OnDemandMigrationConfigError::UnexpectedProviderBlock("azure", Provider::S3))
);
let mut cfg = sample();
cfg.source.gcs = gcs_native_cfg().source.gcs;
assert_eq!(
cfg.validate(empty_ctx()),
Err(OnDemandMigrationConfigError::UnexpectedProviderBlock("gcs", Provider::S3))
);
let mut cfg = azure_cfg();
cfg.source.azure = None;
assert_eq!(
cfg.validate(empty_ctx()),
Err(OnDemandMigrationConfigError::MissingProviderBlock("azure", Provider::Azure))
);
let mut cfg = gcs_native_cfg();
cfg.source.gcs = None;
assert_eq!(
cfg.validate(empty_ctx()),
Err(OnDemandMigrationConfigError::MissingProviderBlock("gcs", Provider::GcsNative))
);
}
#[test]
fn azure_block_rules() {
let with = |account: &str, key: Option<&str>, sas: Option<&str>| {
let mut cfg = azure_cfg();
cfg.source.azure = Some(AzureSourceConfig {
account: account.to_string(),
account_key: key.map(str::to_string),
sas_token: sas.map(str::to_string),
});
cfg.validate(empty_ctx())
};
with("legacyaccount", None, Some("sv=2021-08-06&sig=abc%3D")).expect("a SAS token is a complete credential");
with("legacyaccount", Some("c2VjcmV0LWtleQ=="), None).expect("an account key is a complete credential");
for (label, result) in [
("empty account", with("", Some("c2VjcmV0LWtleQ=="), None)),
// The account becomes the first label of the derived hostname.
("account with a dot", with("legacy.account", Some("c2VjcmV0LWtleQ=="), None)),
("account with a slash", with("legacy/account", Some("c2VjcmV0LWtleQ=="), None)),
("no credential", with("legacyaccount", None, None)),
("both credentials", with("legacyaccount", Some("c2VjcmV0LWtleQ=="), Some("sv=1"))),
("empty key", with("legacyaccount", Some(""), None)),
("key that is not base64", with("legacyaccount", Some("not base64!"), None)),
("empty sas", with("legacyaccount", None, Some(""))),
("sas with a leading question mark", with("legacyaccount", None, Some("?sv=1"))),
("sas with whitespace", with("legacyaccount", None, Some("sv=1 &sig=a"))),
] {
assert!(
matches!(result, Err(OnDemandMigrationConfigError::InvalidProviderBlock("azure", _))),
"{label}: {result:?}"
);
}
}
#[test]
fn gcs_native_block_requires_a_usable_service_account_key() {
let with = |json: &str| {
let mut cfg = gcs_native_cfg();
cfg.source.gcs = Some(GcsSourceConfig {
service_account_json: json.to_string(),
});
cfg.validate(empty_ctx())
};
with(SERVICE_ACCOUNT_JSON).expect("a service-account key is accepted");
for (label, json) in [
("empty", ""),
("not json", "not json"),
("not an object", "[]"),
("wrong type", r#"{"type":"authorized_user","client_email":"a@b","private_key":"k"}"#),
("no private key", r#"{"type":"service_account","client_email":"a@b"}"#),
("empty client email", r#"{"type":"service_account","client_email":"","private_key":"k"}"#),
] {
let result = with(json);
assert!(
matches!(result, Err(OnDemandMigrationConfigError::InvalidProviderBlock("gcs", _))),
"{label}: {result:?}"
);
}
}
#[test]
fn native_secrets_never_survive_redaction_or_debug() {
let mut azure = azure_cfg();
azure.source.azure.as_mut().expect("block").sas_token = Some("sv=2021-08-06&sig=top-secret".to_string());
azure.source.azure.as_mut().expect("block").account_key = None;
let gcs = gcs_native_cfg();
for rendered in [
format!("{:?}", azure.redacted()),
format!("{azure:?}"),
String::from_utf8(azure.redacted().to_json().expect("json")).expect("utf8"),
] {
assert!(!rendered.contains("top-secret"), "{rendered}");
assert!(rendered.contains("legacyaccount"), "the account name is not a secret: {rendered}");
}
for rendered in [
format!("{:?}", gcs.redacted()),
format!("{gcs:?}"),
String::from_utf8(gcs.redacted().to_json().expect("json")).expect("utf8"),
] {
assert!(!rendered.contains("PRIVATE KEY-----"), "{rendered}");
assert!(!rendered.contains("gserviceaccount"), "{rendered}");
}
}
#[test] #[test]
fn bucket_rules() { fn bucket_rules() {
let mut cfg = sample(); let mut cfg = sample();
@@ -1509,55 +1105,4 @@ mod tests {
assert!(!rendered.contains("topsecret"), "{rendered}"); assert!(!rendered.contains("topsecret"), "{rendered}");
assert!(!rendered.contains("SK"), "{rendered}"); assert!(!rendered.contains("SK"), "{rendered}");
} }
/// rustfs/backlog#2148: the accessor reports absence as `Ok(None)` and a
/// stored payload it cannot parse as a typed error, never as a default
/// and never as `ConfigNotFound`.
#[tokio::test]
async fn get_on_demand_migration_config_distinguishes_absent_from_corrupt() {
use super::super::storage_api::StorageError as Error;
use super::super::storage_api::test_support::{
BUCKET_ON_DEMAND_MIGRATION_CONFIG, BucketMetadata, BucketMetadataSys, isolated_store_over_temp_disks,
};
use std::sync::Arc;
const ODM_JSON: &[u8] = br#"{"source":{"provider":"minio","endpoint":"https://legacy.example.com:9000","region":"auto","bucket":"legacy-bucket","credentials":{"access_key":"AK","secret_key":"SK"}}}"#;
let (_dirs, ecstore) = isolated_store_over_temp_disks().await;
let sys = BucketMetadataSys::new(ecstore);
let bucket = "odm-accessor";
sys.set(bucket.to_string(), Arc::new(BucketMetadata::new(bucket))).await;
assert_eq!(
decode_stored_config(sys.get_on_demand_migration_config(bucket).await.unwrap()).unwrap(),
None
);
let mut corrupt = BucketMetadata::new(bucket);
corrupt.on_demand_migration_config_json = br#"{"source":{"provider":"s3"},"bogus":1}"#.to_vec();
sys.set(bucket.to_string(), Arc::new(corrupt)).await;
let err = decode_stored_config(sys.get_on_demand_migration_config(bucket).await.unwrap())
.expect_err("corrupt config must not read as a default");
assert_ne!(err, Error::ConfigNotFound, "corruption must not be reported as absence");
let typed = match &err {
Error::Io(io) => io
.get_ref()
.and_then(|source| source.downcast_ref::<OnDemandMigrationConfigError>()),
_ => None,
};
assert!(
matches!(typed, Some(OnDemandMigrationConfigError::Malformed(_))),
"typed parse error must survive the Result boundary, got: {err:?}"
);
let mut valid = BucketMetadata::new(bucket);
valid
.update_config(BUCKET_ON_DEMAND_MIGRATION_CONFIG, ODM_JSON.to_vec())
.unwrap();
let stamped = valid.on_demand_migration_config_updated_at;
sys.set(bucket.to_string(), Arc::new(valid)).await;
let (config, updated_at) = decode_stored_config(sys.get_on_demand_migration_config(bucket).await.unwrap())
.unwrap()
.expect("stored config is returned");
assert_eq!(config, OnDemandMigrationConfig::from_json(ODM_JSON).unwrap());
assert_eq!(updated_at, stamped);
}
} }
@@ -25,21 +25,13 @@ use parking_lot::Mutex;
use serde::{Deserialize, Serialize}; use serde::{Deserialize, Serialize};
use std::time::{Duration, Instant}; use std::time::{Duration, Instant};
/// The continuation-token version used by ordinary progressing pages. /// The only continuation-token envelope version this build reads and writes.
pub const LIST_THROUGH_TOKEN_VERSION: u32 = 1; pub const LIST_THROUGH_TOKEN_VERSION: u32 = 1;
const LIST_THROUGH_PROGRESS_TOKEN_VERSION: u32 = 2;
/// The sixteenth consecutive merged page without a key or new EOF fails.
/// This also bounds legitimate sparse listings; it is not a cycle detector.
pub const MAX_LIST_NO_PROGRESS_PAGES: u8 = 16;
/// Envelope marker. A bucket that is *not* merging hands out the local /// Envelope marker. A bucket that is *not* merging hands out the local
/// listing's own marker, so the decoder needs a positive signal before it /// listing's own marker, so the decoder needs a positive signal before it
/// treats an opaque token as a merged one. /// treats an opaque token as a merged one.
const LIST_THROUGH_TOKEN_TAG: &str = "odm-list"; const LIST_THROUGH_TOKEN_TAG: &str = "odm-list";
// Object keys cannot contain NUL (bucket::utils::is_valid_object_prefix),
// so this framing cannot collide with a local key used as an opaque marker.
const LIST_THROUGH_TOKEN_PREFIX: &str = "\0odm-list:";
/// Pages fetched per side per request: the first page, plus at most one refill /// Pages fetched per side per request: the first page, plus at most one refill
/// when the first one was mostly consumed by the previous page. Two pages of /// when the first one was mostly consumed by the previous page. Two pages of
@@ -94,7 +86,8 @@ pub struct MergePick {
} }
/// The continuation-token envelope. Opaque to clients: it is serialized as /// The continuation-token envelope. Opaque to clients: it is serialized as
/// framed JSON and then base64-encoded by the same helper as a local marker. /// JSON and then base64-encoded by the same helper that encodes a plain local
/// marker, so the wire shape is `base64(json)`.
/// ///
/// A `null` cursor with `done = false` means "list that side from the start"; /// A `null` cursor with `done = false` means "list that side from the start";
/// `done = true` means the side is finished and must not be listed again. /// `done = true` means the side is finished and must not be listed again.
@@ -118,10 +111,6 @@ pub struct ListThroughToken {
/// common prefix compares as itself, never as its members. /// common prefix compares as itself, never as its members.
#[serde(default)] #[serde(default)]
pub last_key: Option<String>, pub last_key: Option<String>,
/// Consecutive empty truncated merged pages, present only in v2 tokens.
/// Ordinary v1 tokens retain their original serialized shape.
#[serde(default, skip_serializing_if = "Option::is_none")]
pub no_progress: Option<u8>,
} }
impl ListThroughToken { impl ListThroughToken {
@@ -134,14 +123,13 @@ impl ListThroughToken {
source: source.token, source: source.token,
source_done: source.done, source_done: source.done,
last_key, last_key,
no_progress: None,
} }
} }
pub fn encode(&self) -> String { pub fn encode(&self) -> String {
// The envelope is built here from owned strings, so serialization // The envelope is built here from owned strings, so serialization
// cannot fail; the fallback keeps the signature infallible. // cannot fail; the fallback keeps the signature infallible.
format!("{LIST_THROUGH_TOKEN_PREFIX}{}", serde_json::to_string(self).unwrap_or_default()) serde_json::to_string(self).unwrap_or_default()
} }
} }
@@ -165,35 +153,24 @@ pub enum ListThroughTokenError {
/// Classifies an already base64-decoded continuation token. /// Classifies an already base64-decoded continuation token.
/// ///
/// Only a framed JSON object is read as a merged token; /// Only a JSON object carrying the envelope marker is read as a merged token;
/// anything else is a local marker, so a bucket that turns `list_through` off /// anything else is a local marker, so a bucket that turns `list_through` off
/// keeps paginating with the tokens it handed out. A token that *is* an /// keeps paginating with the tokens it handed out. A token that *is* an
/// envelope but was tampered with (unknown version, unknown field, truncated /// envelope but was tampered with (unknown version, unknown field, truncated
/// JSON) is an error, never a silent fallback. /// JSON) is an error, never a silent fallback.
pub fn decode_continuation_token(decoded: &str) -> Result<ListThroughCursor, ListThroughTokenError> { pub fn decode_continuation_token(decoded: &str) -> Result<ListThroughCursor, ListThroughTokenError> {
let Some(payload) = decoded.strip_prefix(LIST_THROUGH_TOKEN_PREFIX) else { if !decoded.starts_with('{') {
return Ok(ListThroughCursor::Local(decoded.to_string()));
}
let Ok(value) = serde_json::from_str::<serde_json::Value>(decoded) else {
// Not JSON at all: an object key may legitimately start with '{'.
return Ok(ListThroughCursor::Local(decoded.to_string())); return Ok(ListThroughCursor::Local(decoded.to_string()));
}; };
let value = serde_json::from_str::<serde_json::Value>(payload).map_err(|_| ListThroughTokenError::Malformed)?;
if value.get("t").and_then(serde_json::Value::as_str) != Some(LIST_THROUGH_TOKEN_TAG) { if value.get("t").and_then(serde_json::Value::as_str) != Some(LIST_THROUGH_TOKEN_TAG) {
return Err(ListThroughTokenError::Malformed); return Ok(ListThroughCursor::Local(decoded.to_string()));
} }
match value.get("v").and_then(serde_json::Value::as_u64) { match value.get("v").and_then(serde_json::Value::as_u64) {
Some(version) if version == u64::from(LIST_THROUGH_TOKEN_VERSION) => { Some(version) if version == u64::from(LIST_THROUGH_TOKEN_VERSION) => {}
// v1 readers reject this field even when it is null or zero.
if value.get("no_progress").is_some() {
return Err(ListThroughTokenError::Malformed);
}
}
Some(version) if version == u64::from(LIST_THROUGH_PROGRESS_TOKEN_VERSION) => {
if !value
.get("no_progress")
.and_then(serde_json::Value::as_u64)
.is_some_and(|count| (1..u64::from(MAX_LIST_NO_PROGRESS_PAGES)).contains(&count))
{
return Err(ListThroughTokenError::Malformed);
}
}
Some(version) => return Err(ListThroughTokenError::UnsupportedVersion(version.min(u64::from(u32::MAX)) as u32)), Some(version) => return Err(ListThroughTokenError::UnsupportedVersion(version.min(u64::from(u32::MAX)) as u32)),
None => return Err(ListThroughTokenError::Malformed), None => return Err(ListThroughTokenError::Malformed),
} }
@@ -212,8 +189,8 @@ pub enum SourceListPlan {
/// delimiter — the source's own roll-up boundary matches the request's. /// delimiter — the source's own roll-up boundary matches the request's.
Page { prefix: String }, Page { prefix: String },
/// `filter.prefix` reaches past a delimiter, so every key the source could /// `filter.prefix` reaches past a delimiter, so every key the source could
/// contribute rolls into this one common prefix. Bounded probes follow /// contribute rolls into this one common prefix. One bounded probe listing
/// empty progressing pages until a key proves existence or the source ends. /// decides whether it exists; there is nothing to paginate.
Folded { probe_prefix: String, common_prefix: String }, Folded { probe_prefix: String, common_prefix: String },
} }
@@ -302,31 +279,6 @@ pub struct FetchRequest {
pub token: Option<String>, pub token: Option<String>,
} }
/// Invalid pagination metadata. Opaque cursor values are never included in errors.
#[derive(Clone, Copy, Debug, PartialEq, Eq, thiserror::Error)]
pub enum ListPageError {
#[error("truncated listing has no continuation token")]
Missing,
#[error("truncated listing has an empty continuation token")]
Empty,
#[error("truncated listing repeats a continuation token")]
Repeated,
#[error("listing exhausted its consecutive no-progress page budget")]
NoProgress(MergeSide),
}
pub(crate) fn validate_list_page(is_truncated: bool, token: Option<&str>, next_token: Option<&str>) -> Result<(), ListPageError> {
if is_truncated {
match next_token {
None => return Err(ListPageError::Missing),
Some("") => return Err(ListPageError::Empty),
Some(next) if Some(next) == token => return Err(ListPageError::Repeated),
Some(_) => {}
}
}
Ok(())
}
#[derive(Debug, Default)] #[derive(Debug, Default)]
struct SideState { struct SideState {
start: SideCursor, start: SideCursor,
@@ -377,7 +329,6 @@ pub struct MergeOutcome {
#[derive(Debug)] #[derive(Debug)]
pub struct ListThroughMerger { pub struct ListThroughMerger {
max_keys: usize, max_keys: usize,
no_progress: Option<u8>,
last_key: Option<String>, last_key: Option<String>,
local: SideState, local: SideState,
source: SideState, source: SideState,
@@ -397,7 +348,6 @@ impl ListThroughMerger {
}; };
Self { Self {
max_keys, max_keys,
no_progress: token.and_then(|token| token.no_progress),
last_key, last_key,
local, local,
source, source,
@@ -414,11 +364,6 @@ impl ListThroughMerger {
/// or `filter.prefix` excludes it. /// or `filter.prefix` excludes it.
pub fn disable_source(&mut self) { pub fn disable_source(&mut self) {
self.source.disabled = true; self.source.disabled = true;
// A refill can fail after a valid first page. A local-only response
// must discard both that source payload and its ordering horizon.
self.source.entries.clear();
self.source.pages.clear();
self.source.more = false;
} }
pub fn next_fetch(&self) -> Option<FetchRequest> { pub fn next_fetch(&self) -> Option<FetchRequest> {
@@ -433,13 +378,7 @@ impl ListThroughMerger {
/// Records one fetched page. `entries` must be sorted by `name` and already /// Records one fetched page. `entries` must be sorted by `name` and already
/// filtered with [`Self::accepts`]; the caller keeps the matching payloads /// filtered with [`Self::accepts`]; the caller keeps the matching payloads
/// in the same order. /// in the same order.
pub fn push_page( pub fn push_page(&mut self, side: MergeSide, entries: Vec<ListEntryKey>, is_truncated: bool, next_token: Option<String>) {
&mut self,
side: MergeSide,
entries: Vec<ListEntryKey>,
is_truncated: bool,
next_token: Option<String>,
) -> Result<(), ListPageError> {
let state = match side { let state = match side {
MergeSide::Local => &mut self.local, MergeSide::Local => &mut self.local,
MergeSide::Source => &mut self.source, MergeSide::Source => &mut self.source,
@@ -448,33 +387,24 @@ impl ListThroughMerger {
Some(last) => last.next_token.clone(), Some(last) => last.next_token.clone(),
None => state.start.token.clone(), None => state.start.token.clone(),
}; };
validate_list_page(is_truncated, token.as_deref(), next_token.as_deref())?; // A truncated page without a cursor cannot be continued; treating the
// Also reject a cycle through an earlier page in this bounded fetch. // side as finished is the only alternative to looping on it forever.
if is_truncated && state.pages.iter().any(|page| page.token == next_token) { state.more = is_truncated && next_token.is_some();
return Err(ListPageError::Repeated);
}
state.more = is_truncated;
state.pages.push(FetchedPage { state.pages.push(FetchedPage {
token, token,
count: entries.len(), count: entries.len(),
next_token: is_truncated.then_some(next_token).flatten(), next_token: is_truncated.then_some(next_token).flatten(),
}); });
state.entries.extend(entries); state.entries.extend(entries);
Ok(())
} }
/// `issue_progress_tokens` allows a v1 chain to start carrying a budget. pub fn finish(self) -> MergeOutcome {
/// An existing v2 budget is always enforced, including on reader-only nodes.
/// Borrowing lets a source failure re-merge the fetched local buffers.
pub fn finish(&self, issue_progress_tokens: bool) -> Result<MergeOutcome, ListPageError> {
let Self { let Self {
max_keys, max_keys,
no_progress,
last_key, last_key,
local, local,
source, source,
} = self; } = self;
let max_keys = *max_keys;
// A side with more pages behind it can only be trusted up to the last // A side with more pages behind it can only be trusted up to the last
// key it handed over: past that horizon the other side's entries could // key it handed over: past that horizon the other side's entries could
@@ -540,44 +470,12 @@ impl ListThroughMerger {
let source_left = !source.disabled && (!source_cursor.done || consumed_source < source.entries.len()); let source_left = !source.disabled && (!source_cursor.done || consumed_source < source.entries.len());
let is_truncated = local_left || source_left; let is_truncated = local_left || source_left;
let reached_eof = (!local.start.done && local_cursor.done) || (!source.start.done && source_cursor.done); let last_key = consumed_key.or(last_key);
let next_no_progress = if !is_truncated || !picks.is_empty() || reached_eof { MergeOutcome {
None
} else if max_keys == 0 {
// A zero-sized request cannot consume entries. Preserve an existing
// budget without spending it or starting a new one.
*no_progress
} else if issue_progress_tokens || no_progress.is_some() {
let count = no_progress.unwrap_or(0).saturating_add(1);
if count >= MAX_LIST_NO_PROGRESS_PAGES {
// An empty truncated side closes the merge horizon. Local
// failure takes precedence; disabling the source cannot fix it.
let side = if local.more && local.entries.is_empty() {
MergeSide::Local
} else if !source.disabled && source.more && source.entries.is_empty() {
MergeSide::Source
} else {
MergeSide::Local
};
return Err(ListPageError::NoProgress(side));
}
Some(count)
} else {
None
};
let last_key = consumed_key.or_else(|| last_key.clone());
Ok(MergeOutcome {
picks, picks,
is_truncated, is_truncated,
next_token: is_truncated.then(|| { next_token: is_truncated.then(|| ListThroughToken::new(local_cursor, source_cursor, last_key)),
let mut token = ListThroughToken::new(local_cursor, source_cursor, last_key); }
if let Some(count) = next_no_progress {
token.v = LIST_THROUGH_PROGRESS_TOKEN_VERSION;
token.no_progress = Some(count);
}
token
}),
})
} }
} }
@@ -701,15 +599,9 @@ mod tests {
let (entries, truncated, next) = reference_page(keys, prefix, delimiter, fetch.token.as_deref(), max_keys); let (entries, truncated, next) = reference_page(keys, prefix, delimiter, fetch.token.as_deref(), max_keys);
let kept: Vec<ListEntryKey> = entries.into_iter().filter(|entry| merger.accepts(&entry.name)).collect(); let kept: Vec<ListEntryKey> = entries.into_iter().filter(|entry| merger.accepts(&entry.name)).collect();
buffers[usize::from(fetch.side == MergeSide::Source)].extend(kept.iter().cloned()); buffers[usize::from(fetch.side == MergeSide::Source)].extend(kept.iter().cloned());
merger merger.push_page(fetch.side, kept, truncated, next);
.push_page(fetch.side, kept, truncated, next)
.expect("reference provider pages must advance");
}
let outcome = merger.finish(false).expect("valid merge outcome");
assert_eq!(outcome.is_truncated, outcome.next_token.is_some());
if outcome.is_truncated {
assert_ne!(outcome.next_token, token, "every truncated merged page must make progress");
} }
let outcome = merger.finish();
page_sizes.push(outcome.picks.len()); page_sizes.push(outcome.picks.len());
for pick in &outcome.picks { for pick in &outcome.picks {
let entry = buffers[usize::from(pick.side == MergeSide::Source)][pick.index].clone(); let entry = buffers[usize::from(pick.side == MergeSide::Source)][pick.index].clone();
@@ -724,25 +616,11 @@ mod tests {
} }
fn expected(local: &[String], source: &[String], prefix: &str, delimiter: Option<&str>) -> Vec<ListEntryKey> { fn expected(local: &[String], source: &[String], prefix: &str, delimiter: Option<&str>) -> Vec<ListEntryKey> {
// This oracle builds the complete namespace independently of the let mut all: Vec<String> = local.iter().chain(source.iter()).cloned().collect();
// provider's page/marker helper and the production merger. all.sort();
let mut namespace = std::collections::BTreeMap::new(); all.dedup();
for key in local.iter().chain(source) { let (entries, _, _) = reference_page(&all, prefix, delimiter, None, usize::MAX);
let Some(suffix) = key.strip_prefix(prefix) else { entries
continue;
};
if let Some(delimiter) = delimiter.filter(|delimiter| !delimiter.is_empty())
&& let Some((directory, _)) = suffix.split_once(delimiter)
{
namespace.insert(format!("{prefix}{directory}{delimiter}"), true);
continue;
}
namespace.insert(key.clone(), false);
}
namespace
.into_iter()
.map(|(name, is_prefix)| ListEntryKey { name, is_prefix })
.collect()
} }
#[test] #[test]
@@ -784,11 +662,9 @@ mod tests {
token: None token: None
}) })
); );
merger merger.push_page(MergeSide::Local, vec![ListEntryKey::object("a")], false, None);
.push_page(MergeSide::Local, vec![ListEntryKey::object("a")], false, None)
.expect("local EOF is valid");
assert_eq!(merger.next_fetch(), None); assert_eq!(merger.next_fetch(), None);
let outcome = merger.finish(false).expect("valid merge outcome"); let outcome = merger.finish();
assert_eq!(outcome.picks.len(), 1); assert_eq!(outcome.picks.len(), 1);
assert!(!outcome.is_truncated); assert!(!outcome.is_truncated);
assert!(outcome.next_token.is_none()); assert!(outcome.next_token.is_none());
@@ -804,19 +680,16 @@ mod tests {
source: Some("source-1".to_string()), source: Some("source-1".to_string()),
source_done: false, source_done: false,
last_key: Some("a".to_string()), last_key: Some("a".to_string()),
no_progress: None,
}; };
let mut merger = ListThroughMerger::new(1, Some(&resume)); let mut merger = ListThroughMerger::new(1, Some(&resume));
merger.disable_source(); merger.disable_source();
merger merger.push_page(
.push_page( MergeSide::Local,
MergeSide::Local, vec![ListEntryKey::object("b"), ListEntryKey::object("c")],
vec![ListEntryKey::object("b"), ListEntryKey::object("c")], true,
true, Some("local-2".to_string()),
Some("local-2".to_string()), );
) let outcome = merger.finish();
.expect("local cursor advances");
let outcome = merger.finish(false).expect("valid merge outcome");
assert!(outcome.is_truncated); assert!(outcome.is_truncated);
let token = outcome.next_token.expect("truncated page carries a token"); let token = outcome.next_token.expect("truncated page carries a token");
assert_eq!(token.source.as_deref(), Some("source-1"), "the source cursor must not move"); assert_eq!(token.source.as_deref(), Some("source-1"), "the source cursor must not move");
@@ -825,212 +698,6 @@ mod tests {
assert_eq!(token.local.as_deref(), Some("local-1"), "a partly read page is re-listed"); assert_eq!(token.local.as_deref(), Some("local-1"), "a partly read page is re-listed");
} }
#[test]
fn truncated_pages_require_a_nonempty_advancing_cursor() {
for side in [MergeSide::Local, MergeSide::Source] {
for entries in [vec![], vec![ListEntryKey::object("a")]] {
for (next, expected) in [
(None, Err(ListPageError::Missing)),
(Some(""), Err(ListPageError::Empty)),
(Some("stuck"), Err(ListPageError::Repeated)),
(Some("advances"), Ok(())),
] {
let resume = ListThroughToken::new(
SideCursor {
token: Some("stuck".into()),
done: false,
},
SideCursor {
token: Some("stuck".into()),
done: false,
},
None,
);
let mut merger = ListThroughMerger::new(2, Some(&resume));
let result = merger.push_page(side, entries.clone(), true, next.map(str::to_string));
assert_eq!(result, expected, "{side:?}, {entries:?}, {next:?}");
let state = if side == MergeSide::Local {
&merger.local
} else {
&merger.source
};
assert_eq!(state.pages.len(), usize::from(result.is_ok()), "invalid page must not be accepted");
}
}
}
}
#[test]
fn repeated_empty_cursor_is_rejected_before_an_identical_page_can_escape() {
let resume = ListThroughToken::new(
SideCursor { token: None, done: true },
SideCursor {
token: Some("stuck".into()),
done: false,
},
None,
);
let mut merger = ListThroughMerger::new(2, Some(&resume));
assert_eq!(
merger.next_fetch(),
Some(FetchRequest {
side: MergeSide::Source,
token: Some("stuck".into())
})
);
assert_eq!(
merger.push_page(MergeSide::Source, vec![], true, Some("stuck".into())),
Err(ListPageError::Repeated)
);
}
#[test]
fn empty_pages_may_advance_within_the_fetch_budget_until_eof() {
let mut merger = ListThroughMerger::new(2, None);
merger.push_page(MergeSide::Local, vec![], false, None).expect("local EOF");
for next in ["opaque-z", "opaque-a"] {
assert_eq!(merger.next_fetch().expect("bounded source fetch").side, MergeSide::Source);
merger
.push_page(MergeSide::Source, vec![], true, Some(next.into()))
.expect("opaque cursor advances regardless of sort order");
}
assert!(merger.next_fetch().is_none(), "two source fetches exhaust the request budget");
let outcome = merger.finish(false).expect("valid merge outcome");
assert!(outcome.picks.is_empty());
assert!(outcome.is_truncated);
let token = outcome.next_token.expect("empty progressing page has a cursor");
assert_eq!(token.source.as_deref(), Some("opaque-a"));
let mut merger = ListThroughMerger::new(2, Some(&token));
assert_eq!(merger.next_fetch().expect("source resumes").token.as_deref(), Some("opaque-a"));
merger
.push_page(MergeSide::Source, vec![ListEntryKey::object("result")], false, None)
.expect("source EOF");
let outcome = merger.finish(false).expect("valid merge outcome");
assert_eq!(
outcome.picks,
vec![MergePick {
side: MergeSide::Source,
index: 0
}]
);
assert!(!outcome.is_truncated);
assert!(outcome.next_token.is_none());
}
#[test]
fn a_cursor_cycle_inside_the_fetch_budget_is_rejected() {
let resume = ListThroughToken::new(
SideCursor { token: None, done: true },
SideCursor {
token: Some("first".into()),
done: false,
},
None,
);
let mut merger = ListThroughMerger::new(2, Some(&resume));
merger
.push_page(MergeSide::Source, vec![], true, Some("second".into()))
.expect("first page advances");
assert_eq!(
merger.push_page(MergeSide::Source, vec![], true, Some("first".into())),
Err(ListPageError::Repeated)
);
}
#[test]
fn source_refill_failure_discards_buffered_source_entries_and_horizon() {
let mut merger = ListThroughMerger::new(2, None);
merger
.push_page(MergeSide::Local, vec![ListEntryKey::object("z")], false, None)
.expect("local EOF");
merger
.push_page(MergeSide::Source, vec![ListEntryKey::object("a")], true, Some("stuck".into()))
.expect("first source page advances");
assert_eq!(merger.next_fetch().expect("source refill is required").token.as_deref(), Some("stuck"));
assert_eq!(
merger.push_page(MergeSide::Source, vec![], true, Some("stuck".into())),
Err(ListPageError::Repeated)
);
merger.disable_source();
let outcome = merger.finish(false).expect("valid merge outcome");
assert_eq!(
outcome.picks,
vec![MergePick {
side: MergeSide::Local,
index: 0
}]
);
assert!(!outcome.is_truncated);
assert!(outcome.next_token.is_none());
}
#[test]
fn list_through_static_namespace_boundary_matrix() {
let corpus = [
"a",
"a/",
"a/b",
"a/b/child",
"a0",
"b",
"b/leaf",
"quote\"&<",
"space key",
"z",
"é",
"中/文",
];
for count in [0, 1, 3, 4, corpus.len()] {
let keys: Vec<String> = corpus[..count].iter().map(|key| (*key).to_string()).collect();
for placement in 0..3 {
let (local, source): (Vec<_>, Vec<_>) =
keys.iter()
.enumerate()
.fold((vec![], vec![]), |(mut local, mut source), (index, key)| {
if placement != 1 || index % 2 == 0 {
local.push(key.clone());
}
if placement != 0 || index % 2 == 0 {
source.push(key.clone());
}
(local, source)
});
for prefix in ["", "a", "a/", "中/"] {
for delimiter in [None, Some("/")] {
for max_keys in [1, 3, 4] {
let oracle = expected(&local, &source, prefix, delimiter);
let (emitted, sizes) = walk(&local, &source, prefix, delimiter, max_keys);
assert_eq!(
emitted.iter().map(|(entry, _)| entry.clone()).collect::<Vec<_>>(),
oracle,
"count={count}, placement={placement}, prefix={prefix}, delimiter={delimiter:?}, max={max_keys}"
);
let expected_sizes: Vec<_> = if oracle.is_empty() {
vec![0]
} else {
oracle.chunks(max_keys).map(<[ListEntryKey]>::len).collect()
};
assert_eq!(sizes, expected_sizes, "exact max and max+1 boundaries must agree");
}
}
}
}
}
}
#[test]
fn list_through_large_overlap_walk_keeps_all_5300_keys() {
let source: Vec<_> = (0..5000).map(|index| format!("k{index:05}")).collect();
let local: Vec<_> = (4800..5300).map(|index| format!("k{index:05}")).collect();
let (emitted, sizes) = walk(&local, &source, "", None, 333);
assert_eq!(emitted.len(), 5300);
for (index, (entry, side)) in emitted.iter().enumerate() {
assert_eq!(entry.name, format!("k{index:05}"));
assert_eq!(*side, if index >= 4800 { MergeSide::Local } else { MergeSide::Source });
}
assert_eq!(sizes, [vec![333; 15], vec![305]].concat());
}
#[test] #[test]
fn token_round_trips_and_rejects_tampering() { fn token_round_trips_and_rejects_tampering() {
let token = ListThroughToken::new( let token = ListThroughToken::new(
@@ -1044,287 +711,21 @@ mod tests {
let encoded = token.encode(); let encoded = token.encode();
assert_eq!(decode_continuation_token(&encoded), Ok(ListThroughCursor::Merged(Box::new(token)))); assert_eq!(decode_continuation_token(&encoded), Ok(ListThroughCursor::Merged(Box::new(token))));
let bumped = encoded.replace("\"v\":1", "\"v\":3"); let bumped = encoded.replace("\"v\":1", "\"v\":2");
assert_eq!(decode_continuation_token(&bumped), Err(ListThroughTokenError::UnsupportedVersion(3))); assert_eq!(decode_continuation_token(&bumped), Err(ListThroughTokenError::UnsupportedVersion(2)));
let extra = encoded.replace("{", "{\"x\":1,"); let extra = encoded.replace("{", "{\"x\":1,");
assert_eq!(decode_continuation_token(&extra), Err(ListThroughTokenError::Malformed)); assert_eq!(decode_continuation_token(&extra), Err(ListThroughTokenError::Malformed));
let truncated = &encoded[..encoded.len() - 3]; let truncated = &encoded[..encoded.len() - 3];
assert_eq!(decode_continuation_token(truncated), Err(ListThroughTokenError::Malformed)); assert_eq!(decode_continuation_token(truncated), Ok(ListThroughCursor::Local(truncated.to_string())));
let no_version = "\0odm-list:{\"t\":\"odm-list\"}"; let no_version = "{\"t\":\"odm-list\"}";
assert_eq!(decode_continuation_token(no_version), Err(ListThroughTokenError::Malformed)); assert_eq!(decode_continuation_token(no_version), Err(ListThroughTokenError::Malformed));
} }
fn progress_token(count: Option<u8>, local_done: bool, source_done: bool) -> ListThroughToken {
let mut token = ListThroughToken::new(
SideCursor {
token: None,
done: local_done,
},
SideCursor {
token: Some("A".into()),
done: source_done,
},
Some("last-key".into()),
);
if let Some(count) = count {
token.v = LIST_THROUGH_PROGRESS_TOKEN_VERSION;
token.no_progress = Some(count);
}
token
}
fn push_empty_pages(merger: &mut ListThroughMerger, side: MergeSide) {
for _ in 0..MAX_LIST_FETCHES_PER_SIDE {
let fetch = merger.next_fetch().expect("empty truncated side must be fetched");
assert_eq!(fetch.side, side);
let next = format!("{}:next", fetch.token.unwrap_or_default());
merger
.push_page(side, vec![], true, Some(next))
.expect("opaque cursor advances");
}
}
#[test]
fn progress_tokens_preserve_v1_bytes_and_validate_v2_counts() {
fn framed(payload: &str) -> String {
format!("{LIST_THROUGH_TOKEN_PREFIX}{payload}")
}
let token = progress_token(None, true, false);
assert_eq!(
token.encode(),
concat!(
"\0odm-list:",
r#"{"t":"odm-list","v":1,"local":null,"local_done":true,"source":"A","source_done":false,"last_key":"last-key"}"#
)
);
for count in 1..MAX_LIST_NO_PROGRESS_PAGES {
let token = progress_token(Some(count), true, false);
assert_eq!(decode_continuation_token(&token.encode()), Ok(ListThroughCursor::Merged(Box::new(token))));
}
for version in [1, 2] {
for value in ["null", "0", "16", "-1", "1.5", "256", "18446744073709551616", "\"1\""] {
let encoded = framed(&format!(r#"{{"t":"odm-list","v":{version},"no_progress":{value}}}"#));
assert_eq!(decode_continuation_token(&encoded), Err(ListThroughTokenError::Malformed), "{encoded}");
}
}
for payload in [
r#"{"t":"odm-list","v":1,"no_progress":1}"#,
r#"{"t":"odm-list","v":2}"#,
r#"{"t":"odm-list","v":2,"no_progress":1,"extra":true}"#,
] {
let encoded = framed(payload);
assert_eq!(decode_continuation_token(&encoded), Err(ListThroughTokenError::Malformed), "{encoded}");
}
}
#[test]
fn reader_only_nodes_do_not_start_a_budget_but_mixed_readers_preserve_one() {
let mut token = progress_token(None, true, false);
for _ in 0..MAX_LIST_NO_PROGRESS_PAGES {
let mut merger = ListThroughMerger::new(2, Some(&token));
push_empty_pages(&mut merger, MergeSide::Source);
token = merger
.finish(false)
.expect("reader-only v1 behavior")
.next_token
.expect("truncated cursor");
assert_eq!(token.v, 1);
assert_eq!(token.no_progress, None);
}
for count in 1..=MAX_LIST_NO_PROGRESS_PAGES {
let mut merger = ListThroughMerger::new(2, Some(&token));
push_empty_pages(&mut merger, MergeSide::Source);
assert!(merger.next_fetch().is_none(), "the per-request two-fetch limit stays intact");
let outcome = merger.finish(count % 2 == 1);
if count == MAX_LIST_NO_PROGRESS_PAGES {
assert_eq!(outcome, Err(ListPageError::NoProgress(MergeSide::Source)));
break;
}
token = outcome.expect("budget not exhausted").next_token.expect("truncated cursor");
assert_eq!(token.no_progress, Some(count));
let ListThroughCursor::Merged(decoded) = decode_continuation_token(&token.encode()).expect("round-trip v2") else {
panic!("merged cursor expected");
};
token = *decoded;
}
}
#[test]
fn objects_and_common_prefixes_reset_a_budget_at_the_boundary() {
for entry in [ListEntryKey::object("result"), ListEntryKey::prefix("result/")] {
for issue_tokens in [false, true] {
let resume = progress_token(Some(MAX_LIST_NO_PROGRESS_PAGES - 1), true, false);
let mut merger = ListThroughMerger::new(2, Some(&resume));
merger
.push_page(MergeSide::Source, vec![], true, Some("B".into()))
.expect("empty advancing page");
merger
.push_page(MergeSide::Source, vec![entry.clone()], true, Some("C".into()))
.expect("real progress");
let outcome = merger
.finish(issue_tokens)
.expect("real progress does not exhaust the budget");
assert_eq!(
outcome.picks,
vec![MergePick {
side: MergeSide::Source,
index: 0
}]
);
let next = outcome.next_token.expect("source remains truncated");
assert_eq!(next.last_key.as_deref(), Some(entry.name.as_str()));
assert_eq!(next.v, 1);
assert_eq!(next.no_progress, None);
assert!(!next.encode().contains("no_progress"));
}
}
}
#[test]
fn only_a_new_eof_transition_resets_the_empty_page_budget() {
for finished_side in [MergeSide::Local, MergeSide::Source] {
let resume = progress_token(Some(MAX_LIST_NO_PROGRESS_PAGES - 1), false, false);
let mut merger = ListThroughMerger::new(2, Some(&resume));
if finished_side == MergeSide::Local {
merger
.push_page(MergeSide::Local, vec![], false, None)
.expect("new local EOF");
push_empty_pages(&mut merger, MergeSide::Source);
} else {
push_empty_pages(&mut merger, MergeSide::Local);
merger
.push_page(MergeSide::Source, vec![], false, None)
.expect("new source EOF");
}
let next = merger
.finish(false)
.expect("new EOF is progress")
.next_token
.expect("other side truncated");
assert_eq!(next.no_progress, None);
assert_eq!(next.v, 1);
assert_eq!(next.local_done, finished_side == MergeSide::Local);
assert_eq!(next.source_done, finished_side == MergeSide::Source);
let mut merger = ListThroughMerger::new(2, Some(&next));
let remaining = if finished_side == MergeSide::Local {
MergeSide::Source
} else {
MergeSide::Local
};
push_empty_pages(&mut merger, remaining);
let next = merger
.finish(true)
.expect("a new budget starts")
.next_token
.expect("truncated");
assert_eq!(next.no_progress, Some(1), "an already-done side cannot reset every page");
}
let resume = progress_token(Some(MAX_LIST_NO_PROGRESS_PAGES - 1), true, false);
let mut merger = ListThroughMerger::new(2, Some(&resume));
merger.push_page(MergeSide::Source, vec![], false, None).expect("final EOF");
let outcome = merger.finish(false).expect("EOF succeeds at the budget boundary");
assert!(!outcome.is_truncated);
assert!(outcome.next_token.is_none());
}
#[test]
fn filtered_duplicates_cannot_reset_the_no_progress_budget() {
let resume = progress_token(Some(MAX_LIST_NO_PROGRESS_PAGES - 1), true, false);
let mut merger = ListThroughMerger::new(2, Some(&resume));
for next in ["B", "C"] {
let entries = [ListEntryKey::object("last-key"), ListEntryKey::object("earlier")]
.into_iter()
.filter(|entry| merger.accepts(&entry.name))
.collect::<Vec<_>>();
assert!(entries.is_empty(), "both provider entries were already consumed");
merger
.push_page(MergeSide::Source, entries, true, Some(next.into()))
.expect("advancing cursor");
}
assert_eq!(merger.finish(false), Err(ListPageError::NoProgress(MergeSide::Source)));
}
#[test]
fn no_progress_is_attributed_to_local_when_source_cannot_unblock_it() {
for source_mode in ["disabled", "done", "empty", "data"] {
let resume = progress_token(Some(MAX_LIST_NO_PROGRESS_PAGES - 1), false, source_mode == "done");
let mut merger = ListThroughMerger::new(2, Some(&resume));
if source_mode == "disabled" {
merger.disable_source();
}
push_empty_pages(&mut merger, MergeSide::Local);
match source_mode {
"empty" => push_empty_pages(&mut merger, MergeSide::Source),
"data" => merger
.push_page(MergeSide::Source, vec![ListEntryKey::object("source")], false, None)
.expect("source data"),
_ => {}
}
assert_eq!(merger.finish(false), Err(ListPageError::NoProgress(MergeSide::Local)), "{source_mode}");
}
}
#[test]
fn source_budget_failure_remerges_local_objects_and_prefixes_without_refetching() {
let resume = progress_token(Some(MAX_LIST_NO_PROGRESS_PAGES - 1), false, false);
let mut merger = ListThroughMerger::new(2, Some(&resume));
merger
.push_page(MergeSide::Local, vec![ListEntryKey::object("local")], true, Some("L1".into()))
.expect("local object");
merger
.push_page(MergeSide::Local, vec![ListEntryKey::prefix("prefix/")], true, Some("L2".into()))
.expect("local prefix");
push_empty_pages(&mut merger, MergeSide::Source);
assert_eq!(merger.finish(false), Err(ListPageError::NoProgress(MergeSide::Source)));
merger.disable_source();
assert!(merger.next_fetch().is_none(), "fallback does not perform another fetch");
let outcome = merger.finish(false).expect("local data makes progress");
assert_eq!(
outcome.picks,
vec![
MergePick {
side: MergeSide::Local,
index: 0
},
MergePick {
side: MergeSide::Local,
index: 1
}
]
);
let token = outcome.next_token.expect("remaining local page");
assert_eq!(token.local.as_deref(), Some("L2"));
assert_eq!(token.source.as_deref(), Some("A"));
assert_eq!(token.last_key.as_deref(), Some("prefix/"));
assert_eq!(token.no_progress, None);
assert_eq!(token.v, 1);
}
#[test]
fn a_zero_sized_merge_preserves_an_existing_budget() {
let resume = progress_token(Some(MAX_LIST_NO_PROGRESS_PAGES - 1), true, false);
let mut merger = ListThroughMerger::new(0, Some(&resume));
merger
.push_page(MergeSide::Source, vec![ListEntryKey::object("result")], true, Some("B".into()))
.expect("source page");
let outcome = merger.finish(false).expect("a zero-sized request cannot consume entries");
assert!(outcome.picks.is_empty());
assert_eq!(outcome.next_token.expect("unconsumed source").no_progress, resume.no_progress);
}
#[test] #[test]
fn a_plain_local_marker_stays_local() { fn a_plain_local_marker_stays_local() {
for marker in [
r#"{"t":"odm-list","v":1}"#,
r#"{"t":"odm-list","v":2,"local_done":true}"#,
r#"{"t":"odm-list"}"#,
] {
assert_eq!(decode_continuation_token(marker), Ok(ListThroughCursor::Local(marker.to_string())));
}
assert_eq!( assert_eq!(
decode_continuation_token("photos/2024/01.jpg"), decode_continuation_token("photos/2024/01.jpg"),
Ok(ListThroughCursor::Local("photos/2024/01.jpg".to_string())) Ok(ListThroughCursor::Local("photos/2024/01.jpg".to_string()))
@@ -1395,10 +796,7 @@ mod tests {
} }
proptest! { proptest! {
#![proptest_config(ProptestConfig { #![proptest_config(ProptestConfig::with_cases(256))]
rng_seed: proptest::test_runner::RngSeed::Fixed(0xec5706),
..ProptestConfig::with_cases(256)
})]
/// Full pagination of a merged listing equals the sorted, deduplicated /// Full pagination of a merged listing equals the sorted, deduplicated
/// union of both sides, with every shared key served by local, and no /// union of both sides, with every shared key served by local, and no
@@ -19,45 +19,30 @@
//! client, and the per-node runtime (`sys`) that turns configs into live //! client, and the per-node runtime (`sys`) that turns configs into live
//! clients guarded by a breaker, a negative cache, singleflight and a pull //! clients guarded by a breaker, a negative cache, singleflight and a pull
//! concurrency limit (rustfs/backlog#2147). //! concurrency limit (rustfs/backlog#2147).
//!
//! A source is reached through one `SourceBackend`: the S3 dialect for every
//! S3-compatible provider, and a native backend for the providers that have no
//! S3 API (`azure`, `gcs_native`).
pub mod azure;
#[cfg(test)]
mod backend_contract;
pub mod backfill; pub mod backfill;
pub mod breaker; pub mod breaker;
pub mod config; pub mod config;
#[cfg(feature = "gcs")]
pub mod gcs;
pub mod list_through; pub mod list_through;
mod metrics;
mod native_http;
pub mod negative_cache; pub mod negative_cache;
pub mod pull; pub mod pull;
pub mod source_client; pub mod source_client;
pub mod stats; pub mod stats;
mod storage_api;
pub mod sys; pub mod sys;
#[cfg(test)]
mod test_http_fixture;
pub use breaker::{ pub use breaker::{
BREAKER_FAILURE_THRESHOLD, BREAKER_FAILURE_WINDOW, BREAKER_HALF_OPEN_MAX_PROBES, BREAKER_OPEN_DURATION, Breaker, BREAKER_FAILURE_THRESHOLD, BREAKER_FAILURE_WINDOW, BREAKER_HALF_OPEN_MAX_PROBES, BREAKER_OPEN_DURATION, Breaker,
BreakerState, BreakerTransition, BreakerVerdict, BreakerState, BreakerTransition, BreakerVerdict,
}; };
pub use config::{ pub use config::{
AzureSourceConfig, FilterConfig, GcsSourceConfig, HeadPolicy, ON_DEMAND_MIGRATION_CONFIG_VERSION, OnDemandMigrationConfig, ConfigPublishHook, FilterConfig, HeadPolicy, ON_DEMAND_MIGRATION_CONFIG_HOOK, ON_DEMAND_MIGRATION_CONFIG_VERSION,
OnDemandMigrationConfigError, PathStyle, PolicyConfig, Provider, RangeGetPolicy, SourceConfig, SourceCredentials, OnDemandMigrationConfig, OnDemandMigrationConfigError, PathStyle, PolicyConfig, Provider, RangeGetPolicy, SourceConfig,
SourceErrorPolicy, SourceTimeout, TlsConfig, ValidationContext, SourceCredentials, SourceErrorPolicy, SourceTimeout, TlsConfig, ValidationContext,
}; };
pub use list_through::{ pub use list_through::{
FetchRequest, LIST_THROUGH_TOKEN_VERSION, ListEntryKey, ListPageError, ListThroughCursor, ListThroughMerger, FetchRequest, LIST_THROUGH_TOKEN_VERSION, ListEntryKey, ListThroughCursor, ListThroughMerger, ListThroughToken,
ListThroughToken, ListThroughTokenError, MAX_LIST_FETCHES_PER_SIDE, MAX_LIST_NO_PROGRESS_PAGES, MergeOutcome, MergePick, ListThroughTokenError, MAX_LIST_FETCHES_PER_SIDE, MergeOutcome, MergePick, MergeSide, SOURCE_LIST_MAX_RATE_WAIT,
MergeSide, SOURCE_LIST_MAX_RATE_WAIT, SOURCE_LIST_RATE_PER_SEC, SourceListPlan, SourceListRateLimiter, SOURCE_LIST_RATE_PER_SEC, SourceListPlan, SourceListRateLimiter, decode_continuation_token, source_list_plan,
decode_continuation_token, source_list_plan,
}; };
pub use negative_cache::{NEGATIVE_CACHE_MAX_ENTRIES, NegativeCache}; pub use negative_cache::{NEGATIVE_CACHE_MAX_ENTRIES, NegativeCache};
pub use pull::{ pub use pull::{
@@ -65,19 +50,11 @@ pub use pull::{
PullQueue, PullReason, PullSource, QueuedPullOutcome, SourceBody, SourceIdleGuard, WriteBackBody, WriteBackError, PullQueue, PullReason, PullSource, QueuedPullOutcome, SourceBody, SourceIdleGuard, WriteBackBody, WriteBackError,
WriteBackOutcome, WriteBackPart, WriteBackRequest, commit_inline, commit_inline_with, idle_guarded_body, WriteBackOutcome, WriteBackPart, WriteBackRequest, commit_inline, commit_inline_with, idle_guarded_body,
}; };
pub use source_client::{
SourceClient, SourceError, SourceGet, SourceHead, SourceListRequest, SourceObject, SourcePage, SourceSse, is_multipart_etag,
};
pub use stats::{ pub use stats::{
GaugeGuard, LastSourceError, LatencyBucketSnapshot, OdmOp, OdmOutcome, OdmStats, OdmStatsSnapshot, PullFailureReason, GaugeGuard, LastSourceError, LatencyBucketSnapshot, OdmOp, OdmOutcome, OdmStats, OdmStatsSnapshot, PullFailureReason,
PullPath, SOURCE_LATENCY_BUCKET_BOUNDS_MS, SourceLatencySnapshot, PullPath, SOURCE_LATENCY_BUCKET_BOUNDS_MS, SourceLatencySnapshot,
}; };
pub use sys::{ pub use sys::{
ApplyOutcome, BucketOdmState, GLOBAL_ON_DEMAND_MIGRATION_SYS, OdmBucketSnapshot, OdmLookup, OdmStateError, ApplyOutcome, BucketOdmState, GLOBAL_ON_DEMAND_MIGRATION_SYS, OdmBucketSnapshot, OdmLookup, OdmStateError,
OnDemandMigrationSys, PullError, PullFollower, PullLeader, PullOutcome, PullResult, PullSlot, source_backend_spec, OnDemandMigrationSys, PullError, PullFollower, PullLeader, PullOutcome, PullResult, PullSlot, source_client_spec,
source_client_spec,
}; };
pub(crate) fn register_metrics() {
metrics::register();
}
@@ -46,10 +46,10 @@ use super::stats::{PullFailureReason, PullPath};
use super::sys::{BucketOdmState, OnDemandMigrationSys, PullError, PullOutcome, PullSlot}; use super::sys::{BucketOdmState, OnDemandMigrationSys, PullError, PullOutcome, PullSlot};
use async_trait::async_trait; use async_trait::async_trait;
use bytes::Bytes; use bytes::Bytes;
use futures::{FutureExt, Stream, StreamExt, future::Shared}; use futures::{Stream, StreamExt};
use parking_lot::Mutex; use parking_lot::Mutex;
use rand::RngExt; use rand::RngExt;
use std::collections::HashMap; use std::collections::{HashMap, HashSet};
use std::fmt; use std::fmt;
use std::io; use std::io;
use std::pin::Pin; use std::pin::Pin;
@@ -133,8 +133,6 @@ pub enum QueuedPullOutcome {
Failed(PullError), Failed(PullError),
} }
pub type QueuedPullReport = Shared<oneshot::Receiver<QueuedPullOutcome>>;
/// Result of [`PullQueue::enqueue`]. /// Result of [`PullQueue::enqueue`].
#[derive(Clone, Copy, Debug, PartialEq, Eq, Hash)] #[derive(Clone, Copy, Debug, PartialEq, Eq, Hash)]
pub enum EnqueueOutcome { pub enum EnqueueOutcome {
@@ -243,8 +241,6 @@ impl PullSource for SourceClient {
#[derive(Clone, Debug)] #[derive(Clone, Debug)]
pub struct WriteBackRequest { pub struct WriteBackRequest {
pub bucket: String, pub bucket: String,
/// Identity captured with the source configuration, retained through cleanup.
pub bucket_incarnation_id: uuid::Uuid,
pub key: String, pub key: String,
/// Source HEAD/GET of the whole object. /// Source HEAD/GET of the whole object.
pub head: SourceHead, pub head: SourceHead,
@@ -255,7 +251,6 @@ pub struct WriteBackRequest {
pub preserve_etag: bool, pub preserve_etag: bool,
/// `policy.emit_events`. /// `policy.emit_events`.
pub emit_events: bool, pub emit_events: bool,
pub respect_delete_marker: bool,
/// Source tags to copy (`policy.copy_tags`), `None` to skip. /// Source tags to copy (`policy.copy_tags`), `None` to skip.
pub tags: Option<HashMap<String, String>>, pub tags: Option<HashMap<String, String>>,
} }
@@ -265,14 +260,12 @@ impl WriteBackRequest {
let config = state.config(); let config = state.config();
Self { Self {
bucket: state.bucket().to_string(), bucket: state.bucket().to_string(),
bucket_incarnation_id: state.incarnation_id(),
key: key.to_string(), key: key.to_string(),
head, head,
source_label: format!("{}:{}", config.source.provider.as_str(), config.source.bucket), source_label: format!("{}:{}", config.source.provider.as_str(), config.source.bucket),
pulled_at: OffsetDateTime::now_utc(), pulled_at: OffsetDateTime::now_utc(),
preserve_etag: config.policy.preserve_etag, preserve_etag: config.policy.preserve_etag,
emit_events: config.policy.emit_events, emit_events: config.policy.emit_events,
respect_delete_marker: config.policy.respect_local_delete_marker,
tags, tags,
} }
} }
@@ -360,7 +353,7 @@ pub trait OdmWriteBack: Send + Sync {
parts: Vec<WriteBackPart>, parts: Vec<WriteBackPart>,
) -> Result<WriteBackOutcome, WriteBackError>; ) -> Result<WriteBackOutcome, WriteBackError>;
async fn abort_multipart_upload(&self, request: &WriteBackRequest, upload_id: &str) -> Result<(), WriteBackError>; async fn abort_multipart_upload(&self, bucket: &str, key: &str, upload_id: &str) -> Result<(), WriteBackError>;
} }
/// Why the pump stopped feeding the write-back before EOF. /// Why the pump stopped feeding the write-back before EOF.
@@ -665,7 +658,9 @@ async fn write_multipart(
Err(err) => Err(err), Err(err) => Err(err),
}; };
if completed.is_err() if completed.is_err()
&& let Err(abort_err) = write_back.abort_multipart_upload(request, &upload_id).await && let Err(abort_err) = write_back
.abort_multipart_upload(&request.bucket, &request.key, &upload_id)
.await
{ {
debug!( debug!(
event = EVENT_ODM_PULL_FAILED, event = EVENT_ODM_PULL_FAILED,
@@ -835,7 +830,7 @@ pub struct PullQueue {
bucket: String, bucket: String,
tx: mpsc::Sender<PullJob>, tx: mpsc::Sender<PullJob>,
/// Keys queued or running; the job removes its key when it ends. /// Keys queued or running; the job removes its key when it ends.
pending: Mutex<HashMap<String, QueuedPullReport>>, pending: Mutex<HashSet<String>>,
capacity: usize, capacity: usize,
cancel: CancellationToken, cancel: CancellationToken,
stats: Arc<super::stats::OdmStats>, stats: Arc<super::stats::OdmStats>,
@@ -874,7 +869,7 @@ impl PullQueue {
let queue = Arc::new(Self { let queue = Arc::new(Self {
bucket: state.bucket().to_string(), bucket: state.bucket().to_string(),
tx, tx,
pending: Mutex::new(HashMap::new()), pending: Mutex::new(HashSet::new()),
capacity, capacity,
cancel: state.cancel_token(), cancel: state.cancel_token(),
stats: Arc::clone(state.stats()), stats: Arc::clone(state.stats()),
@@ -908,24 +903,29 @@ impl PullQueue {
self.enqueue_with_report(key, reason).0 self.enqueue_with_report(key, reason).0
} }
/// [`Self::enqueue`] with a shared report, including for coalesced pulls. /// [`Self::enqueue`] that also hands back the job's report channel when
pub fn enqueue_with_report(&self, key: &str, reason: PullReason) -> (EnqueueOutcome, Option<QueuedPullReport>) { /// a new job was queued (`Coalesced` pulls report to their first
/// requester only).
pub fn enqueue_with_report(
&self,
key: &str,
reason: PullReason,
) -> (EnqueueOutcome, Option<oneshot::Receiver<QueuedPullOutcome>>) {
if self.cancel.is_cancelled() { if self.cancel.is_cancelled() {
return (EnqueueOutcome::Unavailable, None); return (EnqueueOutcome::Unavailable, None);
} }
let mut pending = self.pending.lock(); let mut pending = self.pending.lock();
if let Some(report) = pending.get(key) { if pending.contains(key) {
return (EnqueueOutcome::Coalesced, Some(report.clone())); return (EnqueueOutcome::Coalesced, None);
} }
let (report_tx, report_rx) = oneshot::channel(); let (report_tx, report_rx) = oneshot::channel();
let report_rx = report_rx.shared();
match self.tx.try_send(PullJob { match self.tx.try_send(PullJob {
key: key.to_string(), key: key.to_string(),
reason, reason,
report: Some(report_tx), report: Some(report_tx),
}) { }) {
Ok(()) => { Ok(()) => {
pending.insert(key.to_string(), report_rx.clone()); pending.insert(key.to_string());
(EnqueueOutcome::Enqueued, Some(report_rx)) (EnqueueOutcome::Enqueued, Some(report_rx))
} }
Err(TrySendError::Full(_)) => { Err(TrySendError::Full(_)) => {
@@ -1072,7 +1072,7 @@ impl BucketOdmState {
self: &Arc<Self>, self: &Arc<Self>,
key: &str, key: &str,
reason: PullReason, reason: PullReason,
) -> (EnqueueOutcome, Option<QueuedPullReport>) { ) -> (EnqueueOutcome, Option<oneshot::Receiver<QueuedPullOutcome>>) {
match self.pull_queue() { match self.pull_queue() {
Some(queue) => queue.enqueue_with_report(key, reason), Some(queue) => queue.enqueue_with_report(key, reason),
None => (EnqueueOutcome::Unavailable, None), None => (EnqueueOutcome::Unavailable, None),
@@ -1094,7 +1094,7 @@ impl OnDemandMigrationSys {
#[cfg(test)] #[cfg(test)]
mod tests { mod tests {
use super::*; use super::*;
use crate::on_demand_migration::config::{ use crate::bucket::on_demand_migration::config::{
FilterConfig, OnDemandMigrationConfig, PathStyle as ConfigPathStyle, PolicyConfig, Provider, SourceConfig, FilterConfig, OnDemandMigrationConfig, PathStyle as ConfigPathStyle, PolicyConfig, Provider, SourceConfig,
SourceCredentials, TlsConfig, SourceCredentials, TlsConfig,
}; };
@@ -1119,8 +1119,6 @@ mod tests {
session_token: None, session_token: None,
}), }),
tls: TlsConfig::default(), tls: TlsConfig::default(),
azure: None,
gcs: None,
}, },
filter: FilterConfig::default(), filter: FilterConfig::default(),
policy: PolicyConfig::default(), policy: PolicyConfig::default(),
@@ -1341,7 +1339,7 @@ mod tests {
}) })
} }
async fn abort_multipart_upload(&self, _request: &WriteBackRequest, upload_id: &str) -> Result<(), WriteBackError> { async fn abort_multipart_upload(&self, _bucket: &str, _key: &str, upload_id: &str) -> Result<(), WriteBackError> {
self.aborted.lock().push(upload_id.to_string()); self.aborted.lock().push(upload_id.to_string());
Ok(()) Ok(())
} }
@@ -1401,21 +1399,13 @@ mod tests {
assert_eq!(queue.capacity(), 1024); assert_eq!(queue.capacity(), 1024);
let mut outcomes = HashMap::new(); let mut outcomes = HashMap::new();
let mut shared_report = None;
for _ in 0..100 { for _ in 0..100 {
let (outcome, report) = queue.enqueue_with_report("a", PullReason::RangeGet); *outcomes.entry(queue.enqueue("a", PullReason::RangeGet)).or_insert(0) += 1;
*outcomes.entry(outcome).or_insert(0) += 1;
shared_report = report;
} }
assert_eq!(outcomes.get(&EnqueueOutcome::Enqueued), Some(&1)); assert_eq!(outcomes.get(&EnqueueOutcome::Enqueued), Some(&1));
assert_eq!(outcomes.get(&EnqueueOutcome::Coalesced), Some(&99)); assert_eq!(outcomes.get(&EnqueueOutcome::Coalesced), Some(&99));
assert_eq!(queue.pending_keys(), 1); assert_eq!(queue.pending_keys(), 1);
assert_eq!(
shared_report.expect("coalesced report").await,
Ok(QueuedPullOutcome::Stored { size: 1000 })
);
wait_until("first pull to finish", || queue.pending_keys() == 0).await; wait_until("first pull to finish", || queue.pending_keys() == 0).await;
assert_eq!(source.head_calls.load(Ordering::SeqCst), 1); assert_eq!(source.head_calls.load(Ordering::SeqCst), 1);
assert_eq!(source.get_calls.load(Ordering::SeqCst), 1); assert_eq!(source.get_calls.load(Ordering::SeqCst), 1);
@@ -1448,23 +1438,6 @@ mod tests {
assert_eq!(queue.enqueue("a", PullReason::RangeGet), EnqueueOutcome::Unavailable); assert_eq!(queue.enqueue("a", PullReason::RangeGet), EnqueueOutcome::Unavailable);
} }
#[tokio::test]
async fn coalesced_enqueues_share_failure_reports() {
let sys = OnDemandMigrationSys::new();
let state = enabled_state(&sys, &config()).await;
let source = MockSource::with_object("missing", 1000, BodyKind::Bytes(body_bytes(1000)));
let queue = PullQueue::start(Arc::clone(&state), source, Arc::new(MockWriteBack::default()));
let (first, first_report) = queue.enqueue_with_report("absent", PullReason::RangeGet);
let (second, second_report) = queue.enqueue_with_report("absent", PullReason::Backfill);
assert_eq!(first, EnqueueOutcome::Enqueued);
assert_eq!(second, EnqueueOutcome::Coalesced);
let (first, second) = tokio::join!(first_report.expect("leader report"), second_report.expect("coalesced report"));
assert_eq!(first, second);
assert!(matches!(first, Ok(QueuedPullOutcome::Failed(_))));
sys.remove(BUCKET);
queue.wait_until_stopped().await;
}
#[tokio::test] #[tokio::test]
async fn queue_full_is_reported_and_cancel_drains_without_leaking_tasks() { async fn queue_full_is_reported_and_cancel_drains_without_leaking_tasks() {
let sys = OnDemandMigrationSys::new(); let sys = OnDemandMigrationSys::new();
@@ -1494,23 +1467,16 @@ mod tests {
wait_until("dispatcher to wait for a slot", || state.stats().queue_depth() == 1).await; wait_until("dispatcher to wait for a slot", || state.stats().queue_depth() == 1).await;
assert_eq!(queue.enqueue("c", PullReason::LargeObject), EnqueueOutcome::Enqueued); assert_eq!(queue.enqueue("c", PullReason::LargeObject), EnqueueOutcome::Enqueued);
assert_eq!(queue.enqueue("d", PullReason::LargeObject), EnqueueOutcome::QueueFull); assert_eq!(queue.enqueue("d", PullReason::LargeObject), EnqueueOutcome::QueueFull);
let (coalesced, canceled_report) = queue.enqueue_with_report("c", PullReason::LargeObject); assert_eq!(queue.enqueue("c", PullReason::LargeObject), EnqueueOutcome::Coalesced);
assert_eq!(coalesced, EnqueueOutcome::Coalesced);
assert_eq!(queue.pending_keys(), 3); assert_eq!(queue.pending_keys(), 3);
assert_eq!(failures(&state).get("queue_full"), Some(&1)); assert_eq!(failures(&state).get("queue_full"), Some(&1));
assert!(!queue.is_stopped()); assert!(!queue.is_stopped());
assert_eq!(sys.remove(BUCKET), crate::on_demand_migration::ApplyOutcome::Removed); assert_eq!(sys.remove(BUCKET), crate::bucket::on_demand_migration::ApplyOutcome::Removed);
tokio::time::timeout(Duration::from_secs(5), queue.wait_until_stopped()) tokio::time::timeout(Duration::from_secs(5), queue.wait_until_stopped())
.await .await
.expect("dispatcher and in-flight job must exit after cancel"); .expect("dispatcher and in-flight job must exit after cancel");
assert!(queue.is_stopped()); assert!(queue.is_stopped());
assert!(
tokio::time::timeout(Duration::from_secs(5), canceled_report.expect("coalesced cancellation report"))
.await
.expect("cancellation closes the report")
.is_err()
);
assert_eq!(queue.pending_keys(), 0); assert_eq!(queue.pending_keys(), 0);
assert_eq!(state.inflight_keys(), 0); assert_eq!(state.inflight_keys(), 0);
assert_eq!(state.stats().inflight_pulls(), 0); assert_eq!(state.stats().inflight_pulls(), 0);
@@ -25,14 +25,10 @@
//! Client-supplied `If-*`, `Authorization`, `Host` and SSE-C headers are never //! Client-supplied `If-*`, `Authorization`, `Host` and SSE-C headers are never
//! forwarded: v1 rejects SSE-C source objects outright. //! forwarded: v1 rejects SSE-C source objects outright.
use super::azure::AzureSourceBackend; use crate::bucket::remote_s3_client::{
#[cfg(feature = "gcs")]
use super::gcs::GcsNativeSourceBackend;
use super::list_through::{ListPageError, validate_list_page};
use super::storage_api::HTTPRangeSpec;
use super::storage_api::remote_s3_client::{
PathStyle, RemoteCredentials, RemoteS3ClientError, RemoteS3EndpointSpec, RemoteS3RetryPolicy, build_remote_s3_config, PathStyle, RemoteCredentials, RemoteS3ClientError, RemoteS3EndpointSpec, RemoteS3RetryPolicy, build_remote_s3_config,
}; };
use crate::storage_api_contracts::range::HTTPRangeSpec;
use aws_sdk_s3::Client as S3Client; use aws_sdk_s3::Client as S3Client;
use aws_sdk_s3::error::{ProvideErrorMetadata, SdkError}; use aws_sdk_s3::error::{ProvideErrorMetadata, SdkError};
use aws_sdk_s3::operation::get_object::GetObjectOutput; use aws_sdk_s3::operation::get_object::GetObjectOutput;
@@ -68,10 +64,6 @@ pub enum SourceProvider {
/// Generic S3-compatible service. /// Generic S3-compatible service.
#[default] #[default]
S3, S3,
/// Native Azure Blob service; not an S3 dialect.
Azure,
/// Native GCS JSON API with a service-account key; not an S3 dialect.
GcsNative,
} }
impl SourceProvider { impl SourceProvider {
@@ -83,8 +75,6 @@ impl SourceProvider {
"minio" => Some(Self::Minio), "minio" => Some(Self::Minio),
"rustfs" => Some(Self::Rustfs), "rustfs" => Some(Self::Rustfs),
"s3" => Some(Self::S3), "s3" => Some(Self::S3),
"azure" => Some(Self::Azure),
"gcs_native" => Some(Self::GcsNative),
_ => None, _ => None,
} }
} }
@@ -97,8 +87,6 @@ impl SourceProvider {
Self::Minio => "minio", Self::Minio => "minio",
Self::Rustfs => "rustfs", Self::Rustfs => "rustfs",
Self::S3 => "s3", Self::S3 => "s3",
Self::Azure => "azure",
Self::GcsNative => "gcs_native",
} }
} }
@@ -164,75 +152,12 @@ pub struct SourceClientSpec {
/// Wire requests one logical source call may cost. The pull pipeline and /// Wire requests one logical source call may cost. The pull pipeline and
/// the backfill job own the retry budget (`pull.rs` `PULL_MAX_RETRIES`, /// the backfill job own the retry budget (`pull.rs` `PULL_MAX_RETRIES`,
/// `backfill.rs` `LIST_MAX_RETRIES`) and the breaker counts logical calls, /// `backfill.rs` `LIST_MAX_RETRIES`) and the breaker counts logical calls,
/// so ODM declares [`RemoteS3RetryPolicy::Disabled`]. An ambiguous HEAD /// so ODM declares [`RemoteS3RetryPolicy::Disabled`] and keeps one counted
/// 404 additionally probes the bucket before declaring a key absent. /// failure equal to one request against a struggling source.
pub retry: RemoteS3RetryPolicy, pub retry: RemoteS3RetryPolicy,
/// Bytes per second the pull pipeline may consume from this source; /// Bytes per second the pull pipeline may consume from this source;
/// `None` means unlimited. Enforced by the consumer, not by this client. /// `None` means unlimited. Enforced by the consumer, not by this client.
pub bandwidth_limit: Option<NonZeroU64>, pub bandwidth_limit: Option<NonZeroU64>,
/// Which [`SourceBackend`] to build. The S3 variant reads `region`,
/// `path_style` and `credentials`; the native variants ignore all three
/// and carry their own credentials.
pub backend: SourceBackendSpec,
}
/// Provider-specific half of [`SourceClientSpec`].
#[derive(Clone, Debug, Default, PartialEq, Eq)]
pub enum SourceBackendSpec {
#[default]
S3,
Azure(AzureSourceSpec),
Gcs(GcsSourceSpec),
}
/// Native Azure Blob parameters. The container is [`SourceClientSpec::bucket`].
#[derive(Clone, PartialEq, Eq)]
pub struct AzureSourceSpec {
pub account: String,
pub auth: AzureAuth,
}
impl fmt::Debug for AzureSourceSpec {
fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result {
f.debug_struct("AzureSourceSpec")
.field("account", &self.account)
.field("auth", &self.auth)
.finish()
}
}
/// How Azure requests are authorized.
#[derive(Clone, PartialEq, Eq)]
pub enum AzureAuth {
/// Base64 storage-account key, signed per request with Shared Key.
SharedKey(String),
/// SAS query string without the leading `?`, appended to every URL.
Sas(String),
}
impl fmt::Debug for AzureAuth {
fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result {
// Both variants are secrets; only the scheme may be rendered.
f.write_str(match self {
Self::SharedKey(_) => "SharedKey(REDACTED)",
Self::Sas(_) => "Sas(REDACTED)",
})
}
}
/// Native GCS parameters. The bucket is [`SourceClientSpec::bucket`].
#[derive(Clone, PartialEq, Eq)]
pub struct GcsSourceSpec {
/// Service-account key JSON.
pub service_account_json: String,
}
impl fmt::Debug for GcsSourceSpec {
fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result {
f.debug_struct("GcsSourceSpec")
.field("service_account_json", &"REDACTED")
.finish()
}
} }
impl SourceClientSpec { impl SourceClientSpec {
@@ -298,8 +223,6 @@ pub enum SourceError {
ServerError(u16), ServerError(u16),
#[error("unsupported source object: {0}")] #[error("unsupported source object: {0}")]
Unsupported(String), Unsupported(String),
#[error("invalid source listing: {0}")]
InvalidPagination(#[from] ListPageError),
#[error("source request failed: {0}")] #[error("source request failed: {0}")]
Other(String), Other(String),
} }
@@ -322,7 +245,6 @@ impl SourceError {
SourceError::Connect(_) => "connect", SourceError::Connect(_) => "connect",
SourceError::ServerError(_) => "server_error", SourceError::ServerError(_) => "server_error",
SourceError::Unsupported(_) => "unsupported", SourceError::Unsupported(_) => "unsupported",
SourceError::InvalidPagination(_) => "invalid_pagination",
SourceError::Other(_) => "other", SourceError::Other(_) => "other",
} }
} }
@@ -335,9 +257,8 @@ const THROTTLE_CODES: &[&str] = &[
"RequestLimitExceeded", "RequestLimitExceeded",
"TooManyRequests", "TooManyRequests",
"RequestThrottled", "RequestThrottled",
"ServerBusy",
]; ];
const NOT_FOUND_CODES: &[&str] = &["NoSuchKey", "BlobNotFound"]; const NOT_FOUND_CODES: &[&str] = &["NoSuchKey", "NotFound", "NoSuchBucket", "NoSuchVersion"];
const ACCESS_DENIED_CODES: &[&str] = &[ const ACCESS_DENIED_CODES: &[&str] = &[
"AccessDenied", "AccessDenied",
"InvalidAccessKeyId", "InvalidAccessKeyId",
@@ -345,10 +266,9 @@ const ACCESS_DENIED_CODES: &[&str] = &[
"AllAccessDisabled", "AllAccessDisabled",
"ExpiredToken", "ExpiredToken",
"InvalidToken", "InvalidToken",
"AuthorizationPermissionMismatch",
]; ];
pub(super) fn classify_status(status: u16, code: Option<&str>, message: String) -> SourceError { fn classify_status(status: u16, code: Option<&str>, message: String) -> SourceError {
if let Some(code) = code { if let Some(code) = code {
if THROTTLE_CODES.contains(&code) { if THROTTLE_CODES.contains(&code) {
return SourceError::Throttled; return SourceError::Throttled;
@@ -361,6 +281,7 @@ pub(super) fn classify_status(status: u16, code: Option<&str>, message: String)
} }
} }
match status { match status {
404 => SourceError::NotFound,
401 | 403 => SourceError::AccessDenied, 401 | 403 => SourceError::AccessDenied,
429 | 503 => SourceError::Throttled, 429 | 503 => SourceError::Throttled,
500..=599 => SourceError::ServerError(status), 500..=599 => SourceError::ServerError(status),
@@ -420,11 +341,6 @@ pub struct SourceHead {
pub storage_class: Option<String>, pub storage_class: Option<String>,
pub sse: Option<SourceSse>, pub sse: Option<SourceSse>,
pub is_multipart_etag: bool, pub is_multipart_etag: bool,
/// The provider's ETag is not derived from the object bytes (Azure
/// stamps an opaque concurrency token). Such an ETag is recorded for
/// provenance but must never be read as a content digest, so the
/// write-back path refuses to use it as the expected MD5.
pub etag_is_opaque: bool,
} }
/// Per-operation fields shared by HEAD and GET outputs. /// Per-operation fields shared by HEAD and GET outputs.
@@ -446,7 +362,7 @@ struct HeadParts {
sse_customer_algorithm: Option<String>, sse_customer_algorithm: Option<String>,
} }
pub(super) fn normalize_etag(etag: Option<String>) -> Option<String> { fn normalize_etag(etag: Option<String>) -> Option<String> {
etag.map(|etag| etag.trim().trim_matches('"').to_string()) etag.map(|etag| etag.trim().trim_matches('"').to_string())
.filter(|etag| !etag.is_empty()) .filter(|etag| !etag.is_empty())
} }
@@ -495,7 +411,6 @@ fn source_head(parts: HeadParts) -> Result<SourceHead, SourceError> {
storage_class: parts.storage_class, storage_class: parts.storage_class,
sse, sse,
is_multipart_etag, is_multipart_etag,
etag_is_opaque: false,
}) })
} }
@@ -706,55 +621,14 @@ impl fmt::Debug for SourceClient {
impl SourceClient { impl SourceClient {
pub async fn new(spec: &SourceClientSpec) -> Result<Self, RemoteS3ClientError> { pub async fn new(spec: &SourceClientSpec) -> Result<Self, RemoteS3ClientError> {
match &spec.backend { let endpoint = spec.endpoint_spec()?;
SourceBackendSpec::S3 => { let config = build_remote_s3_config(&endpoint).await?;
let endpoint = spec.endpoint_spec()?; Ok(Self::from_config_builder(config, endpoint.endpoint_url(), spec))
let config = build_remote_s3_config(&endpoint).await?;
Ok(Self::from_config_builder(config, endpoint.endpoint_url(), spec))
}
SourceBackendSpec::Azure(azure) => {
let backend = AzureSourceBackend::new(
&spec.endpoint,
&spec.bucket,
azure,
spec.timeouts,
spec.skip_tls_verify,
spec.ca_cert_pem.as_deref(),
)?;
Ok(Self::from_backend(Box::new(backend), spec))
}
#[cfg(not(feature = "gcs"))]
SourceBackendSpec::Gcs(_) => Err(RemoteS3ClientError::BackendNotCompiled("gcs_native")),
#[cfg(feature = "gcs")]
SourceBackendSpec::Gcs(gcs) => {
let backend = GcsNativeSourceBackend::new(
&spec.endpoint,
&spec.bucket,
gcs,
spec.timeouts,
spec.skip_tls_verify,
spec.ca_cert_pem.as_deref(),
)?;
Ok(Self::from_backend(Box::new(backend), spec))
}
}
}
/// Wraps a ready backend in the prefix-mapping client. The endpoint is
/// kept only for `Debug` and admin status.
fn from_backend(backend: Box<dyn SourceBackend>, spec: &SourceClientSpec) -> Self {
Self {
backend,
endpoint: spec.endpoint.clone(),
bucket: spec.bucket.clone(),
source_prefix: spec.source_prefix.clone().filter(|prefix| !prefix.is_empty()),
timeouts: spec.timeouts,
bandwidth_limit: spec.bandwidth_limit,
}
} }
/// `config` must come from [`SourceClientSpec::endpoint_spec`], which is /// `config` must come from [`SourceClientSpec::endpoint_spec`], which is
/// where the policy disabling SDK-level retries is declared. /// where the retry policy that keeps one logical call equal to one wire
/// request is declared.
fn from_config_builder(config: aws_sdk_s3::config::Builder, endpoint: String, spec: &SourceClientSpec) -> Self { fn from_config_builder(config: aws_sdk_s3::config::Builder, endpoint: String, spec: &SourceClientSpec) -> Self {
let client = S3Client::from_conf(config.interceptor(SourceProxyMarkerInterceptor::new()).build()); let client = S3Client::from_conf(config.interceptor(SourceProxyMarkerInterceptor::new()).build());
Self { Self {
@@ -840,7 +714,6 @@ impl SourceClient {
..*request ..*request
}) })
.await?; .await?;
validate_list_page(page.is_truncated, request.continuation_token, page.next_continuation_token.as_deref())?;
page.objects = page page.objects = page
.objects .objects
.into_iter() .into_iter()
@@ -876,16 +749,15 @@ impl SourceClient {
#[async_trait::async_trait] #[async_trait::async_trait]
impl SourceBackend for S3SourceBackend { impl SourceBackend for S3SourceBackend {
async fn head(&self, key: &str) -> Result<SourceHead, SourceError> { async fn head(&self, key: &str) -> Result<SourceHead, SourceError> {
match self.client.head_object().bucket(&self.bucket).key(key).send().await { let output = self
Ok(output) => source_head_from_head_output(output), .client
Err(err) if err.raw_response().is_some_and(|response| response.status().as_u16() == 404) => { .head_object()
// HEAD has no error body: a missing bucket must not poison .bucket(&self.bucket)
// the per-key negative cache as though only the key was absent. .key(key)
self.probe().await?; .send()
Err(SourceError::NotFound) .await
} .map_err(classify_sdk_error)?;
Err(err) => Err(classify_sdk_error(err)), source_head_from_head_output(output)
}
} }
/// Streams the object; `range` is passed through as an HTTP `Range` /// Streams the object; `range` is passed through as an HTTP `Range`
@@ -928,12 +800,17 @@ impl SourceBackend for S3SourceBackend {
let is_truncated = output.is_truncated.unwrap_or(false); let is_truncated = output.is_truncated.unwrap_or(false);
let next_continuation_token = output.next_continuation_token; let next_continuation_token = output.next_continuation_token;
if is_truncated && next_continuation_token.is_none() {
return Err(SourceError::Other(
"source reported a truncated listing without a continuation token".to_string(),
));
}
let objects = output let objects = output
.contents .contents
.unwrap_or_default() .unwrap_or_default()
.into_iter() .into_iter()
.map(s3_source_object) .filter_map(s3_source_object)
.collect::<Result<Vec<_>, _>>()?; .collect();
let common_prefixes = output let common_prefixes = output
.common_prefixes .common_prefixes
.unwrap_or_default() .unwrap_or_default()
@@ -972,20 +849,14 @@ impl SourceBackend for S3SourceBackend {
} }
} }
fn s3_source_object(object: SdkObject) -> Result<SourceObject, SourceError> { fn s3_source_object(object: SdkObject) -> Option<SourceObject> {
let key = object let key = object.key?;
.key
.ok_or_else(|| SourceError::Other("source listing object has no key".to_string()))?;
let size = object
.size
.and_then(|size| u64::try_from(size).ok())
.ok_or_else(|| SourceError::Other("source listing object has no valid size".to_string()))?;
let etag = normalize_etag(object.e_tag); let etag = normalize_etag(object.e_tag);
let is_multipart_etag = etag.as_deref().is_some_and(is_multipart_etag); let is_multipart_etag = etag.as_deref().is_some_and(is_multipart_etag);
Ok(SourceObject { Some(SourceObject {
key, key,
etag, etag,
size, size: object.size.and_then(|size| u64::try_from(size).ok()).unwrap_or(0),
last_modified: system_time(object.last_modified), last_modified: system_time(object.last_modified),
storage_class: object.storage_class.map(|class| class.as_str().to_string()), storage_class: object.storage_class.map(|class| class.as_str().to_string()),
is_multipart_etag, is_multipart_etag,
@@ -995,7 +866,6 @@ fn s3_source_object(object: SdkObject) -> Result<SourceObject, SourceError> {
#[cfg(test)] #[cfg(test)]
mod tests { mod tests {
use super::*; use super::*;
use crate::on_demand_migration::backend_contract::{BackendCapabilities, OBJECT_MD5, assert_backend_contract};
use aws_smithy_runtime_api::client::http::{HttpConnector, HttpConnectorFuture, SharedHttpConnector, http_client_fn}; use aws_smithy_runtime_api::client::http::{HttpConnector, HttpConnectorFuture, SharedHttpConnector, http_client_fn};
use aws_smithy_runtime_api::client::orchestrator::HttpRequest; use aws_smithy_runtime_api::client::orchestrator::HttpRequest;
use aws_smithy_runtime_api::client::result::ConnectorError; use aws_smithy_runtime_api::client::result::ConnectorError;
@@ -1113,32 +983,9 @@ mod tests {
retry: RemoteS3RetryPolicy::Disabled, retry: RemoteS3RetryPolicy::Disabled,
timeouts: SourceTimeouts::default(), timeouts: SourceTimeouts::default(),
bandwidth_limit: NonZeroU64::new(1_000_000), bandwidth_limit: NonZeroU64::new(1_000_000),
backend: SourceBackendSpec::S3,
} }
} }
#[cfg(not(feature = "gcs"))]
#[tokio::test]
async fn gcs_backend_not_compiled_keeps_hmac_s3_available() {
let mut native = spec(None);
native.provider = SourceProvider::GcsNative;
native.credentials = None;
native.backend = SourceBackendSpec::Gcs(GcsSourceSpec {
service_account_json: "{}".to_string(),
});
assert!(matches!(
SourceClient::new(&native).await,
Err(RemoteS3ClientError::BackendNotCompiled("gcs_native"))
));
let mut hmac = spec(None);
hmac.provider = SourceProvider::Gcs;
hmac.endpoint = "https://storage.googleapis.com".to_string();
SourceClient::new(&hmac)
.await
.expect("GCS HMAC uses the always-available S3 backend");
}
async fn scripted_client(spec: &SourceClientSpec, responses: Vec<Scripted>) -> (SourceClient, Recorded) { async fn scripted_client(spec: &SourceClientSpec, responses: Vec<Scripted>) -> (SourceClient, Recorded) {
let requests: Recorded = Arc::new(Mutex::new(Vec::new())); let requests: Recorded = Arc::new(Mutex::new(Vec::new()));
let connector = SharedHttpConnector::new(ScriptedConnector { let connector = SharedHttpConnector::new(ScriptedConnector {
@@ -1427,9 +1274,7 @@ mod tests {
<CommonPrefixes><Prefix>data/photos/</Prefix></CommonPrefixes> <CommonPrefixes><Prefix>data/photos/</Prefix></CommonPrefixes>
<CommonPrefixes><Prefix>outside/</Prefix></CommonPrefixes> <CommonPrefixes><Prefix>outside/</Prefix></CommonPrefixes>
</ListBucketResult>"#; </ListBucketResult>"#;
let next_body = body.replace("data/opaque", "data/next"); let (client, requests) = scripted_client(&spec(Some("data/")), vec![ok(Vec::new(), body), ok(Vec::new(), body)]).await;
let (client, requests) =
scripted_client(&spec(Some("data/")), vec![ok(Vec::new(), body), ok(Vec::new(), &next_body)]).await;
let first = client let first = client
.list_page(&SourceListRequest { .list_page(&SourceListRequest {
prefix: Some("photos/"), prefix: Some("photos/"),
@@ -1491,104 +1336,7 @@ mod tests {
.list_objects_v2(None, None, 10) .list_objects_v2(None, None, 10)
.await .await
.expect_err("truncated page without token is corrupt"); .expect_err("truncated page without token is corrupt");
assert!(matches!(err, SourceError::InvalidPagination(ListPageError::Missing)), "{err:?}"); assert!(matches!(err, SourceError::Other(_)), "{err:?}");
}
#[tokio::test]
async fn list_page_validates_s3_cursor_progress_before_mapping_entries() {
for contents in ["", "<Contents><Key>data/a</Key><Size>1</Size></Contents>"] {
for (truncated, next, expected) in [
(true, None, Some(ListPageError::Missing)),
(true, Some(""), Some(ListPageError::Empty)),
(true, Some("stuck"), Some(ListPageError::Repeated)),
(true, Some("opaque-next"), None),
(false, None, None),
(false, Some("stuck"), None),
] {
let next_xml = next
.map(|next| format!("<NextContinuationToken>{next}</NextContinuationToken>"))
.unwrap_or_default();
let body = format!(
"<ListBucketResult xmlns=\"http://s3.amazonaws.com/doc/2006-03-01/\"><IsTruncated>{truncated}</IsTruncated>{next_xml}{contents}</ListBucketResult>"
);
let (client, requests) = scripted_client(&spec(Some("data/")), vec![ok(Vec::new(), &body)]).await;
let result = client
.list_page(&SourceListRequest {
continuation_token: Some("stuck"),
max_keys: 2,
..Default::default()
})
.await;
match expected {
Some(expected) => {
let error = result.expect_err("malformed pagination must fail at the provider boundary");
assert!(
matches!(&error, SourceError::InvalidPagination(actual) if *actual == expected),
"{error:?}"
);
assert_eq!(error.class_label(), "invalid_pagination");
assert!(!error.is_retryable());
assert!(!error.to_string().contains("stuck"), "errors must not echo opaque tokens");
}
None => {
let page = result.expect("progressing empty/nonempty pages and EOF are valid");
assert_eq!(page.is_truncated, truncated);
assert_eq!(page.next_continuation_token.as_deref(), next);
assert_eq!(page.objects.len(), usize::from(!contents.is_empty()));
if let Some(object) = page.objects.first() {
assert_eq!(object.key, "a");
}
}
}
let requests = recorded(&requests);
assert_eq!(requests.len(), 1, "invalid pagination must not be retried");
assert!(requests[0].uri.contains("continuation-token=stuck"));
}
}
}
struct ListOnlyBackend(SourcePage);
#[async_trait::async_trait]
impl SourceBackend for ListOnlyBackend {
async fn list(&self, request: &SourceListRequest<'_>) -> Result<SourcePage, SourceError> {
assert_eq!(request.continuation_token, Some("stuck"), "opaque cursors reach every provider unchanged");
Ok(self.0.clone())
}
async fn head(&self, _key: &str) -> Result<SourceHead, SourceError> {
panic!("unexpected HEAD in list test")
}
async fn get(&self, _key: &str, _range: Option<&HTTPRangeSpec>) -> Result<SourceGet, SourceError> {
panic!("unexpected GET in list test")
}
async fn tagging(&self, _key: &str) -> Result<HashMap<String, String>, SourceError> {
panic!("unexpected tagging in list test")
}
async fn probe(&self) -> Result<(), SourceError> {
panic!("unexpected probe in list test")
}
}
#[tokio::test]
async fn list_page_validates_non_s3_provider_cursors_at_the_common_boundary() {
for (next, expected) in [
(None, ListPageError::Missing),
(Some(""), ListPageError::Empty),
(Some("stuck"), ListPageError::Repeated),
] {
let mut client = prefix_client(Some("data/".into()));
client.backend = Box::new(ListOnlyBackend(SourcePage {
is_truncated: true,
next_continuation_token: next.map(str::to_string),
..Default::default()
}));
let error = client
.list_objects_v2(None, Some("stuck"), 2)
.await
.expect_err("all providers must advance pagination");
assert!(matches!(error, SourceError::InvalidPagination(actual) if actual == expected));
}
} }
const TAGGING_BODY: &str = r#"<?xml version="1.0" encoding="UTF-8"?> const TAGGING_BODY: &str = r#"<?xml version="1.0" encoding="UTF-8"?>
@@ -1642,10 +1390,7 @@ mod tests {
#[tokio::test] #[tokio::test]
async fn source_error_classification_covers_every_class() { async fn source_error_classification_covers_every_class() {
let cases: Vec<(Scripted, &str, bool)> = vec![ let cases: Vec<(Scripted, &str, bool)> = vec![
(status(404, ""), "other", false), (status(404, ""), "not_found", false),
(status(404, "<Error><Code>NoSuchKey</Code></Error>"), "not_found", false),
(status(404, "<Error><Code>NoSuchBucket</Code></Error>"), "other", false),
(status(404, "<Error><Code>NoSuchVersion</Code></Error>"), "other", false),
(status(403, ACCESS_DENIED_BODY), "access_denied", false), (status(403, ACCESS_DENIED_BODY), "access_denied", false),
(status(401, ""), "access_denied", false), (status(401, ""), "access_denied", false),
(status(429, ""), "throttled", true), (status(429, ""), "throttled", true),
@@ -1668,35 +1413,14 @@ mod tests {
} }
} }
let (client, requests) = scripted_client(&spec(None), vec![status(404, ""), status(200, "")]).await; // HEAD carries no error body, so the classification must work from the
// status alone as well.
let (client, _) = scripted_client(&spec(None), vec![status(404, "")]).await;
assert!(matches!(client.head_object("missing").await, Err(SourceError::NotFound))); assert!(matches!(client.head_object("missing").await, Err(SourceError::NotFound)));
assert_eq!(recorded(&requests).len(), 2, "ambiguous HEAD 404 must check the bucket");
let (client, _) = scripted_client(&spec(None), vec![status(404, ""), status(404, "")]).await;
assert!(matches!(client.head_object("missing").await, Err(SourceError::Other(_))));
let (client, _) = scripted_client(&spec(None), vec![status(404, ""), status(403, "")]).await;
assert!(matches!(client.head_object("missing").await, Err(SourceError::AccessDenied)));
let (client, _) = scripted_client(&spec(None), vec![status(403, "")]).await; let (client, _) = scripted_client(&spec(None), vec![status(403, "")]).await;
assert!(matches!(client.head_object("secret").await, Err(SourceError::AccessDenied))); assert!(matches!(client.head_object("secret").await, Err(SourceError::AccessDenied)));
} }
#[test]
fn source_listing_rejects_missing_and_negative_sizes() {
for size in [None, Some(-1)] {
let object = SdkObject::builder().key("key").set_size(size).build();
assert!(matches!(s3_source_object(object), Err(SourceError::Other(_))));
}
assert!(matches!(
s3_source_object(SdkObject::builder().size(0).build()),
Err(SourceError::Other(_))
));
assert_eq!(
s3_source_object(SdkObject::builder().key("empty").size(0).build())
.expect("empty object")
.size,
0
);
}
#[tokio::test] #[tokio::test]
async fn source_client_debug_redacts_credentials() { async fn source_client_debug_redacts_credentials() {
let (client, _) = scripted_client(&spec(Some("data/")), Vec::new()).await; let (client, _) = scripted_client(&spec(Some("data/")), Vec::new()).await;
@@ -1763,102 +1487,7 @@ mod tests {
assert_eq!(resolve_path_style(PathStyle::VirtualHost, Minio, "10.0.0.1"), PathStyle::VirtualHost); assert_eq!(resolve_path_style(PathStyle::VirtualHost, Minio, "10.0.0.1"), PathStyle::VirtualHost);
assert_eq!(resolve_path_style(PathStyle::Path, Aws, "s3.amazonaws.com"), PathStyle::Path); assert_eq!(resolve_path_style(PathStyle::Path, Aws, "s3.amazonaws.com"), PathStyle::Path);
assert_eq!(SourceProvider::from_label(" AWS "), Some(Aws)); assert_eq!(SourceProvider::from_label(" AWS "), Some(Aws));
assert_eq!(SourceProvider::from_label(" Azure "), Some(Azure)); assert_eq!(SourceProvider::from_label("azure"), None);
assert_eq!(SourceProvider::from_label("gcs_native"), Some(GcsNative));
assert_eq!(SourceProvider::from_label("swift"), None);
}
const CONTRACT_LIST_PAGE_ONE: &str = r#"<?xml version="1.0" encoding="UTF-8"?>
<ListBucketResult xmlns="http://s3.amazonaws.com/doc/2006-03-01/">
<Name>source-bucket</Name>
<IsTruncated>true</IsTruncated>
<NextContinuationToken>cursor-1</NextContinuationToken>
<Contents>
<Key>dir/a.txt</Key>
<LastModified>2015-10-21T07:28:00.000Z</LastModified>
<ETag>&quot;5d41402abc4b2a76b9719d911017c592&quot;</ETag>
<Size>5</Size>
<StorageClass>STANDARD</StorageClass>
</Contents>
<CommonPrefixes><Prefix>dir/sub/</Prefix></CommonPrefixes>
</ListBucketResult>"#;
const CONTRACT_LIST_PAGE_TWO: &str = r#"<?xml version="1.0" encoding="UTF-8"?>
<ListBucketResult xmlns="http://s3.amazonaws.com/doc/2006-03-01/">
<Name>source-bucket</Name>
<IsTruncated>false</IsTruncated>
<Contents>
<Key>dir/b.txt</Key>
<LastModified>2015-10-21T07:28:00.000Z</LastModified>
<ETag>&quot;7d41402abc4b2a76b9719d911017c592&quot;</ETag>
<Size>7</Size>
</Contents>
</ListBucketResult>"#;
const CONTRACT_TAGGING: &str = r#"<?xml version="1.0" encoding="UTF-8"?>
<Tagging xmlns="http://s3.amazonaws.com/doc/2006-03-01/"><TagSet>
<Tag><Key>env</Key><Value>prod</Value></Tag>
</TagSet></Tagging>"#;
fn contract_object_headers(content_length: u64) -> Vec<(&'static str, String)> {
vec![
("etag", format!("\"{OBJECT_MD5}\"")),
("content-length", content_length.to_string()),
("content-type", "text/plain".to_string()),
("last-modified", "Wed, 21 Oct 2015 07:28:00 GMT".to_string()),
("x-amz-meta-owner", "alice".to_string()),
("x-amz-storage-class", "STANDARD".to_string()),
]
}
/// The S3 backend behind the scripted connector, without the prefix-mapping
/// client on top: the contract is a property of the backend itself.
async fn scripted_s3_backend(responses: Vec<Scripted>) -> S3SourceBackend {
let spec = spec(None);
let connector = SharedHttpConnector::new(ScriptedConnector {
requests: Arc::new(Mutex::new(Vec::new())),
responses: Arc::new(Mutex::new(responses.into_iter().collect())),
});
let http_client = http_client_fn(move |_settings, _components| connector.clone());
let endpoint = spec.endpoint_spec().expect("test spec endpoint should parse");
let config = build_remote_s3_config(&endpoint)
.await
.expect("test spec should build")
.http_client(http_client)
.interceptor(SourceProxyMarkerInterceptor::new());
S3SourceBackend {
client: S3Client::from_conf(config.build()),
bucket: spec.bucket.clone(),
}
}
#[tokio::test]
async fn s3_backend_satisfies_the_shared_backend_contract() {
let mut ranged = contract_object_headers(3);
ranged.push(("content-range", "bytes 1-3/5".to_string()));
let backend = scripted_s3_backend(vec![
ok(contract_object_headers(5), ""),
ok(contract_object_headers(5), "hello"),
ok(ranged, "ell"),
ok(Vec::new(), CONTRACT_LIST_PAGE_ONE),
ok(Vec::new(), CONTRACT_LIST_PAGE_TWO),
ok(Vec::new(), CONTRACT_TAGGING),
ok(Vec::new(), ""),
status(404, ""),
ok(Vec::new(), ""),
status(403, ACCESS_DENIED_BODY),
])
.await;
assert_backend_contract(
&backend,
BackendCapabilities {
etag_is_opaque: false,
supports_start_after: true,
supports_tagging: true,
},
)
.await;
} }
fn prefix_client(prefix: Option<String>) -> SourceClient { fn prefix_client(prefix: Option<String>) -> SourceClient {
@@ -177,7 +177,7 @@ impl From<&SourceError> for PullFailureReason {
SourceError::Connect(_) => PullFailureReason::SourceConnect, SourceError::Connect(_) => PullFailureReason::SourceConnect,
SourceError::ServerError(_) => PullFailureReason::SourceServerError, SourceError::ServerError(_) => PullFailureReason::SourceServerError,
SourceError::Unsupported(_) => PullFailureReason::SourceUnsupported, SourceError::Unsupported(_) => PullFailureReason::SourceUnsupported,
SourceError::InvalidPagination(_) | SourceError::Other(_) => PullFailureReason::SourceOther, SourceError::Other(_) => PullFailureReason::SourceOther,
} }
} }
} }
@@ -19,12 +19,13 @@
//! [`SourceClient`], a circuit breaker, a negative cache, a per-key //! [`SourceClient`], a circuit breaker, a negative cache, a per-key
//! singleflight table, a pull concurrency limit and counters. Its lifecycle //! singleflight table, a pull concurrency limit and counters. Its lifecycle
//! follows the bucket metadata cache through the publish hook registered in //! follows the bucket metadata cache through the publish hook registered in
//! [`BUCKET_CONFIG_PUBLISH_HOOK`]; the hook fires on every cache install //! [`ON_DEMAND_MIGRATION_CONFIG_HOOK`]; the hook fires on every cache install
//! path (initial load, admin update, peer reload, refresh loop, lazy load). //! path (initial load, admin update, peer reload, refresh loop, lazy load).
//! //!
//! Change detection compares the config by value (`PartialEq`) rather than //! Change detection compares the config by value (`PartialEq`) rather than
//! by `updated_at`. The bucket incarnation is part of this comparison: //! by `updated_at`: the hook does not carry the timestamp, fetching it would
//! recreating a bucket must cancel old work even with identical configuration. //! re-enter the metadata system from inside its own publish path, and a
//! byte-identical config never needs a new client anyway.
//! //!
//! Client construction is async (TLS material may be read from disk), so //! Client construction is async (TLS material may be read from disk), so
//! the hook does not build inline: `publish` removes state synchronously and //! the hook does not build inline: `publish` removes state synchronously and
@@ -40,19 +41,17 @@
use super::backfill::{PriorityPullPermits, PullPermit, PullPriority}; use super::backfill::{PriorityPullPermits, PullPermit, PullPriority};
use super::breaker::{Breaker, BreakerState, BreakerTransition, BreakerVerdict}; use super::breaker::{Breaker, BreakerState, BreakerTransition, BreakerVerdict};
use super::config::{OnDemandMigrationConfig, PathStyle as ConfigPathStyle, Provider, SourceConfig}; use super::config::{
ON_DEMAND_MIGRATION_CONFIG_HOOK, OnDemandMigrationConfig, PathStyle as ConfigPathStyle, Provider, SourceConfig,
};
use super::list_through::{SOURCE_LIST_RATE_PER_SEC, SourceListRateLimiter}; use super::list_through::{SOURCE_LIST_RATE_PER_SEC, SourceListRateLimiter};
use super::negative_cache::NegativeCache; use super::negative_cache::NegativeCache;
use super::pull::{OdmWriteBack, PullQueue}; use super::pull::{OdmWriteBack, PullQueue};
use super::source_client::{ use super::source_client::{SourceClient, SourceClientSpec, SourceError, SourceProvider, SourceTimeouts};
AzureAuth, AzureSourceSpec, GcsSourceSpec, SourceBackendSpec, SourceClient, SourceClientSpec, SourceError, SourceProvider,
SourceTimeouts,
};
use super::stats::{GaugeGuard, OdmStats, OdmStatsSnapshot, PullFailureReason}; use super::stats::{GaugeGuard, OdmStats, OdmStatsSnapshot, PullFailureReason};
use super::storage_api::remote_s3_client::{ use crate::bucket::remote_s3_client::{
PathStyle as ClientPathStyle, RemoteCredentials, RemoteS3ClientError, RemoteS3RetryPolicy, PathStyle as ClientPathStyle, RemoteCredentials, RemoteS3ClientError, RemoteS3RetryPolicy,
}; };
use super::storage_api::{BUCKET_CONFIG_PUBLISH_HOOK, BUCKET_ON_DEMAND_MIGRATION_CONFIG};
use parking_lot::{Mutex, RwLock}; use parking_lot::{Mutex, RwLock};
use serde::{Deserialize, Serialize}; use serde::{Deserialize, Serialize};
use std::collections::HashMap; use std::collections::HashMap;
@@ -83,8 +82,6 @@ pub static GLOBAL_ON_DEMAND_MIGRATION_SYS: OnceLock<OnDemandMigrationSys> = Once
/// `resolve` as [`OdmLookup::Unavailable`] and through status snapshots. /// `resolve` as [`OdmLookup::Unavailable`] and through status snapshots.
#[derive(Clone, Debug, PartialEq, Eq, thiserror::Error)] #[derive(Clone, Debug, PartialEq, Eq, thiserror::Error)]
pub enum OdmStateError { pub enum OdmStateError {
#[error("the {0} backend is not included in this build")]
BackendNotCompiled(&'static str),
/// `source.credentials` is `null`; the shared client builder has no /// `source.credentials` is `null`; the shared client builder has no
/// anonymous mode yet (rustfs/backlog#2149 follow-up). /// anonymous mode yet (rustfs/backlog#2149 follow-up).
#[error("anonymous source access is not supported yet; configure source credentials")] #[error("anonymous source access is not supported yet; configure source credentials")]
@@ -269,7 +266,6 @@ impl Drop for InflightEntryGuard<'_> {
/// config change (counters excepted), removed when the config goes away. /// config change (counters excepted), removed when the config goes away.
pub struct BucketOdmState { pub struct BucketOdmState {
bucket: String, bucket: String,
incarnation_id: uuid::Uuid,
config: OnDemandMigrationConfig, config: OnDemandMigrationConfig,
applied_at: OffsetDateTime, applied_at: OffsetDateTime,
endpoint_host: String, endpoint_host: String,
@@ -307,24 +303,21 @@ impl BucketOdmState {
async fn build( async fn build(
bucket: &str, bucket: &str,
config: &OnDemandMigrationConfig, config: &OnDemandMigrationConfig,
incarnation_id: uuid::Uuid,
stats: Arc<OdmStats>, stats: Arc<OdmStats>,
write_back: Option<Arc<dyn OdmWriteBack>>, write_back: Option<Arc<dyn OdmWriteBack>>,
) -> Arc<Self> { ) -> Arc<Self> {
let spec = source_client_spec(config); let spec = source_client_spec(config);
let client = if config.source.credentials.is_none() && !config.source.provider.is_native() { let client = if config.source.credentials.is_none() {
Err(OdmStateError::AnonymousUnsupported) Err(OdmStateError::AnonymousUnsupported)
} else { } else {
SourceClient::new(&spec).await.map(Arc::new).map_err(|err| match err { SourceClient::new(&spec).await.map(Arc::new).map_err(|err| match err {
RemoteS3ClientError::MissingCredentials => OdmStateError::AnonymousUnsupported, RemoteS3ClientError::MissingCredentials => OdmStateError::AnonymousUnsupported,
RemoteS3ClientError::BackendNotCompiled(provider) => OdmStateError::BackendNotCompiled(provider),
other => OdmStateError::ClientBuild(other.to_string()), other => OdmStateError::ClientBuild(other.to_string()),
}) })
}; };
let policy = &config.policy; let policy = &config.policy;
Arc::new(Self { Arc::new(Self {
bucket: bucket.to_string(), bucket: bucket.to_string(),
incarnation_id,
endpoint_host: endpoint_host(&config.source), endpoint_host: endpoint_host(&config.source),
config: config.clone(), config: config.clone(),
applied_at: OffsetDateTime::now_utc(), applied_at: OffsetDateTime::now_utc(),
@@ -342,18 +335,10 @@ impl BucketOdmState {
}) })
} }
pub fn filter_incarnation(self: Arc<Self>, incarnation_id: uuid::Uuid) -> Option<Arc<Self>> {
(self.incarnation_id == incarnation_id && !self.is_cancelled()).then_some(self)
}
pub fn bucket(&self) -> &str { pub fn bucket(&self) -> &str {
&self.bucket &self.bucket
} }
pub fn incarnation_id(&self) -> uuid::Uuid {
self.incarnation_id
}
pub fn config(&self) -> &OnDemandMigrationConfig { pub fn config(&self) -> &OnDemandMigrationConfig {
&self.config &self.config
} }
@@ -634,7 +619,6 @@ pub fn source_client_spec(config: &OnDemandMigrationConfig) -> SourceClientSpec
// load on a source that is already failing. // load on a source that is already failing.
retry: RemoteS3RetryPolicy::Disabled, retry: RemoteS3RetryPolicy::Disabled,
bandwidth_limit: policy.bandwidth_limit_bytes_per_sec.and_then(NonZeroU64::new), bandwidth_limit: policy.bandwidth_limit_bytes_per_sec.and_then(NonZeroU64::new),
backend: source_backend_spec(source),
} }
} }
@@ -646,31 +630,6 @@ fn source_provider(provider: Provider) -> SourceProvider {
Provider::Rustfs => SourceProvider::Rustfs, Provider::Rustfs => SourceProvider::Rustfs,
Provider::R2 => SourceProvider::R2, Provider::R2 => SourceProvider::R2,
Provider::Gcs => SourceProvider::Gcs, Provider::Gcs => SourceProvider::Gcs,
Provider::Azure => SourceProvider::Azure,
Provider::GcsNative => SourceProvider::GcsNative,
}
}
/// Which backend the client builds. A native provider whose block is missing
/// falls back to the S3 spec, where the builder reports the missing
/// credentials: the config layer already refuses to store that shape, so this
/// only covers a config written by an older or hand-edited build.
pub fn source_backend_spec(source: &SourceConfig) -> SourceBackendSpec {
match (source.provider, source.azure.as_ref(), source.gcs.as_ref()) {
(Provider::Azure, Some(azure), _) => SourceBackendSpec::Azure(AzureSourceSpec {
account: azure.account.clone(),
auth: match (&azure.account_key, &azure.sas_token) {
(Some(key), _) => AzureAuth::SharedKey(key.clone()),
(None, Some(sas)) => AzureAuth::Sas(sas.clone()),
// Refused by `SourceConfig::validate`; an empty shared key
// fails closed at the builder rather than signing with none.
(None, None) => AzureAuth::SharedKey(String::new()),
},
}),
(Provider::GcsNative, _, Some(gcs)) => SourceBackendSpec::Gcs(GcsSourceSpec {
service_account_json: gcs.service_account_json.clone(),
}),
_ => SourceBackendSpec::S3,
} }
} }
@@ -749,51 +708,21 @@ impl OnDemandMigrationSys {
/// Registers `publish` as the bucket-metadata publish hook. Returns /// Registers `publish` as the bucket-metadata publish hook. Returns
/// `false` when a hook was already registered. /// `false` when a hook was already registered.
pub fn register_config_hook(&'static self) -> bool { pub fn register_config_hook(&'static self) -> bool {
BUCKET_CONFIG_PUBLISH_HOOK ON_DEMAND_MIGRATION_CONFIG_HOOK
.set(Box::new(move |bucket, config_file, stored| { .set(Box::new(move |bucket, config| self.publish(bucket, config)))
if config_file == BUCKET_ON_DEMAND_MIGRATION_CONFIG {
self.publish_stored(bucket, stored.map(|(bytes, _, incarnation)| (bytes, incarnation)));
}
}))
.is_ok() .is_ok()
} }
/// Corrupt persisted bytes withdraw state synchronously, just like deletion.
fn publish_stored(&'static self, bucket: &str, stored: Option<(&[u8], uuid::Uuid)>) {
let incarnation_id = stored.map(|(_, id)| id).unwrap_or_default();
match stored.map(|(bytes, _)| OnDemandMigrationConfig::from_json(bytes)).transpose() {
Ok(config) => self.publish_for_incarnation(bucket, incarnation_id, config.as_ref()),
Err(err) => {
warn!(
event = EVENT_ODM_BUCKET_STATE_APPLIED,
component = LOG_COMPONENT_ECSTORE,
subsystem = LOG_SUBSYSTEM_ON_DEMAND_MIGRATION,
result = "invalid",
bucket = %bucket,
error = %err,
"Failed to parse on-demand migration config"
);
self.publish_for_incarnation(bucket, incarnation_id, None);
}
}
}
/// Hook entry point: removals apply immediately, installs are spawned /// Hook entry point: removals apply immediately, installs are spawned
/// (client construction is async). Requires a Tokio runtime for the /// (client construction is async). Requires a Tokio runtime for the
/// install path; without one the config is logged and skipped. /// install path; without one the config is logged and skipped.
pub fn publish_for_incarnation( pub fn publish(&'static self, bucket: &str, config: Option<&OnDemandMigrationConfig>) {
&'static self, let generation = self.next_generation();
bucket: &str, let Some(config) = self.desired(config) else {
incarnation_id: uuid::Uuid,
config: Option<&OnDemandMigrationConfig>,
) {
let config = self.desired(config).filter(|_| !incarnation_id.is_nil());
let generation = self.reserve_generation(bucket, config.is_some());
let Some(config) = config else {
self.remove_with_generation(bucket, generation); self.remove_with_generation(bucket, generation);
return; return;
}; };
if self.is_unchanged(bucket, incarnation_id, config, generation) { if self.is_unchanged(bucket, config, generation) {
return; return;
} }
let Ok(handle) = tokio::runtime::Handle::try_current() else { let Ok(handle) = tokio::runtime::Handle::try_current() else {
@@ -812,49 +741,31 @@ impl OnDemandMigrationSys {
let bucket = bucket.to_string(); let bucket = bucket.to_string();
let config = config.clone(); let config = config.clone();
handle.spawn(async move { handle.spawn(async move {
self.apply_with_generation(&bucket, incarnation_id, Some(&config), generation) self.apply_with_generation(&bucket, Some(&config), generation).await;
.await;
}); });
} }
/// Installs, rebuilds, or removes the bucket state for `config`. /// Installs, rebuilds, or removes the bucket state for `config`.
/// Idempotent: the same config on an installed bucket is a no-op. /// Idempotent: the same config on an installed bucket is a no-op.
#[cfg(test)]
pub async fn apply(&self, bucket: &str, config: Option<&OnDemandMigrationConfig>) -> ApplyOutcome { pub async fn apply(&self, bucket: &str, config: Option<&OnDemandMigrationConfig>) -> ApplyOutcome {
self.apply_for_incarnation(bucket, uuid::Uuid::from_u128(1), config).await let generation = self.next_generation();
} self.apply_with_generation(bucket, config, generation).await
#[cfg(test)]
pub fn publish(&'static self, bucket: &str, config: Option<&OnDemandMigrationConfig>) {
self.publish_for_incarnation(bucket, uuid::Uuid::from_u128(1), config);
}
pub async fn apply_for_incarnation(
&self,
bucket: &str,
incarnation_id: uuid::Uuid,
config: Option<&OnDemandMigrationConfig>,
) -> ApplyOutcome {
let config = self.desired(config).filter(|_| !incarnation_id.is_nil());
let generation = self.reserve_generation(bucket, config.is_some());
self.apply_with_generation(bucket, incarnation_id, config, generation).await
} }
async fn apply_with_generation( async fn apply_with_generation(
&self, &self,
bucket: &str, bucket: &str,
incarnation_id: uuid::Uuid,
config: Option<&OnDemandMigrationConfig>, config: Option<&OnDemandMigrationConfig>,
generation: u64, generation: u64,
) -> ApplyOutcome { ) -> ApplyOutcome {
let Some(config) = self.desired(config) else { let Some(config) = self.desired(config) else {
return self.remove_with_generation(bucket, generation); return self.remove_with_generation(bucket, generation);
}; };
if self.is_unchanged(bucket, incarnation_id, config, generation) { if self.is_unchanged(bucket, config, generation) {
return ApplyOutcome::Unchanged; return ApplyOutcome::Unchanged;
} }
let stats = self.state(bucket).map(|state| Arc::clone(&state.stats)).unwrap_or_default(); let stats = self.state(bucket).map(|state| Arc::clone(&state.stats)).unwrap_or_default();
let state = BucketOdmState::build(bucket, config, incarnation_id, stats, self.write_back()).await; let state = BucketOdmState::build(bucket, config, stats, self.write_back()).await;
let (outcome, previous) = { let (outcome, previous) = {
let mut buckets = self.buckets.write(); let mut buckets = self.buckets.write();
@@ -898,13 +809,12 @@ impl OnDemandMigrationSys {
/// Removes a bucket's state (idempotent), cancelling its token. /// Removes a bucket's state (idempotent), cancelling its token.
pub fn remove(&self, bucket: &str) -> ApplyOutcome { pub fn remove(&self, bucket: &str) -> ApplyOutcome {
let generation = self.reserve_generation(bucket, false); let generation = self.next_generation();
self.remove_with_generation(bucket, generation) self.remove_with_generation(bucket, generation)
} }
/// One-shot lookup: module switch, bucket state, prefix filter, /// One-shot lookup: module switch, bucket state, prefix filter,
/// client availability, negative cache, breaker, in that order. /// client availability, negative cache, breaker, in that order.
#[cfg(test)]
pub fn resolve(&self, bucket: &str, key: &str) -> Option<OdmLookup> { pub fn resolve(&self, bucket: &str, key: &str) -> Option<OdmLookup> {
if !self.is_module_enabled() { if !self.is_module_enabled() {
return None; return None;
@@ -912,13 +822,6 @@ impl OnDemandMigrationSys {
self.state(bucket)?.resolve_key(key) self.state(bucket)?.resolve_key(key)
} }
pub fn resolve_for_incarnation(&self, bucket: &str, key: &str, incarnation_id: uuid::Uuid) -> Option<OdmLookup> {
if !self.is_module_enabled() {
return None;
}
self.state(bucket)?.filter_incarnation(incarnation_id)?.resolve_key(key)
}
pub fn state(&self, bucket: &str) -> Option<Arc<BucketOdmState>> { pub fn state(&self, bucket: &str) -> Option<Arc<BucketOdmState>> {
self.buckets.read().get(bucket).and_then(|slot| slot.state.clone()) self.buckets.read().get(bucket).and_then(|slot| slot.state.clone())
} }
@@ -947,17 +850,8 @@ impl OnDemandMigrationSys {
snapshots snapshots
} }
fn reserve_generation(&self, bucket: &str, installing: bool) -> u64 { fn next_generation(&self) -> u64 {
// Reserve a desired install before its async client build, under the self.generation.fetch_add(1, Ordering::Relaxed) + 1
// same lock that orders removals. Unconfigured buckets need no slot.
let mut buckets = self.buckets.write();
let generation = self.generation.fetch_add(1, Ordering::Relaxed) + 1;
if installing {
buckets.entry(bucket.to_string()).or_default().generation = generation;
} else if let Some(slot) = buckets.get_mut(bucket) {
slot.generation = generation;
}
generation
} }
fn desired<'c>(&self, config: Option<&'c OnDemandMigrationConfig>) -> Option<&'c OnDemandMigrationConfig> { fn desired<'c>(&self, config: Option<&'c OnDemandMigrationConfig>) -> Option<&'c OnDemandMigrationConfig> {
@@ -966,26 +860,15 @@ impl OnDemandMigrationSys {
/// Claims `generation` for the bucket when the installed state already /// Claims `generation` for the bucket when the installed state already
/// matches `config` and has a usable client. /// matches `config` and has a usable client.
fn is_unchanged(&self, bucket: &str, incarnation_id: uuid::Uuid, config: &OnDemandMigrationConfig, generation: u64) -> bool { fn is_unchanged(&self, bucket: &str, config: &OnDemandMigrationConfig, generation: u64) -> bool {
let mut buckets = self.buckets.write(); let mut buckets = self.buckets.write();
let Some(slot) = buckets.get_mut(bucket) else { let Some(slot) = buckets.get_mut(bucket) else {
return false; return false;
}; };
if slot.generation > generation {
return false;
}
if slot
.state
.as_ref()
.is_some_and(|state| state.incarnation_id != incarnation_id)
&& let Some(previous) = slot.state.take()
{
previous.cancel.cancel();
}
let unchanged = slot let unchanged = slot
.state .state
.as_ref() .as_ref()
.is_some_and(|state| state.client.is_ok() && state.incarnation_id == incarnation_id && state.config == *config); .is_some_and(|state| state.client.is_ok() && state.config == *config);
if unchanged && slot.generation < generation { if unchanged && slot.generation < generation {
slot.generation = generation; slot.generation = generation;
} }
@@ -1025,8 +908,8 @@ impl OnDemandMigrationSys {
#[cfg(test)] #[cfg(test)]
mod tests { mod tests {
use super::*; use super::*;
use crate::on_demand_migration::breaker::BREAKER_FAILURE_THRESHOLD; use crate::bucket::on_demand_migration::breaker::BREAKER_FAILURE_THRESHOLD;
use crate::on_demand_migration::config::{FilterConfig, PolicyConfig, SourceCredentials, SourceTimeout, TlsConfig}; use crate::bucket::on_demand_migration::config::{FilterConfig, PolicyConfig, SourceCredentials, SourceTimeout, TlsConfig};
use std::sync::atomic::AtomicUsize; use std::sync::atomic::AtomicUsize;
use tokio::sync::Barrier; use tokio::sync::Barrier;
@@ -1046,8 +929,6 @@ mod tests {
session_token: None, session_token: None,
}), }),
tls: TlsConfig::default(), tls: TlsConfig::default(),
azure: None,
gcs: None,
}, },
filter: FilterConfig { filter: FilterConfig {
prefix: prefix.map(str::to_string), prefix: prefix.map(str::to_string),
@@ -1164,45 +1045,6 @@ mod tests {
assert_eq!(sys.apply("b", Some(&cfg)).await, ApplyOutcome::Rebuilt); assert_eq!(sys.apply("b", Some(&cfg)).await, ApplyOutcome::Rebuilt);
} }
#[tokio::test]
async fn native_azure_uses_provider_credentials_without_s3_credentials() {
let sys = enabled_sys();
let mut cfg = config(None);
cfg.source.provider = Provider::Azure;
cfg.source.endpoint = None;
cfg.source.credentials = None;
cfg.source.azure = Some(super::super::config::AzureSourceConfig {
account: "legacyaccount".to_string(),
account_key: Some("c2VjcmV0LWtleQ==".to_string()),
sas_token: None,
});
assert_eq!(sys.apply("b", Some(&cfg)).await, ApplyOutcome::Installed);
let state = ready_state(sys.resolve("b", "k"));
assert!(state.client().is_ok(), "native credentials must not be classified as anonymous S3");
}
#[cfg(not(feature = "gcs"))]
#[tokio::test]
async fn gcs_backend_not_compiled_is_unavailable_not_anonymous() {
let sys = enabled_sys();
let mut cfg = config(None);
cfg.source.provider = Provider::GcsNative;
cfg.source.credentials = None;
cfg.source.gcs = Some(super::super::config::GcsSourceConfig {
service_account_json: "{}".to_string(),
});
let encoded = cfg.to_json().expect("GCS config is serializable without the backend");
let restored: OnDemandMigrationConfig = serde_json::from_slice(&encoded).expect("GCS config stays readable");
assert_eq!(restored, cfg);
assert_eq!(sys.apply("b", Some(&cfg)).await, ApplyOutcome::Installed);
match sys.resolve("b", "k") {
Some(OdmLookup::Unavailable { error, .. }) => {
assert_eq!(error, OdmStateError::BackendNotCompiled("gcs_native"));
}
other => panic!("expected unavailable backend, got {other:?}"),
}
}
#[tokio::test(flavor = "multi_thread", worker_threads = 4)] #[tokio::test(flavor = "multi_thread", worker_threads = 4)]
async fn singleflight_admits_one_leader_per_key() { async fn singleflight_admits_one_leader_per_key() {
let sys = enabled_sys(); let sys = enabled_sys();
@@ -1418,128 +1260,24 @@ mod tests {
assert!(state.is_cancelled()); assert!(state.is_cancelled());
} }
#[tokio::test]
async fn identical_config_on_recreated_bucket_cancels_old_state() {
let sys = enabled_sys();
let cfg = config(None);
let old_id = uuid::Uuid::new_v4();
let new_id = uuid::Uuid::new_v4();
sys.apply_for_incarnation("recreated", old_id, Some(&cfg)).await;
let old = sys.state("recreated").expect("old state installed");
assert!(sys.resolve_for_incarnation("recreated", "key", new_id).is_none());
sys.apply_for_incarnation("recreated", new_id, Some(&cfg)).await;
let replacement = sys.state("recreated").expect("replacement state installed");
assert!(old.is_cancelled());
assert!(!Arc::ptr_eq(&old, &replacement));
assert_eq!(replacement.incarnation_id(), new_id);
assert!(sys.resolve_for_incarnation("recreated", "key", old_id).is_none());
assert!(sys.resolve_for_incarnation("recreated", "key", new_id).is_some());
}
#[tokio::test]
async fn changed_delete_marker_policy_withdraws_the_captured_lookup() {
let sys = enabled_sys();
let incarnation = uuid::Uuid::new_v4();
let mut cfg = config(None);
cfg.policy.respect_local_delete_marker = false;
sys.apply_for_incarnation("policy-snapshot", incarnation, Some(&cfg)).await;
let captured = sys.state("policy-snapshot").expect("policy A installed");
assert!(!captured.config().policy.respect_local_delete_marker);
cfg.policy.respect_local_delete_marker = true;
sys.apply_for_incarnation("policy-snapshot", incarnation, Some(&cfg)).await;
let replacement = sys.state("policy-snapshot").expect("policy B installed");
assert!(replacement.config().policy.respect_local_delete_marker);
assert!(captured.is_cancelled());
assert!(
captured
.filter_incarnation(incarnation)
.and_then(|state| state.resolve_key("key"))
.is_none(),
"a request that evaluated policy A cannot continue through policy B"
);
assert!(
replacement
.clone()
.filter_incarnation(incarnation)
.and_then(|state| state.resolve_key("key"))
.is_some()
);
assert_eq!(
replacement
.stats()
.snapshot(replacement.breaker().state())
.source_latency
.count,
0
);
}
#[tokio::test]
async fn missing_incarnation_cannot_install_or_retain_source_state() {
let sys: &'static OnDemandMigrationSys = Box::leak(Box::new(enabled_sys()));
let cfg = config(None);
assert_eq!(
sys.apply_for_incarnation("missing", uuid::Uuid::nil(), Some(&cfg)).await,
ApplyOutcome::NotDesired
);
sys.publish_for_incarnation("missing", uuid::Uuid::nil(), Some(&cfg));
assert!(sys.state("missing").is_none());
sys.apply_for_incarnation("missing", uuid::Uuid::new_v4(), Some(&cfg)).await;
let state = sys.state("missing").expect("valid identity installed");
sys.publish_for_incarnation("missing", uuid::Uuid::nil(), Some(&cfg));
assert!(sys.state("missing").is_none());
assert!(state.is_cancelled());
}
#[tokio::test]
async fn corrupt_stored_config_withdraws_runtime_state() {
let sys: &'static OnDemandMigrationSys = Box::leak(Box::new(enabled_sys()));
let cfg = config(None);
assert_eq!(sys.apply("corrupt", Some(&cfg)).await, ApplyOutcome::Installed);
let state = sys.state("corrupt").expect("state installed");
sys.publish_stored("corrupt", Some((b"not-json", uuid::Uuid::from_u128(1))));
assert!(sys.state("corrupt").is_none(), "corruption cannot keep an older source active");
assert!(state.is_cancelled(), "corruption cancels in-flight work");
}
#[tokio::test]
async fn absent_config_updates_do_not_allocate_bucket_slots() {
let sys = enabled_sys();
for index in 0..1000 {
let bucket = format!("unconfigured-{index}");
assert_eq!(sys.apply(&bucket, None).await, ApplyOutcome::NotDesired);
assert_eq!(sys.remove(&bucket), ApplyOutcome::NotDesired);
}
assert!(sys.buckets.read().is_empty(), "unconfigured buckets must not accumulate tombstones");
}
#[tokio::test] #[tokio::test]
async fn stale_install_cannot_overwrite_a_later_removal() { async fn stale_install_cannot_overwrite_a_later_removal() {
let sys = enabled_sys(); let sys = enabled_sys();
let cfg = config(None); let cfg = config(None);
let older = sys.reserve_generation("b", true); let older = sys.next_generation();
let newer = sys.reserve_generation("b", false); let newer = sys.next_generation();
assert_eq!(sys.remove_with_generation("b", newer), ApplyOutcome::NotDesired); assert_eq!(sys.remove_with_generation("b", newer), ApplyOutcome::NotDesired);
assert_eq!( // The removal above did not create a slot; simulate an install that
sys.apply_with_generation("b", uuid::Uuid::from_u128(1), Some(&cfg), older) // started before it and finishes after.
.await, sys.apply_with_generation("b", Some(&cfg), older).await;
ApplyOutcome::Superseded assert!(sys.state("b").is_some(), "no slot yet, so the older install lands");
);
assert!(sys.state("b").is_none(), "removal must supersede an in-flight first install");
assert_eq!(sys.apply("b", Some(&cfg)).await, ApplyOutcome::Installed);
let installed = sys.state("b").unwrap(); let installed = sys.state("b").unwrap();
let older = sys.reserve_generation("b", true); let older = sys.next_generation();
let newer = sys.reserve_generation("b", false); let newer = sys.next_generation();
assert_eq!(sys.remove_with_generation("b", newer), ApplyOutcome::Removed); assert_eq!(sys.remove_with_generation("b", newer), ApplyOutcome::Removed);
assert!(installed.is_cancelled()); assert!(installed.is_cancelled());
assert_eq!( assert_eq!(sys.apply_with_generation("b", Some(&cfg), older).await, ApplyOutcome::Superseded);
sys.apply_with_generation("b", uuid::Uuid::from_u128(1), Some(&cfg), older)
.await,
ApplyOutcome::Superseded
);
assert!(sys.state("b").is_none(), "the stale install is discarded"); assert!(sys.state("b").is_none(), "the stale install is discarded");
} }
+5 -174
View File
@@ -180,8 +180,6 @@ impl RemoteS3EndpointSpec {
#[derive(Debug, thiserror::Error)] #[derive(Debug, thiserror::Error)]
pub enum RemoteS3ClientError { pub enum RemoteS3ClientError {
#[error("the {0} backend is not included in this build")]
BackendNotCompiled(&'static str),
#[error("remote endpoint requires credentials")] #[error("remote endpoint requires credentials")]
MissingCredentials, MissingCredentials,
#[error("{0}")] #[error("{0}")]
@@ -283,7 +281,9 @@ impl Intercept for UserAgentSuffixInterceptor {
/// Builds the SDK config for `spec` without finalizing it, so callers can add /// Builds the SDK config for `spec` without finalizing it, so callers can add
/// interceptors or (in tests) swap the HTTP client before `build()`. /// interceptors or (in tests) swap the HTTP client before `build()`.
pub async fn build_remote_s3_config(spec: &RemoteS3EndpointSpec) -> Result<aws_sdk_s3::config::Builder, RemoteS3ClientError> { pub(crate) async fn build_remote_s3_config(
spec: &RemoteS3EndpointSpec,
) -> Result<aws_sdk_s3::config::Builder, RemoteS3ClientError> {
let Some(credentials) = &spec.credentials else { let Some(credentials) = &spec.credentials else {
return Err(RemoteS3ClientError::MissingCredentials); return Err(RemoteS3ClientError::MissingCredentials);
}; };
@@ -523,7 +523,7 @@ fn validate_ca_pem_bundle(ca_cert_pem: &[u8]) -> Result<(), String> {
Ok(()) Ok(())
} }
pub fn validate_target_ca_pem(ca_cert_pem: &str) -> Result<(), RemoteS3ClientError> { pub(crate) fn validate_target_ca_pem(ca_cert_pem: &str) -> Result<(), RemoteS3ClientError> {
validate_ca_pem_bundle(ca_cert_pem.as_bytes()).map_err(RemoteS3ClientError::InvalidCaPem) validate_ca_pem_bundle(ca_cert_pem.as_bytes()).map_err(RemoteS3ClientError::InvalidCaPem)
} }
@@ -652,10 +652,9 @@ async fn build_aws_s3_http_client_from_tls_path() -> Option<SharedHttpClient> {
#[cfg(test)] #[cfg(test)]
mod tests { mod tests {
use super::*; use super::*;
use aws_smithy_async::time::TimeSource;
use aws_smithy_runtime_api::http::StatusCode as SmithyStatusCode; use aws_smithy_runtime_api::http::StatusCode as SmithyStatusCode;
use std::sync::Mutex; use std::sync::Mutex;
use std::sync::atomic::{AtomicU64, AtomicUsize, Ordering}; use std::sync::atomic::{AtomicUsize, Ordering};
fn spec(endpoint: &str, secure: bool) -> RemoteS3EndpointSpec { fn spec(endpoint: &str, secure: bool) -> RemoteS3EndpointSpec {
RemoteS3EndpointSpec { RemoteS3EndpointSpec {
@@ -825,174 +824,6 @@ mod tests {
); );
} }
#[derive(Clone, Debug)]
struct ClockSkewTimeSource(Arc<AtomicU64>);
impl TimeSource for ClockSkewTimeSource {
fn now(&self) -> SystemTime {
SystemTime::UNIX_EPOCH + Duration::from_secs(self.0.load(Ordering::SeqCst))
}
}
#[derive(Clone, Debug)]
struct ClockSkewConnector {
request_headers: RecordedHeaders,
error_code: &'static str,
skew_seconds: i64,
clock: ClockSkewTimeSource,
}
fn recorded_header<'a>(headers: &'a [(String, String)], name: &str) -> &'a str {
headers
.iter()
.find(|(key, _)| key.eq_ignore_ascii_case(name))
.map(|(_, value)| value.as_str())
.unwrap_or_else(|| panic!("signed request must contain {name}"))
}
fn signing_time(headers: &[(String, String)]) -> chrono::NaiveDateTime {
chrono::NaiveDateTime::parse_from_str(recorded_header(headers, "x-amz-date"), "%Y%m%dT%H%M%SZ")
.expect("SDK signing timestamp must use the SigV4 format")
}
impl SmithyHttpConnector for ClockSkewConnector {
fn call(&self, request: HttpRequest) -> HttpConnectorFuture {
let mut headers = self.request_headers.lock().expect("clock skew request capture lock");
assert!(headers.len() < 3, "clock skew fixture must not exceed two GET attempts and one HEAD");
headers.push(
request
.headers()
.iter()
.map(|(key, value)| (key.to_string(), value.to_string()))
.collect(),
);
let server_time = chrono::DateTime::<chrono::Utc>::from(self.clock.now()).naive_utc()
+ chrono::Duration::seconds(self.skew_seconds);
let (status, body) = if headers.len() == 1 {
(
403,
format!("<Error><Code>{}</Code><Message>Clock skew fixture</Message></Error>", self.error_code),
)
} else {
(200, String::new())
};
let response = http::Response::builder()
.status(status)
.header("date", server_time.format("%a, %d %b %Y %H:%M:%S GMT").to_string())
.header("content-type", "application/xml")
.header("content-length", body.len())
.body(SdkBody::from(body))
.expect("clock skew fixture response");
HttpConnectorFuture::ready(Ok(HttpResponse::try_from(response).expect("Smithy fixture response")))
}
}
async fn clock_skew_client(
error_code: &'static str,
skew_seconds: i64,
retry: RemoteS3RetryPolicy,
) -> (S3Client, RecordedHeaders, ClockSkewTimeSource) {
let headers: RecordedHeaders = Arc::new(Mutex::new(Vec::new()));
let clock = ClockSkewTimeSource(Arc::new(AtomicU64::new(1_700_000_000)));
let connector = SharedHttpConnector::new(ClockSkewConnector {
request_headers: Arc::clone(&headers),
error_code,
skew_seconds,
clock: clock.clone(),
});
let mut spec = spec("s3.example.com", true);
spec.retry = retry;
let config = build_remote_s3_config(&spec)
.await
.expect("clock skew fixture uses the production outbound configuration")
.http_client(http_client_fn(move |_settings, _components| connector.clone()))
.time_source(clock.clone())
.build();
(S3Client::from_conf(config), headers, clock)
}
#[tokio::test(start_paused = true)]
async fn remote_s3_clock_skew_retries_resign_and_seed_next_operation() {
for error_code in ["RequestTimeTooSkewed", "SignatureDoesNotMatch"] {
for skew_seconds in [-600, 600] {
let (client, headers, clock) = clock_skew_client(error_code, skew_seconds, REPLICATION_TARGET_RETRY_POLICY).await;
let initial = chrono::DateTime::<chrono::Utc>::from(clock.now()).naive_utc();
client
.get_object()
.bucket("bucket")
.key("object")
.send()
.await
.expect("clock skew GET must retry successfully");
assert_eq!(
headers.lock().expect("captured requests").len(),
2,
"{error_code}: GET needs exactly one retry"
);
clock.0.fetch_add(17, Ordering::SeqCst);
// SDK signing time is independent of Tokio's retry/scheduler clock.
tokio::time::advance(Duration::from_secs(61)).await;
client
.head_bucket()
.bucket("bucket")
.send()
.await
.expect("subsequent HEAD must use the client's cached skew");
let headers = headers.lock().expect("captured signed requests");
assert_eq!(headers.len(), 3, "subsequent operation must succeed on its first attempt");
assert_eq!(signing_time(&headers[0]), initial, "the first attempt must use the injected clock");
assert_eq!(
signing_time(&headers[1]),
initial + chrono::Duration::seconds(skew_seconds),
"{error_code}: retry must apply the measured offset exactly"
);
assert_eq!(
signing_time(&headers[2]),
initial + chrono::Duration::seconds(skew_seconds + 17),
"{error_code}: the next operation must apply cached skew to the advanced signing clock"
);
let signature = |index: usize| {
recorded_header(&headers[index], "authorization")
.rsplit_once("Signature=")
.expect("SigV4 authorization contains a signature")
.1
};
assert_ne!(
signature(0),
signature(1),
"{error_code}: retry must be signed again after adjusting its date"
);
}
}
}
#[tokio::test(start_paused = true)]
async fn remote_s3_clock_skew_respects_one_attempt_policy() {
use aws_smithy_types::error::metadata::ProvideErrorMetadata;
for error_code in ["RequestTimeTooSkewed", "SignatureDoesNotMatch"] {
for retry in [
RemoteS3RetryPolicy::Disabled,
RemoteS3RetryPolicy::Standard { max_attempts: 1 },
] {
let (client, headers, _clock) = clock_skew_client(error_code, 600, retry).await;
let error = client
.get_object()
.bucket("bucket")
.key("object")
.send()
.await
.expect_err("clock skew must not override the caller's one-attempt budget");
assert_eq!(error.as_service_error().and_then(ProvideErrorMetadata::code), Some(error_code));
assert_eq!(
headers.lock().expect("captured requests").len(),
1,
"{error_code}: {retry:?} must send exactly one request"
);
}
}
}
#[test] #[test]
fn path_style_auto_and_path_force_path_style() { fn path_style_auto_and_path_force_path_style() {
assert!(PathStyle::Auto.force_path_style()); assert!(PathStyle::Auto.force_path_style());
@@ -46,7 +46,7 @@ use super::replication_storage_boundary::{
HTTPPreconditions, ObjectInfo, ObjectOptions, ObjectToDelete, ReplicationDeletedObject, ReplicationObjectIO, HTTPPreconditions, ObjectInfo, ObjectOptions, ObjectToDelete, ReplicationDeletedObject, ReplicationObjectIO,
ReplicationStorage, ReplicationStorage,
}; };
use super::replication_target_boundary::{BucketTargetError, ReplicationTargetStore, replication_object_is_ssec_encrypted}; use super::replication_target_boundary::{ReplicationTargetStore, replication_object_is_ssec_encrypted};
use super::replication_versioning_boundary::ReplicationVersioningStore; use super::replication_versioning_boundary::ReplicationVersioningStore;
use super::runtime_boundary as runtime_sources; use super::runtime_boundary as runtime_sources;
use futures_util::stream::{self, StreamExt}; use futures_util::stream::{self, StreamExt};
@@ -3084,23 +3084,6 @@ pub async fn queue_replication_heal(bucket: &str, oi: ObjectInfo, retry_count: u
let tgts = match ReplicationTargetStore::list_bucket_targets(bucket).await { let tgts = match ReplicationTargetStore::list_bucket_targets(bucket).await {
Ok(targets) => Some(targets), Ok(targets) => Some(targets),
// A bucket whose persisted target configuration cannot be decoded has
// an unknown target set, not an empty one: scheduling against `None`
// here would drop every heal for it without a trace
// (rustfs/backlog#2282). Report it missed so the object is retried
// once the configuration is readable again.
Err(BucketTargetError::BucketRemoteTargetsUnreadable { .. }) => {
warn!(
event = EVENT_REPLICATION_CONFIG_LOOKUP_SKIPPED,
component = LOG_COMPONENT_ECSTORE,
subsystem = LOG_SUBSYSTEM_REPLICATION,
bucket,
reason = "target_config_unreadable",
"Bucket replication targets are unreadable; replication heal queue fails closed"
);
return ReplicationQueueAdmission::Missed;
}
Err(err) => { Err(err) => {
debug!( debug!(
event = EVENT_REPLICATION_CONFIG_LOOKUP_SKIPPED, event = EVENT_REPLICATION_CONFIG_LOOKUP_SKIPPED,
@@ -15,8 +15,7 @@
use std::collections::HashMap; use std::collections::HashMap;
use std::sync::Arc; use std::sync::Arc;
pub(crate) use crate::bucket::bucket_target_sys::BucketTargetError; use crate::bucket::bucket_target_sys::{BucketTargetError, BucketTargetSys};
use crate::bucket::bucket_target_sys::BucketTargetSys;
use aws_sdk_s3::operation::head_object::HeadObjectOutput; use aws_sdk_s3::operation::head_object::HeadObjectOutput;
use aws_sdk_s3::types::{ObjectLockLegalHoldStatus, ObjectLockRetentionMode}; use aws_sdk_s3::types::{ObjectLockLegalHoldStatus, ObjectLockRetentionMode};
use http::HeaderMap; use http::HeaderMap;
@@ -203,7 +203,7 @@ mod tests {
use parking_lot::Mutex; use parking_lot::Mutex;
use std::collections::BTreeMap; use std::collections::BTreeMap;
fn encode_context(context: &BTreeMap<String, String>) -> String { fn encode_context(context: &HashMap<String, String>) -> String {
let ordered = context.iter().collect::<BTreeMap<_, _>>(); let ordered = context.iter().collect::<BTreeMap<_, _>>();
serde_json::to_string(&ordered).expect("context serializes") serde_json::to_string(&ordered).expect("context serializes")
} }
-19
View File
@@ -30,7 +30,6 @@ use rustfs_protos::{
ChannelClass, create_new_channel, get_channel_for_class, ChannelClass, create_new_channel, get_channel_for_class,
proto_gen::node_service::{ proto_gen::node_service::{
heal_control_service_client::HealControlServiceClient, node_service_client::NodeServiceClient, heal_control_service_client::HealControlServiceClient, node_service_client::NodeServiceClient,
scanner_control_service_client::ScannerControlServiceClient,
tier_mutation_control_service_client::TierMutationControlServiceClient, tier_mutation_control_service_client::TierMutationControlServiceClient,
}, },
}; };
@@ -61,24 +60,6 @@ pub async fn node_service_time_out_client(
node_service_time_out_client_for_class(addr, interceptor, ChannelClass::Control).await node_service_time_out_client_for_class(addr, interceptor, ChannelClass::Control).await
} }
pub(crate) async fn scanner_control_time_out_client(
addr: &str,
interceptor: TonicInterceptor,
) -> crate::error::Result<ScannerControlServiceClient<InterceptedService<AuthenticatedChannel, TonicInterceptor>>> {
let interceptor = interceptor.with_rpc_audience(addr)?;
let channel = match runtime_sources::cached_node_channel(addr).await {
Some(channel) => channel,
None => create_new_channel(addr)
.await
.map_err(|err| crate::error::Error::other(err.to_string()))?,
};
let channel = ReplayScopeChannel::new(channel, interceptor.replay_scope_audience());
let limit = rustfs_protos::scoped_dirty_usage::SCOPED_DIRTY_USAGE_MAX_REQUEST_BYTES as usize;
Ok(ScannerControlServiceClient::with_interceptor(channel, interceptor)
.max_decoding_message_size(limit)
.max_encoding_message_size(limit))
}
pub async fn heal_control_time_out_client( pub async fn heal_control_time_out_client(
addr: &str, addr: &str,
interceptor: TonicInterceptor, interceptor: TonicInterceptor,
@@ -2050,53 +2050,6 @@ impl PeerRestClient {
.await .await
} }
/// Probe only: scoped ACK production requires a durable per-bucket proof.
pub async fn scanner_scoped_dirty_usage_capability(
&self,
owner_id: String,
instance_id: String,
entries: Vec<rustfs_protos::proto_gen::node_service::ScannerScopedDirtyUsageEntry>,
) -> Result<bool> {
use rustfs_protos::scoped_dirty_usage::*;
let payload = rustfs_protos::proto_gen::node_service::ScannerScopedDirtyUsageAckRequest {
challenge: Uuid::new_v4().as_bytes().to_vec().into(),
protocol_version: SCOPED_DIRTY_USAGE_PROTOCOL_VERSION,
owner_id,
instance_id,
scope: SCOPED_DIRTY_USAGE_BUCKET_SCOPE,
probe_only: true,
entries,
};
let canonical = canonical_scoped_dirty_usage_request(&payload).map_err(|err| Error::other(err.to_string()))?;
self.finalize_result(
async {
let mut client = super::client::scanner_control_time_out_client(
&self.grid_host,
TonicInterceptor::Signature(gen_tonic_signature_interceptor()),
)
.await?;
let mut request = Request::new(payload.clone());
set_tonic_canonical_body_digest(&mut request, &canonical)?;
let response = client.scanner_scoped_dirty_usage_ack(request).await?.into_inner();
let body = canonical_scoped_dirty_usage_response(&canonical, &response)
.map_err(|_| Error::other("scoped dirty usage capability response is too large"))?;
verify_tonic_rpc_response_proof(&body, response.response_proof.as_ref())?;
if response.protocol_version != SCOPED_DIRTY_USAGE_PROTOCOL_VERSION
|| response.owner_id != payload.owner_id
|| response.instance_id != payload.instance_id
|| response.max_entries != SCOPED_DIRTY_USAGE_MAX_ENTRIES
|| response.max_request_bytes != SCOPED_DIRTY_USAGE_MAX_REQUEST_BYTES
|| response.cleared != 0
{
return Err(Error::other("scoped dirty usage capability response does not match request"));
}
Ok(response.supported)
}
.await,
)
.await
}
pub async fn acknowledge_scanner_dirty_usage(&self, instance_id: String, generation: u64) -> Result<ScannerPeerActivity> { pub async fn acknowledge_scanner_dirty_usage(&self, instance_id: String, generation: u64) -> Result<ScannerPeerActivity> {
let result = self let result = self
.scanner_activity_request_with_protocol(instance_id.clone(), generation, SCANNER_ACTIVITY_PROTOCOL_VERSION) .scanner_activity_request_with_protocol(instance_id.clone(), generation, SCANNER_ACTIVITY_PROTOCOL_VERSION)
-4
View File
@@ -5493,7 +5493,6 @@ where
fence.ensure_held()?; fence.ensure_held()?;
let mut opts = ObjectOptions { let mut opts = ObjectOptions {
max_parity: true, max_parity: true,
write_completion: crate::object_api::WriteCompletion::TailDrained,
no_lock: true, no_lock: true,
http_preconditions: Some(pool_meta_cas_preconditions(token, object)?), http_preconditions: Some(pool_meta_cas_preconditions(token, object)?),
..Default::default() ..Default::default()
@@ -14413,7 +14412,6 @@ impl ECStore {
encoded.clone(), encoded.clone(),
&ObjectOptions { &ObjectOptions {
max_parity: true, max_parity: true,
write_completion: crate::object_api::WriteCompletion::TailDrained,
http_preconditions: Some(HTTPPreconditions { http_preconditions: Some(HTTPPreconditions {
if_none_match: Some("*".to_string()), if_none_match: Some("*".to_string()),
..Default::default() ..Default::default()
@@ -14568,7 +14566,6 @@ impl ECStore {
encoded, encoded,
&ObjectOptions { &ObjectOptions {
max_parity: true, max_parity: true,
write_completion: crate::object_api::WriteCompletion::TailDrained,
http_preconditions: Some(http_preconditions), http_preconditions: Some(http_preconditions),
..Default::default() ..Default::default()
}, },
@@ -14960,7 +14957,6 @@ impl ECStore {
encoded, encoded,
&ObjectOptions { &ObjectOptions {
max_parity: true, max_parity: true,
write_completion: crate::object_api::WriteCompletion::TailDrained,
http_preconditions: Some(HTTPPreconditions { http_preconditions: Some(HTTPPreconditions {
if_match: Some(etag), if_match: Some(etag),
..Default::default() ..Default::default()
+8 -77
View File
@@ -317,22 +317,6 @@ impl DiskStoreRenameDataExt for LocalDiskWrapper {
dst_path: &str, dst_path: &str,
external_guard: Option<Arc<dyn Send + Sync>>, external_guard: Option<Arc<dyn Send + Sync>>,
) -> Result<RenameDataResp> { ) -> Result<RenameDataResp> {
self.rename_data_observed(src_volume, src_path, fi, dst_volume, dst_path, external_guard)
.await
.result
}
}
impl LocalDiskWrapper {
pub(in crate::disk) async fn rename_data_observed(
&self,
src_volume: &str,
src_path: &str,
fi: &FileInfo,
dst_volume: &str,
dst_path: &str,
external_guard: Option<Arc<dyn Send + Sync>>,
) -> super::RenameDataObservation {
let operation = self.clone(); let operation = self.clone();
let src_volume = src_volume.to_owned(); let src_volume = src_volume.to_owned();
let src_path = src_path.to_owned(); let src_path = src_path.to_owned();
@@ -349,35 +333,22 @@ impl LocalDiskWrapper {
} else { } else {
get_max_timeout_duration() get_max_timeout_duration()
}; };
let observed = run_owned_mutation(external_guard, move || async move { run_owned_mutation(external_guard, move || async move {
let mut preflight_rejection = None; operation
let result = operation
.track_disk_health_mutation( .track_disk_health_mutation(
"rename_data", "rename_data",
DiskMetricMutation::Write, DiskMetricMutation::Write,
|| async { || async {
// Preserve the former DiskAPI future's single boxing boundary. operation
let observed = .disk
Box::pin( .rename_data_borrowed(&src_volume, &src_path, &fi, &dst_volume, &dst_path)
operation .await
.disk
.rename_data_observed(&src_volume, &src_path, &fi, &dst_volume, &dst_path),
)
.await;
preflight_rejection = observed.preflight_rejection;
observed.result
}, },
timeout_duration, timeout_duration,
) )
.await; .await
// Health tracking must observe the real disk error, not an Ok tuple.
Ok(super::RenameDataObservation {
result,
preflight_rejection,
})
}) })
.await; .await
observed.unwrap_or_else(|error| super::RenameDataObservation::unknown(Err(error)))
} }
} }
@@ -2617,46 +2588,6 @@ mod tests {
assert_eq!(wrapper.metrics_snapshot().api_calls.get("unknown"), Some(&1)); assert_eq!(wrapper.metrics_snapshot().api_calls.get("unknown"), Some(&1));
} }
#[tokio::test]
async fn rename_preflight_evidence_preserves_health_errors_and_owned_reply() {
for source_exists in [false, true] {
for guarded in [false, true] {
let dir = tempfile::tempdir().expect("temp dir should be created");
let endpoint = Endpoint::try_from(dir.path().to_str().expect("temp dir should be valid UTF-8"))
.expect("endpoint should parse");
let disk = Arc::new(LocalDisk::new(&endpoint, false).await.expect("local disk should be created"));
if source_exists {
disk.make_volume("source").await.expect("source volume should exist");
}
let wrapper = LocalDiskWrapper::new(disk, false);
let drops = Arc::new(std::sync::atomic::AtomicUsize::new(0));
let external_guard = guarded.then(|| Arc::new(DropProbe(Arc::clone(&drops))) as Arc<dyn Send + Sync>);
let mut file_info = FileInfo::new("object", 1, 0);
file_info.mod_time = Some(::time::OffsetDateTime::now_utc());
file_info.erasure.index = 1;
let observed = wrapper
.rename_data_observed("source", "object", &file_info, "missing-destination", "object", external_guard)
.await;
assert!(observed.rejected_before_publication(), "normal access rejection must carry proof");
assert!(matches!(observed.result, Err(DiskError::VolumeNotFound)));
let snapshot = wrapper.metrics_snapshot();
assert_eq!(snapshot.api_calls.get("rename_data"), Some(&1));
assert_eq!(snapshot.total_writes, 0, "health tracking must not observe the rejection as Ok");
assert_eq!(drops.load(Ordering::SeqCst), usize::from(guarded));
wrapper.health.force_runtime_state_for_test(RuntimeDriveHealthState::Offline);
let observed = wrapper
.rename_data_observed("source", "object", &file_info, "missing-destination", "object", None)
.await;
assert!(!observed.rejected_before_publication(), "wrapper errors carry no local preflight proof");
assert!(matches!(observed.result, Err(DiskError::FaultyDisk)));
let snapshot = wrapper.metrics_snapshot();
assert_eq!(snapshot.total_errors_availability, 1);
assert_eq!(snapshot.total_writes, 0);
}
}
}
#[tokio::test] #[tokio::test]
async fn local_disk_health_wrapper_counts_returned_availability_errors() { async fn local_disk_health_wrapper_counts_returned_availability_errors() {
let dir = tempfile::tempdir().expect("temp dir should be created"); let dir = tempfile::tempdir().expect("temp dir should be created");
File diff suppressed because it is too large Load Diff
File diff suppressed because it is too large Load Diff
-49
View File
@@ -75,25 +75,6 @@ use time::OffsetDateTime;
use tokio::io::{AsyncRead, AsyncWrite}; use tokio::io::{AsyncRead, AsyncWrite};
use uuid::Uuid; use uuid::Uuid;
/// Local preflight evidence stays outside DiskAPI and the RPC response format.
pub(crate) struct RenameDataObservation {
pub(crate) result: Result<RenameDataResp>,
preflight_rejection: Option<local::LocalRenamePreflightRejection>,
}
impl RenameDataObservation {
fn unknown(result: Result<RenameDataResp>) -> Self {
Self {
result,
preflight_rejection: None,
}
}
pub(crate) fn rejected_before_publication(&self) -> bool {
self.result.is_err() && self.preflight_rejection.is_some()
}
}
const QUOTA_MUTATION_FENCE_PREFIX: &str = "tmp/quota-mutation-fences/"; const QUOTA_MUTATION_FENCE_PREFIX: &str = "tmp/quota-mutation-fences/";
pub(crate) const QUOTA_MUTATION_FENCE_METADATA_SUFFIX: &str = "quota-mutation-fence-token"; pub(crate) const QUOTA_MUTATION_FENCE_METADATA_SUFFIX: &str = "quota-mutation-fence-token";
@@ -730,36 +711,6 @@ impl Disk {
.await .await
} }
pub(crate) async fn rename_data_borrowed_with_fence_observed(
&self,
src_volume: &str,
src_path: &str,
fi: &FileInfo,
dst_volume: &str,
dst_path: &str,
scanner_publication_lease_token: Option<Uuid>,
) -> RenameDataObservation {
match self {
Disk::Local(local_disk) => {
local_disk
.rename_data_observed(src_volume, src_path, fi, dst_volume, dst_path, None)
.await
}
Disk::Remote(remote_disk) => RenameDataObservation::unknown(
remote_disk
.rename_data_borrowed_with_fence(
src_volume,
src_path,
fi,
dst_volume,
dst_path,
scanner_publication_lease_token,
)
.await,
),
}
}
pub(crate) async fn rename_data_borrowed_with_fence( pub(crate) async fn rename_data_borrowed_with_fence(
&self, &self,
src_volume: &str, src_volume: &str,
-19
View File
@@ -870,18 +870,6 @@ impl TierFreeVersionReceiptSink {
} }
} }
/// Internal PUT completion boundary; this does not change fsync or write quorum.
#[doc(hidden)]
#[derive(Debug, Default, Clone, Copy, PartialEq, Eq)]
pub enum WriteCompletion {
/// Return at write quorum when the commit owner can retain its guards.
#[default]
Quorum,
/// Drain the rename fan-out before returning. Minority failures still heal
/// after a successful quorum commit; this does not require every disk to succeed.
TailDrained,
}
#[derive(Default, Clone)] #[derive(Default, Clone)]
pub struct ObjectOptions { pub struct ObjectOptions {
// Use the maximum parity (N/2), used when saving server configuration files // Use the maximum parity (N/2), used when saving server configuration files
@@ -908,10 +896,6 @@ pub struct ObjectOptions {
/// Persisted bucket incarnation observed before authorization. /// Persisted bucket incarnation observed before authorization.
pub expected_bucket_incarnation_id: Option<Uuid>, pub expected_bucket_incarnation_id: Option<Uuid>,
pub no_lock: bool, pub no_lock: bool,
/// Control-plane writers that immediately read or CAS the same namespace
/// key use TailDrained without changing namespace lock ownership.
#[doc(hidden)]
pub write_completion: WriteCompletion,
/// True when an upper layer already holds the object read lock before /// True when an upper layer already holds the object read lock before
/// forwarding a no_lock read to the set layer. /// forwarding a no_lock read to the set layer.
pub metadata_cache_safe: bool, pub metadata_cache_safe: bool,
@@ -956,9 +940,6 @@ pub struct ObjectOptions {
pub preserve_etag: Option<String>, pub preserve_etag: Option<String>,
pub metadata_chg: bool, pub metadata_chg: bool,
pub http_preconditions: Option<HTTPPreconditions>, pub http_preconditions: Option<HTTPPreconditions>,
/// Internal create-only writes may also preserve an acknowledged deletion.
/// Evaluated with `http_preconditions` under the namespace commit lock.
pub preserve_delete_marker: bool,
pub delete_replication: Option<ReplicationState>, pub delete_replication: Option<ReplicationState>,
pub delete_replication_config_snapshot: Option<Arc<DeleteReplicationConfigSnapshot>>, pub delete_replication_config_snapshot: Option<Arc<DeleteReplicationConfigSnapshot>>,
-112
View File
@@ -78,21 +78,6 @@ pub(crate) struct ScannerPublicationLeaseEntry {
pub(crate) _operation_guard: OwnedRwLockReadGuard<()>, pub(crate) _operation_guard: OwnedRwLockReadGuard<()>,
} }
pub(crate) struct NamespaceCommitGuard {
ctx: Arc<InstanceContext>,
counted: bool,
}
impl Drop for NamespaceCommitGuard {
fn drop(&mut self) {
if self.counted {
// Publish the new generation before a zero-pending publication probe.
self.ctx.advance_namespace_commit_generation();
self.ctx.namespace_commits.fetch_sub(1, Ordering::AcqRel);
}
}
}
/// Runtime state owned by a single `ECStore` instance. /// Runtime state owned by a single `ECStore` instance.
/// ///
/// This is intentionally minimal in the first migration slice; subsequent /// This is intentionally minimal in the first migration slice; subsequent
@@ -224,13 +209,9 @@ pub struct InstanceContext {
/// Last storage-owned movement snapshot observed under the operation /// Last storage-owned movement snapshot observed under the operation
/// gate. SetDisks cache writers fail closed until ECStore refreshes it. /// gate. SetDisks cache writers fail closed until ECStore refreshes it.
scanner_publication_state: AtomicU8, scanner_publication_state: AtomicU8,
namespace_commits: AtomicU64,
namespace_commit_generation: AtomicU64,
/// Resolves object-encryption material at the application boundary. /// Resolves object-encryption material at the application boundary.
object_encryption_resolver: OnceLock<Arc<dyn ObjectEncryptionResolver>>, object_encryption_resolver: OnceLock<Arc<dyn ObjectEncryptionResolver>>,
tier_delete_journal_recovery_stores: std::sync::Mutex<HashSet<Uuid>>, tier_delete_journal_recovery_stores: std::sync::Mutex<HashSet<Uuid>>,
#[cfg(test)]
suppress_tier_delete_journal_recovery: bool,
transition_transaction_recovery_stores: std::sync::Mutex<HashSet<Uuid>>, transition_transaction_recovery_stores: std::sync::Mutex<HashSet<Uuid>>,
tier_delete_journal_recovery_wakeup: tokio::sync::Notify, tier_delete_journal_recovery_wakeup: tokio::sync::Notify,
} }
@@ -275,12 +256,8 @@ impl InstanceContext {
data_movement_generation_exhausted: AtomicBool::new(false), data_movement_generation_exhausted: AtomicBool::new(false),
data_movement_generation_notify: Arc::new(Notify::new()), data_movement_generation_notify: Arc::new(Notify::new()),
scanner_publication_state: AtomicU8::new(SCANNER_PUBLICATION_STATE_UNKNOWN), scanner_publication_state: AtomicU8::new(SCANNER_PUBLICATION_STATE_UNKNOWN),
namespace_commits: AtomicU64::new(0),
namespace_commit_generation: AtomicU64::new(0),
object_encryption_resolver: OnceLock::new(), object_encryption_resolver: OnceLock::new(),
tier_delete_journal_recovery_stores: std::sync::Mutex::new(HashSet::new()), tier_delete_journal_recovery_stores: std::sync::Mutex::new(HashSet::new()),
#[cfg(test)]
suppress_tier_delete_journal_recovery: false,
transition_transaction_recovery_stores: std::sync::Mutex::new(HashSet::new()), transition_transaction_recovery_stores: std::sync::Mutex::new(HashSet::new()),
tier_delete_journal_recovery_wakeup: tokio::sync::Notify::new(), tier_delete_journal_recovery_wakeup: tokio::sync::Notify::new(),
} }
@@ -408,36 +385,6 @@ impl InstanceContext {
&& self.scanner_publication_state.load(Ordering::Acquire) == SCANNER_PUBLICATION_STATE_ALLOWED && self.scanner_publication_state.load(Ordering::Acquire) == SCANNER_PUBLICATION_STATE_ALLOWED
} }
pub(crate) fn begin_namespace_commit(self: &Arc<Self>) -> Arc<NamespaceCommitGuard> {
let counted = self
.namespace_commits
.fetch_update(Ordering::AcqRel, Ordering::Acquire, |count| count.checked_add(1))
.is_ok();
if counted {
self.advance_namespace_commit_generation();
} else {
self.namespace_commit_generation.store(u64::MAX, Ordering::Release);
}
Arc::new(NamespaceCommitGuard {
ctx: Arc::clone(self),
counted,
})
}
fn advance_namespace_commit_generation(&self) {
let _ = self
.namespace_commit_generation
.fetch_update(Ordering::AcqRel, Ordering::Acquire, |generation| Some(generation.saturating_add(1)));
}
pub(crate) fn namespace_commit_generation(&self) -> u64 {
self.namespace_commit_generation.load(Ordering::Acquire)
}
pub(crate) fn namespace_commits_pending(&self) -> bool {
self.namespace_commits.load(Ordering::Acquire) != 0 || self.namespace_commit_generation() == u64::MAX
}
pub(crate) fn set_scanner_publication_state(&self, blocked: bool) { pub(crate) fn set_scanner_publication_state(&self, blocked: bool) {
self.scanner_publication_state.store( self.scanner_publication_state.store(
if blocked { if blocked {
@@ -693,21 +640,12 @@ impl InstanceContext {
} }
pub(crate) fn mark_tier_delete_journal_recovery_started(&self, store_id: Uuid) -> bool { pub(crate) fn mark_tier_delete_journal_recovery_started(&self, store_id: Uuid) -> bool {
#[cfg(test)]
if self.suppress_tier_delete_journal_recovery {
return false;
}
self.tier_delete_journal_recovery_stores self.tier_delete_journal_recovery_stores
.lock() .lock()
.unwrap_or_else(std::sync::PoisonError::into_inner) .unwrap_or_else(std::sync::PoisonError::into_inner)
.insert(store_id) .insert(store_id)
} }
#[cfg(test)]
pub(crate) fn suppress_tier_delete_journal_recovery_for_test(&mut self) {
self.suppress_tier_delete_journal_recovery = true;
}
pub(crate) fn mark_transition_transaction_recovery_started(&self, store_id: Uuid) -> bool { pub(crate) fn mark_transition_transaction_recovery_started(&self, store_id: Uuid) -> bool {
self.transition_transaction_recovery_stores self.transition_transaction_recovery_stores
.lock() .lock()
@@ -818,50 +756,6 @@ pub fn bootstrap_ctx() -> Arc<InstanceContext> {
mod tests { mod tests {
use super::*; use super::*;
#[test]
fn namespace_commit_guards_are_instance_local_and_count_until_last_owner() {
let first = Arc::new(InstanceContext::new());
let other = Arc::new(InstanceContext::new());
first.set_scanner_publication_state(false);
other.set_scanner_publication_state(false);
assert!(first.scanner_publication_state_allowed());
let one = first.begin_namespace_commit();
let shared_owner = Arc::clone(&one);
let two = first.begin_namespace_commit();
assert!(first.namespace_commits_pending());
assert!(first.scanner_publication_state_allowed(), "pending writes must not block scan admission");
assert_eq!(first.namespace_commit_generation(), 2);
assert!(!other.namespace_commits_pending());
assert_eq!(other.namespace_commit_generation(), 0);
assert!(other.scanner_publication_state_allowed());
drop(one);
assert_eq!(first.namespace_commit_generation(), 2);
drop(shared_owner);
assert!(first.namespace_commits_pending());
assert_eq!(first.namespace_commit_generation(), 3);
drop(two);
assert!(!first.namespace_commits_pending());
assert_eq!(first.namespace_commit_generation(), 4);
assert!(first.scanner_publication_state_allowed());
}
#[test]
fn namespace_commit_counter_exhaustion_keeps_publication_blocked() {
for (count, generation) in [(0, u64::MAX - 1), (u64::MAX, 0)] {
let ctx = Arc::new(InstanceContext::new());
ctx.set_scanner_publication_state(false);
ctx.namespace_commits.store(count, Ordering::Release);
ctx.namespace_commit_generation.store(generation, Ordering::Release);
let guard = ctx.begin_namespace_commit();
assert!(ctx.namespace_commits_pending());
assert_eq!(ctx.namespace_commit_generation(), u64::MAX);
drop(guard);
assert!(ctx.namespace_commits_pending());
assert_eq!(ctx.namespace_commit_generation(), u64::MAX);
assert_eq!(ctx.namespace_commits.load(Ordering::Acquire), count);
}
}
// The SetupType inputs must derive the exact (is_erasure, // The SetupType inputs must derive the exact (is_erasure,
// is_dist_erasure, is_erasure_sd) triples that the original three // is_dist_erasure, is_erasure_sd) triples that the original three
// process-global erasure bools produced via update_erasure_type(). // process-global erasure bools produced via update_erasure_type().
@@ -1179,12 +1073,6 @@ mod tests {
assert!(!ctx_a.mark_tier_delete_journal_recovery_started(store_a)); assert!(!ctx_a.mark_tier_delete_journal_recovery_started(store_a));
assert!(ctx_a.mark_tier_delete_journal_recovery_started(store_b)); assert!(ctx_a.mark_tier_delete_journal_recovery_started(store_b));
assert!(ctx_b.mark_tier_delete_journal_recovery_started(store_a)); assert!(ctx_b.mark_tier_delete_journal_recovery_started(store_a));
let mut manual_ctx = InstanceContext::new();
manual_ctx.suppress_tier_delete_journal_recovery_for_test();
assert!(!manual_ctx.mark_tier_delete_journal_recovery_started(store_a));
assert!(!manual_ctx.mark_tier_delete_journal_recovery_started(store_b));
assert!(ctx_b.mark_tier_delete_journal_recovery_started(store_b));
} }
#[test] #[test]
+12 -483
View File
@@ -62,27 +62,12 @@ const REMOTE_VERSION_STATE_PROOF_TTL: Duration = Duration::from_secs(30);
const CROSS_POOL_FENCE_SUPPORTED_VERSION: u32 = 2; const CROSS_POOL_FENCE_SUPPORTED_VERSION: u32 = 2;
const TIER_DELETE_JOURNAL_POLICY_SUPPORTED_VERSION: u32 = 3; const TIER_DELETE_JOURNAL_POLICY_SUPPORTED_VERSION: u32 = 3;
const DECOMMISSION_TARGET_FENCE_POLICY_SUPPORTED_VERSION: u32 = 4; const DECOMMISSION_TARGET_FENCE_POLICY_SUPPORTED_VERSION: u32 = 4;
// Keep this synchronized with the version served by node_service. Including
// the local member in the minimum prevents an older coordinator from
// self-authorizing a policy implemented only by newer remote peers.
const LOCAL_CROSS_POOL_FENCE_POLICY_SUPPORTED_VERSION: u32 = 4;
/// Version 5 is reserved for a fleet whose every metadata writer preserves
/// explicit transition version state and destination identity, and implements
/// conditional per-generation `xl.meta` writes with strong readback. The node
/// service must not advertise this version until the conditional writer from
/// rustfs/backlog#684 is available.
const LEGACY_TRANSITION_STATE_RECONCILE_POLICY_SUPPORTED_VERSION: u32 = 5;
type CrossPoolFencePolicyResult = Result<BTreeMap<String, Uuid>>; type CrossPoolFencePolicyResult = Result<BTreeMap<String, Uuid>>;
fn cross_pool_fence_policy_results( fn cross_pool_fence_policy_results(
peer_epochs: BTreeMap<String, Uuid>, peer_epochs: BTreeMap<String, Uuid>,
minimum_version: u32, minimum_version: u32,
) -> ( ) -> (CrossPoolFencePolicyResult, CrossPoolFencePolicyResult, CrossPoolFencePolicyResult) {
CrossPoolFencePolicyResult,
CrossPoolFencePolicyResult,
CrossPoolFencePolicyResult,
CrossPoolFencePolicyResult,
) {
let journal_result = if minimum_version >= TIER_DELETE_JOURNAL_POLICY_SUPPORTED_VERSION { let journal_result = if minimum_version >= TIER_DELETE_JOURNAL_POLICY_SUPPORTED_VERSION {
Ok(peer_epochs.clone()) Ok(peer_epochs.clone())
} else { } else {
@@ -93,18 +78,7 @@ fn cross_pool_fence_policy_results(
} else { } else {
Err(Error::other("decommission target fence policy capability version is unsupported")) Err(Error::other("decommission target fence policy capability version is unsupported"))
}; };
let legacy_transition_state_reconcile_result = (Ok(peer_epochs), journal_result, decommission_target_fence_result)
if minimum_version >= LEGACY_TRANSITION_STATE_RECONCILE_POLICY_SUPPORTED_VERSION {
Ok(peer_epochs.clone())
} else {
Err(Error::other("legacy transition state reconcile policy capability version is unsupported"))
};
(
Ok(peer_epochs),
journal_result,
decommission_target_fence_result,
legacy_transition_state_reconcile_result,
)
} }
#[derive(Clone, Debug)] #[derive(Clone, Debug)]
@@ -278,21 +252,10 @@ pub(crate) struct TierDeleteJournalFleetProofToken {
_permit: FleetCapabilityProofPermit, _permit: FleetCapabilityProofPermit,
} }
/// Effect-window authority for one legacy transition-state reconciliation.
///
/// The token intentionally cannot be cloned. Its permit keeps the admitted
/// fleet generation alive until the caller finishes the final strong
/// readback, while revocation makes every later validation fail immediately.
pub struct LegacyTransitionStateReconcileFleetProofToken {
token: FleetCapabilityProofToken,
_permit: FleetCapabilityProofPermit,
}
static REMOTE_VERSION_STATE_FLEET_PROOF: OnceLock<std::sync::RwLock<FleetCapabilityProofState>> = OnceLock::new(); static REMOTE_VERSION_STATE_FLEET_PROOF: OnceLock<std::sync::RwLock<FleetCapabilityProofState>> = OnceLock::new();
static CROSS_POOL_FENCE_FLEET_PROOF: OnceLock<std::sync::RwLock<FleetCapabilityProofState>> = OnceLock::new(); static CROSS_POOL_FENCE_FLEET_PROOF: OnceLock<std::sync::RwLock<FleetCapabilityProofState>> = OnceLock::new();
static TIER_DELETE_JOURNAL_FLEET_PROOF: OnceLock<std::sync::RwLock<FleetCapabilityProofState>> = OnceLock::new(); static TIER_DELETE_JOURNAL_FLEET_PROOF: OnceLock<std::sync::RwLock<FleetCapabilityProofState>> = OnceLock::new();
static DECOMMISSION_TARGET_FENCE_FLEET_PROOF: OnceLock<std::sync::RwLock<FleetCapabilityProofState>> = OnceLock::new(); static DECOMMISSION_TARGET_FENCE_FLEET_PROOF: OnceLock<std::sync::RwLock<FleetCapabilityProofState>> = OnceLock::new();
static LEGACY_TRANSITION_STATE_RECONCILE_FLEET_PROOF: OnceLock<std::sync::RwLock<FleetCapabilityProofState>> = OnceLock::new();
static REMOTE_VERSION_STATE_PROBE_TOPOLOGY: OnceLock<String> = OnceLock::new(); static REMOTE_VERSION_STATE_PROBE_TOPOLOGY: OnceLock<String> = OnceLock::new();
fn cross_pool_fence_fleet_proof_slot() -> &'static std::sync::RwLock<FleetCapabilityProofState> { fn cross_pool_fence_fleet_proof_slot() -> &'static std::sync::RwLock<FleetCapabilityProofState> {
@@ -311,10 +274,6 @@ fn decommission_target_fence_fleet_proof_slot() -> &'static std::sync::RwLock<Fl
DECOMMISSION_TARGET_FENCE_FLEET_PROOF.get_or_init(|| std::sync::RwLock::new(FleetCapabilityProofState::default())) DECOMMISSION_TARGET_FENCE_FLEET_PROOF.get_or_init(|| std::sync::RwLock::new(FleetCapabilityProofState::default()))
} }
fn legacy_transition_state_reconcile_fleet_proof_slot() -> &'static std::sync::RwLock<FleetCapabilityProofState> {
LEGACY_TRANSITION_STATE_RECONCILE_FLEET_PROOF.get_or_init(|| std::sync::RwLock::new(FleetCapabilityProofState::default()))
}
fn revoke_fleet_capability_proof_state(state: &mut FleetCapabilityProofState) { fn revoke_fleet_capability_proof_state(state: &mut FleetCapabilityProofState) {
if let Some(proof) = state.proof.take() { if let Some(proof) = state.proof.take() {
proof.generation.revoke(); proof.generation.revoke();
@@ -485,125 +444,6 @@ pub(crate) fn tier_delete_journal_topology_generation(proof: &TierDeleteJournalF
stable_tier_delete_journal_topology_generation(&proof.token.topology_fingerprint) stable_tier_delete_journal_topology_generation(&proof.token.topology_fingerprint)
} }
/// Acquire one non-cloneable authority that must span the complete reconcile
/// effect window, including its final strong readback.
pub async fn acquire_legacy_transition_state_reconcile_fleet_proof() -> Option<LegacyTransitionStateReconcileFleetProofToken> {
let expected_topology = REMOTE_VERSION_STATE_PROBE_TOPOLOGY.get()?;
let proof = {
let state = legacy_transition_state_reconcile_fleet_proof_slot()
.read()
.unwrap_or_else(std::sync::PoisonError::into_inner);
acquire_legacy_transition_state_reconcile_fleet_proof_from(&state, expected_topology, Instant::now())?
};
let observed_peer_epochs = observe_legacy_transition_state_reconcile_fleet(expected_topology).await?;
let state = legacy_transition_state_reconcile_fleet_proof_slot()
.read()
.unwrap_or_else(std::sync::PoisonError::into_inner);
legacy_transition_state_reconcile_fleet_proof_matches_observation_at(
&state,
&proof,
expected_topology,
&observed_peer_epochs,
Instant::now(),
)
.then_some(proof)
}
fn acquire_legacy_transition_state_reconcile_fleet_proof_from(
state: &FleetCapabilityProofState,
expected_topology: &str,
now: Instant,
) -> Option<LegacyTransitionStateReconcileFleetProofToken> {
let token = acquire_fleet_capability_proof_from(state, expected_topology, now)?;
let permit = state.proof.as_ref()?.generation.try_acquire()?;
Some(LegacyTransitionStateReconcileFleetProofToken { token, _permit: permit })
}
async fn observe_legacy_transition_state_reconcile_fleet(expected_topology: &str) -> Option<BTreeMap<String, Uuid>> {
let notification_sys = get_global_notification_sys()?;
let (peer_epochs, minimum_version) = timeout(
REMOTE_VERSION_STATE_PROBE_TIMEOUT,
notification_sys.probe_cross_pool_fence_fleet(expected_topology),
)
.await
.ok()?
.ok()?;
let (_, _, _, reconcile_result) = cross_pool_fence_policy_results(peer_epochs, minimum_version);
reconcile_result.ok()
}
/// Revalidate the exact fleet generation captured by a reconcile token with a
/// fresh synchronous observation. Callers must await this before each
/// conditional metadata write and after the final strong readback.
pub async fn legacy_transition_state_reconcile_fleet_proof_matches(
proof: &LegacyTransitionStateReconcileFleetProofToken,
) -> bool {
let Some(expected_topology) = REMOTE_VERSION_STATE_PROBE_TOPOLOGY.get() else {
return false;
};
legacy_transition_state_reconcile_fleet_proof_matches_with_observer(
legacy_transition_state_reconcile_fleet_proof_slot(),
proof,
expected_topology,
|| observe_legacy_transition_state_reconcile_fleet(expected_topology),
)
.await
}
async fn legacy_transition_state_reconcile_fleet_proof_matches_with_observer<F, Fut>(
slot: &std::sync::RwLock<FleetCapabilityProofState>,
proof: &LegacyTransitionStateReconcileFleetProofToken,
expected_topology: &str,
observe: F,
) -> bool
where
F: FnOnce() -> Fut,
Fut: Future<Output = Option<BTreeMap<String, Uuid>>>,
{
{
let state = slot.read().unwrap_or_else(std::sync::PoisonError::into_inner);
if !legacy_transition_state_reconcile_fleet_proof_matches_at(&state, proof, expected_topology, Instant::now()) {
return false;
}
}
let Some(observed_peer_epochs) = observe().await else {
return false;
};
let state = slot.read().unwrap_or_else(std::sync::PoisonError::into_inner);
legacy_transition_state_reconcile_fleet_proof_matches_observation_at(
&state,
proof,
expected_topology,
&observed_peer_epochs,
Instant::now(),
)
}
fn legacy_transition_state_reconcile_fleet_proof_matches_at(
state: &FleetCapabilityProofState,
proof: &LegacyTransitionStateReconcileFleetProofToken,
expected_topology: &str,
now: Instant,
) -> bool {
proof._permit.generation.is_accepting()
&& fleet_capability_proof_matches_at(state, &proof.token, expected_topology, now)
&& state
.proof
.as_ref()
.is_some_and(|current| Arc::ptr_eq(&current.generation, &proof._permit.generation))
}
fn legacy_transition_state_reconcile_fleet_proof_matches_observation_at(
state: &FleetCapabilityProofState,
proof: &LegacyTransitionStateReconcileFleetProofToken,
expected_topology: &str,
observed_peer_epochs: &BTreeMap<String, Uuid>,
now: Instant,
) -> bool {
legacy_transition_state_reconcile_fleet_proof_matches_at(state, proof, expected_topology, now)
&& proof.token.peer_epochs.as_ref() == observed_peer_epochs
}
#[cfg(all(test, feature = "test-util"))] #[cfg(all(test, feature = "test-util"))]
pub(crate) fn tier_delete_journal_fleet_proof_has_inflight_for_test() -> bool { pub(crate) fn tier_delete_journal_fleet_proof_has_inflight_for_test() -> bool {
let state = tier_delete_journal_fleet_proof_slot() let state = tier_delete_journal_fleet_proof_slot()
@@ -926,7 +766,6 @@ pub fn start_remote_version_state_fleet_probe(topology_fingerprint: String) {
cross_pool_fence_fleet_proof_slot(), cross_pool_fence_fleet_proof_slot(),
tier_delete_journal_fleet_proof_slot(), tier_delete_journal_fleet_proof_slot(),
decommission_target_fence_fleet_proof_slot(), decommission_target_fence_fleet_proof_slot(),
legacy_transition_state_reconcile_fleet_proof_slot(),
] { ] {
mark_fleet_capability_topology_conflict(slot); mark_fleet_capability_topology_conflict(slot);
} }
@@ -959,12 +798,11 @@ pub fn start_remote_version_state_fleet_probe(topology_fingerprint: String) {
.unwrap_or_else(|_| Err(Error::other("cross-pool fence fleet capability probe timed out"))), .unwrap_or_else(|_| Err(Error::other("cross-pool fence fleet capability probe timed out"))),
None => Err(Error::other("cross-pool fence fleet capability notification system is unavailable")), None => Err(Error::other("cross-pool fence fleet capability notification system is unavailable")),
}; };
let (fence_result, journal_result, decommission_target_fence_result, reconcile_result) = match fence_probe { let (fence_result, journal_result, decommission_target_fence_result) = match fence_probe {
Ok((peer_epochs, minimum_version)) => cross_pool_fence_policy_results(peer_epochs, minimum_version), Ok((peer_epochs, minimum_version)) => cross_pool_fence_policy_results(peer_epochs, minimum_version),
Err(err) => { Err(err) => {
let message = err.to_string(); let message = err.to_string();
( (
Err(Error::other(message.clone())),
Err(Error::other(message.clone())), Err(Error::other(message.clone())),
Err(Error::other(message.clone())), Err(Error::other(message.clone())),
Err(Error::other(message)), Err(Error::other(message)),
@@ -980,7 +818,6 @@ pub fn start_remote_version_state_fleet_probe(topology_fingerprint: String) {
revoke_fleet_capability_proof(cross_pool_fence_fleet_proof_slot()); revoke_fleet_capability_proof(cross_pool_fence_fleet_proof_slot());
revoke_fleet_capability_proof(tier_delete_journal_fleet_proof_slot()); revoke_fleet_capability_proof(tier_delete_journal_fleet_proof_slot());
revoke_fleet_capability_proof(decommission_target_fence_fleet_proof_slot()); revoke_fleet_capability_proof(decommission_target_fence_fleet_proof_slot());
revoke_fleet_capability_proof(legacy_transition_state_reconcile_fleet_proof_slot());
} else if let Some(err) = publish_fleet_capability_probe_result( } else if let Some(err) = publish_fleet_capability_probe_result(
remote_version_state_fleet_proof_slot(), remote_version_state_fleet_proof_slot(),
&topology_fingerprint, &topology_fingerprint,
@@ -1043,24 +880,6 @@ pub fn start_remote_version_state_fleet_probe(topology_fingerprint: String) {
"notification capability probe" "notification capability probe"
); );
} }
if !topology_conflict
&& let Some(err) = publish_fleet_capability_probe_result(
legacy_transition_state_reconcile_fleet_proof_slot(),
&topology_fingerprint,
reconcile_result,
Instant::now(),
)
{
debug!(
event = EVENT_NOTIFICATION_CAPABILITY_PROBE,
component = LOG_COMPONENT_ECSTORE,
subsystem = LOG_SUBSYSTEM_NOTIFICATION,
capability = "legacy_transition_state_reconcile_v1",
state = "failed_closed",
error = %err,
"notification capability probe"
);
}
sleep(REMOTE_VERSION_STATE_PROBE_INTERVAL).await; sleep(REMOTE_VERSION_STATE_PROBE_INTERVAL).await;
} }
}); });
@@ -1140,7 +959,7 @@ impl NotificationSys {
client.probe_cross_pool_fence(topology_fingerprint.to_string()).await client.probe_cross_pool_fence(topology_fingerprint.to_string()).await
}); });
let mut peer_epochs = BTreeMap::new(); let mut peer_epochs = BTreeMap::new();
let mut minimum_version = LOCAL_CROSS_POOL_FENCE_POLICY_SUPPORTED_VERSION; let mut minimum_version = u32::MAX;
for result in join_all(probes).await { for result in join_all(probes).await {
let (peer, version, epoch) = result?; let (peer, version, epoch) = result?;
if version < CROSS_POOL_FENCE_SUPPORTED_VERSION { if version < CROSS_POOL_FENCE_SUPPORTED_VERSION {
@@ -1149,6 +968,11 @@ impl NotificationSys {
minimum_version = minimum_version.min(version); minimum_version = minimum_version.min(version);
insert_remote_version_state_peer(&mut peer_epochs, peer, epoch)?; insert_remote_version_state_peer(&mut peer_epochs, peer, epoch)?;
} }
// A single-node deployment has no remote member to lower the local
// policy version advertised by this binary.
if minimum_version == u32::MAX {
minimum_version = DECOMMISSION_TARGET_FENCE_POLICY_SUPPORTED_VERSION;
}
Ok((peer_epochs, minimum_version)) Ok((peer_epochs, minimum_version))
} }
} }
@@ -3366,36 +3190,20 @@ mod tests {
#[test] #[test]
fn cross_pool_policy_versions_authorize_only_their_supported_protocols() { fn cross_pool_policy_versions_authorize_only_their_supported_protocols() {
let peers = BTreeMap::from([("node-b:9000".to_string(), Uuid::new_v4())]); let peers = BTreeMap::from([("node-b:9000".to_string(), Uuid::new_v4())]);
let (generic_v2, journal_v2, decommission_v2, reconcile_v2) = cross_pool_fence_policy_results(peers.clone(), 2); let (generic_v2, journal_v2, decommission_v2) = cross_pool_fence_policy_results(peers.clone(), 2);
assert!(generic_v2.is_ok(), "v2 remains valid for existing cross-pool fencing"); assert!(generic_v2.is_ok(), "v2 remains valid for existing cross-pool fencing");
assert!(journal_v2.is_err(), "a mixed v2/v3 fleet must fail closed for journal-v6 deletion"); assert!(journal_v2.is_err(), "a mixed v2/v3 fleet must fail closed for journal-v6 deletion");
assert!(decommission_v2.is_err(), "v2 cannot authorize the sticky per-target decommission fence"); assert!(decommission_v2.is_err(), "v2 cannot authorize the sticky per-target decommission fence");
assert!(reconcile_v2.is_err(), "v2 cannot authorize legacy transition-state reconciliation");
let (generic_v3, journal_v3, decommission_v3, reconcile_v3) = cross_pool_fence_policy_results(peers.clone(), 3); let (generic_v3, journal_v3, decommission_v3) = cross_pool_fence_policy_results(peers.clone(), 3);
assert!(generic_v3.is_ok()); assert!(generic_v3.is_ok());
assert!(journal_v3.is_ok(), "an all-v3 fleet may authorize journal-v6 deletion"); assert!(journal_v3.is_ok(), "an all-v3 fleet may authorize journal-v6 deletion");
assert!(decommission_v3.is_err(), "v3 members do not understand the per-target decommission fence"); assert!(decommission_v3.is_err(), "v3 members do not understand the per-target decommission fence");
assert!(reconcile_v3.is_err());
let (generic_v4, journal_v4, decommission_v4, reconcile_v4) = let (generic_v4, journal_v4, decommission_v4) = cross_pool_fence_policy_results(peers, 4);
cross_pool_fence_policy_results(peers.clone(), LOCAL_CROSS_POOL_FENCE_POLICY_SUPPORTED_VERSION);
assert!(generic_v4.is_ok()); assert!(generic_v4.is_ok());
assert!(journal_v4.is_ok()); assert!(journal_v4.is_ok());
assert!(decommission_v4.is_ok(), "an all-v4 fleet may create sticky per-target reservations"); assert!(decommission_v4.is_ok(), "an all-v4 fleet may create sticky per-target reservations");
assert!(
reconcile_v4.is_err(),
"the current local policy lacks the conditional xl.meta writer required by reconcile"
);
let (generic_v5, journal_v5, decommission_v5, reconcile_v5) = cross_pool_fence_policy_results(peers, 5);
assert!(generic_v5.is_ok());
assert!(journal_v5.is_ok());
assert!(decommission_v5.is_ok());
assert!(
reconcile_v5.is_ok(),
"only an all-v5 fleet preserves destination identity and conditional reconcile writes"
);
} }
#[test] #[test]
@@ -3650,234 +3458,6 @@ mod tests {
); );
} }
#[test]
fn legacy_transition_state_reconcile_admits_only_compatible_single_and_multi_node_fleets() {
let now = Instant::now();
for peers in [
BTreeMap::new(),
BTreeMap::from([("peer-a".to_string(), Uuid::new_v4()), ("peer-b".to_string(), Uuid::new_v4())]),
] {
let slot = std::sync::RwLock::new(FleetCapabilityProofState::default());
let (_, _, _, result) =
cross_pool_fence_policy_results(peers, LEGACY_TRANSITION_STATE_RECONCILE_POLICY_SUPPORTED_VERSION);
assert!(publish_fleet_capability_probe_result(&slot, "topology-a", result, now).is_none());
let admitted = {
let state = slot.read().expect("reconcile proof slot should not poison");
acquire_legacy_transition_state_reconcile_fleet_proof_from(&state, "topology-a", now)
.expect("an all-compatible fleet should admit reconciliation")
};
let state = slot.read().expect("reconcile proof slot should not poison");
assert!(legacy_transition_state_reconcile_fleet_proof_matches_at(
&state,
&admitted,
"topology-a",
now,
));
}
}
#[test]
fn legacy_transition_state_reconcile_restart_drains_concurrent_effect_windows() {
let slot = std::sync::RwLock::new(FleetCapabilityProofState::default());
let now = Instant::now();
let original_peers = BTreeMap::from([("peer-a".to_string(), Uuid::new_v4())]);
let (_, _, _, original_result) =
cross_pool_fence_policy_results(original_peers, LEGACY_TRANSITION_STATE_RECONCILE_POLICY_SUPPORTED_VERSION);
assert!(publish_fleet_capability_probe_result(&slot, "topology-a", original_result, now).is_none());
let (first, second) = {
let state = slot.read().expect("reconcile proof slot should not poison");
(
acquire_legacy_transition_state_reconcile_fleet_proof_from(&state, "topology-a", now)
.expect("the first reconcile writer should be admitted"),
acquire_legacy_transition_state_reconcile_fleet_proof_from(&state, "topology-a", now)
.expect("the second reconcile writer should be admitted"),
)
};
let restarted_peers = BTreeMap::from([("peer-a".to_string(), Uuid::new_v4())]);
let (_, _, _, restarted_result) =
cross_pool_fence_policy_results(restarted_peers.clone(), LEGACY_TRANSITION_STATE_RECONCILE_POLICY_SUPPORTED_VERSION);
let blocked =
publish_fleet_capability_probe_result(&slot, "topology-a", restarted_result, now + Duration::from_millis(1))
.expect("a restarted member must revoke the old generation and wait for both writers");
assert!(blocked.to_string().contains("previous generation to drain"));
{
let state = slot.read().expect("reconcile proof slot should not poison");
assert!(state.proof.is_none());
assert!(state.draining_generation.is_some());
assert!(!legacy_transition_state_reconcile_fleet_proof_matches_at(
&state,
&first,
"topology-a",
now + Duration::from_millis(1),
));
assert!(!legacy_transition_state_reconcile_fleet_proof_matches_at(
&state,
&second,
"topology-a",
now + Duration::from_millis(1),
));
}
drop(first);
let (_, _, _, still_blocked_result) =
cross_pool_fence_policy_results(restarted_peers.clone(), LEGACY_TRANSITION_STATE_RECONCILE_POLICY_SUPPORTED_VERSION);
assert!(
publish_fleet_capability_probe_result(&slot, "topology-a", still_blocked_result, now + Duration::from_millis(2),)
.is_some(),
"one remaining writer must keep the successor generation closed"
);
drop(second);
let (_, _, _, admitted_result) =
cross_pool_fence_policy_results(restarted_peers, LEGACY_TRANSITION_STATE_RECONCILE_POLICY_SUPPORTED_VERSION);
assert!(
publish_fleet_capability_probe_result(&slot, "topology-a", admitted_result, now + Duration::from_millis(3),)
.is_none(),
"the restarted generation may publish only after every old writer drains"
);
}
#[test]
fn legacy_transition_state_reconcile_fresh_observation_closes_the_polling_window() {
let slot = std::sync::RwLock::new(FleetCapabilityProofState::default());
let now = Instant::now();
let original_peers = BTreeMap::from([("peer-a".to_string(), Uuid::new_v4())]);
let (_, _, _, original_result) =
cross_pool_fence_policy_results(original_peers.clone(), LEGACY_TRANSITION_STATE_RECONCILE_POLICY_SUPPORTED_VERSION);
assert!(publish_fleet_capability_probe_result(&slot, "topology-a", original_result, now).is_none());
let admitted = {
let state = slot.read().expect("reconcile proof slot should not poison");
acquire_legacy_transition_state_reconcile_fleet_proof_from(&state, "topology-a", now)
.expect("the original fleet should admit reconciliation")
};
let restarted_peers = BTreeMap::from([("peer-a".to_string(), Uuid::new_v4())]);
let state = slot.read().expect("reconcile proof slot should not poison");
assert!(
legacy_transition_state_reconcile_fleet_proof_matches_at(&state, &admitted, "topology-a", now),
"the periodic cache has not observed the restart yet"
);
assert!(!legacy_transition_state_reconcile_fleet_proof_matches_observation_at(
&state,
&admitted,
"topology-a",
&restarted_peers,
now,
));
let (_, _, _, downgraded) = cross_pool_fence_policy_results(original_peers, 4);
assert!(
downgraded.is_err(),
"a synchronous observation of a downgraded peer must fail before any cached proof can authorize a write"
);
}
#[tokio::test]
async fn legacy_transition_state_reconcile_invalid_token_skips_fleet_observation() {
let slot = std::sync::RwLock::new(FleetCapabilityProofState::default());
let now = Instant::now();
let peers = BTreeMap::from([("peer-a".to_string(), Uuid::new_v4())]);
assert!(publish_fleet_capability_probe_result(&slot, "topology-a", Ok(peers), now).is_none());
let admitted = {
let state = slot.read().expect("reconcile proof slot should not poison");
acquire_legacy_transition_state_reconcile_fleet_proof_from(&state, "topology-a", now)
.expect("the original fleet should admit reconciliation")
};
revoke_fleet_capability_proof(&slot);
assert!(
!legacy_transition_state_reconcile_fleet_proof_matches_with_observer(&slot, &admitted, "topology-a", || async {
panic!("an invalid local generation must not trigger a fleet observation");
},)
.await
);
}
#[test]
fn legacy_transition_state_reconcile_membership_and_topology_changes_revoke_authority() {
let now = Instant::now();
for replacement in [
BTreeMap::from([("peer-a".to_string(), Uuid::new_v4()), ("peer-b".to_string(), Uuid::new_v4())]),
BTreeMap::from([("peer-a".to_string(), Uuid::new_v4())]),
] {
let slot = std::sync::RwLock::new(FleetCapabilityProofState::default());
let original = BTreeMap::from([("peer-a".to_string(), Uuid::new_v4())]);
assert!(publish_fleet_capability_probe_result(&slot, "topology-a", Ok(original), now).is_none());
let admitted = {
let state = slot.read().expect("reconcile proof slot should not poison");
acquire_legacy_transition_state_reconcile_fleet_proof_from(&state, "topology-a", now)
.expect("the original fleet should admit reconciliation")
};
assert!(
publish_fleet_capability_probe_result(&slot, "topology-a", Ok(replacement), now + Duration::from_millis(1),)
.is_some(),
"membership or process-epoch replacement must wait for the admitted writer"
);
let state = slot.read().expect("reconcile proof slot should not poison");
assert!(!legacy_transition_state_reconcile_fleet_proof_matches_at(
&state,
&admitted,
"topology-a",
now + Duration::from_millis(1),
));
}
let slot = std::sync::RwLock::new(FleetCapabilityProofState::default());
assert!(publish_fleet_capability_probe_result(&slot, "topology-a", Ok(BTreeMap::new()), now).is_none());
let admitted = {
let state = slot.read().expect("reconcile proof slot should not poison");
acquire_legacy_transition_state_reconcile_fleet_proof_from(&state, "topology-a", now)
.expect("the original topology should admit reconciliation")
};
mark_fleet_capability_topology_conflict(&slot);
let state = slot.read().expect("reconcile proof slot should not poison");
assert!(state.topology_conflict);
assert!(!legacy_transition_state_reconcile_fleet_proof_matches_at(
&state,
&admitted,
"topology-a",
now,
));
}
#[test]
fn legacy_transition_state_reconcile_capability_downgrade_fails_closed() {
let slot = std::sync::RwLock::new(FleetCapabilityProofState::default());
let now = Instant::now();
let peers = BTreeMap::from([("peer-a".to_string(), Uuid::new_v4())]);
let (_, _, _, compatible_result) =
cross_pool_fence_policy_results(peers.clone(), LEGACY_TRANSITION_STATE_RECONCILE_POLICY_SUPPORTED_VERSION);
assert!(publish_fleet_capability_probe_result(&slot, "topology-a", compatible_result, now).is_none());
let admitted = {
let state = slot.read().expect("reconcile proof slot should not poison");
acquire_legacy_transition_state_reconcile_fleet_proof_from(&state, "topology-a", now)
.expect("v5 should admit reconciliation")
};
let (_, _, _, downgraded_result) =
cross_pool_fence_policy_results(peers, LEGACY_TRANSITION_STATE_RECONCILE_POLICY_SUPPORTED_VERSION - 1);
let err = publish_fleet_capability_probe_result(&slot, "topology-a", downgraded_result, now + Duration::from_millis(1))
.expect("a v4 member must revoke reconcile authority");
assert!(err.to_string().contains("reconcile policy capability version is unsupported"));
let state = slot.read().expect("reconcile proof slot should not poison");
assert!(state.proof.is_none());
assert!(!legacy_transition_state_reconcile_fleet_proof_matches_at(
&state,
&admitted,
"topology-a",
now + Duration::from_millis(1),
));
assert!(
acquire_legacy_transition_state_reconcile_fleet_proof_from(&state, "topology-a", now + Duration::from_millis(1),)
.is_none(),
"a downgraded fleet must remain inspect-only"
);
}
#[test] #[test]
fn remote_version_state_fleet_proof_conflict_revokes_atomic_snapshot() { fn remote_version_state_fleet_proof_conflict_revokes_atomic_snapshot() {
let now = Instant::now(); let now = Instant::now();
@@ -3959,57 +3539,6 @@ mod tests {
assert!(err.to_string().contains("incomplete")); assert!(err.to_string().contains("incomplete"));
} }
#[tokio::test]
async fn legacy_transition_state_reconcile_probe_rejects_missing_or_unreachable_members() {
let missing = NotificationSys {
peer_clients: Vec::new(),
all_peer_clients: vec![None],
peer_topology_hosts: vec!["peer-a".to_string()],
peer_admin_caches: Vec::new(),
tier_config_reload_workers: Default::default(),
};
let missing_err = missing
.probe_cross_pool_fence_fleet("topology-a")
.await
.expect_err("a missing member slot must prevent reconcile capability proof");
assert!(missing_err.to_string().contains("incomplete"));
let unreachable = NotificationSys {
peer_clients: vec![None],
all_peer_clients: vec![None, None],
peer_topology_hosts: vec!["peer-a".to_string()],
peer_admin_caches: vec![Mutex::new(PeerAdminCache::new())],
tier_config_reload_workers: Default::default(),
};
let unreachable_err = unreachable
.probe_cross_pool_fence_fleet("topology-a")
.await
.expect_err("an unreachable member must prevent reconcile capability proof");
assert!(unreachable_err.to_string().contains("unreachable"));
}
#[tokio::test]
async fn legacy_transition_state_reconcile_single_node_stays_closed_before_local_cas_support() {
let notification_sys = NotificationSys {
peer_clients: Vec::new(),
all_peer_clients: vec![None],
peer_topology_hosts: Vec::new(),
peer_admin_caches: Vec::new(),
tier_config_reload_workers: Default::default(),
};
let (peers, minimum_version) = notification_sys
.probe_cross_pool_fence_fleet("topology-a")
.await
.expect("a single-node capability probe should complete");
assert!(peers.is_empty());
assert_eq!(minimum_version, LOCAL_CROSS_POOL_FENCE_POLICY_SUPPORTED_VERSION);
let (_, _, _, reconcile_result) = cross_pool_fence_policy_results(peers, minimum_version);
assert!(
reconcile_result.is_err(),
"the current node must not self-authorize reconcile before the conditional writer lands"
);
}
fn build_props(endpoint: &str) -> ServerProperties { fn build_props(endpoint: &str) -> ServerProperties {
ServerProperties { ServerProperties {
endpoint: endpoint.to_string(), endpoint: endpoint.to_string(),
-1
View File
@@ -25,7 +25,6 @@ pub(crate) mod tier_probe_intent;
pub mod warm_backend; pub mod warm_backend;
pub mod warm_backend_aliyun; pub mod warm_backend_aliyun;
pub mod warm_backend_azure; pub mod warm_backend_azure;
#[cfg(feature = "gcs")]
pub mod warm_backend_gcs; pub mod warm_backend_gcs;
pub mod warm_backend_huaweicloud; pub mod warm_backend_huaweicloud;
pub mod warm_backend_minio; pub mod warm_backend_minio;
@@ -701,7 +701,7 @@ impl WarmBackend for MockWarmBackend {
Ok(version) Ok(version)
} }
async fn get(&self, object: &str, rv: &str, opts: WarmBackendGetOpts) -> Result<ReadCloser, std::io::Error> { async fn get(&self, object: &str, _rv: &str, opts: WarmBackendGetOpts) -> Result<ReadCloser, std::io::Error> {
self.precondition().await?; self.precondition().await?;
let barrier = self.inner.get_barrier.lock().await.take(); let barrier = self.inner.get_barrier.lock().await.take();
if let Some(barrier) = barrier { if let Some(barrier) = barrier {
@@ -719,9 +719,6 @@ impl WarmBackend for MockWarmBackend {
let Some(stored) = objects.get(object) else { let Some(stored) = objects.get(object) else {
return Err(std::io::Error::new(std::io::ErrorKind::NotFound, "mock object not found")); return Err(std::io::Error::new(std::io::ErrorKind::NotFound, "mock object not found"));
}; };
if !rv.is_empty() && stored.remote_version_id != rv {
return Err(std::io::Error::new(std::io::ErrorKind::NotFound, "NoSuchVersion"));
}
let bytes = &stored.bytes; let bytes = &stored.bytes;
let start = opts.start_offset.max(0) as usize; let start = opts.start_offset.max(0) as usize;
+19 -181
View File
@@ -2346,10 +2346,6 @@ impl WarmBackend for SharedWarmBackendProxy {
self.0.probe_transition_candidate(object).await self.0.probe_transition_candidate(object).await
} }
async fn probe_transition_version(&self, object: &str, remote_version_id: &str) -> io::Result<TransitionCandidateProbe> {
self.0.probe_transition_version(object, remote_version_id).await
}
async fn in_use(&self) -> io::Result<bool> { async fn in_use(&self) -> io::Result<bool> {
self.0.in_use().await self.0.in_use().await
} }
@@ -2462,15 +2458,6 @@ impl TierOperationLease {
Ok(()) Ok(())
} }
pub(crate) async fn probe_transition_version(
&self,
object: &str,
remote_version_id: &str,
) -> io::Result<TransitionCandidateProbe> {
self.validate_remote_version_id(remote_version_id)?;
self.inner.driver.probe_transition_version(object, remote_version_id).await
}
pub(crate) fn is_current_generation(&self) -> bool { pub(crate) fn is_current_generation(&self) -> bool {
lock_unpoisoned(&self.runtime) lock_unpoisoned(&self.runtime)
.generations .generations
@@ -3541,7 +3528,7 @@ impl TierConfigMgr {
// Get tier configuration and create new driver // Get tier configuration and create new driver
let tier_config = self.tiers.get(tier_name).ok_or_else(|| ERR_TIER_NOT_FOUND.clone())?; let tier_config = self.tiers.get(tier_name).ok_or_else(|| ERR_TIER_NOT_FOUND.clone())?;
let driver = construct_warm_backend(tier_config).await?; let driver = new_warm_backend(tier_config, false).await?;
self.replace_driver(tier_name, driver)?; self.replace_driver(tier_name, driver)?;
Ok(self Ok(self
@@ -4486,11 +4473,6 @@ impl TierConfigMgr {
let committed_coordinator_intent = let committed_coordinator_intent =
committed_tier_mutation_intent(coordinator_intent.as_ref(), &committed_config_etag) committed_tier_mutation_intent(coordinator_intent.as_ref(), &committed_config_etag)
.map_err(TierConfigUpdateError::Save)?; .map_err(TierConfigUpdateError::Save)?;
// Persist Committed before notifying refresh; a Prepared disk record
// would restore the prepared block and invalidate our publish allowance.
let coordinator_commit =
commit_coordinator_tier_mutation_intent(api.clone(), coordinator_intent.as_ref(), &committed_config_etag)
.await;
if let Some(intent) = committed_coordinator_intent.as_ref() { if let Some(intent) = committed_coordinator_intent.as_ref() {
TierConfigMgr::apply_committed_mutation_intent_block(&handle, intent) TierConfigMgr::apply_committed_mutation_intent_block(&handle, intent)
.await .await
@@ -4501,9 +4483,9 @@ impl TierConfigMgr {
.map_err(TierConfigUpdateError::Publish)?, .map_err(TierConfigUpdateError::Publish)?,
); );
} }
// Config is already saved: retain the committed fence and wake recovery commit_coordinator_tier_mutation_intent(api.clone(), coordinator_intent.as_ref(), &committed_config_etag)
// even when the coordinator commit failed or its outcome is unknown. .await
coordinator_commit.map_err(TierConfigUpdateError::Save)?; .map_err(TierConfigUpdateError::Save)?;
if coordinated_config_update { if coordinated_config_update {
drop(update.take()); drop(update.take());
drop(config_lock.take()); drop(config_lock.take());
@@ -10608,11 +10590,6 @@ mod tests {
.expect_err("coordinator committed-state CAS failure must be observable"); .expect_err("coordinator committed-state CAS failure must be observable");
assert!(matches!(err, TierConfigUpdateError::Save(_))); assert!(matches!(err, TierConfigUpdateError::Save(_)));
assert!(manager.read().await.tiers.contains_key("COLD-A")); assert!(manager.read().await.tiers.contains_key("COLD-A"));
assert!(TierConfigMgr::has_committed_mutation_block(&manager).await);
let refresh = TierConfigMgr::mutation_refresh_notifier(&manager).await;
tokio::time::timeout(Duration::from_secs(1), refresh.notified())
.await
.expect("failed coordinator commit must notify recovery after saving config");
let blocked = match TierConfigMgr::acquire_operation_lease(&manager, "COLD-A").await { let blocked = match TierConfigMgr::acquire_operation_lease(&manager, "COLD-A").await {
Ok(_) => panic!("failed coordinator commit CAS must retain the local committed fence"), Ok(_) => panic!("failed coordinator commit CAS must retain the local committed fence"),
Err(err) => err, Err(err) => err,
@@ -14339,12 +14316,6 @@ mod tests {
after_commit: bool, after_commit: bool,
} }
#[derive(Debug, Default)]
struct CasCoordinatorCommitBarrier {
arrived: tokio::sync::Notify,
release: tokio::sync::Notify,
}
#[derive(Debug)] #[derive(Debug)]
struct CasConfigStore { struct CasConfigStore {
objects: tokio::sync::Mutex<HashMap<String, (Vec<u8>, String)>>, objects: tokio::sync::Mutex<HashMap<String, (Vec<u8>, String)>>,
@@ -14357,7 +14328,6 @@ mod tests {
fail_delete_prefix: tokio::sync::Mutex<Option<(String, usize)>>, fail_delete_prefix: tokio::sync::Mutex<Option<(String, usize)>>,
delete_log: tokio::sync::Mutex<Vec<String>>, delete_log: tokio::sync::Mutex<Vec<String>>,
list_barrier: tokio::sync::Mutex<Option<Arc<CasListBarrier>>>, list_barrier: tokio::sync::Mutex<Option<Arc<CasListBarrier>>>,
coordinator_commit_barrier: tokio::sync::Mutex<Option<Arc<CasCoordinatorCommitBarrier>>>,
intent_list_calls: AtomicUsize, intent_list_calls: AtomicUsize,
fail_reference_walk: AtomicBool, fail_reference_walk: AtomicBool,
reference_walk_send_count: AtomicUsize, reference_walk_send_count: AtomicUsize,
@@ -14380,7 +14350,6 @@ mod tests {
fail_delete_prefix: tokio::sync::Mutex::new(None), fail_delete_prefix: tokio::sync::Mutex::new(None),
delete_log: tokio::sync::Mutex::new(Vec::new()), delete_log: tokio::sync::Mutex::new(Vec::new()),
list_barrier: tokio::sync::Mutex::new(None), list_barrier: tokio::sync::Mutex::new(None),
coordinator_commit_barrier: tokio::sync::Mutex::new(None),
intent_list_calls: AtomicUsize::new(0), intent_list_calls: AtomicUsize::new(0),
fail_reference_walk: AtomicBool::new(false), fail_reference_walk: AtomicBool::new(false),
reference_walk_send_count: AtomicUsize::new(0), reference_walk_send_count: AtomicUsize::new(0),
@@ -14572,19 +14541,6 @@ mod tests {
} }
let mut payload = Vec::new(); let mut payload = Vec::new();
tokio::io::AsyncReadExt::read_to_end(&mut data.stream, &mut payload).await?; tokio::io::AsyncReadExt::read_to_end(&mut data.stream, &mut payload).await?;
if object.starts_with(crate::services::tier::tier_mutation_intent::TIER_COORDINATOR_MUTATION_INTENT_RECORD_PREFIX)
&& opts
.http_preconditions
.as_ref()
.and_then(HTTPPreconditions::if_match_value)
.is_some()
{
let barrier = self.coordinator_commit_barrier.lock().await.take();
if let Some(barrier) = barrier {
barrier.arrived.notify_one();
barrier.release.notified().await;
}
}
let race_rewrite = if opts let race_rewrite = if opts
.http_preconditions .http_preconditions
.as_ref() .as_ref()
@@ -15682,7 +15638,14 @@ mod tests {
); );
} }
async fn assert_lifecycle_only_reference_obeys_force(clear: bool, force: bool) { #[tokio::test]
async fn force_remove_and_save_bypasses_lifecycle_only_reference() {
// rustfs/rustfs#6832: reproduces the admin RemoveTier path (not just the lower-level
// reference-proof function) for a tier with zero transitioned objects but a lifecycle
// rule still pointing at it — the exact shape of
// `test_manual_transition_async_tier_failure_reports_terminal_partial` in e2e_test,
// which force-removes a tier a lifecycle rule still references to simulate a
// decommissioned backend.
let store = Arc::new(CasConfigStore::default()); let store = Arc::new(CasConfigStore::default());
let tier = build_rustfs_tier("COLD-A"); let tier = build_rustfs_tier("COLD-A");
let mut persisted = empty_mgr(); let mut persisted = empty_mgr();
@@ -15723,55 +15686,22 @@ mod tests {
let manager = TierConfigMgr::new(); let manager = TierConfigMgr::new();
manager.write().await.tiers.insert("COLD-A".to_string(), tier); manager.write().await.tiers.insert("COLD-A".to_string(), tier);
let mutation = if clear { TierConfigMgr::remove_and_save_with(&manager, store.clone(), "COLD-A", true)
TierCandidateMutation::Clear(force) .await
} else { .expect("force remove must bypass a lifecycle-config-only reference");
TierCandidateMutation::Remove("COLD-A".to_string(), force)
};
let result = TIER_DRIVER_TEST_FACTORY
.scope(
healthy_driver_factory(),
TierConfigMgr::update_candidate_with_config_lock(&manager, store.clone(), mutation),
)
.await;
if force {
result.expect("force mutation must bypass a lifecycle-config-only reference");
} else {
let err = result.expect_err("non-force mutation must reject a lifecycle-only reference");
let TierConfigUpdateError::Publish(err) = err else {
panic!("non-force mutation must fail during reference proof: {err:?}");
};
assert_eq!(err.code, ERR_TIER_BACKEND_IN_USE.code);
assert!(err.message.contains("move-current"), "{err}");
}
assert_eq!(manager.read().await.tiers.contains_key("COLD-A"), !force); assert!(!manager.read().await.tiers.contains_key("COLD-A"));
assert_eq!( assert!(
load_tier_config_for_update(store) !load_tier_config_for_update(store)
.await .await
.expect("config should still reload") .expect("config should still reload")
.0 .0
.tiers .tiers
.contains_key("COLD-A"), .contains_key("COLD-A"),
!force, "force removal must persist the empty candidate"
"persisted state must match the force mutation result"
); );
} }
#[tokio::test]
async fn remove_with_config_lock_obeys_force_for_lifecycle_only_reference() {
for force in [false, true] {
assert_lifecycle_only_reference_obeys_force(false, force).await;
}
}
#[tokio::test]
async fn clear_with_config_lock_obeys_force_for_lifecycle_only_reference() {
for force in [false, true] {
assert_lifecycle_only_reference_obeys_force(true, force).await;
}
}
#[tokio::test] #[tokio::test]
async fn zero_reference_proof_blocks_clear_before_config_save() { async fn zero_reference_proof_blocks_clear_before_config_save() {
let store = Arc::new(CasConfigStore::default()); let store = Arc::new(CasConfigStore::default());
@@ -17312,98 +17242,6 @@ mod tests {
assert_ne!(manager_a.read().await.empty(), manager_b.read().await.empty()); assert_ne!(manager_a.read().await.empty(), manager_b.read().await.empty());
} }
async fn assert_coordinator_commit_refresh_succeeds(mutation: TierCandidateMutation) {
let adding = matches!(mutation, TierCandidateMutation::Add(..));
let manager = TierConfigMgr::new();
let store = Arc::new(CasConfigStore::default());
if !adding {
let mut persisted = empty_mgr();
persisted.tiers.insert("COLD-A".to_string(), build_rustfs_tier("COLD-A"));
persisted
.save_tiering_config_if_current(store.clone(), None)
.await
.expect("existing tier fixture should persist");
let mut guard = manager.write().await;
install_lease_backend(&mut guard, "COLD-A", LeaseTestBackend::ready("old"));
}
let barrier = Arc::new(CasCoordinatorCommitBarrier::default());
*store.coordinator_commit_barrier.lock().await = Some(barrier.clone());
let update_manager = manager.clone();
let update_store = store.clone();
let update = tokio::spawn(async move {
TIER_DRIVER_TEST_FACTORY
.scope(
healthy_driver_factory(),
TIER_MUTATION_TEST_PEERS.scope(
Vec::new(),
TierConfigMgr::update_candidate_with_config_lock(&update_manager, update_store, mutation),
),
)
.await
});
tokio::time::timeout(Duration::from_secs(5), barrier.arrived.notified())
.await
.expect("mutation should reach coordinator commit after saving config");
assert_eq!(
load_tier_config_for_update(store.clone())
.await
.expect("saved config should be readable before coordinator commit")
.0
.tiers
.contains_key("COLD-A"),
adding
);
assert_eq!(
TierConfigMgr::load_coordinator_mutation_intents(store.clone())
.await
.expect("coordinator intent should remain readable")[0]
.state,
TierMutationIntentState::Prepared
);
let lock_requests = lock_unpoisoned(&store.lock_requests).len();
// Also exercise an independently scheduled refresh while the durable
// coordinator record is still Prepared, before its commit notification.
TierConfigMgr::request_committed_mutation_refresh(&manager).await;
TIER_MUTATION_TEST_PEERS
.scope(Vec::new(), async {
let worker = TierConfigMgr::refresh_tier_config_handle_with(manager.clone(), store.clone());
tokio::pin!(worker);
tokio::time::timeout(Duration::from_secs(5), async {
while lock_unpoisoned(&store.lock_requests).len() == lock_requests {
tokio::select! {
_ = &mut worker => panic!("refresh worker must remain available"),
_ = tokio::task::yield_now() => {}
}
}
})
.await
.expect("refresh should reconcile the Prepared record before waiting for the config lock");
barrier.release.notify_one();
let result = tokio::time::timeout(Duration::from_secs(5), async {
tokio::select! {
_ = &mut worker => panic!("refresh worker must remain available"),
result = update => result.expect("tier mutation task should join"),
}
})
.await
.expect("tier mutation should finish with refresh running");
result.expect("saved tier mutation must publish successfully on the first attempt");
})
.await;
assert_eq!(manager.read().await.tiers.contains_key("COLD-A"), adding);
}
#[tokio::test]
async fn tier_add_succeeds_with_refresh_during_coordinator_commit() {
assert_coordinator_commit_refresh_succeeds(TierCandidateMutation::Add(build_rustfs_tier("COLD-A"), true)).await;
}
#[tokio::test]
async fn tier_remove_succeeds_with_refresh_during_coordinator_commit() {
assert_coordinator_commit_refresh_succeeds(TierCandidateMutation::Remove("COLD-A".to_string(), true)).await;
}
async fn committed_refresh_fixture(fail_cleanup: bool) -> (Arc<RwLock<TierConfigMgr>>, Arc<CasConfigStore>, uuid::Uuid) { async fn committed_refresh_fixture(fail_cleanup: bool) -> (Arc<RwLock<TierConfigMgr>>, Arc<CasConfigStore>, uuid::Uuid) {
let manager = TierConfigMgr::new(); let manager = TierConfigMgr::new();
{ {
@@ -15,6 +15,8 @@
#![allow(unused_variables)] #![allow(unused_variables)]
#![allow(unused_mut)] #![allow(unused_mut)]
#![allow(unused_assignments)] #![allow(unused_assignments)]
#![allow(unused_must_use)]
#![allow(clippy::all)]
use serde::{Deserialize, Deserializer, Serialize, Serializer, de}; use serde::{Deserialize, Deserializer, Serialize, Serializer, de};
@@ -143,7 +145,7 @@ mod tests {
assert_eq!(creds.access_key, "access"); assert_eq!(creds.access_key, "access");
assert_eq!(creds.secret_key, "secret"); assert_eq!(creds.secret_key, "secret");
assert_eq!(creds.creds_json.as_slice(), service_account); assert_eq!(creds.creds_json.as_slice(), &service_account[..]);
let wire = serde_json::to_value(&creds).expect("madmin tier credentials should encode"); let wire = serde_json::to_value(&creds).expect("madmin tier credentials should encode");
assert_eq!(wire["access"], "access"); assert_eq!(wire["access"], "access");
@@ -160,7 +162,7 @@ mod tests {
.expect("the former RustFS field names and byte-array encoding should remain readable"); .expect("the former RustFS field names and byte-array encoding should remain readable");
assert_eq!(legacy.access_key, "legacy-access"); assert_eq!(legacy.access_key, "legacy-access");
assert_eq!(legacy.secret_key, "legacy-secret"); assert_eq!(legacy.secret_key, "legacy-secret");
assert_eq!(legacy.creds_json.as_slice(), service_account); assert_eq!(legacy.creds_json.as_slice(), &service_account[..]);
} }
#[test] #[test]
@@ -460,7 +460,6 @@ where
data, data,
&ObjectOptions { &ObjectOptions {
max_parity: true, max_parity: true,
write_completion: crate::object_api::WriteCompletion::TailDrained,
http_preconditions: Some(HTTPPreconditions { http_preconditions: Some(HTTPPreconditions {
if_none_match: Some("*".to_string()), if_none_match: Some("*".to_string()),
..Default::default() ..Default::default()
@@ -557,7 +556,6 @@ where
data, data,
&ObjectOptions { &ObjectOptions {
max_parity: true, max_parity: true,
write_completion: crate::object_api::WriteCompletion::TailDrained,
http_preconditions: Some(HTTPPreconditions { http_preconditions: Some(HTTPPreconditions {
if_match: Some(current_etag.to_string()), if_match: Some(current_etag.to_string()),
..Default::default() ..Default::default()
@@ -494,7 +494,6 @@ where
data, data,
&ObjectOptions { &ObjectOptions {
max_parity: true, max_parity: true,
write_completion: crate::object_api::WriteCompletion::TailDrained,
http_preconditions: Some(HTTPPreconditions { http_preconditions: Some(HTTPPreconditions {
if_none_match: Some("*".to_string()), if_none_match: Some("*".to_string()),
..Default::default() ..Default::default()
@@ -550,7 +549,6 @@ where
data, data,
&ObjectOptions { &ObjectOptions {
max_parity: true, max_parity: true,
write_completion: crate::object_api::WriteCompletion::TailDrained,
http_preconditions: Some(HTTPPreconditions { http_preconditions: Some(HTTPPreconditions {
if_match: Some(current.record_etag.clone()), if_match: Some(current.record_etag.clone()),
..Default::default() ..Default::default()
@@ -19,14 +19,13 @@
#![allow(clippy::all)] #![allow(clippy::all)]
use crate::error::is_err_bucket_not_found; use crate::error::is_err_bucket_not_found;
#[cfg(feature = "gcs")]
use crate::services::tier::warm_backend_gcs::WarmBackendGCS;
use crate::services::tier::{ use crate::services::tier::{
tier::{ERR_TIER_BACKEND_IN_USE, ERR_TIER_INVALID_CONFIG, ERR_TIER_TYPE_UNSUPPORTED}, tier::{ERR_TIER_BACKEND_IN_USE, ERR_TIER_INVALID_CONFIG, ERR_TIER_TYPE_UNSUPPORTED},
tier_config::{TierConfig, TierType}, tier_config::{TierConfig, TierType},
tier_handlers::{ERR_TIER_BUCKET_NOT_FOUND, ERR_TIER_NOT_FOUND, ERR_TIER_PERM_ERR}, tier_handlers::{ERR_TIER_BUCKET_NOT_FOUND, ERR_TIER_NOT_FOUND, ERR_TIER_PERM_ERR},
warm_backend_aliyun::WarmBackendAliyun, warm_backend_aliyun::WarmBackendAliyun,
warm_backend_azure::WarmBackendAzure, warm_backend_azure::WarmBackendAzure,
warm_backend_gcs::WarmBackendGCS,
warm_backend_huaweicloud::WarmBackendHuaweicloud, warm_backend_huaweicloud::WarmBackendHuaweicloud,
warm_backend_minio::WarmBackendMinIO, warm_backend_minio::WarmBackendMinIO,
warm_backend_r2::WarmBackendR2, warm_backend_r2::WarmBackendR2,
@@ -38,10 +37,9 @@ use crate::services::tier::{
use bytes::Bytes; use bytes::Bytes;
use http::StatusCode; use http::StatusCode;
use rustfs_s3_client::credentials::{Credentials, SignatureType, Static, Value}; use rustfs_s3_client::credentials::{Credentials, SignatureType, Static, Value};
use rustfs_s3_client::transition_api::{BucketLookupType, Options, TransitionClient, TransitionClientTimeouts, TransitionCore}; use rustfs_s3_client::transition_api::{BucketLookupType, Options, TransitionClient, TransitionCore};
use rustfs_s3_client::{ use rustfs_s3_client::{
admin_handler_utils::AdminError, admin_handler_utils::AdminError,
api_error_response::to_error_response,
api_put_object::{AdvancedPutOptions, PutObjectOptions}, api_put_object::{AdvancedPutOptions, PutObjectOptions},
transition_api::{ReadCloser, ReaderImpl}, transition_api::{ReadCloser, ReaderImpl},
}; };
@@ -50,14 +48,11 @@ use rustfs_utils::egress::validate_outbound_url;
use rustfs_utils::http::headers::{ use rustfs_utils::http::headers::{
CACHE_CONTROL, CONTENT_DISPOSITION, CONTENT_ENCODING, CONTENT_LANGUAGE, CONTENT_TYPE, EXPIRES, HeaderExt as _, CACHE_CONTROL, CONTENT_DISPOSITION, CONTENT_ENCODING, CONTENT_LANGUAGE, CONTENT_TYPE, EXPIRES, HeaderExt as _,
}; };
use s3s::dto::{ObjectLockLegalHoldStatus, ObjectLockRetentionMode, ReplicationStatus};
use s3s::header::{ use s3s::header::{
X_AMZ_OBJECT_LOCK_LEGAL_HOLD, X_AMZ_OBJECT_LOCK_MODE, X_AMZ_OBJECT_LOCK_RETAIN_UNTIL_DATE, X_AMZ_REPLICATION_STATUS, X_AMZ_OBJECT_LOCK_LEGAL_HOLD, X_AMZ_OBJECT_LOCK_MODE, X_AMZ_OBJECT_LOCK_RETAIN_UNTIL_DATE, X_AMZ_REPLICATION_STATUS,
X_AMZ_STORAGE_CLASS, X_AMZ_STORAGE_CLASS,
}; };
use s3s::{
S3ErrorCode,
dto::{ObjectLockLegalHoldStatus, ObjectLockRetentionMode, ReplicationStatus},
};
use std::collections::HashMap; use std::collections::HashMap;
use std::sync::Arc; use std::sync::Arc;
use std::time::Duration; use std::time::Duration;
@@ -146,42 +141,6 @@ pub trait WarmBackend {
async fn probe_transition_candidate(&self, _object: &str) -> Result<TransitionCandidateProbe, std::io::Error> { async fn probe_transition_candidate(&self, _object: &str) -> Result<TransitionCandidateProbe, std::io::Error> {
Ok(TransitionCandidateProbe::Unsupported) Ok(TransitionCandidateProbe::Unsupported)
} }
async fn probe_transition_version(
&self,
object: &str,
remote_version_id: &str,
) -> Result<TransitionCandidateProbe, std::io::Error> {
if remote_version_id.is_empty() {
return Err(std::io::Error::new(
std::io::ErrorKind::InvalidInput,
"an exact tier probe requires a remote version ID",
));
}
self.validate_remote_version_id(remote_version_id)?;
match self
.get(
object,
remote_version_id,
WarmBackendGetOpts {
start_offset: 0,
length: 1,
},
)
.await
{
Ok(_) => Ok(TransitionCandidateProbe::VersionedPresent(remote_version_id.to_string())),
Err(err) if matches!(to_error_response(&err).code, S3ErrorCode::InvalidRange) => {
Ok(TransitionCandidateProbe::VersionedPresent(remote_version_id.to_string()))
}
Err(err)
if err.kind() == std::io::ErrorKind::NotFound
|| matches!(to_error_response(&err).code, S3ErrorCode::NoSuchKey | S3ErrorCode::NoSuchVersion) =>
{
Ok(TransitionCandidateProbe::Missing)
}
Err(err) => Err(err),
}
}
async fn in_use(&self) -> Result<bool, std::io::Error>; async fn in_use(&self) -> Result<bool, std::io::Error>;
} }
@@ -321,27 +280,6 @@ pub(crate) fn endpoint_authority(url: &url::Url) -> Result<String, std::io::Erro
} }
} }
fn transition_timeout_from_env(env_key: &str, default_secs: u64) -> Duration {
Duration::from_secs(rustfs_utils::get_env_u64(env_key, default_secs))
}
pub(crate) fn transition_client_timeouts_from_env() -> TransitionClientTimeouts {
TransitionClientTimeouts::new(
transition_timeout_from_env(
rustfs_config::ENV_TIER_REMOTE_CONNECT_TIMEOUT_SECS,
rustfs_config::DEFAULT_TIER_REMOTE_CONNECT_TIMEOUT_SECS,
),
transition_timeout_from_env(
rustfs_config::ENV_TIER_REMOTE_REQUEST_TIMEOUT_SECS,
rustfs_config::DEFAULT_TIER_REMOTE_REQUEST_TIMEOUT_SECS,
),
transition_timeout_from_env(
rustfs_config::ENV_TIER_REMOTE_RESPONSE_BODY_IDLE_TIMEOUT_SECS,
rustfs_config::DEFAULT_TIER_REMOTE_RESPONSE_BODY_IDLE_TIMEOUT_SECS,
),
)
}
/// Build the [`WarmBackendS3`] shared by the S3-compatible warm backend providers. /// Build the [`WarmBackendS3`] shared by the S3-compatible warm backend providers.
/// ///
/// Credential, bucket, and endpoint validation run in this order because the /// Credential, bucket, and endpoint validation run in this order because the
@@ -372,7 +310,6 @@ pub(crate) async fn new_s3_compatible_warm_backend(
signer_type: SignatureType::SignatureV4, signer_type: SignatureType::SignatureV4,
..Default::default() ..Default::default()
})); }));
let timeouts = transition_client_timeouts_from_env();
let opts = Options { let opts = Options {
creds, creds,
secure: u.scheme() == "https", secure: u.scheme() == "https",
@@ -385,7 +322,7 @@ pub(crate) async fn new_s3_compatible_warm_backend(
// Run the SSRF guard after the host-presence check so a host-less endpoint // Run the SSRF guard after the host-presence check so a host-less endpoint
// keeps this constructor's stable error text. // keeps this constructor's stable error text.
(params.validate_endpoint)(&u).map_err(|err| std::io::Error::other(format!("tier endpoint is not allowed: {err}")))?; (params.validate_endpoint)(&u).map_err(|err| std::io::Error::other(format!("tier endpoint is not allowed: {err}")))?;
let client = TransitionClient::new_with_timeouts(&endpoint, opts, params.provider_tag, timeouts).await?; let client = TransitionClient::new(&endpoint, opts, params.provider_tag).await?;
let client = Arc::new(client); let client = Arc::new(client);
let core = TransitionCore(Arc::clone(&client)); let core = TransitionCore(Arc::clone(&client));
@@ -500,17 +437,6 @@ impl WarmBackend for MeteredWarmBackend {
Self::record(TierRequestOperation::Probe, result) Self::record(TierRequestOperation::Probe, result)
} }
async fn probe_transition_version(
&self,
object: &str,
remote_version_id: &str,
) -> Result<TransitionCandidateProbe, std::io::Error> {
Self::record(
TierRequestOperation::Probe,
self.inner.probe_transition_version(object, remote_version_id).await,
)
}
async fn in_use(&self) -> Result<bool, std::io::Error> { async fn in_use(&self) -> Result<bool, std::io::Error> {
Self::record(TierRequestOperation::InUse, self.inner.in_use().await) Self::record(TierRequestOperation::InUse, self.inner.in_use().await)
} }
@@ -913,15 +839,6 @@ pub async fn new_warm_backend(tier: &TierConfig, probe: bool) -> Result<WarmBack
}); });
} }
} }
#[cfg(not(feature = "gcs"))]
TierType::GCS => {
return Err(AdminError {
code: ERR_TIER_TYPE_UNSUPPORTED.code.clone(),
message: "This build does not include the GCS backend; rebuild with the gcs feature".to_string(),
status_code: StatusCode::NOT_IMPLEMENTED,
});
}
#[cfg(feature = "gcs")]
TierType::GCS => { TierType::GCS => {
if let Some(gcs_config) = tier.gcs.as_ref() { if let Some(gcs_config) = tier.gcs.as_ref() {
let dd = WarmBackendGCS::new(gcs_config, &tier.name).await; let dd = WarmBackendGCS::new(gcs_config, &tier.name).await;
@@ -1038,27 +955,6 @@ mod tests {
const PROBE_VERSION: &str = "remote-v2"; const PROBE_VERSION: &str = "remote-v2";
#[cfg(not(feature = "gcs"))]
#[tokio::test]
async fn gcs_backend_not_compiled_preserves_config() {
let json = r#"{"name":"ARCHIVE","type":"gcs","gcs":{"bucket":"archive","creds":"secret"}}"#;
let tier: TierConfig = serde_json::from_str(json).expect("GCS config remains readable without the backend");
assert_eq!(tier.tier_type, TierType::GCS);
let encoded = serde_json::to_vec(&tier).expect("GCS config remains writable");
let restored: TierConfig = serde_json::from_slice(&encoded).expect("GCS config round trips");
assert_eq!(restored.tier_type, TierType::GCS);
let restored_gcs = restored.gcs.as_ref().expect("GCS settings preserved");
assert_eq!(restored_gcs.bucket, "archive");
assert_eq!(restored_gcs.creds, "secret");
assert_eq!(tier.redacted().gcs.expect("redacted GCS settings").creds, "REDACTED");
let error = match new_warm_backend(&tier, false).await {
Ok(_) => panic!("an excluded GCS backend cannot be constructed"),
Err(error) => error,
};
assert_eq!(error.code, ERR_TIER_TYPE_UNSUPPORTED.code);
assert_eq!(error.status_code, StatusCode::NOT_IMPLEMENTED);
}
struct CountingBackend { struct CountingBackend {
put_result: fn() -> Result<String, std::io::Error>, put_result: fn() -> Result<String, std::io::Error>,
removes: Arc<AtomicUsize>, removes: Arc<AtomicUsize>,
@@ -15,6 +15,8 @@
#![allow(unused_variables)] #![allow(unused_variables)]
#![allow(unused_mut)] #![allow(unused_mut)]
#![allow(unused_assignments)] #![allow(unused_assignments)]
#![allow(unused_must_use)]
#![allow(clippy::all)]
use std::collections::HashMap; use std::collections::HashMap;
@@ -15,6 +15,8 @@
#![allow(unused_variables)] #![allow(unused_variables)]
#![allow(unused_mut)] #![allow(unused_mut)]
#![allow(unused_assignments)] #![allow(unused_assignments)]
#![allow(unused_must_use)]
#![allow(clippy::all)]
use std::collections::HashMap; use std::collections::HashMap;
@@ -15,6 +15,8 @@
#![allow(unused_variables)] #![allow(unused_variables)]
#![allow(unused_mut)] #![allow(unused_mut)]
#![allow(unused_assignments)] #![allow(unused_assignments)]
#![allow(unused_must_use)]
#![allow(clippy::all)]
use std::collections::{HashMap, HashSet}; use std::collections::{HashMap, HashSet};
use std::future::Future; use std::future::Future;
@@ -144,11 +146,11 @@ pub struct WarmBackendGCS {
impl WarmBackendGCS { impl WarmBackendGCS {
pub async fn new(conf: &TierGCS, tier: &str) -> Result<Self, std::io::Error> { pub async fn new(conf: &TierGCS, tier: &str) -> Result<Self, std::io::Error> {
if conf.creds.is_empty() { if conf.creds == "" {
return Err(std::io::Error::other("both access and secret keys are required")); return Err(std::io::Error::other("both access and secret keys are required"));
} }
if conf.bucket.is_empty() { if conf.bucket == "" {
return Err(std::io::Error::other("no bucket name was provided")); return Err(std::io::Error::other("no bucket name was provided"));
} }
@@ -193,11 +195,11 @@ impl WarmBackendGCS {
} }
pub fn get_dest(&self, object: &str) -> String { pub fn get_dest(&self, object: &str) -> String {
if self.prefix.is_empty() { let mut dest_obj = object.to_string();
object.to_string() if self.prefix != "" {
} else { dest_obj = format!("{}/{}", &self.prefix, object);
format!("{}/{}", self.prefix, object)
} }
return dest_obj;
} }
} }
@@ -221,7 +223,7 @@ impl WarmBackend for WarmBackendGCS {
let bucket = gcs_bucket_resource_name(&self.bucket); let bucket = gcs_bucket_resource_name(&self.bucket);
let Ok(res) = Box::pin( let Ok(res) = Box::pin(
self.client self.client
.write_object(&bucket, self.get_dest(object), Bytes::from(d)) .write_object(&bucket, &self.get_dest(object), Bytes::from(d))
.send_buffered(), .send_buffered(),
) )
.await .await
@@ -238,7 +240,7 @@ impl WarmBackend for WarmBackendGCS {
async fn get(&self, object: &str, rv: &str, opts: WarmBackendGetOpts) -> Result<ReadCloser, std::io::Error> { async fn get(&self, object: &str, rv: &str, opts: WarmBackendGetOpts) -> Result<ReadCloser, std::io::Error> {
let bucket = gcs_bucket_resource_name(&self.bucket); let bucket = gcs_bucket_resource_name(&self.bucket);
let mut req = self.client.read_object(&bucket, self.get_dest(object)); let mut req = self.client.read_object(&bucket, &self.get_dest(object));
let mut max_response_bytes = None; let mut max_response_bytes = None;
if let Some(generation) = parse_generation(rv)? { if let Some(generation) = parse_generation(rv)? {
req = req.set_generation(generation); req = req.set_generation(generation);
@@ -15,6 +15,8 @@
#![allow(unused_variables)] #![allow(unused_variables)]
#![allow(unused_mut)] #![allow(unused_mut)]
#![allow(unused_assignments)] #![allow(unused_assignments)]
#![allow(unused_must_use)]
#![allow(clippy::all)]
use std::collections::HashMap; use std::collections::HashMap;
@@ -15,6 +15,8 @@
#![allow(unused_variables)] #![allow(unused_variables)]
#![allow(unused_mut)] #![allow(unused_mut)]
#![allow(unused_assignments)] #![allow(unused_assignments)]
#![allow(unused_must_use)]
#![allow(clippy::all)]
use std::collections::HashMap; use std::collections::HashMap;
@@ -15,6 +15,8 @@
#![allow(unused_variables)] #![allow(unused_variables)]
#![allow(unused_mut)] #![allow(unused_mut)]
#![allow(unused_assignments)] #![allow(unused_assignments)]
#![allow(unused_must_use)]
#![allow(clippy::all)]
use std::collections::HashMap; use std::collections::HashMap;
@@ -15,6 +15,8 @@
#![allow(unused_variables)] #![allow(unused_variables)]
#![allow(unused_mut)] #![allow(unused_mut)]
#![allow(unused_assignments)] #![allow(unused_assignments)]
#![allow(unused_must_use)]
#![allow(clippy::all)]
use std::collections::HashMap; use std::collections::HashMap;
@@ -26,7 +26,7 @@ use crate::services::tier::{
tier_config::TierS3, tier_config::TierS3,
warm_backend::{ warm_backend::{
TransitionCandidateIdentity, TransitionCandidateProbe, TransitionCandidateReconciler, WarmBackend, WarmBackendGetOpts, TransitionCandidateIdentity, TransitionCandidateProbe, TransitionCandidateReconciler, WarmBackend, WarmBackendGetOpts,
build_transition_put_options, endpoint_authority, transition_client_timeouts_from_env, build_transition_put_options, endpoint_authority,
}, },
}; };
use http::HeaderMap; use http::HeaderMap;
@@ -139,7 +139,6 @@ impl WarmBackendS3 {
} else { } else {
return Err(std::io::Error::other("insufficient parameters for S3 backend authentication")); return Err(std::io::Error::other("insufficient parameters for S3 backend authentication"));
} }
let timeouts = transition_client_timeouts_from_env();
let opts = Options { let opts = Options {
creds, creds,
secure: u.scheme() == "https", secure: u.scheme() == "https",
@@ -148,7 +147,7 @@ impl WarmBackendS3 {
..Default::default() ..Default::default()
}; };
let endpoint = endpoint_authority(&u)?; let endpoint = endpoint_authority(&u)?;
let client = TransitionClient::new_with_timeouts(&endpoint, opts, tier_type, timeouts).await?; let client = TransitionClient::new(&endpoint, opts, tier_type).await?;
let client = Arc::new(client); let client = Arc::new(client);
let core = TransitionCore(Arc::clone(&client)); let core = TransitionCore(Arc::clone(&client));
@@ -530,10 +529,6 @@ mod tests {
"HTTP/1.1 404 Not Found\r\nContent-Type: application/xml\r\nContent-Length: 63\r\nConnection: close\r\n\r\n<Error><Code>NoSuchKey</Code><Message>missing</Message></Error>", "HTTP/1.1 404 Not Found\r\nContent-Type: application/xml\r\nContent-Length: 63\r\nConnection: close\r\n\r\n<Error><Code>NoSuchKey</Code><Message>missing</Message></Error>",
"HTTP/1.1 404 Not Found\r\nContent-Type: application/xml\r\nContent-Length: 66\r\nConnection: close\r\n\r\n<Error><Code>NoSuchObject</Code><Message>missing</Message></Error>", "HTTP/1.1 404 Not Found\r\nContent-Type: application/xml\r\nContent-Length: 66\r\nConnection: close\r\n\r\n<Error><Code>NoSuchObject</Code><Message>missing</Message></Error>",
"HTTP/1.1 403 Forbidden\r\nContent-Type: application/xml\r\nContent-Length: 65\r\nConnection: close\r\n\r\n<Error><Code>AccessDenied</Code><Message>denied</Message></Error>", "HTTP/1.1 403 Forbidden\r\nContent-Type: application/xml\r\nContent-Length: 65\r\nConnection: close\r\n\r\n<Error><Code>AccessDenied</Code><Message>denied</Message></Error>",
"HTTP/1.1 404 Not Found\r\nContent-Type: application/xml\r\nContent-Length: 63\r\nConnection: close\r\n\r\n<Error><Code>NoSuchKey</Code><Message>missing</Message></Error>",
"HTTP/1.1 416 Range Not Satisfiable\r\nContent-Type: application/xml\r\nContent-Length: 72\r\nConnection: close\r\n\r\n<Error><Code>InvalidRange</Code><Message>empty version</Message></Error>",
"HTTP/1.1 404 Not Found\r\nContent-Type: application/xml\r\nContent-Length: 67\r\nConnection: close\r\n\r\n<Error><Code>NoSuchVersion</Code><Message>missing</Message></Error>",
"HTTP/1.1 404 Not Found\r\nContent-Type: application/xml\r\nContent-Length: 63\r\nConnection: close\r\n\r\n<Error><Code>NoSuchKey</Code><Message>missing</Message></Error>",
]; ];
let mut requests = Vec::new(); let mut requests = Vec::new();
for response in responses { for response in responses {
@@ -627,52 +622,15 @@ mod tests {
.await .await
.expect_err("an authorization failure must not be mistaken for a missing key"); .expect_err("an authorization failure must not be mistaken for a missing key");
assert_eq!(to_error_response(&err).code, S3ErrorCode::AccessDenied); assert_eq!(to_error_response(&err).code, S3ErrorCode::AccessDenied);
assert_eq!(
backend
.probe_transition_candidate("delete-marker-hidden")
.await
.expect("a current delete marker should hide the data version"),
TransitionCandidateProbe::Missing
);
assert_eq!(
backend
.probe_transition_version("delete-marker-hidden", "historical-version")
.await
.expect("the stored historical version should be probed exactly"),
TransitionCandidateProbe::VersionedPresent("historical-version".to_string())
);
assert_eq!(
backend
.probe_transition_version("delete-marker-hidden", "missing-version")
.await
.expect("a missing exact version should be classified"),
TransitionCandidateProbe::Missing
);
assert_eq!(
backend
.probe_transition_version("missing-object", "historical-version")
.await
.expect("a missing key for an exact version probe should be classified"),
TransitionCandidateProbe::Missing
);
let requests = fixture.await.expect("candidate fixture should join"); let requests = fixture.await.expect("candidate fixture should join");
for request in &requests[..6] { for request in requests {
let request = request.to_ascii_lowercase(); let request = request.to_ascii_lowercase();
assert!(request.starts_with("get /bucket/"), "candidate discovery must use object GET"); assert!(request.starts_with("get /bucket/"), "candidate discovery must use object GET");
assert!(request.contains("\r\nrange: bytes=0-0\r\n")); assert!(request.contains("\r\nrange: bytes=0-0\r\n"));
assert!(!request.contains("?versioning")); assert!(!request.contains("?versioning"));
assert!(!request.contains("?versions")); assert!(!request.contains("?versions"));
} }
for request in &requests[6..] {
let request = request.to_ascii_lowercase();
assert!(request.starts_with("get /bucket/"), "exact discovery must use object GET");
assert!(request.contains("\r\nrange: bytes=0-0\r\n"));
}
assert!(!requests[5].to_ascii_lowercase().contains("versionid="));
assert!(requests[6].to_ascii_lowercase().contains("?versionid=historical-version"));
assert!(requests[7].to_ascii_lowercase().contains("?versionid=missing-version"));
assert!(requests[8].to_ascii_lowercase().contains("?versionid=historical-version"));
} }
fn list_versions(versions: &[(&str, &str)], delete_markers: &[(&str, &str)], is_truncated: bool) -> ListVersionsResult { fn list_versions(versions: &[(&str, &str)], delete_markers: &[(&str, &str)], is_truncated: bool) -> ListVersionsResult {
@@ -15,6 +15,8 @@
#![allow(unused_variables)] #![allow(unused_variables)]
#![allow(unused_mut)] #![allow(unused_mut)]
#![allow(unused_assignments)] #![allow(unused_assignments)]
#![allow(unused_must_use)]
#![allow(clippy::all)]
use std::collections::HashMap; use std::collections::HashMap;
File diff suppressed because it is too large Load Diff
@@ -1,385 +0,0 @@
// Copyright 2024 RustFS Team
//
// Licensed under the Apache License, Version 2.0 (the "License");
// you may not use this file except in compliance with the License.
// You may obtain a copy of the License at
//
// http://www.apache.org/licenses/LICENSE-2.0
//
// Unless required by applicable law or agreed to in writing, software
// distributed under the License is distributed on an "AS IS" BASIS,
// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
// See the License for the specific language governing permissions and
// limitations under the License.
//! Pure metadata quorum and early-stop decisions for `SetDisks` reads.
//!
//! Disk scheduling, coalescing, cancellation, and late shard materialization
//! remain with their existing owners; this module only classifies observations.
use crate::diagnostics::get::{
GET_METADATA_EARLY_STOP_REASON_CONFLICTING_METADATA, GET_METADATA_EARLY_STOP_REASON_DELETE_MARKER,
GET_METADATA_EARLY_STOP_REASON_ERROR, GET_METADATA_EARLY_STOP_REASON_INSUFFICIENT_QUORUM,
GET_METADATA_EARLY_STOP_REASON_NOT_FOUND, GET_METADATA_EARLY_STOP_REASON_UNSAFE_REQUEST,
GET_METADATA_EARLY_STOP_REASON_VALID_QUORUM, GET_METADATA_EARLY_STOP_REASON_VERSION_MATCH_QUORUM,
GET_METADATA_EARLY_STOP_REASON_VERSION_NOT_FOUND,
};
use crate::disk::error::DiskError;
use crate::disk::error_reduce::OBJECT_OP_IGNORED_ERRS;
use crate::set_disk::file_info_is_valid_for_metadata;
use rustfs_filemeta::FileInfo;
#[derive(Clone, Copy, Debug, Eq, PartialEq)]
pub(in crate::set_disk) struct MetadataEarlyStopDecision {
pub(in crate::set_disk) reason: &'static str,
}
#[derive(Clone, Debug)]
pub(in crate::set_disk) struct MetadataQuorumAccumulator {
pub(in crate::set_disk) total_disks: usize,
pub(in crate::set_disk) default_parity_count: usize,
pub(in crate::set_disk) allow_early_stop: bool,
pub(in crate::set_disk) valid_responses: usize,
pub(in crate::set_disk) not_found_responses: usize,
pub(in crate::set_disk) version_not_found_responses: usize,
pub(in crate::set_disk) ignored_errors: usize,
pub(in crate::set_disk) hard_errors: usize,
pub(in crate::set_disk) candidate: Option<FileInfo>,
pub(in crate::set_disk) candidate_votes: usize,
// Bitset of shard indexes whose metadata matches the candidate. Erasure
// layouts are capped at 16 shards, so this stays allocation-free on the
// GET metadata hot path.
candidate_shard_mask: u16,
pub(in crate::set_disk) conflicting_metadata: bool,
pub(in crate::set_disk) delete_marker_seen: bool,
pub(in crate::set_disk) delete_marker_candidates: Vec<(FileInfo, usize)>,
pub(in crate::set_disk) delete_marker_votes: usize,
pub(in crate::set_disk) requested_version_id: String,
pub(in crate::set_disk) matching_version_votes: usize,
}
impl MetadataQuorumAccumulator {
pub(in crate::set_disk) fn new(total_disks: usize, default_parity_count: usize, allow_early_stop: bool) -> Self {
Self {
total_disks,
default_parity_count,
allow_early_stop,
valid_responses: 0,
not_found_responses: 0,
version_not_found_responses: 0,
ignored_errors: 0,
hard_errors: 0,
candidate: None,
candidate_votes: 0,
candidate_shard_mask: 0,
conflicting_metadata: false,
delete_marker_seen: false,
delete_marker_candidates: Vec::new(),
delete_marker_votes: 0,
requested_version_id: String::new(),
matching_version_votes: 0,
}
}
pub(in crate::set_disk) fn with_requested_version_id(mut self, version_id: &str) -> Self {
self.requested_version_id = version_id.to_string();
self
}
pub(in crate::set_disk) fn observe_file_info(&mut self, file_info: &FileInfo) {
self.observe_file_info_with_index(None, file_info);
}
pub(in crate::set_disk) fn observe_file_info_at(&mut self, disk_index: usize, file_info: &FileInfo) {
self.observe_file_info_with_index(Some(disk_index), file_info);
}
fn observe_file_info_with_index(&mut self, disk_index: Option<usize>, file_info: &FileInfo) {
if !file_info_is_valid_for_metadata(file_info) {
self.hard_errors = self.hard_errors.saturating_add(1);
return;
}
self.valid_responses = self.valid_responses.saturating_add(1);
// Track version match for versioned requests
if !self.requested_version_id.is_empty()
&& let Some(ref vid) = file_info.version_id
&& vid.to_string() == self.requested_version_id
{
self.matching_version_votes = self.matching_version_votes.saturating_add(1);
}
if file_info.is_canonical_delete_marker() {
self.delete_marker_seen = true;
if let Some((_, votes)) = self
.delete_marker_candidates
.iter_mut()
.find(|(candidate, _)| metadata_early_stop_candidate_matches(candidate, file_info))
{
*votes = votes.saturating_add(1);
} else {
self.delete_marker_candidates.push((file_info.clone(), 1));
}
self.delete_marker_votes = self
.delete_marker_candidates
.iter()
.map(|(_, votes)| *votes)
.max()
.unwrap_or_default();
self.conflicting_metadata |= self.delete_marker_candidates.len() > 1;
return;
}
match &self.candidate {
Some(candidate) if metadata_early_stop_candidate_matches(candidate, file_info) => {
self.candidate_votes = self.candidate_votes.saturating_add(1);
if let Some(disk_index) = disk_index
&& let Some(bit) = Self::candidate_shard_bit(candidate, file_info, disk_index)
{
self.candidate_shard_mask |= bit;
}
}
Some(_) => {
self.conflicting_metadata = true;
}
None => {
self.candidate = Some(file_info.clone());
self.candidate_votes = 1;
if let Some(disk_index) = disk_index
&& let Some(bit) = Self::candidate_shard_bit(file_info, file_info, disk_index)
{
self.candidate_shard_mask |= bit;
}
}
}
}
fn candidate_shard_bit(candidate: &FileInfo, file_info: &FileInfo, disk_index: usize) -> Option<u16> {
let &erasure_index = candidate.erasure.distribution.get(disk_index)?;
if erasure_index == 0 || erasure_index > u16::BITS as usize || file_info.erasure.index != erasure_index {
return None;
}
Some(1u16 << (erasure_index - 1))
}
pub(in crate::set_disk) fn candidate_has_read_reserve(&self) -> bool {
self.candidate_read_reserve_target()
.is_some_and(|required| self.candidate_shard_mask.count_ones() as usize >= required)
}
pub(in crate::set_disk) fn candidate_read_reserve_target(&self) -> Option<usize> {
let candidate = self.candidate.as_ref()?;
Some(
candidate
.erasure
.data_blocks
.saturating_add(usize::from(candidate.erasure.parity_blocks > 0)),
)
}
pub(in crate::set_disk) fn observe_error(&mut self, err: &DiskError) {
match err {
DiskError::FileNotFound | DiskError::VolumeNotFound => {
self.not_found_responses = self.not_found_responses.saturating_add(1);
}
DiskError::FileVersionNotFound => {
self.version_not_found_responses = self.version_not_found_responses.saturating_add(1);
}
_ if is_metadata_fanout_ignored_error(err) => {
self.ignored_errors = self.ignored_errors.saturating_add(1);
}
_ => {
self.hard_errors = self.hard_errors.saturating_add(1);
}
}
}
pub(in crate::set_disk) fn early_stop_decision(&self) -> Option<MetadataEarlyStopDecision> {
if !self.allow_early_stop {
return None;
}
if self.delete_marker_votes >= self.default_write_quorum() {
return Some(MetadataEarlyStopDecision {
reason: GET_METADATA_EARLY_STOP_REASON_DELETE_MARKER,
});
}
if self.conflicting_metadata
|| self.delete_marker_seen
|| self.not_found_responses > 0
|| self.version_not_found_responses > 0
|| self.hard_errors > 0
{
return None;
}
if self
.candidate
.as_ref()
.and_then(|candidate| self.candidate_latest_quorum(candidate))
.is_some_and(|latest_quorum| self.candidate_votes >= latest_quorum)
{
return Some(MetadataEarlyStopDecision {
reason: GET_METADATA_EARLY_STOP_REASON_VALID_QUORUM,
});
}
None
}
/// Check if a versioned request can early-stop because the requested
/// version_id has reached quorum across disks.
pub(in crate::set_disk) fn version_early_stop_decision(&self) -> Option<MetadataEarlyStopDecision> {
if !self.allow_early_stop {
return None;
}
if self.requested_version_id.is_empty() {
return None;
}
if self.conflicting_metadata
|| self.delete_marker_seen
|| self.not_found_responses > 0
|| self.version_not_found_responses > 0
|| self.hard_errors > 0
{
return None;
}
if self.matching_version_votes >= self.read_quorum_for_version() {
return Some(MetadataEarlyStopDecision {
reason: GET_METADATA_EARLY_STOP_REASON_VERSION_MATCH_QUORUM,
});
}
None
}
pub(in crate::set_disk) fn can_still_reach_early_stop_with_pending(&self, pending: usize) -> bool {
if !self.allow_early_stop {
return false;
}
if self.delete_marker_votes.saturating_add(pending) >= self.default_write_quorum() {
return true;
}
if self.conflicting_metadata
|| self.delete_marker_seen
|| self.not_found_responses > 0
|| self.version_not_found_responses > 0
|| self.hard_errors > 0
{
return false;
}
if !self.requested_version_id.is_empty()
&& self.matching_version_votes.saturating_add(pending) >= self.read_quorum_for_version()
{
return true;
}
match &self.candidate {
Some(candidate) => self
.candidate_latest_quorum(candidate)
.is_some_and(|latest_quorum| self.candidate_votes.saturating_add(pending) >= latest_quorum),
None => pending >= self.default_write_quorum(),
}
}
/// Compute the read quorum threshold for version-aware early-stop.
/// Uses `total_disks / 2` (like `missing_response_quorum`) when
/// `default_parity_count` is set, otherwise requires all disks.
pub(in crate::set_disk) fn read_quorum_for_version(&self) -> usize {
self.missing_response_quorum()
}
pub(in crate::set_disk) fn final_miss_reason(&self) -> &'static str {
if !self.allow_early_stop {
return GET_METADATA_EARLY_STOP_REASON_UNSAFE_REQUEST;
}
if self.conflicting_metadata {
return GET_METADATA_EARLY_STOP_REASON_CONFLICTING_METADATA;
}
if self.delete_marker_seen {
return GET_METADATA_EARLY_STOP_REASON_DELETE_MARKER;
}
let missing_response_quorum = self.missing_response_quorum();
if self.version_not_found_responses >= missing_response_quorum {
return GET_METADATA_EARLY_STOP_REASON_VERSION_NOT_FOUND;
}
if self.not_found_responses >= missing_response_quorum {
return GET_METADATA_EARLY_STOP_REASON_NOT_FOUND;
}
if self.hard_errors > 0 {
return GET_METADATA_EARLY_STOP_REASON_ERROR;
}
if self.ignored_errors > 0 {
return GET_METADATA_EARLY_STOP_REASON_INSUFFICIENT_QUORUM;
}
GET_METADATA_EARLY_STOP_REASON_INSUFFICIENT_QUORUM
}
pub(in crate::set_disk) fn candidate_latest_quorum(&self, candidate: &FileInfo) -> Option<usize> {
if self.default_parity_count == 0 {
return Some(self.total_disks);
}
if candidate.is_canonical_delete_marker() || candidate.size == 0 || candidate.erasure.parity_blocks >= self.total_disks {
return None;
}
let data_blocks = candidate.erasure.data_blocks;
Some(if data_blocks == candidate.erasure.parity_blocks {
data_blocks.saturating_add(1)
} else {
data_blocks
})
}
pub(crate) fn default_write_quorum(&self) -> usize {
if self.default_parity_count == 0 || self.default_parity_count >= self.total_disks {
return self.total_disks;
}
let data_blocks = self.total_disks.saturating_sub(self.default_parity_count);
if data_blocks == self.default_parity_count {
data_blocks.saturating_add(1)
} else {
data_blocks
}
}
pub(in crate::set_disk) fn missing_response_quorum(&self) -> usize {
if self.default_parity_count == 0 || self.default_parity_count >= self.total_disks {
self.total_disks
} else {
self.total_disks / 2
}
}
}
pub(in crate::set_disk) fn metadata_early_stop_candidate_matches(left: &FileInfo, right: &FileInfo) -> bool {
left.volume == right.volume
&& left.name == right.name
&& left.version_id == right.version_id
&& left.is_latest == right.is_latest
&& left.deleted == right.deleted
&& left.mark_deleted == right.mark_deleted
&& left.transition_status == right.transition_status
&& left.transitioned_objname == right.transitioned_objname
&& left.transition_tier == right.transition_tier
&& left.transition_version_id == right.transition_version_id
&& left.transition_version == right.transition_version
&& left.transition_version_state == right.transition_version_state
&& left.expire_restored == right.expire_restored
&& left.size == right.size
&& left.mod_time == right.mod_time
&& left.mode == right.mode
&& left.written_by_version == right.written_by_version
&& left.metadata == right.metadata
&& left.replication_state_internal == right.replication_state_internal
&& left.parts == right.parts
&& left.checksum == right.checksum
&& left.versioned == right.versioned
&& left.num_versions == right.num_versions
&& left.successor_mod_time == right.successor_mod_time
&& left.data_dir == right.data_dir
&& left.erasure.algorithm == right.erasure.algorithm
&& left.erasure.data_blocks == right.erasure.data_blocks
&& left.erasure.parity_blocks == right.erasure.parity_blocks
&& left.erasure.block_size == right.erasure.block_size
&& left.erasure.distribution == right.erasure.distribution
}
pub(in crate::set_disk) fn is_metadata_fanout_ignored_error(err: &DiskError) -> bool {
OBJECT_OP_IGNORED_ERRS.iter().any(|ignored| ignored == err)
}
-1
View File
@@ -18,4 +18,3 @@
//! duplicating read/write/erasure logic. //! duplicating read/write/erasure logic.
pub(crate) mod io_primitives; pub(crate) mod io_primitives;
mod metadata_quorum;
+1 -1
View File
@@ -876,7 +876,7 @@ pub use ops::multipart::{MultipartCommitBarrier, MultipartCommitPause};
pub(crate) use ops::object::DeleteObjectCommitBarrier; pub(crate) use ops::object::DeleteObjectCommitBarrier;
#[cfg(any(test, feature = "test-util"))] #[cfg(any(test, feature = "test-util"))]
pub(crate) use ops::object::TransitionCleanupStoreBarrier as SetDiskTransitionCleanupStoreBarrier; pub(crate) use ops::object::TransitionCleanupStoreBarrier as SetDiskTransitionCleanupStoreBarrier;
#[cfg(all(test, feature = "test-util"))] #[cfg(test)]
pub(crate) use ops::object::TransitionUploadedCommitBarrier as SetDiskTransitionUploadedCommitBarrier; pub(crate) use ops::object::TransitionUploadedCommitBarrier as SetDiskTransitionUploadedCommitBarrier;
pub(crate) use ops::object::body_cache_plaintext_len; pub(crate) use ops::object::body_cache_plaintext_len;
#[cfg(all(test, feature = "test-util"))] #[cfg(all(test, feature = "test-util"))]
+3 -363
View File
@@ -2490,9 +2490,9 @@ impl crate::storage_api_contracts::heal::HealOperations for SetDisks {
return Ok((result, err.map(|e| e.into()))); return Ok((result, err.map(|e| e.into())));
} }
// The inner heal and missing-object report read the registry again; let disks = self.disks.read().await;
// release this snapshot guard before a topology writer can queue between reads.
let disks = self.get_disks_internal().await; let disks = disks.clone();
let (_, errs) = Self::read_all_fileinfo(&disks, "", bucket, object, version_id, false, false, false) let (_, errs) = Self::read_all_fileinfo(&disks, "", bucket, object, version_id, false, false, false)
.await .await
.map_err(|e| to_object_err(e.into(), vec![bucket, object]))?; .map_err(|e| to_object_err(e.into(), vec![bucket, object]))?;
@@ -3419,366 +3419,6 @@ mod heal_result_report_tests {
assert_eq!(unformatted, DiskError::UnformattedDisk); assert_eq!(unformatted, DiskError::UnformattedDisk);
} }
#[derive(Clone, Copy)]
enum InventoryWriterHealCase {
Existing,
Missing,
MissingVersion,
}
async fn assert_heal_object_inventory_writer(case: InventoryWriterHealCase) {
use crate::set_disk::core::io_primitives::disk_call_counters;
use std::time::Duration;
use tokio::io::AsyncReadExt;
let (_temp_dirs, disks, set) = hermetic_set_disks_isolated(4).await;
let bucket = "heal-inventory-writer-bucket";
let object = match case {
InventoryWriterHealCase::Existing => "heal-inventory-writer-existing",
InventoryWriterHealCase::Missing => "heal-inventory-writer-missing",
InventoryWriterHealCase::MissingVersion => "heal-inventory-writer-missing-version",
};
set.make_bucket(
bucket,
&MakeBucketOptions {
versioning_enabled: true,
..Default::default()
},
)
.await
.expect("heal fixture bucket should be created");
let body = vec![0x67; 64 * 1024];
let stored_version = Uuid::new_v4();
let stored_version_string = stored_version.to_string();
let published = if matches!(case, InventoryWriterHealCase::Missing) {
None
} else {
let mut reader = PutObjReader::from_vec(body.clone());
let info = set
.put_object(
bucket,
object,
&mut reader,
&ObjectOptions {
no_lock: true,
versioned: true,
version_id: Some(stored_version_string.clone()),
..Default::default()
},
)
.await
.expect("full-fanout PUT should seed the heal fixture");
for disk in &disks {
let metadata = disk
.read_version("", bucket, object, &stored_version_string, &ReadOptions::default())
.await
.expect("the seeded version must be present on every disk");
assert_eq!(metadata.version_id, Some(stored_version));
assert_eq!(metadata.size, i64::try_from(body.len()).expect("fixture size should fit i64"));
}
Some(info)
};
let requested_version = match case {
InventoryWriterHealCase::Existing => stored_version_string.clone(),
InventoryWriterHealCase::Missing => String::new(),
InventoryWriterHealCase::MissingVersion => Uuid::new_v4().to_string(),
};
let opts = HealOpts {
no_lock: true,
..Default::default()
};
let calls = disk_call_counters::observe(object);
let read_gate = set.disks.read().await;
// UFCS selects the trait's outer precheck, not the same-named inherent heal.
let heal = <SetDisks as crate::storage_api_contracts::heal::HealOperations>::heal_object(
set.as_ref(),
bucket,
object,
&requested_version,
&opts,
);
tokio::pin!(heal);
assert!(matches!(
futures::poll!(tokio::task::unconstrained(heal.as_mut())),
std::task::Poll::Pending
));
// These tests use the current-thread runtime: full-wait metadata tasks
// have been spawned, but cannot run during the single unconstrained poll.
assert_eq!(calls.total(disk_call_counters::KIND_READ_VERSION), 0);
let writer = set.disks.write();
tokio::pin!(writer);
assert!(matches!(
futures::poll!(tokio::task::unconstrained(writer.as_mut())),
std::task::Poll::Pending
));
assert!(set.disks.try_read().is_err(), "the writer must already block new inventory readers");
tokio::time::timeout(Duration::from_secs(5), async {
while calls.total(disk_call_counters::KIND_READ_VERSION) < 4 {
tokio::task::yield_now().await;
}
})
.await
.expect("the suspended trait heal must have started the real metadata fanout");
for disk_index in 0..4 {
assert_eq!(calls.for_disk(disk_call_counters::KIND_READ_VERSION, disk_index), 1);
}
drop(read_gate);
let (_, outcome) =
tokio::time::timeout(Duration::from_secs(5), async { tokio::join!(async { drop(writer.await) }, heal) })
.await
.expect("trait heal must not deadlock its nested inventory read with the queued writer");
let (result, error) = outcome.expect("heal should report the object's outcome");
match case {
InventoryWriterHealCase::Existing => assert!(error.is_none(), "existing object heal failed: {error:?}"),
InventoryWriterHealCase::Missing => assert!(matches!(error, Some(Error::FileNotFound))),
InventoryWriterHealCase::MissingVersion => assert!(matches!(error, Some(Error::FileVersionNotFound))),
}
assert_eq!(result.bucket, bucket);
assert_eq!(result.object, object);
assert_eq!(result.version_id, requested_version);
assert_eq!(result.disk_count, 4);
assert_eq!(result.before.drives.len(), 4);
assert_eq!(result.after.drives.len(), 4);
for disk_index in 0..4 {
let endpoint = set.set_endpoints[disk_index].to_string();
assert_eq!(result.before.drives[disk_index].endpoint, endpoint);
assert_eq!(result.after.drives[disk_index].endpoint, endpoint);
}
if let Some(published) = published {
tokio::time::timeout(Duration::from_secs(10), async {
let mut reader = set
.get_object_reader(
bucket,
object,
None,
Default::default(),
&ObjectOptions {
versioned: true,
version_id: Some(stored_version_string),
..Default::default()
},
)
.await
.expect("the stored version must remain readable after heal");
assert_eq!(reader.object_info.etag, published.etag);
assert_eq!(reader.object_info.version_id, Some(stored_version));
let mut observed_body = Vec::new();
reader
.stream
.read_to_end(&mut observed_body)
.await
.expect("stored body should stream");
assert_eq!(observed_body, body);
})
.await
.expect("GET must finish after the inventory writer and heal");
}
}
#[tokio::test]
async fn heal_object_inventory_writer_existing() {
assert_heal_object_inventory_writer(InventoryWriterHealCase::Existing).await;
}
#[tokio::test]
async fn heal_object_inventory_writer_missing() {
assert_heal_object_inventory_writer(InventoryWriterHealCase::Missing).await;
}
#[tokio::test]
async fn heal_object_inventory_writer_missing_version() {
assert_heal_object_inventory_writer(InventoryWriterHealCase::MissingVersion).await;
}
#[tokio::test]
#[serial_test::serial]
async fn heal_object_with_queued_disk_renewal() {
use crate::layout::endpoints::SetupType;
use crate::runtime::instance::InstanceContext;
use crate::set_disk::core::io_primitives::disk_call_counters;
use std::collections::HashMap;
use std::future::Future;
use std::task::Poll;
use std::time::Duration;
use tokio::io::AsyncReadExt;
// renew_disk still registers local disks on the ambient context. Match
// the default serial group used by its other setup/registry fixtures,
// and restore only this temporary endpoint, including on a failed join.
struct RenewDiskTestState {
ctx: Arc<InstanceContext>,
was_dist_erasure: bool,
map: Arc<RwLock<HashMap<String, Option<DiskStore>>>>,
endpoint: String,
previous_disk: Option<Option<DiskStore>>,
}
impl Drop for RenewDiskTestState {
fn drop(&mut self) {
let ctx = self.ctx.clone();
let was_dist_erasure = self.was_dist_erasure;
let map = self.map.clone();
let endpoint = self.endpoint.clone();
let previous_disk = self.previous_disk.take();
let handle = tokio::runtime::Handle::current();
std::thread::spawn(move || {
handle.block_on(async move {
let mut map = map.write().await;
match previous_disk {
Some(disk) => {
map.insert(endpoint, disk);
}
None => {
map.remove(&endpoint);
}
}
drop(map);
if was_dist_erasure {
ctx.update_erasure_type(SetupType::DistErasure).await;
}
});
})
.join()
.expect("renew fixture state restoration should finish");
}
}
let (_temp_dirs, disks, set) = hermetic_set_disks_isolated(4).await;
let endpoint = set.set_endpoints[0].clone();
let ctx = crate::runtime::global::current_ctx();
let map = ctx.local_disk_map();
let restore = RenewDiskTestState {
ctx: ctx.clone(),
was_dist_erasure: ctx.is_dist_erasure().await,
map: map.clone(),
endpoint: endpoint.to_string(),
previous_disk: map.read().await.get(&endpoint.to_string()).cloned(),
};
// Only distributed erasure needs an override to avoid the ambient slot array.
if restore.was_dist_erasure {
ctx.update_erasure_type(SetupType::Erasure).await;
}
let bucket = "heal-disk-renewal-bucket";
let object = "heal-disk-renewal-object";
set.make_bucket(bucket, &MakeBucketOptions::default())
.await
.expect("renew fixture bucket should be created");
let body = vec![0x73; 64 * 1024];
let mut reader = PutObjReader::from_vec(body.clone());
let published = set
.put_object(
bucket,
object,
&mut reader,
&ObjectOptions {
no_lock: true,
..Default::default()
},
)
.await
.expect("full-fanout PUT should seed the renewal fixture");
for disk in &disks {
let metadata = disk
.read_version("", bucket, object, "", &ReadOptions::default())
.await
.expect("the seeded object must be present on every disk");
assert_eq!(metadata.size, i64::try_from(body.len()).expect("fixture size should fit i64"));
}
let opts = HealOpts {
no_lock: true,
..Default::default()
};
let calls = disk_call_counters::observe(object);
let read_gate = set.disks.read().await;
let heal = <SetDisks as crate::storage_api_contracts::heal::HealOperations>::heal_object(
set.as_ref(),
bucket,
object,
"",
&opts,
);
tokio::pin!(heal);
assert!(matches!(futures::poll!(tokio::task::unconstrained(heal.as_mut())), Poll::Pending));
assert_eq!(calls.total(disk_call_counters::KIND_READ_VERSION), 0);
let renew = set.renew_disk(&endpoint);
tokio::pin!(renew);
tokio::time::timeout(
Duration::from_secs(5),
futures::future::poll_fn(|cx| {
assert!(
std::pin::pin!(tokio::task::unconstrained(renew.as_mut()))
.poll(cx)
.is_pending(),
"renewal must reach its inventory write before returning"
);
if set.disks.try_read().is_err() {
Poll::Ready(())
} else {
Poll::Pending
}
}),
)
.await
.expect("real renewal must queue its topology writer behind the read gate");
let registered = map
.read()
.await
.get(&endpoint.to_string())
.cloned()
.flatten()
.expect("renewal must register the connected disk before its inventory write");
assert!(!Arc::ptr_eq(&registered, &disks[0]), "renewal must construct a new disk handle");
tokio::time::timeout(Duration::from_secs(5), async {
while calls.total(disk_call_counters::KIND_READ_VERSION) < 4 {
tokio::task::yield_now().await;
}
})
.await
.expect("the suspended trait heal must have started the real metadata fanout");
for disk_index in 0..4 {
assert_eq!(calls.for_disk(disk_call_counters::KIND_READ_VERSION, disk_index), 1);
}
drop(read_gate);
let (_, outcome) = tokio::time::timeout(Duration::from_secs(5), async { tokio::join!(renew, heal) })
.await
.expect("trait heal and real disk renewal must finish without a nested inventory read deadlock");
let (report, error) = outcome.expect("heal should report the existing object");
assert!(error.is_none(), "existing object heal failed after renewal: {error:?}");
assert_eq!(report.bucket, bucket);
assert_eq!(report.object, object);
assert_eq!(report.disk_count, 4);
let renewed = set.get_disks_internal().await[0]
.clone()
.expect("the renewed slot must remain online");
assert!(Arc::ptr_eq(&renewed, &registered), "the set must publish the newly connected handle");
assert_eq!(renewed.endpoint(), endpoint);
let format = load_format_erasure(&renewed, false)
.await
.expect("renewed disk format should remain readable");
assert_eq!(format.erasure.this, set.format.erasure.sets[0][0]);
tokio::time::timeout(Duration::from_secs(10), async {
let mut reader = set
.get_object_reader(bucket, object, None, Default::default(), &ObjectOptions::default())
.await
.expect("the object must remain readable after renewal and heal");
assert_eq!(reader.object_info.etag, published.etag);
let mut observed_body = Vec::new();
reader
.stream
.read_to_end(&mut observed_body)
.await
.expect("stored body should stream");
assert_eq!(observed_body, body);
})
.await
.expect("GET must finish after renewal and heal");
}
// Regression for #955: an offline disk must contribute exactly one drive // Regression for #955: an offline disk must contribute exactly one drive
// record. Before the fix the offline branch fell through and pushed a second // record. Before the fix the offline branch fell through and pushed a second
// (Corrupt) record for the same disk, so `before/after.drives` grew to // (Corrupt) record for the same disk, so `before/after.drives` grew to
+23 -119
View File
@@ -2452,9 +2452,10 @@ impl crate::storage_api_contracts::multipart::MultipartOperations for SetDisks {
let write_quorum = fi.write_quorum(self.default_write_quorum()); let write_quorum = fi.write_quorum(self.default_write_quorum());
let read_quorum = fi.read_quorum(self.default_read_quorum()); let read_quorum = fi.read_quorum(self.default_read_quorum());
// Release the registry guard before recovery and cleanup read it again: let disks = self.disks.read().await;
// a queued topology writer would otherwise deadlock those nested reads.
let disks = self.get_disks_internal().await; let disks = disks.clone();
// let disks = Self::shuffle_disks(&disks, &fi.erasure.distribution);
let part_path = format!("{}/{}/", upload_id_path, fi.data_dir.unwrap_or(Uuid::nil())); let part_path = format!("{}/{}/", upload_id_path, fi.data_dir.unwrap_or(Uuid::nil()));
self.recover_part_transactions(&part_path, read_quorum, write_quorum) self.recover_part_transactions(&part_path, read_quorum, write_quorum)
@@ -4050,7 +4051,6 @@ mod tests {
let _ = drain_global_dirty_scopes(); let _ = drain_global_dirty_scopes();
let rename_barrier = rename_fanout_barrier::arm(object, 0, rename_fanout_barrier::PHASE_RENAME); let rename_barrier = rename_fanout_barrier::arm(object, 0, rename_fanout_barrier::PHASE_RENAME);
let rename_tasks = rename_fanout_barrier::observe_tasks(object);
let complete_store = Arc::clone(&set_disks); let complete_store = Arc::clone(&set_disks);
let mut complete = tokio::spawn(async move { let mut complete = tokio::spawn(async move {
let mut opts = ObjectOptions::default(); let mut opts = ObjectOptions::default();
@@ -4062,6 +4062,16 @@ mod tests {
tokio::time::timeout(Duration::from_secs(30), rename_barrier.wait_until_paused()) tokio::time::timeout(Duration::from_secs(30), rename_barrier.wait_until_paused())
.await .await
.expect("multipart completion should pause one tail disk during rename"); .expect("multipart completion should pause one tail disk during rename");
assert!(
tokio::time::timeout(Duration::from_millis(100), &mut complete).await.is_err(),
"multipart completion must not publish success while a tail rename is still paused"
);
let initial = drain_global_dirty_scopes().into_iter().collect::<HashSet<_>>();
assert!(
initial.is_empty(),
"capacity must not be marked as committed before the full multipart rename finishes"
);
let abort_store = Arc::clone(&set_disks); let abort_store = Arc::clone(&set_disks);
let abort = tokio::spawn(async move { let abort = tokio::spawn(async move {
@@ -4070,46 +4080,21 @@ mod tests {
.await .await
}); });
signaling.wait_for_attempts(2).await; signaling.wait_for_attempts(2).await;
assert!(!abort.is_finished(), "the in-flight completion must retain the multipart upload guard");
// A paused rename does not establish that the other disks reached quorum. let retained_staging = futures::future::join_all(
let retained_staging = tokio::time::timeout(Duration::from_secs(30), async { disk_stores
loop { .iter()
let mut retained = 0; .map(|disk| disk.read_all(RUSTFS_META_MULTIPART_BUCKET, &staged_part)),
for result in futures::future::join_all( )
disk_stores
.iter()
.map(|disk| disk.read_all(RUSTFS_META_MULTIPART_BUCKET, &staged_part)),
)
.await
{
match result {
Ok(_) => retained += 1,
Err(DiskError::FileNotFound) => {}
Err(error) => panic!("staged rename source lookup failed: {error}"),
}
}
if retained <= 1 && rename_tasks.running() == 1 {
break retained;
}
tokio::time::sleep(Duration::from_millis(10)).await;
}
})
.await .await
.expect("unpaused multipart renames should finish before the tail is released"); .into_iter()
.filter(|result| result.is_ok())
.count();
assert_eq!( assert_eq!(
retained_staging, 1, retained_staging, 1,
"only the paused tail disk should still retain the multipart rename source" "only the paused tail disk should still retain the multipart rename source"
); );
assert!(
tokio::time::timeout(Duration::from_millis(100), &mut complete).await.is_err(),
"multipart completion must not publish success while a tail rename is still paused"
);
let initial = drain_global_dirty_scopes().into_iter().collect::<HashSet<_>>();
assert!(
initial.is_empty(),
"capacity must not be marked as committed before the full multipart rename finishes"
);
assert!(!abort.is_finished(), "the in-flight completion must retain the multipart upload guard");
signaling.set_target(rustfs_lock::ObjectKey::new(bucket, object)); signaling.set_target(rustfs_lock::ObjectKey::new(bucket, object));
let object_attempt = signaling.attempts.load(Ordering::Acquire) + 1; let object_attempt = signaling.attempts.load(Ordering::Acquire) + 1;
@@ -6758,87 +6743,6 @@ mod tests {
.await; .await;
} }
#[tokio::test(flavor = "multi_thread")]
#[serial]
async fn complete_multipart_releases_disk_snapshot_before_cleanup() {
let (temp_dirs, disk_stores, set_disks) = hermetic_set_disks(4).await;
let bucket = "multipart-topology-lock-bucket";
let object = "object";
let body = vec![0x65; 4096];
make_bucket_on_all(&disk_stores, bucket).await;
let (upload_id, parts) =
stage_upload_with_create_opts(&set_disks, bucket, object, &body, &ObjectOptions::default()).await;
let upload_id_path = SetDisks::get_upload_id_dir(bucket, object, &upload_id);
for dir in &temp_dirs {
assert!(
dir.path().join(RUSTFS_META_MULTIPART_BUCKET).join(&upload_id_path).exists(),
"the test must create real upload staging on every disk"
);
}
let barrier = MultipartCommitBarrier::install(bucket, object, MultipartCommitPause::AfterObjectPublication);
let complete_store = set_disks.clone();
let complete_upload_id = upload_id.clone();
let complete = tokio::spawn(async move {
complete_store
.complete_multipart_upload(bucket, object, &complete_upload_id, parts, &ObjectOptions::default())
.await
});
barrier.wait_until_paused().await;
// Hold a separate read gate so the real writer queues even when completion
// correctly releases its snapshot guard. Polling Pending proves admission
// to Tokio's write-preferring queue before the cleanup attempts another read.
let read_gate = set_disks.disks.read().await;
let writer = set_disks.disks.write();
tokio::pin!(writer);
assert!(matches!(
futures::poll!(tokio::task::unconstrained(writer.as_mut())),
std::task::Poll::Pending
));
assert!(
set_disks.disks.try_read().is_err(),
"the pending writer must already block new readers before the cleanup resumes"
);
drop(read_gate);
barrier.release();
let writer_guard = tokio::time::timeout(Duration::from_secs(5), writer)
.await
.expect("a queued topology writer must not deadlock with multipart cleanup's disk snapshot");
// A reconnect can publish the same handles; this test isolates admission
// order without changing the disks that contain the committed object.
drop(writer_guard);
tokio::time::timeout(Duration::from_secs(10), complete)
.await
.expect("multipart cleanup must finish after the topology writer releases")
.expect("completion task should not panic")
.expect("completion should preserve the successful object commit");
let mut reader = tokio::time::timeout(
Duration::from_secs(10),
set_disks.get_object_reader(bucket, object, None, HeaderMap::new(), &ObjectOptions::default()),
)
.await
.expect("GET should finish after completion")
.expect("the completed object should remain readable");
let mut observed_body = Vec::new();
tokio::time::timeout(Duration::from_secs(10), reader.stream.read_to_end(&mut observed_body))
.await
.expect("the completed object body should finish streaming")
.expect("the completed object body should be readable");
assert_eq!(observed_body, body);
assert!(matches!(
set_disks.check_upload_id_exists(bucket, object, &upload_id, false).await,
Err(StorageError::InvalidUploadID(..))
));
for dir in &temp_dirs {
assert!(
!dir.path().join(RUSTFS_META_MULTIPART_BUCKET).join(&upload_id_path).exists(),
"successful completion must remove its upload staging from every disk"
);
}
}
#[tokio::test(flavor = "multi_thread")] #[tokio::test(flavor = "multi_thread")]
#[serial] #[serial]
async fn complete_releases_object_lock_before_cleanup_and_keeps_upload_lock() { async fn complete_releases_object_lock_before_cleanup_and_keeps_upload_lock() {
+31 -447
View File
@@ -299,11 +299,11 @@ use crate::error::is_err_invalid_upload_id;
use crate::object_api::{GetObjectBodySource, get_object_body_cache_hook_suppressed}; use crate::object_api::{GetObjectBodySource, get_object_body_cache_hook_suppressed};
use crate::object_api::{ use crate::object_api::{
NamespaceLockFence, ReplicationStatusWritebackCondition, ReplicationStatusWritebackMode, NamespaceLockFence, ReplicationStatusWritebackCondition, ReplicationStatusWritebackMode,
SCANNER_PUBLICATION_LEASE_FENCE_METADATA_KEY, WriteCompletion, SCANNER_PUBLICATION_LEASE_FENCE_METADATA_KEY,
}; };
use crate::services::notification_sys::RemoteVersionStateFleetProofToken; use crate::services::notification_sys::RemoteVersionStateFleetProofToken;
use crate::services::tier::tier::{TierConfigMgr, TierDestinationId, TierOperationLease, tier_destination_id_from_metadata}; use crate::services::tier::tier::{TierConfigMgr, TierDestinationId, TierOperationLease, tier_destination_id_from_metadata};
use crate::set_disk::core::io_primitives::{RenameRollbackReceipt, RenameTailCleanup, finish_rename_tail_heal}; use crate::set_disk::core::io_primitives::{RenameTailCleanup, finish_rename_tail_heal};
#[cfg(test)] #[cfg(test)]
use crate::storage_api_contracts::namespace::NamespaceLocking; use crate::storage_api_contracts::namespace::NamespaceLocking;
#[cfg(test)] #[cfg(test)]
@@ -3548,7 +3548,6 @@ impl SetDisks {
(None, None, None) (None, None, None)
}; };
let mut tmp_cleanup_owned = false; let mut tmp_cleanup_owned = false;
let rollback_receipt = RenameRollbackReceipt::default();
let operation = async { let operation = async {
let erasure = Arc::new(erasure_from_file_info(&fi, false)?); let erasure = Arc::new(erasure_from_file_info(&fi, false)?);
@@ -4257,7 +4256,6 @@ impl SetDisks {
let commit_bucket = bucket.to_owned(); let commit_bucket = bucket.to_owned();
let commit_object = object.to_owned(); let commit_object = object.to_owned();
let commit_tmp_dir = tmp_dir.clone(); let commit_tmp_dir = tmp_dir.clone();
let commit_rollback_receipt = rollback_receipt.clone();
let commit_object_lock_guard = object_lock_guard.take(); let commit_object_lock_guard = object_lock_guard.take();
let commit_decommission_object_lock_guard = decommission_object_lock_guard.take(); let commit_decommission_object_lock_guard = decommission_object_lock_guard.take();
let commit_publication_guard = publication_commit_guard.take(); let commit_publication_guard = publication_commit_guard.take();
@@ -4268,17 +4266,13 @@ impl SetDisks {
// complete rename fan-out drains. Keep this path synchronous so // complete rename fan-out drains. Keep this path synchronous so
// its terminal state is known before the coordinator releases // its terminal state is known before the coordinator releases
// remote leases. // remote leases.
let commit_owns_namespace_guard = commit_object_lock_guard.is_some() let commit_allows_early_ack = !(opts.data_movement && opts.has_decommission_capacity_reservation())
|| commit_decommission_object_lock_guard.is_some() && (commit_object_lock_guard.is_some()
|| commit_publication_guard.is_some(); || commit_decommission_object_lock_guard.is_some()
let commit_allows_early_ack = opts.write_completion == WriteCompletion::Quorum || commit_publication_guard.is_some())
&& !(opts.data_movement && opts.has_decommission_capacity_reservation())
&& commit_owns_namespace_guard
&& commit_scanner_publication_scope.is_none(); && commit_scanner_publication_scope.is_none();
// Full-tail callers also transfer owned guards to the coordinator:
// cancelling their ACK waiter must not cancel an in-flight rename.
let detach_commit_owner = commit_scanner_publication_scope.is_some() let detach_commit_owner = commit_scanner_publication_scope.is_some()
|| commit_owns_namespace_guard || commit_allows_early_ack
|| commit_bucket_lifecycle_guard.is_some() || commit_bucket_lifecycle_guard.is_some()
|| quota_mutation_fence; || quota_mutation_fence;
let commit_write_path_label = write_path.metric_label(); let commit_write_path_label = write_path.metric_label();
@@ -4458,11 +4452,7 @@ impl SetDisks {
write_quorum, write_quorum,
commit_scanner_publication_lease_tokens.as_ref(), commit_scanner_publication_lease_tokens.as_ref(),
) )
.with_publication_scope(commit_scanner_publication_scope.clone()) .with_publication_scope(commit_scanner_publication_scope.clone()),
.with_rollback_receipt(commit_rollback_receipt.clone())
.with_namespace_commit_guard(
(!is_meta_bucketname(&commit_bucket)).then(|| commit_set.ctx.begin_namespace_commit()),
),
) )
.await; .await;
if let Some(scope) = commit_scanner_publication_scope.as_ref() { if let Some(scope) = commit_scanner_publication_scope.as_ref() {
@@ -4595,11 +4585,6 @@ impl SetDisks {
let rename_commit = match rename_result { let rename_commit = match rename_result {
Ok(commit) => commit, Ok(commit) => commit,
Err(err) => { Err(err) => {
if commit_rollback_receipt.is_incomplete() {
// Incomplete undo retains the staging source and
// rollback backup for recovery; cleanup is unsafe.
return Err(err.into());
}
if let Err(cleanup_err) = commit_set.delete_all(RUSTFS_META_TMP_BUCKET, &commit_tmp_dir).await { if let Err(cleanup_err) = commit_set.delete_all(RUSTFS_META_TMP_BUCKET, &commit_tmp_dir).await {
warn!(tmp_dir = %commit_tmp_dir, error = ?cleanup_err, "failed to cleanup put_object temporary data"); warn!(tmp_dir = %commit_tmp_dir, error = ?cleanup_err, "failed to cleanup put_object temporary data");
} else if issue3031_diag_enabled() { } else if issue3031_diag_enabled() {
@@ -4632,8 +4617,9 @@ impl SetDisks {
request.object_version_id = committed_version_id request.object_version_id = committed_version_id
.or_else(|| commit_version_suspended.then(Uuid::nil)) .or_else(|| commit_version_suspended.then(Uuid::nil))
.map(|version_id| version_id.to_string()); .map(|version_id| version_id.to_string());
let heal_set = commit_set.clone(); tokio::spawn(async move {
tokio::spawn(async move { heal_set.submit_rename_tail_heal(request).await }); let _ = rustfs_heal_contracts::heal_channel::send_heal_request(request).await;
});
} }
let rename_stage_elapsed = rename_stage_start.elapsed(); let rename_stage_elapsed = rename_stage_start.elapsed();
@@ -4899,7 +4885,7 @@ impl SetDisks {
); );
} }
}); });
} else if !rollback_receipt.is_incomplete() { } else {
// Failure path (quorum loss / rollback): keep the cleanup inline so // Failure path (quorum loss / rollback): keep the cleanup inline so
// a failed PUT never returns while its tmp shards are still on disk // a failed PUT never returns while its tmp shards are still on disk
// (state-residue hardening tracked by backlog#864 / backlog#898). // (state-residue hardening tracked by backlog#864 / backlog#898).
@@ -17508,69 +17494,27 @@ mod put_object_tmp_cleanup_tests {
} }
#[tokio::test] #[tokio::test]
#[serial_test::serial(capacity_dirty_scope)]
async fn put_object_failure_cleans_tmp_workspace_inline() { async fn put_object_failure_cleans_tmp_workspace_inline() {
temp_env::async_with_vars([(ENV_RUSTFS_PUT_RENAME_EARLY_ACK_ENABLE, Some("true"))], async { let (temp_dirs, _disk_stores, set_disks) = hermetic_set_disks(4).await;
for write_completion in [WriteCompletion::Quorum, WriteCompletion::TailDrained] {
let (temp_dirs, _disk_stores, set_disks) = hermetic_set_disks(4).await;
let bucket = "tmp-clean-missing-bucket";
let object = "orphan-object";
let barrier = PutObjectCommitBarrier::install(bucket, object, PutObjectCommitPause::BeforeNamespace);
let writer = Arc::clone(&set_disks);
let put = tokio::spawn(async move {
let mut reader = PutObjReader::from_vec(vec![9u8; TEST_OBJECT_SIZE]);
writer
.put_object(
bucket,
object,
&mut reader,
&ObjectOptions {
write_completion,
..Default::default()
},
)
.await
});
tokio::time::timeout(Duration::from_secs(30), barrier.wait_until_paused())
.await
.expect("missing-bucket PUT must stage before rename");
let staged = non_trash_tmp_entries(&temp_dirs).await;
assert_eq!(staged.len(), 4, "every disk must have a staged workspace before rejection");
for workspace in staged {
let mut entries = tokio::fs::read_dir(&workspace)
.await
.expect("staged workspace should be readable");
let mut shards = 0;
while let Some(entry) = entries.next_entry().await.expect("staged data directory should be readable") {
if entry.file_type().await.expect("staged entry type").is_dir() {
let part = tokio::fs::metadata(entry.path().join("part.1"))
.await
.expect("staging must contain an actual erasure shard");
assert!(part.len() > 0, "the shard must be written before the missing-bucket failure");
shards += 1;
}
}
assert_eq!(shards, 1);
}
assert!(temp_dirs.iter().all(|dir| !dir.path().join(bucket).exists()));
barrier.release();
let err = tokio::time::timeout(Duration::from_secs(30), put)
.await
.expect("missing-bucket PUT must finish")
.expect("PUT task should join")
.expect_err("put_object into a missing bucket volume must fail");
assert!(matches!(err, StorageError::VolumeNotFound), "original disk error expected: {err}");
// No polling: known pre-publication rejection must clean staging // The bucket volume is never created, so the shards are written into
// inline, before PUT returns (backlog#864 / backlog#898). // the tmp workspace and the commit fails at rename_data with a quorum
let leftovers = non_trash_tmp_entries(&temp_dirs).await; // error — exercising the failure-path cleanup.
assert!( let mut reader = PutObjReader::from_vec(vec![9u8; TEST_OBJECT_SIZE]);
leftovers.is_empty(), let err = set_disks
"failed PUT must not leave tmp shards behind, leftovers: {leftovers:?}, err: {err}" .put_object("tmp-clean-missing-bucket", "orphan-object", &mut reader, &ObjectOptions::default())
); .await
} .expect_err("put_object into a missing bucket volume must fail");
})
.await; // No polling: the failure path must clean the tmp workspace inline,
// before put_object returns (backlog#864 / backlog#898 hardening).
let leftovers = non_trash_tmp_entries(&temp_dirs).await;
assert!(
leftovers.is_empty(),
"failed PUT must not leave tmp shards behind, leftovers: {leftovers:?}, err: {err}"
);
drop(temp_dirs);
} }
#[tokio::test] #[tokio::test]
@@ -18213,354 +18157,6 @@ mod put_object_tmp_cleanup_tests {
.await; .await;
} }
async fn make_completion_test_bucket(disks: &[DiskStore], bucket: &str) {
for disk in disks {
disk.make_volume(bucket)
.await
.expect("completion test bucket should be created");
}
}
/// Observe the actual metadata quorum while the remaining rename is parked.
/// A completed task count alone can race tasks that have not started yet.
async fn wait_for_paused_tail_metadata_quorum(disks: &[DiskStore], bucket: &str, object: &str) {
tokio::time::timeout(Duration::from_secs(30), async {
loop {
let mut committed = 0;
for disk in disks {
match disk.read_version("", bucket, object, "", &ReadOptions::default()).await {
Ok(_) => committed += 1,
Err(DiskError::FileNotFound | DiskError::FileVersionNotFound) => {}
Err(err) => panic!("unexpected metadata error while observing {bucket}/{object}: {err}"),
}
}
if committed == 3 {
break;
}
tokio::task::yield_now().await;
}
})
.await
.expect("three disks must publish metadata while the fourth rename remains paused");
}
#[tokio::test]
#[serial_test::serial(capacity_dirty_scope)]
async fn tail_drained_put_waits_for_tail_and_allows_immediate_cas() {
temp_env::async_with_vars([(ENV_RUSTFS_PUT_RENAME_EARLY_ACK_ENABLE, Some("true"))], async {
for size in [4096, 1024 * 1024] {
let (_dirs, disks, set) = hermetic_set_disks(4).await;
let bucket = "put-full-tail-cas";
let object = "full-tail-cas-object";
make_completion_test_bucket(&disks, bucket).await;
let tasks = rename_fanout_barrier::observe_tasks(object);
let barrier = rename_fanout_barrier::arm(object, 0, rename_fanout_barrier::PHASE_RENAME);
let writer = Arc::clone(&set);
let put = tokio::spawn(async move {
let mut reader = PutObjReader::from_vec(vec![b'1'; size]);
writer
.put_object(
bucket,
object,
&mut reader,
&ObjectOptions {
write_completion: WriteCompletion::TailDrained,
..Default::default()
},
)
.await
});
tokio::time::timeout(Duration::from_secs(30), barrier.wait_until_paused())
.await
.expect("full-tail PUT must reach the rename barrier");
wait_for_paused_tail_metadata_quorum(&disks, bucket, object).await;
assert!(!put.is_finished(), "full-tail PUT must remain pending after metadata quorum");
let mut lock_probe = Box::pin(set.acquire_write_lock_diag("full_tail_probe", bucket, object));
assert!(
futures::poll!(lock_probe.as_mut()).is_pending(),
"the owned namespace guard must remain held"
);
barrier.release();
let written = tokio::time::timeout(Duration::from_secs(30), put)
.await
.expect("full-tail PUT should finish after release")
.expect("full-tail PUT task should join")
.expect("full-tail PUT must commit");
assert_eq!(tasks.running(), 0, "full-tail response must follow every rename task");
drop(
tokio::time::timeout(Duration::from_secs(5), lock_probe)
.await
.expect("same-key lock should be available on return")
.expect("same-key lock probe should succeed"),
);
for disk in &disks {
disk.read_version("", bucket, object, "", &ReadOptions::default())
.await
.expect("successful full-tail PUT must publish on every healthy disk");
}
drop(barrier);
let mut replacement = PutObjReader::from_vec(b"cas successor".to_vec());
set.put_object(
bucket,
object,
&mut replacement,
&ObjectOptions {
write_completion: WriteCompletion::TailDrained,
http_preconditions: Some(HTTPPreconditions {
if_match: written.etag,
..Default::default()
}),
..Default::default()
},
)
.await
.expect("immediate same-key CAS must acquire the namespace guard");
let mut read = set
.get_object_reader(bucket, object, None, HeaderMap::new(), &ObjectOptions::default())
.await
.expect("CAS successor must be immediately readable");
let mut body = Vec::new();
read.stream.read_to_end(&mut body).await.expect("successor body must drain");
assert_eq!(body, b"cas successor");
}
})
.await;
}
#[tokio::test]
#[serial_test::serial(capacity_dirty_scope)]
async fn tail_drained_put_preserves_quorum_success_and_heals_failed_tail() {
let (_dirs, disks, set) = hermetic_set_disks(4).await;
let bucket = "put-full-tail-heal";
let object = "full-tail-heal-object";
make_completion_test_bucket(&disks, bucket).await;
let mut heals = set.capture_test_rename_tail_heals();
let tasks = rename_fanout_barrier::observe_tasks(object);
let barrier = rename_fanout_barrier::arm(object, 0, rename_fanout_barrier::PHASE_RENAME);
let _fault = rename_fault_injection::fail_rename_on(object, &[0]);
let writer = Arc::clone(&set);
let put = tokio::spawn(async move {
let mut reader = PutObjReader::from_vec(vec![b'1'; TEST_OBJECT_SIZE]);
writer
.put_object(
bucket,
object,
&mut reader,
&ObjectOptions {
write_completion: WriteCompletion::TailDrained,
..Default::default()
},
)
.await
});
tokio::time::timeout(Duration::from_secs(30), barrier.wait_until_paused())
.await
.expect("failed tail must first reach the rename barrier");
wait_for_paused_tail_metadata_quorum(&disks, bucket, object).await;
assert!(!put.is_finished(), "committed quorum must still wait for the failing tail");
barrier.release();
tokio::time::timeout(Duration::from_secs(30), put)
.await
.expect("failed tail should drain")
.expect("PUT task should join")
.expect("a minority tail error must not negate committed quorum");
assert_eq!(tasks.running(), 0);
let heal = tokio::time::timeout(Duration::from_secs(30), heals.recv())
.await
.expect("failed tail must schedule heal")
.expect("heal capture must remain connected");
assert_eq!(heal.bucket, bucket);
assert_eq!(heal.object_prefix.as_deref(), Some(object));
let info = set
.get_object_info(bucket, object, &ObjectOptions::default())
.await
.expect("committed object must remain readable despite the failed tail");
assert_eq!(info.size, TEST_OBJECT_SIZE as i64);
}
#[tokio::test]
#[serial_test::serial(capacity_dirty_scope)]
async fn tail_drained_put_rejects_quorum_minus_one() {
let (_dirs, disks, set) = hermetic_set_disks(4).await;
let bucket = "put-full-tail-no-quorum";
let object = "full-tail-no-quorum-object";
make_completion_test_bucket(&disks, bucket).await;
let _fault = rename_fault_injection::fail_rename_on(object, &[0, 1]);
let tasks = rename_fanout_barrier::observe_tasks(object);
let mut reader = PutObjReader::from_vec(vec![b'1'; TEST_OBJECT_SIZE]);
let err = set
.put_object(
bucket,
object,
&mut reader,
&ObjectOptions {
write_completion: WriteCompletion::TailDrained,
..Default::default()
},
)
.await
.expect_err("draining two successful disks cannot satisfy write quorum three");
assert!(
matches!(err, Error::ErasureWriteQuorum | Error::InsufficientWriteQuorum(_, _)),
"original quorum error expected: {err}"
);
assert_eq!(tasks.running(), 0, "failed fan-out and rollback must complete before return");
assert!(
set.get_object_info(bucket, object, &ObjectOptions::default()).await.is_err(),
"failed fresh write must not become visible"
);
}
#[tokio::test]
#[serial_test::serial(capacity_dirty_scope)]
async fn put_incomplete_rollback_preserves_staging_and_old_version_backup() {
use crate::set_disk::core::io_primitives::rollback_fault_injection;
temp_env::async_with_vars([(ENV_RUSTFS_PUT_RENAME_EARLY_ACK_ENABLE, Some("true"))], async {
for write_completion in [WriteCompletion::Quorum, WriteCompletion::TailDrained] {
for fault in [
rollback_fault_injection::Fault::Io,
rollback_fault_injection::Fault::VolumeNotFoundAfterRename,
] {
let (dirs, disks, set) = hermetic_set_disks(4).await;
let bucket = "put-incomplete-undo";
let object = "incomplete-undo-object";
make_completion_test_bucket(&disks, bucket).await;
let mut old_reader = PutObjReader::from_vec(vec![b'0'; TEST_OBJECT_SIZE]);
set.put_object(
bucket,
object,
&mut old_reader,
&ObjectOptions {
write_completion: WriteCompletion::TailDrained,
..Default::default()
},
)
.await
.expect("old generation should be completely committed");
wait_for_tmp_workspace_to_drain(&dirs, "old PUT must leave no unrelated staging").await;
let old = disks[0]
.read_version("", bucket, object, "", &ReadOptions::default())
.await
.expect("old metadata must be readable");
let old_data_dir = old.data_dir.expect("non-inline old version needs a data directory");
let tasks = rename_fanout_barrier::observe_tasks(object);
let barrier = rename_fanout_barrier::arm(object, 0, rename_fanout_barrier::PHASE_RENAME);
let _rename_fault = rename_fault_injection::fail_rename_on(object, &[2, 3]);
let _undo_fault = rollback_fault_injection::arm(object, 0, fault);
let writer = Arc::clone(&set);
let put = tokio::spawn(async move {
let mut reader = PutObjReader::from_vec(vec![b'1'; TEST_OBJECT_SIZE]);
writer
.put_object(
bucket,
object,
&mut reader,
&ObjectOptions {
write_completion,
..Default::default()
},
)
.await
});
tokio::time::timeout(Duration::from_secs(30), barrier.wait_until_paused())
.await
.expect("overwrite must enter the actual rename fan-out before failure injection");
barrier.release();
let err = tokio::time::timeout(Duration::from_secs(30), put)
.await
.expect("incomplete undo must return without hanging")
.expect("PUT task should join")
.expect_err("two renamed disks cannot satisfy write quorum three");
assert!(
matches!(err, Error::ErasureWriteQuorum | Error::InsufficientWriteQuorum(_, _)),
"original quorum error expected: {err}"
);
assert_eq!(tasks.running(), 0, "every rename and undo task must be reaped before return");
let leftovers = non_trash_tmp_entries(&dirs).await;
assert!(!leftovers.is_empty(), "incomplete undo must retain the new staging source for recovery");
let backups = dirs
.iter()
.filter(|dir| {
dir.path()
.join(bucket)
.join(object)
.join(old_data_dir.to_string())
.join(crate::disk::STORAGE_FORMAT_FILE_BACKUP)
.exists()
})
.count();
assert_eq!(backups, 1, "exactly the failed undo disk must retain its old-version backup");
// The remaining three disks still serve the old generation;
// the failed minority must never become an acknowledged write.
let mut read = set
.get_object_reader(bucket, object, None, HeaderMap::new(), &ObjectOptions::default())
.await
.expect("old generation must remain readable after incomplete rollback");
let mut body = Vec::new();
read.stream
.read_to_end(&mut body)
.await
.expect("old generation should stream");
assert_eq!(body, vec![b'0'; TEST_OBJECT_SIZE]);
}
}
})
.await;
}
#[tokio::test]
#[serial_test::serial(capacity_dirty_scope)]
async fn tail_drained_put_owned_commit_survives_waiter_cancellation() {
let (dirs, disks, set) = hermetic_set_disks(4).await;
let bucket = RUSTFS_META_BUCKET;
let object = "full-tail-cancelled-receipt";
// Internal config writes do not own a bucket lifecycle guard. The object
// guard alone must keep the full-tail coordinator alive after cancellation.
let tasks = rename_fanout_barrier::observe_tasks(object);
let barrier = rename_fanout_barrier::arm(object, 0, rename_fanout_barrier::PHASE_RENAME);
let writer = Arc::clone(&set);
let put = tokio::spawn(async move {
let mut reader = PutObjReader::from_vec(vec![b'1'; TEST_OBJECT_SIZE]);
writer
.put_object(
bucket,
object,
&mut reader,
&ObjectOptions {
write_completion: WriteCompletion::TailDrained,
..Default::default()
},
)
.await
});
tokio::time::timeout(Duration::from_secs(30), barrier.wait_until_paused())
.await
.expect("cancelled receipt must first reach the rename barrier");
wait_for_paused_tail_metadata_quorum(&disks, bucket, object).await;
put.abort();
assert!(put.await.expect_err("ACK waiter should cancel").is_cancelled());
let mut lock_probe = Box::pin(set.acquire_write_lock_diag("cancelled_full_tail_probe", bucket, object));
assert!(
futures::poll!(lock_probe.as_mut()).is_pending(),
"owned coordinator must retain the namespace guard after waiter cancellation"
);
barrier.release();
drop(
tokio::time::timeout(Duration::from_secs(30), lock_probe)
.await
.expect("cancelled coordinator must eventually release its guard")
.expect("post-commit lock probe should succeed"),
);
assert_eq!(tasks.running(), 0, "cancelled coordinator must reap every rename task");
for disk in &disks {
disk.read_version("", bucket, object, "", &ReadOptions::default())
.await
.expect("caller cancellation must not interrupt committed receipt materialization");
}
wait_for_tmp_workspace_to_drain(&dirs, "cancelled full-tail commit should release staging ownership").await;
}
#[tokio::test] #[tokio::test]
#[serial_test::serial(capacity_dirty_scope)] #[serial_test::serial(capacity_dirty_scope)]
async fn no_lock_put_waits_for_rename_tail_under_outer_guard() { async fn no_lock_put_waits_for_rename_tail_under_outer_guard() {
@@ -18588,7 +18184,6 @@ mod put_object_tmp_cleanup_tests {
&mut reader, &mut reader,
&ObjectOptions { &ObjectOptions {
no_lock: true, no_lock: true,
write_completion: WriteCompletion::TailDrained,
..Default::default() ..Default::default()
}, },
) )
@@ -18614,18 +18209,7 @@ mod put_object_tmp_cleanup_tests {
put.await put.await
.expect("no-lock PUT task should join") .expect("no-lock PUT task should join")
.expect("no-lock PUT should commit after the rename tail releases"); .expect("no-lock PUT should commit after the rename tail releases");
let mut lock_probe = Box::pin(set_disks.acquire_write_lock_diag("borrowed_full_tail_probe", bucket, object));
assert!(
futures::poll!(lock_probe.as_mut()).is_pending(),
"full-tail PUT must not release the caller's outer guard"
);
drop(outer_guard); drop(outer_guard);
drop(
tokio::time::timeout(Duration::from_secs(5), lock_probe)
.await
.expect("outer owner releasing its guard should unblock the probe")
.expect("post-outer-guard probe should succeed"),
);
}) })
.await; .await;
} }
@@ -18,7 +18,6 @@ use super::{
}; };
use crate::bucket::lifecycle::lifecycle::{TRANSITION_COMPLETE, TRANSITION_PENDING, TransitionOptions, expected_expiry_time}; use crate::bucket::lifecycle::lifecycle::{TRANSITION_COMPLETE, TRANSITION_PENDING, TransitionOptions, expected_expiry_time};
use crate::ecstore_validation_blackbox::make_local_set_disks; use crate::ecstore_validation_blackbox::make_local_set_disks;
use crate::object_api::WriteCompletion;
use crate::services::tier::test_util::register_mock_tier; use crate::services::tier::test_util::register_mock_tier;
use crate::storage_api_contracts::bucket::BucketOperations; use crate::storage_api_contracts::bucket::BucketOperations;
use crate::storage_api_contracts::object::{ObjectIO as _, ObjectOperations as _}; use crate::storage_api_contracts::object::{ObjectIO as _, ObjectOperations as _};
@@ -73,7 +72,7 @@ async fn transition_and_restore_reclaim_prior_metadata_generations() {
object, object,
&mut reader, &mut reader,
&ObjectOptions { &ObjectOptions {
write_completion: WriteCompletion::TailDrained, no_lock: true,
..Default::default() ..Default::default()
}, },
) )
@@ -186,7 +185,7 @@ async fn prepared_snapshot_transition_duplicate_and_late_get_use_committed_remot
object, object,
&mut reader, &mut reader,
&ObjectOptions { &ObjectOptions {
write_completion: WriteCompletion::TailDrained, no_lock: true,
..Default::default() ..Default::default()
}, },
) )
+4 -14
View File
@@ -329,11 +329,11 @@ impl ECStore {
/// reuse its result, which is sound because bucket deletion/recreation /// reuse its result, which is sound because bucket deletion/recreation
/// requires the lifecycle WRITE lock and therefore cannot have run while /// requires the lifecycle WRITE lock and therefore cannot have run while
/// any read guard was continuously held. /// any read guard was continuously held.
pub async fn acquire_bucket_incarnation_fence( pub(crate) async fn acquire_bucket_incarnation_fence(
&self, &self,
bucket: &str, bucket: &str,
expected: uuid::Uuid, expected: uuid::Uuid,
) -> Result<super::BucketIncarnationFenceGuard> { ) -> Result<super::bucket_fence::BucketIncarnationFenceGuard> {
let inner = self.acquire_bucket_lifecycle_read_lock(bucket).await?; let inner = self.acquire_bucket_lifecycle_read_lock(bucket).await?;
let pieces = super::bucket_fence::FencePieces { let pieces = super::bucket_fence::FencePieces {
registry: self.bucket_fence_registry.clone(), registry: self.bucket_fence_registry.clone(),
@@ -1059,7 +1059,6 @@ mod tests {
use crate::storage_api_contracts::{ use crate::storage_api_contracts::{
bucket::{BucketOperations as _, BucketOptions, DeleteBucketOptions, MakeBucketOptions, SRBucketDeleteOp}, bucket::{BucketOperations as _, BucketOptions, DeleteBucketOptions, MakeBucketOptions, SRBucketDeleteOp},
list::ListOperations as _, list::ListOperations as _,
namespace::NamespaceLocking as _,
object::{ObjectIO as _, ObjectOperations as _}, object::{ObjectIO as _, ObjectOperations as _},
}; };
use crate::store::{ECStore, init_local_disks_with_instance_ctx}; use crate::store::{ECStore, init_local_disks_with_instance_ctx};
@@ -1487,19 +1486,10 @@ mod tests {
.put_object(bucket, object, &mut reader, &ObjectOptions::default()) .put_object(bucket, object, &mut reader, &ObjectOptions::default())
.await .await
.expect("object should be written"); .expect("object should be written");
let lock = ecstore.pools[0].disk_set[0]
.new_ns_lock(bucket, object)
.await
.expect("fixture namespace lock should be created");
drop(
lock.get_write_lock(Duration::from_secs(30))
.await
.expect("fixture rename tail should finish before checking its generation"),
);
assert_eq!( assert_eq!(
ecstore.scanner_namespace_mutation_generation(), ecstore.scanner_namespace_mutation_generation(),
generation_before_put.saturating_add(3), generation_before_put.saturating_add(1),
"successful object creation must observe the logical mutation and both fanout boundaries" "successful object creation should advance scanner namespace activity"
); );
ecstore ecstore
.get_object_info(bucket, object, &ObjectOptions::default()) .get_object_info(bucket, object, &ObjectOptions::default())
+1 -39
View File
@@ -150,7 +150,7 @@ impl BucketFenceRegistry {
/// A held bucket lifecycle read lock plus its registration in the fence /// A held bucket lifecycle read lock plus its registration in the fence
/// registry. Dropping the guard deregisters it; the memo is cleared when the /// registry. Dropping the guard deregisters it; the memo is cleared when the
/// last guard for the bucket drops (or a lost lock is observed). /// last guard for the bucket drops (or a lost lock is observed).
pub struct BucketIncarnationFenceGuard { pub(crate) struct BucketIncarnationFenceGuard {
inner: Option<NamespaceLockGuard>, inner: Option<NamespaceLockGuard>,
registry: Arc<BucketFenceRegistry>, registry: Arc<BucketFenceRegistry>,
bucket: String, bucket: String,
@@ -158,14 +158,6 @@ pub struct BucketIncarnationFenceGuard {
} }
impl BucketIncarnationFenceGuard { impl BucketIncarnationFenceGuard {
/// Propagate lifecycle lock loss into the storage commit checks.
/// The caller still owns this guard until the complete write tail drains.
pub fn attach_to_object_options(&self, opts: &mut crate::object_api::ObjectOptions) {
if let Some(guard) = self.namespace_lock_guard() {
opts.add_bucket_lifecycle_lock_guard(guard);
}
}
pub(crate) fn is_lock_lost(&self) -> bool { pub(crate) fn is_lock_lost(&self) -> bool {
self.inner.as_ref().is_some_and(NamespaceLockGuard::is_lock_lost) self.inner.as_ref().is_some_and(NamespaceLockGuard::is_lock_lost)
} }
@@ -354,36 +346,6 @@ mod tests {
first_pieces.abandon("b", first.token); first_pieces.abandon("b", first.token);
} }
#[tokio::test]
async fn checkpoint_options_inherit_bucket_fence_lock_loss() {
let lock = NamespaceLock::new("bucket-fence-options".to_string(), Arc::new(LocalClient::new()));
let inner = lock
.acquire_guard(&lock_request("options"))
.await
.expect("acquire")
.expect("quorum");
let pieces = FencePieces {
registry: Arc::default(),
inner,
};
let registration = pieces.enter("b");
let fence = pieces.into_guard("b", registration.token);
let mut opts = crate::object_api::ObjectOptions::default();
fence.attach_to_object_options(&mut opts);
let inherited = opts
.bucket_lifecycle_lock_fence
.as_ref()
.expect("checkpoint inherits lifecycle guard");
assert!(!inherited.is_lock_lost());
tokio::time::timeout(
Duration::from_secs(2),
fence.namespace_lock_guard().expect("held guard").lock_lost_notified(),
)
.await
.expect("distributed guard expires");
assert!(inherited.is_lock_lost(), "the actual pre-rename options must observe lifecycle lock loss");
}
#[test] #[test]
fn buckets_are_isolated() { fn buckets_are_isolated() {
let reg = BucketFenceRegistry::default(); let reg = BucketFenceRegistry::default();
+47 -736
View File
@@ -353,11 +353,6 @@ async fn resume_rebalance_after_init(store: Arc<ECStore>, rx: CancellationToken)
} }
impl ECStore { impl ECStore {
/// Shutdown token owned by this store instance.
pub fn background_cancel_token(&self) -> Option<CancellationToken> {
self.ctx.background_cancel_token()
}
/// Validate topology and process storage-class overrides before any disk is opened. /// Validate topology and process storage-class overrides before any disk is opened.
pub fn validate_startup_storage_class(endpoint_pools: &EndpointServerPools) -> Result<()> { pub fn validate_startup_storage_class(endpoint_pools: &EndpointServerPools) -> Result<()> {
let drive_counts = startup_pool_drive_counts(endpoint_pools); let drive_counts = startup_pool_drive_counts(endpoint_pools);
@@ -792,12 +787,6 @@ impl ECStore {
pub fn single_pool(&self) -> bool { pub fn single_pool(&self) -> bool {
self.pools.len() == 1 self.pools.len() == 1
} }
/// The set-local create-only check is atomic only when every object
/// mutation uses that same, enabled namespace lock domain.
pub fn supports_atomic_create_only_write_back(&self) -> bool {
!self.ctx.lock_manager().is_disabled() && self.pools.len() == 1 && self.pools[0].disk_set.len() == 1
}
} }
#[cfg(test)] #[cfg(test)]
@@ -2138,7 +2127,7 @@ mod tests {
.iter() .iter()
.map(|&drives_per_set| (1, drives_per_set)) .map(|&drives_per_set| (1, drives_per_set))
.collect::<Vec<_>>(); .collect::<Vec<_>>();
build_isolated_test_store_with_layout(temp_dir, cmd_line, &pool_layouts, shutdown, None).await build_isolated_test_store_with_layout(temp_dir, cmd_line, &pool_layouts, shutdown).await
} }
async fn build_isolated_test_store_with_layout( async fn build_isolated_test_store_with_layout(
@@ -2146,7 +2135,6 @@ mod tests {
cmd_line: &str, cmd_line: &str,
pool_layouts: &[(usize, usize)], pool_layouts: &[(usize, usize)],
shutdown: CancellationToken, shutdown: CancellationToken,
instance_ctx: Option<Arc<crate::runtime::instance::InstanceContext>>,
) -> ( ) -> (
Arc<crate::runtime::instance::InstanceContext>, Arc<crate::runtime::instance::InstanceContext>,
Arc<crate::store::ECStore>, Arc<crate::store::ECStore>,
@@ -2179,7 +2167,7 @@ mod tests {
let endpoint_pools = EndpointServerPools(pools); let endpoint_pools = EndpointServerPools(pools);
crate::services::notification_sys::install_cross_pool_fence_fleet_proof_for_test(); crate::services::notification_sys::install_cross_pool_fence_fleet_proof_for_test();
let instance_ctx = instance_ctx.unwrap_or_else(|| Arc::new(crate::runtime::instance::InstanceContext::new())); let instance_ctx = Arc::new(crate::runtime::instance::InstanceContext::new());
crate::store::init_local_disks_with_instance_ctx(&instance_ctx, endpoint_pools.clone()) crate::store::init_local_disks_with_instance_ctx(&instance_ctx, endpoint_pools.clone())
.await .await
.expect("register local disks into the fresh context"); .expect("register local disks into the fresh context");
@@ -2547,348 +2535,6 @@ mod tests {
shutdown.cancel(); shutdown.cancel();
} }
#[cfg(feature = "test-util")]
#[tokio::test(flavor = "multi_thread", worker_threads = 2)]
#[serial_test::serial(storage_class_env)]
async fn early_ack_put_tails_block_scanner_publication_until_all_renames_finish() {
use crate::storage_api_contracts::namespace::NamespaceLocking as _;
let temp_dir = tempfile::tempdir().expect("create scanner PUT tail store dir");
let (ctx, store, shutdown) =
without_storage_class_env(build_isolated_test_store(temp_dir.path(), "scanner-put-tails", &[4])).await;
crate::bucket::metadata_sys::init_bucket_metadata_sys(Arc::clone(&store), Vec::new()).await;
let bucket = format!("scanner-put-tails-{}", Uuid::new_v4());
store
.make_bucket(&bucket, &MakeBucketOptions::default())
.await
.expect("create scanner PUT tail bucket");
let set = &store.pools[0].disk_set[0];
let objects = [("scanner-tail-a", vec![0xA1; 273]), ("scanner-tail-b", vec![0xB2; 379])];
temp_env::async_with_vars([(crate::set_disk::ENV_RUSTFS_PUT_RENAME_EARLY_ACK_ENABLE, Some("true"))], async {
let (active, blocked, movement_generation) = store.scanner_data_movement_activity().await;
assert!(!active && !blocked);
assert!(ctx.scanner_publication_state_allowed(), "the set admission cache should start allowed");
let (old_lease, _) = store
.acquire_scanner_publication_lease(movement_generation, crate::runtime::instance::SCANNER_PUBLICATION_LEASE_TTL)
.await
.expect("publication lease should be admitted before either PUT starts");
let barriers: Vec<_> = objects
.iter()
.map(|(object, _)| {
crate::set_disk::rename_fanout_barrier::arm(object, 0, crate::set_disk::rename_fanout_barrier::PHASE_RENAME)
})
.collect();
let trackers: Vec<_> = objects
.iter()
.map(|(object, _)| crate::set_disk::rename_fanout_barrier::observe_tasks(object))
.collect();
let puts: Vec<_> = objects
.iter()
.map(|(object, body)| {
let put_store = Arc::clone(&store);
let put_bucket = bucket.clone();
let object = *object;
let body = body.clone();
tokio::spawn(async move {
let mut reader = PutObjReader::from_vec(body);
put_store
.put_object(&put_bucket, object, &mut reader, &ObjectOptions::default())
.await
})
})
.collect();
let committed = tokio::time::timeout(Duration::from_secs(30), async {
for barrier in &barriers {
barrier.wait_until_paused().await;
}
let mut committed = Vec::with_capacity(puts.len());
for put in puts {
committed.push(
put.await
.expect("early-ACK PUT task should join while its tail is paused")
.expect("root PUT should return after quorum without waiting for its tail"),
);
}
committed
})
.await
.expect("both root PUTs must quorum-ACK while their tail disks remain paused");
assert!(trackers.iter().all(|tracker| tracker.running() >= 1));
assert!(ctx.namespace_commits_pending());
assert!(
ctx.scanner_publication_state_allowed(),
"pending PUT tails must not disable scanner namespace walks"
);
let (active, blocked, observed_movement_generation) = store.scanner_data_movement_activity().await;
assert!(!active, "ordinary PUT tails are not decommission or rebalance work");
assert!(!blocked, "ordinary PUT tails must not block the movement-only scan baseline");
assert_eq!(observed_movement_generation, movement_generation);
assert!(store.scanner_data_usage_publication_blocked().await);
assert!(store.scanner_data_usage_publication_admission_guard().await.is_some());
assert!(set.scanner_data_usage_publication_admission_guard().await.is_some());
for error in [
store
.acquire_scanner_publication_lease(
movement_generation,
crate::runtime::instance::SCANNER_PUBLICATION_LEASE_TTL,
)
.await
.expect_err("a new remote publication lease must reject pending PUT tails"),
store
.validate_scanner_publication_lease(old_lease, movement_generation)
.await
.expect_err("an existing remote lease must not bypass pending PUT tails"),
store
.acquire_scanner_publication_lease_guard(old_lease)
.await
.expect_err("target-side publication admission must reject pending PUT tails"),
] {
assert!(
error.to_string().contains("blocked"),
"publication must fail because of active tails: {error}"
);
}
store.release_scanner_publication_lease(old_lease).await;
for (index, barrier) in barriers.iter().enumerate() {
let commit_generation = ctx.namespace_commit_generation();
let namespace_generation = store.scanner_namespace_mutation_generation();
barrier.release();
tokio::time::timeout(Duration::from_secs(30), async {
while trackers[index].running() != 0 || ctx.namespace_commit_generation() <= commit_generation {
tokio::task::yield_now().await;
}
if index + 1 == barriers.len() {
while ctx.namespace_commits_pending() {
tokio::task::yield_now().await;
}
}
})
.await
.expect("released tail must drain and publish its terminal namespace generation");
assert!(store.scanner_namespace_mutation_generation() > namespace_generation);
let pending = index + 1 < barriers.len();
assert_eq!(ctx.namespace_commits_pending(), pending);
assert_eq!(store.scanner_data_usage_publication_blocked().await, pending);
assert!(!store.scanner_data_movement_activity().await.1);
assert!(store.scanner_data_usage_publication_admission_guard().await.is_some());
assert!(set.scanner_data_usage_publication_admission_guard().await.is_some());
}
let (lease, generation) = store
.acquire_scanner_publication_lease(movement_generation, crate::runtime::instance::SCANNER_PUBLICATION_LEASE_TTL)
.await
.expect("remote publication lease should resume after both tails drain");
store
.validate_scanner_publication_lease(lease, generation)
.await
.expect("a resumed remote publication lease should validate");
drop(
store
.acquire_scanner_publication_lease_guard(lease)
.await
.expect("target-side publication admission should resume after both tails drain"),
);
assert!(store.release_scanner_publication_lease(lease).await);
let disks = set.disk_inventory().await;
assert_eq!(disks.len(), 4);
for ((object, body), committed) in objects.iter().zip(&committed) {
let logical_size = i64::try_from(body.len()).expect("fixture payload size should fit i64");
let etag = committed.etag.as_ref().expect("root PUT should return a committed ETag");
for (disk_index, disk) in disks.iter().enumerate() {
let file_info = disk
.as_ref()
.expect("every fixture disk should remain online")
.read_version(
"",
&bucket,
object,
"",
&crate::disk::ReadOptions {
read_data: true,
..Default::default()
},
)
.await
.unwrap_or_else(|err| panic!("disk {disk_index} should publish {object} after its tail finishes: {err}"));
assert_eq!(file_info.size, logical_size);
assert_eq!(file_info.metadata.get(http::header::ETAG.as_str()), Some(etag));
assert!(
file_info.inline_data(),
"small fixture payloads should have an inline shard on every disk"
);
let inline_data = file_info.data.as_ref().expect("every disk should retain its inline shard");
let erasure = crate::erasure::coding::Erasure::try_new_with_options(
file_info.erasure.data_blocks,
file_info.erasure.parity_blocks,
file_info.erasure.block_size,
file_info.uses_legacy_checksum,
)
.expect("persisted erasure geometry should be valid");
let shard_size =
usize::try_from(erasure.shard_file_size(logical_size)).expect("fixture shard size should fit usize");
crate::erasure::coding::bitrot_verify(
Cursor::new(inline_data.clone()),
inline_data.len(),
shard_size,
rustfs_utils::HashAlgorithm::HighwayHash256S,
erasure.shard_size(),
)
.await
.unwrap_or_else(|err| panic!("disk {disk_index} should retain a complete valid shard for {object}: {err}"));
}
let mut reader = store
.get_object_reader(&bucket, object, None, HeaderMap::new(), &ObjectOptions::default())
.await
.expect("fully drained PUT should be readable");
let mut actual = Vec::new();
reader.stream.read_to_end(&mut actual).await.expect("PUT body should drain");
assert_eq!(&actual, body);
}
let generation_before_internal_put = ctx.namespace_commit_generation();
let internal_object = "scanner-tail-regression/internal-metadata";
let internal_body = b"scanner metadata must not invalidate its own publication";
let mut internal_reader = PutObjReader::from_vec(internal_body.to_vec());
store
.put_object(RUSTFS_META_BUCKET, internal_object, &mut internal_reader, &ObjectOptions::default())
.await
.expect("internal metadata PUT should commit without scanner self-invalidation");
let internal_lock = set
.new_ns_lock(RUSTFS_META_BUCKET, internal_object)
.await
.expect("internal metadata tail lock should be available");
drop(
internal_lock
.get_write_lock(Duration::from_secs(30))
.await
.expect("internal metadata tail should drain"),
);
assert_eq!(ctx.namespace_commit_generation(), generation_before_internal_put);
assert!(!ctx.namespace_commits_pending());
assert!(store.scanner_data_usage_publication_admission_guard().await.is_some());
assert!(set.scanner_data_usage_publication_admission_guard().await.is_some());
let mut internal_reader = store
.get_object_reader(RUSTFS_META_BUCKET, internal_object, None, HeaderMap::new(), &ObjectOptions::default())
.await
.expect("internal metadata should remain readable");
let mut actual = Vec::new();
internal_reader
.stream
.read_to_end(&mut actual)
.await
.expect("internal metadata body should drain");
assert_eq!(actual, internal_body);
})
.await;
shutdown.cancel();
}
#[cfg(feature = "test-util")]
#[tokio::test(flavor = "multi_thread", worker_threads = 2)]
#[serial_test::serial(storage_class_env)]
async fn cancelled_early_ack_put_keeps_scanner_publication_blocked_until_tail_finishes() {
let temp_dir = tempfile::tempdir().expect("create cancelled scanner PUT tail store dir");
let (ctx, store, shutdown) =
without_storage_class_env(build_isolated_test_store(temp_dir.path(), "scanner-cancelled-put-tail", &[4])).await;
crate::bucket::metadata_sys::init_bucket_metadata_sys(Arc::clone(&store), Vec::new()).await;
let bucket = format!("scanner-cancelled-put-tail-{}", Uuid::new_v4());
let object = "scanner-cancelled-tail";
let body = vec![0xC3; 273];
store
.make_bucket(&bucket, &MakeBucketOptions::default())
.await
.expect("create cancelled scanner PUT tail bucket");
temp_env::async_with_vars([(crate::set_disk::ENV_RUSTFS_PUT_RENAME_EARLY_ACK_ENABLE, Some("true"))], async {
let tracker = crate::set_disk::rename_fanout_barrier::observe_tasks(object);
let tail =
crate::set_disk::rename_fanout_barrier::arm(object, 0, crate::set_disk::rename_fanout_barrier::PHASE_RENAME);
let quorum = crate::set_disk::PutObjectCommitBarrier::install(
&bucket,
object,
crate::set_disk::PutObjectCommitPause::AfterRenameQuorum,
);
let handoff = crate::set_disk::PutObjectCommitBarrier::install(
&bucket,
object,
crate::set_disk::PutObjectCommitPause::AfterRenameHandoff,
);
let put_store = Arc::clone(&store);
let put_bucket = bucket.clone();
let put_body = body.clone();
let put = tokio::spawn(async move {
let mut reader = PutObjReader::from_vec(put_body);
put_store
.put_object(&put_bucket, object, &mut reader, &ObjectOptions::default())
.await
});
tokio::time::timeout(Duration::from_secs(30), tail.wait_until_paused())
.await
.expect("cancelled PUT should pause one disk before rename");
quorum.wait_until_paused().await;
put.abort();
assert!(
put.await
.expect_err("caller should be cancelled after rename quorum")
.is_cancelled()
);
quorum.release();
handoff.wait_until_paused().await;
assert!(tracker.running() >= 1);
assert!(ctx.namespace_commits_pending());
assert!(!store.scanner_data_movement_activity().await.1);
assert!(store.scanner_data_usage_publication_blocked().await);
assert!(store.scanner_data_usage_publication_admission_guard().await.is_some());
assert!(
store.pools[0].disk_set[0]
.scanner_data_usage_publication_admission_guard()
.await
.is_some()
);
let generation = store.scanner_namespace_mutation_generation();
handoff.release();
tail.release();
tokio::time::timeout(Duration::from_secs(30), async {
while tracker.running() != 0 || ctx.namespace_commits_pending() {
tokio::task::yield_now().await;
}
})
.await
.expect("cancelled request's detached fanout must release scanner admission after finishing");
assert!(store.scanner_namespace_mutation_generation() > generation);
assert!(!store.scanner_data_usage_publication_blocked().await);
assert!(store.scanner_data_usage_publication_admission_guard().await.is_some());
for (disk_index, disk) in store.pools[0].disk_set[0].disk_inventory().await.iter().enumerate() {
let file_info = disk
.as_ref()
.expect("cancelled PUT fixture disk should remain online")
.read_version("", &bucket, object, "", &crate::disk::ReadOptions::default())
.await
.unwrap_or_else(|err| panic!("cancelled PUT must still publish on disk {disk_index}: {err}"));
assert_eq!(file_info.size, i64::try_from(body.len()).expect("fixture body size should fit i64"));
}
let mut reader = store
.get_object_reader(&bucket, object, None, HeaderMap::new(), &ObjectOptions::default())
.await
.expect("a cancelled caller must not discard its quorum-committed object");
let mut actual = Vec::new();
reader
.stream
.read_to_end(&mut actual)
.await
.expect("cancelled PUT body should drain");
assert_eq!(actual, body);
})
.await;
shutdown.cancel();
}
#[cfg(feature = "test-util")] #[cfg(feature = "test-util")]
#[test] #[test]
#[serial_test::serial(storage_class_env)] #[serial_test::serial(storage_class_env)]
@@ -3333,35 +2979,6 @@ mod tests {
#[cfg(feature = "test-util")] #[cfg(feature = "test-util")]
const DECOMMISSION_TEST_FAULT_STAGE_TIERED: &str = "decommission_tiered_object"; const DECOMMISSION_TEST_FAULT_STAGE_TIERED: &str = "decommission_tiered_object";
fn decommission_retry_fault_hook(
bucket: &str,
object: &str,
faults: Arc<AtomicUsize>,
) -> crate::core::pools::DecommissionTestFaultDecision {
let target_bucket = bucket.to_string();
let target_object = object.to_string();
Arc::new(move |stage, bucket, object, attempt, succeeded| {
if !succeeded
|| attempt >= crate::core::pools::DECOMMISSION_VERSION_COPY_ATTEMPTS
|| stage != DECOMMISSION_TEST_FAULT_STAGE_MIGRATE_OBJECT
|| bucket != target_bucket
|| object != target_object
{
return false;
}
// Entry retries reset the local attempt; real copy errors can skip
// successful attempts. Only injected faults spend this global budget.
// A real failure may consume an attempt, so preserve the final chance.
faults
.fetch_update(Ordering::SeqCst, Ordering::SeqCst, |faults| {
(faults < crate::core::pools::DECOMMISSION_VERSION_COPY_ATTEMPTS.saturating_sub(1))
.then_some(faults.saturating_add(1))
})
.is_ok()
})
}
async fn seed_decommission_source( async fn seed_decommission_source(
store: &Arc<crate::store::ECStore>, store: &Arc<crate::store::ECStore>,
bucket: &str, bucket: &str,
@@ -5374,7 +4991,6 @@ mod tests {
"decommission-delete-fence", "decommission-delete-fence",
&[(2, 4), (1, 4)], &[(2, 4), (1, 4)],
CancellationToken::new(), CancellationToken::new(),
None,
)) ))
.await; .await;
crate::bucket::metadata_sys::init_bucket_metadata_sys(store.clone(), Vec::new()).await; crate::bucket::metadata_sys::init_bucket_metadata_sys(store.clone(), Vec::new()).await;
@@ -5504,43 +5120,6 @@ mod tests {
shutdown.cancel(); shutdown.cancel();
} }
#[test]
fn decommission_retry_fault_budget_counts_successes_across_attempt_changes() {
let cases: &[&[(usize, bool, bool)]] = &[
&[(1, true, true), (2, true, true), (3, true, false)],
&[(1, true, true), (1, true, true), (2, true, false)],
&[(1, true, true), (3, true, false), (3, true, false)],
&[(1, true, true), (2, false, false), (1, true, true), (2, true, false)],
&[(1, true, true), (2, false, false), (3, true, false)],
&[(3, true, false), (4, true, false)],
];
for case in cases {
let faults = Arc::new(AtomicUsize::new(0));
let hook = decommission_retry_fault_hook("bucket", "object", Arc::clone(&faults));
for (stage, bucket, object, succeeded) in [
("other-stage", "bucket", "object", true),
(DECOMMISSION_TEST_FAULT_STAGE_MIGRATE_OBJECT, "other-bucket", "object", true),
(DECOMMISSION_TEST_FAULT_STAGE_MIGRATE_OBJECT, "bucket", "other-object", true),
(DECOMMISSION_TEST_FAULT_STAGE_MIGRATE_OBJECT, "bucket", "object", false),
] {
assert!(!hook(stage, bucket, object, 1, succeeded));
}
assert_eq!(faults.load(Ordering::SeqCst), 0, "unrelated or failed copies must not consume faults");
let mut expected_faults = 0;
for &(attempt, succeeded, expected) in *case {
assert_eq!(
hook(DECOMMISSION_TEST_FAULT_STAGE_MIGRATE_OBJECT, "bucket", "object", attempt, succeeded),
expected,
"fault plan {case:?} at attempt {attempt}"
);
expected_faults += usize::from(expected);
assert_eq!(faults.load(Ordering::SeqCst), expected_faults);
}
}
}
#[test] #[test]
#[serial_test::serial(storage_class_env)] #[serial_test::serial(storage_class_env)]
fn decommission_entry_retries_source_changed_without_canceling_other_bucket() { fn decommission_entry_retries_source_changed_without_canceling_other_bucket() {
@@ -5635,8 +5214,31 @@ mod tests {
)); ));
let ordinary_faults = Arc::new(AtomicUsize::new(0)); let ordinary_faults = Arc::new(AtomicUsize::new(0));
let fault_hook = decommission_retry_fault_hook(&other_bucket, other_object, Arc::clone(&ordinary_faults)); let ordinary_faults_for_hook = Arc::clone(&ordinary_faults);
let _fault_guard = crate::core::pools::DecommissionTestFaultGuard::install(fault_hook); let fault_bucket = other_bucket.clone();
let _fault_guard = crate::core::pools::DecommissionTestFaultGuard::install(Arc::new(
move |stage, bucket, object, attempt, succeeded| {
let candidate = succeeded
&& stage == DECOMMISSION_TEST_FAULT_STAGE_MIGRATE_OBJECT
&& bucket == fault_bucket.as_str()
&& object == other_object;
if !candidate {
return false;
}
// Keep the fault budget global across any
// entry-level re-list; its inner attempt counter
// restarts after SourceChanged.
ordinary_faults_for_hook
.fetch_update(Ordering::SeqCst, Ordering::SeqCst, |faults| {
let next_fault = faults.saturating_add(1);
(faults < crate::core::pools::DECOMMISSION_VERSION_COPY_ATTEMPTS.saturating_sub(1)
&& attempt == next_fault)
.then_some(next_fault)
})
.is_ok()
},
));
let rx = CancellationToken::new(); let rx = CancellationToken::new();
let source_changed_exhaustions = Arc::new(AtomicUsize::new(0)); let source_changed_exhaustions = Arc::new(AtomicUsize::new(0));
@@ -5673,15 +5275,6 @@ mod tests {
changed_result.expect("SourceChanged entry retry must converge"); changed_result.expect("SourceChanged entry retry must converge");
other_result.expect("other bucket entry must continue through ordinary copy retries"); other_result.expect("other bucket entry must continue through ordinary copy retries");
assert_eq!(
store.pool_meta.read().await.pools[0]
.decommission
.as_ref()
.expect("decommission progress should remain available")
.items_decommission_failed,
0,
"entry completion must not hide an exhausted copy failure"
);
assert!(!rx.is_cancelled(), "entry-level SourceChanged must not cancel the shared worker token"); assert!(!rx.is_cancelled(), "entry-level SourceChanged must not cancel the shared worker token");
assert_eq!(mutation_calls.load(Ordering::SeqCst), 2, "entry must be re-listed after SourceChanged"); assert_eq!(mutation_calls.load(Ordering::SeqCst), 2, "entry must be re-listed after SourceChanged");
assert_eq!(ordinary_faults.load(Ordering::SeqCst), 2, "ordinary copy must consume the retry budget"); assert_eq!(ordinary_faults.load(Ordering::SeqCst), 2, "ordinary copy must consume the retry budget");
@@ -6310,7 +5903,6 @@ mod tests {
"reverse-decommission-fixed-target", "reverse-decommission-fixed-target",
&[(1, 4), (1, 4)], &[(1, 4), (1, 4)],
CancellationToken::new(), CancellationToken::new(),
None,
)) ))
.await; .await;
crate::bucket::metadata_sys::init_bucket_metadata_sys(store.clone(), Vec::new()).await; crate::bucket::metadata_sys::init_bucket_metadata_sys(store.clone(), Vec::new()).await;
@@ -6732,7 +6324,6 @@ mod tests {
"multi-set-decommission-source-cleanup", "multi-set-decommission-source-cleanup",
&[(2, 4)], &[(2, 4)],
CancellationToken::new(), CancellationToken::new(),
None,
)) ))
.await; .await;
crate::bucket::metadata_sys::init_bucket_metadata_sys(store.clone(), Vec::new()).await; crate::bucket::metadata_sys::init_bucket_metadata_sys(store.clone(), Vec::new()).await;
@@ -8454,15 +8045,10 @@ mod tests {
); );
assert!(com::read_config(store.pools[0].clone(), &second_page_path).await.is_ok()); assert!(com::read_config(store.pools[0].clone(), &second_page_path).await.is_ok());
let full_tail = ObjectOptions { com::save_config(store.pools[target_pool_idx].clone(), &second_page_path, receipt_bytes.clone())
max_parity: true,
write_completion: crate::object_api::WriteCompletion::TailDrained,
..Default::default()
};
com::save_config_with_opts(store.pools[target_pool_idx].clone(), &second_page_path, receipt_bytes.clone(), &full_tail)
.await .await
.expect("second page receipt should restore"); .expect("second page receipt should restore");
com::save_config_with_opts(store.pools[target_pool_idx].clone(), &second_page_path, b"{corrupt".to_vec(), &full_tail) com::save_config(store.pools[target_pool_idx].clone(), &second_page_path, b"{corrupt".to_vec())
.await .await
.expect("second page receipt should corrupt deterministically"); .expect("second page receipt should corrupt deterministically");
let corrupt = store let corrupt = store
@@ -9248,17 +8834,18 @@ mod tests {
const MANIFEST_COUNT: usize = 10; const MANIFEST_COUNT: usize = 10;
let temp_dir = tempfile::tempdir().expect("create fast manifest pass recovery store dir"); let temp_dir = tempfile::tempdir().expect("create fast manifest pass recovery store dir");
let mut instance_ctx = crate::runtime::instance::InstanceContext::new(); let (ctx, store, _shutdown) =
instance_ctx.suppress_tier_delete_journal_recovery_for_test(); without_storage_class_env(build_isolated_test_store(temp_dir.path(), "tier-delete-fast-manifest-pass", &[4])).await;
let (ctx, store, shutdown) = without_storage_class_env(build_isolated_test_store_with_layout(
temp_dir.path(),
"tier-delete-fast-manifest-pass",
&[(1, 4)],
CancellationToken::new(),
Some(Arc::new(instance_ctx)),
))
.await;
crate::bucket::metadata_sys::init_bucket_metadata_sys(store.clone(), Vec::new()).await; crate::bucket::metadata_sys::init_bucket_metadata_sys(store.clone(), Vec::new()).await;
let bucket = "tier-delete-fast-manifest-pass-bucket";
store
.make_bucket(bucket, &MakeBucketOptions::default())
.await
.expect("fast manifest pass bucket should be created");
let incarnation = store
.bucket_incarnation_id(bucket)
.await
.expect("fast manifest pass bucket incarnation should resolve");
let tier_name = "FAST-MANIFEST-PASS"; let tier_name = "FAST-MANIFEST-PASS";
let backend = register_mock_tier(&ctx.tier_config_mgr(), tier_name).await; let backend = register_mock_tier(&ctx.tier_config_mgr(), tier_name).await;
let backend_identity = TierConfigMgr::acquire_operation_lease(&ctx.tier_config_mgr(), tier_name) let backend_identity = TierConfigMgr::acquire_operation_lease(&ctx.tier_config_mgr(), tier_name)
@@ -9266,19 +8853,9 @@ mod tests {
.expect("fast manifest pass tier lease should resolve") .expect("fast manifest pass tier lease should resolve")
.backend_identity(); .backend_identity();
for index in 0..MANIFEST_COUNT { for index in 0..MANIFEST_COUNT {
// Pagination must not depend on same-bucket lock wait deadlines.
let bucket = format!("tier-delete-fast-manifest-pass-{index}");
store
.make_bucket(&bucket, &MakeBucketOptions::default())
.await
.expect("fast manifest pass bucket should be created");
let incarnation = store
.bucket_incarnation_id(&bucket)
.await
.expect("fast manifest pass bucket incarnation should resolve");
install_aborting_dispatch_fixture( install_aborting_dispatch_fixture(
store.clone(), store.clone(),
&bucket, bucket,
incarnation, incarnation,
&format!("manifest-page-{index:06}/"), &format!("manifest-page-{index:06}/"),
tier_name, tier_name,
@@ -9309,78 +8886,12 @@ mod tests {
"one production pass must cross the default eight-manifest page limit" "one production pass must cross the default eight-manifest page limit"
); );
assert_eq!(stats.manifests.scanned, MANIFEST_COUNT); assert_eq!(stats.manifests.scanned, MANIFEST_COUNT);
assert_eq!(stats.manifests.deleted, MANIFEST_COUNT, "full recovery result: {stats:?}"); assert_eq!(stats.manifests.deleted, MANIFEST_COUNT);
assert_eq!(stats.manifests.failed, 0, "full recovery result: {stats:?}"); assert_eq!(stats.manifests.failed, 0);
assert_eq!(manifest_marker, None); assert_eq!(manifest_marker, None);
assert_eq!(tier_delete_dispatch_manifest_count(store.clone()).await, 0); assert_eq!(tier_delete_dispatch_manifest_count(store.clone()).await, 0);
assert_eq!(tier_delete_journal_count(store).await, 0); assert_eq!(tier_delete_journal_count(store).await, 0);
assert_eq!(backend.remove_count().await, 0, "rollback recovery must not call the remote tier"); assert_eq!(backend.remove_count().await, 0, "rollback recovery must not call the remote tier");
shutdown.cancel();
}
#[cfg(feature = "test-util")]
#[tokio::test]
#[serial_test::serial(storage_class_env)]
async fn tier_delete_manual_pass_retains_manifest_owned_by_startup_recovery() {
let temp_dir = tempfile::tempdir().expect("create automatic recovery ownership store dir");
let (ctx, store, shutdown) =
without_storage_class_env(build_isolated_test_store(temp_dir.path(), "tier-delete-auto-owner", &[4])).await;
crate::bucket::metadata_sys::init_bucket_metadata_sys(store.clone(), Vec::new()).await;
let bucket = "tier-delete-auto-owner-bucket";
store
.make_bucket(bucket, &MakeBucketOptions::default())
.await
.expect("automatic recovery bucket should be created");
let incarnation = store.bucket_incarnation_id(bucket).await.expect("bucket incarnation");
let tier_name = "AUTO-OWNER";
let backend = register_mock_tier(&ctx.tier_config_mgr(), tier_name).await;
let identity = TierConfigMgr::acquire_operation_lease(&ctx.tier_config_mgr(), tier_name)
.await
.expect("automatic recovery tier lease")
.backend_identity();
// The automatic worker must not observe a partially installed fixture.
let lifecycle_guard = store
.acquire_bucket_lifecycle_write_lock(bucket)
.await
.expect("fixture lifecycle lock");
let (manifest_name, entries) =
install_aborting_dispatch_fixture(store.clone(), bucket, incarnation, "auto-owner/", tier_name, identity, 1).await;
let journal_name = tier_delete_journal_object_name(&entries[0]);
let hook = TierDeleteDispatchRollbackTestHook::install_slow_delete(&journal_name, &journal_name);
drop(lifecycle_guard);
ctx.wake_tier_delete_journal_recovery();
tokio::time::timeout(Duration::from_secs(30), hook.wait_until_delete_paused())
.await
.expect("startup recovery should own the manifest before a manual pass");
assert!(tier_delete_dispatch_manifest_recovery_inflight_for_test(&store, &manifest_name));
let stats = recover_tier_delete_dispatch_manifests(store.clone(), 8, None)
.await
.expect("manual recovery scan");
assert_eq!(stats.scanned, 1, "{stats:?}");
assert_eq!(stats.retained, 1, "{stats:?}");
assert_eq!(stats.deleted, 0, "{stats:?}");
assert_eq!(stats.failed, 0, "{stats:?}");
assert_eq!(tier_delete_dispatch_manifest_count(store.clone()).await, 1);
assert_eq!(tier_delete_journal_count(store.clone()).await, 1);
hook.release_delete();
tokio::time::timeout(Duration::from_secs(30), async {
loop {
let manifest_gone = matches!(com::read_config(store.clone(), &manifest_name).await, Err(Error::ConfigNotFound));
if manifest_gone && !tier_delete_dispatch_manifest_recovery_inflight_for_test(&store, &manifest_name) {
break;
}
tokio::time::sleep(Duration::from_millis(10)).await;
}
})
.await
.expect("automatic recovery should converge without a manual retry");
assert_eq!(tier_delete_dispatch_manifest_count(store.clone()).await, 0);
assert_eq!(tier_delete_journal_count(store).await, 0);
assert_eq!(backend.remove_count().await, 0, "rollback must not delete from the remote tier");
shutdown.cancel();
} }
#[cfg(feature = "test-util")] #[cfg(feature = "test-util")]
@@ -10755,17 +10266,8 @@ mod tests {
const JOURNAL_COUNT: usize = 40; const JOURNAL_COUNT: usize = 40;
let temp_dir = tempfile::tempdir().expect("create rollback retry store dir"); let temp_dir = tempfile::tempdir().expect("create rollback retry store dir");
// Manual retries must own progress between fault removal and the next attempt. let (ctx, store, _shutdown) =
let mut instance_ctx = crate::runtime::instance::InstanceContext::new(); without_storage_class_env(build_isolated_test_store(temp_dir.path(), "dispatch-rollback-retry", &[4])).await;
instance_ctx.suppress_tier_delete_journal_recovery_for_test();
let (ctx, store, shutdown) = without_storage_class_env(build_isolated_test_store_with_layout(
temp_dir.path(),
"dispatch-rollback-retry",
&[(1, 4)],
CancellationToken::new(),
Some(Arc::new(instance_ctx)),
))
.await;
crate::bucket::metadata_sys::init_bucket_metadata_sys(store.clone(), Vec::new()).await; crate::bucket::metadata_sys::init_bucket_metadata_sys(store.clone(), Vec::new()).await;
let bucket = "dispatch-rollback-retry-bucket"; let bucket = "dispatch-rollback-retry-bucket";
store store
@@ -10839,7 +10341,6 @@ mod tests {
assert_eq!(tier_delete_dispatch_manifest_count(store.clone()).await, 0); assert_eq!(tier_delete_dispatch_manifest_count(store.clone()).await, 0);
assert_eq!(backend.remove_count().await, 0, "rollback retries must never call the remote tier"); assert_eq!(backend.remove_count().await, 0, "rollback retries must never call the remote tier");
shutdown.cancel();
} }
#[cfg(feature = "test-util")] #[cfg(feature = "test-util")]
@@ -12074,7 +11575,6 @@ mod tests {
pool_index: usize, pool_index: usize,
bucket: &str, bucket: &str,
object: &str, object: &str,
minio_unversioned: bool,
) { ) {
for disk_index in 0..4 { for disk_index in 0..4 {
let metadata_path = let metadata_path =
@@ -12108,11 +11608,6 @@ mod tests {
] { ] {
rustfs_utils::http::metadata_compat::remove_bytes(&mut object_meta.meta_sys, suffix); rustfs_utils::http::metadata_compat::remove_bytes(&mut object_meta.meta_sys, suffix);
} }
if minio_unversioned {
object_meta
.meta_sys
.insert("x-minio-internal-transitioned-versionID".to_string(), Vec::new());
}
*shallow = rustfs_filemeta::FileMetaShallowVersion::try_from(version) *shallow = rustfs_filemeta::FileMetaShallowVersion::try_from(version)
.expect("legacy transitioned version should re-encode"); .expect("legacy transitioned version should re-encode");
} }
@@ -12123,152 +11618,6 @@ mod tests {
} }
} }
#[cfg(feature = "test-util")]
async fn read_store_body(
store: &Arc<crate::store::ECStore>,
bucket: &str,
object: &str,
range: Option<HTTPRangeSpec>,
opts: &ObjectOptions,
) -> Vec<u8> {
let mut reader = store
.get_object_reader(bucket, object, range, HeaderMap::new(), opts)
.await
.expect("object reader should open");
let mut body = Vec::new();
reader.stream.read_to_end(&mut body).await.expect("object body should drain");
body
}
#[cfg(feature = "test-util")]
#[tokio::test]
#[serial_test::serial(storage_class_env)]
async fn legacy_unknown_unversioned_transition_supports_head_get_and_range_without_backfill() {
let temp_dir = tempfile::tempdir().expect("create legacy unknown unversioned store dir");
let (ctx, store, _shutdown) =
without_storage_class_env(build_isolated_test_store(temp_dir.path(), "legacy-unknown-unversioned-read", &[4])).await;
crate::bucket::metadata_sys::init_bucket_metadata_sys(store.clone(), Vec::new()).await;
let tier_name = "LEGACY-UNKNOWN-UNVERSIONED-READ";
let backend = register_mock_tier(&ctx.tier_config_mgr(), tier_name).await;
backend.set_put_remote_version(Some(String::new())).await;
let bucket = "legacy-unknown-unversioned-read-bucket";
let object = "object.bin";
let payload = b"legacy unversioned remote tier object remains readable".repeat(1024);
store
.make_bucket(bucket, &MakeBucketOptions::default())
.await
.expect("legacy source bucket should be created");
let mut reader = PutObjReader::from_vec(payload.clone());
let source = store
.put_object(bucket, object, &mut reader, &ObjectOptions::default())
.await
.expect("legacy source should be written");
store
.transition_object(
bucket,
object,
&ObjectOptions {
transition: TransitionOptions {
status: TRANSITION_PENDING.to_string(),
tier: tier_name.to_string(),
etag: source.etag.clone().expect("legacy source should have an etag"),
..Default::default()
},
mod_time: source.mod_time,
..Default::default()
},
)
.await
.expect("legacy source should transition");
rewrite_transitioned_xlmeta_as_legacy_unknown(temp_dir.path(), 0, bucket, object, true).await;
backend.clear_op_log().await;
let opts = ObjectOptions {
metadata_cache_safe: false,
..Default::default()
};
let head = store
.get_object_info(bucket, object, &opts)
.await
.expect("legacy transitioned HEAD should use local metadata");
assert_eq!(head.transition_version_state, rustfs_filemeta::TransitionVersionState::Unknown);
assert!(head.transitioned_object.version_id.is_empty());
assert_eq!(
head.user_defined
.get("x-minio-internal-transitioned-versionID")
.map(String::as_str),
Some(""),
"the MinIO empty version-key provenance must survive xl.meta decoding"
);
assert!(
!rustfs_utils::http::metadata_compat::contains_key_str(
&head.user_defined,
rustfs_utils::http::metadata_compat::SUFFIX_TRANSITIONED_VERSION_STATE,
),
"the compatibility read must not synthesize version-state metadata"
);
let full_body = read_store_body(&store, bucket, object, None, &opts).await;
assert_eq!(full_body, payload);
let range = HTTPRangeSpec {
is_suffix_length: false,
start: 7,
end: 38,
};
let ranged_body = read_store_body(&store, bucket, object, Some(range), &opts).await;
assert_eq!(ranged_body, &payload[7..=38]);
let after_read = store.pools[0]
.get_disks_by_key(object)
.load_file_info_versions_exact(bucket, object)
.await
.expect("legacy metadata should remain readable after GET")
.expect("legacy object metadata should remain on disk")
.versions
.into_iter()
.find(|version| version.transition_status == rustfs_filemeta::TRANSITION_COMPLETE)
.expect("legacy transitioned source should remain visible after GET");
assert_eq!(after_read.transition_version_state, rustfs_filemeta::TransitionVersionState::Unknown);
assert!(after_read.transition_version.is_none());
assert!(after_read.transition_version_id.is_none());
assert_eq!(
after_read
.metadata
.get("x-minio-internal-transitioned-versionID")
.map(String::as_str),
Some(""),
"the MinIO empty version-key provenance must remain after GET and Range GET"
);
assert!(
!rustfs_utils::http::metadata_compat::contains_key_str(
&after_read.metadata,
rustfs_utils::http::metadata_compat::SUFFIX_TRANSITIONED_VERSION_STATE,
),
"the compatibility read must remain side-effect free"
);
assert_eq!(
backend.op_log().await,
vec![
MockWarmOp::Probe {
object: after_read.transitioned_objname.clone(),
},
MockWarmOp::Get {
object: after_read.transitioned_objname.clone(),
},
MockWarmOp::Probe {
object: after_read.transitioned_objname.clone(),
},
MockWarmOp::Get {
object: after_read.transitioned_objname,
},
],
"legacy reads should probe before each unversioned GET and never mutate local metadata"
);
assert_eq!(backend.remove_count().await, 0);
}
#[cfg(feature = "test-util")] #[cfg(feature = "test-util")]
#[tokio::test] #[tokio::test]
#[serial_test::serial(storage_class_env)] #[serial_test::serial(storage_class_env)]
@@ -12309,7 +11658,7 @@ mod tests {
) )
.await .await
.expect("legacy source should transition"); .expect("legacy source should transition");
rewrite_transitioned_xlmeta_as_legacy_unknown(temp_dir.path(), 0, bucket, object, false).await; rewrite_transitioned_xlmeta_as_legacy_unknown(temp_dir.path(), 0, bucket, object).await;
let legacy = store.pools[0] let legacy = store.pools[0]
.get_disks_by_key(object) .get_disks_by_key(object)
.load_file_info_versions_exact(bucket, object) .load_file_info_versions_exact(bucket, object)
@@ -13450,7 +12799,7 @@ mod tests {
.expect("merge-loser source should transition"); .expect("merge-loser source should transition");
copy_test_xlmeta_between_pools(temp_dir.path(), 0, 1, bucket, object).await; copy_test_xlmeta_between_pools(temp_dir.path(), 0, 1, bucket, object).await;
} }
rewrite_transitioned_xlmeta_as_legacy_unknown(temp_dir.path(), 1, bucket, "legacy/item.bin", false).await; rewrite_transitioned_xlmeta_as_legacy_unknown(temp_dir.path(), 1, bucket, "legacy/item.bin").await;
backend.set_remove_failure(true); backend.set_remove_failure(true);
store.pools[1] store.pools[1]
.delete_object(bucket, "hidden/item.bin", ObjectOptions::default()) .delete_object(bucket, "hidden/item.bin", ObjectOptions::default())
@@ -13667,7 +13016,6 @@ mod tests {
"partial-set-prefix-delete", "partial-set-prefix-delete",
&[(2, 4)], &[(2, 4)],
CancellationToken::new(), CancellationToken::new(),
None,
)) ))
.await; .await;
crate::bucket::metadata_sys::init_bucket_metadata_sys(store.clone(), Vec::new()).await; crate::bucket::metadata_sys::init_bucket_metadata_sys(store.clone(), Vec::new()).await;
@@ -17040,7 +16388,6 @@ mod tests {
"prepared-directory-recovery", "prepared-directory-recovery",
&[(2, 4)], &[(2, 4)],
shutdown, shutdown,
None,
)) ))
.await; .await;
crate::bucket::metadata_sys::init_bucket_metadata_sys(store.clone(), Vec::new()).await; crate::bucket::metadata_sys::init_bucket_metadata_sys(store.clone(), Vec::new()).await;
@@ -17519,10 +16866,6 @@ mod tests {
.find(|version| version.version_id == history.version_id) .find(|version| version.version_id == history.version_id)
.expect("transitioned history should exist"); .expect("transitioned history should exist");
transitioned.transition_version_state = rustfs_filemeta::TransitionVersionState::Unknown; transitioned.transition_version_state = rustfs_filemeta::TransitionVersionState::Unknown;
rustfs_utils::http::metadata_compat::remove_str(
&mut transitioned.metadata,
rustfs_utils::http::metadata_compat::SUFFIX_TRANSITIONED_VERSION_STATE,
);
metadata metadata
.add_version(transitioned) .add_version(transitioned)
.expect("unknown state should replace the transitioned version"); .expect("unknown state should replace the transitioned version");
@@ -17693,38 +17036,6 @@ mod tests {
.expect("test thread should complete"); .expect("test thread should complete");
} }
#[cfg(feature = "test-util")]
#[tokio::test]
#[serial_test::serial(storage_class_env)]
async fn odm_write_back_requires_one_set_and_enabled_namespace_locking() {
for (layout, locking, supported) in [
(&[(1, 4)][..], true, true),
(&[(1, 4), (1, 4)][..], true, false),
(&[(2, 4)][..], true, false),
(&[(1, 4)][..], false, false),
] {
temp_env::async_with_vars([("RUSTFS_LOCK_ENABLED", Some(if locking { "true" } else { "false" }))], async {
let dir = tempfile::tempdir().expect("isolated topology");
let shutdown = CancellationToken::new();
let (_ctx, store, _) = without_storage_class_env(build_isolated_test_store_with_layout(
dir.path(),
"odm-topology",
layout,
shutdown.clone(),
None,
))
.await;
assert_eq!(
store.supports_atomic_create_only_write_back(),
supported,
"layout={layout:?}, locking={locking}"
);
shutdown.cancel();
})
.await;
}
}
#[cfg(feature = "test-util")] #[cfg(feature = "test-util")]
#[tokio::test] #[tokio::test]
#[serial_test::serial(storage_class_env)] #[serial_test::serial(storage_class_env)]
+8 -10
View File
@@ -417,7 +417,6 @@ const MAX_UPLOADS_LIST: usize = 10000;
mod bucket; mod bucket;
mod bucket_fence; mod bucket_fence;
pub(crate) use bucket::await_bucket_namespace_operation; pub(crate) use bucket::await_bucket_namespace_operation;
pub use bucket_fence::BucketIncarnationFenceGuard;
mod heal; mod heal;
mod heal_walk; mod heal_walk;
pub use heal_walk::HealWalkVersion; pub use heal_walk::HealWalkVersion;
@@ -426,7 +425,7 @@ pub(crate) mod init_format;
pub(crate) mod list_objects; pub(crate) mod list_objects;
mod multipart; mod multipart;
mod object; mod object;
#[cfg(feature = "test-util")] #[cfg(any(test, feature = "test-util"))]
pub use object::DeleteAfterObjectLockSnapshotBarrier; pub use object::DeleteAfterObjectLockSnapshotBarrier;
pub(crate) use object::{ pub(crate) use object::{
DecommissionFixedReadAnchor, ObjectLockDiagGuard, RemoteTuplePublicationCommitGuard, RemoteTuplePublicationFence, DecommissionFixedReadAnchor, ObjectLockDiagGuard, RemoteTuplePublicationCommitGuard, RemoteTuplePublicationFence,
@@ -849,7 +848,7 @@ impl ECStore {
} }
pub fn scanner_namespace_mutation_generation(&self) -> u64 { pub fn scanner_namespace_mutation_generation(&self) -> u64 {
list_objects::scanner_namespace_mutation_generation().saturating_add(self.ctx.namespace_commit_generation()) list_objects::scanner_namespace_mutation_generation()
} }
pub async fn scanner_data_movement_active(&self) -> bool { pub async fn scanner_data_movement_active(&self) -> bool {
@@ -858,7 +857,7 @@ impl ECStore {
} }
/// Return the storage-owned movement state and generation as one /// Return the storage-owned movement state and generation as one
/// authenticated activity snapshot. The read lock is acquired before /// authenticated activity snapshot. The read lock is acquired before
/// the state locks (cancelers, pool metadata, then rebalance metadata), /// the state locks (cancelers, pool metadata, then rebalance metadata),
/// matching the transition writer order and preventing a terminal state /// matching the transition writer order and preventing a terminal state
/// from being reported with the preceding generation. /// from being reported with the preceding generation.
@@ -887,12 +886,11 @@ impl ECStore {
/// Returns whether scanner metadata may still be hidden by a local /// Returns whether scanner metadata may still be hidden by a local
/// data-movement state. Terminal failed/canceled decommission entries /// data-movement state. Terminal failed/canceled decommission entries
/// remain suspended until an operator clears or retries them, so they are /// remain suspended until an operator clears or retries them, so they are
/// a publication barrier even after the worker has stopped. Active PUT /// a publication barrier even after the worker has stopped.
/// rename fanouts also defer publication, including post-ACK tails.
pub async fn scanner_data_usage_publication_blocked(&self) -> bool { pub async fn scanner_data_usage_publication_blocked(&self) -> bool {
let operation_gate = self.ctx.data_movement_operation_gate(); let operation_gate = self.ctx.data_movement_operation_gate();
let _operation_guard = operation_gate.read_owned().await; let _operation_guard = operation_gate.read_owned().await;
self.scanner_data_usage_publication_snapshot_blocked().await || self.ctx.namespace_commits_pending() self.scanner_data_usage_publication_snapshot_blocked().await
} }
pub async fn scanner_data_movement_pause_status(&self) -> ScannerDataMovementPauseStatus { pub async fn scanner_data_movement_pause_status(&self) -> ScannerDataMovementPauseStatus {
@@ -1072,7 +1070,7 @@ impl ECStore {
{ {
return Err(Error::other("scanner publication lease generation is stale")); return Err(Error::other("scanner publication lease generation is stale"));
} }
if self.scanner_data_movement_snapshot_locked().await.1 || self.ctx.namespace_commits_pending() { if self.scanner_data_movement_snapshot_locked().await.1 {
return Err(Error::other("scanner publication lease is blocked by data movement")); return Err(Error::other("scanner publication lease is blocked by data movement"));
} }
@@ -1111,7 +1109,7 @@ impl ECStore {
{ {
return Err(Error::other("scanner publication lease generation is stale")); return Err(Error::other("scanner publication lease generation is stale"));
} }
if self.scanner_data_movement_snapshot_locked().await.1 || self.ctx.namespace_commits_pending() { if self.scanner_data_movement_snapshot_locked().await.1 {
return Err(Error::other("scanner publication lease is blocked by data movement")); return Err(Error::other("scanner publication lease is blocked by data movement"));
} }
if !self.ctx.scanner_publication_lease_is_active(token).await { if !self.ctx.scanner_publication_lease_is_active(token).await {
@@ -1131,7 +1129,7 @@ impl ECStore {
if self.ctx.data_movement_generation_exhausted() || self.ctx.data_movement_operation_epoch_exhausted() { if self.ctx.data_movement_generation_exhausted() || self.ctx.data_movement_operation_epoch_exhausted() {
return Err(Error::other("scanner publication lease generation is exhausted")); return Err(Error::other("scanner publication lease generation is exhausted"));
} }
if self.scanner_data_movement_snapshot_locked().await.1 || self.ctx.namespace_commits_pending() { if self.scanner_data_movement_snapshot_locked().await.1 {
return Err(Error::other("scanner publication lease is blocked by data movement")); return Err(Error::other("scanner publication lease is blocked by data movement"));
} }
let Some(lease_generation) = self.ctx.scanner_publication_lease_generation(token).await else { let Some(lease_generation) = self.ctx.scanner_publication_lease_generation(token).await else {
+10 -115
View File
@@ -297,20 +297,6 @@ fn transitioned_version_from_bytes(value: Option<&[u8]>, state: TransitionVersio
} }
} }
fn transition_version_metadata_value(raw: &[u8], decoded: Option<&str>) -> String {
decoded.map(str::to_owned).unwrap_or_else(|| {
if raw.is_empty() {
String::new()
} else {
String::from_utf8_lossy(raw).into_owned()
}
})
}
fn is_transition_version_metadata_key(key: &str) -> bool {
strip_internal_prefix_preserving_case(key).is_some_and(|suffix| suffix.eq_ignore_ascii_case(SUFFIX_TRANSITIONED_VERSION_ID))
}
fn validate_transition_version_state(state: TransitionVersionState, version: Option<&str>) -> Result<()> { fn validate_transition_version_state(state: TransitionVersionState, version: Option<&str>) -> Result<()> {
let valid = match state { let valid = match state {
TransitionVersionState::Unknown | TransitionVersionState::KnownDisabled => version.is_none(), TransitionVersionState::Unknown | TransitionVersionState::KnownDisabled => version.is_none(),
@@ -380,26 +366,14 @@ impl<'a> DerivedInternalMetadata<'a> {
} }
*slot = Some(value.as_slice()); *slot = Some(value.as_slice());
} }
fn merge_consistent<'a>(canonical: Option<&'a [u8]>, legacy: Option<&'a [u8]>) -> Result<Option<&'a [u8]>> {
if let (Some(canonical), Some(legacy)) = (canonical, legacy)
&& canonical != legacy
{
return Err(Error::FileCorrupt);
}
Ok(canonical.or(legacy))
}
Ok(Self { Ok(Self {
checksum: canonical.checksum.or(legacy.checksum), checksum: canonical.checksum.or(legacy.checksum),
part_checksums: canonical.part_checksums.or(legacy.part_checksums), part_checksums: canonical.part_checksums.or(legacy.part_checksums),
transition_status: merge_consistent(canonical.transition_status, legacy.transition_status)?, transition_status: canonical.transition_status.or(legacy.transition_status),
transitioned_object: merge_consistent(canonical.transitioned_object, legacy.transitioned_object)?, transitioned_object: canonical.transitioned_object.or(legacy.transitioned_object),
transitioned_version: merge_consistent(canonical.transitioned_version, legacy.transitioned_version)?, transitioned_version: canonical.transitioned_version.or(legacy.transitioned_version),
transitioned_version_state: merge_consistent( transitioned_version_state: canonical.transitioned_version_state.or(legacy.transitioned_version_state),
canonical.transitioned_version_state, transition_tier: canonical.transition_tier.or(legacy.transition_tier),
legacy.transitioned_version_state,
)?,
transition_tier: merge_consistent(canonical.transition_tier, legacy.transition_tier)?,
}) })
} }
} }
@@ -464,14 +438,8 @@ impl FileInfo {
} }
} }
fn set_transition_version_state( fn set_transition_version_state(meta_sys: &mut HashMap<String, Vec<u8>>, state: TransitionVersionState) {
meta_sys: &mut HashMap<String, Vec<u8>>, if state == TransitionVersionState::Unknown {
state: TransitionVersionState,
source_metadata: &HashMap<String, String>,
) {
if state == TransitionVersionState::Unknown
&& !rustfs_utils::http::metadata_compat::contains_key_str(source_metadata, SUFFIX_TRANSITIONED_VERSION_STATE)
{
remove_bytes(meta_sys, SUFFIX_TRANSITIONED_VERSION_STATE); remove_bytes(meta_sys, SUFFIX_TRANSITIONED_VERSION_STATE);
} else { } else {
insert_bytes(meta_sys, SUFFIX_TRANSITIONED_VERSION_STATE, state.as_str().as_bytes().to_vec()); insert_bytes(meta_sys, SUFFIX_TRANSITIONED_VERSION_STATE, state.as_str().as_bytes().to_vec());
@@ -2675,11 +2643,6 @@ impl MetaObject {
if derived_metadata.transitioned_version_state.is_some() { if derived_metadata.transitioned_version_state.is_some() {
validate_transition_version_state(transition_version_state, transition_version.as_deref())?; validate_transition_version_state(transition_version_state, transition_version.as_deref())?;
} }
for (key, value) in &self.meta_sys {
if is_transition_version_metadata_key(key) {
metadata.insert(key.to_owned(), transition_version_metadata_value(value, transition_version.as_deref()));
}
}
let transition_version_id = transition_version.as_deref().and_then(|value| Uuid::parse_str(value).ok()); let transition_version_id = transition_version.as_deref().and_then(|value| Uuid::parse_str(value).ok());
let transition_tier = derived_metadata let transition_tier = derived_metadata
.transition_tier .transition_tier
@@ -2726,7 +2689,7 @@ impl MetaObject {
} else { } else {
remove_bytes(&mut self.meta_sys, SUFFIX_TRANSITIONED_VERSION_ID); remove_bytes(&mut self.meta_sys, SUFFIX_TRANSITIONED_VERSION_ID);
} }
set_transition_version_state(&mut self.meta_sys, fi.transition_version_state, &fi.metadata); set_transition_version_state(&mut self.meta_sys, fi.transition_version_state);
insert_bytes(&mut self.meta_sys, SUFFIX_TRANSITION_TIER, fi.transition_tier.as_bytes().to_vec()); insert_bytes(&mut self.meta_sys, SUFFIX_TRANSITION_TIER, fi.transition_tier.as_bytes().to_vec());
if let Some(destination_id) = get_str(&fi.metadata, SUFFIX_TRANSITION_TIER_DESTINATION_ID) { if let Some(destination_id) = get_str(&fi.metadata, SUFFIX_TRANSITION_TIER_DESTINATION_ID) {
insert_bytes(&mut self.meta_sys, SUFFIX_TRANSITION_TIER_DESTINATION_ID, destination_id.into_bytes()); insert_bytes(&mut self.meta_sys, SUFFIX_TRANSITION_TIER_DESTINATION_ID, destination_id.into_bytes());
@@ -2867,7 +2830,7 @@ impl From<FileInfo> for MetaObject {
insert_bytes(&mut meta_sys, SUFFIX_TRANSITIONED_VERSION_ID, transition_version); insert_bytes(&mut meta_sys, SUFFIX_TRANSITIONED_VERSION_ID, transition_version);
} }
if !value.transition_status.is_empty() { if !value.transition_status.is_empty() {
set_transition_version_state(&mut meta_sys, value.transition_version_state, &value.metadata); set_transition_version_state(&mut meta_sys, value.transition_version_state);
} }
if !value.transition_tier.is_empty() { if !value.transition_tier.is_empty() {
@@ -3022,12 +2985,6 @@ impl MetaDeleteMarker {
fi.transition_version_state = transition_version_state_from_bytes(derived_metadata.transitioned_version_state)?; fi.transition_version_state = transition_version_state_from_bytes(derived_metadata.transitioned_version_state)?;
fi.transition_version = fi.transition_version =
transitioned_version_from_bytes(derived_metadata.transitioned_version, fi.transition_version_state); transitioned_version_from_bytes(derived_metadata.transitioned_version, fi.transition_version_state);
for (key, value) in &self.meta_sys {
if is_transition_version_metadata_key(key) {
fi.metadata
.insert(key.to_owned(), transition_version_metadata_value(value, fi.transition_version.as_deref()));
}
}
fi.transition_version_id = fi.transition_version.as_deref().and_then(|value| Uuid::parse_str(value).ok()); fi.transition_version_id = fi.transition_version.as_deref().and_then(|value| Uuid::parse_str(value).ok());
if derived_metadata.transitioned_version_state.is_some() { if derived_metadata.transitioned_version_state.is_some() {
validate_transition_version_state(fi.transition_version_state, fi.transition_version.as_deref())?; validate_transition_version_state(fi.transition_version_state, fi.transition_version.as_deref())?;
@@ -3195,7 +3152,7 @@ impl From<FileInfo> for MetaDeleteMarker {
insert_bytes(&mut meta_sys, SUFFIX_TRANSITIONED_VERSION_ID, transition_version); insert_bytes(&mut meta_sys, SUFFIX_TRANSITIONED_VERSION_ID, transition_version);
} }
if !value.transition_status.is_empty() || value.tier_free_version() { if !value.transition_status.is_empty() || value.tier_free_version() {
set_transition_version_state(&mut meta_sys, value.transition_version_state, &value.metadata); set_transition_version_state(&mut meta_sys, value.transition_version_state);
} }
if !value.transition_tier.is_empty() { if !value.transition_tier.is_empty() {
insert_bytes(&mut meta_sys, SUFFIX_TRANSITION_TIER, value.transition_tier.as_bytes().to_vec()); insert_bytes(&mut meta_sys, SUFFIX_TRANSITION_TIER, value.transition_tier.as_bytes().to_vec());
@@ -4617,7 +4574,6 @@ mod tests {
.into_fileinfo("b", "k", false) .into_fileinfo("b", "k", false)
.expect("into_fileinfo"); .expect("into_fileinfo");
assert_eq!(fi.transition_version_id, None); assert_eq!(fi.transition_version_id, None);
assert_eq!(get_str(&fi.metadata, SUFFIX_TRANSITIONED_VERSION_ID), Some(String::new()));
} }
#[test] #[test]
@@ -4629,10 +4585,6 @@ mod tests {
.into_fileinfo("b", "k", false) .into_fileinfo("b", "k", false)
.expect("into_fileinfo"); .expect("into_fileinfo");
assert_eq!(fi.transition_version_id, None); assert_eq!(fi.transition_version_id, None);
assert!(
get_str(&fi.metadata, SUFFIX_TRANSITIONED_VERSION_ID).is_some_and(|value| !value.is_empty()),
"nil UUID bytes must remain distinguishable from an empty MinIO version"
);
} }
#[test] #[test]
@@ -4646,7 +4598,6 @@ mod tests {
assert_eq!(fi.transition_version_id, Some(id)); assert_eq!(fi.transition_version_id, Some(id));
assert_eq!(fi.transition_version, Some(id.to_string())); assert_eq!(fi.transition_version, Some(id.to_string()));
assert_eq!(fi.transition_version_state, TransitionVersionState::Unknown); assert_eq!(fi.transition_version_state, TransitionVersionState::Unknown);
assert_eq!(get_str(&fi.metadata, SUFFIX_TRANSITIONED_VERSION_ID), Some(id.to_string()));
} }
#[test] #[test]
@@ -4686,36 +4637,6 @@ mod tests {
assert_eq!(fi.transition_version_state, TransitionVersionState::Unknown); assert_eq!(fi.transition_version_state, TransitionVersionState::Unknown);
} }
#[test]
fn meta_object_transition_version_state_explicit_unknown_is_not_legacy_missing() {
let mut metadata = HashMap::new();
rustfs_utils::http::metadata_compat::insert_str(
&mut metadata,
SUFFIX_TRANSITIONED_VERSION_STATE,
TransitionVersionState::Unknown.as_str().to_string(),
);
let fi = FileInfo {
transition_status: "complete".to_string(),
transition_version_state: TransitionVersionState::Unknown,
metadata,
..Default::default()
};
let object = MetaObject::from(fi);
assert_eq!(
get_consistent_bytes(&object.meta_sys, SUFFIX_TRANSITIONED_VERSION_STATE),
Some(b"unknown".as_slice())
);
let decoded = object
.into_fileinfo("b", "k", false)
.expect("explicit unknown state should decode");
assert_eq!(decoded.transition_version_state, TransitionVersionState::Unknown);
assert_eq!(
rustfs_utils::http::metadata_compat::get_consistent_str(&decoded.metadata, SUFFIX_TRANSITIONED_VERSION_STATE,),
Some("unknown")
);
}
#[test] #[test]
fn meta_object_transition_version_state_exact_round_trips_dual_keys() { fn meta_object_transition_version_state_exact_round_trips_dual_keys() {
let id = sample_version_id(); let id = sample_version_id();
@@ -4832,10 +4753,6 @@ mod tests {
.expect("invalid transition version bytes must not fail the object read"); .expect("invalid transition version bytes must not fail the object read");
assert_eq!(fi.transition_version_id, None); assert_eq!(fi.transition_version_id, None);
assert_eq!(fi.transition_version, None); assert_eq!(fi.transition_version, None);
assert!(
get_str(&fi.metadata, SUFFIX_TRANSITIONED_VERSION_ID).is_some_and(|value| !value.is_empty()),
"invalid raw bytes must remain distinguishable from an empty MinIO version"
);
} }
#[test] #[test]
@@ -4878,10 +4795,6 @@ mod tests {
.into_fileinfo("b", "k", false) .into_fileinfo("b", "k", false)
.expect("nil tier version should remain an absent remote version"); .expect("nil tier version should remain an absent remote version");
assert_eq!(fi.transition_version_id, None); assert_eq!(fi.transition_version_id, None);
assert!(
get_str(&fi.metadata, SUFFIX_TRANSITIONED_VERSION_ID).is_some_and(|value| !value.is_empty()),
"nil UUID bytes must remain distinguishable from an empty MinIO version"
);
} }
#[test] #[test]
@@ -4899,7 +4812,6 @@ mod tests {
.expect("legacy binary UUID tier version should decode"); .expect("legacy binary UUID tier version should decode");
assert_eq!(fi.transition_version_id, Some(id)); assert_eq!(fi.transition_version_id, Some(id));
assert_eq!(fi.transition_version, Some(id.to_string())); assert_eq!(fi.transition_version, Some(id.to_string()));
assert_eq!(get_str(&fi.metadata, SUFFIX_TRANSITIONED_VERSION_ID), Some(id.to_string()));
} }
#[test] #[test]
@@ -4998,23 +4910,6 @@ mod tests {
assert_eq!(err, Error::FileCorrupt); assert_eq!(err, Error::FileCorrupt);
} }
#[test]
fn meta_object_transition_version_state_mixed_case_alias_conflict_fails_closed() {
let sys = HashMap::from([
(
format!("{RUSTFS_INTERNAL_PREFIX}{SUFFIX_TRANSITIONED_VERSION_STATE}"),
b"unknown".to_vec(),
),
("X-Minio-Internal-transitioned-version-state".to_string(), b"exact".to_vec()),
]);
let err = make_meta_object_with_sys(sys)
.into_fileinfo("b", "k", false)
.expect_err("mixed-case transition state aliases must agree");
assert_eq!(err, Error::FileCorrupt);
}
#[test] #[test]
fn version_header_sorts_before_prefers_object_over_delete_marker_on_equal_mod_time() { fn version_header_sorts_before_prefers_object_over_delete_marker_on_equal_mod_time() {
let object = FileMetaVersionHeader { let object = FileMetaVersionHeader {
+12 -60
View File
@@ -45,11 +45,6 @@ use tracing::{debug, error, info, warn};
use super::{DiskError, Endpoint, HealDiskExt as _, local_disk_map_read}; use super::{DiskError, Endpoint, HealDiskExt as _, local_disk_map_read};
const KEEP_HEAL_TASK_STATUS_DURATION: Duration = Duration::from_secs(10 * 60); const KEEP_HEAL_TASK_STATUS_DURATION: Duration = Duration::from_secs(10 * 60);
// Each cache includes alias tokens in its count and byte budget. Eviction
// removes every token sharing a snapshot; neither cache retains repair state.
const MAX_COMPLETED_HEAL_TOKENS: usize = 1024;
const MAX_COMPLETED_HEAL_BYTES: usize = 64 * 1024 * 1024;
const MAX_COMPLETED_HEAL_RESULT_BYTES: usize = 1024 * 1024;
const DISPLACED_HEAL_REASON: &str = "reason=displaced; retry_hint=submit_again"; const DISPLACED_HEAL_REASON: &str = "reason=displaced; retry_hint=submit_again";
const LOG_COMPONENT_HEAL: &str = "heal"; const LOG_COMPONENT_HEAL: &str = "heal";
const LOG_SUBSYSTEM_DISK_SCANNER: &str = "disk_scanner"; const LOG_SUBSYSTEM_DISK_SCANNER: &str = "disk_scanner";
@@ -185,8 +180,6 @@ fn record_displaced_terminal(
request: &HealRequest, request: &HealRequest,
) -> Arc<CompletedHealStatus> { ) -> Arc<CompletedHealStatus> {
let terminal = Arc::new(CompletedHealStatus { let terminal = Arc::new(CompletedHealStatus {
progress: None,
retained_bytes: std::sync::OnceLock::new(),
heal_type: request.heal_type.clone(), heal_type: request.heal_type.clone(),
status: HealTaskStatus::Failed { status: HealTaskStatus::Failed {
error: format!("heal task displaced by a higher-priority request ({DISPLACED_HEAL_REASON})"), error: format!("heal task displaced by a higher-priority request ({DISPLACED_HEAL_REASON})"),
@@ -200,7 +193,6 @@ fn record_displaced_terminal(
let mut terminals = lock_displaced_terminals(registry); let mut terminals = lock_displaced_terminals(registry);
prune_completed_heal_statuses(&mut terminals); prune_completed_heal_statuses(&mut terminals);
terminals.insert(request.id.clone(), Arc::clone(&terminal)); terminals.insert(request.id.clone(), Arc::clone(&terminal));
prune_completed_heal_statuses(&mut terminals);
terminal terminal
} }
@@ -217,15 +209,9 @@ async fn remove_displaced_task_aliases(
.collect::<Vec<_>>(); .collect::<Vec<_>>();
let mut displaced_terminals = lock_displaced_terminals(terminals); let mut displaced_terminals = lock_displaced_terminals(terminals);
prune_completed_heal_statuses(&mut displaced_terminals); prune_completed_heal_statuses(&mut displaced_terminals);
if displaced_terminals for alias_id in alias_ids {
.get(task_id) displaced_terminals.insert(alias_id, Arc::clone(terminal));
.is_some_and(|current| Arc::ptr_eq(current, terminal))
{
for alias_id in alias_ids {
displaced_terminals.insert(alias_id, Arc::clone(terminal));
}
} }
prune_completed_heal_statuses(&mut displaced_terminals);
aliases.retain(|alias_id, alias| alias_id != task_id && alias.task_id != task_id); aliases.retain(|alias_id, alias| alias_id != task_id && alias.task_id != task_id);
} }
@@ -236,36 +222,6 @@ async fn remove_task_aliases_for_task(registry: &Arc<Mutex<HashMap<String, HealT
.retain(|alias_id, alias| alias_id != task_id && alias.task_id != task_id); .retain(|alias_id, alias| alias_id != task_id && alias.task_id != task_id);
} }
// Callers hold active ownership until publication. Lock order is active ->
// retrying (when needed) -> aliases -> completed; queries release aliases
// before looking up active state. Publishing aliases before removing their
// mapping keeps both an already-resolved token and a new lookup valid.
async fn publish_completed_heal(
completed_heals: &Mutex<HashMap<String, Arc<CompletedHealStatus>>>,
task_aliases: &Mutex<HashMap<String, HealTaskAlias>>,
task_id: &str,
completed: CompletedHealStatus,
terminal: bool,
) {
let completed = Arc::new(completed);
completed.retained_bytes();
let mut aliases = task_aliases.lock().await;
let mut retained = completed_heals.lock().await;
if let Some(previous) = retained.get(task_id).cloned() {
for entry in retained.values_mut().filter(|entry| Arc::ptr_eq(entry, &previous)) {
*entry = Arc::clone(&completed);
}
}
retained.insert(task_id.to_owned(), Arc::clone(&completed));
if terminal {
for (alias_id, _) in aliases.iter().filter(|(_, alias)| alias.task_id == task_id) {
retained.insert(alias_id.clone(), Arc::clone(&completed));
}
aliases.retain(|alias_id, alias| alias_id != task_id && alias.task_id != task_id);
}
prune_completed_heal_statuses(&mut retained);
}
#[derive(Debug, Clone)] #[derive(Debug, Clone)]
pub struct HealTaskReport { pub struct HealTaskReport {
pub status: HealTaskStatus, pub status: HealTaskStatus,
@@ -312,7 +268,7 @@ fn completed_task_report(completed: &CompletedHealStatus, since: Option<u64>) ->
let result_items = match since { let result_items = match since {
None => completed.seqed_items.iter().map(|(_, item)| item.clone()).collect(), None => completed.seqed_items.iter().map(|(_, item)| item.clone()).collect(),
Some(cursor) => { Some(cursor) => {
if cursor.saturating_add(1) < completed.min_seq { if cursor + 1 < completed.min_seq {
lagged = true; lagged = true;
} }
completed completed
@@ -327,7 +283,7 @@ fn completed_task_report(completed: &CompletedHealStatus, since: Option<u64>) ->
status: completed.status.clone(), status: completed.status.clone(),
result_items, result_items,
result_items_truncated: completed.result_items_truncated || lagged, result_items_truncated: completed.result_items_truncated || lagged,
progress: completed.progress.clone(), progress: None,
next_seq: completed.next_seq, next_seq: completed.next_seq,
min_seq: completed.min_seq, min_seq: completed.min_seq,
} }
@@ -1891,14 +1847,14 @@ impl HealManager {
pub async fn get_task_progress(&self, task_id: &str) -> Result<HealProgress> { pub async fn get_task_progress(&self, task_id: &str) -> Result<HealProgress> {
let canonical_task_id = self.canonical_task_id(task_id).await; let canonical_task_id = self.canonical_task_id(task_id).await;
let progress = match self.lookup_task_state(&canonical_task_id, None).await { let active_heals = self.active_heals.lock().await;
TaskStateLookup::Active(task) => Some(task.get_progress().await), if let Some(task) = active_heals.get(&canonical_task_id) {
TaskStateLookup::Completed(completed) => completed.progress.clone(), Ok(task.get_progress().await)
_ => None, } else {
}; Err(Error::TaskNotFound {
progress.ok_or_else(|| Error::TaskNotFound { task_id: task_id.to_string(),
task_id: task_id.to_string(), })
}) }
} }
/// Cancel task /// Cancel task
@@ -1908,8 +1864,6 @@ impl HealManager {
let mut active_heals = self.active_heals.lock().await; let mut active_heals = self.active_heals.lock().await;
if let Some(task) = active_heals.get(&canonical_task_id) { if let Some(task) = active_heals.get(&canonical_task_id) {
task.cancel().await?; task.cancel().await?;
let completed = CompletedHealStatus::snapshot(task, HealTaskStatus::Cancelled).await;
publish_completed_heal(&self.completed_heals, &self.task_aliases, &canonical_task_id, completed, true).await;
active_heals.remove(&canonical_task_id); active_heals.remove(&canonical_task_id);
publish_active_heal_count(&active_heals); publish_active_heal_count(&active_heals);
info!( info!(
@@ -1986,8 +1940,6 @@ impl HealManager {
for task_id in &task_ids { for task_id in &task_ids {
if let Some(task) = active_heals.get(task_id) { if let Some(task) = active_heals.get(task_id) {
task.cancel().await?; task.cancel().await?;
let completed = CompletedHealStatus::snapshot(task, HealTaskStatus::Cancelled).await;
publish_completed_heal(&self.completed_heals, &self.task_aliases, task_id, completed, true).await;
} }
active_heals.remove(task_id); active_heals.remove(task_id);
cancelled += 1; cancelled += 1;
-129
View File
@@ -82,8 +82,6 @@ pub(super) enum QueuePushOutcome {
pub(super) struct CompletedHealStatus { pub(super) struct CompletedHealStatus {
pub(super) heal_type: HealType, pub(super) heal_type: HealType,
pub(super) status: HealTaskStatus, pub(super) status: HealTaskStatus,
pub(super) progress: Option<HealProgress>,
pub(super) retained_bytes: std::sync::OnceLock<usize>,
pub(super) result_items_truncated: bool, pub(super) result_items_truncated: bool,
pub(super) completed_at: SystemTime, pub(super) completed_at: SystemTime,
/// Sequence-stamped retained window, archived with the completion so /// Sequence-stamped retained window, archived with the completion so
@@ -94,133 +92,6 @@ pub(super) struct CompletedHealStatus {
pub(super) min_seq: u64, pub(super) min_seq: u64,
} }
impl CompletedHealStatus {
// Account for owned capacities, including nested drive arrays. Aliases
// conservatively charge the shared allocation again, keeping both token
// count and retained payload bounded without a second ownership index.
pub(super) fn retained_bytes(&self) -> usize {
*self.retained_bytes.get_or_init(|| self.measure_retained_bytes())
}
fn measure_retained_bytes(&self) -> usize {
let mut bytes = size_of::<Self>();
let mut add = |amount: usize| bytes = bytes.saturating_add(amount);
match &self.heal_type {
HealType::Cluster => {}
HealType::Bucket { bucket } => add(bucket.capacity()),
HealType::Object {
bucket,
object,
version_id,
}
| HealType::ECDecode {
bucket,
object,
version_id,
} => {
add(bucket.capacity());
add(object.capacity());
add(version_id.as_ref().map_or(0, String::capacity));
}
HealType::Prefix { bucket, prefix } => {
add(bucket.capacity());
add(prefix.capacity());
}
HealType::Metadata { bucket, object } => {
add(bucket.capacity());
add(object.capacity());
}
HealType::ErasureSet { buckets, set_disk_id } => {
add(buckets.capacity().saturating_mul(size_of::<String>()));
for bucket in buckets {
add(bucket.capacity());
}
add(set_disk_id.capacity());
}
}
if let HealTaskStatus::Failed { error } | HealTaskStatus::Retrying { error, .. } = &self.status {
add(error.capacity());
}
add(self
.progress
.as_ref()
.and_then(|progress| progress.current_object.as_ref())
.map_or(0, String::capacity));
add(self.seqed_items.capacity().saturating_mul(size_of::<(u64, HealResultItem)>()));
for (_, item) in &self.seqed_items {
add(Self::result_item_heap_bytes(item));
}
bytes
}
fn result_item_heap_bytes(item: &HealResultItem) -> usize {
let mut bytes = 0usize;
let mut add = |amount: usize| bytes = bytes.saturating_add(amount);
for value in [
&item.heal_item_type,
&item.bucket,
&item.object,
&item.version_id,
&item.detail,
] {
add(value.capacity());
}
for infos in [&item.before, &item.after] {
add(infos
.drives
.capacity()
.saturating_mul(size_of::<rustfs_madmin::heal_commands::HealDriveInfo>()));
for drive in &infos.drives {
add(drive.uuid.capacity());
add(drive.endpoint.capacity());
add(drive.state.capacity());
}
}
bytes
}
pub(super) fn bound_result_window(&mut self) {
let mut bytes = 0usize;
let retained = self
.seqed_items
.iter()
.rev()
.take_while(|(_, item)| {
bytes = bytes
.saturating_add(size_of::<(u64, HealResultItem)>())
.saturating_add(Self::result_item_heap_bytes(item));
bytes <= MAX_COMPLETED_HEAL_RESULT_BYTES
})
.count();
let truncated = retained < self.seqed_items.len();
if truncated {
self.seqed_items.drain(..self.seqed_items.len() - retained);
self.seqed_items.shrink_to_fit();
self.min_seq = self.seqed_items.first().map_or(self.next_seq, |(seq, _)| *seq);
self.result_items_truncated = true;
self.retained_bytes.take();
}
}
pub(super) async fn snapshot(task: &HealTask, status: HealTaskStatus) -> Self {
let seqed_items = task.get_seqed_result_items().await;
let (next_seq, min_seq) = task.result_seq_cursors();
let mut snapshot = Self {
heal_type: task.heal_type.clone(),
status,
progress: Some(task.get_progress().await),
retained_bytes: std::sync::OnceLock::new(),
result_items_truncated: task.result_items_truncated(),
completed_at: SystemTime::now(),
seqed_items,
next_seq,
min_seq,
};
snapshot.bound_result_window();
snapshot
}
}
#[derive(Debug, Clone)] #[derive(Debug, Clone)]
pub(super) struct HealTaskAlias { pub(super) struct HealTaskAlias {
pub(super) task_id: String, pub(super) task_id: String,

Some files were not shown because too many files have changed in this diff Show More