mirror of
https://github.com/rustfs/rustfs.git
synced 2026-09-04 19:25:40 +00:00
Compare commits
1 Commits
| Author | SHA1 | Date | |
|---|---|---|---|
| f7385b7b4c |
@@ -1,2 +1,2 @@
|
||||
sha256-darwin=ef914ec0b8daa9c2c5e52f501d339914662f42d6f6ed9d33877d56b97adf16f9
|
||||
sha256-linux=a8a816d7bb0e7cb5632b1863b33794bcb9fc7e765f150aa5e1bf16518e28dfb4
|
||||
sha256-darwin=d6aa36cfaae2c4d8590482c7e47138c5965b335b34a75f50d11ffc3366e9021e
|
||||
sha256-linux=e3eb4ab7fc72224abf58c546ac0706d6605d3bd26bac7d8ce338829fd3daecc2
|
||||
|
||||
@@ -1 +1 @@
|
||||
sha256=51da41c54167602f2bd6c45921b39a44562bf3cfcdf468d992bb992c62cad7fd
|
||||
sha256=26003ce03eca11391d1c080491e4f408526717e4b47967b62db08abe1edd189a
|
||||
|
||||
@@ -1 +1 @@
|
||||
sha256=dbebfbab9b9efd4eff31211e69dd32235dc00e207f2ab0dd919a1b2ac9e724c2
|
||||
sha256=294350518743cac8d7c41880a2835216e4b697908d7b0b1bc92b62816d94c59d
|
||||
|
||||
@@ -23,4 +23,4 @@ coverage: core-deps ## Workspace line coverage (cargo-llvm-cov + nextest; slow,
|
||||
@mkdir -p target/llvm-cov
|
||||
cargo llvm-cov report --lcov --output-path target/llvm-cov/lcov.info
|
||||
cargo llvm-cov report --json --output-path target/llvm-cov/coverage.json
|
||||
$(RUSTFS_PYTHON_BIN) scripts/coverage_per_crate.py target/llvm-cov/coverage.json
|
||||
python3 scripts/coverage_per_crate.py target/llvm-cov/coverage.json
|
||||
|
||||
@@ -88,7 +88,7 @@ offline-enrollment-e2e-check: core-deps ## Build and exercise the dedicated offl
|
||||
.PHONY: test-wiring-check
|
||||
test-wiring-check: ## Check tests stay registered and selected by their intended runners
|
||||
@echo "🧪 Checking test wiring..."
|
||||
$(RUSTFS_PYTHON_BIN) ./scripts/check_test_wiring.py
|
||||
python3 ./scripts/check_test_wiring.py
|
||||
|
||||
.PHONY: log-analyzer-rules-check
|
||||
log-analyzer-rules-check: core-deps ## Check log-analyzer rule anchors still exist verbatim in source
|
||||
|
||||
@@ -35,14 +35,13 @@ script-tests: ## Run shell script tests
|
||||
./scripts/test_pinned_paired_abba_bench.sh
|
||||
./scripts/test_manual_transition_runbooks.sh
|
||||
./scripts/test_fuzz_runner.sh
|
||||
./scripts/test_python_bin.sh
|
||||
./scripts/check_embedded_secrets.sh --self-test
|
||||
$(RUSTFS_PYTHON_BIN) ./scripts/check_test_wiring.py --self-test
|
||||
$(RUSTFS_PYTHON_BIN) ./scripts/check_security_coverage.py --self-test
|
||||
$(RUSTFS_PYTHON_BIN) ./scripts/check_scheduled_validation_freshness.py --self-test
|
||||
$(RUSTFS_PYTHON_BIN) ./scripts/s3-tests/test_report_compat.py
|
||||
python3 ./scripts/check_test_wiring.py --self-test
|
||||
python3 ./scripts/check_security_coverage.py --self-test
|
||||
python3 ./scripts/check_scheduled_validation_freshness.py --self-test
|
||||
python3 ./scripts/s3-tests/test_report_compat.py
|
||||
bash -n ./scripts/validate_object_data_cache_cold_stampede.sh
|
||||
$(RUSTFS_PYTHON_BIN) ./scripts/check_object_data_cache_follower_samples.py --self-test
|
||||
python3 ./scripts/check_object_data_cache_follower_samples.py --self-test
|
||||
./scripts/validate_object_data_cache_cold_stampede.sh --self-test
|
||||
|
||||
.PHONY: test
|
||||
|
||||
+15
-65
@@ -46,11 +46,6 @@ e2e-reliability = { max-threads = 1 }
|
||||
e2e-inline-boundaries = { max-threads = 1 }
|
||||
e2e-cluster-nightly = { max-threads = 1 }
|
||||
|
||||
# Deep async storage futures are composed into tests across several crates.
|
||||
# Keep the test stack bounded but above libtest's 2 MiB default.
|
||||
[scripts.setup.ecstore-base-stack]
|
||||
command = ['sh', '-c', 'echo RUST_MIN_STACK=4194304 >> "$NEXTEST_ENV"']
|
||||
|
||||
# These exact regression scenarios build deep async storage futures that exceed
|
||||
# libtest's 2 MiB spawned-thread stack on Linux. Give only their test processes
|
||||
# the same 32 MiB stack already used by the crate's dedicated large-stack tests.
|
||||
@@ -65,13 +60,9 @@ command = ['sh', '-c', 'echo RUST_MIN_STACK=33554432 >> "$NEXTEST_ENV"']
|
||||
|
||||
# --- default profile (local): serialize the flaky groups, never retry --------
|
||||
[[profile.default.scripts]]
|
||||
filter = 'package(rustfs-ecstore) & test(/^(bucket::lifecycle::bucket_lifecycle_ops::tests::manual_transition_worker_result_recovery_marks_unknown_for_corrupt_marker|services::rebalance::entry::tests::real_rebalance_run_fence_loss_blocks_multipart_publication|store::init::tests::(batch_transitioned_delete_uses_free_version_per_item|decommission_entry_(allows_free_version_consumed_before_source_lock|rejects_subquorum_free_version_conflict_and_retains_source|skips_cleanup_only_marker_when_free_version_is_present)|dispatched_tier_delete_recovery_(checks_later_pool_then_commits_after_source_removal|finds_directory_source_on_encoded_set|retains_journal_on_source_metadata_error)|force_tier_remove_blocks_on_physical_free_version_hidden_by_other_pool|legacy_unknown_transition_delete_falls_back_for_single_batch_and_blocks_prefix|multi_pool_(recursive_prefix_rejects_legacy_or_hidden_merge_loser_before_delete|same_remote_tuple_(batch|single)_delete_waits_for_all_sources|same_tuple_recursive_prefix_uses_one_journal_owner|transitioned_delete_persists_one_free_version_per_remote_tuple)|recursive_prefix_partial_(pool|set)_failure_keeps_prepared_cleanup_owners|restored_transitioned_delete_uses_free_version_as_cleanup_owner|stable_transitioned_recursive_prefix_delete_uses_journal_owners|suspended_null_transition_delete_uses_free_version_as_sole_owner|tier_mutation_peer_handler_applies_prepare_commit_and_abort_idempotently|transition_response_loss_persists_unknown_outcome_for_provider_recovery|transition_transaction_recovery_(drops_record_after_confirmed_local_commit|keeps_cleanup_pending_local_commit)|transitioned_delete_(free_version_replays_after_store_restart|local_quorum_failure_rolls_back_without_cleanup_owner|uses_free_version_as_cleanup_owner)|versioned_delete_marker_keeps_transitioned_source_and_remote_object|versioned_explicit_transition_delete_preserves_other_version_then_allows_bucket_delete))$/)'
|
||||
filter = 'package(rustfs-ecstore) & test(/^(bucket::lifecycle::bucket_lifecycle_ops::tests::manual_transition_worker_result_recovery_marks_unknown_for_corrupt_marker|services::rebalance::entry::tests::real_rebalance_run_fence_loss_blocks_multipart_publication|store::init::tests::(decommission_entry_(allows_free_version_consumed_before_source_lock|rejects_subquorum_free_version_conflict_and_retains_source|skips_cleanup_only_marker_when_free_version_is_present)|prepared_tier_delete_recovery_(checks_later_pool_then_commits_after_source_removal|finds_directory_source_on_encoded_set|retains_journal_on_source_metadata_error)|tier_mutation_peer_handler_applies_prepare_commit_and_abort_idempotently|transition_response_loss_persists_unknown_outcome_for_provider_recovery|transition_transaction_recovery_(drops_record_after_confirmed_local_commit|keeps_cleanup_pending_local_commit)))$/)'
|
||||
setup = 'ecstore-large-stack'
|
||||
|
||||
[[profile.default.scripts]]
|
||||
filter = 'package(rustfs-ecstore) | package(rustfs-s3select-api) | package(rustfs-scanner) | (package(rustfs) & test(/^(app::multipart_usecase::tests::concurrent_completions_share_durable_bucket_quota_reservations|app::object::delete::tests::compressed_delete_requests_update_observed_usage_without_releasing_quota_floor|storage::access::tests::(delete_object_access_captures_authorized_bucket_incarnation|copy_operations_reject_recreated_source_bucket_after_authorization|request_slot_keeps_bucket_policy_bound_to_its_store))$/))'
|
||||
setup = 'ecstore-base-stack'
|
||||
|
||||
[[profile.default.scripts]]
|
||||
filter = 'binary(lifecycle_integration_test) | (package(rustfs) & test(/^app::lifecycle_transition_api_test::/))'
|
||||
setup = 'lifecycle-large-stack'
|
||||
@@ -89,29 +80,6 @@ test-group = 'ecstore-serial-flaky'
|
||||
filter = 'package(rustfs-ecstore) & test(/^set_disk::ops::multipart::tests::crash_consistency::/)'
|
||||
test-group = 'ecstore-serial-flaky'
|
||||
|
||||
# Serialize the heal result-report tests. Every test in the module builds a
|
||||
# real-disk (TempDir-backed) hermetic erasure set and drives MiB-scale writes
|
||||
# plus deep-scan heal — the same load-sensitive cross-disk IO shape as the
|
||||
# crash_consistency scenarios above. Under a heavily parallel run a single
|
||||
# disk's IO can fail while write quorum still holds, which flips per-disk
|
||||
# readback and aggregate-outcome assertions nondeterministically (different
|
||||
# tests each round; all pass standalone). Preventive serialization only, no
|
||||
# retries. The matching ci-profile override is after [profile.ci].
|
||||
[[profile.default.overrides]]
|
||||
filter = 'package(rustfs-ecstore) & test(/^set_disk::ops::heal::heal_result_report_tests::/)'
|
||||
test-group = 'ecstore-serial-flaky'
|
||||
|
||||
# Serialize the metadata-cache generation-retirement pair. Both carry
|
||||
# #[serial(metadata_cache_invalidation_probe)] — a no-op across nextest's
|
||||
# process boundary — and assert get_object_metadata_cache generation
|
||||
# semantics on a 4-disk hermetic set, the same load-sensitive shape that
|
||||
# forced the transition matrix tests into this group. Preventive
|
||||
# serialization only, no retries. The matching ci-profile override is after
|
||||
# [profile.ci].
|
||||
[[profile.default.overrides]]
|
||||
filter = 'package(rustfs-ecstore) & test(retires_cached_snapshot)'
|
||||
test-group = 'ecstore-serial-flaky'
|
||||
|
||||
# The production-handler relocation regression builds an isolated 8-disk,
|
||||
# 2-pool store and commits a 72 MiB multipart object. Keep that cross-disk IO
|
||||
# from overlapping the ecstore commit fixtures above.
|
||||
@@ -148,13 +116,6 @@ test-group = 'ecstore-serial-flaky'
|
||||
filter = 'package(rustfs-ecstore) & (test(decommission_migrates_and_verifies_registered_durable_ilm_records) | test(decommission_durable_ilm_target_read_error_is_not_masked_by_peer_success) | test(decommission_durable_ilm_terminal_receipt_recovers_failed_source_cleanup) | test(decommission_durable_ilm_receipt_pagination_fails_closed_on_second_page) | test(decommission_durable_ilm_recovery_keeps_multiple_active_sources))'
|
||||
test-group = 'ecstore-serial-flaky'
|
||||
|
||||
# Decommission entry and marker/barrier tests share process-wide fault hooks and
|
||||
# deterministic commit barriers. Keep the whole init decommission family in one
|
||||
# nextest group; serial_test alone cannot isolate separate test processes.
|
||||
[[profile.default.overrides]]
|
||||
filter = 'package(rustfs-ecstore) & test(/^store::init::tests::(decommission_|suspended_.*decommission)$/)'
|
||||
test-group = 'ecstore-serial-flaky'
|
||||
|
||||
# Serialize the bucket-incarnation / lifecycle-fence tests. They drive
|
||||
# init_bucket_metadata_sys and bucket_metadata_sys_of, i.e. process-global
|
||||
# OnceLock state that serial_test's #[serial] cannot protect across nextest's
|
||||
@@ -206,13 +167,9 @@ fail-fast = false
|
||||
path = "junit.xml"
|
||||
|
||||
[[profile.ci.scripts]]
|
||||
filter = 'package(rustfs-ecstore) & test(/^(bucket::lifecycle::bucket_lifecycle_ops::tests::manual_transition_worker_result_recovery_marks_unknown_for_corrupt_marker|services::rebalance::entry::tests::real_rebalance_run_fence_loss_blocks_multipart_publication|store::init::tests::(batch_transitioned_delete_uses_free_version_per_item|decommission_entry_(allows_free_version_consumed_before_source_lock|rejects_subquorum_free_version_conflict_and_retains_source|skips_cleanup_only_marker_when_free_version_is_present)|dispatched_tier_delete_recovery_(checks_later_pool_then_commits_after_source_removal|finds_directory_source_on_encoded_set|retains_journal_on_source_metadata_error)|force_tier_remove_blocks_on_physical_free_version_hidden_by_other_pool|legacy_unknown_transition_delete_falls_back_for_single_batch_and_blocks_prefix|multi_pool_(recursive_prefix_rejects_legacy_or_hidden_merge_loser_before_delete|same_remote_tuple_(batch|single)_delete_waits_for_all_sources|same_tuple_recursive_prefix_uses_one_journal_owner|transitioned_delete_persists_one_free_version_per_remote_tuple)|recursive_prefix_partial_(pool|set)_failure_keeps_prepared_cleanup_owners|restored_transitioned_delete_uses_free_version_as_cleanup_owner|stable_transitioned_recursive_prefix_delete_uses_journal_owners|suspended_null_transition_delete_uses_free_version_as_sole_owner|tier_mutation_peer_handler_applies_prepare_commit_and_abort_idempotently|transition_response_loss_persists_unknown_outcome_for_provider_recovery|transition_transaction_recovery_(drops_record_after_confirmed_local_commit|keeps_cleanup_pending_local_commit)|transitioned_delete_(free_version_replays_after_store_restart|local_quorum_failure_rolls_back_without_cleanup_owner|uses_free_version_as_cleanup_owner)|versioned_delete_marker_keeps_transitioned_source_and_remote_object|versioned_explicit_transition_delete_preserves_other_version_then_allows_bucket_delete))$/)'
|
||||
filter = 'package(rustfs-ecstore) & test(/^(bucket::lifecycle::bucket_lifecycle_ops::tests::manual_transition_worker_result_recovery_marks_unknown_for_corrupt_marker|services::rebalance::entry::tests::real_rebalance_run_fence_loss_blocks_multipart_publication|store::init::tests::(decommission_entry_(allows_free_version_consumed_before_source_lock|rejects_subquorum_free_version_conflict_and_retains_source|skips_cleanup_only_marker_when_free_version_is_present)|prepared_tier_delete_recovery_(checks_later_pool_then_commits_after_source_removal|finds_directory_source_on_encoded_set|retains_journal_on_source_metadata_error)|tier_mutation_peer_handler_applies_prepare_commit_and_abort_idempotently|transition_response_loss_persists_unknown_outcome_for_provider_recovery|transition_transaction_recovery_(drops_record_after_confirmed_local_commit|keeps_cleanup_pending_local_commit)))$/)'
|
||||
setup = 'ecstore-large-stack'
|
||||
|
||||
[[profile.ci.scripts]]
|
||||
filter = 'package(rustfs-ecstore) | package(rustfs-s3select-api) | package(rustfs-scanner) | (package(rustfs) & test(/^(app::multipart_usecase::tests::concurrent_completions_share_durable_bucket_quota_reservations|app::object::delete::tests::compressed_delete_requests_update_observed_usage_without_releasing_quota_floor|storage::access::tests::(delete_object_access_captures_authorized_bucket_incarnation|copy_operations_reject_recreated_source_bucket_after_authorization|request_slot_keeps_bucket_policy_bound_to_its_store))$/))'
|
||||
setup = 'ecstore-base-stack'
|
||||
|
||||
[[profile.ci.scripts]]
|
||||
filter = 'binary(lifecycle_integration_test) | (package(rustfs) & test(/^app::lifecycle_transition_api_test::/))'
|
||||
setup = 'lifecycle-large-stack'
|
||||
@@ -273,20 +230,6 @@ test-group = 'e2e-reliability'
|
||||
filter = 'package(rustfs-ecstore) & test(/^set_disk::ops::multipart::tests::crash_consistency::/)'
|
||||
test-group = 'ecstore-serial-flaky'
|
||||
|
||||
# Serialize the heal result-report tests under the ci profile too (see the
|
||||
# matching default-profile override near the top). Not a quarantine: no
|
||||
# retries, just serialized real-disk heal IO.
|
||||
[[profile.ci.overrides]]
|
||||
filter = 'package(rustfs-ecstore) & test(/^set_disk::ops::heal::heal_result_report_tests::/)'
|
||||
test-group = 'ecstore-serial-flaky'
|
||||
|
||||
# Serialize the metadata-cache generation-retirement pair under the ci
|
||||
# profile too (see the matching default-profile override near the top). Not a
|
||||
# quarantine: no retries.
|
||||
[[profile.ci.overrides]]
|
||||
filter = 'package(rustfs-ecstore) & test(retires_cached_snapshot)'
|
||||
test-group = 'ecstore-serial-flaky'
|
||||
|
||||
# Match the default-profile embedded test isolation without quarantining or
|
||||
# retrying failures in CI.
|
||||
[[profile.ci.overrides]]
|
||||
@@ -309,10 +252,6 @@ test-group = 'ecstore-serial-flaky'
|
||||
filter = 'package(rustfs-ecstore) & (test(decommission_migrates_and_verifies_registered_durable_ilm_records) | test(decommission_durable_ilm_target_read_error_is_not_masked_by_peer_success) | test(decommission_durable_ilm_terminal_receipt_recovers_failed_source_cleanup) | test(decommission_durable_ilm_receipt_pagination_fails_closed_on_second_page) | test(decommission_durable_ilm_recovery_keeps_multiple_active_sources))'
|
||||
test-group = 'ecstore-serial-flaky'
|
||||
|
||||
[[profile.ci.overrides]]
|
||||
filter = 'package(rustfs-ecstore) & test(/^store::init::tests::(decommission_|suspended_.*decommission)$/)'
|
||||
test-group = 'ecstore-serial-flaky'
|
||||
|
||||
# Serialize the bucket-incarnation / lifecycle-fence tests under the ci profile
|
||||
# too (see the matching default-profile override near the top). No retries.
|
||||
[[profile.ci.overrides]]
|
||||
@@ -477,7 +416,7 @@ path = "junit.xml"
|
||||
[profile.e2e-nightly]
|
||||
default-filter = """
|
||||
package(e2e_test)
|
||||
& test(/^(admin_timeout_regression_test|cluster_concurrency_test|cluster_multidrive_pool_test|degraded_listing_availability_test|heal_erasure_disk_rebuild_test|namespace_lock_quorum_test|object_lambda_test|stale_multipart_cleanup_cluster_test)::/)
|
||||
& test(/^(admin_timeout_regression_test|cluster_concurrency_test|cluster_multidrive_pool_test|heal_erasure_disk_rebuild_test|namespace_lock_quorum_test|object_lambda_test|stale_multipart_cleanup_cluster_test)::/)
|
||||
"""
|
||||
fail-fast = false
|
||||
|
||||
@@ -529,12 +468,23 @@ path = "junit.xml"
|
||||
# parallel-safe — the same property e2e-smoke relies on. The exceptions are the
|
||||
# 4-disk reliability / degraded-read fault-injection tests and the fixed-port
|
||||
# Vault tests, both serialized below.
|
||||
# KNOWN-FAILURE EXCLUSIONS (characterization run 29381309848, 2026-07-15:
|
||||
# 341 ran / 32 failed on the suites' first automated run ever). Deterministic
|
||||
# product failures cannot be quarantined away with retries, so each family is
|
||||
# excluded here with its tracking issue, under the same discipline as the
|
||||
# ci-profile quarantine (docs/testing/README.md): every entry MUST cite one
|
||||
# OPEN issue, and the fixing PR MUST delete the exclusion. The passing
|
||||
# negative-path siblings of each family stay in as regression guards.
|
||||
# * rustfs#4843 — over-limit archive entry paths hard-reject the whole
|
||||
# archive even under ignore-errors semantics.
|
||||
[profile.e2e-full]
|
||||
default-filter = """
|
||||
package(e2e_test)
|
||||
& !test(/^protocols::/)
|
||||
& !test(/^(admin_timeout_regression_test|cluster_concurrency_test|cluster_multidrive_pool_test|degraded_listing_availability_test|heal_erasure_disk_rebuild_test|namespace_lock_quorum_test|object_lambda_test|stale_multipart_cleanup_cluster_test)::/)
|
||||
& !test(/^(admin_timeout_regression_test|cluster_concurrency_test|cluster_multidrive_pool_test|heal_erasure_disk_rebuild_test|namespace_lock_quorum_test|object_lambda_test|stale_multipart_cleanup_cluster_test)::/)
|
||||
& !test(/^replication_extension_test::/)
|
||||
& !test(/^multipart_auth_test::test_signed_put_object_extract_skips_invalid_entry_when_ignore_errors_enabled$/)
|
||||
& !test(/^snowball_auto_extract_test::tests::snowball_auto_extract_(ignores_invalid_entries_when_requested|supports_standard_headers_with_combined_extract_options)$/)
|
||||
"""
|
||||
fail-fast = false
|
||||
|
||||
|
||||
@@ -24,11 +24,8 @@ on:
|
||||
- '.github/actions/**'
|
||||
- '.github/workflows/**'
|
||||
- 'scripts/release/create_or_update_release.sh'
|
||||
- 'scripts/release/package_versions.sh'
|
||||
- 'scripts/test_package_versions.sh'
|
||||
- 'scripts/security/check_performance_ab_workflow.sh'
|
||||
- 'scripts/security/check_preview_release_workflow.sh'
|
||||
- 'scripts/security/check_tier_artifact_workflow.sh'
|
||||
- 'scripts/security/check_workflow_pins.sh'
|
||||
pull_request:
|
||||
types: [ opened, synchronize, reopened, closed ]
|
||||
@@ -40,11 +37,8 @@ on:
|
||||
- '.github/actions/**'
|
||||
- '.github/workflows/**'
|
||||
- 'scripts/release/create_or_update_release.sh'
|
||||
- 'scripts/release/package_versions.sh'
|
||||
- 'scripts/test_package_versions.sh'
|
||||
- 'scripts/security/check_performance_ab_workflow.sh'
|
||||
- 'scripts/security/check_preview_release_workflow.sh'
|
||||
- 'scripts/security/check_tier_artifact_workflow.sh'
|
||||
- 'scripts/security/check_workflow_pins.sh'
|
||||
schedule:
|
||||
# Daily, not weekly. This schedule exists to catch RustSec advisories
|
||||
@@ -152,12 +146,6 @@ jobs:
|
||||
- name: Check performance A/B workflow trust boundary
|
||||
run: ./scripts/security/check_performance_ab_workflow.sh
|
||||
|
||||
- name: Check tier evidence workflow isolation
|
||||
run: ./scripts/security/check_tier_artifact_workflow.sh
|
||||
|
||||
- name: Check package version contract
|
||||
run: ./scripts/test_package_versions.sh
|
||||
|
||||
dependency-review:
|
||||
name: Dependency Review
|
||||
runs-on: ubuntu-latest
|
||||
|
||||
@@ -244,7 +244,7 @@ jobs:
|
||||
needs: [ build-check, prepare-platform-matrix ]
|
||||
if: needs.build-check.outputs.should_build == 'true' && needs.prepare-platform-matrix.result == 'success'
|
||||
runs-on: ${{ matrix.os }}
|
||||
timeout-minutes: 180
|
||||
timeout-minutes: 150
|
||||
env:
|
||||
FORCE_JAVASCRIPT_ACTIONS_TO_NODE24: "true"
|
||||
# Release binaries ship without dial9 telemetry and therefore do not need
|
||||
@@ -408,9 +408,9 @@ jobs:
|
||||
|
||||
if [[ "${{ matrix.cross }}" == "true" ]]; then
|
||||
# All cross targets in the matrix are Linux; zigbuild handles them.
|
||||
cargo zigbuild --release --target ${{ matrix.target }} -p rustfs --bin rustfs
|
||||
cargo zigbuild --release --target ${{ matrix.target }} -p rustfs --bins
|
||||
else
|
||||
cargo build --release --target ${{ matrix.target }} -p rustfs --bin rustfs
|
||||
cargo build --release --target ${{ matrix.target }} -p rustfs --bins
|
||||
fi
|
||||
|
||||
- name: Create release package
|
||||
|
||||
@@ -49,20 +49,8 @@ env:
|
||||
UPGRADE_SOURCE_SHA256: 7c789386bf85278f865b8e0d359bf4edb84d5aa408cc3fa54a18c25ca74cd6e7
|
||||
|
||||
jobs:
|
||||
upgrade:
|
||||
name: ${{ matrix.name }}
|
||||
strategy:
|
||||
fail-fast: false
|
||||
matrix:
|
||||
include:
|
||||
- name: Direct upgrade from rc.2
|
||||
cache_key: e2e-direct-upgrade
|
||||
test: direct_upgrade_from_rc2_preserves_object_contracts
|
||||
artifact: direct-upgrade
|
||||
- name: Mixed-version rolling upgrade from rc.2
|
||||
cache_key: e2e-mixed-version-upgrade
|
||||
test: rolling_upgrade_from_rc2_preserves_mixed_version_contracts
|
||||
artifact: mixed-version-upgrade
|
||||
direct-upgrade:
|
||||
name: Direct upgrade from rc.2
|
||||
runs-on: ubuntu-latest
|
||||
timeout-minutes: 60
|
||||
env:
|
||||
@@ -76,7 +64,7 @@ jobs:
|
||||
- name: Setup Rust environment
|
||||
uses: ./.github/actions/setup
|
||||
with:
|
||||
cache-shared-key: ${{ matrix.cache_key }}
|
||||
cache-shared-key: e2e-direct-upgrade
|
||||
cache-save-if: ${{ github.ref == 'refs/heads/main' }}
|
||||
install-build-packaging-tools: "false"
|
||||
|
||||
@@ -101,17 +89,17 @@ jobs:
|
||||
cargo build --locked -p rustfs --bin rustfs
|
||||
: > target/debug/rustfs.features
|
||||
|
||||
- name: Run upgrade compatibility test
|
||||
- name: Run direct-upgrade compatibility test
|
||||
run: |
|
||||
cargo test --locked -p e2e_test \
|
||||
"upgrade_compatibility_test::${{ matrix.test }}" \
|
||||
upgrade_compatibility_test::direct_upgrade_from_rc2_preserves_object_contracts \
|
||||
-- --ignored --exact --nocapture
|
||||
|
||||
- name: Upload server logs
|
||||
if: always()
|
||||
uses: actions/upload-artifact@b7c566a772e6b6bfb58ed0dc250532a479d7789f # v6
|
||||
with:
|
||||
name: ${{ matrix.artifact }}-server-logs-${{ github.run_number }}
|
||||
name: direct-upgrade-server-logs-${{ github.run_number }}
|
||||
path: ${{ runner.temp }}/rustfs-upgrade-logs
|
||||
if-no-files-found: warn
|
||||
retention-days: 14
|
||||
|
||||
+92
-162
@@ -21,10 +21,10 @@
|
||||
# - workflow_run: automatically package after "Build and Release" completes
|
||||
# for a release tag (the mac/windows/linux binaries are already uploaded
|
||||
# to the GitHub release before packaging starts)
|
||||
# - workflow_dispatch: manual fallback with a release tag and/or exact build run ID
|
||||
# - workflow_dispatch: manual fallback (backfill / re-run) with optional tag/run_id
|
||||
#
|
||||
# Flow:
|
||||
# 1. Resolve and validate the selected Build workflow run and source identity
|
||||
# 1. Resolve the triggering Build workflow run for the release tag
|
||||
# 2. Download Linux binaries (x86_64-gnu, aarch64-gnu) from build artifacts
|
||||
# 3. Build DEB packages for amd64 and arm64
|
||||
# 4. Build RPM packages for x86_64 and aarch64
|
||||
@@ -51,7 +51,7 @@ on:
|
||||
required: false
|
||||
type: string
|
||||
build_run_id:
|
||||
description: "Build workflow run ID (when combined with tag, both must identify the same release commit)"
|
||||
description: "Build workflow run ID (overrides tag lookup)"
|
||||
required: false
|
||||
type: string
|
||||
|
||||
@@ -82,9 +82,6 @@ jobs:
|
||||
version: ${{ steps.resolve.outputs.version }}
|
||||
build_type: ${{ steps.resolve.outputs.build_type }}
|
||||
build_run_id: ${{ steps.resolve.outputs.build_run_id }}
|
||||
build_run_number: ${{ steps.resolve.outputs.build_run_number }}
|
||||
head_sha: ${{ steps.resolve.outputs.head_sha }}
|
||||
dev_sequence: ${{ steps.resolve.outputs.dev_sequence }}
|
||||
tag: ${{ steps.resolve.outputs.tag }}
|
||||
steps:
|
||||
- name: Resolve build run
|
||||
@@ -92,129 +89,90 @@ jobs:
|
||||
shell: bash
|
||||
env:
|
||||
GH_TOKEN: ${{ github.token }}
|
||||
EVENT_NAME: ${{ github.event_name }}
|
||||
REPOSITORY: ${{ github.repository }}
|
||||
INPUT_TAG: ${{ github.event.inputs.tag }}
|
||||
INPUT_RUN_ID: ${{ github.event.inputs.build_run_id }}
|
||||
run: |
|
||||
set -euo pipefail
|
||||
|
||||
fail() {
|
||||
echo "❌ $1" >&2
|
||||
exit 1
|
||||
}
|
||||
|
||||
TAG=""
|
||||
BUILD_RUN_ID=""
|
||||
case "$EVENT_NAME" in
|
||||
workflow_run)
|
||||
TAG="$HEAD_BRANCH"
|
||||
BUILD_RUN_ID="$WORKFLOW_RUN_ID"
|
||||
;;
|
||||
workflow_dispatch)
|
||||
TAG="$INPUT_TAG"
|
||||
BUILD_RUN_ID="$INPUT_RUN_ID"
|
||||
;;
|
||||
*) fail "unsupported event: $EVENT_NAME" ;;
|
||||
esac
|
||||
|
||||
# Validate and classify tags before using them in API paths or logs.
|
||||
semver_core='(0|[1-9][0-9]*)\.(0|[1-9][0-9]*)\.(0|[1-9][0-9]*)'
|
||||
prerelease_id='(alpha|beta|rc)\.(0|[1-9][0-9]*)'
|
||||
if [[ -n "$TAG" ]]; then
|
||||
if [[ "$TAG" =~ ^${semver_core}-${prerelease_id}-preview\.(0|[1-9][0-9]*)$ ]]; then
|
||||
BUILD_TYPE=preview
|
||||
elif [[ "$TAG" =~ ^${semver_core}-${prerelease_id}$ ]]; then
|
||||
BUILD_TYPE=prerelease
|
||||
elif [[ "$TAG" =~ ^${semver_core}$ ]]; then
|
||||
BUILD_TYPE=release
|
||||
else
|
||||
fail "tag is not a supported strict package version"
|
||||
fi
|
||||
# Determine tag
|
||||
if [[ "${{ github.event_name }}" == "workflow_run" ]]; then
|
||||
TAG="${HEAD_BRANCH}"
|
||||
elif [[ -n "$INPUT_TAG" ]]; then
|
||||
TAG="$INPUT_TAG"
|
||||
else
|
||||
BUILD_TYPE=development
|
||||
TAG=""
|
||||
fi
|
||||
|
||||
if [[ -n "$BUILD_RUN_ID" ]]; then
|
||||
[[ "$BUILD_RUN_ID" =~ ^[1-9][0-9]*$ ]] || fail "build run ID must be a positive decimal integer"
|
||||
echo "Using selected build run: $BUILD_RUN_ID"
|
||||
elif [[ -n "$TAG" ]]; then
|
||||
echo "Looking for build run for tag: $TAG"
|
||||
BUILD_RUN_ID=$(gh api --method GET \
|
||||
"repos/${REPOSITORY}/actions/workflows/build.yml/runs" \
|
||||
-f branch="$TAG" -f status=success -F per_page=1 \
|
||||
--jq '.workflow_runs[0].id // empty' 2>/dev/null || true)
|
||||
echo "Tag: ${TAG:-<none>}"
|
||||
|
||||
if [[ -z "$BUILD_RUN_ID" ]]; then
|
||||
BUILD_RUN_ID=$(gh api --method GET \
|
||||
"repos/${REPOSITORY}/actions/workflows/build.yml/runs" \
|
||||
-f event=push -f status=success -F per_page=100 2>/dev/null |
|
||||
jq -r --arg tag "$TAG" \
|
||||
'[.workflow_runs[] | select(.head_branch == $tag)][0].id // empty' || true)
|
||||
# Determine build run ID
|
||||
BUILD_RUN_ID=""
|
||||
|
||||
if [[ -n "$INPUT_RUN_ID" ]]; then
|
||||
# Explicit run ID takes priority
|
||||
BUILD_RUN_ID="$INPUT_RUN_ID"
|
||||
echo "Using explicit build run ID: $BUILD_RUN_ID"
|
||||
|
||||
elif [[ "${{ github.event_name }}" == "workflow_run" ]]; then
|
||||
# Use the Build and Release run that triggered this workflow
|
||||
BUILD_RUN_ID="${WORKFLOW_RUN_ID}"
|
||||
echo "Using triggering workflow run: $BUILD_RUN_ID"
|
||||
|
||||
elif [[ -n "$TAG" ]]; then
|
||||
# Find the build run that produced this tag
|
||||
echo "Looking for build run for tag: $TAG"
|
||||
BUILD_RUN_ID=$(gh api \
|
||||
"repos/${{ github.repository }}/actions/workflows/build.yml/runs?branch=${TAG}&status=success&per_page=1" \
|
||||
--jq '.workflow_runs[0].id' 2>/dev/null || echo "")
|
||||
|
||||
if [[ -z "$BUILD_RUN_ID" || "$BUILD_RUN_ID" == "null" ]]; then
|
||||
# Tag might not be a branch; try event=push with head_branch matching
|
||||
BUILD_RUN_ID=$(gh api \
|
||||
"repos/${{ github.repository }}/actions/workflows/build.yml/runs?event=push&status=success&per_page=100" \
|
||||
--jq ".workflow_runs[] | select(.head_branch == \"$TAG\") | .id" 2>/dev/null | head -1 || echo "")
|
||||
fi
|
||||
|
||||
if [[ -z "$BUILD_RUN_ID" || "$BUILD_RUN_ID" == "null" ]]; then
|
||||
echo "❌ No successful build run found for tag: $TAG"
|
||||
exit 1
|
||||
fi
|
||||
[[ "$BUILD_RUN_ID" =~ ^[1-9][0-9]*$ ]] || fail "no successful build run found for tag"
|
||||
echo "Found build run: $BUILD_RUN_ID"
|
||||
|
||||
else
|
||||
# No tag — latest successful main build
|
||||
echo "No tag specified, looking for latest main build"
|
||||
BUILD_RUN_ID=$(gh api --method GET \
|
||||
"repos/${REPOSITORY}/actions/workflows/build.yml/runs" \
|
||||
-f branch=main -f status=success -F per_page=1 \
|
||||
--jq '.workflow_runs[0].id // empty' 2>/dev/null || true)
|
||||
[[ "$BUILD_RUN_ID" =~ ^[1-9][0-9]*$ ]] || fail "no successful main build found"
|
||||
BUILD_RUN_ID=$(gh api \
|
||||
"repos/${{ github.repository }}/actions/workflows/build.yml/runs?branch=main&status=success&per_page=1" \
|
||||
--jq '.workflow_runs[0].id' 2>/dev/null || echo "")
|
||||
|
||||
if [[ -z "$BUILD_RUN_ID" || "$BUILD_RUN_ID" == "null" ]]; then
|
||||
echo "❌ No successful main build found"
|
||||
exit 1
|
||||
fi
|
||||
echo "Latest main build: $BUILD_RUN_ID"
|
||||
fi
|
||||
|
||||
# Fetch once and use the same immutable run metadata for identity,
|
||||
# ordering, workflow provenance, and release-channel validation.
|
||||
RUN_JSON=$(gh api "repos/${REPOSITORY}/actions/runs/${BUILD_RUN_ID}") ||
|
||||
fail "cannot read selected build run"
|
||||
RUN_ID=$(jq -r '.id // empty' <<<"$RUN_JSON")
|
||||
RUN_NUMBER=$(jq -r '.run_number // empty' <<<"$RUN_JSON")
|
||||
RUN_STATUS=$(jq -r '.status // empty' <<<"$RUN_JSON")
|
||||
RUN_CONCLUSION=$(jq -r '.conclusion // empty' <<<"$RUN_JSON")
|
||||
RUN_PATH=$(jq -r '.path // empty' <<<"$RUN_JSON")
|
||||
HEAD_SHA=$(jq -r '.head_sha // empty' <<<"$RUN_JSON")
|
||||
RUN_HEAD_BRANCH=$(jq -r '.head_branch // empty' <<<"$RUN_JSON")
|
||||
|
||||
[[ "$RUN_ID" == "$BUILD_RUN_ID" ]] || fail "run metadata ID mismatch"
|
||||
[[ "$RUN_NUMBER" =~ ^[1-9][0-9]*$ ]] || fail "build run number must be a positive decimal integer"
|
||||
[[ "$RUN_STATUS" == completed && "$RUN_CONCLUSION" == success ]] || fail "selected build run is not successful"
|
||||
[[ "$RUN_PATH" == .github/workflows/build.yml ]] || fail "selected run is not Build and Release"
|
||||
[[ "$HEAD_SHA" =~ ^[0-9a-f]{40}$ ]] || fail "selected build run has an invalid head SHA"
|
||||
[[ "$RUN_HEAD_BRANCH" != *$'\n'* && -n "$RUN_HEAD_BRANCH" ]] || fail "selected build run has an invalid head branch"
|
||||
|
||||
# Determine version and build type
|
||||
if [[ -n "$TAG" ]]; then
|
||||
[[ "$RUN_HEAD_BRANCH" == "$TAG" ]] || fail "tag and build run head branch do not match"
|
||||
|
||||
TAG_REF_JSON=$(gh api "repos/${REPOSITORY}/git/ref/tags/${TAG}") ||
|
||||
fail "cannot resolve release tag ref"
|
||||
TAG_OBJECT_TYPE=$(jq -r '.object.type // empty' <<<"$TAG_REF_JSON")
|
||||
TAG_OBJECT_SHA=$(jq -r '.object.sha // empty' <<<"$TAG_REF_JSON")
|
||||
depth=0
|
||||
while [[ "$TAG_OBJECT_TYPE" == tag && $depth -lt 5 ]]; do
|
||||
TAG_OBJECT_JSON=$(gh api "repos/${REPOSITORY}/git/tags/${TAG_OBJECT_SHA}") ||
|
||||
fail "cannot peel annotated release tag"
|
||||
TAG_OBJECT_TYPE=$(jq -r '.object.type // empty' <<<"$TAG_OBJECT_JSON")
|
||||
TAG_OBJECT_SHA=$(jq -r '.object.sha // empty' <<<"$TAG_OBJECT_JSON")
|
||||
depth=$((depth + 1))
|
||||
done
|
||||
[[ "$TAG_OBJECT_TYPE" == commit && "$TAG_OBJECT_SHA" =~ ^[0-9a-f]{40}$ ]] ||
|
||||
fail "release tag does not resolve to a commit"
|
||||
[[ "$TAG_OBJECT_SHA" == "$HEAD_SHA" ]] || fail "release tag commit and build run head SHA do not match"
|
||||
VERSION="$TAG"
|
||||
DEV_SEQUENCE=""
|
||||
if [[ "$TAG" == *"-preview"* ]]; then
|
||||
BUILD_TYPE="preview"
|
||||
elif [[ "$TAG" == *"alpha"* || "$TAG" == *"beta"* || "$TAG" == *"rc"* ]]; then
|
||||
BUILD_TYPE="prerelease"
|
||||
else
|
||||
BUILD_TYPE="release"
|
||||
fi
|
||||
else
|
||||
VERSION="dev-${HEAD_SHA}"
|
||||
DEV_SEQUENCE="$RUN_NUMBER"
|
||||
SHORT_SHA=$(gh api "repos/${{ github.repository }}/actions/runs/${BUILD_RUN_ID}" \
|
||||
--jq '.head_sha' 2>/dev/null | head -c 7)
|
||||
VERSION="dev-${SHORT_SHA}"
|
||||
BUILD_TYPE="development"
|
||||
fi
|
||||
|
||||
{
|
||||
echo "version=$VERSION"
|
||||
echo "build_type=$BUILD_TYPE"
|
||||
echo "build_run_id=$BUILD_RUN_ID"
|
||||
echo "build_run_number=$RUN_NUMBER"
|
||||
echo "head_sha=$HEAD_SHA"
|
||||
echo "dev_sequence=$DEV_SEQUENCE"
|
||||
echo "tag=${TAG}"
|
||||
} >> "$GITHUB_OUTPUT"
|
||||
|
||||
@@ -222,7 +180,6 @@ jobs:
|
||||
echo " Version: $VERSION"
|
||||
echo " Build type: $BUILD_TYPE"
|
||||
echo " Build run ID: $BUILD_RUN_ID"
|
||||
echo " Build run number: $RUN_NUMBER"
|
||||
|
||||
# Build DEB and RPM packages for each architecture
|
||||
package:
|
||||
@@ -249,22 +206,6 @@ jobs:
|
||||
with:
|
||||
persist-credentials: false
|
||||
|
||||
- name: Normalize package metadata
|
||||
id: versions
|
||||
shell: bash
|
||||
env:
|
||||
BUILD_TYPE: ${{ needs.resolve.outputs.build_type }}
|
||||
SOURCE_VERSION: ${{ needs.resolve.outputs.version }}
|
||||
DEV_SEQUENCE: ${{ needs.resolve.outputs.dev_sequence }}
|
||||
DEB_ARCH: ${{ matrix.deb_arch }}
|
||||
RPM_ARCH: ${{ matrix.rpm_arch }}
|
||||
run: |
|
||||
set -euo pipefail
|
||||
|
||||
normalized=$(./scripts/release/package_versions.sh \
|
||||
"$BUILD_TYPE" "$SOURCE_VERSION" "$DEV_SEQUENCE" "$DEB_ARCH" "$RPM_ARCH")
|
||||
printf '%s\n' "$normalized" >> "$GITHUB_OUTPUT"
|
||||
|
||||
- name: Download binary artifact from build run
|
||||
uses: actions/download-artifact@37930b1c2abaa49bbe596cd826c3c89aef350131 # v7
|
||||
with:
|
||||
@@ -304,16 +245,18 @@ jobs:
|
||||
- name: Build DEB package
|
||||
id: deb
|
||||
shell: bash
|
||||
env:
|
||||
DEB_VERSION: ${{ steps.versions.outputs.deb_version }}
|
||||
DEB_ARCH: ${{ matrix.deb_arch }}
|
||||
DEB_FILE: ${{ steps.versions.outputs.deb_file }}
|
||||
run: |
|
||||
set -euo pipefail
|
||||
|
||||
PKG_DIR="${DEB_FILE%.deb}"
|
||||
VERSION="${{ needs.resolve.outputs.version }}"
|
||||
DEB_ARCH="${{ matrix.deb_arch }}"
|
||||
# DEB version: replace - with ~ (1.0.0-beta.12 -> 1.0.0~beta.12)
|
||||
# Use a variable for ~ to prevent tilde expansion by bash
|
||||
TILDE='~'
|
||||
DEB_VERSION="${VERSION/-/$TILDE}"
|
||||
PKG_DIR="rustfs_${DEB_VERSION}_${DEB_ARCH}"
|
||||
|
||||
echo "Building DEB: ${DEB_FILE}"
|
||||
echo "Building DEB: ${PKG_DIR}.deb"
|
||||
|
||||
mkdir -p "${PKG_DIR}/DEBIAN"
|
||||
mkdir -p "${PKG_DIR}/usr/bin"
|
||||
@@ -390,12 +333,9 @@ jobs:
|
||||
cp LICENSE "${PKG_DIR}/usr/share/doc/rustfs/"
|
||||
cp README.md "${PKG_DIR}/usr/share/doc/rustfs/"
|
||||
|
||||
fakeroot dpkg-deb --build "${PKG_DIR}" "$DEB_FILE"
|
||||
fakeroot dpkg-deb --build "${PKG_DIR}"
|
||||
|
||||
[[ $(dpkg-deb -f "$DEB_FILE" Package) == rustfs ]]
|
||||
[[ $(dpkg-deb -f "$DEB_FILE" Version) == "$DEB_VERSION" ]]
|
||||
[[ $(dpkg-deb -f "$DEB_FILE" Architecture) == "$DEB_ARCH" ]]
|
||||
dpkg-deb --fsys-tarfile "$DEB_FILE" | tar -tf - | grep -Fx './usr/bin/rustfs' >/dev/null
|
||||
DEB_FILE="${PKG_DIR}.deb"
|
||||
stat --printf='%n %s bytes\n' "$DEB_FILE"
|
||||
echo "deb_file=$DEB_FILE" >> "$GITHUB_OUTPUT"
|
||||
echo "✅ DEB built: $DEB_FILE"
|
||||
@@ -403,19 +343,16 @@ jobs:
|
||||
- name: Build RPM package
|
||||
id: rpm
|
||||
shell: bash
|
||||
env:
|
||||
RPM_VERSION: ${{ steps.versions.outputs.rpm_version }}
|
||||
RPM_RELEASE: ${{ steps.versions.outputs.rpm_release }}
|
||||
RPM_ARCH: ${{ matrix.rpm_arch }}
|
||||
RPM_FILE: ${{ steps.versions.outputs.rpm_file }}
|
||||
run: |
|
||||
set -euo pipefail
|
||||
|
||||
VERSION="${{ needs.resolve.outputs.version }}"
|
||||
RPM_ARCH="${{ matrix.rpm_arch }}"
|
||||
|
||||
echo "Building RPM for ${RPM_ARCH}"
|
||||
|
||||
sudo apt-get update && sudo apt-get install -y ruby ruby-dev build-essential rpm
|
||||
sudo apt-get update && sudo apt-get install -y ruby ruby-dev build-essential
|
||||
sudo gem install fpm
|
||||
./scripts/test_package_versions.sh --require-package-managers
|
||||
|
||||
# Create config file for fpm (DEB build creates it in its package dir structure,
|
||||
# but fpm needs the file to exist before packaging)
|
||||
@@ -430,10 +367,8 @@ jobs:
|
||||
|
||||
fpm -s dir -t rpm \
|
||||
--name rustfs \
|
||||
--version "$RPM_VERSION" \
|
||||
--iteration "$RPM_RELEASE" \
|
||||
--version "$VERSION" \
|
||||
--architecture "$RPM_ARCH" \
|
||||
--package "$RPM_FILE" \
|
||||
--depends "glibc >= 2.31" \
|
||||
--maintainer "RustFS Team <support@rustfs.com>" \
|
||||
--description "High-performance distributed object storage" \
|
||||
@@ -475,15 +410,13 @@ jobs:
|
||||
LICENSE=/usr/share/doc/rustfs/LICENSE \
|
||||
README.md=/usr/share/doc/rustfs/README.md
|
||||
|
||||
if [[ ! -f "$RPM_FILE" ]]; then
|
||||
RPM_FILE=$(find . -maxdepth 1 -type f -name 'rustfs-*.rpm' -print | head -1)
|
||||
RPM_FILE="${RPM_FILE#./}"
|
||||
if [[ -z "$RPM_FILE" ]]; then
|
||||
echo "❌ RPM build failed"
|
||||
exit 1
|
||||
fi
|
||||
|
||||
RPM_METADATA=$(rpm -qp --qf '%{NAME}\n%{VERSION}\n%{RELEASE}\n%{ARCH}\n' "$RPM_FILE")
|
||||
EXPECTED_METADATA=$(printf 'rustfs\n%s\n%s\n%s' "$RPM_VERSION" "$RPM_RELEASE" "$RPM_ARCH")
|
||||
[[ "$RPM_METADATA" == "$EXPECTED_METADATA" ]]
|
||||
rpm -qpl "$RPM_FILE" | grep -Fx '/usr/bin/rustfs' >/dev/null
|
||||
stat --printf='%n %s bytes\n' "$RPM_FILE"
|
||||
echo "rpm_file=$RPM_FILE" >> "$GITHUB_OUTPUT"
|
||||
echo "✅ RPM built: $RPM_FILE"
|
||||
@@ -505,9 +438,6 @@ jobs:
|
||||
R2_ENDPOINT: ${{ secrets.R2_ENDPOINT }}
|
||||
R2_BUCKET: ${{ secrets.R2_BUCKET }}
|
||||
AWS_EC2_METADATA_DISABLED: true
|
||||
BUILD_TYPE: ${{ needs.resolve.outputs.build_type }}
|
||||
DEB_FILE: ${{ steps.deb.outputs.deb_file }}
|
||||
RPM_FILE: ${{ steps.rpm.outputs.rpm_file }}
|
||||
shell: bash
|
||||
run: |
|
||||
set -euo pipefail
|
||||
@@ -525,6 +455,7 @@ jobs:
|
||||
export AWS_SECRET_ACCESS_KEY="$R2_SECRET_ACCESS_KEY"
|
||||
export AWS_DEFAULT_REGION="auto"
|
||||
|
||||
BUILD_TYPE="${{ needs.resolve.outputs.build_type }}"
|
||||
if [[ "$BUILD_TYPE" == "development" ]]; then
|
||||
R2_PREFIX="artifacts/rustfs/packages/dev"
|
||||
else
|
||||
@@ -534,6 +465,9 @@ jobs:
|
||||
|
||||
echo "📤 Uploading to $R2_PATH"
|
||||
|
||||
DEB_FILE="${{ steps.deb.outputs.deb_file }}"
|
||||
RPM_FILE="${{ steps.rpm.outputs.rpm_file }}"
|
||||
|
||||
for f in "$DEB_FILE" "$RPM_FILE"; do
|
||||
if [[ -n "$f" && -f "$f" ]]; then
|
||||
echo "Uploading: $f"
|
||||
@@ -559,13 +493,14 @@ jobs:
|
||||
if: needs.resolve.outputs.tag != ''
|
||||
env:
|
||||
GH_TOKEN: ${{ github.token }}
|
||||
TAG: ${{ needs.resolve.outputs.tag }}
|
||||
DEB_FILE: ${{ steps.deb.outputs.deb_file }}
|
||||
RPM_FILE: ${{ steps.rpm.outputs.rpm_file }}
|
||||
shell: bash
|
||||
run: |
|
||||
set -euo pipefail
|
||||
|
||||
TAG="${{ needs.resolve.outputs.tag }}"
|
||||
DEB_FILE="${{ steps.deb.outputs.deb_file }}"
|
||||
RPM_FILE="${{ steps.rpm.outputs.rpm_file }}"
|
||||
|
||||
# Upload the packages, then refresh the release checksums so the new
|
||||
# assets are covered, matching the binary release flow.
|
||||
for f in "$DEB_FILE" "$RPM_FILE"; do
|
||||
@@ -617,19 +552,14 @@ jobs:
|
||||
steps:
|
||||
- name: Print summary
|
||||
shell: bash
|
||||
env:
|
||||
SUMMARY_VERSION: ${{ needs.resolve.outputs.version }}
|
||||
SUMMARY_BUILD_TYPE: ${{ needs.resolve.outputs.build_type }}
|
||||
SUMMARY_BUILD_RUN_ID: ${{ needs.resolve.outputs.build_run_id }}
|
||||
SUMMARY_PACKAGE_STATUS: ${{ needs.package.result }}
|
||||
run: |
|
||||
{
|
||||
echo "## 📦 Package Summary"
|
||||
echo ""
|
||||
echo "| Item | Value |"
|
||||
echo "|------|-------|"
|
||||
echo "| Version | \`${SUMMARY_VERSION}\` |"
|
||||
echo "| Build Type | ${SUMMARY_BUILD_TYPE} |"
|
||||
echo "| Build Run | #${SUMMARY_BUILD_RUN_ID} |"
|
||||
echo "| Package Status | ${SUMMARY_PACKAGE_STATUS} |"
|
||||
echo "| Version | \`${{ needs.resolve.outputs.version }}\` |"
|
||||
echo "| Build Type | ${{ needs.resolve.outputs.build_type }} |"
|
||||
echo "| Build Run | #${{ needs.resolve.outputs.build_run_id }} |"
|
||||
echo "| Package Status | ${{ needs.package.result }} |"
|
||||
} >> "$GITHUB_STEP_SUMMARY"
|
||||
|
||||
@@ -1,74 +0,0 @@
|
||||
# Copyright 2024 RustFS Team
|
||||
#
|
||||
# Licensed under the Apache License, Version 2.0 (the "License");
|
||||
# you may not use this file except in compliance with the License.
|
||||
# You may obtain a copy of the License at
|
||||
#
|
||||
# http://www.apache.org/licenses/LICENSE-2.0
|
||||
#
|
||||
# Unless required by applicable law or agreed to in writing, software
|
||||
# distributed under the License is distributed on an "AS IS" BASIS,
|
||||
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||
# See the License for the specific language governing permissions and
|
||||
# limitations under the License.
|
||||
|
||||
# Functional chain driver: runs the nine functional suites in a fixed order
|
||||
# (upgrade -> s3 -> kms -> tier -> storage -> heal -> pool -> security, with
|
||||
# performance on its own runner in parallel) and guarantees the chain keeps
|
||||
# moving even when individual suites fail.
|
||||
#
|
||||
# Each suite workflow can still be dispatched standalone (workflow_dispatch);
|
||||
# only chain-triggered runs forward to the next suite via repository_dispatch,
|
||||
# so a standalone run never drags the rest of the chain behind it.
|
||||
#
|
||||
# Why not workflow_run chaining: GitHub does not guarantee delivery of
|
||||
# workflow_run events (they are fire-and-forget), and the head-SHA filter made
|
||||
# newly added suites (storage) unable to trigger at all. Explicit
|
||||
# repository_dispatch handoffs are verifiable and re-drivable.
|
||||
|
||||
name: RustFS Functional Chain
|
||||
|
||||
on:
|
||||
workflow_dispatch:
|
||||
workflow_run:
|
||||
# Entry point: start the chain after the nightly build completes. The
|
||||
# build's own conclusion does not gate the chain; each suite reports its
|
||||
# own result to rustfs/backlog and the dashboard.
|
||||
workflows: ["Nightly GNU Build"]
|
||||
types: [completed]
|
||||
|
||||
permissions:
|
||||
contents: read
|
||||
|
||||
jobs:
|
||||
start-chain:
|
||||
name: Start functional chain (upgrade first)
|
||||
runs-on: ubuntu-latest
|
||||
timeout-minutes: 10
|
||||
if: ${{ github.event_name == 'workflow_dispatch' || (github.event_name == 'workflow_run' && github.event.workflow_run.event == 'schedule') }}
|
||||
steps:
|
||||
- name: Dispatch first suite (upgrade)
|
||||
env:
|
||||
GH_TOKEN: ${{ secrets.PF_TESTING_GH_TOKEN }}
|
||||
run: |
|
||||
set -euo pipefail
|
||||
if [ -z "${GH_TOKEN:-}" ]; then
|
||||
echo "PF_TESTING_GH_TOKEN is not configured; cannot start the functional chain" >&2
|
||||
exit 1
|
||||
fi
|
||||
gh api --method POST repos/rustfs/rustfs/dispatches \
|
||||
-f event_type='rustfs-chain-upgrade' \
|
||||
-F 'client_payload[from_suite]=nightly-build'
|
||||
|
||||
- name: Dispatch performance suite (parallel, own runner)
|
||||
env:
|
||||
GH_TOKEN: ${{ secrets.PF_TESTING_GH_TOKEN }}
|
||||
run: |
|
||||
set -euo pipefail
|
||||
if [ -z "${GH_TOKEN:-}" ]; then
|
||||
echo "PF_TESTING_GH_TOKEN is not configured; cannot dispatch performance" >&2
|
||||
exit 1
|
||||
fi
|
||||
gh api --method POST repos/rustfs/rustfs/dispatches \
|
||||
-f event_type='rustfs-chain-performance' \
|
||||
-F 'client_payload[from_suite]=nightly-build'
|
||||
@@ -23,11 +23,6 @@ on:
|
||||
description: 'Reset the nodes after the test (DESTROYS test data/config)'
|
||||
type: boolean
|
||||
default: true
|
||||
repository_dispatch:
|
||||
# Chain handoff: dispatched when the storage suite finishes. Heal runs
|
||||
# exactly once per chain; the pool expansion workflow no longer embeds
|
||||
# its own heal pass.
|
||||
types: [rustfs-chain-heal]
|
||||
|
||||
permissions:
|
||||
contents: read
|
||||
@@ -48,39 +43,24 @@ env:
|
||||
RUSTFS_API_ENDPOINT: ${{ secrets.RUSTFS_API_ENDPOINT || vars.RUSTFS_API_ENDPOINT || vars.RUSTFS_RC_ENDPOINT }}
|
||||
RUSTFS_NODES: ${{ secrets.RUSTFS_NODES || vars.RUSTFS_NODES }}
|
||||
RUSTFS_SSH_USER: ${{ secrets.RUSTFS_SSH_USER || vars.RUSTFS_SSH_USER }}
|
||||
PF_TESTING_GH_TOKEN: ${{ secrets.PF_TESTING_GH_TOKEN }}
|
||||
RUSTFS_NIGHTLY_PACKAGE_URL: ${{ vars.RUSTFS_NIGHTLY_PACKAGE_URL || 'https://dl.rustfs.com/artifacts/rustfs/packages/nightly/rustfs-nightly-latest.deb' }}
|
||||
|
||||
jobs:
|
||||
heal-test:
|
||||
runs-on: smoke-testing
|
||||
# Requirement: a failing suite must not fail the workflow; failures
|
||||
# are filed to rustfs/backlog and the chain continues.
|
||||
continue-on-error: true
|
||||
timeout-minutes: 480
|
||||
# Standalone manual run, or one link of the nightly functional chain
|
||||
# (storage -> heal -> pool). Pool expansion no longer re-runs heal.
|
||||
if: ${{ github.event_name == 'workflow_dispatch' || github.event_name == 'repository_dispatch' }}
|
||||
# Manual-only standalone run. Nightly chain already runs heal in
|
||||
# rustfs-pool-expand-test.yml to avoid duplicate heal executions.
|
||||
if: ${{ github.event_name == 'workflow_dispatch' }}
|
||||
steps:
|
||||
# auto-testing is private: clone it with the dedicated PF token (not
|
||||
# GITHUB_TOKEN) and retry transient GitHub/network failures.
|
||||
- name: Checkout auto-testing scripts (with retry)
|
||||
env:
|
||||
GH_TOKEN: ${{ secrets.PF_TESTING_GH_TOKEN }}
|
||||
run: |
|
||||
set -euo pipefail
|
||||
rm -rf auto-testing
|
||||
for attempt in 1 2 3 4 5; do
|
||||
if gh repo clone rustfs/auto-testing auto-testing -- --depth 1 --quiet; then
|
||||
echo "auto-testing cloned (attempt ${attempt})"
|
||||
exit 0
|
||||
fi
|
||||
rm -rf auto-testing
|
||||
echo "clone attempt ${attempt} failed; retrying in $((attempt * 15))s" >&2
|
||||
sleep $((attempt * 15))
|
||||
done
|
||||
echo "ERROR: unable to clone rustfs/auto-testing after 5 attempts" >&2
|
||||
exit 1
|
||||
- name: Checkout auto-testing scripts
|
||||
uses: actions/checkout@9c091bb21b7c1c1d1991bb908d89e4e9dddfe3e0 # v7
|
||||
with:
|
||||
repository: rustfs/auto-testing
|
||||
ref: main
|
||||
path: auto-testing
|
||||
persist-credentials: false
|
||||
token: ${{ secrets.PF_TESTING_GH_TOKEN }}
|
||||
|
||||
- name: Show environment
|
||||
run: |
|
||||
@@ -90,24 +70,11 @@ jobs:
|
||||
warp --version || true
|
||||
df -h /data | tail -1
|
||||
|
||||
- name: Cleanup environment (before)
|
||||
- name: Reset test environment (before)
|
||||
if: ${{ inputs.cleanup_before != 'false' }}
|
||||
run: |
|
||||
set -euo pipefail
|
||||
read -r -a NODES <<< "${RUSTFS_NODES:-vm000 vm001 vm002}"
|
||||
SSH_USER="${RUSTFS_SSH_USER:-azureuser}"
|
||||
for node in "${NODES[@]}"; do
|
||||
ssh -o BatchMode=yes -o ConnectTimeout=10 -o StrictHostKeyChecking=accept-new "${SSH_USER}@${node}" '
|
||||
set -euo pipefail
|
||||
SUDO=""; [ "$(id -u)" -ne 0 ] && SUDO="sudo -n"
|
||||
${SUDO} systemctl stop rustfs 2>/dev/null || true
|
||||
if ${SUDO} dpkg -l rustfs 2>/dev/null | grep -q "^ii"; then
|
||||
${SUDO} dpkg -P rustfs
|
||||
fi
|
||||
for i in 1 2 3 4; do ${SUDO} rm -rf /data/rustfs${i}/mnmd; done
|
||||
${SUDO} rm -rf /var/log/rustfs /var/lib/rustfs/kms /var/lib/rustfs/kms-backup
|
||||
'
|
||||
done
|
||||
chmod +x auto-testing/rustfs_heal_test.sh
|
||||
./auto-testing/rustfs_heal_test.sh --reset -y
|
||||
|
||||
- name: Install RustFS package & start cluster
|
||||
run: |
|
||||
@@ -130,129 +97,14 @@ jobs:
|
||||
./auto-testing/rustfs_heal_test.sh "${ARGS[@]}"
|
||||
|
||||
- name: Run heal test (write -> outage -> heal -> verify)
|
||||
id: test
|
||||
run: |
|
||||
./auto-testing/rustfs_heal_test.sh \
|
||||
--steps "3,4,5,6,7" -y \
|
||||
--endpoint "${{ env.RUSTFS_API_ENDPOINT }}" \
|
||||
--stop-node-gb "${{ inputs.stop_node_gb || '15' }}" \
|
||||
--warp-stop-gb "${{ inputs.warp_stop_gb || '40' }}" \
|
||||
--stop-node-gb "${{ inputs.stop_node_gb }}" \
|
||||
--warp-stop-gb "${{ inputs.warp_stop_gb }}" \
|
||||
--log-file /tmp/rustfs-heal-test.log
|
||||
|
||||
- name: Generate report
|
||||
if: always()
|
||||
env:
|
||||
LOG_FILE: /tmp/rustfs-heal-test.log
|
||||
REPORT_FILE: /tmp/rustfs-heal-report.md
|
||||
run: |
|
||||
set -euo pipefail
|
||||
PACKAGE_URL='${{ inputs.package_url }}'
|
||||
if [ -n "${PACKAGE_URL}" ]; then
|
||||
PACKAGE_SOURCE="${PACKAGE_URL}"
|
||||
else
|
||||
PACKAGE_SOURCE="${RUSTFS_NIGHTLY_PACKAGE_URL}"
|
||||
fi
|
||||
{
|
||||
echo "# RustFS heal test report"
|
||||
echo ""
|
||||
echo "- Run: ${{ github.server_url }}/${{ github.repository }}/actions/runs/${{ github.run_id }}"
|
||||
echo "- Trigger: ${{ github.event_name }}"
|
||||
echo "- Package: ${PACKAGE_SOURCE}"
|
||||
echo "- Test Step Outcome: ${{ steps.test.outcome }}"
|
||||
echo ""
|
||||
echo "## Log tail"
|
||||
echo '```text'
|
||||
tail -n 200 "${LOG_FILE}" || true
|
||||
echo '```'
|
||||
} | tee "${REPORT_FILE}"
|
||||
cat "${REPORT_FILE}" >> "${GITHUB_STEP_SUMMARY}"
|
||||
|
||||
- name: Upload functional report to dashboard
|
||||
if: always()
|
||||
continue-on-error: true
|
||||
env:
|
||||
GH_TOKEN: ${{ env.PF_TESTING_GH_TOKEN }}
|
||||
REPORT_FILE: /tmp/rustfs-heal-report.md
|
||||
SUITE: heal
|
||||
run: |
|
||||
set -euo pipefail
|
||||
if [ -z "${GH_TOKEN:-}" ]; then
|
||||
echo "PF_TESTING_GH_TOKEN is not configured; skipping dashboard upload"
|
||||
exit 0
|
||||
fi
|
||||
DATE="$(date -u +%Y-%m-%d)"
|
||||
REPORT_PATH="functional-reports/${SUITE}/${DATE}.md"
|
||||
CONTENT="$(python3 -c 'import base64,sys;print(base64.b64encode(open(sys.argv[1],"rb").read()).decode())' "${REPORT_FILE}")"
|
||||
SHA="$(gh api "repos/rustfs/dashboard/contents/${REPORT_PATH}" -q '.sha' 2>/dev/null || true)"
|
||||
if [ -n "${SHA}" ]; then
|
||||
jq -n --arg msg "report(${SUITE}): ${DATE}" --arg content "${CONTENT}" --arg sha "${SHA}" \
|
||||
'{message:$msg, content:$content, sha:$sha}' \
|
||||
| gh api --method PUT "repos/rustfs/dashboard/contents/${REPORT_PATH}" --input - >/dev/null
|
||||
else
|
||||
jq -n --arg msg "report(${SUITE}): ${DATE}" --arg content "${CONTENT}" \
|
||||
'{message:$msg, content:$content}' \
|
||||
| gh api --method PUT "repos/rustfs/dashboard/contents/${REPORT_PATH}" --input - >/dev/null
|
||||
fi
|
||||
|
||||
- name: File failure issue in rustfs/backlog
|
||||
if: ${{ always() && (failure() || steps.test.outcome == 'failure' || steps.test.outcome == 'cancelled') }}
|
||||
continue-on-error: true
|
||||
env:
|
||||
GH_TOKEN: ${{ secrets.PF_TESTING_GH_TOKEN }}
|
||||
SUITE: 'heal'
|
||||
SUITE_LABEL: 'Heal'
|
||||
RUN_URL: ${{ github.server_url }}/${{ github.repository }}/actions/runs/${{ github.run_id }}
|
||||
REPORT_FILE: '/tmp/rustfs-heal-report.md'
|
||||
LOG_FILE: '/tmp/rustfs-heal-test.log'
|
||||
run: |
|
||||
set -euo pipefail
|
||||
if [ -z "${GH_TOKEN:-}" ]; then
|
||||
echo "PF_TESTING_GH_TOKEN is not configured; skipping backlog issue"
|
||||
exit 0
|
||||
fi
|
||||
TITLE="[functional][${SUITE}] ${SUITE_LABEL} suite failed (run ${GITHUB_RUN_ID})"
|
||||
EXISTING="$(gh issue list -R rustfs/backlog --state all \
|
||||
--search "in:title \"run ${GITHUB_RUN_ID}\"" \
|
||||
--json number --jq '.[].number' || true)"
|
||||
if [ -n "${EXISTING}" ]; then
|
||||
echo "backlog issue already exists for run ${GITHUB_RUN_ID}; skipping"
|
||||
exit 0
|
||||
fi
|
||||
redact() {
|
||||
sed -E \
|
||||
-e 's/(RUSTFS_(ACCESS_KEY|SECRET_KEY)[=: ]+)[^[:space:]]+/\1[REDACTED]/Ig' \
|
||||
-e 's/(Authorization:).*/\1 [REDACTED]/Ig' \
|
||||
-e 's/(X-Amz-Signature=)[^&[:space:]]+/\1[REDACTED]/Ig' \
|
||||
-e 's/^.*(password|secret|token)[=: ].*/[REDACTED SENSITIVE LINE]/Ig'
|
||||
}
|
||||
BODY_FILE="$(mktemp)"
|
||||
{
|
||||
echo "The **${SUITE_LABEL}** functional suite failed."
|
||||
echo ""
|
||||
echo "- Suite: \`${SUITE}\`"
|
||||
echo "- Run: ${RUN_URL}"
|
||||
echo "- Trigger: ${GITHUB_EVENT_NAME}"
|
||||
echo "- Date: $(date -u +%Y-%m-%d)"
|
||||
echo ""
|
||||
echo "## Report (errors and symptoms)"
|
||||
echo ""
|
||||
if [ -s "${REPORT_FILE}" ]; then
|
||||
redact < "${REPORT_FILE}"
|
||||
elif [ -s "${LOG_FILE:-}" ]; then
|
||||
echo "(report file missing; log tail below)"
|
||||
echo ""
|
||||
tail -n 200 "${LOG_FILE}" | redact
|
||||
else
|
||||
echo "(no report or log file was produced)"
|
||||
fi
|
||||
} | head -c 55000 > "${BODY_FILE}"
|
||||
gh label create functional-test -R rustfs/backlog --color d73a4a 2>/dev/null || true
|
||||
if ! gh issue create -R rustfs/backlog --title "${TITLE}" \
|
||||
--body-file "${BODY_FILE}" --label functional-test; then
|
||||
gh issue create -R rustfs/backlog --title "${TITLE}" --body-file "${BODY_FILE}"
|
||||
fi
|
||||
echo "filed backlog issue for suite ${SUITE}"
|
||||
|
||||
- name: Upload test logs
|
||||
if: always()
|
||||
uses: actions/upload-artifact@b7c566a772e6b6bfb58ed0dc250532a479d7789f # v6
|
||||
@@ -263,42 +115,10 @@ jobs:
|
||||
/tmp/rustfs-warp.*.log
|
||||
if-no-files-found: warn
|
||||
|
||||
- name: Cleanup environment (after)
|
||||
- name: Reset test environment (after)
|
||||
if: ${{ always() && inputs.cleanup_after != 'false' }}
|
||||
run: |
|
||||
set -euo pipefail
|
||||
read -r -a NODES <<< "${RUSTFS_NODES:-vm000 vm001 vm002}"
|
||||
SSH_USER="${RUSTFS_SSH_USER:-azureuser}"
|
||||
for node in "${NODES[@]}"; do
|
||||
ssh -o BatchMode=yes -o ConnectTimeout=10 -o StrictHostKeyChecking=accept-new "${SSH_USER}@${node}" '
|
||||
set -euo pipefail
|
||||
SUDO=""; [ "$(id -u)" -ne 0 ] && SUDO="sudo -n"
|
||||
${SUDO} systemctl stop rustfs 2>/dev/null || true
|
||||
if ${SUDO} dpkg -l rustfs 2>/dev/null | grep -q "^ii"; then
|
||||
${SUDO} dpkg -P rustfs
|
||||
fi
|
||||
for i in 1 2 3 4; do ${SUDO} rm -rf /data/rustfs${i}/mnmd; done
|
||||
${SUDO} rm -rf /var/log/rustfs /var/lib/rustfs/kms /var/lib/rustfs/kms-backup
|
||||
'
|
||||
done
|
||||
|
||||
- name: "Continue functional chain (next: Pool expansion)"
|
||||
# Only chain-triggered runs forward to the next suite; standalone
|
||||
# workflow_dispatch runs stop after their own cleanup.
|
||||
if: ${{ always() && github.event_name == 'repository_dispatch' }}
|
||||
continue-on-error: true
|
||||
env:
|
||||
GH_TOKEN: ${{ secrets.PF_TESTING_GH_TOKEN }}
|
||||
run: |
|
||||
set -euo pipefail
|
||||
if [ -z "${GH_TOKEN:-}" ]; then
|
||||
echo "PF_TESTING_GH_TOKEN is not configured; cannot dispatch the next suite" >&2
|
||||
exit 1
|
||||
fi
|
||||
echo "Dispatching next functional suite: Pool expansion"
|
||||
gh api --method POST repos/rustfs/rustfs/dispatches \
|
||||
-f event_type='rustfs-chain-pool' \
|
||||
-F 'client_payload[from_suite]=heal'
|
||||
./auto-testing/rustfs_heal_test.sh --reset -y
|
||||
|
||||
- name: Notify on failure
|
||||
if: failure()
|
||||
|
||||
@@ -11,21 +11,10 @@ on:
|
||||
description: 'Direct .deb URL (nightly/R2/dev). Overrides rustfs_version.'
|
||||
required: false
|
||||
type: string
|
||||
enforce_sse_key_policy:
|
||||
description: 'Enable RUSTFS_KMS_ENFORCE_SSE_KEY_POLICY (runs KMS-401/402)'
|
||||
type: boolean
|
||||
default: false
|
||||
frame_v2:
|
||||
description: 'Enable RUSTFS_ENCRYPTION_FRAME_V2 (runs KMS-318)'
|
||||
type: boolean
|
||||
default: false
|
||||
config_secret:
|
||||
description: 'Set RUSTFS_KMS_CONFIG_SECRET (runs KMS-107 config sealing)'
|
||||
required: false
|
||||
type: string
|
||||
repository_dispatch:
|
||||
# Chain handoff: dispatched when the S3 compatibility suite finishes.
|
||||
types: [rustfs-chain-kms]
|
||||
workflow_run:
|
||||
# Strict shared-environment order: run after S3 compatibility test succeeds.
|
||||
workflows: ["RustFS S3 Compatibility Test"]
|
||||
types: [completed]
|
||||
|
||||
permissions:
|
||||
contents: read
|
||||
@@ -49,29 +38,17 @@ env:
|
||||
jobs:
|
||||
kms-test:
|
||||
runs-on: smoke-testing
|
||||
continue-on-error: true
|
||||
timeout-minutes: 420
|
||||
if: ${{ github.event_name == 'workflow_dispatch' || github.event_name == 'repository_dispatch' }}
|
||||
if: ${{ github.event_name == 'workflow_dispatch' || github.event.workflow_run.conclusion == 'success' }}
|
||||
steps:
|
||||
# auto-testing is private: clone it with the dedicated PF token (not
|
||||
# GITHUB_TOKEN) and retry transient GitHub/network failures.
|
||||
- name: Checkout auto-testing scripts (with retry)
|
||||
env:
|
||||
GH_TOKEN: ${{ secrets.PF_TESTING_GH_TOKEN }}
|
||||
run: |
|
||||
set -euo pipefail
|
||||
rm -rf auto-testing
|
||||
for attempt in 1 2 3 4 5; do
|
||||
if gh repo clone rustfs/auto-testing auto-testing -- --depth 1 --quiet; then
|
||||
echo "auto-testing cloned (attempt ${attempt})"
|
||||
exit 0
|
||||
fi
|
||||
rm -rf auto-testing
|
||||
echo "clone attempt ${attempt} failed; retrying in $((attempt * 15))s" >&2
|
||||
sleep $((attempt * 15))
|
||||
done
|
||||
echo "ERROR: unable to clone rustfs/auto-testing after 5 attempts" >&2
|
||||
exit 1
|
||||
- name: Checkout auto-testing scripts
|
||||
uses: actions/checkout@9c091bb21b7c1c1d1991bb908d89e4e9dddfe3e0 # v7
|
||||
with:
|
||||
repository: rustfs/auto-testing
|
||||
ref: main
|
||||
path: auto-testing
|
||||
persist-credentials: false
|
||||
token: ${{ secrets.PF_TESTING_GH_TOKEN }}
|
||||
|
||||
- name: Show environment
|
||||
run: |
|
||||
@@ -109,7 +86,6 @@ jobs:
|
||||
|
||||
- name: Run KMS suite
|
||||
id: test
|
||||
continue-on-error: true
|
||||
env:
|
||||
LOG_FILE: /tmp/rustfs-kms.log
|
||||
run: |
|
||||
@@ -118,19 +94,6 @@ jobs:
|
||||
PACKAGE_URL='${{ inputs.package_url }}'
|
||||
RUSTFS_VERSION='${{ inputs.rustfs_version }}'
|
||||
ARGS=(--all-topologies --backends "local,vault-kv2" -y --log-file "${LOG_FILE}")
|
||||
EXTRA_ENV=""
|
||||
if [ "${{ inputs.enforce_sse_key_policy }}" = "true" ]; then
|
||||
EXTRA_ENV+="RUSTFS_KMS_ENFORCE_SSE_KEY_POLICY=true"$'\n'
|
||||
fi
|
||||
if [ "${{ inputs.frame_v2 }}" = "true" ]; then
|
||||
EXTRA_ENV+="RUSTFS_ENCRYPTION_FRAME_V2=true"$'\n'
|
||||
fi
|
||||
if [ -n "${{ inputs.config_secret }}" ]; then
|
||||
EXTRA_ENV+="RUSTFS_KMS_CONFIG_SECRET=${{ inputs.config_secret }}"$'\n'
|
||||
fi
|
||||
if [ -n "${EXTRA_ENV}" ]; then
|
||||
ARGS+=(--extra-env "${EXTRA_ENV}")
|
||||
fi
|
||||
if [ -n "${PACKAGE_URL}" ]; then
|
||||
ARGS+=(--package-url "${PACKAGE_URL}")
|
||||
elif [ -n "${RUSTFS_VERSION}" ]; then
|
||||
@@ -156,65 +119,12 @@ jobs:
|
||||
else
|
||||
PACKAGE_SOURCE="${RUSTFS_NIGHTLY_PACKAGE_URL}"
|
||||
fi
|
||||
CASE_TABLE="/tmp/rustfs-kms-cases.md"
|
||||
python3 - "${LOG_FILE}" "${CASE_TABLE}" <<'PY'
|
||||
import re
|
||||
import sys
|
||||
|
||||
log_file, out_file = sys.argv[1], sys.argv[2]
|
||||
ansi = re.compile(r'\x1b\[[0-9;]*m')
|
||||
start_re = re.compile(r'^---\s+([A-Z]+-[0-9]+)\s+(.+?)\s+---$')
|
||||
done_re = re.compile(r'^\[(PASS|FAIL|UNSUPPORTED)\]\s+([A-Z]+-[0-9]+)\b')
|
||||
|
||||
rows = []
|
||||
index = {}
|
||||
try:
|
||||
with open(log_file, 'r', encoding='utf-8', errors='replace') as fh:
|
||||
for raw in fh:
|
||||
line = ansi.sub('', raw).strip()
|
||||
m = start_re.match(line)
|
||||
if m:
|
||||
case_id, name = m.group(1), m.group(2)
|
||||
if case_id not in index:
|
||||
index[case_id] = len(rows)
|
||||
rows.append([case_id, name, 'RUNNING'])
|
||||
continue
|
||||
m = done_re.match(line)
|
||||
if m:
|
||||
status, case_id = m.group(1), m.group(2)
|
||||
if case_id in index:
|
||||
rows[index[case_id]][2] = status
|
||||
else:
|
||||
rows.append([case_id, case_id, status])
|
||||
index[case_id] = len(rows) - 1
|
||||
except FileNotFoundError:
|
||||
rows = []
|
||||
|
||||
counts = {'PASS': 0, 'FAIL': 0, 'UNSUPPORTED': 0, 'RUNNING': 0}
|
||||
for _, _, status in rows:
|
||||
counts[status] = counts.get(status, 0) + 1
|
||||
|
||||
with open(out_file, 'w', encoding='utf-8') as out:
|
||||
out.write('## Case Summary\n\n')
|
||||
out.write(f"- Total: {len(rows)}\\n")
|
||||
out.write(f"- PASS: {counts.get('PASS', 0)}\\n")
|
||||
out.write(f"- FAIL: {counts.get('FAIL', 0)}\\n")
|
||||
out.write(f"- UNSUPPORTED: {counts.get('UNSUPPORTED', 0)}\\n")
|
||||
out.write('\\n')
|
||||
out.write('| Case | Name | Status |\\n')
|
||||
out.write('| --- | --- | --- |\\n')
|
||||
for case_id, name, status in rows:
|
||||
out.write(f'| {case_id} | {name} | {status} |\\n')
|
||||
PY
|
||||
{
|
||||
echo "# RustFS KMS test report"
|
||||
echo ""
|
||||
echo "- Run: ${{ github.server_url }}/${{ github.repository }}/actions/runs/${{ github.run_id }}"
|
||||
echo "- Trigger: ${{ github.event_name }}"
|
||||
echo "- Package: ${PACKAGE_SOURCE}"
|
||||
echo "- Test Step Outcome: ${{ steps.test.outcome }}"
|
||||
echo ""
|
||||
cat "${CASE_TABLE}" || true
|
||||
echo ""
|
||||
echo "## Log tail"
|
||||
echo '```text'
|
||||
@@ -223,92 +133,6 @@ jobs:
|
||||
} | tee "${REPORT_FILE}"
|
||||
cat "${REPORT_FILE}" >> "${GITHUB_STEP_SUMMARY}"
|
||||
|
||||
- name: Upload functional report to dashboard
|
||||
if: always()
|
||||
continue-on-error: true
|
||||
env:
|
||||
GH_TOKEN: ${{ env.PF_TESTING_GH_TOKEN }}
|
||||
REPORT_FILE: /tmp/rustfs-kms-report.md
|
||||
SUITE: kms
|
||||
run: |
|
||||
set -euo pipefail
|
||||
if [ -z "${GH_TOKEN:-}" ]; then
|
||||
echo "PF_TESTING_GH_TOKEN is not configured; skipping dashboard upload"
|
||||
exit 0
|
||||
fi
|
||||
DATE="$(date -u +%Y-%m-%d)"
|
||||
REPORT_PATH="functional-reports/${SUITE}/${DATE}.md"
|
||||
CONTENT="$(python3 -c 'import base64,sys;print(base64.b64encode(open(sys.argv[1],"rb").read()).decode())' "${REPORT_FILE}")"
|
||||
SHA="$(gh api "repos/rustfs/dashboard/contents/${REPORT_PATH}" -q '.sha' 2>/dev/null || true)"
|
||||
if [ -n "${SHA}" ]; then
|
||||
jq -n --arg msg "report(${SUITE}): ${DATE}" --arg content "${CONTENT}" --arg sha "${SHA}" \
|
||||
'{message:$msg, content:$content, sha:$sha}' \
|
||||
| gh api --method PUT "repos/rustfs/dashboard/contents/${REPORT_PATH}" --input - >/dev/null
|
||||
else
|
||||
jq -n --arg msg "report(${SUITE}): ${DATE}" --arg content "${CONTENT}" \
|
||||
'{message:$msg, content:$content}' \
|
||||
| gh api --method PUT "repos/rustfs/dashboard/contents/${REPORT_PATH}" --input - >/dev/null
|
||||
fi
|
||||
|
||||
- name: File failure issue in rustfs/backlog
|
||||
if: ${{ always() && (failure() || steps.test.outcome == 'failure' || steps.test.outcome == 'cancelled') }}
|
||||
continue-on-error: true
|
||||
env:
|
||||
GH_TOKEN: ${{ secrets.PF_TESTING_GH_TOKEN }}
|
||||
SUITE: 'kms'
|
||||
SUITE_LABEL: 'KMS'
|
||||
RUN_URL: ${{ github.server_url }}/${{ github.repository }}/actions/runs/${{ github.run_id }}
|
||||
REPORT_FILE: '/tmp/rustfs-kms-report.md'
|
||||
LOG_FILE: '/tmp/rustfs-kms.log'
|
||||
run: |
|
||||
set -euo pipefail
|
||||
if [ -z "${GH_TOKEN:-}" ]; then
|
||||
echo "PF_TESTING_GH_TOKEN is not configured; skipping backlog issue"
|
||||
exit 0
|
||||
fi
|
||||
TITLE="[functional][${SUITE}] ${SUITE_LABEL} suite failed (run ${GITHUB_RUN_ID})"
|
||||
EXISTING="$(gh issue list -R rustfs/backlog --state all \
|
||||
--search "in:title \"run ${GITHUB_RUN_ID}\"" \
|
||||
--json number --jq '.[].number' || true)"
|
||||
if [ -n "${EXISTING}" ]; then
|
||||
echo "backlog issue already exists for run ${GITHUB_RUN_ID}; skipping"
|
||||
exit 0
|
||||
fi
|
||||
redact() {
|
||||
sed -E \
|
||||
-e 's/(RUSTFS_(ACCESS_KEY|SECRET_KEY)[=: ]+)[^[:space:]]+/\1[REDACTED]/Ig' \
|
||||
-e 's/(Authorization:).*/\1 [REDACTED]/Ig' \
|
||||
-e 's/(X-Amz-Signature=)[^&[:space:]]+/\1[REDACTED]/Ig' \
|
||||
-e 's/^.*(password|secret|token)[=: ].*/[REDACTED SENSITIVE LINE]/Ig'
|
||||
}
|
||||
BODY_FILE="$(mktemp)"
|
||||
{
|
||||
echo "The **${SUITE_LABEL}** functional suite failed."
|
||||
echo ""
|
||||
echo "- Suite: \`${SUITE}\`"
|
||||
echo "- Run: ${RUN_URL}"
|
||||
echo "- Trigger: ${GITHUB_EVENT_NAME}"
|
||||
echo "- Date: $(date -u +%Y-%m-%d)"
|
||||
echo ""
|
||||
echo "## Report (errors and symptoms)"
|
||||
echo ""
|
||||
if [ -s "${REPORT_FILE}" ]; then
|
||||
redact < "${REPORT_FILE}"
|
||||
elif [ -s "${LOG_FILE:-}" ]; then
|
||||
echo "(report file missing; log tail below)"
|
||||
echo ""
|
||||
tail -n 200 "${LOG_FILE}" | redact
|
||||
else
|
||||
echo "(no report or log file was produced)"
|
||||
fi
|
||||
} | head -c 55000 > "${BODY_FILE}"
|
||||
gh label create functional-test -R rustfs/backlog --color d73a4a 2>/dev/null || true
|
||||
if ! gh issue create -R rustfs/backlog --title "${TITLE}" \
|
||||
--body-file "${BODY_FILE}" --label functional-test; then
|
||||
gh issue create -R rustfs/backlog --title "${TITLE}" --body-file "${BODY_FILE}"
|
||||
fi
|
||||
echo "filed backlog issue for suite ${SUITE}"
|
||||
|
||||
- name: Upload report and logs
|
||||
if: always()
|
||||
uses: actions/upload-artifact@b7c566a772e6b6bfb58ed0dc250532a479d7789f # v6
|
||||
@@ -338,24 +162,6 @@ jobs:
|
||||
'
|
||||
done
|
||||
|
||||
- name: "Continue functional chain (next: Tier)"
|
||||
# Only chain-triggered runs forward to the next suite; standalone
|
||||
# workflow_dispatch runs stop after their own cleanup.
|
||||
if: ${{ always() && github.event_name == 'repository_dispatch' }}
|
||||
continue-on-error: true
|
||||
env:
|
||||
GH_TOKEN: ${{ secrets.PF_TESTING_GH_TOKEN }}
|
||||
run: |
|
||||
set -euo pipefail
|
||||
if [ -z "${GH_TOKEN:-}" ]; then
|
||||
echo "PF_TESTING_GH_TOKEN is not configured; cannot dispatch the next suite" >&2
|
||||
exit 1
|
||||
fi
|
||||
echo "Dispatching next functional suite: Tier"
|
||||
gh api --method POST repos/rustfs/rustfs/dispatches \
|
||||
-f event_type='rustfs-chain-tier' \
|
||||
-F 'client_payload[from_suite]=kms'
|
||||
|
||||
- name: Notify on failure
|
||||
if: failure()
|
||||
run: |
|
||||
|
||||
@@ -48,10 +48,10 @@ on:
|
||||
description: 'Reset the nodes after the test (DESTROYS test data/config)'
|
||||
type: boolean
|
||||
default: true
|
||||
repository_dispatch:
|
||||
# Chain entry: dispatched by rustfs-functional-chain.yml (runs on its own
|
||||
# pf-testing runner, in parallel with the shared-VM chain).
|
||||
types: [rustfs-chain-performance]
|
||||
workflow_run:
|
||||
# Run after the nightly build completes; the nightly deb is what the test installs.
|
||||
workflows: ["Nightly GNU Build"]
|
||||
types: [completed]
|
||||
|
||||
permissions:
|
||||
contents: read
|
||||
@@ -84,33 +84,19 @@ env:
|
||||
jobs:
|
||||
performance-test:
|
||||
runs-on: pf-testing
|
||||
# Requirement: a failing benchmark must not fail the workflow;
|
||||
# failures are filed to rustfs/backlog.
|
||||
continue-on-error: true
|
||||
timeout-minutes: 900
|
||||
# Run on manual dispatch, or when the nightly build completed successfully.
|
||||
# Skipped when nightly failed.
|
||||
if: ${{ github.event_name == 'workflow_dispatch' || github.event_name == 'repository_dispatch' }}
|
||||
if: ${{ github.event_name == 'workflow_dispatch' || github.event.workflow_run.conclusion == 'success' }}
|
||||
steps:
|
||||
# auto-testing is private: clone it with the dedicated PF token (not
|
||||
# GITHUB_TOKEN) and retry transient GitHub/network failures.
|
||||
- name: Checkout auto-testing scripts (with retry)
|
||||
env:
|
||||
GH_TOKEN: ${{ secrets.PF_TESTING_GH_TOKEN }}
|
||||
run: |
|
||||
set -euo pipefail
|
||||
rm -rf auto-testing
|
||||
for attempt in 1 2 3 4 5; do
|
||||
if gh repo clone rustfs/auto-testing auto-testing -- --depth 1 --quiet; then
|
||||
echo "auto-testing cloned (attempt ${attempt})"
|
||||
exit 0
|
||||
fi
|
||||
rm -rf auto-testing
|
||||
echo "clone attempt ${attempt} failed; retrying in $((attempt * 15))s" >&2
|
||||
sleep $((attempt * 15))
|
||||
done
|
||||
echo "ERROR: unable to clone rustfs/auto-testing after 5 attempts" >&2
|
||||
exit 1
|
||||
- name: Checkout auto-testing scripts
|
||||
uses: actions/checkout@9c091bb21b7c1c1d1991bb908d89e4e9dddfe3e0 # v7
|
||||
with:
|
||||
repository: rustfs/auto-testing
|
||||
ref: main
|
||||
path: auto-testing
|
||||
persist-credentials: false
|
||||
token: ${{ secrets.PF_TESTING_GH_TOKEN }}
|
||||
|
||||
- name: Show environment
|
||||
run: |
|
||||
@@ -226,65 +212,6 @@ jobs:
|
||||
echo "created ${REPORT_PATH} in rustfs/dashboard"
|
||||
fi
|
||||
|
||||
- name: File failure issue in rustfs/backlog
|
||||
if: ${{ always() && (failure() || steps.benchmark.outcome == 'failure' || steps.benchmark.outcome == 'cancelled') }}
|
||||
continue-on-error: true
|
||||
env:
|
||||
GH_TOKEN: ${{ secrets.PF_TESTING_GH_TOKEN }}
|
||||
SUITE: 'performance'
|
||||
SUITE_LABEL: 'Performance'
|
||||
RUN_URL: ${{ github.server_url }}/${{ github.repository }}/actions/runs/${{ github.run_id }}
|
||||
REPORT_FILE: '/tmp/rustfs-perf-report.md'
|
||||
LOG_FILE: '/tmp/rustfs-perf-test.log'
|
||||
run: |
|
||||
set -euo pipefail
|
||||
if [ -z "${GH_TOKEN:-}" ]; then
|
||||
echo "PF_TESTING_GH_TOKEN is not configured; skipping backlog issue"
|
||||
exit 0
|
||||
fi
|
||||
TITLE="[functional][${SUITE}] ${SUITE_LABEL} suite failed (run ${GITHUB_RUN_ID})"
|
||||
EXISTING="$(gh issue list -R rustfs/backlog --state all \
|
||||
--search "in:title \"run ${GITHUB_RUN_ID}\"" \
|
||||
--json number --jq '.[].number' || true)"
|
||||
if [ -n "${EXISTING}" ]; then
|
||||
echo "backlog issue already exists for run ${GITHUB_RUN_ID}; skipping"
|
||||
exit 0
|
||||
fi
|
||||
redact() {
|
||||
sed -E \
|
||||
-e 's/(RUSTFS_(ACCESS_KEY|SECRET_KEY)[=: ]+)[^[:space:]]+/\1[REDACTED]/Ig' \
|
||||
-e 's/(Authorization:).*/\1 [REDACTED]/Ig' \
|
||||
-e 's/(X-Amz-Signature=)[^&[:space:]]+/\1[REDACTED]/Ig' \
|
||||
-e 's/^.*(password|secret|token)[=: ].*/[REDACTED SENSITIVE LINE]/Ig'
|
||||
}
|
||||
BODY_FILE="$(mktemp)"
|
||||
{
|
||||
echo "The **${SUITE_LABEL}** functional suite failed."
|
||||
echo ""
|
||||
echo "- Suite: \`${SUITE}\`"
|
||||
echo "- Run: ${RUN_URL}"
|
||||
echo "- Trigger: ${GITHUB_EVENT_NAME}"
|
||||
echo "- Date: $(date -u +%Y-%m-%d)"
|
||||
echo ""
|
||||
echo "## Report (errors and symptoms)"
|
||||
echo ""
|
||||
if [ -s "${REPORT_FILE}" ]; then
|
||||
redact < "${REPORT_FILE}"
|
||||
elif [ -s "${LOG_FILE:-}" ]; then
|
||||
echo "(report file missing; log tail below)"
|
||||
echo ""
|
||||
tail -n 200 "${LOG_FILE}" | redact
|
||||
else
|
||||
echo "(no report or log file was produced)"
|
||||
fi
|
||||
} | head -c 55000 > "${BODY_FILE}"
|
||||
gh label create functional-test -R rustfs/backlog --color d73a4a 2>/dev/null || true
|
||||
if ! gh issue create -R rustfs/backlog --title "${TITLE}" \
|
||||
--body-file "${BODY_FILE}" --label functional-test; then
|
||||
gh issue create -R rustfs/backlog --title "${TITLE}" --body-file "${BODY_FILE}"
|
||||
fi
|
||||
echo "filed backlog issue for suite ${SUITE}"
|
||||
|
||||
- name: Upload test logs & results
|
||||
if: always()
|
||||
uses: actions/upload-artifact@b7c566a772e6b6bfb58ed0dc250532a479d7789f # v6
|
||||
|
||||
@@ -1,4 +1,4 @@
|
||||
name: RustFS Pool Expansion Test
|
||||
name: RustFS Pool Expansion / Heal Test
|
||||
|
||||
on:
|
||||
workflow_dispatch:
|
||||
@@ -25,14 +25,18 @@ on:
|
||||
description: 'warp write duration (e.g. 5m, 10m)'
|
||||
required: false
|
||||
default: '10m'
|
||||
warp_concurrent:
|
||||
description: 'Pool fill: concurrent warp operations'
|
||||
required: false
|
||||
default: '32'
|
||||
run_decommission:
|
||||
description: 'Run the pool decommission step (3-pool topology only)'
|
||||
type: boolean
|
||||
default: true
|
||||
stop_node_gb:
|
||||
description: 'Heal: stop the outage node when surviving nodes reach N GiB'
|
||||
required: false
|
||||
default: '15'
|
||||
warp_stop_gb:
|
||||
description: 'Heal: stop warp when surviving nodes reach N GiB'
|
||||
required: false
|
||||
default: '40'
|
||||
cleanup_before:
|
||||
description: 'Reset the nodes before the test (DESTROYS existing data/config)'
|
||||
type: boolean
|
||||
@@ -41,16 +45,17 @@ on:
|
||||
description: 'Reset the nodes after the test (DESTROYS test data/config)'
|
||||
type: boolean
|
||||
default: true
|
||||
repository_dispatch:
|
||||
# Chain handoff: dispatched when the heal suite finishes.
|
||||
types: [rustfs-chain-pool]
|
||||
workflow_run:
|
||||
# Strict shared-environment order: run after tier test succeeds.
|
||||
workflows: ["RustFS Tier Test"]
|
||||
types: [completed]
|
||||
|
||||
permissions:
|
||||
contents: read
|
||||
|
||||
# Only one test run at a time: the job mutates the same shared test
|
||||
# Only one test run at a time: every job mutates the same shared test
|
||||
# environment (vm000/vm001/vm002), so concurrent runs must not clobber each
|
||||
# other.
|
||||
# other. Jobs inside a run are chained with needs to serialize them.
|
||||
concurrency:
|
||||
group: rustfs-shared-functional-tests
|
||||
cancel-in-progress: false
|
||||
@@ -65,55 +70,25 @@ env:
|
||||
RUSTFS_API_ENDPOINT: ${{ secrets.RUSTFS_API_ENDPOINT || vars.RUSTFS_API_ENDPOINT || vars.RUSTFS_RC_ENDPOINT }}
|
||||
RUSTFS_NODES: ${{ secrets.RUSTFS_NODES || vars.RUSTFS_NODES }}
|
||||
RUSTFS_SSH_USER: ${{ secrets.RUSTFS_SSH_USER || vars.RUSTFS_SSH_USER }}
|
||||
PF_TESTING_GH_TOKEN: ${{ secrets.PF_TESTING_GH_TOKEN }}
|
||||
# Package used by the nightly run (workflow_dispatch inputs are empty for
|
||||
# workflow_run events), i.e. the latest nightly deb published by nightly-gnu.yml.
|
||||
RUSTFS_NIGHTLY_PACKAGE_URL: ${{ vars.RUSTFS_NIGHTLY_PACKAGE_URL || 'https://dl.rustfs.com/artifacts/rustfs/packages/nightly/rustfs-nightly-latest.deb' }}
|
||||
|
||||
jobs:
|
||||
# Pool expansion: dispatched by the heal suite's chain handoff. Heal
|
||||
# itself lives in rustfs-heal-test.yml and runs exactly once per chain.
|
||||
pool-expansion-test:
|
||||
name: Pool expansion / decommission test
|
||||
runs-on: smoke-testing
|
||||
# Requirement: a failing suite must not fail the workflow; failures
|
||||
# are filed to rustfs/backlog and the chain continues.
|
||||
continue-on-error: true
|
||||
timeout-minutes: 360
|
||||
if: ${{ github.event_name == 'workflow_dispatch' || github.event_name == 'repository_dispatch' }}
|
||||
env:
|
||||
RUSTFS_POOL_ADMIN_ENDPOINT: ${{ secrets.RUSTFS_POOL_ADMIN_ENDPOINT || vars.RUSTFS_POOL_ADMIN_ENDPOINT || 'http://rustfs-node1:9000' }}
|
||||
RUSTFS_POOL_PROXY_ENDPOINT: http://127.0.0.1:19000
|
||||
RUSTFS_POOL_WARP_ENDPOINT: http://127.0.0.1:19000
|
||||
RUSTFS_SHARED_PROXY_ENDPOINT: ${{ secrets.RUSTFS_API_ENDPOINT || vars.RUSTFS_API_ENDPOINT || vars.RUSTFS_RC_ENDPOINT }}
|
||||
RUSTFS_POOL_NODE_ENDPOINTS: ${{ secrets.RUSTFS_POOL_NODE_ENDPOINTS || vars.RUSTFS_POOL_NODE_ENDPOINTS || 'http://rustfs-node1:9000 http://rustfs-node2:9000 http://rustfs-node3:9000' }}
|
||||
if: ${{ github.event_name == 'workflow_dispatch' || github.event.workflow_run.conclusion == 'success' }}
|
||||
steps:
|
||||
# auto-testing is private: clone it with the dedicated PF token (not
|
||||
# GITHUB_TOKEN) and retry transient GitHub/network failures.
|
||||
- name: Checkout auto-testing scripts (with retry)
|
||||
env:
|
||||
GH_TOKEN: ${{ secrets.PF_TESTING_GH_TOKEN }}
|
||||
run: |
|
||||
set -euo pipefail
|
||||
rm -rf auto-testing
|
||||
for attempt in 1 2 3 4 5; do
|
||||
if gh repo clone rustfs/auto-testing auto-testing -- --depth 1 --quiet; then
|
||||
echo "auto-testing cloned (attempt ${attempt})"
|
||||
exit 0
|
||||
fi
|
||||
rm -rf auto-testing
|
||||
echo "clone attempt ${attempt} failed; retrying in $((attempt * 15))s" >&2
|
||||
sleep $((attempt * 15))
|
||||
done
|
||||
echo "ERROR: unable to clone rustfs/auto-testing after 5 attempts" >&2
|
||||
exit 1
|
||||
|
||||
- name: Initialize pool test artifacts
|
||||
run: |
|
||||
set -euo pipefail
|
||||
ARTIFACT_DIR="${RUNNER_TEMP}/rustfs-pool-${GITHUB_RUN_ID}-${GITHUB_RUN_ATTEMPT}"
|
||||
mkdir -p "${ARTIFACT_DIR}"
|
||||
echo "POOL_ARTIFACT_DIR=${ARTIFACT_DIR}" >> "${GITHUB_ENV}"
|
||||
- name: Checkout auto-testing scripts
|
||||
uses: actions/checkout@9c091bb21b7c1c1d1991bb908d89e4e9dddfe3e0 # v7
|
||||
with:
|
||||
repository: rustfs/auto-testing
|
||||
ref: main
|
||||
path: auto-testing
|
||||
persist-credentials: false
|
||||
token: ${{ secrets.PF_TESTING_GH_TOKEN }}
|
||||
|
||||
- name: Show environment
|
||||
run: |
|
||||
@@ -123,32 +98,15 @@ jobs:
|
||||
warp --version || true
|
||||
df -h /data | tail -1
|
||||
|
||||
- name: Cleanup environment (before)
|
||||
- name: Reset test environment (before)
|
||||
if: ${{ inputs.cleanup_before != 'false' }}
|
||||
run: |
|
||||
set -euo pipefail
|
||||
read -r -a NODES <<< "${RUSTFS_NODES:-vm000 vm001 vm002}"
|
||||
SSH_USER="${RUSTFS_SSH_USER:-azureuser}"
|
||||
for node in "${NODES[@]}"; do
|
||||
ssh -o BatchMode=yes -o ConnectTimeout=10 -o StrictHostKeyChecking=accept-new "${SSH_USER}@${node}" '
|
||||
set -euo pipefail
|
||||
SUDO=""; [ "$(id -u)" -ne 0 ] && SUDO="sudo -n"
|
||||
${SUDO} systemctl stop rustfs 2>/dev/null || true
|
||||
if ${SUDO} dpkg -l rustfs 2>/dev/null | grep -q "^ii"; then
|
||||
${SUDO} dpkg -P rustfs
|
||||
fi
|
||||
for i in 1 2 3 4; do ${SUDO} rm -rf /data/rustfs${i}/mnmd; done
|
||||
${SUDO} rm -rf /var/log/rustfs /var/lib/rustfs/kms /var/lib/rustfs/kms-backup
|
||||
'
|
||||
done
|
||||
chmod +x auto-testing/rustfs_pool_expand.sh
|
||||
./auto-testing/rustfs_pool_expand.sh --reset -y
|
||||
|
||||
- name: Install RustFS package & start cluster
|
||||
- name: Install RustFS package & start first pool
|
||||
run: |
|
||||
ARGS=(--steps "1,2,3" -y \
|
||||
--admin-endpoint "${RUSTFS_POOL_ADMIN_ENDPOINT}" \
|
||||
--warp-endpoint "${RUSTFS_POOL_WARP_ENDPOINT}" \
|
||||
--node-endpoints "${RUSTFS_POOL_NODE_ENDPOINTS}" \
|
||||
--log-file "${POOL_ARTIFACT_DIR}/pool-test.log")
|
||||
ARGS=(--steps "1,2,3" -y --endpoint "${{ env.RUSTFS_API_ENDPOINT }}")
|
||||
if [ -n "${{ inputs.package_url }}" ]; then
|
||||
ARGS+=(--package-url "${{ inputs.package_url }}")
|
||||
elif [ -n "${{ inputs.rustfs_version }}" ]; then
|
||||
@@ -160,11 +118,7 @@ jobs:
|
||||
|
||||
- name: Preflight checks
|
||||
run: |
|
||||
ARGS=(--preflight \
|
||||
--admin-endpoint "${RUSTFS_POOL_ADMIN_ENDPOINT}" \
|
||||
--warp-endpoint "${RUSTFS_POOL_WARP_ENDPOINT}" \
|
||||
--node-endpoints "${RUSTFS_POOL_NODE_ENDPOINTS}" \
|
||||
--log-file "${POOL_ARTIFACT_DIR}/pool-test.log")
|
||||
ARGS=(--preflight --endpoint "${{ env.RUSTFS_API_ENDPOINT }}")
|
||||
if [ -n "${{ inputs.package_url }}" ]; then
|
||||
ARGS+=(--package-url "${{ inputs.package_url }}")
|
||||
elif [ -n "${{ inputs.rustfs_version }}" ]; then
|
||||
@@ -174,60 +128,6 @@ jobs:
|
||||
fi
|
||||
./auto-testing/rustfs_pool_expand.sh "${ARGS[@]}"
|
||||
|
||||
- name: Reset dedicated pool proxy
|
||||
run: |
|
||||
set -euo pipefail
|
||||
RUSTFS_POOL_NGINX_CONFIG_PATH=/etc/nginx/conf.d/rustfs-pool-test.conf \
|
||||
RUSTFS_POOL_NGINX_LISTEN="${RUSTFS_POOL_PROXY_ENDPOINT#http://}" \
|
||||
RUSTFS_POOL_NGINX_ACCESS_LOG=/var/log/nginx/rustfs-pool-test-access.log \
|
||||
RUSTFS_POOL_NGINX_ERROR_LOG=/var/log/nginx/rustfs-pool-test-error.log \
|
||||
./auto-testing/rustfs_pool_nginx_stage.sh cleanup
|
||||
|
||||
- name: Capture pool test baseline
|
||||
run: |
|
||||
set -uo pipefail
|
||||
BASELINE_FILE="${POOL_ARTIFACT_DIR}/pool-baseline.log"
|
||||
read -r -a NODES <<< "${RUSTFS_NODES:-vm000 vm001 vm002}"
|
||||
read -r -a DIRECT_ENDPOINTS <<< "${RUSTFS_POOL_NODE_ENDPOINTS}"
|
||||
SSH_USER="${RUSTFS_SSH_USER:-azureuser}"
|
||||
failed=0
|
||||
: > "${BASELINE_FILE}"
|
||||
|
||||
if [ "${#DIRECT_ENDPOINTS[@]}" -lt "${#NODES[@]}" ]; then
|
||||
echo "not enough direct endpoints for the configured nodes" | tee -a "${BASELINE_FILE}" >&2
|
||||
exit 1
|
||||
fi
|
||||
|
||||
for index in "${!NODES[@]}"; do
|
||||
node="${NODES[$index]}"
|
||||
endpoint="${DIRECT_ENDPOINTS[$index]}"
|
||||
body_file="${POOL_ARTIFACT_DIR}/ready-baseline-$((index + 1)).body"
|
||||
{
|
||||
echo "--- node=${node} endpoint=${endpoint} ---"
|
||||
if ! ssh -o BatchMode=yes -o ConnectTimeout=10 -o StrictHostKeyChecking=accept-new \
|
||||
"${SSH_USER}@${node}" '
|
||||
set -euo pipefail
|
||||
SUDO=""; [ "$(id -u)" -ne 0 ] && SUDO="sudo -n"
|
||||
echo "--- rustfs version ---"
|
||||
rustfs --version
|
||||
echo "--- systemd state ---"
|
||||
${SUDO} systemctl show rustfs --no-pager \
|
||||
--property=ActiveState,SubState,Result,ExecMainPID,ExecMainStartTimestamp,NRestarts
|
||||
'; then
|
||||
echo "baseline collection failed for ${node}"
|
||||
failed=1
|
||||
fi
|
||||
curl -sS --connect-timeout 5 --max-time 15 -o "${body_file}" \
|
||||
-w "baseline_ready=${endpoint} http=%{http_code} connect=%{time_connect} ttfb=%{time_starttransfer} total=%{time_total}\n" \
|
||||
"${endpoint%/}/health/ready" || true
|
||||
echo "--- readiness body ---"
|
||||
cat "${body_file}" 2>/dev/null || true
|
||||
echo
|
||||
} >> "${BASELINE_FILE}" 2>&1
|
||||
done
|
||||
|
||||
[ "${failed}" -eq 0 ] || exit 1
|
||||
|
||||
- name: Run pool expansion & decommission test
|
||||
id: pool_test
|
||||
run: |
|
||||
@@ -240,16 +140,10 @@ jobs:
|
||||
fi
|
||||
fi
|
||||
ARGS=(--steps "$STEPS" --with-warp -y \
|
||||
--admin-endpoint "${RUSTFS_POOL_ADMIN_ENDPOINT}" \
|
||||
--warp-endpoint "${RUSTFS_POOL_WARP_ENDPOINT}" \
|
||||
--node-endpoints "${RUSTFS_POOL_NODE_ENDPOINTS}" \
|
||||
--endpoint "${{ env.RUSTFS_API_ENDPOINT }}" \
|
||||
--storage-threshold "${{ inputs.storage_threshold || '50' }}" \
|
||||
--warp-duration "${{ inputs.warp_duration || '10m' }}" \
|
||||
--warp-concurrent "${{ inputs.warp_concurrent || '32' }}" \
|
||||
--log-file "${POOL_ARTIFACT_DIR}/pool-test.log")
|
||||
if [ -n "${RUSTFS_POOL_PROXY_ENDPOINT}" ]; then
|
||||
ARGS+=(--proxy-endpoint "${RUSTFS_POOL_PROXY_ENDPOINT}")
|
||||
fi
|
||||
--log-file /tmp/rustfs-pool-test.log)
|
||||
if [ -n "${{ inputs.package_url }}" ]; then
|
||||
ARGS+=(--package-url "${{ inputs.package_url }}")
|
||||
elif [ -n "${{ inputs.rustfs_version }}" ]; then
|
||||
@@ -257,360 +151,22 @@ jobs:
|
||||
else
|
||||
ARGS+=(--package-url "${{ env.RUSTFS_NIGHTLY_PACKAGE_URL }}")
|
||||
fi
|
||||
RUSTFS_WARP_LOG_FILE="${POOL_ARTIFACT_DIR}/warp.log" \
|
||||
RUSTFS_PROXY_STAGE_HOOK=./auto-testing/rustfs_pool_nginx_stage.sh \
|
||||
RUSTFS_POOL_NGINX_CONFIG_PATH=/etc/nginx/conf.d/rustfs-pool-test.conf \
|
||||
RUSTFS_POOL_NGINX_LISTEN="${RUSTFS_POOL_PROXY_ENDPOINT#http://}" \
|
||||
RUSTFS_POOL_NGINX_ACCESS_LOG=/var/log/nginx/rustfs-pool-test-access.log \
|
||||
RUSTFS_POOL_NGINX_ERROR_LOG=/var/log/nginx/rustfs-pool-test-error.log \
|
||||
./auto-testing/rustfs_pool_expand.sh "${ARGS[@]}"
|
||||
|
||||
- name: Collect pool test diagnostics
|
||||
if: always()
|
||||
run: |
|
||||
set -uo pipefail
|
||||
ARTIFACT_DIR="${POOL_ARTIFACT_DIR:-${RUNNER_TEMP}/rustfs-pool-${GITHUB_RUN_ID}-${GITHUB_RUN_ATTEMPT}}"
|
||||
mkdir -p "${ARTIFACT_DIR}"
|
||||
echo "POOL_ARTIFACT_DIR=${ARTIFACT_DIR}" >> "${GITHUB_ENV}"
|
||||
|
||||
redact() {
|
||||
sed -E \
|
||||
-e 's/(RUSTFS_(ACCESS_KEY|SECRET_KEY)=).*/\1[REDACTED]/Ig' \
|
||||
-e 's/(Authorization:).*/\1 [REDACTED]/Ig' \
|
||||
-e 's/(X-Amz-Signature=)[^&[:space:]]+/\1[REDACTED]/Ig' \
|
||||
-e 's/(proxy_set_header[[:space:]]+Authorization[[:space:]]+).*/\1[REDACTED];/Ig' \
|
||||
-e 's/^.*(password|secret|token).*/[REDACTED SENSITIVE LINE]/Ig'
|
||||
}
|
||||
|
||||
if [ "$(id -u)" -eq 0 ]; then
|
||||
SUDO=()
|
||||
else
|
||||
SUDO=(sudo -n)
|
||||
fi
|
||||
|
||||
{
|
||||
echo "captured_at=$(date -u +%Y-%m-%dT%H:%M:%SZ)"
|
||||
echo "run_id=${GITHUB_RUN_ID}"
|
||||
echo "run_attempt=${GITHUB_RUN_ATTEMPT}"
|
||||
if command -v nginx >/dev/null 2>&1; then
|
||||
"${SUDO[@]}" nginx -T 2>&1 || echo "nginx -T failed"
|
||||
else
|
||||
echo "nginx is not installed on the runner"
|
||||
fi
|
||||
} | redact > "${ARTIFACT_DIR}/nginx-config-redacted.txt"
|
||||
|
||||
for log_path in \
|
||||
/var/log/nginx/access.log \
|
||||
/var/log/nginx/error.log \
|
||||
/var/log/nginx/rustfs-pool-test-access.log \
|
||||
/var/log/nginx/rustfs-pool-test-error.log; do
|
||||
log_name="$(basename "${log_path}")"
|
||||
if "${SUDO[@]}" test -r "${log_path}" 2>/dev/null; then
|
||||
"${SUDO[@]}" cat "${log_path}" 2>&1 | redact \
|
||||
> "${ARTIFACT_DIR}/nginx-${log_name%.log}-redacted.log"
|
||||
else
|
||||
echo "unavailable: ${log_path}" > "${ARTIFACT_DIR}/nginx-${log_name%.log}-redacted.log"
|
||||
fi
|
||||
done
|
||||
"${SUDO[@]}" journalctl -u nginx --no-pager -n 5000 2>&1 | redact \
|
||||
> "${ARTIFACT_DIR}/nginx-journal-redacted.log" || true
|
||||
|
||||
read -r -a NODES <<< "${RUSTFS_NODES:-vm000 vm001 vm002}"
|
||||
SSH_USER="${RUSTFS_SSH_USER:-azureuser}"
|
||||
for node in "${NODES[@]}"; do
|
||||
safe_node="${node//[^A-Za-z0-9_.-]/_}"
|
||||
{
|
||||
if ! ssh -o BatchMode=yes -o ConnectTimeout=10 -o StrictHostKeyChecking=accept-new \
|
||||
"${SSH_USER}@${node}" '
|
||||
SUDO=""; [ "$(id -u)" -ne 0 ] && SUDO="sudo -n"
|
||||
echo "--- rustfs version ---"
|
||||
rustfs --version 2>&1 || true
|
||||
echo "--- systemd state ---"
|
||||
${SUDO} systemctl show rustfs --no-pager \
|
||||
--property=ActiveState,SubState,Result,ExecMainPID,ExecMainStartTimestamp,NRestarts 2>&1 || true
|
||||
echo "--- rustfs journal ---"
|
||||
${SUDO} journalctl -u rustfs --no-pager -n 10000 2>&1 || true
|
||||
echo "--- rustfs file logs ---"
|
||||
if ${SUDO} test -d /var/log/rustfs; then
|
||||
${SUDO} find /var/log/rustfs -maxdepth 2 -type f -print 2>/dev/null | while IFS= read -r file; do
|
||||
echo "--- ${file} (last 5000 lines) ---"
|
||||
${SUDO} tail -n 5000 "${file}" 2>&1 || true
|
||||
done
|
||||
else
|
||||
echo "/var/log/rustfs is unavailable"
|
||||
fi
|
||||
'; then
|
||||
echo "SSH diagnostics failed for ${node}"
|
||||
fi
|
||||
} 2>&1 | redact > "${ARTIFACT_DIR}/${safe_node}-rustfs-redacted.log"
|
||||
done
|
||||
|
||||
: > "${ARTIFACT_DIR}/endpoint-ready-probes.log"
|
||||
read -r -a DIRECT_ENDPOINTS <<< "${RUSTFS_POOL_NODE_ENDPOINTS}"
|
||||
probe_index=0
|
||||
for endpoint in "${DIRECT_ENDPOINTS[@]}"; do
|
||||
probe_index=$((probe_index + 1))
|
||||
curl -sS --connect-timeout 5 --max-time 15 -o "${ARTIFACT_DIR}/ready-direct-${probe_index}.body" \
|
||||
-w "direct[${probe_index}]=${endpoint} http=%{http_code} connect=%{time_connect} ttfb=%{time_starttransfer} total=%{time_total}\n" \
|
||||
"${endpoint%/}/health/ready" >> "${ARTIFACT_DIR}/endpoint-ready-probes.log" 2>&1 || true
|
||||
done
|
||||
if [ -n "${RUSTFS_POOL_PROXY_ENDPOINT}" ]; then
|
||||
curl -sS --connect-timeout 5 --max-time 15 -o "${ARTIFACT_DIR}/ready-proxy.body" \
|
||||
-w "proxy=${RUSTFS_POOL_PROXY_ENDPOINT} http=%{http_code} connect=%{time_connect} ttfb=%{time_starttransfer} total=%{time_total}\n" \
|
||||
"${RUSTFS_POOL_PROXY_ENDPOINT%/}/health/ready" >> "${ARTIFACT_DIR}/endpoint-ready-probes.log" 2>&1 || true
|
||||
fi
|
||||
if [ -n "${RUSTFS_SHARED_PROXY_ENDPOINT}" ]; then
|
||||
curl -sS --connect-timeout 5 --max-time 15 -o "${ARTIFACT_DIR}/ready-shared-proxy.body" \
|
||||
-w "shared_proxy=${RUSTFS_SHARED_PROXY_ENDPOINT} http=%{http_code} connect=%{time_connect} ttfb=%{time_starttransfer} total=%{time_total}\n" \
|
||||
"${RUSTFS_SHARED_PROXY_ENDPOINT%/}/health/ready" >> "${ARTIFACT_DIR}/endpoint-ready-probes.log" 2>&1 || true
|
||||
fi
|
||||
|
||||
- name: Generate report
|
||||
if: always()
|
||||
run: |
|
||||
set -euo pipefail
|
||||
LOG_FILE="${POOL_ARTIFACT_DIR}/pool-test.log"
|
||||
REPORT_FILE="${POOL_ARTIFACT_DIR}/pool-report.md"
|
||||
PACKAGE_URL='${{ inputs.package_url }}'
|
||||
RUSTFS_VERSION='${{ inputs.rustfs_version }}'
|
||||
if [ -n "${PACKAGE_URL}" ]; then
|
||||
PACKAGE_SOURCE="${PACKAGE_URL}"
|
||||
elif [ -n "${RUSTFS_VERSION}" ]; then
|
||||
PACKAGE_SOURCE="version ${RUSTFS_VERSION}"
|
||||
else
|
||||
PACKAGE_SOURCE="${RUSTFS_NIGHTLY_PACKAGE_URL}"
|
||||
fi
|
||||
{
|
||||
echo "# RustFS pool expansion test report"
|
||||
echo ""
|
||||
echo "- Run: ${{ github.server_url }}/${{ github.repository }}/actions/runs/${{ github.run_id }}"
|
||||
echo "- Trigger: ${{ github.event_name }}"
|
||||
echo "- Package: ${PACKAGE_SOURCE}"
|
||||
echo "- Warp concurrent: ${{ inputs.warp_concurrent || '32' }}"
|
||||
echo "- Test Step Outcome: ${{ steps.pool_test.outcome }}"
|
||||
echo ""
|
||||
echo "## Log tail"
|
||||
echo '```text'
|
||||
tail -n 200 "${LOG_FILE}" || true
|
||||
echo '```'
|
||||
} | tee "${REPORT_FILE}"
|
||||
cat "${REPORT_FILE}" >> "${GITHUB_STEP_SUMMARY}"
|
||||
|
||||
- name: Validate pool diagnostic completeness
|
||||
if: always()
|
||||
run: |
|
||||
set -euo pipefail
|
||||
failed=0
|
||||
require_nonempty() {
|
||||
if [ ! -s "$1" ]; then
|
||||
echo "required diagnostic is missing or empty: $1" >&2
|
||||
failed=1
|
||||
fi
|
||||
}
|
||||
require_available() {
|
||||
if [ ! -e "$1" ]; then
|
||||
echo "required diagnostic is missing: $1" >&2
|
||||
failed=1
|
||||
elif grep -Fq 'unavailable:' "$1" 2>/dev/null; then
|
||||
echo "required diagnostic could not be collected: $1" >&2
|
||||
failed=1
|
||||
fi
|
||||
}
|
||||
|
||||
require_nonempty "${POOL_ARTIFACT_DIR}/pool-test.log"
|
||||
require_nonempty "${POOL_ARTIFACT_DIR}/warp.log"
|
||||
require_nonempty "${POOL_ARTIFACT_DIR}/pool-report.md"
|
||||
require_nonempty "${POOL_ARTIFACT_DIR}/pool-baseline.log"
|
||||
require_nonempty "${POOL_ARTIFACT_DIR}/nginx-config-redacted.txt"
|
||||
require_nonempty "${POOL_ARTIFACT_DIR}/nginx-rustfs-pool-test-access-redacted.log"
|
||||
require_available "${POOL_ARTIFACT_DIR}/nginx-rustfs-pool-test-access-redacted.log"
|
||||
require_available "${POOL_ARTIFACT_DIR}/nginx-rustfs-pool-test-error-redacted.log"
|
||||
require_nonempty "${POOL_ARTIFACT_DIR}/endpoint-ready-probes.log"
|
||||
|
||||
read -r -a NODES <<< "${RUSTFS_NODES:-vm000 vm001 vm002}"
|
||||
if grep -Fq 'baseline collection failed' "${POOL_ARTIFACT_DIR}/pool-baseline.log" 2>/dev/null; then
|
||||
echo "one or more node baselines could not be collected" >&2
|
||||
failed=1
|
||||
fi
|
||||
for node in "${NODES[@]}"; do
|
||||
safe_node="${node//[^A-Za-z0-9_.-]/_}"
|
||||
node_log="${POOL_ARTIFACT_DIR}/${safe_node}-rustfs-redacted.log"
|
||||
require_nonempty "${node_log}"
|
||||
if grep -Fq "SSH diagnostics failed for ${node}" "${node_log}" 2>/dev/null; then
|
||||
echo "node diagnostics failed: ${node_log}" >&2
|
||||
failed=1
|
||||
fi
|
||||
if ! grep -Eq '^rustfs @' "${node_log}" 2>/dev/null \
|
||||
|| ! grep -Eq '^NRestarts=[0-9]+$' "${node_log}" 2>/dev/null; then
|
||||
echo "node version or restart evidence is incomplete: ${node_log}" >&2
|
||||
failed=1
|
||||
elif grep -Eq '^NRestarts=[1-9][0-9]*$' "${node_log}"; then
|
||||
echo "RustFS restarted unexpectedly during the run: ${node_log}" >&2
|
||||
failed=1
|
||||
fi
|
||||
done
|
||||
|
||||
if ! grep -Fq "upstream_status=\"\$upstream_status\"" \
|
||||
"${POOL_ARTIFACT_DIR}/nginx-config-redacted.txt"; then
|
||||
echo "Nginx config does not expose upstream status fields" >&2
|
||||
failed=1
|
||||
fi
|
||||
if ! grep -Eq '^proxy=.* http=200([[:space:]]|$)' "${POOL_ARTIFACT_DIR}/endpoint-ready-probes.log"; then
|
||||
echo "dedicated proxy readiness probe did not return HTTP 200" >&2
|
||||
failed=1
|
||||
fi
|
||||
if grep -Eq 'status=50(2|4)|upstream_status="[^"]*50(2|4)' \
|
||||
"${POOL_ARTIFACT_DIR}/nginx-rustfs-pool-test-access-redacted.log"; then
|
||||
echo "dedicated proxy access log contains a 502/504 response" >&2
|
||||
failed=1
|
||||
fi
|
||||
if grep -Eiq 'upstream prematurely closed connection|upstream timed out|(connect\(\)|recv\(\)|send\(\)) failed.*upstream|connection reset by peer.*upstream' \
|
||||
"${POOL_ARTIFACT_DIR}/nginx-rustfs-pool-test-error-redacted.log"; then
|
||||
echo "dedicated proxy error log contains an upstream timeout or connection failure" >&2
|
||||
failed=1
|
||||
fi
|
||||
[ "${failed}" -eq 0 ] || exit 1
|
||||
|
||||
- name: Upload functional report to dashboard
|
||||
if: always()
|
||||
continue-on-error: true
|
||||
env:
|
||||
GH_TOKEN: ${{ env.PF_TESTING_GH_TOKEN }}
|
||||
SUITE: pool
|
||||
run: |
|
||||
set -euo pipefail
|
||||
REPORT_FILE="${POOL_ARTIFACT_DIR}/pool-report.md"
|
||||
if [ -z "${GH_TOKEN:-}" ]; then
|
||||
echo "PF_TESTING_GH_TOKEN is not configured; skipping dashboard upload"
|
||||
exit 0
|
||||
fi
|
||||
DATE="$(date -u +%Y-%m-%d)"
|
||||
REPORT_PATH="functional-reports/${SUITE}/${DATE}.md"
|
||||
CONTENT="$(python3 -c 'import base64,sys;print(base64.b64encode(open(sys.argv[1],"rb").read()).decode())' "${REPORT_FILE}")"
|
||||
SHA="$(gh api "repos/rustfs/dashboard/contents/${REPORT_PATH}" -q '.sha' 2>/dev/null || true)"
|
||||
if [ -n "${SHA}" ]; then
|
||||
jq -n --arg msg "report(${SUITE}): ${DATE}" --arg content "${CONTENT}" --arg sha "${SHA}" \
|
||||
'{message:$msg, content:$content, sha:$sha}' \
|
||||
| gh api --method PUT "repos/rustfs/dashboard/contents/${REPORT_PATH}" --input - >/dev/null
|
||||
else
|
||||
jq -n --arg msg "report(${SUITE}): ${DATE}" --arg content "${CONTENT}" \
|
||||
'{message:$msg, content:$content}' \
|
||||
| gh api --method PUT "repos/rustfs/dashboard/contents/${REPORT_PATH}" --input - >/dev/null
|
||||
fi
|
||||
|
||||
- name: File failure issue in rustfs/backlog
|
||||
if: ${{ always() && (failure() || steps.pool_test.outcome == 'failure' || steps.pool_test.outcome == 'cancelled') }}
|
||||
continue-on-error: true
|
||||
env:
|
||||
GH_TOKEN: ${{ secrets.PF_TESTING_GH_TOKEN }}
|
||||
SUITE: 'pool'
|
||||
SUITE_LABEL: 'Pool expansion'
|
||||
RUN_URL: ${{ github.server_url }}/${{ github.repository }}/actions/runs/${{ github.run_id }}
|
||||
REPORT_FILE: '${{ env.POOL_ARTIFACT_DIR }}/pool-report.md'
|
||||
LOG_FILE: '${{ env.POOL_ARTIFACT_DIR }}/pool-test.log'
|
||||
run: |
|
||||
set -euo pipefail
|
||||
if [ -z "${GH_TOKEN:-}" ]; then
|
||||
echo "PF_TESTING_GH_TOKEN is not configured; skipping backlog issue"
|
||||
exit 0
|
||||
fi
|
||||
TITLE="[functional][${SUITE}] ${SUITE_LABEL} suite failed (run ${GITHUB_RUN_ID})"
|
||||
EXISTING="$(gh issue list -R rustfs/backlog --state all \
|
||||
--search "in:title \"run ${GITHUB_RUN_ID}\"" \
|
||||
--json number --jq '.[].number' || true)"
|
||||
if [ -n "${EXISTING}" ]; then
|
||||
echo "backlog issue already exists for run ${GITHUB_RUN_ID}; skipping"
|
||||
exit 0
|
||||
fi
|
||||
redact() {
|
||||
sed -E \
|
||||
-e 's/(RUSTFS_(ACCESS_KEY|SECRET_KEY)[=: ]+)[^[:space:]]+/\1[REDACTED]/Ig' \
|
||||
-e 's/(Authorization:).*/\1 [REDACTED]/Ig' \
|
||||
-e 's/(X-Amz-Signature=)[^&[:space:]]+/\1[REDACTED]/Ig' \
|
||||
-e 's/^.*(password|secret|token)[=: ].*/[REDACTED SENSITIVE LINE]/Ig'
|
||||
}
|
||||
BODY_FILE="$(mktemp)"
|
||||
{
|
||||
echo "The **${SUITE_LABEL}** functional suite failed."
|
||||
echo ""
|
||||
echo "- Suite: \`${SUITE}\`"
|
||||
echo "- Run: ${RUN_URL}"
|
||||
echo "- Trigger: ${GITHUB_EVENT_NAME}"
|
||||
echo "- Date: $(date -u +%Y-%m-%d)"
|
||||
echo ""
|
||||
echo "## Report (errors and symptoms)"
|
||||
echo ""
|
||||
if [ -s "${REPORT_FILE}" ]; then
|
||||
redact < "${REPORT_FILE}"
|
||||
elif [ -s "${LOG_FILE:-}" ]; then
|
||||
echo "(report file missing; log tail below)"
|
||||
echo ""
|
||||
tail -n 200 "${LOG_FILE}" | redact
|
||||
else
|
||||
echo "(no report or log file was produced)"
|
||||
fi
|
||||
} | head -c 55000 > "${BODY_FILE}"
|
||||
gh label create functional-test -R rustfs/backlog --color d73a4a 2>/dev/null || true
|
||||
if ! gh issue create -R rustfs/backlog --title "${TITLE}" \
|
||||
--body-file "${BODY_FILE}" --label functional-test; then
|
||||
gh issue create -R rustfs/backlog --title "${TITLE}" --body-file "${BODY_FILE}"
|
||||
fi
|
||||
echo "filed backlog issue for suite ${SUITE}"
|
||||
./auto-testing/rustfs_pool_expand.sh "${ARGS[@]}"
|
||||
|
||||
- name: Upload test logs
|
||||
if: always()
|
||||
uses: actions/upload-artifact@b7c566a772e6b6bfb58ed0dc250532a479d7789f # v6
|
||||
with:
|
||||
name: rustfs-pool-test-${{ github.run_id }}-${{ github.run_attempt }}
|
||||
path: ${{ runner.temp }}/rustfs-pool-${{ github.run_id }}-${{ github.run_attempt }}
|
||||
name: rustfs-pool-test-${{ github.run_id }}
|
||||
path: |
|
||||
/tmp/rustfs-pool-test*.log
|
||||
/tmp/rustfs-warp.*.log
|
||||
if-no-files-found: warn
|
||||
|
||||
- name: Restore dedicated pool proxy
|
||||
if: always()
|
||||
run: |
|
||||
set -euo pipefail
|
||||
RUSTFS_POOL_NGINX_CONFIG_PATH=/etc/nginx/conf.d/rustfs-pool-test.conf \
|
||||
RUSTFS_POOL_NGINX_LISTEN="${RUSTFS_POOL_PROXY_ENDPOINT#http://}" \
|
||||
RUSTFS_POOL_NGINX_ACCESS_LOG=/var/log/nginx/rustfs-pool-test-access.log \
|
||||
RUSTFS_POOL_NGINX_ERROR_LOG=/var/log/nginx/rustfs-pool-test-error.log \
|
||||
./auto-testing/rustfs_pool_nginx_stage.sh cleanup
|
||||
|
||||
- name: Cleanup environment (after)
|
||||
- name: Reset test environment (after)
|
||||
if: ${{ always() && inputs.cleanup_after != 'false' }}
|
||||
run: |
|
||||
set -euo pipefail
|
||||
read -r -a NODES <<< "${RUSTFS_NODES:-vm000 vm001 vm002}"
|
||||
SSH_USER="${RUSTFS_SSH_USER:-azureuser}"
|
||||
for node in "${NODES[@]}"; do
|
||||
ssh -o BatchMode=yes -o ConnectTimeout=10 -o StrictHostKeyChecking=accept-new "${SSH_USER}@${node}" '
|
||||
set -euo pipefail
|
||||
SUDO=""; [ "$(id -u)" -ne 0 ] && SUDO="sudo -n"
|
||||
${SUDO} systemctl stop rustfs 2>/dev/null || true
|
||||
if ${SUDO} dpkg -l rustfs 2>/dev/null | grep -q "^ii"; then
|
||||
${SUDO} dpkg -P rustfs
|
||||
fi
|
||||
for i in 1 2 3 4; do ${SUDO} rm -rf /data/rustfs${i}/mnmd; done
|
||||
${SUDO} rm -rf /var/log/rustfs /var/lib/rustfs/kms /var/lib/rustfs/kms-backup
|
||||
'
|
||||
done
|
||||
|
||||
- name: "Continue functional chain (next: Security)"
|
||||
# Only chain-triggered runs forward to the next suite; standalone
|
||||
# workflow_dispatch runs stop after their own cleanup.
|
||||
if: ${{ always() && github.event_name == 'repository_dispatch' }}
|
||||
continue-on-error: true
|
||||
env:
|
||||
GH_TOKEN: ${{ secrets.PF_TESTING_GH_TOKEN }}
|
||||
run: |
|
||||
set -euo pipefail
|
||||
if [ -z "${GH_TOKEN:-}" ]; then
|
||||
echo "PF_TESTING_GH_TOKEN is not configured; cannot dispatch the next suite" >&2
|
||||
exit 1
|
||||
fi
|
||||
echo "Dispatching next functional suite: Security"
|
||||
gh api --method POST repos/rustfs/rustfs/dispatches \
|
||||
-f event_type='rustfs-chain-security' \
|
||||
-F 'client_payload[from_suite]=pool'
|
||||
./auto-testing/rustfs_pool_expand.sh --reset -y
|
||||
|
||||
- name: Notify on failure
|
||||
if: failure()
|
||||
@@ -618,3 +174,82 @@ jobs:
|
||||
echo "RustFS pool expansion test failed"
|
||||
echo "Package source: ${{ inputs.package_url || inputs.rustfs_version || 'nightly (R2 latest)' }}"
|
||||
echo "See the uploaded log artifact for details."
|
||||
|
||||
# Heal regression runs after the pool test regardless of its outcome: a pool
|
||||
# failure must be reported (it makes the run red) but must not block heal.
|
||||
heal-test:
|
||||
name: Heal test (after pool test)
|
||||
runs-on: smoke-testing
|
||||
timeout-minutes: 480
|
||||
needs: pool-expansion-test
|
||||
if: ${{ always() && (github.event_name == 'workflow_dispatch' || github.event.workflow_run.conclusion == 'success') }}
|
||||
steps:
|
||||
- name: Checkout auto-testing scripts
|
||||
uses: actions/checkout@9c091bb21b7c1c1d1991bb908d89e4e9dddfe3e0 # v7
|
||||
with:
|
||||
repository: rustfs/auto-testing
|
||||
ref: main
|
||||
path: auto-testing
|
||||
persist-credentials: false
|
||||
token: ${{ secrets.PF_TESTING_GH_TOKEN }}
|
||||
|
||||
- name: Reset test environment (before)
|
||||
if: ${{ inputs.cleanup_before != 'false' }}
|
||||
run: |
|
||||
chmod +x auto-testing/rustfs_heal_test.sh
|
||||
./auto-testing/rustfs_heal_test.sh --reset -y
|
||||
|
||||
- name: Install RustFS package & start cluster
|
||||
run: |
|
||||
ARGS=(--steps "1,2" -y --endpoint "${{ env.RUSTFS_API_ENDPOINT }}")
|
||||
if [ -n "${{ inputs.package_url }}" ]; then
|
||||
ARGS+=(--package-url "${{ inputs.package_url }}")
|
||||
else
|
||||
ARGS+=(--package-url "${{ env.RUSTFS_NIGHTLY_PACKAGE_URL }}")
|
||||
fi
|
||||
./auto-testing/rustfs_heal_test.sh "${ARGS[@]}"
|
||||
|
||||
- name: Preflight checks
|
||||
run: |
|
||||
ARGS=(--preflight --endpoint "${{ env.RUSTFS_API_ENDPOINT }}")
|
||||
if [ -n "${{ inputs.package_url }}" ]; then
|
||||
ARGS+=(--package-url "${{ inputs.package_url }}")
|
||||
else
|
||||
ARGS+=(--package-url "${{ env.RUSTFS_NIGHTLY_PACKAGE_URL }}")
|
||||
fi
|
||||
./auto-testing/rustfs_heal_test.sh "${ARGS[@]}"
|
||||
|
||||
- name: Run heal test (write -> outage -> heal -> verify)
|
||||
run: |
|
||||
ARGS=(--steps "3,4,5,6,7" -y \
|
||||
--endpoint "${{ env.RUSTFS_API_ENDPOINT }}" \
|
||||
--stop-node-gb "${{ inputs.stop_node_gb || '15' }}" \
|
||||
--warp-stop-gb "${{ inputs.warp_stop_gb || '40' }}" \
|
||||
--log-file /tmp/rustfs-heal-test.log)
|
||||
if [ -n "${{ inputs.package_url }}" ]; then
|
||||
ARGS+=(--package-url "${{ inputs.package_url }}")
|
||||
else
|
||||
ARGS+=(--package-url "${{ env.RUSTFS_NIGHTLY_PACKAGE_URL }}")
|
||||
fi
|
||||
./auto-testing/rustfs_heal_test.sh "${ARGS[@]}"
|
||||
|
||||
- name: Upload test logs
|
||||
if: always()
|
||||
uses: actions/upload-artifact@b7c566a772e6b6bfb58ed0dc250532a479d7789f # v6
|
||||
with:
|
||||
name: rustfs-heal-test-${{ github.run_id }}
|
||||
path: |
|
||||
/tmp/rustfs-heal-test.log
|
||||
/tmp/rustfs-warp.*.log
|
||||
if-no-files-found: warn
|
||||
|
||||
- name: Reset test environment (after)
|
||||
if: ${{ always() && inputs.cleanup_after != 'false' }}
|
||||
run: |
|
||||
./auto-testing/rustfs_heal_test.sh --reset -y
|
||||
|
||||
- name: Notify on failure
|
||||
if: failure()
|
||||
run: |
|
||||
echo "RustFS heal test failed"
|
||||
echo "See the uploaded log artifact for details."
|
||||
|
||||
@@ -11,9 +11,10 @@ on:
|
||||
description: 'Direct .deb URL (nightly/R2/dev). Overrides rustfs_version.'
|
||||
required: false
|
||||
type: string
|
||||
repository_dispatch:
|
||||
# Chain handoff: dispatched when the upgrade suite finishes.
|
||||
types: [rustfs-chain-s3]
|
||||
workflow_run:
|
||||
# Run after the nightly build completes; the nightly deb is what the test installs.
|
||||
workflows: ["Nightly GNU Build"]
|
||||
types: [completed]
|
||||
|
||||
permissions:
|
||||
contents: read
|
||||
@@ -32,34 +33,21 @@ env:
|
||||
RUSTFS_NODES: ${{ secrets.RUSTFS_NODES || vars.RUSTFS_NODES }}
|
||||
RUSTFS_SSH_USER: ${{ secrets.RUSTFS_SSH_USER || vars.RUSTFS_SSH_USER }}
|
||||
RUSTFS_NIGHTLY_PACKAGE_URL: ${{ vars.RUSTFS_NIGHTLY_PACKAGE_URL || 'https://dl.rustfs.com/artifacts/rustfs/packages/nightly/rustfs-nightly-latest.deb' }}
|
||||
PF_TESTING_GH_TOKEN: ${{ secrets.PF_TESTING_GH_TOKEN }}
|
||||
|
||||
jobs:
|
||||
s3-compat-test:
|
||||
runs-on: smoke-testing
|
||||
continue-on-error: true
|
||||
timeout-minutes: 360
|
||||
if: ${{ github.event_name == 'workflow_dispatch' || github.event_name == 'repository_dispatch' }}
|
||||
if: ${{ github.event_name == 'workflow_dispatch' || github.event.workflow_run.conclusion == 'success' }}
|
||||
steps:
|
||||
# auto-testing is private: clone it with the dedicated PF token (not
|
||||
# GITHUB_TOKEN) and retry transient GitHub/network failures.
|
||||
- name: Checkout auto-testing scripts (with retry)
|
||||
env:
|
||||
GH_TOKEN: ${{ secrets.PF_TESTING_GH_TOKEN }}
|
||||
run: |
|
||||
set -euo pipefail
|
||||
rm -rf auto-testing
|
||||
for attempt in 1 2 3 4 5; do
|
||||
if gh repo clone rustfs/auto-testing auto-testing -- --depth 1 --quiet; then
|
||||
echo "auto-testing cloned (attempt ${attempt})"
|
||||
exit 0
|
||||
fi
|
||||
rm -rf auto-testing
|
||||
echo "clone attempt ${attempt} failed; retrying in $((attempt * 15))s" >&2
|
||||
sleep $((attempt * 15))
|
||||
done
|
||||
echo "ERROR: unable to clone rustfs/auto-testing after 5 attempts" >&2
|
||||
exit 1
|
||||
- name: Checkout auto-testing scripts
|
||||
uses: actions/checkout@9c091bb21b7c1c1d1991bb908d89e4e9dddfe3e0 # v7
|
||||
with:
|
||||
repository: rustfs/auto-testing
|
||||
ref: main
|
||||
path: auto-testing
|
||||
persist-credentials: false
|
||||
token: ${{ secrets.PF_TESTING_GH_TOKEN }}
|
||||
|
||||
- name: Show environment
|
||||
run: |
|
||||
@@ -88,7 +76,6 @@ jobs:
|
||||
|
||||
- name: Run S3 compatibility suite
|
||||
id: test
|
||||
continue-on-error: true
|
||||
env:
|
||||
LOG_FILE: /tmp/rustfs-s3-compat.log
|
||||
run: |
|
||||
@@ -122,79 +109,12 @@ jobs:
|
||||
else
|
||||
PACKAGE_SOURCE="${RUSTFS_NIGHTLY_PACKAGE_URL}"
|
||||
fi
|
||||
read -r -a NODES <<< "${RUSTFS_NODES:-vm000 vm001 vm002}"
|
||||
SSH_USER="${RUSTFS_SSH_USER:-azureuser}"
|
||||
RUSTFS_VERSION_INFO="N/A"
|
||||
if [ "${#NODES[@]}" -gt 0 ]; then
|
||||
DETECTED_VERSION="$(ssh -o BatchMode=yes -o ConnectTimeout=10 -o StrictHostKeyChecking=accept-new \
|
||||
"${SSH_USER}@${NODES[0]}" 'rustfs --version' 2>/dev/null | tr -d '\r' | head -n 1 || true)"
|
||||
if [ -n "${DETECTED_VERSION}" ]; then
|
||||
RUSTFS_VERSION_INFO="${DETECTED_VERSION}"
|
||||
fi
|
||||
fi
|
||||
CASE_TABLE="/tmp/rustfs-s3-compat-cases.md"
|
||||
python3 - "${LOG_FILE}" "${CASE_TABLE}" <<'PY'
|
||||
import re
|
||||
import sys
|
||||
|
||||
log_file, out_file = sys.argv[1], sys.argv[2]
|
||||
ansi = re.compile(r'\x1b\[[0-9;]*m')
|
||||
start_re = re.compile(r'^---\s+([A-Z0-9]+-[0-9]+)\s+(.+?)\s+---$')
|
||||
done_re = re.compile(r'^\[(PASS|FAIL|UNSUPPORTED)\]\s+([A-Z0-9]+-[0-9]+)\b')
|
||||
|
||||
rows = []
|
||||
index = {}
|
||||
current = None
|
||||
try:
|
||||
with open(log_file, 'r', encoding='utf-8', errors='replace') as fh:
|
||||
for raw in fh:
|
||||
line = ansi.sub('', raw).strip()
|
||||
m = start_re.match(line)
|
||||
if m:
|
||||
case_id, name = m.group(1), m.group(2)
|
||||
current = case_id
|
||||
if case_id not in index:
|
||||
index[case_id] = len(rows)
|
||||
rows.append([case_id, name, 'RUNNING'])
|
||||
continue
|
||||
m = done_re.match(line)
|
||||
if m:
|
||||
status, case_id = m.group(1), m.group(2)
|
||||
if case_id in index:
|
||||
rows[index[case_id]][2] = status
|
||||
else:
|
||||
rows.append([case_id, case_id, status])
|
||||
index[case_id] = len(rows) - 1
|
||||
current = None
|
||||
except FileNotFoundError:
|
||||
rows = []
|
||||
|
||||
counts = {'PASS': 0, 'FAIL': 0, 'UNSUPPORTED': 0, 'RUNNING': 0}
|
||||
for _, _, status in rows:
|
||||
counts[status] = counts.get(status, 0) + 1
|
||||
|
||||
with open(out_file, 'w', encoding='utf-8') as out:
|
||||
out.write('## Case Summary\n\n')
|
||||
out.write(f"- Total: {len(rows)}\\n")
|
||||
out.write(f"- PASS: {counts.get('PASS', 0)}\\n")
|
||||
out.write(f"- FAIL: {counts.get('FAIL', 0)}\\n")
|
||||
out.write(f"- UNSUPPORTED: {counts.get('UNSUPPORTED', 0)}\\n")
|
||||
out.write('\\n')
|
||||
out.write('| Case | Name | Status |\\n')
|
||||
out.write('| --- | --- | --- |\\n')
|
||||
for case_id, name, status in rows:
|
||||
out.write(f'| {case_id} | {name} | {status} |\\n')
|
||||
PY
|
||||
{
|
||||
echo "# RustFS S3 compatibility test report"
|
||||
echo ""
|
||||
echo "- Run: ${{ github.server_url }}/${{ github.repository }}/actions/runs/${{ github.run_id }}"
|
||||
echo "- Trigger: ${{ github.event_name }}"
|
||||
echo "- Package: ${PACKAGE_SOURCE}"
|
||||
echo "- RustFS Version: ${RUSTFS_VERSION_INFO}"
|
||||
echo "- Test Step Outcome: ${{ steps.test.outcome }}"
|
||||
echo ""
|
||||
cat "${CASE_TABLE}" || true
|
||||
echo ""
|
||||
echo "## Log tail"
|
||||
echo '```text'
|
||||
@@ -203,92 +123,6 @@ jobs:
|
||||
} | tee "${REPORT_FILE}"
|
||||
cat "${REPORT_FILE}" >> "${GITHUB_STEP_SUMMARY}"
|
||||
|
||||
- name: Upload functional report to dashboard
|
||||
if: always()
|
||||
continue-on-error: true
|
||||
env:
|
||||
GH_TOKEN: ${{ env.PF_TESTING_GH_TOKEN }}
|
||||
REPORT_FILE: /tmp/rustfs-s3-compat-report.md
|
||||
SUITE: s3
|
||||
run: |
|
||||
set -euo pipefail
|
||||
if [ -z "${GH_TOKEN:-}" ]; then
|
||||
echo "PF_TESTING_GH_TOKEN is not configured; skipping dashboard upload"
|
||||
exit 0
|
||||
fi
|
||||
DATE="$(date -u +%Y-%m-%d)"
|
||||
REPORT_PATH="functional-reports/${SUITE}/${DATE}.md"
|
||||
CONTENT="$(python3 -c 'import base64,sys;print(base64.b64encode(open(sys.argv[1],"rb").read()).decode())' "${REPORT_FILE}")"
|
||||
SHA="$(gh api "repos/rustfs/dashboard/contents/${REPORT_PATH}" -q '.sha' 2>/dev/null || true)"
|
||||
if [ -n "${SHA}" ]; then
|
||||
jq -n --arg msg "report(${SUITE}): ${DATE}" --arg content "${CONTENT}" --arg sha "${SHA}" \
|
||||
'{message:$msg, content:$content, sha:$sha}' \
|
||||
| gh api --method PUT "repos/rustfs/dashboard/contents/${REPORT_PATH}" --input - >/dev/null
|
||||
else
|
||||
jq -n --arg msg "report(${SUITE}): ${DATE}" --arg content "${CONTENT}" \
|
||||
'{message:$msg, content:$content}' \
|
||||
| gh api --method PUT "repos/rustfs/dashboard/contents/${REPORT_PATH}" --input - >/dev/null
|
||||
fi
|
||||
|
||||
- name: File failure issue in rustfs/backlog
|
||||
if: ${{ always() && (failure() || steps.test.outcome == 'failure' || steps.test.outcome == 'cancelled') }}
|
||||
continue-on-error: true
|
||||
env:
|
||||
GH_TOKEN: ${{ secrets.PF_TESTING_GH_TOKEN }}
|
||||
SUITE: 's3'
|
||||
SUITE_LABEL: 'S3 compatibility'
|
||||
RUN_URL: ${{ github.server_url }}/${{ github.repository }}/actions/runs/${{ github.run_id }}
|
||||
REPORT_FILE: '/tmp/rustfs-s3-compat-report.md'
|
||||
LOG_FILE: '/tmp/rustfs-s3-compat.log'
|
||||
run: |
|
||||
set -euo pipefail
|
||||
if [ -z "${GH_TOKEN:-}" ]; then
|
||||
echo "PF_TESTING_GH_TOKEN is not configured; skipping backlog issue"
|
||||
exit 0
|
||||
fi
|
||||
TITLE="[functional][${SUITE}] ${SUITE_LABEL} suite failed (run ${GITHUB_RUN_ID})"
|
||||
EXISTING="$(gh issue list -R rustfs/backlog --state all \
|
||||
--search "in:title \"run ${GITHUB_RUN_ID}\"" \
|
||||
--json number --jq '.[].number' || true)"
|
||||
if [ -n "${EXISTING}" ]; then
|
||||
echo "backlog issue already exists for run ${GITHUB_RUN_ID}; skipping"
|
||||
exit 0
|
||||
fi
|
||||
redact() {
|
||||
sed -E \
|
||||
-e 's/(RUSTFS_(ACCESS_KEY|SECRET_KEY)[=: ]+)[^[:space:]]+/\1[REDACTED]/Ig' \
|
||||
-e 's/(Authorization:).*/\1 [REDACTED]/Ig' \
|
||||
-e 's/(X-Amz-Signature=)[^&[:space:]]+/\1[REDACTED]/Ig' \
|
||||
-e 's/^.*(password|secret|token)[=: ].*/[REDACTED SENSITIVE LINE]/Ig'
|
||||
}
|
||||
BODY_FILE="$(mktemp)"
|
||||
{
|
||||
echo "The **${SUITE_LABEL}** functional suite failed."
|
||||
echo ""
|
||||
echo "- Suite: \`${SUITE}\`"
|
||||
echo "- Run: ${RUN_URL}"
|
||||
echo "- Trigger: ${GITHUB_EVENT_NAME}"
|
||||
echo "- Date: $(date -u +%Y-%m-%d)"
|
||||
echo ""
|
||||
echo "## Report (errors and symptoms)"
|
||||
echo ""
|
||||
if [ -s "${REPORT_FILE}" ]; then
|
||||
redact < "${REPORT_FILE}"
|
||||
elif [ -s "${LOG_FILE:-}" ]; then
|
||||
echo "(report file missing; log tail below)"
|
||||
echo ""
|
||||
tail -n 200 "${LOG_FILE}" | redact
|
||||
else
|
||||
echo "(no report or log file was produced)"
|
||||
fi
|
||||
} | head -c 55000 > "${BODY_FILE}"
|
||||
gh label create functional-test -R rustfs/backlog --color d73a4a 2>/dev/null || true
|
||||
if ! gh issue create -R rustfs/backlog --title "${TITLE}" \
|
||||
--body-file "${BODY_FILE}" --label functional-test; then
|
||||
gh issue create -R rustfs/backlog --title "${TITLE}" --body-file "${BODY_FILE}"
|
||||
fi
|
||||
echo "filed backlog issue for suite ${SUITE}"
|
||||
|
||||
- name: Upload report and logs
|
||||
if: always()
|
||||
uses: actions/upload-artifact@b7c566a772e6b6bfb58ed0dc250532a479d7789f # v6
|
||||
@@ -318,24 +152,6 @@ jobs:
|
||||
'
|
||||
done
|
||||
|
||||
- name: "Continue functional chain (next: KMS)"
|
||||
# Only chain-triggered runs forward to the next suite; standalone
|
||||
# workflow_dispatch runs stop after their own cleanup.
|
||||
if: ${{ always() && github.event_name == 'repository_dispatch' }}
|
||||
continue-on-error: true
|
||||
env:
|
||||
GH_TOKEN: ${{ secrets.PF_TESTING_GH_TOKEN }}
|
||||
run: |
|
||||
set -euo pipefail
|
||||
if [ -z "${GH_TOKEN:-}" ]; then
|
||||
echo "PF_TESTING_GH_TOKEN is not configured; cannot dispatch the next suite" >&2
|
||||
exit 1
|
||||
fi
|
||||
echo "Dispatching next functional suite: KMS"
|
||||
gh api --method POST repos/rustfs/rustfs/dispatches \
|
||||
-f event_type='rustfs-chain-kms' \
|
||||
-F 'client_payload[from_suite]=s3'
|
||||
|
||||
- name: Notify on failure
|
||||
if: failure()
|
||||
run: |
|
||||
|
||||
@@ -1,300 +0,0 @@
|
||||
# Copyright 2024 RustFS Team
|
||||
#
|
||||
# Licensed under the Apache License, Version 2.0 (the "License");
|
||||
# you may not use this file except in compliance with the License.
|
||||
# You may obtain a copy of the License at
|
||||
#
|
||||
# http://www.apache.org/licenses/LICENSE-2.0
|
||||
#
|
||||
# Unless required by applicable law or agreed to in writing, software
|
||||
# distributed under the License is distributed on an "AS IS" BASIS,
|
||||
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||
# See the License for the specific language governing permissions and
|
||||
# limitations under the License.
|
||||
|
||||
name: RustFS Security Test
|
||||
|
||||
on:
|
||||
workflow_dispatch:
|
||||
inputs:
|
||||
rustfs_version:
|
||||
description: 'RustFS release tag to test (e.g. 1.0.0-rc.4-preview.1)'
|
||||
required: false
|
||||
default: '1.0.0-rc.4-preview.1'
|
||||
package_url:
|
||||
description: 'Direct .deb URL (nightly/R2/dev). Overrides rustfs_version.'
|
||||
required: false
|
||||
type: string
|
||||
topology:
|
||||
description: 'Topology to run (all = SNSD, SNMD, MNMD)'
|
||||
type: choice
|
||||
options:
|
||||
- all
|
||||
- single-single
|
||||
- single-multi
|
||||
- multi-multi
|
||||
default: all
|
||||
oidc_live:
|
||||
description: 'Run the live Keycloak OIDC/SSO gate as part of the suite'
|
||||
type: boolean
|
||||
default: true
|
||||
cleanup_before:
|
||||
description: 'Reset the nodes before the test (DESTROYS existing data/config)'
|
||||
type: boolean
|
||||
default: true
|
||||
cleanup_after:
|
||||
description: 'Reset the nodes after the test (DESTROYS test data/config)'
|
||||
type: boolean
|
||||
default: true
|
||||
repository_dispatch:
|
||||
# Chain handoff: dispatched when the pool expansion suite finishes (last link).
|
||||
types: [rustfs-chain-security]
|
||||
|
||||
permissions:
|
||||
contents: read
|
||||
|
||||
# The security suite uses the same shared VMs as the other functional tests,
|
||||
# so it must serialize with them instead of running in parallel.
|
||||
concurrency:
|
||||
group: rustfs-shared-functional-tests
|
||||
cancel-in-progress: false
|
||||
|
||||
defaults:
|
||||
run:
|
||||
shell: bash
|
||||
|
||||
env:
|
||||
RUSTFS_ACCESS_KEY: ${{ secrets.RUSTFS_ACCESS_KEY }}
|
||||
RUSTFS_SECRET_KEY: ${{ secrets.RUSTFS_SECRET_KEY }}
|
||||
RUSTFS_NODES: ${{ secrets.RUSTFS_NODES || vars.RUSTFS_NODES }}
|
||||
RUSTFS_SSH_USER: ${{ secrets.RUSTFS_SSH_USER || vars.RUSTFS_SSH_USER }}
|
||||
RUSTFS_NIGHTLY_PACKAGE_URL: ${{ vars.RUSTFS_NIGHTLY_PACKAGE_URL || 'https://dl.rustfs.com/artifacts/rustfs/packages/nightly/rustfs-nightly-latest.deb' }}
|
||||
PF_TESTING_GH_TOKEN: ${{ secrets.PF_TESTING_GH_TOKEN }}
|
||||
|
||||
jobs:
|
||||
security-test:
|
||||
runs-on: smoke-testing
|
||||
continue-on-error: true
|
||||
timeout-minutes: 360
|
||||
if: ${{ github.event_name == 'workflow_dispatch' || github.event_name == 'repository_dispatch' }}
|
||||
steps:
|
||||
# auto-testing is private: clone it with the dedicated PF token (not
|
||||
# GITHUB_TOKEN) and retry transient GitHub/network failures.
|
||||
- name: Checkout auto-testing scripts (with retry)
|
||||
env:
|
||||
GH_TOKEN: ${{ secrets.PF_TESTING_GH_TOKEN }}
|
||||
run: |
|
||||
set -euo pipefail
|
||||
rm -rf auto-testing
|
||||
for attempt in 1 2 3 4 5; do
|
||||
if gh repo clone rustfs/auto-testing auto-testing -- --depth 1 --quiet; then
|
||||
echo "auto-testing cloned (attempt ${attempt})"
|
||||
exit 0
|
||||
fi
|
||||
rm -rf auto-testing
|
||||
echo "clone attempt ${attempt} failed; retrying in $((attempt * 15))s" >&2
|
||||
sleep $((attempt * 15))
|
||||
done
|
||||
echo "ERROR: unable to clone rustfs/auto-testing after 5 attempts" >&2
|
||||
exit 1
|
||||
|
||||
- name: Checkout repository (for the OIDC live gate script)
|
||||
uses: actions/checkout@9c091bb21b7c1c1d1991bb908d89e4e9dddfe3e0 # v7
|
||||
with:
|
||||
persist-credentials: false
|
||||
|
||||
- name: Show environment
|
||||
run: |
|
||||
uname -a
|
||||
jq --version
|
||||
openssl version
|
||||
aws --version || true
|
||||
docker --version || true
|
||||
df -h /data | tail -1
|
||||
|
||||
- name: Cleanup environment (before)
|
||||
if: ${{ inputs.cleanup_before != 'false' || github.event_name != 'workflow_dispatch' }}
|
||||
run: |
|
||||
set -euo pipefail
|
||||
read -r -a NODES <<< "${RUSTFS_NODES:-vm000 vm001 vm002}"
|
||||
SSH_USER="${RUSTFS_SSH_USER:-azureuser}"
|
||||
for node in "${NODES[@]}"; do
|
||||
ssh -o BatchMode=yes -o ConnectTimeout=10 -o StrictHostKeyChecking=accept-new "${SSH_USER}@${node}" '
|
||||
set -euo pipefail
|
||||
SUDO=""; [ "$(id -u)" -ne 0 ] && SUDO="sudo -n"
|
||||
${SUDO} systemctl stop rustfs 2>/dev/null || true
|
||||
if ${SUDO} dpkg -l rustfs 2>/dev/null | grep -q "^ii"; then
|
||||
${SUDO} dpkg -P rustfs
|
||||
fi
|
||||
for i in 1 2 3 4; do ${SUDO} rm -rf /data/rustfs${i}/mnmd; done
|
||||
${SUDO} rm -rf /var/log/rustfs /var/lib/rustfs/kms
|
||||
'
|
||||
done
|
||||
|
||||
- name: Run security suite
|
||||
id: test
|
||||
continue-on-error: true
|
||||
env:
|
||||
REPORT_FILE: /tmp/rustfs-security-report.md
|
||||
RUSTFS_SECURITY_OIDC_LIVE_SCRIPT: ${{ github.workspace }}/scripts/test/oidc_keycloak_live.sh
|
||||
run: |
|
||||
set -euo pipefail
|
||||
chmod +x auto-testing/rustfs-security-test.sh
|
||||
PACKAGE_URL='${{ inputs.package_url }}'
|
||||
RUSTFS_VERSION='${{ inputs.rustfs_version }}'
|
||||
TOPOLOGY='${{ inputs.topology }}'
|
||||
ARGS=(-y)
|
||||
if [ "${TOPOLOGY}" = "all" ] || [ -z "${TOPOLOGY}" ] || [ "${TOPOLOGY}" = "null" ]; then
|
||||
ARGS+=(--all-topologies)
|
||||
else
|
||||
ARGS+=(--topology "${TOPOLOGY}")
|
||||
fi
|
||||
if [ "${{ inputs.oidc_live }}" = "true" ] || [ "${{ github.event_name }}" != "workflow_dispatch" ]; then
|
||||
ARGS+=(--oidc-live)
|
||||
fi
|
||||
if [ -n "${PACKAGE_URL}" ]; then
|
||||
ARGS+=(--package-url "${PACKAGE_URL}")
|
||||
elif [ -n "${RUSTFS_VERSION}" ] && [ "${RUSTFS_VERSION}" != "null" ]; then
|
||||
ARGS+=(--version "${RUSTFS_VERSION}")
|
||||
else
|
||||
ARGS+=(--package-url "${RUSTFS_NIGHTLY_PACKAGE_URL}")
|
||||
fi
|
||||
./auto-testing/rustfs-security-test.sh "${ARGS[@]}"
|
||||
|
||||
- name: Generate report
|
||||
if: always()
|
||||
run: |
|
||||
set -euo pipefail
|
||||
if [ ! -f /tmp/rustfs-security-report.md ]; then
|
||||
{
|
||||
echo "# RustFS security test report"
|
||||
echo ""
|
||||
echo "- Run: ${{ github.server_url }}/${{ github.repository }}/actions/runs/${{ github.run_id }}"
|
||||
echo "- Trigger: ${{ github.event_name }}"
|
||||
echo "- Test Step Outcome: failure (suite did not produce a report)"
|
||||
} > /tmp/rustfs-security-report.md
|
||||
fi
|
||||
cat /tmp/rustfs-security-report.md >> "${GITHUB_STEP_SUMMARY}"
|
||||
|
||||
- name: Upload functional report to dashboard
|
||||
if: always()
|
||||
continue-on-error: true
|
||||
env:
|
||||
GH_TOKEN: ${{ env.PF_TESTING_GH_TOKEN }}
|
||||
REPORT_FILE: /tmp/rustfs-security-report.md
|
||||
SUITE: security
|
||||
run: |
|
||||
set -euo pipefail
|
||||
if [ -z "${GH_TOKEN:-}" ]; then
|
||||
echo "PF_TESTING_GH_TOKEN is not configured; skipping dashboard upload"
|
||||
exit 0
|
||||
fi
|
||||
DATE="$(date -u +%Y-%m-%d)"
|
||||
REPORT_PATH="functional-reports/${SUITE}/${DATE}.md"
|
||||
CONTENT="$(python3 -c 'import base64,sys;print(base64.b64encode(open(sys.argv[1],"rb").read()).decode())' "${REPORT_FILE}")"
|
||||
SHA="$(gh api "repos/rustfs/dashboard/contents/${REPORT_PATH}" -q '.sha' 2>/dev/null || true)"
|
||||
if [ -n "${SHA}" ]; then
|
||||
jq -n --arg msg "report(${SUITE}): ${DATE}" --arg content "${CONTENT}" --arg sha "${SHA}" \
|
||||
'{message:$msg, content:$content, sha:$sha}' \
|
||||
| gh api --method PUT "repos/rustfs/dashboard/contents/${REPORT_PATH}" --input - >/dev/null
|
||||
else
|
||||
jq -n --arg msg "report(${SUITE}): ${DATE}" --arg content "${CONTENT}" \
|
||||
'{message:$msg, content:$content}' \
|
||||
| gh api --method PUT "repos/rustfs/dashboard/contents/${REPORT_PATH}" --input - >/dev/null
|
||||
fi
|
||||
|
||||
- name: File failure issue in rustfs/backlog
|
||||
if: ${{ always() && (failure() || steps.test.outcome == 'failure' || steps.test.outcome == 'cancelled') }}
|
||||
continue-on-error: true
|
||||
env:
|
||||
GH_TOKEN: ${{ secrets.PF_TESTING_GH_TOKEN }}
|
||||
SUITE: 'security'
|
||||
SUITE_LABEL: 'Security'
|
||||
RUN_URL: ${{ github.server_url }}/${{ github.repository }}/actions/runs/${{ github.run_id }}
|
||||
REPORT_FILE: '/tmp/rustfs-security-report.md'
|
||||
LOG_FILE: ''
|
||||
run: |
|
||||
set -euo pipefail
|
||||
if [ -z "${GH_TOKEN:-}" ]; then
|
||||
echo "PF_TESTING_GH_TOKEN is not configured; skipping backlog issue"
|
||||
exit 0
|
||||
fi
|
||||
TITLE="[functional][${SUITE}] ${SUITE_LABEL} suite failed (run ${GITHUB_RUN_ID})"
|
||||
EXISTING="$(gh issue list -R rustfs/backlog --state all \
|
||||
--search "in:title \"run ${GITHUB_RUN_ID}\"" \
|
||||
--json number --jq '.[].number' || true)"
|
||||
if [ -n "${EXISTING}" ]; then
|
||||
echo "backlog issue already exists for run ${GITHUB_RUN_ID}; skipping"
|
||||
exit 0
|
||||
fi
|
||||
redact() {
|
||||
sed -E \
|
||||
-e 's/(RUSTFS_(ACCESS_KEY|SECRET_KEY)[=: ]+)[^[:space:]]+/\1[REDACTED]/Ig' \
|
||||
-e 's/(Authorization:).*/\1 [REDACTED]/Ig' \
|
||||
-e 's/(X-Amz-Signature=)[^&[:space:]]+/\1[REDACTED]/Ig' \
|
||||
-e 's/^.*(password|secret|token)[=: ].*/[REDACTED SENSITIVE LINE]/Ig'
|
||||
}
|
||||
BODY_FILE="$(mktemp)"
|
||||
{
|
||||
echo "The **${SUITE_LABEL}** functional suite failed."
|
||||
echo ""
|
||||
echo "- Suite: \`${SUITE}\`"
|
||||
echo "- Run: ${RUN_URL}"
|
||||
echo "- Trigger: ${GITHUB_EVENT_NAME}"
|
||||
echo "- Date: $(date -u +%Y-%m-%d)"
|
||||
echo ""
|
||||
echo "## Report (errors and symptoms)"
|
||||
echo ""
|
||||
if [ -s "${REPORT_FILE}" ]; then
|
||||
redact < "${REPORT_FILE}"
|
||||
elif [ -s "${LOG_FILE:-}" ]; then
|
||||
echo "(report file missing; log tail below)"
|
||||
echo ""
|
||||
tail -n 200 "${LOG_FILE}" | redact
|
||||
else
|
||||
echo "(no report or log file was produced)"
|
||||
fi
|
||||
} | head -c 55000 > "${BODY_FILE}"
|
||||
gh label create functional-test -R rustfs/backlog --color d73a4a 2>/dev/null || true
|
||||
if ! gh issue create -R rustfs/backlog --title "${TITLE}" \
|
||||
--body-file "${BODY_FILE}" --label functional-test; then
|
||||
gh issue create -R rustfs/backlog --title "${TITLE}" --body-file "${BODY_FILE}"
|
||||
fi
|
||||
echo "filed backlog issue for suite ${SUITE}"
|
||||
|
||||
- name: Upload report and logs
|
||||
if: always()
|
||||
uses: actions/upload-artifact@b7c566a772e6b6bfb58ed0dc250532a479d7789f # v6
|
||||
with:
|
||||
name: rustfs-security-test-${{ github.run_id }}
|
||||
path: |
|
||||
/tmp/rustfs-security-report.md
|
||||
/tmp/rustfs-security.*/*
|
||||
if-no-files-found: ignore
|
||||
retention-days: 3
|
||||
|
||||
- name: Cleanup environment (after)
|
||||
if: ${{ always() && (inputs.cleanup_after != 'false' || github.event_name != 'workflow_dispatch') }}
|
||||
run: |
|
||||
set -euo pipefail
|
||||
read -r -a NODES <<< "${RUSTFS_NODES:-vm000 vm001 vm002}"
|
||||
SSH_USER="${RUSTFS_SSH_USER:-azureuser}"
|
||||
for node in "${NODES[@]}"; do
|
||||
ssh -o BatchMode=yes -o ConnectTimeout=10 -o StrictHostKeyChecking=accept-new "${SSH_USER}@${node}" '
|
||||
set -euo pipefail
|
||||
SUDO=""; [ "$(id -u)" -ne 0 ] && SUDO="sudo -n"
|
||||
${SUDO} systemctl stop rustfs 2>/dev/null || true
|
||||
if ${SUDO} dpkg -l rustfs 2>/dev/null | grep -q "^ii"; then
|
||||
${SUDO} dpkg -P rustfs
|
||||
fi
|
||||
for i in 1 2 3 4; do ${SUDO} rm -rf /data/rustfs${i}/mnmd; done
|
||||
${SUDO} rm -rf /var/log/rustfs /var/lib/rustfs/kms
|
||||
'
|
||||
done
|
||||
|
||||
- name: Notify on failure
|
||||
if: failure()
|
||||
run: |
|
||||
echo "RustFS security test failed"
|
||||
echo "Package source: ${{ inputs.package_url || 'nightly (R2 latest)' }}"
|
||||
echo "See the uploaded report and logs for details."
|
||||
@@ -1,358 +0,0 @@
|
||||
name: RustFS Storage Engine Test
|
||||
|
||||
on:
|
||||
workflow_dispatch:
|
||||
inputs:
|
||||
rustfs_version:
|
||||
description: 'RustFS release tag to test (e.g. 1.0.0-rc.4-preview.1)'
|
||||
required: false
|
||||
default: '1.0.0-rc.4-preview.1'
|
||||
package_url:
|
||||
description: 'Direct .deb URL (nightly/R2/dev). Overrides rustfs_version.'
|
||||
required: false
|
||||
type: string
|
||||
topology:
|
||||
description: 'Topology to run (all = SNSD, SNMD, MNMD)'
|
||||
type: choice
|
||||
options:
|
||||
- all
|
||||
- single-single
|
||||
- single-multi
|
||||
- multi-multi
|
||||
default: all
|
||||
repository_dispatch:
|
||||
# Chain handoff: dispatched when the tier suite finishes.
|
||||
types: [rustfs-chain-storage]
|
||||
|
||||
permissions:
|
||||
contents: read
|
||||
|
||||
concurrency:
|
||||
group: rustfs-shared-functional-tests
|
||||
cancel-in-progress: false
|
||||
|
||||
defaults:
|
||||
run:
|
||||
shell: bash
|
||||
|
||||
env:
|
||||
RUSTFS_ACCESS_KEY: ${{ secrets.RUSTFS_ACCESS_KEY }}
|
||||
RUSTFS_SECRET_KEY: ${{ secrets.RUSTFS_SECRET_KEY }}
|
||||
RUSTFS_NODES: ${{ secrets.RUSTFS_NODES || vars.RUSTFS_NODES }}
|
||||
RUSTFS_SSH_USER: ${{ secrets.RUSTFS_SSH_USER || vars.RUSTFS_SSH_USER }}
|
||||
RUSTFS_NIGHTLY_PACKAGE_URL: ${{ vars.RUSTFS_NIGHTLY_PACKAGE_URL || 'https://dl.rustfs.com/artifacts/rustfs/packages/nightly/rustfs-nightly-latest.deb' }}
|
||||
PF_TESTING_GH_TOKEN: ${{ secrets.PF_TESTING_GH_TOKEN }}
|
||||
|
||||
jobs:
|
||||
storage-test:
|
||||
runs-on: smoke-testing
|
||||
continue-on-error: true
|
||||
timeout-minutes: 360
|
||||
if: ${{ github.event_name == 'workflow_dispatch' || github.event_name == 'repository_dispatch' }}
|
||||
steps:
|
||||
# auto-testing is private: clone it with the dedicated PF token (not
|
||||
# GITHUB_TOKEN) and retry transient GitHub/network failures.
|
||||
- name: Checkout auto-testing scripts (with retry)
|
||||
env:
|
||||
GH_TOKEN: ${{ secrets.PF_TESTING_GH_TOKEN }}
|
||||
run: |
|
||||
set -euo pipefail
|
||||
rm -rf auto-testing
|
||||
for attempt in 1 2 3 4 5; do
|
||||
if gh repo clone rustfs/auto-testing auto-testing -- --depth 1 --quiet; then
|
||||
echo "auto-testing cloned (attempt ${attempt})"
|
||||
exit 0
|
||||
fi
|
||||
rm -rf auto-testing
|
||||
echo "clone attempt ${attempt} failed; retrying in $((attempt * 15))s" >&2
|
||||
sleep $((attempt * 15))
|
||||
done
|
||||
echo "ERROR: unable to clone rustfs/auto-testing after 5 attempts" >&2
|
||||
exit 1
|
||||
|
||||
- name: Show environment
|
||||
run: |
|
||||
uname -a
|
||||
jq --version
|
||||
openssl version
|
||||
df -h /data | tail -1
|
||||
|
||||
- name: Cleanup environment (before)
|
||||
run: |
|
||||
set -euo pipefail
|
||||
read -r -a NODES <<< "${RUSTFS_NODES:-vm000 vm001 vm002}"
|
||||
SSH_USER="${RUSTFS_SSH_USER:-azureuser}"
|
||||
for node in "${NODES[@]}"; do
|
||||
ssh -o BatchMode=yes -o ConnectTimeout=10 -o StrictHostKeyChecking=accept-new "${SSH_USER}@${node}" '
|
||||
set -euo pipefail
|
||||
SUDO=""; [ "$(id -u)" -ne 0 ] && SUDO="sudo -n"
|
||||
${SUDO} systemctl stop rustfs 2>/dev/null || true
|
||||
if ${SUDO} dpkg -l rustfs 2>/dev/null | grep -q "^ii"; then
|
||||
${SUDO} dpkg -P rustfs
|
||||
fi
|
||||
for i in 1 2 3 4; do ${SUDO} rm -rf /data/rustfs${i}/mnmd; done
|
||||
${SUDO} rm -rf /var/log/rustfs /var/lib/rustfs/kms /var/lib/rustfs/kms-backup
|
||||
'
|
||||
done
|
||||
|
||||
- name: Run storage engine suite
|
||||
id: test
|
||||
continue-on-error: true
|
||||
env:
|
||||
LOG_FILE: /tmp/rustfs-storage.log
|
||||
run: |
|
||||
set -euo pipefail
|
||||
chmod +x auto-testing/rustfs-storage-test.sh
|
||||
PACKAGE_URL='${{ inputs.package_url }}'
|
||||
RUSTFS_VERSION='${{ inputs.rustfs_version }}'
|
||||
TOPOLOGY='${{ inputs.topology }}'
|
||||
ARGS=(-y --log-file "${LOG_FILE}")
|
||||
if [ "${TOPOLOGY}" = "all" ] || [ -z "${TOPOLOGY}" ] || [ "${TOPOLOGY}" = "null" ]; then
|
||||
ARGS+=(--all-topologies)
|
||||
else
|
||||
ARGS+=(--topology "${TOPOLOGY}")
|
||||
fi
|
||||
if [ -n "${PACKAGE_URL}" ]; then
|
||||
ARGS+=(--package-url "${PACKAGE_URL}")
|
||||
elif [ -n "${RUSTFS_VERSION}" ]; then
|
||||
ARGS+=(--version "${RUSTFS_VERSION}")
|
||||
else
|
||||
ARGS+=(--package-url "${RUSTFS_NIGHTLY_PACKAGE_URL}")
|
||||
fi
|
||||
./auto-testing/rustfs-storage-test.sh "${ARGS[@]}"
|
||||
|
||||
- name: Generate report
|
||||
if: always()
|
||||
env:
|
||||
LOG_FILE: /tmp/rustfs-storage.log
|
||||
REPORT_FILE: /tmp/rustfs-storage-report.md
|
||||
run: |
|
||||
set -euo pipefail
|
||||
PACKAGE_URL='${{ inputs.package_url }}'
|
||||
RUSTFS_VERSION='${{ inputs.rustfs_version }}'
|
||||
if [ -n "${PACKAGE_URL}" ]; then
|
||||
PACKAGE_SOURCE="${PACKAGE_URL}"
|
||||
elif [ -n "${RUSTFS_VERSION}" ]; then
|
||||
PACKAGE_SOURCE="version ${RUSTFS_VERSION}"
|
||||
else
|
||||
PACKAGE_SOURCE="${RUSTFS_NIGHTLY_PACKAGE_URL}"
|
||||
fi
|
||||
read -r -a NODES <<< "${RUSTFS_NODES:-vm000 vm001 vm002}"
|
||||
SSH_USER="${RUSTFS_SSH_USER:-azureuser}"
|
||||
RUSTFS_VERSION_INFO="N/A"
|
||||
if [ "${#NODES[@]}" -gt 0 ]; then
|
||||
DETECTED_VERSION="$(ssh -o BatchMode=yes -o ConnectTimeout=10 -o StrictHostKeyChecking=accept-new \
|
||||
"${SSH_USER}@${NODES[0]}" 'rustfs --version' 2>/dev/null | tr -d '\r' | head -n 1 || true)"
|
||||
if [ -n "${DETECTED_VERSION}" ]; then
|
||||
RUSTFS_VERSION_INFO="${DETECTED_VERSION}"
|
||||
fi
|
||||
fi
|
||||
CASE_TABLE="/tmp/rustfs-storage-cases.md"
|
||||
python3 - "${LOG_FILE}" "${CASE_TABLE}" <<'PY'
|
||||
import re
|
||||
import sys
|
||||
|
||||
log_file, out_file = sys.argv[1], sys.argv[2]
|
||||
ansi = re.compile(r'\x1b\[[0-9;]*m')
|
||||
start_re = re.compile(r'^---\s+([A-Z0-9]+-[0-9]+)\s+(.+?)\s+---$')
|
||||
done_re = re.compile(r'^\[(PASS|FAIL|UNSUPPORTED)\]\s+([A-Z0-9]+-[0-9]+)\b')
|
||||
|
||||
rows = []
|
||||
index = {}
|
||||
current = None
|
||||
try:
|
||||
with open(log_file, 'r', encoding='utf-8', errors='replace') as fh:
|
||||
for raw in fh:
|
||||
line = ansi.sub('', raw).strip()
|
||||
m = start_re.match(line)
|
||||
if m:
|
||||
case_id, name = m.group(1), m.group(2)
|
||||
current = case_id
|
||||
if case_id not in index:
|
||||
index[case_id] = len(rows)
|
||||
rows.append([case_id, name, 'RUNNING'])
|
||||
continue
|
||||
m = done_re.match(line)
|
||||
if m:
|
||||
status, case_id = m.group(1), m.group(2)
|
||||
if case_id in index:
|
||||
rows[index[case_id]][2] = status
|
||||
else:
|
||||
rows.append([case_id, case_id, status])
|
||||
index[case_id] = len(rows) - 1
|
||||
current = None
|
||||
except FileNotFoundError:
|
||||
rows = []
|
||||
|
||||
counts = {'PASS': 0, 'FAIL': 0, 'UNSUPPORTED': 0, 'RUNNING': 0}
|
||||
for _, _, status in rows:
|
||||
counts[status] = counts.get(status, 0) + 1
|
||||
|
||||
with open(out_file, 'w', encoding='utf-8') as out:
|
||||
out.write('## Case Summary\n\n')
|
||||
out.write(f"- Total: {len(rows)}\\n")
|
||||
out.write(f"- PASS: {counts.get('PASS', 0)}\\n")
|
||||
out.write(f"- FAIL: {counts.get('FAIL', 0)}\\n")
|
||||
out.write(f"- UNSUPPORTED: {counts.get('UNSUPPORTED', 0)}\\n")
|
||||
out.write('\\n')
|
||||
out.write('| Case | Name | Status |\\n')
|
||||
out.write('| --- | --- | --- |\\n')
|
||||
for case_id, name, status in rows:
|
||||
out.write(f'| {case_id} | {name} | {status} |\\n')
|
||||
PY
|
||||
{
|
||||
echo "# RustFS storage engine test report"
|
||||
echo ""
|
||||
echo "- Run: ${{ github.server_url }}/${{ github.repository }}/actions/runs/${{ github.run_id }}"
|
||||
echo "- Trigger: ${{ github.event_name }}"
|
||||
echo "- Package: ${PACKAGE_SOURCE}"
|
||||
echo "- RustFS Version: ${RUSTFS_VERSION_INFO}"
|
||||
echo "- Test Step Outcome: ${{ steps.test.outcome }}"
|
||||
echo ""
|
||||
cat "${CASE_TABLE}" || true
|
||||
echo ""
|
||||
echo "## Log tail"
|
||||
echo '```text'
|
||||
tail -n 200 "${LOG_FILE}" || true
|
||||
echo '```'
|
||||
} | tee "${REPORT_FILE}"
|
||||
cat "${REPORT_FILE}" >> "${GITHUB_STEP_SUMMARY}"
|
||||
|
||||
- name: Upload functional report to dashboard
|
||||
if: always()
|
||||
continue-on-error: true
|
||||
env:
|
||||
GH_TOKEN: ${{ env.PF_TESTING_GH_TOKEN }}
|
||||
REPORT_FILE: /tmp/rustfs-storage-report.md
|
||||
SUITE: storage
|
||||
run: |
|
||||
set -euo pipefail
|
||||
if [ -z "${GH_TOKEN:-}" ]; then
|
||||
echo "PF_TESTING_GH_TOKEN is not configured; skipping dashboard upload"
|
||||
exit 0
|
||||
fi
|
||||
DATE="$(date -u +%Y-%m-%d)"
|
||||
REPORT_PATH="functional-reports/${SUITE}/${DATE}.md"
|
||||
CONTENT="$(python3 -c 'import base64,sys;print(base64.b64encode(open(sys.argv[1],"rb").read()).decode())' "${REPORT_FILE}")"
|
||||
SHA="$(gh api "repos/rustfs/dashboard/contents/${REPORT_PATH}" -q '.sha' 2>/dev/null || true)"
|
||||
if [ -n "${SHA}" ]; then
|
||||
jq -n --arg msg "report(${SUITE}): ${DATE}" --arg content "${CONTENT}" --arg sha "${SHA}" \
|
||||
'{message:$msg, content:$content, sha:$sha}' \
|
||||
| gh api --method PUT "repos/rustfs/dashboard/contents/${REPORT_PATH}" --input - >/dev/null
|
||||
else
|
||||
jq -n --arg msg "report(${SUITE}): ${DATE}" --arg content "${CONTENT}" \
|
||||
'{message:$msg, content:$content}' \
|
||||
| gh api --method PUT "repos/rustfs/dashboard/contents/${REPORT_PATH}" --input - >/dev/null
|
||||
fi
|
||||
|
||||
- name: File failure issue in rustfs/backlog
|
||||
if: ${{ always() && (failure() || steps.test.outcome == 'failure' || steps.test.outcome == 'cancelled') }}
|
||||
continue-on-error: true
|
||||
env:
|
||||
GH_TOKEN: ${{ secrets.PF_TESTING_GH_TOKEN }}
|
||||
SUITE: 'storage'
|
||||
SUITE_LABEL: 'Storage engine'
|
||||
RUN_URL: ${{ github.server_url }}/${{ github.repository }}/actions/runs/${{ github.run_id }}
|
||||
REPORT_FILE: '/tmp/rustfs-storage-report.md'
|
||||
LOG_FILE: '/tmp/rustfs-storage.log'
|
||||
run: |
|
||||
set -euo pipefail
|
||||
if [ -z "${GH_TOKEN:-}" ]; then
|
||||
echo "PF_TESTING_GH_TOKEN is not configured; skipping backlog issue"
|
||||
exit 0
|
||||
fi
|
||||
TITLE="[functional][${SUITE}] ${SUITE_LABEL} suite failed (run ${GITHUB_RUN_ID})"
|
||||
EXISTING="$(gh issue list -R rustfs/backlog --state all \
|
||||
--search "in:title \"run ${GITHUB_RUN_ID}\"" \
|
||||
--json number --jq '.[].number' || true)"
|
||||
if [ -n "${EXISTING}" ]; then
|
||||
echo "backlog issue already exists for run ${GITHUB_RUN_ID}; skipping"
|
||||
exit 0
|
||||
fi
|
||||
redact() {
|
||||
sed -E \
|
||||
-e 's/(RUSTFS_(ACCESS_KEY|SECRET_KEY)[=: ]+)[^[:space:]]+/\1[REDACTED]/Ig' \
|
||||
-e 's/(Authorization:).*/\1 [REDACTED]/Ig' \
|
||||
-e 's/(X-Amz-Signature=)[^&[:space:]]+/\1[REDACTED]/Ig' \
|
||||
-e 's/^.*(password|secret|token)[=: ].*/[REDACTED SENSITIVE LINE]/Ig'
|
||||
}
|
||||
BODY_FILE="$(mktemp)"
|
||||
{
|
||||
echo "The **${SUITE_LABEL}** functional suite failed."
|
||||
echo ""
|
||||
echo "- Suite: \`${SUITE}\`"
|
||||
echo "- Run: ${RUN_URL}"
|
||||
echo "- Trigger: ${GITHUB_EVENT_NAME}"
|
||||
echo "- Date: $(date -u +%Y-%m-%d)"
|
||||
echo ""
|
||||
echo "## Report (errors and symptoms)"
|
||||
echo ""
|
||||
if [ -s "${REPORT_FILE}" ]; then
|
||||
redact < "${REPORT_FILE}"
|
||||
elif [ -s "${LOG_FILE:-}" ]; then
|
||||
echo "(report file missing; log tail below)"
|
||||
echo ""
|
||||
tail -n 200 "${LOG_FILE}" | redact
|
||||
else
|
||||
echo "(no report or log file was produced)"
|
||||
fi
|
||||
} | head -c 55000 > "${BODY_FILE}"
|
||||
gh label create functional-test -R rustfs/backlog --color d73a4a 2>/dev/null || true
|
||||
if ! gh issue create -R rustfs/backlog --title "${TITLE}" \
|
||||
--body-file "${BODY_FILE}" --label functional-test; then
|
||||
gh issue create -R rustfs/backlog --title "${TITLE}" --body-file "${BODY_FILE}"
|
||||
fi
|
||||
echo "filed backlog issue for suite ${SUITE}"
|
||||
|
||||
- name: Upload report and logs
|
||||
if: always()
|
||||
uses: actions/upload-artifact@b7c566a772e6b6bfb58ed0dc250532a479d7789f # v6
|
||||
with:
|
||||
name: rustfs-storage-${{ github.run_id }}
|
||||
path: |
|
||||
/tmp/rustfs-storage.log
|
||||
/tmp/rustfs-storage-report.md
|
||||
if-no-files-found: warn
|
||||
|
||||
- name: Cleanup environment (after)
|
||||
if: always()
|
||||
run: |
|
||||
set -euo pipefail
|
||||
read -r -a NODES <<< "${RUSTFS_NODES:-vm000 vm001 vm002}"
|
||||
SSH_USER="${RUSTFS_SSH_USER:-azureuser}"
|
||||
for node in "${NODES[@]}"; do
|
||||
ssh -o BatchMode=yes -o ConnectTimeout=10 -o StrictHostKeyChecking=accept-new "${SSH_USER}@${node}" '
|
||||
set -euo pipefail
|
||||
SUDO=""; [ "$(id -u)" -ne 0 ] && SUDO="sudo -n"
|
||||
${SUDO} systemctl stop rustfs 2>/dev/null || true
|
||||
if ${SUDO} dpkg -l rustfs 2>/dev/null | grep -q "^ii"; then
|
||||
${SUDO} dpkg -P rustfs
|
||||
fi
|
||||
for i in 1 2 3 4; do ${SUDO} rm -rf /data/rustfs${i}/mnmd; done
|
||||
${SUDO} rm -rf /var/log/rustfs
|
||||
'
|
||||
done
|
||||
|
||||
- name: "Continue functional chain (next: Heal)"
|
||||
# Only chain-triggered runs forward to the next suite; standalone
|
||||
# workflow_dispatch runs stop after their own cleanup.
|
||||
if: ${{ always() && github.event_name == 'repository_dispatch' }}
|
||||
continue-on-error: true
|
||||
env:
|
||||
GH_TOKEN: ${{ secrets.PF_TESTING_GH_TOKEN }}
|
||||
run: |
|
||||
set -euo pipefail
|
||||
if [ -z "${GH_TOKEN:-}" ]; then
|
||||
echo "PF_TESTING_GH_TOKEN is not configured; cannot dispatch the next suite" >&2
|
||||
exit 1
|
||||
fi
|
||||
echo "Dispatching next functional suite: Heal"
|
||||
gh api --method POST repos/rustfs/rustfs/dispatches \
|
||||
-f event_type='rustfs-chain-heal' \
|
||||
-F 'client_payload[from_suite]=storage'
|
||||
|
||||
- name: Notify on failure
|
||||
if: failure()
|
||||
run: |
|
||||
echo "RustFS storage engine suite failed"
|
||||
echo "See the uploaded report and log artifacts for details."
|
||||
@@ -11,22 +11,10 @@ on:
|
||||
description: 'Direct .deb URL (nightly/R2/dev). Overrides rustfs_version.'
|
||||
required: false
|
||||
type: string
|
||||
package_sha256:
|
||||
description: 'Optional SHA-256 for package_url; mismatch is an infrastructure failure.'
|
||||
required: false
|
||||
type: string
|
||||
rc_sha256:
|
||||
description: 'Optional SHA-256 for the preinstalled rc binary; mismatch is an infrastructure failure.'
|
||||
required: false
|
||||
type: string
|
||||
force_case_failure:
|
||||
description: 'Diagnostic only: rewrite single-single/TIER-101 to FAIL after execution to verify artifact and final-gate behavior.'
|
||||
required: false
|
||||
default: false
|
||||
type: boolean
|
||||
repository_dispatch:
|
||||
# Chain handoff: dispatched when the KMS suite finishes.
|
||||
types: [rustfs-chain-tier]
|
||||
workflow_run:
|
||||
# Strict shared-environment order: run after KMS test succeeds.
|
||||
workflows: ["RustFS KMS Test"]
|
||||
types: [completed]
|
||||
|
||||
permissions:
|
||||
contents: read
|
||||
@@ -45,90 +33,22 @@ env:
|
||||
RUSTFS_NODES: ${{ secrets.RUSTFS_NODES || vars.RUSTFS_NODES }}
|
||||
RUSTFS_SSH_USER: ${{ secrets.RUSTFS_SSH_USER || vars.RUSTFS_SSH_USER }}
|
||||
RUSTFS_NIGHTLY_PACKAGE_URL: ${{ vars.RUSTFS_NIGHTLY_PACKAGE_URL || 'https://dl.rustfs.com/artifacts/rustfs/packages/nightly/rustfs-nightly-latest.deb' }}
|
||||
RUSTFS_EXPECTED_RC_SHA256: ${{ inputs.rc_sha256 || vars.RUSTFS_TIER_RC_SHA256 }}
|
||||
PF_TESTING_GH_TOKEN: ${{ secrets.PF_TESTING_GH_TOKEN }}
|
||||
TIER_ARTIFACTS_DIR: /tmp/rustfs-tier-artifacts-${{ github.run_id }}-${{ github.run_attempt }}
|
||||
|
||||
jobs:
|
||||
tier-test:
|
||||
runs-on: smoke-testing
|
||||
# Requirement: a failing suite must not fail the workflow; failures
|
||||
# are filed to rustfs/backlog and the chain continues.
|
||||
continue-on-error: true
|
||||
timeout-minutes: 420
|
||||
if: ${{ github.event_name == 'workflow_dispatch' || github.event_name == 'repository_dispatch' }}
|
||||
if: ${{ github.event_name == 'workflow_dispatch' || github.event.workflow_run.conclusion == 'success' }}
|
||||
steps:
|
||||
- name: Initialize run evidence directory
|
||||
id: evidence
|
||||
run: |
|
||||
set -euo pipefail
|
||||
umask 077
|
||||
if ! mkdir -- "${TIER_ARTIFACTS_DIR}"; then
|
||||
echo "refusing to reuse tier evidence path: ${TIER_ARTIFACTS_DIR}" >&2
|
||||
exit 1
|
||||
fi
|
||||
test -d "${TIER_ARTIFACTS_DIR}"
|
||||
test ! -L "${TIER_ARTIFACTS_DIR}"
|
||||
|
||||
# auto-testing is private: clone it with the dedicated PF token (not
|
||||
# GITHUB_TOKEN) and retry transient GitHub/network failures.
|
||||
- name: Checkout auto-testing scripts (with retry)
|
||||
env:
|
||||
GH_TOKEN: ${{ secrets.PF_TESTING_GH_TOKEN }}
|
||||
AUTO_TESTING_REF: cxymds/fix-2132-tier-log-isolation
|
||||
AUTO_TESTING_COMMIT: 02da54dd62110649dc2860fc5fcd9e08d2e9a1ca
|
||||
run: |
|
||||
set -euo pipefail
|
||||
rm -rf auto-testing
|
||||
for attempt in 1 2 3 4 5; do
|
||||
if gh repo clone rustfs/auto-testing auto-testing -- \
|
||||
--branch "${AUTO_TESTING_REF}" --single-branch --depth 1 --quiet; then
|
||||
actual_commit="$(git -C auto-testing rev-parse HEAD)"
|
||||
if [[ "${actual_commit}" == "${AUTO_TESTING_COMMIT}" ]]; then
|
||||
echo "auto-testing ${actual_commit} cloned (attempt ${attempt})"
|
||||
exit 0
|
||||
fi
|
||||
echo "auto-testing commit mismatch: expected ${AUTO_TESTING_COMMIT}, got ${actual_commit}" >&2
|
||||
fi
|
||||
rm -rf auto-testing
|
||||
echo "clone attempt ${attempt} failed; retrying in $((attempt * 15))s" >&2
|
||||
sleep $((attempt * 15))
|
||||
done
|
||||
echo "ERROR: unable to clone rustfs/auto-testing after 5 attempts" >&2
|
||||
exit 1
|
||||
|
||||
- name: Download exact rc candidate
|
||||
uses: actions/download-artifact@37930b1c2abaa49bbe596cd826c3c89aef350131 # v7
|
||||
- name: Checkout auto-testing scripts
|
||||
uses: actions/checkout@9c091bb21b7c1c1d1991bb908d89e4e9dddfe3e0 # v7
|
||||
with:
|
||||
repository: rustfs/rustfs-release-validation
|
||||
run-id: '33465191972'
|
||||
name: rc-under-test-33465191972-1
|
||||
path: ${{ runner.temp }}/issue-2128-rc
|
||||
github-token: ${{ secrets.PF_TESTING_GH_TOKEN }}
|
||||
|
||||
- name: Verify exact rc candidate
|
||||
env:
|
||||
RC_BIN: ${{ runner.temp }}/issue-2128-rc/rc
|
||||
RC_PROVENANCE: ${{ runner.temp }}/issue-2128-rc/rc-build.json
|
||||
RC_EXPECTED_COMMIT: f6b9b509a60ef172a2b037d638c2cac46e762129
|
||||
RC_EXPECTED_SHA256: 3d128d99f05403f4028c7c9ae24b03d66a3e98f2090e66cb7f9f45a9e11fdce1
|
||||
run: |
|
||||
set -euo pipefail
|
||||
test -s "${RC_BIN}"
|
||||
test -s "${RC_PROVENANCE}"
|
||||
jq -e \
|
||||
--arg commit "${RC_EXPECTED_COMMIT}" \
|
||||
--arg digest "${RC_EXPECTED_SHA256}" \
|
||||
'.repository == "rustfs/cli"
|
||||
and .requestedCommit == $commit
|
||||
and .resolvedCommit == $commit
|
||||
and .binarySha256 == $digest
|
||||
and .target == "x86_64-unknown-linux-gnu"' \
|
||||
"${RC_PROVENANCE}" >/dev/null
|
||||
actual_sha256="$(sha256sum -- "${RC_BIN}" | awk '{print $1}')"
|
||||
test "${actual_sha256}" = "${RC_EXPECTED_SHA256}"
|
||||
chmod 0555 "${RC_BIN}"
|
||||
"${RC_BIN}" --version
|
||||
repository: rustfs/auto-testing
|
||||
ref: main
|
||||
path: auto-testing
|
||||
persist-credentials: false
|
||||
token: ${{ secrets.PF_TESTING_GH_TOKEN }}
|
||||
|
||||
- name: Show environment
|
||||
run: |
|
||||
@@ -187,32 +107,14 @@ jobs:
|
||||
|
||||
- name: Run tier suite
|
||||
id: test
|
||||
continue-on-error: true
|
||||
env:
|
||||
PACKAGE_URL_INPUT: ${{ inputs.package_url }}
|
||||
PACKAGE_SHA256_INPUT: ${{ inputs.package_sha256 }}
|
||||
RUSTFS_VERSION_INPUT: ${{ inputs.rustfs_version }}
|
||||
LOG_FILE: /tmp/rustfs-tier.log
|
||||
run: |
|
||||
set -euo pipefail
|
||||
LOG_FILE="${TIER_ARTIFACTS_DIR}/rustfs-tier.log"
|
||||
chmod +x auto-testing/rustfs-tier-test.sh
|
||||
RC_BIN="${RUNNER_TEMP}/issue-2128-rc/rc"
|
||||
PACKAGE_URL="${PACKAGE_URL_INPUT}"
|
||||
PACKAGE_SHA256="${PACKAGE_SHA256_INPUT}"
|
||||
RUSTFS_VERSION="${RUSTFS_VERSION_INPUT}"
|
||||
ARGS=(
|
||||
--all-topologies
|
||||
-y
|
||||
--log-file "${LOG_FILE}"
|
||||
--rc-bin "${RC_BIN}"
|
||||
--artifacts-dir "${TIER_ARTIFACTS_DIR}"
|
||||
)
|
||||
if [ -n "${RUSTFS_EXPECTED_RC_SHA256}" ]; then
|
||||
ARGS+=(--expected-rc-sha256 "${RUSTFS_EXPECTED_RC_SHA256}")
|
||||
fi
|
||||
if [ -n "${PACKAGE_SHA256}" ]; then
|
||||
ARGS+=(--sha256 "${PACKAGE_SHA256}")
|
||||
fi
|
||||
PACKAGE_URL='${{ inputs.package_url }}'
|
||||
RUSTFS_VERSION='${{ inputs.rustfs_version }}'
|
||||
ARGS=(--all-topologies -y --log-file "${LOG_FILE}")
|
||||
if [ -n "${PACKAGE_URL}" ]; then
|
||||
ARGS+=(--package-url "${PACKAGE_URL}")
|
||||
elif [ -n "${RUSTFS_VERSION}" ]; then
|
||||
@@ -222,34 +124,15 @@ jobs:
|
||||
fi
|
||||
./auto-testing/rustfs-tier-test.sh "${ARGS[@]}"
|
||||
|
||||
- name: Inject diagnostic case failure
|
||||
if: ${{ always() && steps.evidence.outcome == 'success' && inputs.force_case_failure }}
|
||||
run: |
|
||||
set -euo pipefail
|
||||
RESULT_FILE="${TIER_ARTIFACTS_DIR}/cases/single-single--TIER-101.json"
|
||||
test -s "${RESULT_FILE}"
|
||||
TMP_FILE="$(mktemp "${TIER_ARTIFACTS_DIR}/cases/.forced.XXXXXX")"
|
||||
jq '.status = "FAIL" | .case_rc = 97' "${RESULT_FILE}" > "${TMP_FILE}"
|
||||
mv "${TMP_FILE}" "${RESULT_FILE}"
|
||||
|
||||
- name: Generate report
|
||||
if: ${{ always() && steps.evidence.outcome == 'success' }}
|
||||
if: always()
|
||||
env:
|
||||
PACKAGE_URL_INPUT: ${{ inputs.package_url }}
|
||||
RUSTFS_VERSION_INPUT: ${{ inputs.rustfs_version }}
|
||||
TEST_OUTCOME: ${{ steps.test.outcome }}
|
||||
RUN_URL: ${{ github.server_url }}/${{ github.repository }}/actions/runs/${{ github.run_id }}
|
||||
TRIGGER_NAME: ${{ github.event_name }}
|
||||
LOG_FILE: /tmp/rustfs-tier.log
|
||||
REPORT_FILE: /tmp/rustfs-tier-report.md
|
||||
run: |
|
||||
set -euo pipefail
|
||||
test -d "${TIER_ARTIFACTS_DIR}"
|
||||
test ! -L "${TIER_ARTIFACTS_DIR}"
|
||||
LOG_FILE="${TIER_ARTIFACTS_DIR}/rustfs-tier.log"
|
||||
REPORT_FILE="${TIER_ARTIFACTS_DIR}/rustfs-tier-report.md"
|
||||
CASE_TABLE="${TIER_ARTIFACTS_DIR}/rustfs-tier-cases.md"
|
||||
GATE_RC_FILE="${TIER_ARTIFACTS_DIR}/rustfs-tier-gate.rc"
|
||||
PACKAGE_URL="${PACKAGE_URL_INPUT}"
|
||||
RUSTFS_VERSION="${RUSTFS_VERSION_INPUT}"
|
||||
PACKAGE_URL='${{ inputs.package_url }}'
|
||||
RUSTFS_VERSION='${{ inputs.rustfs_version }}'
|
||||
if [ -n "${PACKAGE_URL}" ]; then
|
||||
PACKAGE_SOURCE="${PACKAGE_URL}"
|
||||
elif [ -n "${RUSTFS_VERSION}" ]; then
|
||||
@@ -257,31 +140,12 @@ jobs:
|
||||
else
|
||||
PACKAGE_SOURCE="${RUSTFS_NIGHTLY_PACKAGE_URL}"
|
||||
fi
|
||||
set +e
|
||||
python3 auto-testing/rustfs_tier_report.py \
|
||||
--results-dir "${TIER_ARTIFACTS_DIR}/cases" \
|
||||
--provenance "${TIER_ARTIFACTS_DIR}/provenance.json" \
|
||||
--output "${CASE_TABLE}"
|
||||
CASE_GATE_RC=$?
|
||||
set -e
|
||||
printf '%s\n' "${CASE_GATE_RC}" > "${GATE_RC_FILE}"
|
||||
if [ ! -s "${CASE_TABLE}" ]; then
|
||||
{
|
||||
echo "## Case Summary"
|
||||
echo ""
|
||||
echo "Structured report generation failed before producing output (exit ${CASE_GATE_RC})."
|
||||
} > "${CASE_TABLE}"
|
||||
fi
|
||||
{
|
||||
echo "# RustFS tier test report"
|
||||
echo ""
|
||||
echo "- Run: ${RUN_URL}"
|
||||
echo "- Trigger: ${TRIGGER_NAME}"
|
||||
echo "- Run: ${{ github.server_url }}/${{ github.repository }}/actions/runs/${{ github.run_id }}"
|
||||
echo "- Trigger: ${{ github.event_name }}"
|
||||
echo "- Package: ${PACKAGE_SOURCE}"
|
||||
echo "- Test Step Outcome: ${TEST_OUTCOME}"
|
||||
echo "- Structured Gate Exit: ${CASE_GATE_RC}"
|
||||
echo ""
|
||||
cat "${CASE_TABLE}"
|
||||
echo ""
|
||||
echo "## Log tail"
|
||||
echo '```text'
|
||||
@@ -290,69 +154,15 @@ jobs:
|
||||
} | tee "${REPORT_FILE}"
|
||||
cat "${REPORT_FILE}" >> "${GITHUB_STEP_SUMMARY}"
|
||||
|
||||
- name: Upload functional report to dashboard
|
||||
if: ${{ always() && steps.evidence.outcome == 'success' }}
|
||||
continue-on-error: true
|
||||
env:
|
||||
GH_TOKEN: ${{ env.PF_TESTING_GH_TOKEN }}
|
||||
REPORT_FILE: ${{ env.TIER_ARTIFACTS_DIR }}/rustfs-tier-report.md
|
||||
SUITE: tier
|
||||
run: |
|
||||
set -euo pipefail
|
||||
if [ -z "${GH_TOKEN:-}" ]; then
|
||||
echo "PF_TESTING_GH_TOKEN is not configured; skipping dashboard upload"
|
||||
exit 0
|
||||
fi
|
||||
DATE="$(date -u +%Y-%m-%d)"
|
||||
REPORT_PATH="functional-reports/${SUITE}/${DATE}.md"
|
||||
CONTENT="$(python3 -c 'import base64,sys;print(base64.b64encode(open(sys.argv[1],"rb").read()).decode())' "${REPORT_FILE}")"
|
||||
SHA="$(gh api "repos/rustfs/dashboard/contents/${REPORT_PATH}" -q '.sha' 2>/dev/null || true)"
|
||||
if [ -n "${SHA}" ]; then
|
||||
jq -n --arg msg "report(${SUITE}): ${DATE}" --arg content "${CONTENT}" --arg sha "${SHA}" \
|
||||
'{message:$msg, content:$content, sha:$sha}' \
|
||||
| gh api --method PUT "repos/rustfs/dashboard/contents/${REPORT_PATH}" --input - >/dev/null
|
||||
else
|
||||
jq -n --arg msg "report(${SUITE}): ${DATE}" --arg content "${CONTENT}" \
|
||||
'{message:$msg, content:$content}' \
|
||||
| gh api --method PUT "repos/rustfs/dashboard/contents/${REPORT_PATH}" --input - >/dev/null
|
||||
fi
|
||||
|
||||
- name: Verify required tier evidence
|
||||
id: evidence_verify
|
||||
if: ${{ always() && steps.evidence.outcome == 'success' }}
|
||||
run: |
|
||||
set -euo pipefail
|
||||
failed=0
|
||||
for name in \
|
||||
rustfs-tier.log \
|
||||
rustfs-tier-report.md \
|
||||
rustfs-tier-cases.md \
|
||||
rustfs-tier-gate.rc \
|
||||
provenance.json; do
|
||||
if [ ! -s "${TIER_ARTIFACTS_DIR}/${name}" ]; then
|
||||
echo "required tier evidence is missing or empty: ${name}" >&2
|
||||
failed=1
|
||||
fi
|
||||
done
|
||||
for name in cases logs; do
|
||||
if [ ! -d "${TIER_ARTIFACTS_DIR}/${name}" ]; then
|
||||
echo "required tier evidence directory is missing: ${name}" >&2
|
||||
failed=1
|
||||
fi
|
||||
done
|
||||
if ! find "${TIER_ARTIFACTS_DIR}/cases" -maxdepth 1 -type f -name '*.json' -print -quit 2>/dev/null | grep -q .; then
|
||||
echo "no atomic tier case result was produced" >&2
|
||||
failed=1
|
||||
fi
|
||||
[ "${failed}" -eq 0 ]
|
||||
|
||||
- name: Upload report and logs
|
||||
if: ${{ always() && steps.evidence.outcome == 'success' }}
|
||||
if: always()
|
||||
uses: actions/upload-artifact@b7c566a772e6b6bfb58ed0dc250532a479d7789f # v6
|
||||
with:
|
||||
name: rustfs-tier-test-${{ github.run_id }}-${{ github.run_attempt }}
|
||||
path: ${{ env.TIER_ARTIFACTS_DIR }}/
|
||||
if-no-files-found: error
|
||||
name: rustfs-tier-test-${{ github.run_id }}
|
||||
path: |
|
||||
/tmp/rustfs-tier.log
|
||||
/tmp/rustfs-tier-report.md
|
||||
if-no-files-found: warn
|
||||
|
||||
- name: Cleanup environment (after)
|
||||
if: always()
|
||||
@@ -375,126 +185,6 @@ jobs:
|
||||
'
|
||||
done
|
||||
|
||||
- name: Enforce tier suite result
|
||||
id: gate
|
||||
if: always()
|
||||
env:
|
||||
EVIDENCE_OUTCOME: ${{ steps.evidence.outcome }}
|
||||
TEST_OUTCOME: ${{ steps.test.outcome }}
|
||||
GATE_RC_FILE: ${{ env.TIER_ARTIFACTS_DIR }}/rustfs-tier-gate.rc
|
||||
run: |
|
||||
set -euo pipefail
|
||||
failed=0
|
||||
if [ "${EVIDENCE_OUTCOME}" != "success" ]; then
|
||||
echo "tier evidence directory initialization is ${EVIDENCE_OUTCOME}, expected success" >&2
|
||||
failed=1
|
||||
fi
|
||||
if [ "${TEST_OUTCOME}" != "success" ]; then
|
||||
echo "tier suite step outcome is ${TEST_OUTCOME}, expected success" >&2
|
||||
failed=1
|
||||
fi
|
||||
if [ "${EVIDENCE_OUTCOME}" != "success" ]; then
|
||||
echo "structured gate result is unavailable because evidence initialization failed" >&2
|
||||
elif [ ! -s "${GATE_RC_FILE}" ]; then
|
||||
echo "structured gate result is missing" >&2
|
||||
failed=1
|
||||
else
|
||||
GATE_RC="$(tr -d '[:space:]' < "${GATE_RC_FILE}")"
|
||||
if ! [[ "${GATE_RC}" =~ ^[0-9]+$ ]] || [ "${GATE_RC}" -ne 0 ]; then
|
||||
echo "structured 56-case gate failed with exit ${GATE_RC:-invalid}" >&2
|
||||
failed=1
|
||||
fi
|
||||
fi
|
||||
[ "${failed}" -eq 0 ]
|
||||
|
||||
- name: File failure issue in rustfs/backlog
|
||||
if: ${{ always() && (failure() || steps.test.outcome == 'failure' || steps.test.outcome == 'cancelled' || steps.evidence_verify.outcome == 'failure' || steps.evidence_verify.outcome == 'cancelled' || steps.gate.outcome == 'failure' || steps.gate.outcome == 'cancelled') }}
|
||||
continue-on-error: true
|
||||
env:
|
||||
GH_TOKEN: ${{ secrets.PF_TESTING_GH_TOKEN }}
|
||||
SUITE: 'tier'
|
||||
SUITE_LABEL: 'Tier'
|
||||
RUN_URL: ${{ github.server_url }}/${{ github.repository }}/actions/runs/${{ github.run_id }}
|
||||
EVIDENCE_DIR: ${{ env.TIER_ARTIFACTS_DIR }}
|
||||
EVIDENCE_OUTCOME: ${{ steps.evidence.outcome }}
|
||||
VERIFY_OUTCOME: ${{ steps.evidence_verify.outcome }}
|
||||
GATE_OUTCOME: ${{ steps.gate.outcome }}
|
||||
REPORT_FILE: ${{ env.TIER_ARTIFACTS_DIR }}/rustfs-tier-report.md
|
||||
LOG_FILE: ${{ env.TIER_ARTIFACTS_DIR }}/rustfs-tier.log
|
||||
run: |
|
||||
set -euo pipefail
|
||||
if [ -z "${GH_TOKEN:-}" ]; then
|
||||
echo "PF_TESTING_GH_TOKEN is not configured; skipping backlog issue"
|
||||
exit 0
|
||||
fi
|
||||
TITLE="[functional][${SUITE}] ${SUITE_LABEL} suite failed (run ${GITHUB_RUN_ID})"
|
||||
EXISTING="$(gh issue list -R rustfs/backlog --state all \
|
||||
--search "in:title \"run ${GITHUB_RUN_ID}\"" \
|
||||
--json number --jq '.[].number' || true)"
|
||||
if [ -n "${EXISTING}" ]; then
|
||||
echo "backlog issue already exists for run ${GITHUB_RUN_ID}; skipping"
|
||||
exit 0
|
||||
fi
|
||||
redact() {
|
||||
sed -E \
|
||||
-e 's/(RUSTFS_(ACCESS_KEY|SECRET_KEY)[=: ]+)[^[:space:]]+/\1[REDACTED]/Ig' \
|
||||
-e 's/(Authorization:).*/\1 [REDACTED]/Ig' \
|
||||
-e 's/(X-Amz-Signature=)[^&[:space:]]+/\1[REDACTED]/Ig' \
|
||||
-e 's/^.*(password|secret|token)[=: ].*/[REDACTED SENSITIVE LINE]/Ig'
|
||||
}
|
||||
BODY_FILE="$(mktemp)"
|
||||
{
|
||||
echo "The **${SUITE_LABEL}** functional suite failed."
|
||||
echo ""
|
||||
echo "- Suite: \`${SUITE}\`"
|
||||
echo "- Run: ${RUN_URL}"
|
||||
echo "- Trigger: ${GITHUB_EVENT_NAME}"
|
||||
echo "- Date: $(date -u +%Y-%m-%d)"
|
||||
echo "- Evidence initialization: ${EVIDENCE_OUTCOME}"
|
||||
echo "- Evidence verification: ${VERIFY_OUTCOME}"
|
||||
echo "- Final gate: ${GATE_OUTCOME}"
|
||||
echo ""
|
||||
echo "## Report (errors and symptoms)"
|
||||
echo ""
|
||||
if [ "${EVIDENCE_OUTCOME}" != "success" ]; then
|
||||
echo "(the run evidence directory was rejected; its contents were not read)"
|
||||
elif [ ! -d "${EVIDENCE_DIR}" ] || [ -L "${EVIDENCE_DIR}" ]; then
|
||||
echo "(the run evidence directory is missing or unsafe; its contents were not read)"
|
||||
elif [ -s "${REPORT_FILE}" ]; then
|
||||
redact < "${REPORT_FILE}"
|
||||
elif [ -s "${LOG_FILE:-}" ]; then
|
||||
echo "(report file missing; log tail below)"
|
||||
echo ""
|
||||
tail -n 200 "${LOG_FILE}" | redact
|
||||
else
|
||||
echo "(no report or log file was produced)"
|
||||
fi
|
||||
} | head -c 55000 > "${BODY_FILE}"
|
||||
gh label create functional-test -R rustfs/backlog --color d73a4a 2>/dev/null || true
|
||||
if ! gh issue create -R rustfs/backlog --title "${TITLE}" \
|
||||
--body-file "${BODY_FILE}" --label functional-test; then
|
||||
gh issue create -R rustfs/backlog --title "${TITLE}" --body-file "${BODY_FILE}"
|
||||
fi
|
||||
echo "filed backlog issue for suite ${SUITE}"
|
||||
|
||||
- name: "Continue functional chain (next: Storage engine)"
|
||||
# Only chain-triggered runs forward to the next suite; standalone
|
||||
# workflow_dispatch runs stop after their own cleanup.
|
||||
if: ${{ always() && github.event_name == 'repository_dispatch' }}
|
||||
continue-on-error: true
|
||||
env:
|
||||
GH_TOKEN: ${{ secrets.PF_TESTING_GH_TOKEN }}
|
||||
run: |
|
||||
set -euo pipefail
|
||||
if [ -z "${GH_TOKEN:-}" ]; then
|
||||
echo "PF_TESTING_GH_TOKEN is not configured; cannot dispatch the next suite" >&2
|
||||
exit 1
|
||||
fi
|
||||
echo "Dispatching next functional suite: Storage engine"
|
||||
gh api --method POST repos/rustfs/rustfs/dispatches \
|
||||
-f event_type='rustfs-chain-storage' \
|
||||
-F 'client_payload[from_suite]=tier'
|
||||
|
||||
- name: Notify on failure
|
||||
if: failure()
|
||||
run: |
|
||||
|
||||
@@ -1,413 +0,0 @@
|
||||
# Copyright 2024 RustFS Team
|
||||
#
|
||||
# Licensed under the Apache License, Version 2.0 (the "License");
|
||||
# you may not use this file except in compliance with the License.
|
||||
# You may obtain a copy of the License at
|
||||
#
|
||||
# http://www.apache.org/licenses/LICENSE-2.0
|
||||
#
|
||||
# Unless required by applicable law or agreed to in writing, software
|
||||
# distributed under the License is distributed on an "AS IS" BASIS,
|
||||
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||
# See the License for the specific language governing permissions and
|
||||
# limitations under the License.
|
||||
|
||||
name: RustFS Upgrade Test
|
||||
|
||||
on:
|
||||
workflow_dispatch:
|
||||
inputs:
|
||||
from_version:
|
||||
description: 'OLD RustFS release tag (e.g. 1.0.0-rc.4-preview.1)'
|
||||
required: false
|
||||
default: '1.0.0-rc.4-preview.1'
|
||||
from_url:
|
||||
description: 'OLD .deb URL. Overrides from_version.'
|
||||
required: false
|
||||
type: string
|
||||
to_version:
|
||||
description: 'NEW RustFS release tag (leave empty for latest nightly)'
|
||||
required: false
|
||||
to_url:
|
||||
description: 'NEW .deb URL. Overrides to_version / nightly default.'
|
||||
required: false
|
||||
type: string
|
||||
topology:
|
||||
description: 'Topology to run (all = SNSD, SNMD, MNMD)'
|
||||
type: choice
|
||||
options:
|
||||
- all
|
||||
- single-single
|
||||
- single-multi
|
||||
- multi-multi
|
||||
default: all
|
||||
backends:
|
||||
description: 'KMS backends to run (local,vault-kv2)'
|
||||
required: false
|
||||
default: 'local,vault-kv2'
|
||||
cleanup_before:
|
||||
description: 'Reset the nodes before the test (DESTROYS existing data/config)'
|
||||
type: boolean
|
||||
default: true
|
||||
cleanup_after:
|
||||
description: 'Reset the nodes after the test (DESTROYS test data/config)'
|
||||
type: boolean
|
||||
default: true
|
||||
repository_dispatch:
|
||||
# Functional-chain entry: dispatched by rustfs-functional-chain.yml.
|
||||
types: [rustfs-chain-upgrade]
|
||||
|
||||
permissions:
|
||||
contents: read
|
||||
|
||||
concurrency:
|
||||
group: rustfs-shared-functional-tests
|
||||
cancel-in-progress: false
|
||||
|
||||
defaults:
|
||||
run:
|
||||
shell: bash
|
||||
|
||||
env:
|
||||
RUSTFS_ACCESS_KEY: ${{ secrets.RUSTFS_ACCESS_KEY }}
|
||||
RUSTFS_SECRET_KEY: ${{ secrets.RUSTFS_SECRET_KEY }}
|
||||
RUSTFS_NODES: ${{ secrets.RUSTFS_NODES || vars.RUSTFS_NODES }}
|
||||
RUSTFS_SSH_USER: ${{ secrets.RUSTFS_SSH_USER || vars.RUSTFS_SSH_USER }}
|
||||
RUSTFS_NIGHTLY_PACKAGE_URL: ${{ vars.RUSTFS_NIGHTLY_PACKAGE_URL || 'https://dl.rustfs.com/artifacts/rustfs/packages/nightly/rustfs-nightly-latest.deb' }}
|
||||
PF_TESTING_GH_TOKEN: ${{ secrets.PF_TESTING_GH_TOKEN }}
|
||||
|
||||
jobs:
|
||||
upgrade-test:
|
||||
runs-on: smoke-testing
|
||||
continue-on-error: true
|
||||
timeout-minutes: 420
|
||||
if: ${{ github.event_name == 'workflow_dispatch' || github.event_name == 'repository_dispatch' }}
|
||||
steps:
|
||||
# auto-testing is private: clone it with the dedicated PF token (not
|
||||
# GITHUB_TOKEN) and retry transient GitHub/network failures.
|
||||
- name: Checkout auto-testing scripts (with retry)
|
||||
env:
|
||||
GH_TOKEN: ${{ secrets.PF_TESTING_GH_TOKEN }}
|
||||
run: |
|
||||
set -euo pipefail
|
||||
rm -rf auto-testing
|
||||
for attempt in 1 2 3 4 5; do
|
||||
if gh repo clone rustfs/auto-testing auto-testing -- --depth 1 --quiet; then
|
||||
echo "auto-testing cloned (attempt ${attempt})"
|
||||
exit 0
|
||||
fi
|
||||
rm -rf auto-testing
|
||||
echo "clone attempt ${attempt} failed; retrying in $((attempt * 15))s" >&2
|
||||
sleep $((attempt * 15))
|
||||
done
|
||||
echo "ERROR: unable to clone rustfs/auto-testing after 5 attempts" >&2
|
||||
exit 1
|
||||
|
||||
- name: Show environment
|
||||
run: |
|
||||
uname -a
|
||||
jq --version
|
||||
openssl version
|
||||
aws --version || true
|
||||
docker --version || true
|
||||
df -h /data | tail -1
|
||||
|
||||
- name: Cleanup environment (before)
|
||||
if: ${{ inputs.cleanup_before != 'false' || github.event_name != 'workflow_dispatch' }}
|
||||
run: |
|
||||
set -euo pipefail
|
||||
read -r -a NODES <<< "${RUSTFS_NODES:-vm000 vm001 vm002}"
|
||||
SSH_USER="${RUSTFS_SSH_USER:-azureuser}"
|
||||
for node in "${NODES[@]}"; do
|
||||
ssh -o BatchMode=yes -o ConnectTimeout=10 -o StrictHostKeyChecking=accept-new "${SSH_USER}@${node}" '
|
||||
set -euo pipefail
|
||||
SUDO=""; [ "$(id -u)" -ne 0 ] && SUDO="sudo -n"
|
||||
${SUDO} systemctl stop rustfs 2>/dev/null || true
|
||||
if ${SUDO} dpkg -l rustfs 2>/dev/null | grep -q "^ii"; then
|
||||
${SUDO} dpkg -P rustfs
|
||||
fi
|
||||
for i in 1 2 3 4; do ${SUDO} rm -rf /data/rustfs${i}/mnmd; done
|
||||
${SUDO} rm -rf /var/log/rustfs /var/lib/rustfs/kms /var/lib/rustfs/kms-backup
|
||||
'
|
||||
done
|
||||
|
||||
- name: Ensure docker (Vault container)
|
||||
run: |
|
||||
if ! command -v docker >/dev/null 2>&1; then
|
||||
sudo apt-get update
|
||||
sudo apt-get install -y docker.io
|
||||
fi
|
||||
sudo systemctl enable --now docker
|
||||
docker info >/dev/null 2>&1 || sudo docker info >/dev/null 2>&1
|
||||
|
||||
- name: Run upgrade compatibility suite
|
||||
id: test
|
||||
continue-on-error: true
|
||||
env:
|
||||
LOG_FILE: /tmp/rustfs-upgrade.log
|
||||
run: |
|
||||
set -euo pipefail
|
||||
chmod +x auto-testing/rustfs-upgrade-test.sh
|
||||
FROM_URL='${{ inputs.from_url }}'
|
||||
FROM_VERSION='${{ inputs.from_version }}'
|
||||
TO_URL='${{ inputs.to_url }}'
|
||||
TO_VERSION='${{ inputs.to_version }}'
|
||||
TOPOLOGY='${{ inputs.topology }}'
|
||||
BACKENDS='${{ inputs.backends }}'
|
||||
ARGS=(-y --log-file "${LOG_FILE}")
|
||||
if [ "${TOPOLOGY}" = "all" ] || [ -z "${TOPOLOGY}" ] || [ "${TOPOLOGY}" = "null" ]; then
|
||||
ARGS+=(--all-topologies)
|
||||
else
|
||||
ARGS+=(--topology "${TOPOLOGY}")
|
||||
fi
|
||||
if [ -n "${BACKENDS}" ] && [ "${BACKENDS}" != "null" ]; then
|
||||
ARGS+=(--backends "${BACKENDS}")
|
||||
fi
|
||||
if [ -n "${FROM_URL}" ]; then
|
||||
ARGS+=(--from-url "${FROM_URL}")
|
||||
elif [ -n "${FROM_VERSION}" ] && [ "${FROM_VERSION}" != "null" ]; then
|
||||
ARGS+=(--from-version "${FROM_VERSION}")
|
||||
fi
|
||||
if [ -n "${TO_URL}" ]; then
|
||||
ARGS+=(--to-url "${TO_URL}")
|
||||
elif [ -n "${TO_VERSION}" ] && [ "${TO_VERSION}" != "null" ]; then
|
||||
ARGS+=(--to-version "${TO_VERSION}")
|
||||
else
|
||||
ARGS+=(--to-url "${RUSTFS_NIGHTLY_PACKAGE_URL}")
|
||||
fi
|
||||
./auto-testing/rustfs-upgrade-test.sh "${ARGS[@]}"
|
||||
|
||||
- name: Generate report
|
||||
if: always()
|
||||
env:
|
||||
LOG_FILE: /tmp/rustfs-upgrade.log
|
||||
REPORT_FILE: /tmp/rustfs-upgrade-report.md
|
||||
run: |
|
||||
set -euo pipefail
|
||||
FROM_URL='${{ inputs.from_url }}'
|
||||
FROM_VERSION='${{ inputs.from_version }}'
|
||||
TO_URL='${{ inputs.to_url }}'
|
||||
TO_VERSION='${{ inputs.to_version }}'
|
||||
if [ -n "${FROM_URL}" ]; then
|
||||
FROM_SOURCE="${FROM_URL}"
|
||||
elif [ -n "${FROM_VERSION}" ]; then
|
||||
FROM_SOURCE="version ${FROM_VERSION}"
|
||||
else
|
||||
FROM_SOURCE="release (default)"
|
||||
fi
|
||||
if [ -n "${TO_URL}" ]; then
|
||||
TO_SOURCE="${TO_URL}"
|
||||
elif [ -n "${TO_VERSION}" ]; then
|
||||
TO_SOURCE="version ${TO_VERSION}"
|
||||
else
|
||||
TO_SOURCE="${RUSTFS_NIGHTLY_PACKAGE_URL}"
|
||||
fi
|
||||
CASE_TABLE="/tmp/rustfs-upgrade-cases.md"
|
||||
python3 - "${LOG_FILE}" "${CASE_TABLE}" <<'PY'
|
||||
import re
|
||||
import sys
|
||||
|
||||
log_file, out_file = sys.argv[1], sys.argv[2]
|
||||
ansi = re.compile(r'\x1b\[[0-9;]*m')
|
||||
start_re = re.compile(r'^---\s+([A-Z]+-[0-9]+)\s+(.+?)\s+---$')
|
||||
done_re = re.compile(r'^\[(PASS|FAIL|UNSUPPORTED)\]\s+([A-Z]+-[0-9]+)\b')
|
||||
|
||||
rows = []
|
||||
index = {}
|
||||
try:
|
||||
with open(log_file, 'r', encoding='utf-8', errors='replace') as fh:
|
||||
for raw in fh:
|
||||
line = ansi.sub('', raw).strip()
|
||||
m = start_re.match(line)
|
||||
if m:
|
||||
case_id, name = m.group(1), m.group(2)
|
||||
if case_id not in index:
|
||||
index[case_id] = len(rows)
|
||||
rows.append([case_id, name, 'RUNNING'])
|
||||
continue
|
||||
m = done_re.match(line)
|
||||
if m:
|
||||
status, case_id = m.group(1), m.group(2)
|
||||
if case_id in index:
|
||||
rows[index[case_id]][2] = status
|
||||
else:
|
||||
rows.append([case_id, case_id, status])
|
||||
index[case_id] = len(rows) - 1
|
||||
except FileNotFoundError:
|
||||
rows = []
|
||||
|
||||
counts = {'PASS': 0, 'FAIL': 0, 'UNSUPPORTED': 0, 'RUNNING': 0}
|
||||
for _, _, status in rows:
|
||||
counts[status] = counts.get(status, 0) + 1
|
||||
|
||||
with open(out_file, 'w', encoding='utf-8') as out:
|
||||
out.write('## Case Summary\n\n')
|
||||
out.write(f"- Total: {len(rows)}\\n")
|
||||
out.write(f"- PASS: {counts.get('PASS', 0)}\\n")
|
||||
out.write(f"- FAIL: {counts.get('FAIL', 0)}\\n")
|
||||
out.write(f"- UNSUPPORTED: {counts.get('UNSUPPORTED', 0)}\\n")
|
||||
out.write('\\n')
|
||||
out.write('| Case | Name | Status |\\n')
|
||||
out.write('| --- | --- | --- |\\n')
|
||||
for case_id, name, status in rows:
|
||||
out.write(f'| {case_id} | {name} | {status} |\\n')
|
||||
PY
|
||||
{
|
||||
echo "# RustFS upgrade compatibility report"
|
||||
echo ""
|
||||
echo "- Run: ${{ github.server_url }}/${{ github.repository }}/actions/runs/${{ github.run_id }}"
|
||||
echo "- Trigger: ${{ github.event_name }}"
|
||||
echo "- From: ${FROM_SOURCE}"
|
||||
echo "- To: ${TO_SOURCE}"
|
||||
echo "- Test Step Outcome: ${{ steps.test.outcome }}"
|
||||
echo ""
|
||||
cat "${CASE_TABLE}" || true
|
||||
echo ""
|
||||
echo "## Log tail"
|
||||
echo '```text'
|
||||
tail -n 200 "${LOG_FILE}" || true
|
||||
echo '```'
|
||||
} | tee "${REPORT_FILE}"
|
||||
cat "${REPORT_FILE}" >> "${GITHUB_STEP_SUMMARY}"
|
||||
|
||||
- name: Upload functional report to dashboard
|
||||
if: always()
|
||||
continue-on-error: true
|
||||
env:
|
||||
GH_TOKEN: ${{ env.PF_TESTING_GH_TOKEN }}
|
||||
REPORT_FILE: /tmp/rustfs-upgrade-report.md
|
||||
SUITE: upgrade
|
||||
run: |
|
||||
set -euo pipefail
|
||||
if [ -z "${GH_TOKEN:-}" ]; then
|
||||
echo "PF_TESTING_GH_TOKEN is not configured; skipping dashboard upload"
|
||||
exit 0
|
||||
fi
|
||||
DATE="$(date -u +%Y-%m-%d)"
|
||||
REPORT_PATH="functional-reports/${SUITE}/${DATE}.md"
|
||||
CONTENT="$(python3 -c 'import base64,sys;print(base64.b64encode(open(sys.argv[1],"rb").read()).decode())' "${REPORT_FILE}")"
|
||||
SHA="$(gh api "repos/rustfs/dashboard/contents/${REPORT_PATH}" -q '.sha' 2>/dev/null || true)"
|
||||
if [ -n "${SHA}" ]; then
|
||||
jq -n --arg msg "report(${SUITE}): ${DATE}" --arg content "${CONTENT}" --arg sha "${SHA}" \
|
||||
'{message:$msg, content:$content, sha:$sha}' \
|
||||
| gh api --method PUT "repos/rustfs/dashboard/contents/${REPORT_PATH}" --input - >/dev/null
|
||||
else
|
||||
jq -n --arg msg "report(${SUITE}): ${DATE}" --arg content "${CONTENT}" \
|
||||
'{message:$msg, content:$content}' \
|
||||
| gh api --method PUT "repos/rustfs/dashboard/contents/${REPORT_PATH}" --input - >/dev/null
|
||||
fi
|
||||
|
||||
- name: File failure issue in rustfs/backlog
|
||||
if: ${{ always() && (failure() || steps.test.outcome == 'failure' || steps.test.outcome == 'cancelled') }}
|
||||
continue-on-error: true
|
||||
env:
|
||||
GH_TOKEN: ${{ secrets.PF_TESTING_GH_TOKEN }}
|
||||
SUITE: 'upgrade'
|
||||
SUITE_LABEL: 'Upgrade compatibility'
|
||||
RUN_URL: ${{ github.server_url }}/${{ github.repository }}/actions/runs/${{ github.run_id }}
|
||||
REPORT_FILE: '/tmp/rustfs-upgrade-report.md'
|
||||
LOG_FILE: '/tmp/rustfs-upgrade.log'
|
||||
run: |
|
||||
set -euo pipefail
|
||||
if [ -z "${GH_TOKEN:-}" ]; then
|
||||
echo "PF_TESTING_GH_TOKEN is not configured; skipping backlog issue"
|
||||
exit 0
|
||||
fi
|
||||
TITLE="[functional][${SUITE}] ${SUITE_LABEL} suite failed (run ${GITHUB_RUN_ID})"
|
||||
EXISTING="$(gh issue list -R rustfs/backlog --state all \
|
||||
--search "in:title \"run ${GITHUB_RUN_ID}\"" \
|
||||
--json number --jq '.[].number' || true)"
|
||||
if [ -n "${EXISTING}" ]; then
|
||||
echo "backlog issue already exists for run ${GITHUB_RUN_ID}; skipping"
|
||||
exit 0
|
||||
fi
|
||||
redact() {
|
||||
sed -E \
|
||||
-e 's/(RUSTFS_(ACCESS_KEY|SECRET_KEY)[=: ]+)[^[:space:]]+/\1[REDACTED]/Ig' \
|
||||
-e 's/(Authorization:).*/\1 [REDACTED]/Ig' \
|
||||
-e 's/(X-Amz-Signature=)[^&[:space:]]+/\1[REDACTED]/Ig' \
|
||||
-e 's/^.*(password|secret|token)[=: ].*/[REDACTED SENSITIVE LINE]/Ig'
|
||||
}
|
||||
BODY_FILE="$(mktemp)"
|
||||
{
|
||||
echo "The **${SUITE_LABEL}** functional suite failed."
|
||||
echo ""
|
||||
echo "- Suite: \`${SUITE}\`"
|
||||
echo "- Run: ${RUN_URL}"
|
||||
echo "- Trigger: ${GITHUB_EVENT_NAME}"
|
||||
echo "- Date: $(date -u +%Y-%m-%d)"
|
||||
echo ""
|
||||
echo "## Report (errors and symptoms)"
|
||||
echo ""
|
||||
if [ -s "${REPORT_FILE}" ]; then
|
||||
redact < "${REPORT_FILE}"
|
||||
elif [ -s "${LOG_FILE:-}" ]; then
|
||||
echo "(report file missing; log tail below)"
|
||||
echo ""
|
||||
tail -n 200 "${LOG_FILE}" | redact
|
||||
else
|
||||
echo "(no report or log file was produced)"
|
||||
fi
|
||||
} | head -c 55000 > "${BODY_FILE}"
|
||||
gh label create functional-test -R rustfs/backlog --color d73a4a 2>/dev/null || true
|
||||
if ! gh issue create -R rustfs/backlog --title "${TITLE}" \
|
||||
--body-file "${BODY_FILE}" --label functional-test; then
|
||||
gh issue create -R rustfs/backlog --title "${TITLE}" --body-file "${BODY_FILE}"
|
||||
fi
|
||||
echo "filed backlog issue for suite ${SUITE}"
|
||||
|
||||
- name: Upload report and logs
|
||||
if: always()
|
||||
uses: actions/upload-artifact@b7c566a772e6b6bfb58ed0dc250532a479d7789f # v6
|
||||
with:
|
||||
name: rustfs-upgrade-test-${{ github.run_id }}
|
||||
path: |
|
||||
/tmp/rustfs-upgrade-report.md
|
||||
/tmp/rustfs-upgrade.*/*
|
||||
if-no-files-found: ignore
|
||||
retention-days: 3
|
||||
|
||||
- name: Cleanup environment (after)
|
||||
if: ${{ always() && (inputs.cleanup_after != 'false' || github.event_name != 'workflow_dispatch') }}
|
||||
run: |
|
||||
set -euo pipefail
|
||||
read -r -a NODES <<< "${RUSTFS_NODES:-vm000 vm001 vm002}"
|
||||
SSH_USER="${RUSTFS_SSH_USER:-azureuser}"
|
||||
for node in "${NODES[@]}"; do
|
||||
ssh -o BatchMode=yes -o ConnectTimeout=10 -o StrictHostKeyChecking=accept-new "${SSH_USER}@${node}" '
|
||||
set -euo pipefail
|
||||
SUDO=""; [ "$(id -u)" -ne 0 ] && SUDO="sudo -n"
|
||||
${SUDO} systemctl stop rustfs 2>/dev/null || true
|
||||
if ${SUDO} dpkg -l rustfs 2>/dev/null | grep -q "^ii"; then
|
||||
${SUDO} dpkg -P rustfs
|
||||
fi
|
||||
for i in 1 2 3 4; do ${SUDO} rm -rf /data/rustfs${i}/mnmd; done
|
||||
${SUDO} rm -rf /var/log/rustfs /var/lib/rustfs/kms /var/lib/rustfs/kms-backup
|
||||
'
|
||||
done
|
||||
|
||||
- name: "Continue functional chain (next: S3 compatibility)"
|
||||
# Only chain-triggered runs forward to the next suite; standalone
|
||||
# workflow_dispatch runs stop after their own cleanup.
|
||||
if: ${{ always() && github.event_name == 'repository_dispatch' }}
|
||||
continue-on-error: true
|
||||
env:
|
||||
GH_TOKEN: ${{ secrets.PF_TESTING_GH_TOKEN }}
|
||||
run: |
|
||||
set -euo pipefail
|
||||
if [ -z "${GH_TOKEN:-}" ]; then
|
||||
echo "PF_TESTING_GH_TOKEN is not configured; cannot dispatch the next suite" >&2
|
||||
exit 1
|
||||
fi
|
||||
echo "Dispatching next functional suite: S3 compatibility"
|
||||
gh api --method POST repos/rustfs/rustfs/dispatches \
|
||||
-f event_type='rustfs-chain-s3' \
|
||||
-F 'client_payload[from_suite]=upgrade'
|
||||
|
||||
- name: Notify on failure
|
||||
if: failure()
|
||||
run: |
|
||||
echo "RustFS upgrade compatibility test failed"
|
||||
echo "From: ${{ inputs.from_url || inputs.from_version || 'release (default)' }}"
|
||||
echo "To: ${{ inputs.to_url || inputs.to_version || 'nightly (R2 latest)' }}"
|
||||
echo "See the uploaded report and logs for details."
|
||||
@@ -123,13 +123,12 @@ runtime/build output:
|
||||
- Use `make pre-commit` only when its repository-wide fast checks add confidence
|
||||
beyond the focused checks.
|
||||
|
||||
### Broad Cross-Module Changes
|
||||
### Broad or High-Risk Changes
|
||||
|
||||
Do not run `make pre-pr` by default before opening a PR. Consider it only when
|
||||
the final diff is broad, spans multiple modules, and targeted checks cannot
|
||||
bound the impact. Decide dynamically from the affected boundaries and risks;
|
||||
otherwise use the scoped formatting, linting, compilation, and test checks
|
||||
above.
|
||||
After the required adversarial review, run `make pre-pr` when targeted coverage
|
||||
cannot bound the impact, including dependency/toolchain/build-matrix changes,
|
||||
unbounded cross-crate APIs, or locking, durability, erasure coding, replication,
|
||||
RPC, IAM/KMS/auth, cryptography, on-disk/on-wire, and S3-visible behavior.
|
||||
|
||||
`make pre-pr` includes `make pre-commit`; never run both for the same unchanged
|
||||
diff. Do not repeat a check already covered by a successful umbrella gate.
|
||||
|
||||
@@ -15,7 +15,7 @@ cargo check -p <crate> # fast type-check one crate
|
||||
cargo test -p <crate> # test one crate
|
||||
cargo fmt --all # format (required before PR)
|
||||
make pre-commit # fast gate: fmt + arch checks + quick-check (NO clippy/tests)
|
||||
make pre-pr # optional full gate for broad cross-module changes
|
||||
make pre-pr # full pre-PR gate: fmt + arch checks + clippy + tests
|
||||
make build-docker BUILD_OS=ubuntu22.04
|
||||
```
|
||||
|
||||
|
||||
+7
-20
@@ -62,20 +62,12 @@ make test
|
||||
# Fast pre-commit gate — see below for exactly what it runs
|
||||
make pre-commit
|
||||
|
||||
# Optional full gate for broad cross-module changes (pre-commit + clippy + tests)
|
||||
# Full pre-PR gate (pre-commit gates + clippy + tests)
|
||||
make pre-pr
|
||||
```
|
||||
|
||||
> `make test` requires [cargo-nextest](https://nexte.st) (CI runs it and only nextest honours `.config/nextest.toml` test-groups). Install it with `cargo install cargo-nextest --locked` or a prebuilt binary (see https://nexte.st/docs/installation/). To run the plain `cargo test` fallback anyway (results not authoritative — serialization semantics differ from CI), set `RUSTFS_ALLOW_CARGO_TEST_FALLBACK=1`.
|
||||
|
||||
> Some guard checks are Python (`test-wiring-check` in `make pre-commit`, plus the
|
||||
> security-coverage and scheduled-validation self-tests in `make test`) and import
|
||||
> `tomllib`, so they need **Python 3.11+**. Make resolves the interpreter through
|
||||
> `scripts/python_bin.sh`, which prefers a `python3.11`+ on `PATH` and otherwise falls
|
||||
> back to `uv run --python 3.12`. macOS ships `/usr/bin/python3` at 3.9, so install a
|
||||
> newer one (`brew install python@3.12`) or [uv](https://docs.astral.sh/uv/); pin a
|
||||
> specific interpreter with `RUSTFS_PYTHON=/path/to/python3.12`.
|
||||
|
||||
> For the full test-layer taxonomy (unit / ecstore black-box / e2e / s3s-e2e / S3 compatibility / chaos / fuzz / bench), each layer's entry command, the naming conventions the migration gate depends on, and the serial/nextest rules, see [docs/testing/README.md](docs/testing/README.md).
|
||||
|
||||
> For the event, timeout, required-status, and local reproduction matrix, see [docs/testing/ci-gates.md](docs/testing/ci-gates.md).
|
||||
@@ -96,16 +88,14 @@ make pre-pr
|
||||
8. `quick-check` — `cargo check --workspace --exclude e2e_test`
|
||||
|
||||
**`make pre-commit` does NOT run clippy and does NOT run any tests.**
|
||||
It does not replace the scoped Clippy and test checks applicable to a change.
|
||||
A green `make pre-commit` is not enough to open a pull request.
|
||||
|
||||
`make pre-pr` is the **full** gate: it runs all of the guard checks above,
|
||||
then `clippy-check` (`cargo clippy --all-targets --all-features -- -D warnings`)
|
||||
and `test` (shell script tests, workspace tests excluding `e2e_test`, and doc
|
||||
tests). Complete the applicable multi-role adversarial review described in
|
||||
`AGENTS.md` first. Do not run `make pre-pr` locally by default before opening or
|
||||
updating a pull request. Consider it only for a broad change that spans multiple
|
||||
modules and whose impact cannot be bounded by targeted checks; decide from the
|
||||
affected boundaries and risks. CI still runs its configured repository gates.
|
||||
`AGENTS.md` before running `make pre-pr`; then run the gate before opening or
|
||||
updating a pull request. This is what CI enforces.
|
||||
|
||||
### 🔒 Git Pre-commit Hooks (optional)
|
||||
|
||||
@@ -124,9 +114,8 @@ Or manually:
|
||||
chmod +x .git/hooks/pre-commit
|
||||
```
|
||||
|
||||
With or without a hook, follow the verification tiers in `AGENTS.md`. Run the
|
||||
applicable scoped checks, and reserve `make pre-pr` for broad cross-module
|
||||
changes whose impact cannot be bounded by those checks.
|
||||
With or without a hook, the expectation is the same: run `make pre-commit`
|
||||
before committing and `make pre-pr` before opening a pull request.
|
||||
|
||||
### 📝 Formatting Configuration
|
||||
|
||||
@@ -165,9 +154,7 @@ Example output when formatting fails:
|
||||
3. **Run the fast gate**: `make pre-commit` (no clippy, no tests)
|
||||
4. **Commit your changes**: `git commit -m "your message"`
|
||||
5. **Complete the applicable multi-role adversarial review** for non-exempt changes (see `AGENTS.md`)
|
||||
6. **Run applicable scoped checks before opening/updating a PR**; consider
|
||||
`make pre-pr` only for broad cross-module changes whose impact cannot be
|
||||
bounded by targeted checks
|
||||
6. **Run the full gate before opening/updating a PR**: `make pre-pr` (clippy + tests)
|
||||
7. **Push to your branch**: `git push`
|
||||
|
||||
### 🛠️ IDE Integration
|
||||
|
||||
Generated
+81
-119
@@ -627,7 +627,8 @@ dependencies = [
|
||||
[[package]]
|
||||
name = "astral-tokio-tar"
|
||||
version = "0.7.0"
|
||||
source = "git+https://github.com/cxymds/tokio-tar.git?rev=603756478b7668436e464519c77ccac22a99ba96#603756478b7668436e464519c77ccac22a99ba96"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "6f2e989b33246fe9240d39accf4dd9a01e0b6c1f3ce9dd095e0a47fa02505523"
|
||||
dependencies = [
|
||||
"futures-core",
|
||||
"libc",
|
||||
@@ -2162,7 +2163,6 @@ dependencies = [
|
||||
"compression-core",
|
||||
"flate2",
|
||||
"liblzma",
|
||||
"lz4",
|
||||
"memchr",
|
||||
"zstd",
|
||||
"zstd-safe",
|
||||
@@ -3916,7 +3916,7 @@ checksum = "d0881ea181b1df73ff77ffaaf9c7544ecc11e82fba9b5f27b262a3c73a332555"
|
||||
|
||||
[[package]]
|
||||
name = "e2e_test"
|
||||
version = "1.0.0-rc.5"
|
||||
version = "1.0.0-rc.4"
|
||||
dependencies = [
|
||||
"anyhow",
|
||||
"astral-tokio-tar",
|
||||
@@ -3926,7 +3926,6 @@ dependencies = [
|
||||
"aws-sdk-s3",
|
||||
"aws-sdk-sts",
|
||||
"aws-smithy-http-client",
|
||||
"aws-smithy-types",
|
||||
"base64-simd",
|
||||
"bytes",
|
||||
"chrono",
|
||||
@@ -3944,7 +3943,6 @@ dependencies = [
|
||||
"hyper-util",
|
||||
"local-ip-address",
|
||||
"md-5 0.11.0",
|
||||
"minlz",
|
||||
"opentelemetry-proto",
|
||||
"prost 0.14.4",
|
||||
"rand 0.10.2",
|
||||
@@ -4210,7 +4208,7 @@ source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "39cab71617ae0d63f51a36d69f866391735b51691dbda63cf6f96d042b63efeb"
|
||||
dependencies = [
|
||||
"libc",
|
||||
"windows-sys 0.61.2",
|
||||
"windows-sys 0.52.0",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
@@ -5039,9 +5037,9 @@ checksum = "2304e00983f87ffb38b55b444b5e3b60a884b5d30c0fca7d82fe33449bbe55ea"
|
||||
|
||||
[[package]]
|
||||
name = "hermit-abi"
|
||||
version = "0.5.3"
|
||||
version = "0.5.2"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "e17592d60ebacc7d5e169f4663c5f84f9161cc90328abcfe8456f41e4dfcb284"
|
||||
checksum = "fc0fef456e4baa96da950455cd02c081ca953b141298e41db3fc7e36b1da849c"
|
||||
|
||||
[[package]]
|
||||
name = "hex"
|
||||
@@ -5668,7 +5666,7 @@ checksum = "3640c1c38b8e4e43584d8df18be5fc6b0aa314ce6ebf51b53313d4306cca8e46"
|
||||
dependencies = [
|
||||
"hermit-abi",
|
||||
"libc",
|
||||
"windows-sys 0.61.2",
|
||||
"windows-sys 0.52.0",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
@@ -6650,11 +6648,10 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "mysql_async"
|
||||
version = "0.37.1"
|
||||
version = "0.37.0"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "40d11da0e2d9fad4640c9f9198ee431c6d68444568f83ef1f10f3367270071e4"
|
||||
checksum = "3519e91b0d254ac1ffa495bc42053286cb2172ad7241d5b3b1b9f8a891f21ee2"
|
||||
dependencies = [
|
||||
"arc-swap",
|
||||
"bytes",
|
||||
"crossbeam-queue",
|
||||
"crossbeam-utils",
|
||||
@@ -6987,9 +6984,9 @@ checksum = "a3c00a0c9600379bd32f8972de90676a7672cba3bf4886986bc05902afc1e093"
|
||||
|
||||
[[package]]
|
||||
name = "nvml-wrapper"
|
||||
version = "0.13.0"
|
||||
version = "0.12.1"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "d164abbde0b3c03edb9edb9cb8d31a7f5b79015c692b7c771f6e0840e9106b9f"
|
||||
checksum = "f049ae562349fefb8e837eb15443da1e7c6dcbd8a11f52a228f92220c2e5c85e"
|
||||
dependencies = [
|
||||
"bitflags 2.13.1",
|
||||
"libloading",
|
||||
@@ -7001,9 +6998,9 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "nvml-wrapper-sys"
|
||||
version = "0.10.0"
|
||||
version = "0.9.1"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "5d2079f4c9b6d2170bfb71c6355734ead6c47da75c179847395c31f9f2f66ede"
|
||||
checksum = "6b4d594420fcda43b1c2c4bd44d48974aa3c7a9ab2cbf10dc18e35265767bf0b"
|
||||
dependencies = [
|
||||
"libloading",
|
||||
]
|
||||
@@ -7014,7 +7011,7 @@ version = "5.0.0"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "51e219e79014df21a225b1860a479e2dcd7cbd9130f4defd4bd0e191ea31d67d"
|
||||
dependencies = [
|
||||
"base64 0.21.7",
|
||||
"base64 0.22.1",
|
||||
"chrono",
|
||||
"getrandom 0.2.17",
|
||||
"http 1.5.0",
|
||||
@@ -8037,9 +8034,9 @@ checksum = "439ee305def115ba05938db6eb1644ff94165c5ab5e9420d1c1bcedbba909391"
|
||||
|
||||
[[package]]
|
||||
name = "ppmd-rust"
|
||||
version = "1.4.1"
|
||||
version = "1.4.0"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "9e9219bcb9d7aca6b2f63c83cf100cf78bcd619ac46e6ecbd0dd90869a39345d"
|
||||
checksum = "efca4c95a19a79d1c98f791f10aebd5c1363b473244630bb7dbde1dc98455a24"
|
||||
|
||||
[[package]]
|
||||
name = "ppv-lite86"
|
||||
@@ -8241,7 +8238,7 @@ version = "0.13.5"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "be769465445e8c1474e9c5dac2018218498557af32d9ed057325ec9a41ae81bf"
|
||||
dependencies = [
|
||||
"heck 0.4.1",
|
||||
"heck 0.5.0",
|
||||
"itertools 0.14.0",
|
||||
"log",
|
||||
"multimap",
|
||||
@@ -8261,7 +8258,7 @@ version = "0.14.4"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "03da047801ff44bb6a4d407d4860c05fd70bb81714e6b2f3812603d5b145b042"
|
||||
dependencies = [
|
||||
"heck 0.4.1",
|
||||
"heck 0.5.0",
|
||||
"itertools 0.14.0",
|
||||
"log",
|
||||
"multimap",
|
||||
@@ -8611,7 +8608,7 @@ dependencies = [
|
||||
"once_cell",
|
||||
"socket2",
|
||||
"tracing",
|
||||
"windows-sys 0.61.2",
|
||||
"windows-sys 0.52.0",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
@@ -9387,7 +9384,7 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "rustfs"
|
||||
version = "1.0.0-rc.5"
|
||||
version = "1.0.0-rc.4"
|
||||
dependencies = [
|
||||
"aes-gcm",
|
||||
"anyhow",
|
||||
@@ -9462,7 +9459,6 @@ dependencies = [
|
||||
"rustfs-io-metrics",
|
||||
"rustfs-keystone",
|
||||
"rustfs-kms",
|
||||
"rustfs-license",
|
||||
"rustfs-lock",
|
||||
"rustfs-log-analyzer",
|
||||
"rustfs-madmin",
|
||||
@@ -9514,7 +9510,7 @@ dependencies = [
|
||||
"tokio-util",
|
||||
"tonic",
|
||||
"tower",
|
||||
"tower-http 0.7.1",
|
||||
"tower-http 0.7.0",
|
||||
"tracing",
|
||||
"tracing-opentelemetry",
|
||||
"tracing-subscriber",
|
||||
@@ -9529,7 +9525,7 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "rustfs-audit"
|
||||
version = "1.0.0-rc.5"
|
||||
version = "1.0.0-rc.4"
|
||||
dependencies = [
|
||||
"const-str",
|
||||
"futures",
|
||||
@@ -9551,7 +9547,7 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "rustfs-checksums"
|
||||
version = "1.0.0-rc.5"
|
||||
version = "1.0.0-rc.4"
|
||||
dependencies = [
|
||||
"base64-simd",
|
||||
"bytes",
|
||||
@@ -9567,7 +9563,7 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "rustfs-common"
|
||||
version = "1.0.0-rc.5"
|
||||
version = "1.0.0-rc.4"
|
||||
dependencies = [
|
||||
"hotpath",
|
||||
"metrics",
|
||||
@@ -9580,7 +9576,7 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "rustfs-concurrency"
|
||||
version = "1.0.0-rc.5"
|
||||
version = "1.0.0-rc.4"
|
||||
dependencies = [
|
||||
"hotpath",
|
||||
"insta",
|
||||
@@ -9593,7 +9589,7 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "rustfs-config"
|
||||
version = "1.0.0-rc.5"
|
||||
version = "1.0.0-rc.4"
|
||||
dependencies = [
|
||||
"const-str",
|
||||
"hotpath",
|
||||
@@ -9603,7 +9599,7 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "rustfs-credentials"
|
||||
version = "1.0.0-rc.5"
|
||||
version = "1.0.0-rc.4"
|
||||
dependencies = [
|
||||
"base64-simd",
|
||||
"hmac 0.13.0",
|
||||
@@ -9617,7 +9613,7 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "rustfs-crypto"
|
||||
version = "1.0.0-rc.5"
|
||||
version = "1.0.0-rc.4"
|
||||
dependencies = [
|
||||
"aes-gcm",
|
||||
"argon2",
|
||||
@@ -9638,7 +9634,7 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "rustfs-data-usage"
|
||||
version = "1.0.0-rc.5"
|
||||
version = "1.0.0-rc.4"
|
||||
dependencies = [
|
||||
"hotpath",
|
||||
"rmp-serde",
|
||||
@@ -9648,7 +9644,7 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "rustfs-ecstore"
|
||||
version = "1.0.0-rc.5"
|
||||
version = "1.0.0-rc.4"
|
||||
dependencies = [
|
||||
"arc-swap",
|
||||
"async-channel",
|
||||
@@ -9783,7 +9779,7 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "rustfs-extension-schema"
|
||||
version = "1.0.0-rc.5"
|
||||
version = "1.0.0-rc.4"
|
||||
dependencies = [
|
||||
"hotpath",
|
||||
"serde",
|
||||
@@ -9793,7 +9789,7 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "rustfs-filemeta"
|
||||
version = "1.0.0-rc.5"
|
||||
version = "1.0.0-rc.4"
|
||||
dependencies = [
|
||||
"arc-swap",
|
||||
"byteorder",
|
||||
@@ -9820,7 +9816,7 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "rustfs-heal"
|
||||
version = "1.0.0-rc.5"
|
||||
version = "1.0.0-rc.4"
|
||||
dependencies = [
|
||||
"async-trait",
|
||||
"base64-simd",
|
||||
@@ -9856,7 +9852,7 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "rustfs-heal-contracts"
|
||||
version = "1.0.0-rc.5"
|
||||
version = "1.0.0-rc.4"
|
||||
dependencies = [
|
||||
"serde",
|
||||
"serde_json",
|
||||
@@ -9866,7 +9862,7 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "rustfs-iam"
|
||||
version = "1.0.0-rc.5"
|
||||
version = "1.0.0-rc.4"
|
||||
dependencies = [
|
||||
"arc-swap",
|
||||
"async-trait",
|
||||
@@ -9914,7 +9910,7 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "rustfs-io-core"
|
||||
version = "1.0.0-rc.5"
|
||||
version = "1.0.0-rc.4"
|
||||
dependencies = [
|
||||
"bytes",
|
||||
"hotpath",
|
||||
@@ -9926,7 +9922,7 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "rustfs-io-metrics"
|
||||
version = "1.0.0-rc.5"
|
||||
version = "1.0.0-rc.4"
|
||||
dependencies = [
|
||||
"criterion",
|
||||
"hotpath",
|
||||
@@ -9990,7 +9986,7 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "rustfs-keystone"
|
||||
version = "1.0.0-rc.5"
|
||||
version = "1.0.0-rc.4"
|
||||
dependencies = [
|
||||
"bytes",
|
||||
"futures",
|
||||
@@ -10017,7 +10013,7 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "rustfs-kms"
|
||||
version = "1.0.0-rc.5"
|
||||
version = "1.0.0-rc.4"
|
||||
dependencies = [
|
||||
"aes-gcm",
|
||||
"anyhow",
|
||||
@@ -10065,16 +10061,9 @@ dependencies = [
|
||||
"zeroize",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "rustfs-license"
|
||||
version = "1.0.0-rc.5"
|
||||
dependencies = [
|
||||
"thiserror 2.0.20",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "rustfs-lifecycle"
|
||||
version = "1.0.0-rc.5"
|
||||
version = "1.0.0-rc.4"
|
||||
dependencies = [
|
||||
"async-trait",
|
||||
"hotpath",
|
||||
@@ -10097,7 +10086,7 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "rustfs-lock"
|
||||
version = "1.0.0-rc.5"
|
||||
version = "1.0.0-rc.4"
|
||||
dependencies = [
|
||||
"async-trait",
|
||||
"compact_str",
|
||||
@@ -10120,7 +10109,7 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "rustfs-log-analyzer"
|
||||
version = "1.0.0-rc.5"
|
||||
version = "1.0.0-rc.4"
|
||||
dependencies = [
|
||||
"chrono",
|
||||
"flate2",
|
||||
@@ -10139,7 +10128,7 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "rustfs-madmin"
|
||||
version = "1.0.0-rc.5"
|
||||
version = "1.0.0-rc.4"
|
||||
dependencies = [
|
||||
"hotpath",
|
||||
"http 1.5.0",
|
||||
@@ -10177,7 +10166,7 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "rustfs-notify"
|
||||
version = "1.0.0-rc.5"
|
||||
version = "1.0.0-rc.4"
|
||||
dependencies = [
|
||||
"arc-swap",
|
||||
"async-trait",
|
||||
@@ -10212,7 +10201,7 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "rustfs-object-capacity"
|
||||
version = "1.0.0-rc.5"
|
||||
version = "1.0.0-rc.4"
|
||||
dependencies = [
|
||||
"criterion",
|
||||
"futures",
|
||||
@@ -10231,7 +10220,7 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "rustfs-object-data-cache"
|
||||
version = "1.0.0-rc.5"
|
||||
version = "1.0.0-rc.4"
|
||||
dependencies = [
|
||||
"bytes",
|
||||
"criterion",
|
||||
@@ -10248,7 +10237,7 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "rustfs-obs"
|
||||
version = "1.0.0-rc.5"
|
||||
version = "1.0.0-rc.4"
|
||||
dependencies = [
|
||||
"chrono",
|
||||
"crossbeam-channel",
|
||||
@@ -10306,7 +10295,7 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "rustfs-policy"
|
||||
version = "1.0.0-rc.5"
|
||||
version = "1.0.0-rc.4"
|
||||
dependencies = [
|
||||
"async-trait",
|
||||
"base64-simd",
|
||||
@@ -10337,7 +10326,7 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "rustfs-protocols"
|
||||
version = "1.0.0-rc.5"
|
||||
version = "1.0.0-rc.4"
|
||||
dependencies = [
|
||||
"astral-tokio-tar",
|
||||
"async-compression",
|
||||
@@ -10399,7 +10388,7 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "rustfs-protos"
|
||||
version = "1.0.0-rc.5"
|
||||
version = "1.0.0-rc.4"
|
||||
dependencies = [
|
||||
"flatbuffers",
|
||||
"hotpath",
|
||||
@@ -10424,7 +10413,7 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "rustfs-replication"
|
||||
version = "1.0.0-rc.5"
|
||||
version = "1.0.0-rc.4"
|
||||
dependencies = [
|
||||
"byteorder",
|
||||
"bytes",
|
||||
@@ -10442,7 +10431,7 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "rustfs-rio"
|
||||
version = "1.0.0-rc.5"
|
||||
version = "1.0.0-rc.4"
|
||||
dependencies = [
|
||||
"aes-gcm",
|
||||
"arc-swap",
|
||||
@@ -10459,7 +10448,6 @@ dependencies = [
|
||||
"hyper",
|
||||
"hyper-util",
|
||||
"md-5 0.11.0",
|
||||
"minlz",
|
||||
"pin-project-lite",
|
||||
"rand 0.10.2",
|
||||
"reqwest",
|
||||
@@ -10483,7 +10471,7 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "rustfs-rio-v2"
|
||||
version = "1.0.0-rc.5"
|
||||
version = "1.0.0-rc.4"
|
||||
dependencies = [
|
||||
"aes-gcm",
|
||||
"bytes",
|
||||
@@ -10506,7 +10494,7 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "rustfs-s3-client"
|
||||
version = "1.0.0-rc.5"
|
||||
version = "1.0.0-rc.4"
|
||||
dependencies = [
|
||||
"base64-simd",
|
||||
"bytes",
|
||||
@@ -10550,7 +10538,7 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "rustfs-s3-ops"
|
||||
version = "1.0.0-rc.5"
|
||||
version = "1.0.0-rc.4"
|
||||
dependencies = [
|
||||
"hotpath",
|
||||
"rustfs-s3-types",
|
||||
@@ -10558,7 +10546,7 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "rustfs-s3-types"
|
||||
version = "1.0.0-rc.5"
|
||||
version = "1.0.0-rc.4"
|
||||
dependencies = [
|
||||
"hotpath",
|
||||
"serde",
|
||||
@@ -10567,16 +10555,12 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "rustfs-s3select-api"
|
||||
version = "1.0.0-rc.5"
|
||||
version = "1.0.0-rc.4"
|
||||
dependencies = [
|
||||
"arc-swap",
|
||||
"async-compression",
|
||||
"async-trait",
|
||||
"bytes",
|
||||
"chrono",
|
||||
"crc-fast",
|
||||
"datafusion",
|
||||
"flate2",
|
||||
"futures",
|
||||
"futures-core",
|
||||
"hotpath",
|
||||
@@ -10592,7 +10576,6 @@ dependencies = [
|
||||
"serial_test",
|
||||
"thiserror 2.0.20",
|
||||
"tokio",
|
||||
"tokio-stream",
|
||||
"tokio-util",
|
||||
"tracing",
|
||||
"transform-stream",
|
||||
@@ -10602,7 +10585,7 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "rustfs-s3select-query"
|
||||
version = "1.0.0-rc.5"
|
||||
version = "1.0.0-rc.4"
|
||||
dependencies = [
|
||||
"async-recursion",
|
||||
"async-trait",
|
||||
@@ -10615,15 +10598,13 @@ dependencies = [
|
||||
"rustfs-s3select-api",
|
||||
"rustfs-test-utils",
|
||||
"s3s",
|
||||
"serde",
|
||||
"serde_json",
|
||||
"tokio",
|
||||
"tracing",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "rustfs-scanner"
|
||||
version = "1.0.0-rc.5"
|
||||
version = "1.0.0-rc.4"
|
||||
dependencies = [
|
||||
"async-trait",
|
||||
"bytes",
|
||||
@@ -10666,7 +10647,7 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "rustfs-scanner-contracts"
|
||||
version = "1.0.0-rc.5"
|
||||
version = "1.0.0-rc.4"
|
||||
dependencies = [
|
||||
"chrono",
|
||||
"jiff",
|
||||
@@ -10681,7 +10662,7 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "rustfs-security-governance"
|
||||
version = "1.0.0-rc.5"
|
||||
version = "1.0.0-rc.4"
|
||||
dependencies = [
|
||||
"hotpath",
|
||||
"thiserror 2.0.20",
|
||||
@@ -10689,7 +10670,7 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "rustfs-signer"
|
||||
version = "1.0.0-rc.5"
|
||||
version = "1.0.0-rc.4"
|
||||
dependencies = [
|
||||
"base64-simd",
|
||||
"bytes",
|
||||
@@ -10707,7 +10688,7 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "rustfs-storage-api"
|
||||
version = "1.0.0-rc.5"
|
||||
version = "1.0.0-rc.4"
|
||||
dependencies = [
|
||||
"async-trait",
|
||||
"hotpath",
|
||||
@@ -10722,7 +10703,7 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "rustfs-targets"
|
||||
version = "1.0.0-rc.5"
|
||||
version = "1.0.0-rc.4"
|
||||
dependencies = [
|
||||
"arc-swap",
|
||||
"async-nats",
|
||||
@@ -10776,7 +10757,7 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "rustfs-test-utils"
|
||||
version = "1.0.0-rc.5"
|
||||
version = "1.0.0-rc.4"
|
||||
dependencies = [
|
||||
"hotpath",
|
||||
"rustfs-data-usage",
|
||||
@@ -10792,7 +10773,7 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "rustfs-tls-runtime"
|
||||
version = "1.0.0-rc.5"
|
||||
version = "1.0.0-rc.4"
|
||||
dependencies = [
|
||||
"arc-swap",
|
||||
"hotpath",
|
||||
@@ -10813,7 +10794,7 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "rustfs-trusted-proxies"
|
||||
version = "1.0.0-rc.5"
|
||||
version = "1.0.0-rc.4"
|
||||
dependencies = [
|
||||
"async-trait",
|
||||
"axum",
|
||||
@@ -10850,7 +10831,7 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "rustfs-utils"
|
||||
version = "1.0.0-rc.5"
|
||||
version = "1.0.0-rc.4"
|
||||
dependencies = [
|
||||
"base64-simd",
|
||||
"blake2",
|
||||
@@ -10892,11 +10873,10 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "rustfs-zip"
|
||||
version = "1.0.0-rc.5"
|
||||
version = "1.0.0-rc.4"
|
||||
dependencies = [
|
||||
"async-compression",
|
||||
"hotpath",
|
||||
"rustfs-rio",
|
||||
"thiserror 2.0.20",
|
||||
"tokio",
|
||||
]
|
||||
@@ -10954,7 +10934,7 @@ dependencies = [
|
||||
"errno",
|
||||
"libc",
|
||||
"linux-raw-sys",
|
||||
"windows-sys 0.61.2",
|
||||
"windows-sys 0.52.0",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
@@ -11027,7 +11007,7 @@ dependencies = [
|
||||
"security-framework",
|
||||
"security-framework-sys",
|
||||
"webpki-root-certs",
|
||||
"windows-sys 0.61.2",
|
||||
"windows-sys 0.52.0",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
@@ -11084,7 +11064,7 @@ checksum = "9774ba4a74de5f7b1c1451ed6cd5285a32eddb5cccb8cc655a4e50009e06477f"
|
||||
[[package]]
|
||||
name = "s3s"
|
||||
version = "0.15.0"
|
||||
source = "git+https://github.com/rustfs/s3s.git?rev=28e9ebb23dd2fb7d667084f34121b4aa4807a5c6#28e9ebb23dd2fb7d667084f34121b4aa4807a5c6"
|
||||
source = "git+https://github.com/rustfs/s3s.git?rev=9c4690d8e73fc8d184031a19b2c4539ebc77d180#9c4690d8e73fc8d184031a19b2c4539ebc77d180"
|
||||
dependencies = [
|
||||
"arc-swap",
|
||||
"arrayvec",
|
||||
@@ -11115,7 +11095,6 @@ dependencies = [
|
||||
"pin-project-lite",
|
||||
"quick-xml",
|
||||
"regex",
|
||||
"s3s-rfc2047",
|
||||
"s3s-sigv2",
|
||||
"s3s-sigv4",
|
||||
"serde",
|
||||
@@ -11139,45 +11118,28 @@ dependencies = [
|
||||
"zeroize",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "s3s-rfc2047"
|
||||
version = "0.16.0-alpha.1"
|
||||
source = "git+https://github.com/rustfs/s3s.git?rev=28e9ebb23dd2fb7d667084f34121b4aa4807a5c6#28e9ebb23dd2fb7d667084f34121b4aa4807a5c6"
|
||||
dependencies = [
|
||||
"base64-simd",
|
||||
"thiserror 2.0.20",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "s3s-sigv2"
|
||||
version = "0.16.0-alpha.1"
|
||||
source = "git+https://github.com/rustfs/s3s.git?rev=28e9ebb23dd2fb7d667084f34121b4aa4807a5c6#28e9ebb23dd2fb7d667084f34121b4aa4807a5c6"
|
||||
source = "git+https://github.com/rustfs/s3s.git?rev=9c4690d8e73fc8d184031a19b2c4539ebc77d180#9c4690d8e73fc8d184031a19b2c4539ebc77d180"
|
||||
dependencies = [
|
||||
"base64-simd",
|
||||
"hmac 0.13.0",
|
||||
"jiff",
|
||||
"sha1 0.11.0",
|
||||
"smallvec",
|
||||
"thiserror 2.0.20",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "s3s-sigv4"
|
||||
version = "0.16.0-alpha.1"
|
||||
source = "git+https://github.com/rustfs/s3s.git?rev=28e9ebb23dd2fb7d667084f34121b4aa4807a5c6#28e9ebb23dd2fb7d667084f34121b4aa4807a5c6"
|
||||
source = "git+https://github.com/rustfs/s3s.git?rev=9c4690d8e73fc8d184031a19b2c4539ebc77d180#9c4690d8e73fc8d184031a19b2c4539ebc77d180"
|
||||
dependencies = [
|
||||
"arrayvec",
|
||||
"base64-simd",
|
||||
"hex-simd",
|
||||
"hmac 0.13.0",
|
||||
"jiff",
|
||||
"nom 8.0.0",
|
||||
"serde",
|
||||
"sha2 0.11.0",
|
||||
"smallvec",
|
||||
"std-next",
|
||||
"thiserror 2.0.20",
|
||||
"zeroize",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
@@ -12069,9 +12031,9 @@ checksum = "13c2bddecc57b384dee18652358fb23172facb8a2c51ccc10d74c157bdea3292"
|
||||
|
||||
[[package]]
|
||||
name = "suppaftp"
|
||||
version = "11.0.0"
|
||||
version = "10.0.2"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "46c5095831abc0d7944a2d50d6ec6abcd75b9d165d9377deb3e45798cae2343a"
|
||||
checksum = "821001051ea3d12a60fb790b8c7cb9a6f5f8698dcfdca4cd533a025fefb0b5b8"
|
||||
dependencies = [
|
||||
"async-trait",
|
||||
"chrono",
|
||||
@@ -12262,10 +12224,10 @@ source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "32497e9a4c7b38532efcdebeef879707aa9f794296a4f0244f6f69e9bc8574bd"
|
||||
dependencies = [
|
||||
"fastrand",
|
||||
"getrandom 0.4.3",
|
||||
"getrandom 0.3.4",
|
||||
"once_cell",
|
||||
"rustix",
|
||||
"windows-sys 0.61.2",
|
||||
"windows-sys 0.52.0",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
@@ -12758,9 +12720,9 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "tower-http"
|
||||
version = "0.7.1"
|
||||
version = "0.7.0"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "08a05a66a4fdd61cbbe0a1d755ffe0ca6aba159dd4820936a0ff8a8278245b9c"
|
||||
checksum = "b11f75e912b0c2be01b63d8cf8057b8c3f97cf34abb3d431a3a4c8675498e233"
|
||||
dependencies = [
|
||||
"async-compression",
|
||||
"bitflags 2.13.1",
|
||||
@@ -13384,7 +13346,7 @@ version = "0.1.11"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "c2a7b1c03c876122aa43f3020e6c3c3ee5c05081c9a00739faf7503aeba10d22"
|
||||
dependencies = [
|
||||
"windows-sys 0.61.2",
|
||||
"windows-sys 0.52.0",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
|
||||
+57
-60
@@ -29,7 +29,6 @@ members = [
|
||||
"crates/heal-contracts", # Heal request/response channel contracts
|
||||
"crates/iam", # Identity and Access Management
|
||||
"crates/keystone", # OpenStack Keystone integration
|
||||
"crates/license", # License and entitlement provider contracts
|
||||
"crates/lifecycle", # Lifecycle rule evaluation contracts
|
||||
"crates/kms", # Key Management Service
|
||||
"crates/lock", # Distributed locking implementation
|
||||
@@ -72,8 +71,8 @@ resolver = "3"
|
||||
edition = "2024"
|
||||
license = "Apache-2.0"
|
||||
repository = "https://github.com/rustfs/rustfs"
|
||||
rust-version = "1.98.0"
|
||||
version = "1.0.0-rc.5"
|
||||
rust-version = "1.97.1"
|
||||
version = "1.0.0-rc.4"
|
||||
homepage = "https://rustfs.com"
|
||||
description = "RustFS is a high-performance distributed object storage software built using Rust, one of the most popular languages worldwide. "
|
||||
keywords = ["RustFS", "Minio", "object-storage", "filesystem", "s3"]
|
||||
@@ -90,61 +89,60 @@ redundant_clone = "warn"
|
||||
|
||||
[workspace.dependencies]
|
||||
# RustFS Internal Crates
|
||||
rustfs = { path = "./rustfs", version = "1.0.0-rc.5" }
|
||||
rustfs-heal = { path = "crates/heal", version = "1.0.0-rc.5" }
|
||||
rustfs-heal-contracts = { path = "crates/heal-contracts", version = "1.0.0-rc.5" }
|
||||
rustfs-scanner-contracts = { path = "crates/scanner-contracts", version = "1.0.0-rc.5" }
|
||||
rustfs-audit = { path = "crates/audit", version = "1.0.0-rc.5" }
|
||||
rustfs-checksums = { path = "crates/checksums", version = "1.0.0-rc.5" }
|
||||
rustfs-common = { path = "crates/common", version = "1.0.0-rc.5" }
|
||||
rustfs-data-usage = { path = "crates/data-usage", version = "1.0.0-rc.5" }
|
||||
rustfs-config = { path = "./crates/config", version = "1.0.0-rc.5" }
|
||||
rustfs-concurrency = { path = "./crates/concurrency", version = "1.0.0-rc.5" }
|
||||
rustfs-credentials = { path = "crates/credentials", version = "1.0.0-rc.5" }
|
||||
rustfs-crypto = { path = "crates/crypto", version = "1.0.0-rc.5" }
|
||||
rustfs-ecstore = { path = "crates/ecstore", version = "1.0.0-rc.5" }
|
||||
rustfs-filemeta = { path = "crates/filemeta", version = "1.0.0-rc.5" }
|
||||
rustfs-iam = { path = "crates/iam", version = "1.0.0-rc.5" }
|
||||
rustfs-keystone = { path = "crates/keystone", version = "1.0.0-rc.5" }
|
||||
rustfs-license = { path = "crates/license", version = "1.0.0-rc.5" }
|
||||
rustfs-lifecycle = { path = "crates/lifecycle", version = "1.0.0-rc.5" }
|
||||
rustfs-kms = { path = "crates/kms", version = "1.0.0-rc.5" }
|
||||
rustfs-lock = { path = "crates/lock", version = "1.0.0-rc.5" }
|
||||
rustfs-madmin = { path = "crates/madmin", version = "1.0.0-rc.5" }
|
||||
rustfs-notify = { path = "crates/notify", version = "1.0.0-rc.5" }
|
||||
rustfs-io-metrics = { path = "crates/io-metrics", version = "1.0.0-rc.5" }
|
||||
rustfs-io-core = { path = "crates/io-core", version = "1.0.0-rc.5" }
|
||||
rustfs-object-capacity = { path = "crates/object-capacity", version = "1.0.0-rc.5" }
|
||||
rustfs-object-data-cache = { path = "crates/object-data-cache", version = "1.0.0-rc.5", default-features = false }
|
||||
rustfs-log-analyzer = { path = "crates/log-analyzer", version = "1.0.0-rc.5" }
|
||||
rustfs-obs = { path = "crates/obs", version = "1.0.0-rc.5" }
|
||||
rustfs-policy = { path = "crates/policy", version = "1.0.0-rc.5" }
|
||||
rustfs-protos = { path = "crates/protos", version = "1.0.0-rc.5" }
|
||||
rustfs-protocols = { path = "crates/protocols", version = "1.0.0-rc.5" }
|
||||
rustfs-replication = { path = "crates/replication", version = "1.0.0-rc.5" }
|
||||
rustfs-rio = { path = "crates/rio", version = "1.0.0-rc.5" }
|
||||
rustfs-rio-v2 = { path = "crates/rio-v2", version = "1.0.0-rc.5" }
|
||||
rustfs-s3-client = { path = "crates/s3-client", version = "1.0.0-rc.5" }
|
||||
rustfs-s3-types = { path = "crates/s3-types", version = "1.0.0-rc.5" }
|
||||
rustfs-s3-ops = { path = "crates/s3-ops", version = "1.0.0-rc.5" }
|
||||
rustfs-s3select-api = { path = "crates/s3select-api", version = "1.0.0-rc.5" }
|
||||
rustfs-s3select-query = { path = "crates/s3select-query", version = "1.0.0-rc.5" }
|
||||
rustfs-scanner = { path = "crates/scanner", version = "1.0.0-rc.5" }
|
||||
rustfs-security-governance = { path = "crates/security-governance", version = "1.0.0-rc.5" }
|
||||
rustfs-extension-schema = { path = "crates/extension-schema", version = "1.0.0-rc.5" }
|
||||
rustfs-signer = { path = "crates/signer", version = "1.0.0-rc.5" }
|
||||
rustfs-storage-api = { path = "crates/storage-api", version = "1.0.0-rc.5" }
|
||||
rustfs-trusted-proxies = { path = "crates/trusted-proxies", version = "1.0.0-rc.5" }
|
||||
rustfs-targets = { path = "crates/targets", version = "1.0.0-rc.5" }
|
||||
rustfs-test-utils = { path = "crates/test-utils", version = "1.0.0-rc.5" }
|
||||
rustfs-tls-runtime = { path = "crates/tls-runtime", version = "1.0.0-rc.5" }
|
||||
rustfs-utils = { path = "crates/utils", version = "1.0.0-rc.5" }
|
||||
rustfs-zip = { path = "./crates/zip", version = "1.0.0-rc.5" }
|
||||
rustfs = { path = "./rustfs", version = "1.0.0-rc.4" }
|
||||
rustfs-heal = { path = "crates/heal", version = "1.0.0-rc.4" }
|
||||
rustfs-heal-contracts = { path = "crates/heal-contracts", version = "1.0.0-rc.4" }
|
||||
rustfs-scanner-contracts = { path = "crates/scanner-contracts", version = "1.0.0-rc.4" }
|
||||
rustfs-audit = { path = "crates/audit", version = "1.0.0-rc.4" }
|
||||
rustfs-checksums = { path = "crates/checksums", version = "1.0.0-rc.4" }
|
||||
rustfs-common = { path = "crates/common", version = "1.0.0-rc.4" }
|
||||
rustfs-data-usage = { path = "crates/data-usage", version = "1.0.0-rc.4" }
|
||||
rustfs-config = { path = "./crates/config", version = "1.0.0-rc.4" }
|
||||
rustfs-concurrency = { path = "./crates/concurrency", version = "1.0.0-rc.4" }
|
||||
rustfs-credentials = { path = "crates/credentials", version = "1.0.0-rc.4" }
|
||||
rustfs-crypto = { path = "crates/crypto", version = "1.0.0-rc.4" }
|
||||
rustfs-ecstore = { path = "crates/ecstore", version = "1.0.0-rc.4" }
|
||||
rustfs-filemeta = { path = "crates/filemeta", version = "1.0.0-rc.4" }
|
||||
rustfs-iam = { path = "crates/iam", version = "1.0.0-rc.4" }
|
||||
rustfs-keystone = { path = "crates/keystone", version = "1.0.0-rc.4" }
|
||||
rustfs-lifecycle = { path = "crates/lifecycle", version = "1.0.0-rc.4" }
|
||||
rustfs-kms = { path = "crates/kms", version = "1.0.0-rc.4" }
|
||||
rustfs-lock = { path = "crates/lock", version = "1.0.0-rc.4" }
|
||||
rustfs-madmin = { path = "crates/madmin", version = "1.0.0-rc.4" }
|
||||
rustfs-notify = { path = "crates/notify", version = "1.0.0-rc.4" }
|
||||
rustfs-io-metrics = { path = "crates/io-metrics", version = "1.0.0-rc.4" }
|
||||
rustfs-io-core = { path = "crates/io-core", version = "1.0.0-rc.4" }
|
||||
rustfs-object-capacity = { path = "crates/object-capacity", version = "1.0.0-rc.4" }
|
||||
rustfs-object-data-cache = { path = "crates/object-data-cache", version = "1.0.0-rc.4", default-features = false }
|
||||
rustfs-log-analyzer = { path = "crates/log-analyzer", version = "1.0.0-rc.4" }
|
||||
rustfs-obs = { path = "crates/obs", version = "1.0.0-rc.4" }
|
||||
rustfs-policy = { path = "crates/policy", version = "1.0.0-rc.4" }
|
||||
rustfs-protos = { path = "crates/protos", version = "1.0.0-rc.4" }
|
||||
rustfs-protocols = { path = "crates/protocols", version = "1.0.0-rc.4" }
|
||||
rustfs-replication = { path = "crates/replication", version = "1.0.0-rc.4" }
|
||||
rustfs-rio = { path = "crates/rio", version = "1.0.0-rc.4" }
|
||||
rustfs-rio-v2 = { path = "crates/rio-v2", version = "1.0.0-rc.4" }
|
||||
rustfs-s3-client = { path = "crates/s3-client", version = "1.0.0-rc.4" }
|
||||
rustfs-s3-types = { path = "crates/s3-types", version = "1.0.0-rc.4" }
|
||||
rustfs-s3-ops = { path = "crates/s3-ops", version = "1.0.0-rc.4" }
|
||||
rustfs-s3select-api = { path = "crates/s3select-api", version = "1.0.0-rc.4" }
|
||||
rustfs-s3select-query = { path = "crates/s3select-query", version = "1.0.0-rc.4" }
|
||||
rustfs-scanner = { path = "crates/scanner", version = "1.0.0-rc.4" }
|
||||
rustfs-security-governance = { path = "crates/security-governance", version = "1.0.0-rc.4" }
|
||||
rustfs-extension-schema = { path = "crates/extension-schema", version = "1.0.0-rc.4" }
|
||||
rustfs-signer = { path = "crates/signer", version = "1.0.0-rc.4" }
|
||||
rustfs-storage-api = { path = "crates/storage-api", version = "1.0.0-rc.4" }
|
||||
rustfs-trusted-proxies = { path = "crates/trusted-proxies", version = "1.0.0-rc.4" }
|
||||
rustfs-targets = { path = "crates/targets", version = "1.0.0-rc.4" }
|
||||
rustfs-test-utils = { path = "crates/test-utils", version = "1.0.0-rc.4" }
|
||||
rustfs-tls-runtime = { path = "crates/tls-runtime", version = "1.0.0-rc.4" }
|
||||
rustfs-utils = { path = "crates/utils", version = "1.0.0-rc.4" }
|
||||
rustfs-zip = { path = "./crates/zip", version = "1.0.0-rc.4" }
|
||||
|
||||
# Async Runtime and Networking
|
||||
async-channel = "2.5.0"
|
||||
async_zip = { default-features = false, version = "0.0.19" }
|
||||
mysql_async = { default-features = false, version = "0.37.1" }
|
||||
mysql_async = { default-features = false, version = "0.37" }
|
||||
async-compression = { version = "0.4.43" }
|
||||
async-recursion = "1.1.1"
|
||||
async-trait = "0.1.92"
|
||||
@@ -176,7 +174,7 @@ tonic = { version = "0.14.6" }
|
||||
tonic-prost = { version = "0.14.6" }
|
||||
tonic-prost-build = { version = "0.14.6" }
|
||||
tower = { version = "0.5.3" }
|
||||
tower-http = { version = "0.7.1" }
|
||||
tower-http = { version = "0.7.0" }
|
||||
|
||||
# Serialization and Data Formats
|
||||
apache-avro = { version = "0.22.0", features = ["snappy", "zstandard"] }
|
||||
@@ -234,8 +232,7 @@ tokio-postgres-rustls = "0.14.0"
|
||||
# Utilities and Tools
|
||||
anyhow = "1.0.104"
|
||||
arc-swap = "1.9.2"
|
||||
# RUSTFS_COMPAT_TODO(tokio-tar-extension-limits): keep the fork pin until every parser hardening used by Snowball is released upstream. Remove after astral-sh/tokio-tar#118 is merged and a published release includes extension, physical-entry, and sparse limits, cancellation-safe sparse parsing, and error-fused entry streams.
|
||||
astral-tokio-tar = { git = "https://github.com/cxymds/tokio-tar.git", rev = "603756478b7668436e464519c77ccac22a99ba96" }
|
||||
astral-tokio-tar = "0.7.0"
|
||||
atoi = "3.1.0"
|
||||
atomic_enum = "0.3.0"
|
||||
aws-config = { version = "1.11.0" }
|
||||
@@ -285,7 +282,7 @@ mime_guess = "2.0.5"
|
||||
moka = { version = "0.12.16" }
|
||||
netif = "0.1.6"
|
||||
num_cpus = { version = "1.17.0" }
|
||||
nvml-wrapper = "0.13.0"
|
||||
nvml-wrapper = "0.12.1"
|
||||
parking_lot = "0.12.5"
|
||||
path-absolutize = "4.0.1"
|
||||
percent-encoding = "2.3.2"
|
||||
@@ -307,7 +304,7 @@ rustify = { version = "0.7", default-features = false }
|
||||
rustix = { version = "1.1.4" }
|
||||
rust-embed = { version = "8.12.0" }
|
||||
rustc-hash = { version = "2.1.3" }
|
||||
s3s = { git = "https://github.com/rustfs/s3s.git", rev = "28e9ebb23dd2fb7d667084f34121b4aa4807a5c6", version = "0.15.0", features = ["minio"] }
|
||||
s3s = { git = "https://github.com/rustfs/s3s.git", rev = "9c4690d8e73fc8d184031a19b2c4539ebc77d180", version = "0.15.0", features = ["minio"] }
|
||||
serial_test = "4.0.1"
|
||||
shadow-rs = { default-features = false, version = "2.0.0" }
|
||||
siphasher = "1.0.3"
|
||||
@@ -357,7 +354,7 @@ pyroscope = { version = "2.1.1" }
|
||||
# FTP and SFTP
|
||||
libunftp = { version = "0.23.0" }
|
||||
unftp-core = "0.1.0"
|
||||
suppaftp = { version = "11.0.0" }
|
||||
suppaftp = { version = "10.0.2" }
|
||||
rcgen = { version = "0.14.10", default-features = false, features = ["aws_lc_rs", "crypto", "pem"] }
|
||||
russh = { version = "0.63.1" }
|
||||
russh-sftp = "2.4.0"
|
||||
|
||||
@@ -23,12 +23,6 @@ SHELL := $(shell which bash)
|
||||
.SHELLFLAGS = -eu -o pipefail -c
|
||||
|
||||
DOCKER_CLI ?= docker
|
||||
# Python interpreter for the repository's helper scripts. They import tomllib
|
||||
# (Python 3.11+), while macOS still ships /usr/bin/python3 at 3.9, so calls go
|
||||
# through a resolver that picks a new-enough interpreter (or falls back to uv).
|
||||
# Override with RUSTFS_PYTHON=/path/to/python3.12, or replace the resolver via
|
||||
# RUSTFS_PYTHON_BIN=<command>.
|
||||
RUSTFS_PYTHON_BIN ?= ./scripts/python_bin.sh
|
||||
IMAGE_NAME ?= rustfs:v1.0.0
|
||||
CONTAINER_NAME ?= rustfs-dev
|
||||
# Docker build configurations
|
||||
|
||||
@@ -115,7 +115,7 @@ chown -R 10001:10001 data logs
|
||||
docker run -d -p 9000:9000 -p 9001:9001 -v $(pwd)/data:/data -v $(pwd)/logs:/logs rustfs/rustfs:latest
|
||||
|
||||
# Using specific version
|
||||
docker run -d -p 9000:9000 -p 9001:9001 -v $(pwd)/data:/data -v $(pwd)/logs:/logs rustfs/rustfs:1.0.0-rc.5
|
||||
docker run -d -p 9000:9000 -p 9001:9001 -v $(pwd)/data:/data -v $(pwd)/logs:/logs rustfs/rustfs:1.0.0-rc.4
|
||||
```
|
||||
|
||||
If you use [podman](https://github.com/containers/podman) instead of docker, you can install the RustFS with the below command
|
||||
|
||||
+1
-1
@@ -112,7 +112,7 @@ chown -R 10001:10001 data logs
|
||||
docker run -d -p 9000:9000 -p 9001:9001 -v $(pwd)/data:/data -v $(pwd)/logs:/logs rustfs/rustfs:latest
|
||||
|
||||
# 使用指定版本运行
|
||||
docker run -d -p 9000:9000 -p 9001:9001 -v $(pwd)/data:/data -v $(pwd)/logs:/logs rustfs/rustfs:1.0.0-rc.5
|
||||
docker run -d -p 9000:9000 -p 9001:9001 -v $(pwd)/data:/data -v $(pwd)/logs:/logs rustfs/rustfs:1.0.0-rc.4
|
||||
```
|
||||
|
||||
如果您通过绑定挂载启用 TLS 证书目录,也请用同样方式准备该目录:
|
||||
|
||||
@@ -59,20 +59,20 @@ pub const ENV_CAPACITY_MAX_TIMEOUT: &str = "RUSTFS_CAPACITY_MAX_TIMEOUT";
|
||||
// ============================================================================
|
||||
|
||||
/// Scheduled update interval in seconds
|
||||
/// Default: 600 seconds (10 minutes)
|
||||
pub const DEFAULT_SCHEDULED_UPDATE_INTERVAL_SECS: u64 = 600;
|
||||
/// Default: 120 seconds (2 minutes)
|
||||
pub const DEFAULT_SCHEDULED_UPDATE_INTERVAL_SECS: u64 = 120;
|
||||
|
||||
/// Write trigger delay in seconds
|
||||
/// Default: 30 seconds
|
||||
pub const DEFAULT_WRITE_TRIGGER_DELAY_SECS: u64 = 30;
|
||||
/// Default: 5 seconds
|
||||
pub const DEFAULT_WRITE_TRIGGER_DELAY_SECS: u64 = 5;
|
||||
|
||||
/// Write frequency threshold (writes per minute)
|
||||
/// Default: 20 writes/minute
|
||||
pub const DEFAULT_WRITE_FREQUENCY_THRESHOLD: usize = 20;
|
||||
/// Default: 5 writes/minute
|
||||
pub const DEFAULT_WRITE_FREQUENCY_THRESHOLD: usize = 5;
|
||||
|
||||
/// Fast update threshold in seconds
|
||||
/// Default: 120 seconds
|
||||
pub const DEFAULT_FAST_UPDATE_THRESHOLD_SECS: u64 = 120;
|
||||
/// Default: 30 seconds
|
||||
pub const DEFAULT_FAST_UPDATE_THRESHOLD_SECS: u64 = 30;
|
||||
|
||||
/// Maximum files threshold for sampling
|
||||
/// Default: 200,000 files
|
||||
@@ -129,16 +129,4 @@ mod tests {
|
||||
assert_eq!(ENV_CAPACITY_MIN_TIMEOUT, "RUSTFS_CAPACITY_MIN_TIMEOUT");
|
||||
assert_eq!(ENV_CAPACITY_MAX_TIMEOUT, "RUSTFS_CAPACITY_MAX_TIMEOUT");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_capacity_default_values() {
|
||||
assert_eq!(DEFAULT_SCHEDULED_UPDATE_INTERVAL_SECS, 600);
|
||||
assert_eq!(DEFAULT_WRITE_TRIGGER_DELAY_SECS, 30);
|
||||
assert_eq!(DEFAULT_WRITE_FREQUENCY_THRESHOLD, 20);
|
||||
assert_eq!(DEFAULT_FAST_UPDATE_THRESHOLD_SECS, 120);
|
||||
assert_eq!(DEFAULT_MAX_FILES_THRESHOLD, 200_000);
|
||||
assert_eq!(DEFAULT_STAT_TIMEOUT_SECS, 3);
|
||||
assert_eq!(DEFAULT_SAMPLE_RATE, 200);
|
||||
assert_eq!(DEFAULT_CAPACITY_METRICS_INTERVAL_SECS, 600);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -198,11 +198,11 @@ pub const ENV_SCANNER_IDLE_MODE: &str = "RUSTFS_SCANNER_IDLE_MODE";
|
||||
/// Environment variable that controls scanner cache save timeout in seconds.
|
||||
/// The scanner enforces a minimum value of `1`.
|
||||
/// - Unit: seconds (u64).
|
||||
/// - Example: `export RUSTFS_SCANNER_CACHE_SAVE_TIMEOUT_SECS=14`
|
||||
/// - Example: `export RUSTFS_SCANNER_CACHE_SAVE_TIMEOUT_SECS=30`
|
||||
pub const ENV_SCANNER_CACHE_SAVE_TIMEOUT_SECS: &str = "RUSTFS_SCANNER_CACHE_SAVE_TIMEOUT_SECS";
|
||||
|
||||
/// Default scanner cache save timeout in seconds.
|
||||
pub const DEFAULT_SCANNER_CACHE_SAVE_TIMEOUT_SECS: u64 = 14;
|
||||
pub const DEFAULT_SCANNER_CACHE_SAVE_TIMEOUT_SECS: u64 = 30;
|
||||
|
||||
/// Environment variable that caps concurrent scanner set tasks.
|
||||
/// A value of `0` keeps the existing topology-based concurrency.
|
||||
|
||||
@@ -100,8 +100,7 @@ aws-sdk-s3 = { workspace = true, default-features = false, features = ["sigv4a",
|
||||
aws-sdk-sts = { workspace = true, default-features = false, features = ["default-https-client", "rt-tokio"] }
|
||||
aws-config = { workspace = true }
|
||||
aws-smithy-http-client = { workspace = true, default-features = false, features = ["rustls-aws-lc"] }
|
||||
aws-smithy-types.workspace = true
|
||||
async-compression = { workspace = true, features = ["tokio", "bzip2", "lz4", "xz"] }
|
||||
async-compression = { workspace = true, features = ["tokio", "bzip2", "xz"] }
|
||||
async-trait = { workspace = true }
|
||||
flate2.workspace = true
|
||||
http.workspace = true
|
||||
@@ -115,7 +114,6 @@ rustfs-signer.workspace = true
|
||||
# server's implementation: a shared helper could agree with a bug on both sides.
|
||||
data-encoding = { workspace = true }
|
||||
hmac = { workspace = true }
|
||||
minlz.workspace = true
|
||||
sha1 = { workspace = true }
|
||||
serde_urlencoded = { workspace = true }
|
||||
tracing = { workspace = true }
|
||||
|
||||
@@ -169,7 +169,7 @@ the same profile for membership and execution with one nightly worker.
|
||||
| `s3s-e2e` black-box | `e2e-tests` + `e2e-tests-rio-v2` jobs | **Active** (external conformance tool) |
|
||||
| ILM / lifecycle (ignored) | `test-ilm-integration-serial` lane, `-j1` | **Active** (backlog#1148 ilm-1) |
|
||||
| KMS suite | `e2e-full` job, merge queue + main | **Active** |
|
||||
| Direct and mixed-version rolling upgrades from pinned previous release | `e2e-upgrade.yml`, storage-sensitive PRs + release tags + weekly | **Active** |
|
||||
| Direct upgrade from pinned previous release | `e2e-upgrade.yml`, storage-sensitive PRs + release tags + weekly | **Active** |
|
||||
| Cluster faults (`e2e-nightly` profile) | consolidated nightly workflow | **Active** (backlog#1149 ci-7) |
|
||||
| Protocols (FTPS/WebDAV/SFTP) | consolidated nightly workflow, serial | **Active** (backlog#1149 ci-7) |
|
||||
| Replication (fast subset) | `e2e-smoke` profile, `e2e-tests` job, every PR | **Active** (backlog#1147 repl-1) |
|
||||
|
||||
@@ -27,10 +27,8 @@
|
||||
//! Readiness is established by the harness's `start()` handshake (TCP reachability
|
||||
//! plus an S3 `ListBuckets` poll) — there are no fixed sleeps.
|
||||
//!
|
||||
//! The volume-proxy smoke below also proves that the socket-level fault proxy
|
||||
//! can be installed before startup without changing the client-facing node URL.
|
||||
//! A full lock-plane partition matrix and 5GiB large-object budget remain
|
||||
//! tracked separately.
|
||||
//! Out of scope for this block (tracked separately): network fault injection
|
||||
//! (toxiproxy / socket proxy) and 5GiB large-object budgets.
|
||||
|
||||
use crate::common::{ClusterTopology, RustFSTestClusterEnvironment};
|
||||
|
||||
@@ -127,27 +125,3 @@ async fn cluster_two_pool_smoke() -> TestResult {
|
||||
put_get_roundtrip(&cluster, "twopool/object", &payload).await?;
|
||||
Ok(())
|
||||
}
|
||||
|
||||
/// A real cluster smoke for the volume FaultProxy wiring. The proxy target is
|
||||
/// not listening yet when it is created; cluster startup must still converge
|
||||
/// once the target node starts, and peer disk/RPC traffic must traverse it.
|
||||
#[tokio::test]
|
||||
async fn cluster_volume_fault_proxy_pass_smoke() -> TestResult {
|
||||
crate::common::init_logging();
|
||||
|
||||
let mut cluster = RustFSTestClusterEnvironment::with_topology(ClusterTopology::single_pool_multidrive(2, 2)).await?;
|
||||
let proxy = cluster.start_volume_proxy_for_node(0).await?;
|
||||
let proxied = proxy.local_addr().to_string();
|
||||
assert!(cluster.rustfs_volumes_arg().contains(&proxied));
|
||||
|
||||
let result: TestResult = async {
|
||||
cluster.start().await?;
|
||||
cluster.create_test_bucket(BUCKET).await?;
|
||||
let payload = vec![0x6Du8; 256 * 1024];
|
||||
put_get_roundtrip(&cluster, "volume-proxy/object", &payload).await
|
||||
}
|
||||
.await;
|
||||
|
||||
proxy.shutdown().await;
|
||||
result
|
||||
}
|
||||
|
||||
@@ -1469,18 +1469,31 @@ impl RustFSTestClusterEnvironment {
|
||||
/// times out, or cluster service readiness times out.
|
||||
pub async fn start(&mut self) -> Result<(), Box<dyn std::error::Error + Send + Sync>> {
|
||||
let binary_path = rustfs_binary_path();
|
||||
self.start_with_binary(&binary_path).await
|
||||
}
|
||||
|
||||
/// Start every cluster node with a specific RustFS binary.
|
||||
///
|
||||
/// Upgrade compatibility tests use this to initialize a cluster with a
|
||||
/// pinned previous release before replacing nodes with the workspace build.
|
||||
pub async fn start_with_binary(&mut self, binary_path: &Path) -> Result<(), Box<dyn std::error::Error + Send + Sync>> {
|
||||
let volumes_arg = self.build_volumes_arg();
|
||||
|
||||
for node_idx in 0..self.nodes.len() {
|
||||
self.spawn_node(node_idx, binary_path, &volumes_arg)?;
|
||||
for (i, node) in self.nodes.iter_mut().enumerate() {
|
||||
info!("Starting cluster node {} on {}", i, node.address);
|
||||
|
||||
let mut command = Command::new(&binary_path);
|
||||
command
|
||||
.env("RUSTFS_VOLUMES", &volumes_arg)
|
||||
.env("RUSTFS_ADDRESS", &node.address)
|
||||
.env("RUSTFS_ACCESS_KEY", &self.access_key)
|
||||
.env("RUSTFS_SECRET_KEY", &self.secret_key)
|
||||
.env("RUSTFS_CONSOLE_ENABLE", "false")
|
||||
.env("RUST_LOG", "rustfs=info,rustfs_notify=debug");
|
||||
|
||||
for (key, value) in &self.extra_env {
|
||||
command.env(key, value);
|
||||
}
|
||||
for (key, value) in &self.node_extra_env[i] {
|
||||
command.env(key, value);
|
||||
}
|
||||
capture_command_logs(&mut command, self.node_capture_log_paths[i].as_deref())?;
|
||||
|
||||
let process = command.current_dir(&node.data_dir).spawn()?;
|
||||
|
||||
node.process = Some(process);
|
||||
}
|
||||
|
||||
for (i, node) in self.nodes.iter().enumerate() {
|
||||
@@ -1496,46 +1509,20 @@ impl RustFSTestClusterEnvironment {
|
||||
|
||||
/// Start one node process using the cluster's existing volume layout.
|
||||
pub async fn start_node(&mut self, node_idx: usize) -> Result<(), Box<dyn std::error::Error + Send + Sync>> {
|
||||
let binary_path = rustfs_binary_path();
|
||||
self.start_node_from_binary(node_idx, &binary_path).await
|
||||
}
|
||||
|
||||
/// Start one stopped cluster node with a specific RustFS binary while
|
||||
/// preserving the cluster's volume layout and that node's data directory.
|
||||
pub async fn start_node_from_binary(
|
||||
&mut self,
|
||||
node_idx: usize,
|
||||
binary_path: &Path,
|
||||
) -> Result<(), Box<dyn std::error::Error + Send + Sync>> {
|
||||
let volumes_arg = self.build_volumes_arg();
|
||||
self.spawn_node(node_idx, binary_path, &volumes_arg)?;
|
||||
|
||||
self.wait_for_node_ready(&self.nodes[node_idx].address, node_idx).await?;
|
||||
self.wait_for_node_service_ready(node_idx).await?;
|
||||
Ok(())
|
||||
}
|
||||
|
||||
fn spawn_node(
|
||||
&mut self,
|
||||
node_idx: usize,
|
||||
binary_path: &Path,
|
||||
volumes_arg: &str,
|
||||
) -> Result<(), Box<dyn std::error::Error + Send + Sync>> {
|
||||
self.ensure_node_index(node_idx)?;
|
||||
if self.nodes[node_idx].process.is_some() {
|
||||
return Err(format!("cluster node {node_idx} is already running").into());
|
||||
}
|
||||
if !binary_path.is_file() {
|
||||
return Err(format!("RustFS binary does not exist: {}", binary_path.display()).into());
|
||||
}
|
||||
|
||||
let binary_path = rustfs_binary_path();
|
||||
let volumes_arg = self.build_volumes_arg();
|
||||
let log_path = self.node_capture_log_paths[node_idx].clone();
|
||||
let node = &mut self.nodes[node_idx];
|
||||
info!("Starting cluster node {} on {} with {}", node_idx, node.address, binary_path.display());
|
||||
info!("Starting cluster node {} on {}", node_idx, node.address);
|
||||
|
||||
let mut command = Command::new(binary_path);
|
||||
let mut command = Command::new(&binary_path);
|
||||
command
|
||||
.env("RUSTFS_VOLUMES", volumes_arg)
|
||||
.env("RUSTFS_VOLUMES", &volumes_arg)
|
||||
.env("RUSTFS_ADDRESS", &node.address)
|
||||
.env("RUSTFS_ACCESS_KEY", &self.access_key)
|
||||
.env("RUSTFS_SECRET_KEY", &self.secret_key)
|
||||
@@ -1552,6 +1539,9 @@ impl RustFSTestClusterEnvironment {
|
||||
|
||||
let process = command.current_dir(&node.data_dir).spawn()?;
|
||||
node.process = Some(process);
|
||||
|
||||
self.wait_for_node_ready(&self.nodes[node_idx].address, node_idx).await?;
|
||||
self.wait_for_node_service_ready(node_idx).await?;
|
||||
Ok(())
|
||||
}
|
||||
|
||||
|
||||
@@ -1,174 +0,0 @@
|
||||
// Copyright 2024 RustFS Team
|
||||
//
|
||||
// Licensed under the Apache License, Version 2.0 (the "License");
|
||||
// you may not use this file except in compliance with the License.
|
||||
// You may obtain a copy of the License at
|
||||
//
|
||||
// http://www.apache.org/licenses/LICENSE-2.0
|
||||
//
|
||||
// Unless required by applicable law or agreed to in writing, software
|
||||
// distributed under the License is distributed on an "AS IS" BASIS,
|
||||
// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||
// See the License for the specific language governing permissions and
|
||||
// limitations under the License.
|
||||
|
||||
//! Regression: an object legally committed at degraded write quorum must stay
|
||||
//! listable while a *different* drive is offline.
|
||||
//!
|
||||
//! On a 4-drive EC 2+2 set, a PUT made while one drive is down persists
|
||||
//! `xl.meta` on 3 of 4 drives (write quorum). If a different drive later goes
|
||||
//! offline before heal converges, a strict latest-listing quorum of 3 can only
|
||||
//! ever observe 2 copies, so ListObjectsV2 silently dropped the object even
|
||||
//! though GetObject (read quorum 2) still succeeded. Exposed by the flaky
|
||||
//! "Mixed-version rolling upgrade from rc.2" CI lane (run 33478999853); the
|
||||
//! product fix relaxes the listing's required object quorum by the number of
|
||||
//! set drives the listing could not consult (see
|
||||
//! `latest_listing_required_object_quorum` in
|
||||
//! `crates/ecstore/src/store/list_objects.rs`).
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use crate::common::{RustFSTestClusterEnvironment, init_logging};
|
||||
use aws_sdk_s3::Client;
|
||||
use bytes::Bytes;
|
||||
use std::collections::HashSet;
|
||||
use std::error::Error;
|
||||
use std::time::{Duration, Instant};
|
||||
use tracing::info;
|
||||
|
||||
type TestResult = Result<(), Box<dyn Error + Send + Sync>>;
|
||||
|
||||
const BUCKET: &str = "degraded-listing-availability";
|
||||
const OBJECT_COUNT: usize = 8;
|
||||
/// Well under the observed heal-convergence gap (~50s in the CI incident),
|
||||
/// so a listing that only completes after heal restores the missing copy
|
||||
/// still fails this deadline on a regressed build.
|
||||
const LISTING_DEADLINE: Duration = Duration::from_secs(25);
|
||||
const GET_RETRY_DEADLINE: Duration = Duration::from_secs(15);
|
||||
const PUT_RETRY_DEADLINE: Duration = Duration::from_secs(15);
|
||||
|
||||
fn object_key(idx: usize) -> String {
|
||||
format!("degraded-object-{idx:02}")
|
||||
}
|
||||
|
||||
async fn list_all_keys(client: &Client) -> Result<HashSet<String>, Box<dyn Error + Send + Sync>> {
|
||||
let mut keys = HashSet::new();
|
||||
let mut continuation_token: Option<String> = None;
|
||||
loop {
|
||||
let response = client
|
||||
.list_objects_v2()
|
||||
.bucket(BUCKET)
|
||||
.set_continuation_token(continuation_token.clone())
|
||||
.send()
|
||||
.await?;
|
||||
keys.extend(
|
||||
response
|
||||
.contents()
|
||||
.iter()
|
||||
.filter_map(|object| object.key().map(str::to_owned)),
|
||||
);
|
||||
match response.next_continuation_token() {
|
||||
Some(token) => continuation_token = Some(token.to_owned()),
|
||||
None => break,
|
||||
}
|
||||
}
|
||||
Ok(keys)
|
||||
}
|
||||
|
||||
/// 4-node single-drive cluster (EC 2+2, write quorum 3):
|
||||
/// 1. Stop node 1 and PUT objects — each commits on nodes {0, 2, 3} only.
|
||||
/// 2. Stop node 3 (a holder drive), then bring node 1 back before heal can
|
||||
/// recreate the missing copies there.
|
||||
/// 3. Every object still satisfies read quorum (nodes 0 and 2), so GET
|
||||
/// must succeed AND ListObjectsV2 must report every key well before
|
||||
/// heal converges.
|
||||
#[tokio::test]
|
||||
async fn degraded_write_remains_listable_while_a_different_drive_is_offline() -> TestResult {
|
||||
init_logging();
|
||||
|
||||
let mut cluster = RustFSTestClusterEnvironment::new(4).await?;
|
||||
// Listing availability must not depend on heal convergence: disable
|
||||
// the background healers so the degraded objects keep their metadata
|
||||
// on exactly 3 of 4 drives for the whole test.
|
||||
cluster.set_env("RUSTFS_HEAL_ENABLED", "false");
|
||||
cluster.set_env("RUSTFS_SCANNER_ENABLED", "false");
|
||||
cluster.start().await?;
|
||||
cluster.create_test_bucket(BUCKET).await?;
|
||||
let client = cluster.create_s3_client(0)?;
|
||||
|
||||
info!("stopping node 1 so the uploads commit at degraded write quorum (3 of 4)");
|
||||
cluster.stop_node(1)?;
|
||||
// The first writes after a node drops can see transient 503s while the
|
||||
// survivors notice the dead peer; retry briefly (overwrites of the same
|
||||
// unversioned key are idempotent).
|
||||
for idx in 0..OBJECT_COUNT {
|
||||
let key = object_key(idx);
|
||||
let body = format!("degraded listing payload {idx}");
|
||||
let deadline = Instant::now() + PUT_RETRY_DEADLINE;
|
||||
loop {
|
||||
let request = client
|
||||
.put_object()
|
||||
.bucket(BUCKET)
|
||||
.key(&key)
|
||||
.body(Bytes::from(body.clone()).into());
|
||||
match request.send().await {
|
||||
Ok(_) => break,
|
||||
Err(error) if Instant::now() < deadline => {
|
||||
info!("retrying degraded PUT for {key}: {error}");
|
||||
tokio::time::sleep(Duration::from_millis(500)).await;
|
||||
}
|
||||
Err(error) => return Err(format!("degraded PUT for {key} failed: {error}").into()),
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
info!("stopping node 3 (holds a copy) and restoring node 1 (holds none)");
|
||||
cluster.stop_node(3)?;
|
||||
cluster.start_node(1).await?;
|
||||
|
||||
// The first requests after a node drops can see transient 503s while
|
||||
// the survivors notice the dead peer; retry briefly before asserting.
|
||||
for idx in 0..OBJECT_COUNT {
|
||||
let key = object_key(idx);
|
||||
let deadline = Instant::now() + GET_RETRY_DEADLINE;
|
||||
let body = loop {
|
||||
match client.get_object().bucket(BUCKET).key(&key).send().await {
|
||||
Ok(response) => break response.body.collect().await?.into_bytes(),
|
||||
Err(error) if Instant::now() < deadline => {
|
||||
info!("retrying degraded GET for {key}: {error}");
|
||||
tokio::time::sleep(Duration::from_millis(500)).await;
|
||||
}
|
||||
Err(error) => return Err(format!("degraded object {key} failed read quorum GET: {error}").into()),
|
||||
}
|
||||
};
|
||||
assert!(!body.is_empty(), "degraded object {key} should read back at read quorum");
|
||||
}
|
||||
|
||||
let expected: HashSet<String> = (0..OBJECT_COUNT).map(object_key).collect();
|
||||
let deadline = Instant::now() + LISTING_DEADLINE;
|
||||
let listed = loop {
|
||||
let listed = match list_all_keys(&client).await {
|
||||
Ok(keys) => keys,
|
||||
Err(error) if Instant::now() < deadline => {
|
||||
info!("retrying degraded listing: {error}");
|
||||
tokio::time::sleep(Duration::from_millis(500)).await;
|
||||
continue;
|
||||
}
|
||||
Err(error) => return Err(error),
|
||||
};
|
||||
if expected.is_subset(&listed) {
|
||||
break listed;
|
||||
}
|
||||
assert!(
|
||||
Instant::now() < deadline,
|
||||
"objects readable at read quorum stayed missing from ListObjectsV2 for {LISTING_DEADLINE:?}: \
|
||||
missing={:?} listed={listed:?}",
|
||||
expected.difference(&listed).collect::<Vec<_>>(),
|
||||
);
|
||||
tokio::time::sleep(Duration::from_millis(500)).await;
|
||||
};
|
||||
info!(listed = listed.len(), "degraded objects are listable while node 3 is offline");
|
||||
|
||||
Ok(())
|
||||
}
|
||||
}
|
||||
@@ -16,14 +16,13 @@
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use crate::chaos::{VersionShardCensus, census_object_version_on_disk, signed_admin_post};
|
||||
use crate::common::{RustFSTestClusterEnvironment, RustFSTestEnvironment, admin_request, init_logging};
|
||||
use crate::chaos::signed_admin_post;
|
||||
use crate::common::{RustFSTestClusterEnvironment, RustFSTestEnvironment, init_logging};
|
||||
use aws_sdk_s3::primitives::ByteStream;
|
||||
use http::Method;
|
||||
use std::collections::HashSet;
|
||||
use std::error::Error;
|
||||
use std::path::{Path, PathBuf};
|
||||
use tokio::time::{Duration, Instant, sleep, timeout};
|
||||
use tokio::time::{Duration, sleep, timeout};
|
||||
use tracing::info;
|
||||
|
||||
fn has_file_under(path: &Path) -> bool {
|
||||
@@ -49,110 +48,6 @@ mod tests {
|
||||
disk.join(bucket).join(key).join("xl.meta").is_file()
|
||||
}
|
||||
|
||||
// Healing may rewrite non-identity bookkeeping in xl.meta. The census
|
||||
// therefore compares the canonical selected metadata fields plus every
|
||||
// physical shard, while the payload seed makes object mix-ups observable.
|
||||
#[derive(Debug)]
|
||||
struct PhysicalObjectManifest {
|
||||
key: String,
|
||||
payload_seed: u8,
|
||||
shard_census: VersionShardCensus,
|
||||
}
|
||||
|
||||
fn deterministic_object_body(len: usize, seed: u8) -> Vec<u8> {
|
||||
let mut value = seed;
|
||||
std::iter::repeat_with(|| {
|
||||
value = value.wrapping_mul(31).wrapping_add(17);
|
||||
value
|
||||
})
|
||||
.take(len)
|
||||
.collect()
|
||||
}
|
||||
|
||||
fn matching_manifest_count(
|
||||
disk: &Path,
|
||||
bucket: &str,
|
||||
expected_manifests: &[PhysicalObjectManifest],
|
||||
) -> Result<usize, Box<dyn Error + Send + Sync>> {
|
||||
let mut matching = 0;
|
||||
for expected in expected_manifests {
|
||||
let actual = census_object_version_on_disk(disk, bucket, &expected.key, None)?;
|
||||
if actual.matches_manifest(&expected.shard_census) {
|
||||
matching += 1;
|
||||
}
|
||||
}
|
||||
Ok(matching)
|
||||
}
|
||||
|
||||
fn metadata_count(disk: &Path, bucket: &str, expected_manifests: &[PhysicalObjectManifest]) -> usize {
|
||||
expected_manifests
|
||||
.iter()
|
||||
.filter(|expected| object_metadata_exists_on_disk(disk, bucket, &expected.key))
|
||||
.count()
|
||||
}
|
||||
|
||||
fn heal_task_status_diagnostic(body: &str) -> String {
|
||||
let Ok(status) = serde_json::from_str::<serde_json::Value>(body) else {
|
||||
return body.to_string();
|
||||
};
|
||||
let items = status["items"].as_array();
|
||||
let mut unresolved_states = HashSet::new();
|
||||
for item in items.into_iter().flatten() {
|
||||
for drive in item["after"]["drives"].as_array().into_iter().flatten() {
|
||||
if let Some(state) = drive["state"].as_str()
|
||||
&& state != "ok"
|
||||
{
|
||||
unresolved_states.insert(state.to_string());
|
||||
}
|
||||
}
|
||||
}
|
||||
let mut unresolved_states = unresolved_states.into_iter().collect::<Vec<_>>();
|
||||
unresolved_states.sort();
|
||||
format!(
|
||||
"summary={:?}, detail={:?}, item_count={}, unresolved_drive_states={unresolved_states:?}",
|
||||
status["summary"].as_str(),
|
||||
status["detail"].as_str(),
|
||||
items.map_or(0, Vec::len)
|
||||
)
|
||||
}
|
||||
|
||||
fn cluster_heal_is_idle(status: &serde_json::Value) -> bool {
|
||||
let operations = &status["healOperations"];
|
||||
status["clusterStatusComplete"] == serde_json::Value::Bool(true)
|
||||
&& status["state"].as_str() == Some("idle")
|
||||
&& operations["queueLength"].as_u64() == Some(0)
|
||||
&& operations["activeTasks"].as_u64() == Some(0)
|
||||
&& operations["retryingTasks"].as_u64() == Some(0)
|
||||
}
|
||||
|
||||
fn only_admin_heal_is_active(status: &serde_json::Value) -> bool {
|
||||
let operations = &status["healOperations"];
|
||||
status["clusterStatusComplete"] == serde_json::Value::Bool(true)
|
||||
&& status["state"].as_str() == Some("active")
|
||||
&& operations["queueLength"].as_u64() == Some(0)
|
||||
&& operations["activeTasks"].as_u64() == Some(1)
|
||||
&& operations["retryingTasks"].as_u64() == Some(0)
|
||||
&& operations["activeBySource"]["admin"].as_u64() == Some(1)
|
||||
}
|
||||
|
||||
async fn replacement_recovery_status(
|
||||
cluster: &RustFSTestClusterEnvironment,
|
||||
) -> Result<serde_json::Value, Box<dyn Error + Send + Sync>> {
|
||||
let (status, body) = admin_request(
|
||||
&cluster.nodes[0].url,
|
||||
Method::GET,
|
||||
"/rustfs/admin/v4/heal/replacement-recovery",
|
||||
None,
|
||||
&cluster.access_key,
|
||||
&cluster.secret_key,
|
||||
)
|
||||
.await?;
|
||||
if !status.is_success() {
|
||||
return Err(format!("replacement recovery status failed: {status} {body}").into());
|
||||
}
|
||||
serde_json::from_str(&body).map_err(|err| format!("replacement recovery status is not JSON ({err}): {body}").into())
|
||||
}
|
||||
|
||||
async fn assert_object_body(env: &RustFSTestEnvironment, bucket: &str, key: &str, expected: &[u8]) {
|
||||
let client = env.create_s3_client();
|
||||
let response = client
|
||||
@@ -547,380 +442,6 @@ mod tests {
|
||||
.into())
|
||||
}
|
||||
|
||||
// Keep the original unformatted-disk scenario above. This case retains the
|
||||
// format identity so only the explicit admin task can rebuild missing data.
|
||||
#[tokio::test(flavor = "multi_thread")]
|
||||
async fn test_cluster_root_heal_resumes_missing_remote_shards_after_node_restart() -> Result<(), Box<dyn Error + Send + Sync>>
|
||||
{
|
||||
init_logging();
|
||||
info!(
|
||||
event = "heal_restart_started",
|
||||
component = "e2e_test",
|
||||
subsystem = "heal",
|
||||
"Starting root-heal restart test"
|
||||
);
|
||||
|
||||
let mut cluster = RustFSTestClusterEnvironment::new(4).await?;
|
||||
cluster.set_env("RUSTFS_UNSAFE_BYPASS_DISK_CHECK", "true");
|
||||
cluster.set_env("RUSTFS_HEAL_ENABLED", "true");
|
||||
cluster.set_env("RUSTFS_HEAL_AUTO_HEAL_ENABLE", "false");
|
||||
cluster.set_env("RUSTFS_HEAL_MRF_ENABLE", "false");
|
||||
cluster.set_env("RUSTFS_SCANNER_ENABLED", "false");
|
||||
cluster.set_env("RUSTFS_HEAL_MAX_CONCURRENT_HEALS", "1");
|
||||
cluster.set_env("RUSTFS_HEAL_MAX_CONCURRENT_PER_SET", "1");
|
||||
cluster.set_env("RUSTFS_HEAL_PAGE_OBJECT_CONCURRENCY", "1");
|
||||
cluster.set_env("RUSTFS_HEAL_PAGE_PARALLEL_ENABLE", "false");
|
||||
// Keep all storage nodes' Heal runtimes enabled so their disk services
|
||||
// complete normal registration after restart. Scanner, auto-heal and
|
||||
// MRF are disabled; the pre-root idle barrier below drains the direct
|
||||
// outage-object repair before the explicit admin task starts.
|
||||
let server_rust_log = std::env::var("RUSTFS_HEAL_CHAOS_SERVER_RUST_LOG")
|
||||
.unwrap_or_else(|_| "rustfs::heal::task=info,rustfs=error".to_string());
|
||||
cluster.set_env("RUST_LOG", server_rust_log);
|
||||
if let Ok(log_dir) = std::env::var("RUSTFS_HEAL_CHAOS_LOG_DIR") {
|
||||
std::fs::create_dir_all(&log_dir)?;
|
||||
for node_index in 0..cluster.nodes.len() {
|
||||
cluster.set_node_capture_log_path(node_index, format!("{log_dir}/node{node_index}.log"))?;
|
||||
}
|
||||
}
|
||||
cluster.start().await?;
|
||||
let clients = cluster.create_all_clients()?;
|
||||
|
||||
let bucket = "heal-restart-during-rebuild";
|
||||
clients[0].create_bucket().bucket(bucket).send().await?;
|
||||
|
||||
let replaced_disk = PathBuf::from(&cluster.nodes[1].data_dir);
|
||||
let replacement_format_path = replaced_disk.join(".rustfs.sys").join("format.json");
|
||||
let replacement_format = std::fs::read(&replacement_format_path).map_err(|err| {
|
||||
format!("failed to capture target format before replacement wipe at {replacement_format_path:?}: {err}")
|
||||
})?;
|
||||
let online_object_count = std::env::var("RUSTFS_HEAL_CHAOS_OBJECT_COUNT")
|
||||
.ok()
|
||||
.and_then(|value| value.parse::<usize>().ok())
|
||||
.unwrap_or(24)
|
||||
.clamp(8, 64);
|
||||
let object_size_bytes = std::env::var("RUSTFS_HEAL_CHAOS_OBJECT_SIZE_BYTES")
|
||||
.ok()
|
||||
.and_then(|value| value.parse::<usize>().ok())
|
||||
.unwrap_or(4 * 1024 * 1024)
|
||||
.clamp(1024 * 1024, 16 * 1024 * 1024);
|
||||
let mut expected_manifests = Vec::with_capacity(online_object_count);
|
||||
for index in 0..online_object_count {
|
||||
let key = format!("cluster/online/object-{index:04}.bin");
|
||||
let payload_seed = u8::try_from(index + 1).expect("clamped object count must fit in u8");
|
||||
timeout(
|
||||
Duration::from_secs(30),
|
||||
clients[0]
|
||||
.put_object()
|
||||
.bucket(bucket)
|
||||
.key(&key)
|
||||
.body(ByteStream::from(deterministic_object_body(object_size_bytes, payload_seed)))
|
||||
.send(),
|
||||
)
|
||||
.await??;
|
||||
let shard_census = census_object_version_on_disk(&replaced_disk, bucket, &key, None)?;
|
||||
assert!(
|
||||
shard_census.is_complete(),
|
||||
"node 1 should hold a complete baseline shard for {key}: {shard_census:?}"
|
||||
);
|
||||
assert!(
|
||||
!shard_census.expected_part_numbers.is_empty(),
|
||||
"chaos objects must use physical part shards rather than inline data: {shard_census:?}"
|
||||
);
|
||||
expected_manifests.push(PhysicalObjectManifest {
|
||||
key,
|
||||
payload_seed,
|
||||
shard_census,
|
||||
});
|
||||
}
|
||||
|
||||
cluster.stop_node(1)?;
|
||||
std::fs::remove_dir_all(&replaced_disk)?;
|
||||
std::fs::create_dir_all(
|
||||
replacement_format_path
|
||||
.parent()
|
||||
.ok_or("replacement format path has no parent")?,
|
||||
)?;
|
||||
std::fs::write(&replacement_format_path, replacement_format)?;
|
||||
assert!(
|
||||
replacement_format_path.is_file(),
|
||||
"replacement target must retain only its preformatted topology identity"
|
||||
);
|
||||
|
||||
let outage_key = "cluster/written-while-node-down.bin";
|
||||
let outage_payload_seed = 0xf1;
|
||||
timeout(
|
||||
Duration::from_secs(30),
|
||||
clients[2]
|
||||
.put_object()
|
||||
.bucket(bucket)
|
||||
.key(outage_key)
|
||||
.body(ByteStream::from(deterministic_object_body(object_size_bytes, outage_payload_seed)))
|
||||
.send(),
|
||||
)
|
||||
.await??;
|
||||
|
||||
let mut outage_peer_erasure_indices = HashSet::new();
|
||||
for (node_index, node) in cluster.nodes.iter().enumerate() {
|
||||
if node_index == 1 {
|
||||
continue;
|
||||
}
|
||||
let census = census_object_version_on_disk(Path::new(&node.data_dir), bucket, outage_key, None)?;
|
||||
assert!(
|
||||
census.is_complete(),
|
||||
"online node {node_index} must hold a complete outage-object shard: {census:?}"
|
||||
);
|
||||
let erasure_index = census
|
||||
.erasure_index
|
||||
.ok_or_else(|| format!("online node {node_index} outage-object shard has no erasure index: {census:?}"))?;
|
||||
assert!(
|
||||
(1..=cluster.nodes.len()).contains(&erasure_index),
|
||||
"online node {node_index} outage-object erasure index is out of range: {census:?}"
|
||||
);
|
||||
assert!(
|
||||
outage_peer_erasure_indices.insert(erasure_index),
|
||||
"outage-object erasure index {erasure_index} is duplicated across online nodes"
|
||||
);
|
||||
}
|
||||
assert_eq!(
|
||||
outage_peer_erasure_indices.len(),
|
||||
cluster.nodes.len().saturating_sub(1),
|
||||
"every online node must contribute one unique outage-object erasure index"
|
||||
);
|
||||
let expected_outage_target_erasure_index = (1..=cluster.nodes.len())
|
||||
.find(|index| !outage_peer_erasure_indices.contains(index))
|
||||
.ok_or("online outage-object shards leave no erasure index for the replacement target")?;
|
||||
|
||||
// The PUT path may have admitted a direct Internal object repair while
|
||||
// node 1 was offline. Cancel the isolated bucket path before the target
|
||||
// returns; otherwise it could rebuild the outage object and invalidate
|
||||
// the explicit-root ownership assertion below.
|
||||
let cancel_outage_heal_path = format!("/rustfs/admin/v3/heal/{bucket}?forceStop=true");
|
||||
let (cancel_status, cancel_body) = admin_request(
|
||||
&cluster.nodes[0].url,
|
||||
Method::POST,
|
||||
&cancel_outage_heal_path,
|
||||
Some(
|
||||
r#"{"recursive":true,"dryRun":false,"remove":false,"recreate":true,"scanMode":2,"updateParity":false,"nolock":false}"#
|
||||
.to_string(),
|
||||
),
|
||||
&cluster.access_key,
|
||||
&cluster.secret_key,
|
||||
)
|
||||
.await?;
|
||||
if !cancel_status.is_success() {
|
||||
return Err(format!("cancel outage heal failed: {cancel_status} {cancel_body}").into());
|
||||
}
|
||||
|
||||
cluster.start_node(1).await?;
|
||||
|
||||
let status_url = format!("{}/rustfs/admin/v3/background-heal/status", cluster.nodes[0].url);
|
||||
let recovery_deadline = Instant::now() + Duration::from_secs(60);
|
||||
loop {
|
||||
let status_body = signed_admin_post(&status_url, None, &cluster.access_key, &cluster.secret_key).await?;
|
||||
assert!(
|
||||
!status_body.contains("MissingContentLength"),
|
||||
"background heal status should not fail without an explicit Content-Length: {status_body}"
|
||||
);
|
||||
let recovered: serde_json::Value = serde_json::from_str(&status_body)
|
||||
.map_err(|err| format!("background heal status is not JSON ({err}): {status_body}"))?;
|
||||
if cluster_heal_is_idle(&recovered) {
|
||||
break;
|
||||
}
|
||||
if Instant::now() >= recovery_deadline {
|
||||
return Err(format!("cluster heal operations did not become idle before root heal: {recovered}").into());
|
||||
}
|
||||
sleep(Duration::from_millis(250)).await;
|
||||
}
|
||||
assert_eq!(
|
||||
matching_manifest_count(&replaced_disk, bucket, &expected_manifests)?,
|
||||
0,
|
||||
"non-admin Heal is disabled, so the replacement target must remain empty before the explicit root heal"
|
||||
);
|
||||
assert!(
|
||||
!census_object_version_on_disk(&replaced_disk, bucket, outage_key, None)?.has_xl_meta,
|
||||
"the object written during the outage must be absent before the explicit root heal"
|
||||
);
|
||||
let pre_heal_replacement = replacement_recovery_status(&cluster).await?;
|
||||
assert_eq!(
|
||||
pre_heal_replacement["cluster"]["records"].as_array().map(Vec::len),
|
||||
Some(0),
|
||||
"isolated target must not retain an automatic replacement generation: {pre_heal_replacement}"
|
||||
);
|
||||
|
||||
let heal_body = r#"{"recursive":true,"dryRun":false,"remove":false,"recreate":true,"scanMode":2,"updateParity":false,"nolock":false}"#;
|
||||
let heal_url = format!("{}/rustfs/admin/v3/heal/?forceStart=true", cluster.nodes[0].url);
|
||||
let heal_start_body = signed_admin_post(&heal_url, Some(heal_body), &cluster.access_key, &cluster.secret_key).await?;
|
||||
let heal_start: serde_json::Value = serde_json::from_str(&heal_start_body)
|
||||
.map_err(|err| format!("heal start response is not JSON ({err}): {heal_start_body}"))?;
|
||||
let client_token = heal_start["clientToken"]
|
||||
.as_str()
|
||||
.filter(|token| !token.is_empty())
|
||||
.ok_or_else(|| format!("heal start response has no client token: {heal_start}"))?;
|
||||
let task_status_url = format!("{}/rustfs/admin/v3/heal/?clientToken={client_token}", cluster.nodes[0].url);
|
||||
|
||||
let partial_timeout_secs = std::env::var("RUSTFS_HEAL_CHAOS_PARTIAL_TIMEOUT_SECS")
|
||||
.ok()
|
||||
.and_then(|value| value.parse::<u64>().ok())
|
||||
.unwrap_or(60);
|
||||
let partial_deadline = Instant::now() + Duration::from_secs(partial_timeout_secs);
|
||||
let pre_interrupt_status = loop {
|
||||
let status_body = signed_admin_post(&status_url, None, &cluster.access_key, &cluster.secret_key).await?;
|
||||
let active_status: serde_json::Value = serde_json::from_str(&status_body)
|
||||
.map_err(|err| format!("background heal status is not JSON ({err}): {status_body}"))?;
|
||||
if only_admin_heal_is_active(&active_status) {
|
||||
break active_status;
|
||||
}
|
||||
if Instant::now() >= partial_deadline {
|
||||
return Err(format!("root heal never became active within {partial_timeout_secs}s: {active_status}").into());
|
||||
}
|
||||
sleep(Duration::from_millis(50)).await;
|
||||
};
|
||||
let partial_count = loop {
|
||||
let matching = matching_manifest_count(&replaced_disk, bucket, &expected_manifests)?;
|
||||
if matching > 0 && matching < expected_manifests.len() {
|
||||
break matching;
|
||||
}
|
||||
if matching == expected_manifests.len() {
|
||||
return Err(format!(
|
||||
"root heal rebuilt all {} baseline objects before the target could be interrupted",
|
||||
expected_manifests.len()
|
||||
)
|
||||
.into());
|
||||
}
|
||||
if Instant::now() >= partial_deadline {
|
||||
return Err(format!(
|
||||
"root heal made no observable partial progress on the replacement target within {partial_timeout_secs}s"
|
||||
)
|
||||
.into());
|
||||
}
|
||||
sleep(Duration::from_millis(10)).await;
|
||||
};
|
||||
info!(
|
||||
event = "heal_restart_checkpoint",
|
||||
component = "e2e_test",
|
||||
subsystem = "heal",
|
||||
partial_count,
|
||||
"Verified unique admin owner before target interruption"
|
||||
);
|
||||
|
||||
cluster.stop_node(1)?;
|
||||
let stopped_count = matching_manifest_count(&replaced_disk, bucket, &expected_manifests)?;
|
||||
assert!(
|
||||
stopped_count > 0 && stopped_count < expected_manifests.len(),
|
||||
"the target must stop after a partial rebuild, observed before stop={partial_count}, after stop={stopped_count}, total={}",
|
||||
expected_manifests.len()
|
||||
);
|
||||
let unclean_shutdown_marker = replaced_disk.join(".rustfs.sys").join("unclean-shutdown");
|
||||
match std::fs::remove_file(&unclean_shutdown_marker) {
|
||||
Ok(()) => {}
|
||||
Err(error) if error.kind() == std::io::ErrorKind::NotFound => {}
|
||||
Err(error) => {
|
||||
return Err(format!("failed to isolate unclean recovery marker {unclean_shutdown_marker:?}: {error}").into());
|
||||
}
|
||||
}
|
||||
cluster.start_node(1).await?;
|
||||
|
||||
let heal_timeout_secs = std::env::var("RUSTFS_HEAL_REPLACED_DISK_TIMEOUT_SECS")
|
||||
.ok()
|
||||
.and_then(|value| value.parse::<u64>().ok())
|
||||
.unwrap_or(180);
|
||||
let heal_deadline = Instant::now() + Duration::from_secs(heal_timeout_secs);
|
||||
loop {
|
||||
if metadata_count(&replaced_disk, bucket, &expected_manifests) == expected_manifests.len()
|
||||
&& object_metadata_exists_on_disk(&replaced_disk, bucket, outage_key)
|
||||
{
|
||||
let matching = matching_manifest_count(&replaced_disk, bucket, &expected_manifests)?;
|
||||
let outage_census = census_object_version_on_disk(&replaced_disk, bucket, outage_key, None)?;
|
||||
if matching == expected_manifests.len() && outage_census.is_complete() {
|
||||
break;
|
||||
}
|
||||
}
|
||||
if Instant::now() >= heal_deadline {
|
||||
let matching = matching_manifest_count(&replaced_disk, bucket, &expected_manifests)?;
|
||||
let outage_census = census_object_version_on_disk(&replaced_disk, bucket, outage_key, None)?;
|
||||
let final_status = signed_admin_post(&status_url, None, &cluster.access_key, &cluster.secret_key)
|
||||
.await
|
||||
.unwrap_or_else(|err| format!("status request failed: {err}"));
|
||||
let task_status = match timeout(
|
||||
Duration::from_secs(5),
|
||||
signed_admin_post(&task_status_url, None, &cluster.access_key, &cluster.secret_key),
|
||||
)
|
||||
.await
|
||||
{
|
||||
Ok(Ok(body)) => heal_task_status_diagnostic(&body),
|
||||
Ok(Err(err)) => format!("task status request failed: {err}"),
|
||||
Err(_) => "task status request exceeded 5s diagnostic budget".to_string(),
|
||||
};
|
||||
let replacement_status = match timeout(Duration::from_secs(5), replacement_recovery_status(&cluster)).await {
|
||||
Ok(Ok(status)) => status.to_string(),
|
||||
Ok(Err(err)) => format!("replacement status request failed: {err}"),
|
||||
Err(_) => "replacement status request exceeded 5s diagnostic budget".to_string(),
|
||||
};
|
||||
return Err(format!(
|
||||
"root heal did not resume after target restart within {heal_timeout_secs}s: baseline={matching}/{}, outage={outage_census:?}, status={final_status}, task_status={task_status}, pre_interrupt_status={pre_interrupt_status}, pre_heal_replacement={pre_heal_replacement}, replacement_status={replacement_status}",
|
||||
expected_manifests.len()
|
||||
)
|
||||
.into());
|
||||
}
|
||||
sleep(Duration::from_millis(250)).await;
|
||||
}
|
||||
|
||||
for expected in &expected_manifests {
|
||||
let actual = census_object_version_on_disk(&replaced_disk, bucket, &expected.key, None)?;
|
||||
assert!(
|
||||
actual.matches_manifest(&expected.shard_census),
|
||||
"rebuilt target shard differs from its baseline for {}: {actual:?}",
|
||||
expected.key
|
||||
);
|
||||
}
|
||||
let outage_census = census_object_version_on_disk(&replaced_disk, bucket, outage_key, None)?;
|
||||
assert!(
|
||||
outage_census.is_complete(),
|
||||
"outage object must have a complete target shard: {outage_census:?}"
|
||||
);
|
||||
assert_eq!(
|
||||
outage_census.erasure_index,
|
||||
Some(expected_outage_target_erasure_index),
|
||||
"the outage object must be rebuilt into its own missing erasure slot"
|
||||
);
|
||||
|
||||
let target_client = cluster.create_s3_client(1)?;
|
||||
for expected in &expected_manifests {
|
||||
let response = target_client.get_object().bucket(bucket).key(&expected.key).send().await?;
|
||||
let actual = response.body.collect().await?.into_bytes();
|
||||
let expected_body = deterministic_object_body(object_size_bytes, expected.payload_seed);
|
||||
assert_eq!(actual.as_ref(), expected_body.as_slice(), "object body changed for {}", expected.key);
|
||||
}
|
||||
let response = target_client.get_object().bucket(bucket).key(outage_key).send().await?;
|
||||
let actual = response.body.collect().await?.into_bytes();
|
||||
let expected_outage_body = deterministic_object_body(object_size_bytes, outage_payload_seed);
|
||||
assert_eq!(actual.as_ref(), expected_outage_body.as_slice(), "object body changed for {outage_key}");
|
||||
|
||||
let terminal_deadline = Instant::now() + Duration::from_secs(30);
|
||||
loop {
|
||||
let status_body = signed_admin_post(&status_url, None, &cluster.access_key, &cluster.secret_key).await?;
|
||||
let status: serde_json::Value = serde_json::from_str(&status_body)
|
||||
.map_err(|err| format!("background heal status is not JSON ({err}): {status_body}"))?;
|
||||
if cluster_heal_is_idle(&status) {
|
||||
break;
|
||||
}
|
||||
if Instant::now() >= terminal_deadline {
|
||||
return Err(format!("heal data rebuilt but operations did not converge to terminal idle: {status}").into());
|
||||
}
|
||||
sleep(Duration::from_millis(250)).await;
|
||||
}
|
||||
|
||||
let task_status_body = signed_admin_post(&task_status_url, None, &cluster.access_key, &cluster.secret_key).await?;
|
||||
let task_status: serde_json::Value = serde_json::from_str(&task_status_body)
|
||||
.map_err(|err| format!("heal task status is not JSON ({err}): {task_status_body}"))?;
|
||||
if task_status["summary"].as_str() != Some("finished") {
|
||||
return Err(format!("heal data rebuilt but task did not finish successfully: {task_status}").into());
|
||||
}
|
||||
|
||||
Ok(())
|
||||
}
|
||||
|
||||
/// Issue #5850: `background-heal/status` must answer while a peer is down.
|
||||
///
|
||||
/// Exercises the production path in `read_cluster_heal_status` end to end,
|
||||
|
||||
@@ -348,11 +348,6 @@ mod delete_regression_test;
|
||||
#[cfg(test)]
|
||||
mod listing_regression_test;
|
||||
|
||||
// Cluster regression: objects committed at degraded write quorum must stay
|
||||
// listable while a different drive is offline (CI run 33478999853).
|
||||
#[cfg(test)]
|
||||
mod degraded_listing_availability_test;
|
||||
|
||||
// P1 regression: bucket statistics accuracy (rustfs#5615, #5008, #5116, #5055, #3898, #1012)
|
||||
#[cfg(test)]
|
||||
mod bucket_stats_regression_test;
|
||||
|
||||
@@ -15,7 +15,7 @@
|
||||
//! Regression coverage for anonymous access on multipart control APIs.
|
||||
|
||||
use crate::common::{RustFSTestEnvironment, init_logging, local_http_client};
|
||||
use async_compression::tokio::write::{BzEncoder, Lz4Encoder, XzEncoder};
|
||||
use async_compression::tokio::write::{BzEncoder, XzEncoder};
|
||||
use aws_sdk_s3::error::{ProvideErrorMetadata, SdkError};
|
||||
use aws_sdk_s3::operation::head_object::HeadObjectOutput;
|
||||
use aws_sdk_s3::primitives::ByteStream;
|
||||
@@ -23,10 +23,7 @@ use aws_sdk_s3::types::{
|
||||
ServerSideEncryption, ServerSideEncryptionByDefault, ServerSideEncryptionConfiguration, ServerSideEncryptionRule,
|
||||
};
|
||||
use chrono::{Duration as ChronoDuration, Utc};
|
||||
use flate2::{
|
||||
Compression,
|
||||
write::{GzEncoder, ZlibEncoder},
|
||||
};
|
||||
use flate2::{Compression, write::GzEncoder};
|
||||
use http::HeaderValue;
|
||||
use http::header::{CONTENT_TYPE, HOST};
|
||||
use md5::{Digest as Md5Digest, Md5};
|
||||
@@ -190,12 +187,6 @@ fn gzip_bytes(data: &[u8]) -> Vec<u8> {
|
||||
encoder.finish().expect("gzip encoder should finish")
|
||||
}
|
||||
|
||||
fn zlib_bytes(data: &[u8]) -> Vec<u8> {
|
||||
let mut encoder = ZlibEncoder::new(Vec::new(), Compression::default());
|
||||
encoder.write_all(data).expect("zlib encoder should accept input");
|
||||
encoder.finish().expect("zlib encoder should finish")
|
||||
}
|
||||
|
||||
fn zstd_bytes(data: &[u8]) -> Vec<u8> {
|
||||
let mut encoder = zstd::Encoder::new(Vec::new(), 0).expect("zstd encoder should initialize");
|
||||
encoder.write_all(data).expect("zstd encoder should accept input");
|
||||
@@ -218,45 +209,6 @@ async fn xz_bytes(data: &[u8]) -> Vec<u8> {
|
||||
encoder.into_inner().into_inner()
|
||||
}
|
||||
|
||||
async fn lz4_bytes(data: &[u8]) -> Vec<u8> {
|
||||
let cursor = Cursor::new(Vec::new());
|
||||
let mut encoder = Lz4Encoder::new(cursor);
|
||||
encoder.write_all(data).await.expect("LZ4 encoder should accept input");
|
||||
encoder.shutdown().await.expect("LZ4 encoder should finish");
|
||||
encoder.into_inner().into_inner()
|
||||
}
|
||||
|
||||
/// Encode the S2 framed stream shape emitted by minio-go PutObjectsSnowball
|
||||
/// with `Compress: true`: 1 MiB independent blocks, better compression,
|
||||
/// masked CRC-32C, and the `S2sTwO` stream identifier.
|
||||
fn minio_go_snowball_s2_bytes(data: &[u8]) -> Vec<u8> {
|
||||
const BLOCK_SIZE: usize = 1 << 20;
|
||||
const CHECKSUM_SIZE: usize = 4;
|
||||
|
||||
let mut output = b"\xff\x06\x00\x00S2sTwO".to_vec();
|
||||
let mut encoder = minlz::Encoder::new();
|
||||
for block in data.chunks(BLOCK_SIZE) {
|
||||
let compressed = encoder.encode_better(block);
|
||||
let compressed_limit = block.len().saturating_sub(block.len() / 32).saturating_sub(5);
|
||||
let (chunk_type, payload) = if compressed.len() <= compressed_limit {
|
||||
(0x00, compressed.as_slice())
|
||||
} else {
|
||||
(0x01, block)
|
||||
};
|
||||
let chunk_len = payload.len() + CHECKSUM_SIZE;
|
||||
assert!(chunk_len < 1 << 24, "S2 fixture chunk must fit the 24-bit frame length");
|
||||
output.extend_from_slice(&[
|
||||
chunk_type,
|
||||
(chunk_len & 0xff) as u8,
|
||||
((chunk_len >> 8) & 0xff) as u8,
|
||||
((chunk_len >> 16) & 0xff) as u8,
|
||||
]);
|
||||
output.extend_from_slice(&minlz::crc::crc(block).to_le_bytes());
|
||||
output.extend_from_slice(payload);
|
||||
}
|
||||
output
|
||||
}
|
||||
|
||||
fn assert_s3_error_code<T, E>(result: Result<T, SdkError<E>>, code: &str)
|
||||
where
|
||||
T: std::fmt::Debug,
|
||||
@@ -3504,62 +3456,6 @@ async fn test_signed_put_object_extract_expands_tar_entries_with_prefix_headers(
|
||||
Ok(())
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
async fn test_signed_put_object_extract_ignore_dirs_skips_unauthorized_directory()
|
||||
-> Result<(), Box<dyn std::error::Error + Send + Sync>> {
|
||||
init_logging();
|
||||
|
||||
let mut env = RustFSTestEnvironment::new().await?;
|
||||
env.start_rustfs_server(vec![]).await?;
|
||||
|
||||
let bucket = "signed-extract-ignore-dirs-auth";
|
||||
let archive_key = "bundle.tar";
|
||||
let allowed_member = "allowed/member.txt";
|
||||
let denied_directory = "denied/";
|
||||
let username = "snowball-ignore-dirs";
|
||||
let secret_key = "snowball-ignore-dirs-secret";
|
||||
let expected_body = b"allowed-body";
|
||||
|
||||
let admin_client = env.create_s3_client();
|
||||
admin_client.create_bucket().bucket(bucket).send().await?;
|
||||
create_restricted_user(&env, username, secret_key).await?;
|
||||
|
||||
let policy = serde_json::json!({
|
||||
"Version": "2012-10-17",
|
||||
"Statement": [{
|
||||
"Effect": "Allow",
|
||||
"Principal": { "AWS": [username] },
|
||||
"Action": ["s3:PutObject"],
|
||||
"Resource": [
|
||||
format!("arn:aws:s3:::{bucket}/{archive_key}"),
|
||||
format!("arn:aws:s3:::{bucket}/{allowed_member}")
|
||||
]
|
||||
}]
|
||||
})
|
||||
.to_string();
|
||||
admin_client.put_bucket_policy().bucket(bucket).policy(policy).send().await?;
|
||||
|
||||
let restricted_client = restricted_user_client(&env, username, secret_key);
|
||||
let tar_bytes = make_tar(&[(allowed_member, expected_body)], &[denied_directory]).await;
|
||||
restricted_client
|
||||
.put_object()
|
||||
.bucket(bucket)
|
||||
.key(archive_key)
|
||||
.body(ByteStream::from(tar_bytes))
|
||||
.customize()
|
||||
.mutate_request(|req| {
|
||||
req.headers_mut().insert("x-amz-meta-snowball-auto-extract", "true");
|
||||
req.headers_mut().insert("x-amz-meta-snowball-ignore-dirs", "true");
|
||||
})
|
||||
.send()
|
||||
.await?;
|
||||
|
||||
let stored = admin_client.get_object().bucket(bucket).key(allowed_member).send().await?;
|
||||
assert_eq!(stored.body.collect().await?.into_bytes().as_ref(), expected_body);
|
||||
|
||||
Ok(())
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
async fn test_signed_put_object_extract_preserves_request_metadata_on_extracted_objects()
|
||||
-> Result<(), Box<dyn std::error::Error + Send + Sync>> {
|
||||
@@ -4289,60 +4185,6 @@ async fn test_signed_put_object_extract_returns_archive_etag() -> Result<(), Box
|
||||
Ok(())
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
async fn test_signed_put_object_extract_expands_s2_and_lz4_by_magic_with_raw_etags()
|
||||
-> Result<(), Box<dyn std::error::Error + Send + Sync>> {
|
||||
init_logging();
|
||||
|
||||
let mut env = RustFSTestEnvironment::new().await?;
|
||||
env.start_rustfs_server(vec![]).await?;
|
||||
|
||||
let bucket = "signed-extract-magic-codecs";
|
||||
let client = env.create_s3_client();
|
||||
client.create_bucket().bucket(bucket).send().await?;
|
||||
|
||||
let s2_tar = make_tar(&[("s2/object.txt", b"s2-body")], &[]).await;
|
||||
let s2_archive = minio_go_snowball_s2_bytes(&s2_tar);
|
||||
let expected_s2_etag = format!("\"{}\"", md5_hex(&s2_archive));
|
||||
let s2_response = client
|
||||
.put_object()
|
||||
.bucket(bucket)
|
||||
// minio-go intentionally uploads a compressed S2 stream with a .tar key.
|
||||
.key("snowball-upload-0123456789abcdef.tar")
|
||||
.body(ByteStream::from(s2_archive))
|
||||
.customize()
|
||||
.mutate_request(|req| {
|
||||
req.headers_mut().insert("x-amz-meta-snowball-auto-extract", "true");
|
||||
})
|
||||
.send()
|
||||
.await?;
|
||||
assert_eq!(s2_response.e_tag(), Some(expected_s2_etag.as_str()));
|
||||
|
||||
let s2_object = client.get_object().bucket(bucket).key("s2/object.txt").send().await?;
|
||||
assert_eq!(s2_object.body.collect().await?.into_bytes().as_ref(), b"s2-body");
|
||||
|
||||
let lz4_tar = make_tar(&[("lz4/object.txt", b"lz4-body")], &[]).await;
|
||||
let lz4_archive = lz4_bytes(&lz4_tar).await;
|
||||
let expected_lz4_etag = format!("\"{}\"", md5_hex(&lz4_archive));
|
||||
let lz4_response = client
|
||||
.put_object()
|
||||
.bucket(bucket)
|
||||
.key("also-looks-like-a-plain.tar")
|
||||
.body(ByteStream::from(lz4_archive))
|
||||
.customize()
|
||||
.mutate_request(|req| {
|
||||
req.headers_mut().insert("x-amz-meta-snowball-auto-extract", "true");
|
||||
})
|
||||
.send()
|
||||
.await?;
|
||||
assert_eq!(lz4_response.e_tag(), Some(expected_lz4_etag.as_str()));
|
||||
|
||||
let lz4_object = client.get_object().bucket(bucket).key("lz4/object.txt").send().await?;
|
||||
assert_eq!(lz4_object.body.collect().await?.into_bytes().as_ref(), b"lz4-body");
|
||||
|
||||
Ok(())
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
async fn test_signed_put_object_extract_preserves_entry_mtime() -> Result<(), Box<dyn std::error::Error + Send + Sync>> {
|
||||
init_logging();
|
||||
@@ -4467,15 +4309,9 @@ async fn test_signed_put_object_extract_authorizes_each_pax_privilege_and_retent
|
||||
let context_archive_resources = [
|
||||
format!("arn:aws:s3:::{bucket}/tag-context.tar"),
|
||||
format!("arn:aws:s3:::{bucket}/lock-context.tar"),
|
||||
format!("arn:aws:s3:::{bucket}/legal-hold-context.tar"),
|
||||
format!("arn:aws:s3:::{bucket}/user-agent-bypass.tar"),
|
||||
format!("arn:aws:s3:::{bucket}/sse-bypass.tar"),
|
||||
];
|
||||
let tag_entry_resource = format!("arn:aws:s3:::{bucket}/tag-context-entry.txt");
|
||||
let lock_entry_resource = format!("arn:aws:s3:::{bucket}/lock-context-entry.txt");
|
||||
let legal_hold_entry_resource = format!("arn:aws:s3:::{bucket}/legal-hold-context-entry.txt");
|
||||
let user_agent_entry_resource = format!("arn:aws:s3:::{bucket}/user-agent-bypass-entry.txt");
|
||||
let sse_entry_resource = format!("arn:aws:s3:::{bucket}/sse-bypass-entry.txt");
|
||||
let policy = serde_json::json!({
|
||||
"Version": "2012-10-17",
|
||||
"Statement": [
|
||||
@@ -4535,7 +4371,7 @@ async fn test_signed_put_object_extract_authorizes_each_pax_privilege_and_retent
|
||||
"Sid": "PaxContextArchives",
|
||||
"Effect": "Allow",
|
||||
"Principal": { "AWS": [pax_context_user] },
|
||||
"Action": ["s3:PutObject", "s3:PutObjectRetention", "s3:PutObjectLegalHold", "s3:PutObjectTagging"],
|
||||
"Action": ["s3:PutObject", "s3:PutObjectRetention", "s3:PutObjectTagging"],
|
||||
"Resource": context_archive_resources
|
||||
},
|
||||
{
|
||||
@@ -4575,49 +4411,6 @@ async fn test_signed_put_object_extract_authorizes_each_pax_privilege_and_retent
|
||||
"Principal": { "AWS": [pax_context_user] },
|
||||
"Action": ["s3:PutObjectRetention"],
|
||||
"Resource": [lock_entry_resource]
|
||||
},
|
||||
{
|
||||
"Sid": "PaxLegalHoldContextPut",
|
||||
"Effect": "Allow",
|
||||
"Principal": { "AWS": [pax_context_user] },
|
||||
"Action": ["s3:PutObject"],
|
||||
"Resource": [legal_hold_entry_resource.clone()]
|
||||
},
|
||||
{
|
||||
"Sid": "PaxLegalHoldContextAction",
|
||||
"Effect": "Allow",
|
||||
"Principal": { "AWS": [pax_context_user] },
|
||||
"Action": ["s3:PutObjectLegalHold"],
|
||||
"Resource": [legal_hold_entry_resource],
|
||||
"Condition": {
|
||||
"StringEquals": {
|
||||
"s3:object-lock-legal-hold": "OFF"
|
||||
}
|
||||
}
|
||||
},
|
||||
{
|
||||
"Sid": "MemberUserAgentCondition",
|
||||
"Effect": "Allow",
|
||||
"Principal": { "AWS": [pax_context_user] },
|
||||
"Action": ["s3:PutObject"],
|
||||
"Resource": [user_agent_entry_resource],
|
||||
"Condition": {
|
||||
"StringEquals": {
|
||||
"aws:UserAgent": "trusted"
|
||||
}
|
||||
}
|
||||
},
|
||||
{
|
||||
"Sid": "MemberSseCondition",
|
||||
"Effect": "Allow",
|
||||
"Principal": { "AWS": [pax_context_user] },
|
||||
"Action": ["s3:PutObject"],
|
||||
"Resource": [sse_entry_resource],
|
||||
"Condition": {
|
||||
"StringEquals": {
|
||||
"s3:x-amz-server-side-encryption": "AES256"
|
||||
}
|
||||
}
|
||||
}
|
||||
]
|
||||
})
|
||||
@@ -4630,13 +4423,8 @@ async fn test_signed_put_object_extract_authorizes_each_pax_privilege_and_retent
|
||||
let cases = [
|
||||
(
|
||||
"legal-hold.tar",
|
||||
put_only_client.clone(),
|
||||
HashMap::from([("minio.metadata.x-amz-object-lock-legal-hold", "ON".to_string())]),
|
||||
),
|
||||
(
|
||||
"tagging.tar",
|
||||
put_only_client,
|
||||
HashMap::from([("minio.metadata.x-amz-tagging", "classification=restricted".to_string())]),
|
||||
HashMap::from([("minio.metadata.x-amz-object-lock-legal-hold", "ON".to_string())]),
|
||||
),
|
||||
(
|
||||
"retention-condition.tar",
|
||||
@@ -4724,57 +4512,6 @@ async fn test_signed_put_object_extract_authorizes_each_pax_privilege_and_retent
|
||||
assert_eq!(stored.body.collect().await?.into_bytes().as_ref(), b"condition-body");
|
||||
|
||||
let pax_context_client = restricted_user_client(&env, pax_context_user, pax_context_secret);
|
||||
for (archive_key, entry_key, pax_key, injected_value, outer_user_agent) in [
|
||||
(
|
||||
"user-agent-bypass.tar",
|
||||
"user-agent-bypass-entry.txt",
|
||||
"minio.metadata.user-agent",
|
||||
"trusted",
|
||||
Some("untrusted"),
|
||||
),
|
||||
(
|
||||
"sse-bypass.tar",
|
||||
"sse-bypass-entry.txt",
|
||||
"minio.metadata.x-amz-server-side-encryption",
|
||||
"AES256",
|
||||
None,
|
||||
),
|
||||
] {
|
||||
let pax = HashMap::from([(pax_key, injected_value.to_string())]);
|
||||
let archive = make_tar_with_pax_entry(entry_key, b"must-not-write", None, &pax).await;
|
||||
let err = pax_context_client
|
||||
.put_object()
|
||||
.bucket(bucket)
|
||||
.key(archive_key)
|
||||
.body(ByteStream::from(archive))
|
||||
.customize()
|
||||
.mutate_request(move |req| {
|
||||
req.headers_mut().insert("x-amz-meta-snowball-auto-extract", "true");
|
||||
if let Some(user_agent) = outer_user_agent {
|
||||
req.headers_mut().insert("user-agent", user_agent);
|
||||
}
|
||||
})
|
||||
.send()
|
||||
.await
|
||||
.expect_err("PAX metadata must not satisfy unrelated IAM request conditions");
|
||||
assert_eq!(
|
||||
err.as_service_error().and_then(|error| error.meta().code()),
|
||||
Some("AccessDenied"),
|
||||
"{archive_key}"
|
||||
);
|
||||
let err = admin_client
|
||||
.head_object()
|
||||
.bucket(bucket)
|
||||
.key(entry_key)
|
||||
.send()
|
||||
.await
|
||||
.expect_err("a denied PAX member must not be written");
|
||||
assert!(matches!(
|
||||
err.as_service_error().and_then(|error| error.meta().code()),
|
||||
Some("NoSuchKey" | "NotFound")
|
||||
));
|
||||
}
|
||||
|
||||
let tag_pax = HashMap::from([("minio.metadata.x-amz-tagging", "classification=public".to_string())]);
|
||||
let archive = make_tar_with_pax_entry("tag-context-entry.txt", b"tag-context-body", None, &tag_pax).await;
|
||||
pax_context_client
|
||||
@@ -4838,34 +4575,6 @@ async fn test_signed_put_object_extract_authorizes_each_pax_privilege_and_retent
|
||||
pax_retain_until
|
||||
);
|
||||
|
||||
let legal_hold_pax = HashMap::from([("minio.metadata.x-amz-object-lock-legal-hold", "ON".to_string())]);
|
||||
let archive = make_tar_with_pax_entry("legal-hold-context-entry.txt", b"must-not-write", None, &legal_hold_pax).await;
|
||||
let err = pax_context_client
|
||||
.put_object()
|
||||
.bucket(bucket)
|
||||
.key("legal-hold-context.tar")
|
||||
.object_lock_legal_hold_status(aws_sdk_s3::types::ObjectLockLegalHoldStatus::Off)
|
||||
.body(ByteStream::from(archive))
|
||||
.customize()
|
||||
.mutate_request(|req| {
|
||||
req.headers_mut().insert("x-amz-meta-snowball-auto-extract", "true");
|
||||
})
|
||||
.send()
|
||||
.await
|
||||
.expect_err("PAX legal hold must replace the outer value in the member IAM condition context");
|
||||
assert_eq!(err.as_service_error().and_then(|error| error.meta().code()), Some("AccessDenied"));
|
||||
let err = admin_client
|
||||
.head_object()
|
||||
.bucket(bucket)
|
||||
.key("legal-hold-context-entry.txt")
|
||||
.send()
|
||||
.await
|
||||
.expect_err("a denied PAX legal-hold member must not be written");
|
||||
assert!(matches!(
|
||||
err.as_service_error().and_then(|error| error.meta().code()),
|
||||
Some("NoSuchKey" | "NotFound")
|
||||
));
|
||||
|
||||
Ok(())
|
||||
}
|
||||
|
||||
@@ -5341,8 +5050,8 @@ async fn test_signed_put_object_extract_expands_tzst_archive() -> Result<(), Box
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
async fn test_signed_put_object_extract_uses_magic_without_requiring_or_trusting_extension()
|
||||
-> Result<(), Box<dyn std::error::Error + Send + Sync>> {
|
||||
async fn test_signed_put_object_extract_rejects_missing_archive_extension() -> Result<(), Box<dyn std::error::Error + Send + Sync>>
|
||||
{
|
||||
init_logging();
|
||||
|
||||
let mut env = RustFSTestEnvironment::new().await?;
|
||||
@@ -5355,7 +5064,8 @@ async fn test_signed_put_object_extract_uses_magic_without_requiring_or_trusting
|
||||
admin_client.create_bucket().bucket(bucket).send().await?;
|
||||
|
||||
let tar_bytes = make_tar(&[("plain.txt", b"plain-body")], &[]).await;
|
||||
admin_client
|
||||
|
||||
let result = admin_client
|
||||
.put_object()
|
||||
.bucket(bucket)
|
||||
.key(archive_key)
|
||||
@@ -5365,80 +5075,15 @@ async fn test_signed_put_object_extract_uses_magic_without_requiring_or_trusting
|
||||
req.headers_mut().insert("x-amz-meta-snowball-auto-extract", "true");
|
||||
})
|
||||
.send()
|
||||
.await?;
|
||||
.await;
|
||||
|
||||
let plain = admin_client.get_object().bucket(bucket).key("plain.txt").send().await?;
|
||||
assert_eq!(plain.body.collect().await?.into_bytes().as_ref(), b"plain-body");
|
||||
|
||||
let raw_with_gzip_suffix = make_tar(&[("raw-with-wrong-suffix.txt", b"raw-body")], &[]).await;
|
||||
admin_client
|
||||
.put_object()
|
||||
.bucket(bucket)
|
||||
.key("raw-but-named.tar.gz")
|
||||
.body(ByteStream::from(raw_with_gzip_suffix))
|
||||
.customize()
|
||||
.mutate_request(|req| {
|
||||
req.headers_mut().insert("x-amz-meta-snowball-auto-extract", "true");
|
||||
})
|
||||
.send()
|
||||
.await?;
|
||||
|
||||
let raw = admin_client
|
||||
.get_object()
|
||||
.bucket(bucket)
|
||||
.key("raw-with-wrong-suffix.txt")
|
||||
.send()
|
||||
.await?;
|
||||
assert_eq!(raw.body.collect().await?.into_bytes().as_ref(), b"raw-body");
|
||||
|
||||
let gzip_with_tar_suffix = gzip_bytes(&make_tar(&[("gzip-with-wrong-suffix.txt", b"gzip-body")], &[]).await);
|
||||
admin_client
|
||||
.put_object()
|
||||
.bucket(bucket)
|
||||
.key("gzip-but-named.tar")
|
||||
.body(ByteStream::from(gzip_with_tar_suffix))
|
||||
.customize()
|
||||
.mutate_request(|req| {
|
||||
req.headers_mut().insert("x-amz-meta-snowball-auto-extract", "true");
|
||||
})
|
||||
.send()
|
||||
.await?;
|
||||
|
||||
let gzip = admin_client
|
||||
.get_object()
|
||||
.bucket(bucket)
|
||||
.key("gzip-with-wrong-suffix.txt")
|
||||
.send()
|
||||
.await?;
|
||||
assert_eq!(gzip.body.collect().await?.into_bytes().as_ref(), b"gzip-body");
|
||||
|
||||
let zlib_archive = zlib_bytes(&make_tar(&[("zlib-extension.txt", b"zlib-body")], &[]).await);
|
||||
admin_client
|
||||
.put_object()
|
||||
.bucket(bucket)
|
||||
.key("bundle.zlib")
|
||||
.body(ByteStream::from(zlib_archive))
|
||||
.customize()
|
||||
.mutate_request(|req| {
|
||||
req.headers_mut().insert("x-amz-meta-snowball-auto-extract", "true");
|
||||
})
|
||||
.send()
|
||||
.await?;
|
||||
|
||||
let zlib = admin_client
|
||||
.get_object()
|
||||
.bucket(bucket)
|
||||
.key("zlib-extension.txt")
|
||||
.send()
|
||||
.await?;
|
||||
assert_eq!(zlib.body.collect().await?.into_bytes().as_ref(), b"zlib-body");
|
||||
assert_s3_error_code(result, "InvalidArgument");
|
||||
|
||||
Ok(())
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
async fn test_signed_put_object_extract_rejects_invalid_archive_payload() -> Result<(), Box<dyn std::error::Error + Send + Sync>>
|
||||
{
|
||||
async fn test_signed_put_object_extract_rejects_invalid_tar_gz_payload() -> Result<(), Box<dyn std::error::Error + Send + Sync>> {
|
||||
init_logging();
|
||||
|
||||
let mut env = RustFSTestEnvironment::new().await?;
|
||||
|
||||
@@ -36,10 +36,9 @@
|
||||
use crate::common::{RustFSTestEnvironment, init_logging, local_http_client};
|
||||
use aws_sdk_s3::error::ProvideErrorMetadata;
|
||||
use aws_sdk_s3::primitives::ByteStream;
|
||||
use rustfs_signer::constants::{UNSIGNED_PAYLOAD, UNSIGNED_PAYLOAD_TRAILER};
|
||||
use rustfs_signer::constants::UNSIGNED_PAYLOAD;
|
||||
use rustfs_signer::request_signature_v4::{SIGN_V4_ALGORITHM, get_scope, get_signature, get_signing_key};
|
||||
use std::fmt::Write as _;
|
||||
use std::io::Cursor;
|
||||
use time::macros::format_description;
|
||||
use time::{Duration, OffsetDateTime};
|
||||
use tracing::info;
|
||||
@@ -99,37 +98,15 @@ impl SigV4 {
|
||||
/// header AND folded into the canonical request — pass the hash of the
|
||||
/// body you *claim* to send, which may differ from what you actually send.
|
||||
fn sign(&self, method: &str, path: &str, canonical_query: &str, content_sha256: &str) -> SignedHeaders {
|
||||
self.sign_with_extra_headers(method, path, canonical_query, content_sha256, &[])
|
||||
}
|
||||
|
||||
/// Sign additional request headers while preserving SigV4's lowercase,
|
||||
/// lexicographically sorted canonical-header representation.
|
||||
fn sign_with_extra_headers(
|
||||
&self,
|
||||
method: &str,
|
||||
path: &str,
|
||||
canonical_query: &str,
|
||||
content_sha256: &str,
|
||||
extra_signed_headers: &[(&str, &str)],
|
||||
) -> SignedHeaders {
|
||||
let amz_date = amz_datetime(self.time);
|
||||
let mut canonical_header_values = vec![
|
||||
("host", self.host.as_str()),
|
||||
("x-amz-content-sha256", content_sha256),
|
||||
("x-amz-date", amz_date.as_str()),
|
||||
];
|
||||
canonical_header_values.extend(extra_signed_headers.iter().copied());
|
||||
canonical_header_values.sort_unstable_by(|left, right| left.0.cmp(right.0));
|
||||
let signed_headers = "host;x-amz-content-sha256;x-amz-date";
|
||||
|
||||
let signed_headers = canonical_header_values
|
||||
.iter()
|
||||
.map(|(name, _)| *name)
|
||||
.collect::<Vec<_>>()
|
||||
.join(";");
|
||||
let mut canonical_headers = String::new();
|
||||
for (name, value) in canonical_header_values {
|
||||
let _ = writeln!(canonical_headers, "{name}:{value}");
|
||||
}
|
||||
let canonical_headers = format!(
|
||||
"host:{host}\nx-amz-content-sha256:{sha}\nx-amz-date:{date}\n",
|
||||
host = self.host,
|
||||
sha = content_sha256,
|
||||
date = amz_date,
|
||||
);
|
||||
let canonical_request =
|
||||
format!("{method}\n{path}\n{canonical_query}\n{canonical_headers}\n{signed_headers}\n{content_sha256}");
|
||||
|
||||
@@ -202,34 +179,6 @@ async fn setup(env: &mut RustFSTestEnvironment) -> Result<(), Box<dyn std::error
|
||||
Ok(())
|
||||
}
|
||||
|
||||
async fn build_single_member_archive(
|
||||
member_key: &str,
|
||||
member_body: &[u8],
|
||||
) -> Result<Vec<u8>, Box<dyn std::error::Error + Send + Sync>> {
|
||||
let mut builder = tokio_tar::Builder::new(Cursor::new(Vec::new()));
|
||||
let mut header = tokio_tar::Header::new_gnu();
|
||||
header.set_size(member_body.len() as u64);
|
||||
header.set_mode(0o644);
|
||||
header.set_cksum();
|
||||
builder.append_data(&mut header, member_key, Cursor::new(member_body)).await?;
|
||||
Ok(builder.into_inner().await?.into_inner())
|
||||
}
|
||||
|
||||
fn sha256_base64(data: &[u8]) -> String {
|
||||
use sha2::{Digest, Sha256};
|
||||
|
||||
base64_simd::STANDARD.encode_to_string(Sha256::digest(data))
|
||||
}
|
||||
|
||||
fn encode_unsigned_aws_chunked_with_sha256_trailer(decoded: &[u8]) -> Vec<u8> {
|
||||
let checksum = sha256_base64(decoded);
|
||||
let mut encoded = format!("{:x}\r\n", decoded.len()).into_bytes();
|
||||
encoded.extend_from_slice(decoded);
|
||||
encoded.extend_from_slice(b"\r\n0\r\n\r\n");
|
||||
encoded.extend_from_slice(format!("x-amz-checksum-sha256:{checksum}").as_bytes());
|
||||
encoded
|
||||
}
|
||||
|
||||
/// Positive control: a correctly hand-signed request must succeed. Without
|
||||
/// this, every negative assertion below could pass for the wrong reason (a
|
||||
/// broken signer that never produces a valid signature).
|
||||
@@ -300,128 +249,6 @@ async fn tampered_signature_returns_signature_does_not_match() -> Result<(), Box
|
||||
Ok(())
|
||||
}
|
||||
|
||||
/// `STREAMING-UNSIGNED-PAYLOAD-TRAILER` disables per-chunk signatures, not the
|
||||
/// seed/header SigV4 signature. A forged request must be rejected before the
|
||||
/// Snowball handler can publish any archive member.
|
||||
#[tokio::test]
|
||||
async fn snowball_streaming_unsigned_trailer_rejects_forged_signature() -> Result<(), Box<dyn std::error::Error + Send + Sync>> {
|
||||
init_logging();
|
||||
let mut env = RustFSTestEnvironment::new().await?;
|
||||
setup(&mut env).await?;
|
||||
|
||||
let archive_key = "forged-streaming-snowball.tar";
|
||||
let member_key = "must-not-be-published.txt";
|
||||
let archive = build_single_member_archive(member_key, b"forged request payload").await?;
|
||||
let decoded_content_length = archive.len().to_string();
|
||||
let encoded_body = encode_unsigned_aws_chunked_with_sha256_trailer(&archive);
|
||||
let path = format!("/{BUCKET}/{archive_key}");
|
||||
|
||||
let mut signer = SigV4::new(&env);
|
||||
signer.secret_key = "wrong-secret-for-forged-streaming-request".to_string();
|
||||
let extra_signed_headers = [
|
||||
("content-encoding", "aws-chunked"),
|
||||
("x-amz-decoded-content-length", decoded_content_length.as_str()),
|
||||
("x-amz-meta-snowball-auto-extract", "true"),
|
||||
("x-amz-trailer", "x-amz-checksum-sha256"),
|
||||
];
|
||||
let headers = signer.sign_with_extra_headers("PUT", &path, "", UNSIGNED_PAYLOAD_TRAILER, &extra_signed_headers);
|
||||
|
||||
let response = local_http_client()
|
||||
.put(format!("{}{}", env.url, path))
|
||||
.header("authorization", &headers.authorization)
|
||||
.header("content-encoding", "aws-chunked")
|
||||
.header("x-amz-content-sha256", &headers.content_sha256)
|
||||
.header("x-amz-date", &headers.amz_date)
|
||||
.header("x-amz-decoded-content-length", &decoded_content_length)
|
||||
.header("x-amz-meta-snowball-auto-extract", "true")
|
||||
.header("x-amz-trailer", "x-amz-checksum-sha256")
|
||||
.body(encoded_body)
|
||||
.send()
|
||||
.await?;
|
||||
let status = response.status();
|
||||
let body = response.text().await?;
|
||||
assert_eq!(status.as_u16(), 403, "forged streaming signature must be 403, body:\n{body}");
|
||||
assert_error_code(&body, "SignatureDoesNotMatch");
|
||||
|
||||
let absent = env
|
||||
.create_s3_client()
|
||||
.get_object()
|
||||
.bucket(BUCKET)
|
||||
.key(member_key)
|
||||
.send()
|
||||
.await
|
||||
.expect_err("a forged streaming request must not publish a Snowball member");
|
||||
assert_eq!(absent.raw_response().map(|response| response.status().as_u16()), Some(404));
|
||||
assert_eq!(absent.as_service_error().and_then(ProvideErrorMetadata::code), Some("NoSuchKey"));
|
||||
|
||||
env.stop_server();
|
||||
Ok(())
|
||||
}
|
||||
|
||||
/// Snowball must consume the complete aws-chunked body before reading the
|
||||
/// trailing checksum exported by s3s into the PutObject response.
|
||||
#[tokio::test]
|
||||
async fn snowball_streaming_unsigned_trailer_returns_sha256_checksum() -> Result<(), Box<dyn std::error::Error + Send + Sync>> {
|
||||
init_logging();
|
||||
let mut env = RustFSTestEnvironment::new().await?;
|
||||
setup(&mut env).await?;
|
||||
|
||||
let archive_key = "valid-streaming-snowball.tar";
|
||||
let member_key = "streaming-checksum-member.txt";
|
||||
let member_body = b"valid streaming Snowball payload";
|
||||
let archive = build_single_member_archive(member_key, member_body).await?;
|
||||
let expected_checksum = sha256_base64(&archive);
|
||||
let decoded_content_length = archive.len().to_string();
|
||||
let encoded_body = encode_unsigned_aws_chunked_with_sha256_trailer(&archive);
|
||||
let path = format!("/{BUCKET}/{archive_key}");
|
||||
|
||||
let signer = SigV4::new(&env);
|
||||
let extra_signed_headers = [
|
||||
("content-encoding", "aws-chunked"),
|
||||
("x-amz-decoded-content-length", decoded_content_length.as_str()),
|
||||
("x-amz-meta-snowball-auto-extract", "true"),
|
||||
("x-amz-sdk-checksum-algorithm", "SHA256"),
|
||||
("x-amz-trailer", "x-amz-checksum-sha256"),
|
||||
];
|
||||
let headers = signer.sign_with_extra_headers("PUT", &path, "", UNSIGNED_PAYLOAD_TRAILER, &extra_signed_headers);
|
||||
|
||||
let response = local_http_client()
|
||||
.put(format!("{}{}", env.url, path))
|
||||
.header("authorization", &headers.authorization)
|
||||
.header("content-encoding", "aws-chunked")
|
||||
.header("x-amz-content-sha256", &headers.content_sha256)
|
||||
.header("x-amz-date", &headers.amz_date)
|
||||
.header("x-amz-decoded-content-length", &decoded_content_length)
|
||||
.header("x-amz-meta-snowball-auto-extract", "true")
|
||||
.header("x-amz-sdk-checksum-algorithm", "SHA256")
|
||||
.header("x-amz-trailer", "x-amz-checksum-sha256")
|
||||
.body(encoded_body)
|
||||
.send()
|
||||
.await?;
|
||||
let status = response.status();
|
||||
let response_checksum = response
|
||||
.headers()
|
||||
.get("x-amz-checksum-sha256")
|
||||
.and_then(|value| value.to_str().ok())
|
||||
.map(str::to_owned);
|
||||
let response_body = response.text().await?;
|
||||
assert_eq!(status.as_u16(), 200, "valid streaming Snowball PUT failed, body:\n{response_body}");
|
||||
assert_eq!(response_checksum.as_deref(), Some(expected_checksum.as_str()));
|
||||
|
||||
let member = env
|
||||
.create_s3_client()
|
||||
.get_object()
|
||||
.bucket(BUCKET)
|
||||
.key(member_key)
|
||||
.send()
|
||||
.await?;
|
||||
let stored = member.body.collect().await?.into_bytes();
|
||||
assert_eq!(stored.as_ref(), member_body);
|
||||
|
||||
env.stop_server();
|
||||
Ok(())
|
||||
}
|
||||
|
||||
/// (b) A valid AccessKeyId paired with the wrong secret key must be rejected
|
||||
/// with SignatureDoesNotMatch / 403.
|
||||
#[tokio::test]
|
||||
|
||||
@@ -21,6 +21,5 @@ mod head_tls_bodyless_test;
|
||||
mod lifecycle;
|
||||
mod lock;
|
||||
mod node_interact_test;
|
||||
mod s3_select_compression;
|
||||
mod sql;
|
||||
mod tiering;
|
||||
|
||||
@@ -1,351 +0,0 @@
|
||||
#![cfg(test)]
|
||||
// Copyright 2024 RustFS Team
|
||||
//
|
||||
// Licensed under the Apache License, Version 2.0 (the "License");
|
||||
// you may not use this file except in compliance with the License.
|
||||
// You may obtain a copy of the License at
|
||||
//
|
||||
// http://www.apache.org/licenses/LICENSE-2.0
|
||||
//
|
||||
// Unless required by applicable law or agreed to in writing, software
|
||||
// distributed under the License is distributed on an "AS IS" BASIS,
|
||||
// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||
// See the License for the specific language governing permissions and
|
||||
// limitations under the License.
|
||||
|
||||
use crate::common::{RustFSTestEnvironment, init_logging};
|
||||
use async_compression::tokio::write::BzEncoder;
|
||||
use aws_sdk_s3::{
|
||||
Client,
|
||||
error::ProvideErrorMetadata,
|
||||
operation::select_object_content::{SelectObjectContentOutput, builders::SelectObjectContentFluentBuilder},
|
||||
types::{
|
||||
CompressionType, CsvInput, CsvOutput, ExpressionType, FileHeaderInfo, InputSerialization, JsonInput, JsonOutput,
|
||||
JsonType, OutputSerialization, SelectObjectContentEventStream,
|
||||
},
|
||||
};
|
||||
use aws_smithy_types::event_stream::RawMessage;
|
||||
use bytes::Bytes;
|
||||
use flate2::{Compression, write::GzEncoder};
|
||||
use std::{error::Error, io::Cursor, time::Duration};
|
||||
use tokio::io::AsyncWriteExt;
|
||||
|
||||
const BUCKET: &str = "s3-select-compression";
|
||||
const SELECT_RESPONSE_TIMEOUT: Duration = Duration::from_secs(30);
|
||||
|
||||
type TestResult<T> = Result<T, Box<dyn Error + Send + Sync>>;
|
||||
|
||||
async fn create_test_environment(extra_env: &[(&str, &str)]) -> TestResult<(RustFSTestEnvironment, Client)> {
|
||||
init_logging();
|
||||
let mut env = RustFSTestEnvironment::new().await?;
|
||||
env.start_rustfs_server_with_env(vec![], extra_env).await?;
|
||||
let client = env.create_s3_client();
|
||||
client.create_bucket().bucket(BUCKET).send().await?;
|
||||
Ok((env, client))
|
||||
}
|
||||
|
||||
async fn put_object(client: &Client, key: &str, body: &[u8]) -> TestResult<()> {
|
||||
client
|
||||
.put_object()
|
||||
.bucket(BUCKET)
|
||||
.key(key)
|
||||
.body(Bytes::copy_from_slice(body).into())
|
||||
.send()
|
||||
.await?;
|
||||
Ok(())
|
||||
}
|
||||
|
||||
fn gzip(input: &[u8]) -> TestResult<Vec<u8>> {
|
||||
let mut encoder = GzEncoder::new(Vec::new(), Compression::default());
|
||||
std::io::Write::write_all(&mut encoder, input)?;
|
||||
Ok(encoder.finish()?)
|
||||
}
|
||||
|
||||
async fn bzip2(input: &[u8]) -> TestResult<Vec<u8>> {
|
||||
let mut encoder = BzEncoder::new(Cursor::new(Vec::new()));
|
||||
encoder.write_all(input).await?;
|
||||
encoder.shutdown().await?;
|
||||
Ok(encoder.into_inner().into_inner())
|
||||
}
|
||||
|
||||
fn csv_select_request(
|
||||
client: &Client,
|
||||
key: &str,
|
||||
compression: CompressionType,
|
||||
expression: &str,
|
||||
) -> SelectObjectContentFluentBuilder {
|
||||
client
|
||||
.select_object_content()
|
||||
.bucket(BUCKET)
|
||||
.key(key)
|
||||
.expression(expression)
|
||||
.expression_type(ExpressionType::Sql)
|
||||
.input_serialization(
|
||||
InputSerialization::builder()
|
||||
.compression_type(compression)
|
||||
.csv(CsvInput::builder().file_header_info(FileHeaderInfo::Use).build())
|
||||
.build(),
|
||||
)
|
||||
.output_serialization(OutputSerialization::builder().csv(CsvOutput::builder().build()).build())
|
||||
}
|
||||
|
||||
fn json_select_request(
|
||||
client: &Client,
|
||||
key: &str,
|
||||
compression: CompressionType,
|
||||
json_type: JsonType,
|
||||
) -> SelectObjectContentFluentBuilder {
|
||||
client
|
||||
.select_object_content()
|
||||
.bucket(BUCKET)
|
||||
.key(key)
|
||||
.expression("SELECT name FROM S3Object")
|
||||
.expression_type(ExpressionType::Sql)
|
||||
.input_serialization(
|
||||
InputSerialization::builder()
|
||||
.compression_type(compression)
|
||||
.json(JsonInput::builder().set_type(Some(json_type)).build())
|
||||
.build(),
|
||||
)
|
||||
.output_serialization(OutputSerialization::builder().json(JsonOutput::builder().build()).build())
|
||||
}
|
||||
|
||||
async fn collect_success(
|
||||
mut response: SelectObjectContentOutput,
|
||||
compressed_bytes: usize,
|
||||
processed_bytes: usize,
|
||||
) -> TestResult<Vec<u8>> {
|
||||
tokio::time::timeout(SELECT_RESPONSE_TIMEOUT, async move {
|
||||
let mut records = Vec::new();
|
||||
let mut stats = None;
|
||||
let mut saw_end = false;
|
||||
|
||||
while let Some(event) = response.payload.recv().await? {
|
||||
assert!(!saw_end, "Select emitted an event after End");
|
||||
match event {
|
||||
SelectObjectContentEventStream::Records(event) => {
|
||||
assert!(stats.is_none(), "Select emitted Records after Stats");
|
||||
if let Some(payload) = event.payload {
|
||||
records.extend_from_slice(payload.as_ref());
|
||||
}
|
||||
}
|
||||
SelectObjectContentEventStream::Stats(event) => {
|
||||
assert!(stats.is_none(), "Select emitted more than one Stats event");
|
||||
stats = event.details;
|
||||
}
|
||||
SelectObjectContentEventStream::End(_) => {
|
||||
assert!(stats.is_some(), "Select emitted End before Stats");
|
||||
saw_end = true;
|
||||
}
|
||||
_ => assert!(stats.is_none(), "Select emitted a non-terminal event after Stats"),
|
||||
}
|
||||
}
|
||||
|
||||
let stats = stats.ok_or("Select response ended without a Stats event")?;
|
||||
assert_eq!(stats.bytes_scanned(), Some(i64::try_from(compressed_bytes)?));
|
||||
assert_eq!(stats.bytes_processed(), Some(i64::try_from(processed_bytes)?));
|
||||
assert_eq!(stats.bytes_returned(), Some(i64::try_from(records.len())?));
|
||||
assert!(saw_end, "Select response ended without an End event");
|
||||
Ok::<_, Box<dyn Error + Send + Sync>>(records)
|
||||
})
|
||||
.await
|
||||
.map_err(|_| -> Box<dyn Error + Send + Sync> { "Select response timed out".into() })?
|
||||
}
|
||||
|
||||
async fn assert_truncated_stream_failure(mut response: SelectObjectContentOutput) -> TestResult<()> {
|
||||
tokio::time::timeout(SELECT_RESPONSE_TIMEOUT, async move {
|
||||
loop {
|
||||
match response.payload.recv().await {
|
||||
Err(error) => {
|
||||
// S3 Select request-level errors use `error` frames, which this SDK version exposes as raw response errors.
|
||||
if let Some(code) = error.code() {
|
||||
assert_eq!(code, "TruncatedInput", "unexpected modeled event-stream error: {error:?}");
|
||||
} else if let aws_sdk_s3::error::SdkError::ResponseError(context) = &error
|
||||
&& let RawMessage::Decoded(message) = context.raw()
|
||||
{
|
||||
let header = |name: &str| {
|
||||
message
|
||||
.headers()
|
||||
.iter()
|
||||
.find(|header| header.name().as_str() == name)
|
||||
.and_then(|header| header.value().as_string().ok())
|
||||
.map(|value| value.as_str())
|
||||
};
|
||||
assert_eq!(header(":message-type"), Some("error"));
|
||||
assert_eq!(header(":error-code"), Some("TruncatedInput"));
|
||||
} else {
|
||||
panic!("unexpected event-stream error: {error:?}");
|
||||
}
|
||||
return Ok(());
|
||||
}
|
||||
Ok(Some(SelectObjectContentEventStream::Stats(_))) | Ok(Some(SelectObjectContentEventStream::End(_))) => {
|
||||
return Err("truncated compressed input reached a success terminal event".into());
|
||||
}
|
||||
Ok(Some(_)) => {}
|
||||
Ok(None) => return Err("truncated compressed input ended without an error event".into()),
|
||||
}
|
||||
}
|
||||
})
|
||||
.await
|
||||
.map_err(|_| -> Box<dyn Error + Send + Sync> { "truncated Select response timed out".into() })?
|
||||
}
|
||||
|
||||
#[tokio::test(flavor = "multi_thread", worker_threads = 4)]
|
||||
async fn test_select_object_content_compressed_csv_and_json() -> TestResult<()> {
|
||||
const CSV: &[u8] = b"name,age\nAlice,30\nBob,25\n";
|
||||
const JSON_LINES: &[u8] = b"{\"name\":\"Alice\"}\n{\"name\":\"Bob\"}\n";
|
||||
const JSON_DOCUMENT: &[u8] = br#"[{"name":"Alice"},{"name":"Bob"}]"#;
|
||||
|
||||
let (_env, client) = create_test_environment(&[]).await?;
|
||||
|
||||
let gzip_csv = gzip(CSV)?;
|
||||
put_object(&client, "records.csv.gz", &gzip_csv).await?;
|
||||
let gzip_csv_records = collect_success(
|
||||
csv_select_request(&client, "records.csv.gz", CompressionType::Gzip, "SELECT * FROM S3Object")
|
||||
.send()
|
||||
.await?,
|
||||
gzip_csv.len(),
|
||||
CSV.len(),
|
||||
)
|
||||
.await?;
|
||||
assert_eq!(gzip_csv_records, b"Alice,30\nBob,25\n");
|
||||
|
||||
let bzip_csv = bzip2(CSV).await?;
|
||||
put_object(&client, "records.csv.bz2", &bzip_csv).await?;
|
||||
let bzip_csv_records = collect_success(
|
||||
csv_select_request(&client, "records.csv.bz2", CompressionType::Bzip2, "SELECT * FROM S3Object")
|
||||
.send()
|
||||
.await?,
|
||||
bzip_csv.len(),
|
||||
CSV.len(),
|
||||
)
|
||||
.await?;
|
||||
assert_eq!(bzip_csv_records, gzip_csv_records);
|
||||
|
||||
let gzip_json_lines = gzip(JSON_LINES)?;
|
||||
put_object(&client, "json-lines", &gzip_json_lines).await?;
|
||||
let gzip_json_records = collect_success(
|
||||
json_select_request(&client, "json-lines", CompressionType::Gzip, JsonType::Lines)
|
||||
.send()
|
||||
.await?,
|
||||
gzip_json_lines.len(),
|
||||
JSON_LINES.len(),
|
||||
)
|
||||
.await?;
|
||||
assert_eq!(gzip_json_records, JSON_LINES);
|
||||
|
||||
let bzip_json_lines = bzip2(JSON_LINES).await?;
|
||||
put_object(&client, "records.jsonl.bz2", &bzip_json_lines).await?;
|
||||
let bzip_json_records = collect_success(
|
||||
json_select_request(&client, "records.jsonl.bz2", CompressionType::Bzip2, JsonType::Lines)
|
||||
.send()
|
||||
.await?,
|
||||
bzip_json_lines.len(),
|
||||
JSON_LINES.len(),
|
||||
)
|
||||
.await?;
|
||||
assert_eq!(bzip_json_records, gzip_json_records);
|
||||
|
||||
let gzip_json_document = gzip(JSON_DOCUMENT)?;
|
||||
put_object(&client, "document.json.gz", &gzip_json_document).await?;
|
||||
let document_records = collect_success(
|
||||
json_select_request(&client, "document.json.gz", CompressionType::Gzip, JsonType::Document)
|
||||
.send()
|
||||
.await?,
|
||||
gzip_json_document.len(),
|
||||
JSON_DOCUMENT.len(),
|
||||
)
|
||||
.await?;
|
||||
assert_eq!(document_records, JSON_LINES);
|
||||
|
||||
Ok(())
|
||||
}
|
||||
|
||||
#[tokio::test(flavor = "multi_thread", worker_threads = 4)]
|
||||
async fn test_select_object_content_invalid_compressed_stream_fails() -> TestResult<()> {
|
||||
const CSV: &[u8] = b"name\nAlice\n";
|
||||
|
||||
let (_env, client) = create_test_environment(&[]).await?;
|
||||
|
||||
put_object(&client, "invalid.csv.gz", CSV).await?;
|
||||
let invalid = csv_select_request(&client, "invalid.csv.gz", CompressionType::Gzip, "SELECT * FROM S3Object")
|
||||
.send()
|
||||
.await
|
||||
.expect_err("invalid GZIP header must fail before streaming");
|
||||
assert_eq!(
|
||||
invalid.as_service_error().and_then(ProvideErrorMetadata::code),
|
||||
Some("InvalidCompressionFormat")
|
||||
);
|
||||
|
||||
put_object(&client, "empty.csv.gz", b"").await?;
|
||||
let empty = csv_select_request(&client, "empty.csv.gz", CompressionType::Gzip, "SELECT * FROM S3Object")
|
||||
.send()
|
||||
.await
|
||||
.expect_err("empty GZIP input must fail as truncated");
|
||||
assert_eq!(empty.as_service_error().and_then(ProvideErrorMetadata::code), Some("TruncatedInput"));
|
||||
|
||||
let mut truncated = bzip2(CSV).await?;
|
||||
truncated.pop();
|
||||
put_object(&client, "truncated.csv.bz2", &truncated).await?;
|
||||
let truncated = csv_select_request(&client, "truncated.csv.bz2", CompressionType::Bzip2, "SELECT * FROM S3Object")
|
||||
.send()
|
||||
.await?;
|
||||
assert_truncated_stream_failure(truncated).await?;
|
||||
|
||||
Ok(())
|
||||
}
|
||||
|
||||
#[tokio::test(flavor = "multi_thread", worker_threads = 4)]
|
||||
async fn test_select_object_content_compressed_disconnect_releases_query() -> TestResult<()> {
|
||||
const OBJECT: &str = "disconnect.csv.gz";
|
||||
const ROWS: usize = 16 * 1024;
|
||||
const RELEASE_ATTEMPTS: usize = 20;
|
||||
const RELEASE_BACKOFF: Duration = Duration::from_millis(25);
|
||||
|
||||
let (_env, client) = create_test_environment(&[("RUSTFS_S3SELECT_MAX_CONCURRENT_QUERIES", "1")]).await?;
|
||||
let row = format!("{}\n", "x".repeat(1023));
|
||||
let mut body = Vec::with_capacity("value\n".len() + ROWS * row.len());
|
||||
body.extend_from_slice(b"value\n");
|
||||
for _ in 0..ROWS {
|
||||
body.extend_from_slice(row.as_bytes());
|
||||
}
|
||||
let compressed = gzip(&body)?;
|
||||
put_object(&client, OBJECT, &compressed).await?;
|
||||
|
||||
let first = csv_select_request(&client, OBJECT, CompressionType::Gzip, "SELECT * FROM S3Object")
|
||||
.send()
|
||||
.await?;
|
||||
let saturated = csv_select_request(&client, OBJECT, CompressionType::Gzip, "SELECT * FROM S3Object")
|
||||
.send()
|
||||
.await
|
||||
.expect_err("the unread compressed response should retain the only query permit");
|
||||
assert_eq!(saturated.as_service_error().and_then(ProvideErrorMetadata::code), Some("SlowDown"));
|
||||
|
||||
drop(first);
|
||||
let second = tokio::time::timeout(Duration::from_secs(5), async {
|
||||
for attempt in 0..RELEASE_ATTEMPTS {
|
||||
match csv_select_request(&client, OBJECT, CompressionType::Gzip, "SELECT * FROM S3Object")
|
||||
.send()
|
||||
.await
|
||||
{
|
||||
Ok(response) => return Ok::<_, Box<dyn Error + Send + Sync>>(response),
|
||||
Err(error)
|
||||
if error.as_service_error().and_then(ProvideErrorMetadata::code) == Some("SlowDown")
|
||||
&& attempt + 1 < RELEASE_ATTEMPTS =>
|
||||
{
|
||||
tokio::time::sleep(RELEASE_BACKOFF).await;
|
||||
}
|
||||
Err(error) if error.as_service_error().and_then(ProvideErrorMetadata::code) == Some("SlowDown") => {
|
||||
return Err("disconnected compressed Select retained its query permit".into());
|
||||
}
|
||||
Err(error) => return Err(format!("unexpected Select error after disconnect: {error}").into()),
|
||||
}
|
||||
}
|
||||
Err("query permit release retry loop ended unexpectedly".into())
|
||||
})
|
||||
.await
|
||||
.map_err(|_| -> Box<dyn Error + Send + Sync> { "compressed Select did not release its query permit".into() })??;
|
||||
drop(second);
|
||||
|
||||
Ok(())
|
||||
}
|
||||
@@ -122,24 +122,6 @@ async fn select_json_document(client: &Client, key: &str, expression: &str) -> T
|
||||
process_select_response(response).await
|
||||
}
|
||||
|
||||
fn csv_select_request(
|
||||
client: &Client,
|
||||
key: &str,
|
||||
) -> aws_sdk_s3::operation::select_object_content::builders::SelectObjectContentFluentBuilder {
|
||||
client
|
||||
.select_object_content()
|
||||
.bucket(BUCKET)
|
||||
.key(key)
|
||||
.expression("SELECT * FROM S3Object")
|
||||
.expression_type(ExpressionType::Sql)
|
||||
.input_serialization(
|
||||
InputSerialization::builder()
|
||||
.csv(CsvInput::builder().file_header_info(FileHeaderInfo::Use).build())
|
||||
.build(),
|
||||
)
|
||||
.output_serialization(OutputSerialization::builder().csv(CsvOutput::builder().build()).build())
|
||||
}
|
||||
|
||||
async fn process_select_response(
|
||||
mut event_stream: aws_sdk_s3::operation::select_object_content::SelectObjectContentOutput,
|
||||
) -> TestResult<String> {
|
||||
@@ -206,42 +188,30 @@ async fn assert_input_byte_stats(
|
||||
let mut last_progress: Option<aws_sdk_s3::types::Progress> = None;
|
||||
let mut stats = None;
|
||||
let mut saw_end = false;
|
||||
tokio::time::timeout(SELECT_RESPONSE_TIMEOUT, async {
|
||||
// The AWS SDK validates both event-stream CRCs before yielding an event.
|
||||
while let Some(event) = payload.recv().await? {
|
||||
assert!(!saw_end, "Select emitted an event after End");
|
||||
match event {
|
||||
aws_sdk_s3::types::SelectObjectContentEventStream::Records(records) => {
|
||||
assert!(stats.is_none(), "Select emitted Records after Stats");
|
||||
if let Some(bytes) = records.payload {
|
||||
records_len = records_len.saturating_add(u64::try_from(bytes.as_ref().len())?);
|
||||
}
|
||||
while let Some(event) = payload.recv().await? {
|
||||
match event {
|
||||
aws_sdk_s3::types::SelectObjectContentEventStream::Records(records) => {
|
||||
if let Some(bytes) = records.payload {
|
||||
records_len = records_len.saturating_add(u64::try_from(bytes.as_ref().len())?);
|
||||
}
|
||||
aws_sdk_s3::types::SelectObjectContentEventStream::Progress(event) => {
|
||||
assert!(stats.is_none(), "Select emitted Progress after Stats");
|
||||
let details = event.details.ok_or("Progress event did not contain details")?;
|
||||
if let Some(previous) = last_progress.as_ref() {
|
||||
assert!(details.bytes_scanned() >= previous.bytes_scanned());
|
||||
assert!(details.bytes_processed() >= previous.bytes_processed());
|
||||
assert!(details.bytes_returned() >= previous.bytes_returned());
|
||||
}
|
||||
last_progress = Some(details);
|
||||
}
|
||||
aws_sdk_s3::types::SelectObjectContentEventStream::Stats(event) => {
|
||||
assert!(stats.is_none(), "Select emitted more than one Stats event");
|
||||
stats = event.details;
|
||||
}
|
||||
aws_sdk_s3::types::SelectObjectContentEventStream::End(_) => {
|
||||
assert!(stats.is_some(), "Select emitted End before Stats");
|
||||
saw_end = true;
|
||||
}
|
||||
_ => assert!(stats.is_none(), "Select emitted a non-terminal event after Stats"),
|
||||
}
|
||||
aws_sdk_s3::types::SelectObjectContentEventStream::Progress(event) => {
|
||||
let details = event.details.ok_or("Progress event did not contain details")?;
|
||||
if let Some(previous) = last_progress.as_ref() {
|
||||
assert!(details.bytes_scanned() >= previous.bytes_scanned());
|
||||
assert!(details.bytes_processed() >= previous.bytes_processed());
|
||||
assert!(details.bytes_returned() >= previous.bytes_returned());
|
||||
}
|
||||
last_progress = Some(details);
|
||||
}
|
||||
aws_sdk_s3::types::SelectObjectContentEventStream::Stats(event) => stats = event.details,
|
||||
aws_sdk_s3::types::SelectObjectContentEventStream::End(_) => {
|
||||
saw_end = true;
|
||||
break;
|
||||
}
|
||||
_ => {}
|
||||
}
|
||||
Ok::<(), Box<dyn Error + Send + Sync>>(())
|
||||
})
|
||||
.await
|
||||
.map_err(|_| -> Box<dyn Error + Send + Sync> { "Select response timed out".into() })??;
|
||||
}
|
||||
|
||||
let stats = stats.ok_or("Select response ended without a Stats event")?;
|
||||
let input_len = i64::try_from(body.len())?;
|
||||
@@ -249,11 +219,10 @@ async fn assert_input_byte_stats(
|
||||
assert_eq!(stats.bytes_processed(), Some(input_len));
|
||||
assert_eq!(stats.bytes_returned(), Some(i64::try_from(records_len)?));
|
||||
if progress_enabled {
|
||||
if let Some(progress) = last_progress {
|
||||
assert!(stats.bytes_scanned() >= progress.bytes_scanned());
|
||||
assert!(stats.bytes_processed() >= progress.bytes_processed());
|
||||
assert!(stats.bytes_returned() >= progress.bytes_returned());
|
||||
}
|
||||
let progress = last_progress.ok_or("Select response ended without a Progress event")?;
|
||||
assert_eq!(progress.bytes_scanned(), stats.bytes_scanned());
|
||||
assert_eq!(progress.bytes_processed(), stats.bytes_processed());
|
||||
assert_eq!(progress.bytes_returned(), stats.bytes_returned());
|
||||
} else {
|
||||
assert!(last_progress.is_none(), "disabled request progress emitted a Progress event");
|
||||
}
|
||||
@@ -262,7 +231,7 @@ async fn assert_input_byte_stats(
|
||||
}
|
||||
|
||||
#[tokio::test(flavor = "multi_thread", worker_threads = 4)]
|
||||
async fn test_select_object_content_http_event_order_crc_and_input_byte_stats() -> TestResult<()> {
|
||||
async fn test_select_object_content_reports_input_byte_stats() -> TestResult<()> {
|
||||
const CSV_BODY: &[u8] = b"name,age\nAlice,30\nBob,25\n";
|
||||
const JSON_LINES_BODY: &[u8] = b"{\"name\":\"Alice\"}\n{\"name\":\"Bob\"}\n";
|
||||
const JSON_DOCUMENT_BODY: &[u8] = b"[{\"name\":\"Alice\"},{\"name\":\"Bob\"}]";
|
||||
@@ -320,60 +289,6 @@ async fn test_select_object_content_http_event_order_crc_and_input_byte_stats()
|
||||
Ok(())
|
||||
}
|
||||
|
||||
#[tokio::test(flavor = "multi_thread", worker_threads = 4)]
|
||||
async fn test_select_object_content_http_disconnect_releases_query() -> TestResult<()> {
|
||||
const OBJECT: &str = "disconnect.csv";
|
||||
const ROWS: usize = 16 * 1024;
|
||||
const RELEASE_BACKOFF: Duration = Duration::from_millis(25);
|
||||
|
||||
init_logging();
|
||||
let mut env = RustFSTestEnvironment::new().await?;
|
||||
env.start_rustfs_server_with_env(vec![], &[("RUSTFS_S3SELECT_MAX_CONCURRENT_QUERIES", "1")])
|
||||
.await?;
|
||||
let client = env.create_s3_client();
|
||||
setup_test_bucket(&client).await?;
|
||||
|
||||
let row = format!("{}\n", "x".repeat(1023));
|
||||
let mut body = Vec::with_capacity("value\n".len() + ROWS * row.len());
|
||||
body.extend_from_slice(b"value\n");
|
||||
for _ in 0..ROWS {
|
||||
body.extend_from_slice(row.as_bytes());
|
||||
}
|
||||
client
|
||||
.put_object()
|
||||
.bucket(BUCKET)
|
||||
.key(OBJECT)
|
||||
.body(Bytes::from(body).into())
|
||||
.send()
|
||||
.await?;
|
||||
|
||||
// Leaving this response body unread fills the bounded HTTP/event channels before the query can finish.
|
||||
let first = csv_select_request(&client, OBJECT).send().await?;
|
||||
let saturated = csv_select_request(&client, OBJECT)
|
||||
.send()
|
||||
.await
|
||||
.expect_err("the first HTTP stream should retain the only query permit");
|
||||
assert_eq!(saturated.as_service_error().and_then(ProvideErrorMetadata::code), Some("SlowDown"));
|
||||
|
||||
drop(first);
|
||||
let second = tokio::time::timeout(Duration::from_secs(5), async {
|
||||
loop {
|
||||
match csv_select_request(&client, OBJECT).send().await {
|
||||
Ok(response) => return Ok::<_, Box<dyn Error + Send + Sync>>(response),
|
||||
Err(error) if error.as_service_error().and_then(ProvideErrorMetadata::code) == Some("SlowDown") => {
|
||||
tokio::time::sleep(RELEASE_BACKOFF).await;
|
||||
}
|
||||
Err(error) => return Err(format!("unexpected Select error after disconnect: {error}").into()),
|
||||
}
|
||||
}
|
||||
})
|
||||
.await
|
||||
.map_err(|_| -> Box<dyn Error + Send + Sync> { "disconnected Select did not release its query permit".into() })??;
|
||||
drop(second);
|
||||
|
||||
Ok(())
|
||||
}
|
||||
|
||||
#[tokio::test(flavor = "multi_thread", worker_threads = 4)]
|
||||
async fn test_select_object_content_csv_basic() -> TestResult<()> {
|
||||
let (_env, client) = create_test_environment().await?;
|
||||
|
||||
@@ -17,56 +17,8 @@ mod tests {
|
||||
use crate::common::{RustFSTestEnvironment, init_logging};
|
||||
use aws_sdk_s3::error::ProvideErrorMetadata;
|
||||
use aws_sdk_s3::primitives::ByteStream;
|
||||
use flate2::{Compression, write::GzEncoder};
|
||||
use std::error::Error;
|
||||
use std::io::{Cursor, Write};
|
||||
|
||||
fn pax_record(key: &str, value: &str) -> Vec<u8> {
|
||||
let payload = format!("{key}={value}\n");
|
||||
let mut len = payload.len() + 3;
|
||||
loop {
|
||||
let record = format!("{len} {payload}");
|
||||
if record.len() == len {
|
||||
return record.into_bytes();
|
||||
}
|
||||
len = record.len();
|
||||
}
|
||||
}
|
||||
|
||||
async fn append_pax_header(
|
||||
builder: &mut tokio_tar::Builder<Cursor<Vec<u8>>>,
|
||||
entry_type: tokio_tar::EntryType,
|
||||
records: &[(&str, &str)],
|
||||
) -> Result<(), Box<dyn Error + Send + Sync>> {
|
||||
let mut payload = Vec::new();
|
||||
for (key, value) in records {
|
||||
payload.extend(pax_record(key, value));
|
||||
}
|
||||
let mut header = tokio_tar::Header::new_ustar();
|
||||
header.set_entry_type(entry_type);
|
||||
header.set_size(u64::try_from(payload.len()).expect("PAX payload length should fit in u64"));
|
||||
header.set_mode(0o644);
|
||||
header.set_cksum();
|
||||
builder
|
||||
.append_data(&mut header, "PaxHeaders.X/snowball", Cursor::new(payload))
|
||||
.await?;
|
||||
Ok(())
|
||||
}
|
||||
|
||||
async fn append_typed_entry(
|
||||
builder: &mut tokio_tar::Builder<Cursor<Vec<u8>>>,
|
||||
path: &str,
|
||||
entry_type: tokio_tar::EntryType,
|
||||
body: &[u8],
|
||||
) -> Result<(), Box<dyn Error + Send + Sync>> {
|
||||
let mut header = tokio_tar::Header::new_gnu();
|
||||
header.set_entry_type(entry_type);
|
||||
header.set_size(u64::try_from(body.len()).expect("TAR member length should fit in u64"));
|
||||
header.set_mode(0o644);
|
||||
header.set_cksum();
|
||||
builder.append_data(&mut header, path, Cursor::new(body)).await?;
|
||||
Ok(())
|
||||
}
|
||||
use std::io::Cursor;
|
||||
|
||||
async fn build_test_archive() -> Result<Vec<u8>, Box<dyn Error + Send + Sync>> {
|
||||
let mut builder = tokio_tar::Builder::new(Cursor::new(Vec::new()));
|
||||
@@ -117,50 +69,12 @@ mod tests {
|
||||
Ok(builder.into_inner().await?.into_inner())
|
||||
}
|
||||
|
||||
async fn build_archive_with_invalid_checksum() -> Result<Vec<u8>, Box<dyn Error + Send + Sync>> {
|
||||
let mut archive = build_test_archive().await?;
|
||||
archive[0] ^= 1;
|
||||
Ok(archive)
|
||||
}
|
||||
|
||||
async fn build_archive_with_negative_gnu_mtime() -> Result<Vec<u8>, Box<dyn Error + Send + Sync>> {
|
||||
let mut builder = tokio_tar::Builder::new(Cursor::new(Vec::new()));
|
||||
let mut header = tokio_tar::Header::new_gnu();
|
||||
header.set_size(b"negative-mtime-body".len() as u64);
|
||||
header.set_mode(0o644);
|
||||
header.as_old_mut().mtime.fill(0xff);
|
||||
builder
|
||||
.append_data(&mut header, "negative-mtime.txt", Cursor::new(b"negative-mtime-body".as_slice()))
|
||||
.await?;
|
||||
Ok(builder.into_inner().await?.into_inner())
|
||||
}
|
||||
|
||||
fn gzip_member(payload: &[u8]) -> Result<Vec<u8>, Box<dyn Error + Send + Sync>> {
|
||||
let mut encoder = GzEncoder::new(Vec::new(), Compression::default());
|
||||
encoder.write_all(payload)?;
|
||||
Ok(encoder.finish()?)
|
||||
}
|
||||
|
||||
async fn build_concatenated_gzip_archive() -> Result<Vec<u8>, Box<dyn Error + Send + Sync>> {
|
||||
let archive = build_test_archive().await?;
|
||||
let split_at = archive.len() / 2;
|
||||
let mut encoded = gzip_member(&archive[..split_at])?;
|
||||
encoded.extend(gzip_member(&archive[split_at..])?);
|
||||
Ok(encoded)
|
||||
}
|
||||
|
||||
async fn build_gzip_archive_with_invalid_crc() -> Result<Vec<u8>, Box<dyn Error + Send + Sync>> {
|
||||
let mut encoded = gzip_member(&build_test_archive().await?)?;
|
||||
let crc_offset = encoded.len().checked_sub(8).expect("gzip fixture must contain a trailer");
|
||||
encoded[crc_offset] ^= 1;
|
||||
Ok(encoded)
|
||||
}
|
||||
|
||||
fn append_raw_tar_entry_with_type(archive: &mut Vec<u8>, path: &[u8], data: &[u8], entry_type: u8) {
|
||||
assert!(path.len() <= 100, "raw TAR fixture path must fit in the name field");
|
||||
fn build_archive_with_parent_dir_entry(victim_bucket: &str) -> Vec<u8> {
|
||||
let path = format!("../{victim_bucket}/evil-injected.txt");
|
||||
let data = b"injected-body";
|
||||
let mut header = [0u8; 512];
|
||||
|
||||
header[..path.len()].copy_from_slice(path);
|
||||
header[..path.len()].copy_from_slice(path.as_bytes());
|
||||
header[100..108].copy_from_slice(b"0000644\0");
|
||||
header[108..116].copy_from_slice(b"0000000\0");
|
||||
header[116..124].copy_from_slice(b"0000000\0");
|
||||
@@ -168,7 +82,7 @@ mod tests {
|
||||
header[124..136].copy_from_slice(size.as_bytes());
|
||||
header[136..148].copy_from_slice(b"00000000000\0");
|
||||
header[148..156].fill(b' ');
|
||||
header[156] = entry_type;
|
||||
header[156] = b'0';
|
||||
header[257..263].copy_from_slice(b"ustar\0");
|
||||
header[263..265].copy_from_slice(b"00");
|
||||
|
||||
@@ -176,87 +90,11 @@ mod tests {
|
||||
let checksum = format!("{:06o}\0 ", checksum);
|
||||
header[148..156].copy_from_slice(checksum.as_bytes());
|
||||
|
||||
let mut archive = Vec::new();
|
||||
archive.extend_from_slice(&header);
|
||||
archive.extend_from_slice(data);
|
||||
let padding = (512 - (data.len() % 512)) % 512;
|
||||
archive.extend(std::iter::repeat_n(0, padding));
|
||||
}
|
||||
|
||||
fn append_raw_tar_entry(archive: &mut Vec<u8>, path: &[u8], data: &[u8]) {
|
||||
append_raw_tar_entry_with_type(archive, path, data, b'0');
|
||||
}
|
||||
|
||||
fn build_archive_with_parent_dir_entry(victim_bucket: &str) -> Vec<u8> {
|
||||
let path = format!("../{victim_bucket}/evil-injected.txt");
|
||||
let mut archive = Vec::new();
|
||||
append_raw_tar_entry(&mut archive, path.as_bytes(), b"injected-body");
|
||||
archive.extend_from_slice(&[0u8; 1024]);
|
||||
archive
|
||||
}
|
||||
|
||||
async fn build_member_semantics_archive() -> Result<Vec<u8>, Box<dyn Error + Send + Sync>> {
|
||||
let mut builder = tokio_tar::Builder::new(Cursor::new(Vec::new()));
|
||||
append_pax_header(
|
||||
&mut builder,
|
||||
tokio_tar::EntryType::XGlobalHeader,
|
||||
&[
|
||||
("minio.metadata.x-amz-meta-owner", "global"),
|
||||
("minio.metadata.x-amz-meta-snowball-auto-extract", "true"),
|
||||
],
|
||||
)
|
||||
.await?;
|
||||
append_pax_header(
|
||||
&mut builder,
|
||||
tokio_tar::EntryType::XHeader,
|
||||
&[("minio.metadata.x-amz-meta-owner", "local")],
|
||||
)
|
||||
.await?;
|
||||
append_typed_entry(&mut builder, "regular.txt", tokio_tar::EntryType::Regular, b"regular-body").await?;
|
||||
for (path, entry_type) in [
|
||||
("char", tokio_tar::EntryType::Char),
|
||||
("block", tokio_tar::EntryType::Block),
|
||||
("fifo", tokio_tar::EntryType::Fifo),
|
||||
] {
|
||||
append_typed_entry(&mut builder, path, entry_type, b"").await?;
|
||||
}
|
||||
let mut directory = tokio_tar::Header::new_gnu();
|
||||
directory.set_entry_type(tokio_tar::EntryType::Directory);
|
||||
directory.set_size(0);
|
||||
directory.set_mode(0o755);
|
||||
directory.set_cksum();
|
||||
builder
|
||||
.append_data(&mut directory, "directory/", Cursor::new(Vec::new()))
|
||||
.await?;
|
||||
for (path, entry_type) in [
|
||||
("hard-link", tokio_tar::EntryType::Link),
|
||||
("symlink", tokio_tar::EntryType::Symlink),
|
||||
("continuous", tokio_tar::EntryType::Continuous),
|
||||
("unknown", tokio_tar::EntryType::Other(b'9')),
|
||||
] {
|
||||
append_typed_entry(&mut builder, path, entry_type, b"").await?;
|
||||
}
|
||||
Ok(builder.into_inner().await?.into_inner())
|
||||
}
|
||||
|
||||
async fn build_versioned_member_archive(path: &str, version_id: &str) -> Result<Vec<u8>, Box<dyn Error + Send + Sync>> {
|
||||
let mut builder = tokio_tar::Builder::new(Cursor::new(Vec::new()));
|
||||
append_pax_header(&mut builder, tokio_tar::EntryType::XHeader, &[("minio.versionId", version_id)]).await?;
|
||||
append_typed_entry(&mut builder, path, tokio_tar::EntryType::Regular, b"versioned-body").await?;
|
||||
Ok(builder.into_inner().await?.into_inner())
|
||||
}
|
||||
|
||||
fn build_archive_with_invalid_utf8_entry() -> Vec<u8> {
|
||||
let mut archive = Vec::new();
|
||||
append_raw_tar_entry(&mut archive, b"invalid-\xff.txt", b"ignored-body");
|
||||
append_raw_tar_entry(&mut archive, b"valid.txt", b"valid-body");
|
||||
archive.extend_from_slice(&[0u8; 1024]);
|
||||
archive
|
||||
}
|
||||
|
||||
fn build_archive_with_invalid_utf8_symlink() -> Vec<u8> {
|
||||
let mut archive = Vec::new();
|
||||
append_raw_tar_entry_with_type(&mut archive, b"invalid-\xff-link", b"", b'2');
|
||||
append_raw_tar_entry(&mut archive, b"valid.txt", b"valid-body");
|
||||
archive.extend_from_slice(&[0u8; 1024]);
|
||||
archive
|
||||
}
|
||||
@@ -297,147 +135,6 @@ mod tests {
|
||||
Ok(())
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
async fn snowball_auto_extract_applies_member_semantics_and_metadata_precedence() -> Result<(), Box<dyn Error + Send + Sync>>
|
||||
{
|
||||
init_logging();
|
||||
|
||||
let mut env = RustFSTestEnvironment::new().await?;
|
||||
env.start_rustfs_server(vec![]).await?;
|
||||
|
||||
let client = env.create_s3_client();
|
||||
let bucket = "snowball-member-semantics";
|
||||
client.create_bucket().bucket(bucket).send().await?;
|
||||
client
|
||||
.put_object()
|
||||
.bucket(bucket)
|
||||
.key("fixture.tar")
|
||||
.metadata("Snowball-Auto-Extract", "true")
|
||||
.metadata("Minio-Snowball-Prefix", "members")
|
||||
.metadata("owner", "outer")
|
||||
.body(ByteStream::from(build_member_semantics_archive().await?))
|
||||
.send()
|
||||
.await?;
|
||||
|
||||
let regular = client.head_object().bucket(bucket).key("members/regular.txt").send().await?;
|
||||
let regular_metadata = regular.metadata().expect("regular member should expose metadata");
|
||||
assert_eq!(regular_metadata.get("owner").map(String::as_str), Some("local"));
|
||||
assert!(!regular_metadata.contains_key("snowball-auto-extract"));
|
||||
assert!(!regular_metadata.contains_key("minio-snowball-prefix"));
|
||||
|
||||
for key in ["char", "block", "fifo"] {
|
||||
let head = client
|
||||
.head_object()
|
||||
.bucket(bucket)
|
||||
.key(format!("members/{key}"))
|
||||
.send()
|
||||
.await?;
|
||||
assert_eq!(head.content_length(), Some(0), "{key} should be materialized as an empty object");
|
||||
assert_eq!(
|
||||
head.metadata().and_then(|metadata| metadata.get("owner")).map(String::as_str),
|
||||
Some("outer"),
|
||||
"{key} should not inherit global PAX metadata"
|
||||
);
|
||||
}
|
||||
let directory = client.head_object().bucket(bucket).key("members/directory/").send().await?;
|
||||
assert_eq!(directory.content_length(), Some(0));
|
||||
|
||||
for key in ["hard-link", "symlink", "continuous", "unknown"] {
|
||||
let error = client
|
||||
.head_object()
|
||||
.bucket(bucket)
|
||||
.key(format!("members/{key}"))
|
||||
.send()
|
||||
.await
|
||||
.expect_err("unsupported TAR entry type must be skipped");
|
||||
assert_eq!(error.into_service_error().code(), Some("NotFound"), "{key}");
|
||||
}
|
||||
|
||||
env.stop_server();
|
||||
Ok(())
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
async fn snowball_auto_extract_validates_pax_version_id_against_bucket_state() -> Result<(), Box<dyn Error + Send + Sync>> {
|
||||
init_logging();
|
||||
|
||||
let mut env = RustFSTestEnvironment::new().await?;
|
||||
env.start_rustfs_server(vec![]).await?;
|
||||
|
||||
let client = env.create_s3_client();
|
||||
let bucket = "snowball-version-semantics";
|
||||
client.create_bucket().bucket(bucket).send().await?;
|
||||
|
||||
client
|
||||
.put_object()
|
||||
.bucket(bucket)
|
||||
.key("null.tar")
|
||||
.metadata("Snowball-Auto-Extract", "true")
|
||||
.body(ByteStream::from(build_versioned_member_archive("null.txt", "null").await?))
|
||||
.send()
|
||||
.await?;
|
||||
let null_member = client.get_object().bucket(bucket).key("null.txt").send().await?;
|
||||
assert_eq!(null_member.body.collect().await?.into_bytes().as_ref(), b"versioned-body");
|
||||
|
||||
for (archive_key, member_key, version_id) in [
|
||||
("uuid.tar", "uuid.txt", uuid::Uuid::new_v4().to_string()),
|
||||
("uppercase-null.tar", "uppercase-null.txt", "NULL".to_string()),
|
||||
] {
|
||||
let error = client
|
||||
.put_object()
|
||||
.bucket(bucket)
|
||||
.key(archive_key)
|
||||
.metadata("Snowball-Auto-Extract", "true")
|
||||
.body(ByteStream::from(build_versioned_member_archive(member_key, &version_id).await?))
|
||||
.send()
|
||||
.await
|
||||
.expect_err("invalid or unversioned UUID import must be rejected");
|
||||
assert_eq!(error.into_service_error().code(), Some("InvalidArgument"), "{archive_key}");
|
||||
let missing = client
|
||||
.head_object()
|
||||
.bucket(bucket)
|
||||
.key(member_key)
|
||||
.send()
|
||||
.await
|
||||
.expect_err("rejected version import must not create an object");
|
||||
assert_eq!(missing.into_service_error().code(), Some("NotFound"), "{member_key}");
|
||||
}
|
||||
|
||||
client
|
||||
.put_bucket_versioning()
|
||||
.bucket(bucket)
|
||||
.versioning_configuration(
|
||||
aws_sdk_s3::types::VersioningConfiguration::builder()
|
||||
.status(aws_sdk_s3::types::BucketVersioningStatus::Enabled)
|
||||
.build(),
|
||||
)
|
||||
.send()
|
||||
.await?;
|
||||
let imported_version_id = uuid::Uuid::new_v4().to_string();
|
||||
client
|
||||
.put_object()
|
||||
.bucket(bucket)
|
||||
.key("versioned-uuid.tar")
|
||||
.metadata("Snowball-Auto-Extract", "true")
|
||||
.body(ByteStream::from(
|
||||
build_versioned_member_archive("versioned-uuid.txt", &imported_version_id).await?,
|
||||
))
|
||||
.send()
|
||||
.await?;
|
||||
let imported = client
|
||||
.get_object()
|
||||
.bucket(bucket)
|
||||
.key("versioned-uuid.txt")
|
||||
.version_id(&imported_version_id)
|
||||
.send()
|
||||
.await?;
|
||||
assert_eq!(imported.version_id(), Some(imported_version_id.as_str()));
|
||||
assert_eq!(imported.body.collect().await?.into_bytes().as_ref(), b"versioned-body");
|
||||
|
||||
env.stop_server();
|
||||
Ok(())
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
async fn snowball_auto_extract_supports_standard_headers_with_combined_extract_options()
|
||||
-> Result<(), Box<dyn Error + Send + Sync>> {
|
||||
@@ -566,113 +263,6 @@ mod tests {
|
||||
Ok(())
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
async fn snowball_auto_extract_accepts_negative_gnu_mtime() -> Result<(), Box<dyn Error + Send + Sync>> {
|
||||
init_logging();
|
||||
|
||||
let mut env = RustFSTestEnvironment::new().await?;
|
||||
env.start_rustfs_server(vec![]).await?;
|
||||
|
||||
let client = env.create_s3_client();
|
||||
let bucket = "snowball-negative-mtime";
|
||||
client.create_bucket().bucket(bucket).send().await?;
|
||||
client
|
||||
.put_object()
|
||||
.bucket(bucket)
|
||||
.key("fixture.tar")
|
||||
.metadata("Snowball-Auto-Extract", "true")
|
||||
.body(ByteStream::from(build_archive_with_negative_gnu_mtime().await?))
|
||||
.send()
|
||||
.await?;
|
||||
|
||||
let object = client.get_object().bucket(bucket).key("negative-mtime.txt").send().await?;
|
||||
assert_eq!(object.body.collect().await?.into_bytes().as_ref(), b"negative-mtime-body");
|
||||
|
||||
env.stop_server();
|
||||
Ok(())
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
async fn snowball_auto_extract_consumes_concatenated_gzip_members() -> Result<(), Box<dyn Error + Send + Sync>> {
|
||||
init_logging();
|
||||
|
||||
let mut env = RustFSTestEnvironment::new().await?;
|
||||
env.start_rustfs_server(vec![]).await?;
|
||||
|
||||
let client = env.create_s3_client();
|
||||
let bucket = "snowball-concatenated-gzip";
|
||||
client.create_bucket().bucket(bucket).send().await?;
|
||||
client
|
||||
.put_object()
|
||||
.bucket(bucket)
|
||||
.key("fixture.tar.gz")
|
||||
.metadata("Snowball-Auto-Extract", "true")
|
||||
.body(ByteStream::from(build_concatenated_gzip_archive().await?))
|
||||
.send()
|
||||
.await?;
|
||||
|
||||
let object = client.get_object().bucket(bucket).key("root.txt").send().await?;
|
||||
assert_eq!(object.body.collect().await?.into_bytes().as_ref(), b"root payload\n");
|
||||
|
||||
env.stop_server();
|
||||
Ok(())
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
async fn snowball_auto_extract_rejects_gzip_crc_error_when_ignore_errors_enabled() -> Result<(), Box<dyn Error + Send + Sync>>
|
||||
{
|
||||
init_logging();
|
||||
|
||||
let mut env = RustFSTestEnvironment::new().await?;
|
||||
env.start_rustfs_server(vec![]).await?;
|
||||
|
||||
let client = env.create_s3_client();
|
||||
let bucket = "snowball-gzip-crc-ignore-errors";
|
||||
client.create_bucket().bucket(bucket).send().await?;
|
||||
let err = client
|
||||
.put_object()
|
||||
.bucket(bucket)
|
||||
.key("fixture.tar.gz")
|
||||
.metadata("Snowball-Auto-Extract", "true")
|
||||
.metadata("Minio-Snowball-Ignore-Errors", "true")
|
||||
.body(ByteStream::from(build_gzip_archive_with_invalid_crc().await?))
|
||||
.send()
|
||||
.await
|
||||
.expect_err("gzip integrity failures must remain fatal under ignore-errors");
|
||||
|
||||
assert_eq!(err.into_service_error().code(), Some("InvalidArgument"));
|
||||
|
||||
env.stop_server();
|
||||
Ok(())
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
async fn snowball_auto_extract_rejects_mismatched_content_md5() -> Result<(), Box<dyn Error + Send + Sync>> {
|
||||
init_logging();
|
||||
|
||||
let mut env = RustFSTestEnvironment::new().await?;
|
||||
env.start_rustfs_server(vec![]).await?;
|
||||
|
||||
let client = env.create_s3_client();
|
||||
let bucket = "snowball-content-md5";
|
||||
client.create_bucket().bucket(bucket).send().await?;
|
||||
let err = client
|
||||
.put_object()
|
||||
.bucket(bucket)
|
||||
.key("fixture.tar")
|
||||
.metadata("Snowball-Auto-Extract", "true")
|
||||
.content_md5("AAAAAAAAAAAAAAAAAAAAAA==")
|
||||
.body(ByteStream::from(build_test_archive().await?))
|
||||
.send()
|
||||
.await
|
||||
.expect_err("mismatched Content-MD5 must fail after the raw body reaches EOF");
|
||||
|
||||
assert_eq!(err.into_service_error().code(), Some("BadDigest"));
|
||||
|
||||
env.stop_server();
|
||||
Ok(())
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
async fn snowball_auto_extract_ignores_invalid_entries_when_requested() -> Result<(), Box<dyn Error + Send + Sync>> {
|
||||
init_logging();
|
||||
@@ -709,100 +299,7 @@ mod tests {
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
async fn snowball_auto_extract_skips_non_utf8_symlink_without_ignore_errors() -> Result<(), Box<dyn Error + Send + Sync>> {
|
||||
init_logging();
|
||||
|
||||
let mut env = RustFSTestEnvironment::new().await?;
|
||||
env.start_rustfs_server(vec![]).await?;
|
||||
|
||||
let client = env.create_s3_client();
|
||||
let bucket = "snowball-invalid-utf8-link";
|
||||
client.create_bucket().bucket(bucket).send().await?;
|
||||
|
||||
client
|
||||
.put_object()
|
||||
.bucket(bucket)
|
||||
.key("fixture.tar")
|
||||
.metadata("Snowball-Auto-Extract", "true")
|
||||
.body(ByteStream::from(build_archive_with_invalid_utf8_symlink()))
|
||||
.send()
|
||||
.await?;
|
||||
|
||||
let valid = client.get_object().bucket(bucket).key("valid.txt").send().await?;
|
||||
assert_eq!(valid.body.collect().await?.into_bytes().as_ref(), b"valid-body");
|
||||
let listed = client.list_objects_v2().bucket(bucket).send().await?;
|
||||
let keys: Vec<_> = listed.contents().iter().filter_map(|entry| entry.key()).collect();
|
||||
assert_eq!(keys, vec!["valid.txt"]);
|
||||
|
||||
env.stop_server();
|
||||
Ok(())
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
async fn snowball_auto_extract_skips_non_utf8_member_without_lossy_key_collision() -> Result<(), Box<dyn Error + Send + Sync>>
|
||||
{
|
||||
init_logging();
|
||||
|
||||
let mut env = RustFSTestEnvironment::new().await?;
|
||||
env.start_rustfs_server(vec![]).await?;
|
||||
|
||||
let client = env.create_s3_client();
|
||||
let bucket = "snowball-invalid-utf8";
|
||||
client.create_bucket().bucket(bucket).send().await?;
|
||||
|
||||
client
|
||||
.put_object()
|
||||
.bucket(bucket)
|
||||
.key("fixture.tar")
|
||||
.metadata("Snowball-Auto-Extract", "true")
|
||||
.metadata("Minio-Snowball-Ignore-Errors", "true")
|
||||
.body(ByteStream::from(build_archive_with_invalid_utf8_entry()))
|
||||
.send()
|
||||
.await?;
|
||||
|
||||
let valid = client.get_object().bucket(bucket).key("valid.txt").send().await?;
|
||||
assert_eq!(valid.body.collect().await?.into_bytes().as_ref(), b"valid-body");
|
||||
let listed = client.list_objects_v2().bucket(bucket).send().await?;
|
||||
let keys: Vec<_> = listed.contents().iter().filter_map(|entry| entry.key()).collect();
|
||||
assert_eq!(keys, vec!["valid.txt"]);
|
||||
|
||||
env.stop_server();
|
||||
Ok(())
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
async fn snowball_auto_extract_rejects_corrupt_tar_when_ignore_errors_enabled() -> Result<(), Box<dyn Error + Send + Sync>> {
|
||||
init_logging();
|
||||
|
||||
let mut env = RustFSTestEnvironment::new().await?;
|
||||
env.start_rustfs_server(vec![]).await?;
|
||||
|
||||
let client = env.create_s3_client();
|
||||
let bucket = "snowball-corrupt-ignore-errors";
|
||||
let archive = build_archive_with_invalid_checksum().await?;
|
||||
client.create_bucket().bucket(bucket).send().await?;
|
||||
|
||||
let err = client
|
||||
.put_object()
|
||||
.bucket(bucket)
|
||||
.key("fixture.tar")
|
||||
.metadata("Snowball-Auto-Extract", "true")
|
||||
.metadata("Minio-Snowball-Ignore-Errors", "true")
|
||||
.body(ByteStream::from(archive))
|
||||
.send()
|
||||
.await
|
||||
.expect_err("corrupt TAR structure must remain fatal under ignore-errors");
|
||||
assert_eq!(err.into_service_error().code(), Some("InvalidArgument"));
|
||||
|
||||
let listed = client.list_objects_v2().bucket(bucket).send().await?;
|
||||
assert!(listed.contents().is_empty(), "corrupt archive must not produce objects");
|
||||
|
||||
env.stop_server();
|
||||
Ok(())
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
async fn snowball_auto_extract_rejects_parent_dir_entry_even_when_ignore_errors_enabled()
|
||||
async fn snowball_auto_extract_rejects_parent_dir_entry_without_cross_bucket_write()
|
||||
-> Result<(), Box<dyn Error + Send + Sync>> {
|
||||
init_logging();
|
||||
|
||||
@@ -822,7 +319,6 @@ mod tests {
|
||||
.bucket(attacker_bucket)
|
||||
.key("fixture.tar")
|
||||
.metadata("Snowball-Auto-Extract", "true")
|
||||
.metadata("Minio-Snowball-Ignore-Errors", "true")
|
||||
.body(ByteStream::from(archive))
|
||||
.send()
|
||||
.await
|
||||
|
||||
@@ -12,17 +12,14 @@
|
||||
// See the License for the specific language governing permissions and
|
||||
// limitations under the License.
|
||||
|
||||
use crate::common::{RustFSTestClusterEnvironment, RustFSTestEnvironment, init_logging, rustfs_binary_path};
|
||||
use crate::common::{RustFSTestEnvironment, init_logging};
|
||||
use aws_sdk_s3::Client;
|
||||
use aws_sdk_s3::error::ProvideErrorMetadata;
|
||||
use aws_sdk_s3::primitives::ByteStream;
|
||||
use aws_sdk_s3::types::{
|
||||
BucketVersioningStatus, CompletedMultipartUpload, CompletedPart, ServerSideEncryption, VersioningConfiguration,
|
||||
};
|
||||
use std::path::{Path, PathBuf};
|
||||
use std::time::Duration;
|
||||
use tokio::task::JoinSet;
|
||||
use tokio::time::{Instant, sleep};
|
||||
use std::path::PathBuf;
|
||||
|
||||
type TestResult = Result<(), Box<dyn std::error::Error + Send + Sync>>;
|
||||
|
||||
@@ -31,14 +28,6 @@ const SSE_MASTER_KEY_ENV: &str = "RUSTFS_SSE_S3_MASTER_KEY";
|
||||
const SSE_MASTER_KEY: &str = "QkJCQkJCQkJCQkJCQkJCQkJCQkJCQkJCQkJCQkJCQkI=";
|
||||
const PLAIN_BUCKET: &str = "upgrade-plain-data";
|
||||
const VERSIONED_BUCKET: &str = "upgrade-versioned-data";
|
||||
const MIXED_BUCKET: &str = "upgrade-mixed-version-data";
|
||||
const MIXED_NODE_COUNT: usize = 4;
|
||||
const MULTIPART_WORKERS: usize = 16;
|
||||
const MULTIPART_UPLOADS_PER_WORKER: usize = 16;
|
||||
// Peers keep a restarted node's drive in Suspect/Returning for roughly
|
||||
// probe_interval (2s) x success_threshold (3) after it comes back; 30s
|
||||
// comfortably covers that window plus CI scheduling jitter.
|
||||
const LISTING_CONVERGENCE_TIMEOUT: Duration = Duration::from_secs(30);
|
||||
|
||||
fn source_binary() -> Result<PathBuf, Box<dyn std::error::Error + Send + Sync>> {
|
||||
let path = std::env::var_os(SOURCE_BINARY_ENV)
|
||||
@@ -114,132 +103,6 @@ async fn write_multipart(client: &Client, bucket: &str, key: &str, parts: &[Vec<
|
||||
Ok(())
|
||||
}
|
||||
|
||||
fn configure_cluster_logs(cluster: &mut RustFSTestClusterEnvironment) -> TestResult {
|
||||
let Some(log_dir) = std::env::var_os("RUSTFS_E2E_LOG_DIR") else {
|
||||
return Ok(());
|
||||
};
|
||||
std::fs::create_dir_all(&log_dir)?;
|
||||
for node_idx in 0..cluster.nodes.len() {
|
||||
let path = Path::new(&log_dir).join(format!("mixed-upgrade-node-{node_idx}.log"));
|
||||
cluster.set_node_capture_log_path(node_idx, path.to_string_lossy().into_owned())?;
|
||||
}
|
||||
Ok(())
|
||||
}
|
||||
|
||||
async fn write_multipart_load(clients: &[Client], phase: &str) -> Result<Vec<String>, Box<dyn std::error::Error + Send + Sync>> {
|
||||
let mut tasks = JoinSet::new();
|
||||
for worker in 0..MULTIPART_WORKERS {
|
||||
let client = clients[worker % clients.len()].clone();
|
||||
let phase = phase.to_string();
|
||||
tasks.spawn(async move {
|
||||
let mut keys = Vec::with_capacity(MULTIPART_UPLOADS_PER_WORKER);
|
||||
for upload in 0..MULTIPART_UPLOADS_PER_WORKER {
|
||||
let key = format!("{phase}/multipart/{worker:02}/{upload:02}");
|
||||
let part = vec![u8::try_from(worker)?; 64 * 1024];
|
||||
write_multipart(&client, MIXED_BUCKET, &key, &[part]).await?;
|
||||
keys.push(key);
|
||||
}
|
||||
Ok::<_, Box<dyn std::error::Error + Send + Sync>>(keys)
|
||||
});
|
||||
}
|
||||
|
||||
let mut keys = Vec::with_capacity(MULTIPART_WORKERS * MULTIPART_UPLOADS_PER_WORKER);
|
||||
while let Some(result) = tasks.join_next().await {
|
||||
keys.extend(result??);
|
||||
}
|
||||
Ok(keys)
|
||||
}
|
||||
|
||||
/// Assert that `client` eventually lists exactly `expected` objects under
|
||||
/// `{phase}/`, polling until [`LISTING_CONVERGENCE_TIMEOUT`].
|
||||
///
|
||||
/// A single-snapshot assertion here is racy by construction: each phase both
|
||||
/// writes and lists within seconds of a node restart. While a peer still holds
|
||||
/// the restarted node's drive in Suspect/Returning, strict-quorum listing
|
||||
/// consults only the remaining three drives and drops any object that was
|
||||
/// itself legally written at write quorum (3/4 drives) during an earlier
|
||||
/// node's identical post-restart window — its xl.meta is then visible on only
|
||||
/// two of the three consulted drives, below the required object quorum of
|
||||
/// three. GET still succeeds for such objects; only the listing under-counts
|
||||
/// until drive health converges. A genuine upgrade data-loss regression still
|
||||
/// fails after the deadline.
|
||||
async fn wait_for_phase_listing(client: &Client, phase: &str, expected: usize, context: &str) -> TestResult {
|
||||
let deadline = Instant::now() + LISTING_CONVERGENCE_TIMEOUT;
|
||||
loop {
|
||||
let listed = client
|
||||
.list_objects_v2()
|
||||
.bucket(MIXED_BUCKET)
|
||||
.prefix(format!("{phase}/"))
|
||||
.send()
|
||||
.await?;
|
||||
let count = listed.contents().len();
|
||||
if count == expected {
|
||||
return Ok(());
|
||||
}
|
||||
if Instant::now() >= deadline {
|
||||
return Err(format!(
|
||||
"{context}: listing under {phase}/ returned {count} of {expected} objects even after {}s of post-restart convergence",
|
||||
LISTING_CONVERGENCE_TIMEOUT.as_secs()
|
||||
)
|
||||
.into());
|
||||
}
|
||||
sleep(Duration::from_millis(500)).await;
|
||||
}
|
||||
}
|
||||
|
||||
async fn exercise_mixed_cluster(
|
||||
cluster: &RustFSTestClusterEnvironment,
|
||||
phase: &str,
|
||||
current_node: usize,
|
||||
previous_node: usize,
|
||||
) -> TestResult {
|
||||
let clients = cluster.create_all_clients()?;
|
||||
let current_client = &clients[current_node];
|
||||
let previous_client = &clients[previous_node];
|
||||
|
||||
let current_key = format!("{phase}/written-by-current");
|
||||
let current_body = format!("{phase}: current RustFS build").into_bytes();
|
||||
current_client
|
||||
.put_object()
|
||||
.bucket(MIXED_BUCKET)
|
||||
.key(¤t_key)
|
||||
.body(ByteStream::from(current_body.clone()))
|
||||
.send()
|
||||
.await?;
|
||||
assert_eq!(read_object(previous_client, MIXED_BUCKET, ¤t_key, None).await?.1, current_body);
|
||||
|
||||
let previous_key = format!("{phase}/written-by-previous");
|
||||
let previous_body = format!("{phase}: previous RustFS release").into_bytes();
|
||||
previous_client
|
||||
.put_object()
|
||||
.bucket(MIXED_BUCKET)
|
||||
.key(&previous_key)
|
||||
.body(ByteStream::from(previous_body.clone()))
|
||||
.send()
|
||||
.await?;
|
||||
assert_eq!(read_object(current_client, MIXED_BUCKET, &previous_key, None).await?.1, previous_body);
|
||||
|
||||
let multipart_keys = write_multipart_load(&clients, phase).await?;
|
||||
let expected_count = multipart_keys.len() + 2;
|
||||
for (label, client) in [("current", current_client), ("previous", previous_client)] {
|
||||
wait_for_phase_listing(
|
||||
client,
|
||||
phase,
|
||||
expected_count,
|
||||
&format!("the {label} RustFS version must stream the complete mixed-version listing"),
|
||||
)
|
||||
.await?;
|
||||
}
|
||||
|
||||
let last_multipart_key = format!("{phase}/multipart/{:02}/{:02}", MULTIPART_WORKERS - 1, MULTIPART_UPLOADS_PER_WORKER - 1);
|
||||
assert_eq!(
|
||||
read_object(previous_client, MIXED_BUCKET, &last_multipart_key, None).await?.1,
|
||||
vec![u8::try_from(MULTIPART_WORKERS - 1)?; 64 * 1024]
|
||||
);
|
||||
|
||||
Ok(())
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
#[ignore = "requires a pinned previous RustFS release binary"]
|
||||
async fn direct_upgrade_from_rc2_preserves_object_contracts() -> TestResult {
|
||||
@@ -389,43 +252,3 @@ async fn direct_upgrade_from_rc2_preserves_object_contracts() -> TestResult {
|
||||
|
||||
Ok(())
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
#[ignore = "requires a pinned previous RustFS release binary"]
|
||||
async fn rolling_upgrade_from_rc2_preserves_mixed_version_contracts() -> TestResult {
|
||||
init_logging();
|
||||
let previous_binary = source_binary()?;
|
||||
let current_binary = rustfs_binary_path();
|
||||
let mut cluster = RustFSTestClusterEnvironment::new(MIXED_NODE_COUNT).await?;
|
||||
cluster.set_env("RUST_LOG", "rustfs=warn,rustfs_notify=warn");
|
||||
configure_cluster_logs(&mut cluster)?;
|
||||
cluster.start_with_binary(&previous_binary).await?;
|
||||
cluster.create_test_bucket(MIXED_BUCKET).await?;
|
||||
|
||||
cluster.stop_node(0)?;
|
||||
cluster.start_node_from_binary(0, ¤t_binary).await?;
|
||||
exercise_mixed_cluster(&cluster, "one-current-node", 0, 1).await?;
|
||||
|
||||
for node_idx in [1, 2] {
|
||||
cluster.stop_node(node_idx)?;
|
||||
cluster.start_node_from_binary(node_idx, ¤t_binary).await?;
|
||||
}
|
||||
exercise_mixed_cluster(&cluster, "one-previous-node", 0, 3).await?;
|
||||
|
||||
cluster.stop_node(3)?;
|
||||
cluster.start_node_from_binary(3, ¤t_binary).await?;
|
||||
|
||||
for (node_idx, client) in cluster.create_all_clients()?.iter().enumerate() {
|
||||
for phase in ["one-current-node", "one-previous-node"] {
|
||||
wait_for_phase_listing(
|
||||
client,
|
||||
phase,
|
||||
MULTIPART_WORKERS * MULTIPART_UPLOADS_PER_WORKER + 2,
|
||||
&format!("node {node_idx}: the homogeneous current cluster must preserve every object"),
|
||||
)
|
||||
.await?;
|
||||
}
|
||||
}
|
||||
|
||||
Ok(())
|
||||
}
|
||||
|
||||
@@ -75,10 +75,6 @@ pub mod bucket {
|
||||
delete_transition_candidate_for_operator, finalize_missing_transition_transaction_for_operator,
|
||||
inspect_transition_transaction_for_operator,
|
||||
};
|
||||
#[cfg(feature = "test-util")]
|
||||
pub use crate::bucket::lifecycle::transition_transaction::{
|
||||
TransitionTransactionRecoveryStats, recover_transition_transaction_records,
|
||||
};
|
||||
}
|
||||
|
||||
pub mod evaluator {
|
||||
@@ -103,17 +99,6 @@ pub mod bucket {
|
||||
pub use crate::bucket::lifecycle::tier_delete_journal::{
|
||||
persist_tier_delete_journal_entry, record_tier_delete_journal_backend_identity,
|
||||
};
|
||||
|
||||
#[cfg(feature = "test-util")]
|
||||
pub mod test_util {
|
||||
/// Model a single-node, all-v6 fleet after its capability probe has completed.
|
||||
///
|
||||
/// Call this only once while constructing an isolated test store, before any
|
||||
/// tier-delete journal permit or background worker can be active.
|
||||
pub fn install_all_v6_fleet_capability_proof() {
|
||||
crate::services::notification_sys::install_cross_pool_fence_fleet_proof_for_test();
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
pub mod tier_last_day_stats {
|
||||
@@ -504,9 +489,9 @@ pub mod storage {
|
||||
pub use crate::core::pools::HealLifecycleExpiryContext;
|
||||
pub use crate::store::HealWalkVersion;
|
||||
pub use crate::store::{
|
||||
ECStore, SCANNER_PUBLICATION_LEASE_TTL_MS, ScannerDataMovementPauseStatus, all_local_disk, all_local_disk_path,
|
||||
find_local_disk_by_ref, init_local_disks, init_local_disks_with_instance_ctx, init_lock_clients,
|
||||
prewarm_local_disk_id_map, prewarm_local_disk_id_map_with_instance_ctx,
|
||||
ECStore, SCANNER_PUBLICATION_LEASE_TTL_MS, all_local_disk, all_local_disk_path, find_local_disk_by_ref, init_local_disks,
|
||||
init_local_disks_with_instance_ctx, init_lock_clients, prewarm_local_disk_id_map,
|
||||
prewarm_local_disk_id_map_with_instance_ctx,
|
||||
};
|
||||
}
|
||||
|
||||
|
||||
@@ -41,7 +41,7 @@ use crate::bucket::lifecycle::tier_free_version_recovery::{
|
||||
DEFAULT_FREE_VERSION_RECOVERY_LIMIT, FreeVersionRecoveryStats, recover_tier_free_versions_with_cancel,
|
||||
};
|
||||
use crate::bucket::lifecycle::tier_last_day_stats::{DailyAllTierStats, LastDayTierStats};
|
||||
use crate::bucket::lifecycle::tier_sweeper::{Jentry, delete_object_from_remote_tier_with_lease_idempotent};
|
||||
use crate::bucket::lifecycle::tier_sweeper::{Jentry, delete_object_from_remote_tier_idempotent_with_manager_and_identity};
|
||||
use crate::bucket::lifecycle::transition_transaction::run_transition_transaction_recovery_loop;
|
||||
use crate::bucket::object_lock::ObjectLockApi;
|
||||
use crate::bucket::versioning::VersioningApi as _;
|
||||
@@ -50,10 +50,7 @@ use crate::disk::error::DiskError;
|
||||
use crate::disk::{DeleteOptions, Disk, DiskAPI, RUSTFS_META_BUCKET, RUSTFS_META_MULTIPART_BUCKET, STORAGE_FORMAT_FILE};
|
||||
use crate::error::Error;
|
||||
use crate::error::StorageError;
|
||||
use crate::error::{
|
||||
is_err_object_not_found, is_err_read_quorum, is_err_strict_volume_not_found, is_err_version_not_found,
|
||||
is_network_or_host_down,
|
||||
};
|
||||
use crate::error::{is_err_object_not_found, is_err_read_quorum, is_err_version_not_found, is_network_or_host_down};
|
||||
use crate::object_api::{GetObjectReader, ObjectInfo, ObjectOptions};
|
||||
use crate::object_api::{ObjectEncryptionResolver, ReadPlan};
|
||||
use crate::services::tier::{
|
||||
@@ -589,215 +586,23 @@ impl ExpiryOp for FreeVersionTask {
|
||||
}
|
||||
}
|
||||
|
||||
async fn acquire_free_version_tier_lease(
|
||||
oi: &ObjectInfo,
|
||||
tier_config_mgr: &Arc<RwLock<TierConfigMgr>>,
|
||||
) -> Result<(TierOperationLease, bool), std::io::Error> {
|
||||
let version_id_exact = validate_transition_remote_version(oi)?;
|
||||
let identity = tier_destination_id_from_metadata(&oi.user_defined)?
|
||||
.ok_or_else(|| std::io::Error::other("tier free-version has no durable backend identity"))?;
|
||||
let lease =
|
||||
TierConfigMgr::acquire_operation_lease_for_backend_identity(tier_config_mgr, &oi.transitioned_object.tier, identity)
|
||||
.await
|
||||
.map_err(std::io::Error::other)?;
|
||||
Ok((lease, version_id_exact))
|
||||
}
|
||||
|
||||
async fn delete_free_version_remote_object_with_lease(
|
||||
oi: &ObjectInfo,
|
||||
lease: &TierOperationLease,
|
||||
version_id_exact: bool,
|
||||
) -> Result<(), std::io::Error> {
|
||||
delete_object_from_remote_tier_with_lease_idempotent(
|
||||
&oi.transitioned_object.name,
|
||||
&oi.transitioned_object.version_id,
|
||||
lease,
|
||||
version_id_exact,
|
||||
)
|
||||
.await?;
|
||||
Ok(())
|
||||
}
|
||||
|
||||
fn free_version_physical_topology_generation(api: &ECStore) -> String {
|
||||
let mut hasher = Sha256::new();
|
||||
for pool in &api.pools {
|
||||
hasher.update(pool.pool_idx.to_be_bytes());
|
||||
hasher.update(pool.disk_set.len().to_be_bytes());
|
||||
for set in &pool.disk_set {
|
||||
hasher.update(set.set_index.to_be_bytes());
|
||||
}
|
||||
}
|
||||
rustfs_utils::crypto::hex(hasher.finalize().as_slice())
|
||||
}
|
||||
|
||||
fn free_version_remote_tuple_matches(candidate: &ObjectInfo, expected: &ObjectInfo) -> std::io::Result<bool> {
|
||||
if candidate.transitioned_object.tier != expected.transitioned_object.tier
|
||||
|| candidate.transitioned_object.name != expected.transitioned_object.name
|
||||
{
|
||||
return Ok(false);
|
||||
}
|
||||
let candidate_identity = tier_destination_id_from_metadata(&candidate.user_defined)?
|
||||
.ok_or_else(|| std::io::Error::other("tier free-version is missing its backend identity"))?;
|
||||
let expected_identity = tier_destination_id_from_metadata(&expected.user_defined)?
|
||||
.ok_or_else(|| std::io::Error::other("tier free-version task is missing its backend identity"))?;
|
||||
if candidate_identity != expected_identity {
|
||||
return Ok(false);
|
||||
}
|
||||
if candidate.transition_version_state == rustfs_filemeta::TransitionVersionState::Unknown
|
||||
|| expected.transition_version_state == rustfs_filemeta::TransitionVersionState::Unknown
|
||||
{
|
||||
return Err(std::io::Error::new(
|
||||
std::io::ErrorKind::WouldBlock,
|
||||
"tier free-version remote version state is unknown",
|
||||
));
|
||||
}
|
||||
Ok(candidate.transition_version_state == expected.transition_version_state
|
||||
&& candidate.transitioned_object.version_id == expected.transitioned_object.version_id)
|
||||
}
|
||||
|
||||
async fn scan_exact_free_version_targets(
|
||||
api: &ECStore,
|
||||
oi: &ObjectInfo,
|
||||
local_object: &str,
|
||||
) -> std::io::Result<Vec<(Arc<SetDisks>, FileInfo)>> {
|
||||
let mut targets = Vec::new();
|
||||
for pool in &api.pools {
|
||||
for set in &pool.disk_set {
|
||||
let versions = match set.load_file_info_versions_exact(&oi.bucket, &oi.name).await {
|
||||
Ok(Some(versions)) => versions,
|
||||
Ok(None) => continue,
|
||||
Err(err) if is_err_strict_volume_not_found(&err) => continue,
|
||||
Err(err) => return Err(std::io::Error::other(err)),
|
||||
};
|
||||
for version in versions.versions.iter().chain(versions.free_versions.iter()) {
|
||||
let candidate = ObjectInfo::from_file_info(version, &oi.bucket, &oi.name, true);
|
||||
if free_version_remote_tuple_matches(&candidate, oi)? {
|
||||
if candidate.transitioned_object.free_version {
|
||||
// Data movement can leave the same remote tuple in
|
||||
// several physical pools. Ordinary deletion assigns a
|
||||
// fresh local free-version UUID to each copy, but all
|
||||
// of those markers own the same idempotent remote
|
||||
// DELETE. Consume them together while holding every
|
||||
// physical object lock; treating their local UUIDs as
|
||||
// conflicting would strand cleanup forever.
|
||||
let mut actual = version.clone();
|
||||
actual.name = local_object.to_string();
|
||||
targets.push((Arc::clone(set), actual));
|
||||
} else {
|
||||
return Err(std::io::Error::new(
|
||||
std::io::ErrorKind::WouldBlock,
|
||||
"a live transitioned source still references the free-version remote tuple",
|
||||
));
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
Ok(targets)
|
||||
}
|
||||
|
||||
fn free_version_cleanup_fences_current(
|
||||
topology_generation: &str,
|
||||
api: &ECStore,
|
||||
bucket_guard: &rustfs_lock::NamespaceLockGuard,
|
||||
object_guards: &[crate::store::ObjectLockDiagGuard],
|
||||
lease: &TierOperationLease,
|
||||
cancel: &CancellationToken,
|
||||
deadline: tokio::time::Instant,
|
||||
) -> bool {
|
||||
!cancel.is_cancelled()
|
||||
&& tokio::time::Instant::now() < deadline
|
||||
&& !bucket_guard.is_lock_lost()
|
||||
&& object_guards.iter().all(|guard| !guard.is_lock_lost())
|
||||
&& lease.is_current_generation()
|
||||
&& free_version_physical_topology_generation(api) == topology_generation
|
||||
}
|
||||
|
||||
async fn cleanup_free_version_exact(api: Arc<ECStore>, oi: &ObjectInfo, cancel: &CancellationToken) -> std::io::Result<bool> {
|
||||
const FREE_VERSION_REMOTE_DEADLINE: StdDuration = StdDuration::from_secs(30);
|
||||
|
||||
let topology_generation = free_version_physical_topology_generation(&api);
|
||||
let bucket_guard = api
|
||||
.acquire_bucket_lifecycle_read_lock(&oi.bucket)
|
||||
.await
|
||||
.map_err(std::io::Error::other)?;
|
||||
let (lease, version_id_exact) = acquire_free_version_tier_lease(oi, &api.tier_config_mgr()).await?;
|
||||
let local_object = encode_dir_object(&oi.name);
|
||||
let object_guards = api
|
||||
.acquire_all_physical_object_write_locks("tier_free_version_cleanup", &oi.bucket, &local_object)
|
||||
.await
|
||||
.map_err(std::io::Error::other)?;
|
||||
let targets = scan_exact_free_version_targets(&api, oi, &local_object).await?;
|
||||
if targets.is_empty() {
|
||||
return Ok(false);
|
||||
}
|
||||
|
||||
let deadline = tokio::time::Instant::now() + FREE_VERSION_REMOTE_DEADLINE;
|
||||
if !free_version_cleanup_fences_current(&topology_generation, &api, &bucket_guard, &object_guards, &lease, cancel, deadline) {
|
||||
return Err(std::io::Error::new(
|
||||
std::io::ErrorKind::WouldBlock,
|
||||
"tier free-version cleanup fence is invalid before remote delete",
|
||||
));
|
||||
}
|
||||
tokio::select! {
|
||||
_ = cancel.cancelled() => {
|
||||
return Err(std::io::Error::new(std::io::ErrorKind::Interrupted, "tier free-version cleanup was cancelled"));
|
||||
}
|
||||
result = tokio::time::timeout_at(
|
||||
deadline,
|
||||
delete_free_version_remote_object_with_lease(oi, &lease, version_id_exact),
|
||||
) => {
|
||||
result
|
||||
.map_err(|_| std::io::Error::new(std::io::ErrorKind::TimedOut, "tier free-version remote delete timed out"))??;
|
||||
}
|
||||
}
|
||||
if !free_version_cleanup_fences_current(&topology_generation, &api, &bucket_guard, &object_guards, &lease, cancel, deadline) {
|
||||
// Remote DELETE is idempotent, but a changed fence makes the local
|
||||
// outcome ambiguous. Keep every marker for a fully fenced retry.
|
||||
return Err(std::io::Error::new(
|
||||
std::io::ErrorKind::WouldBlock,
|
||||
"tier free-version cleanup fence changed after remote delete",
|
||||
));
|
||||
}
|
||||
|
||||
let mut first_error = None;
|
||||
for (set, actual) in &targets {
|
||||
let mut delete_request = FileInfo {
|
||||
name: local_object.clone(),
|
||||
version_id: actual.version_id,
|
||||
..Default::default()
|
||||
};
|
||||
delete_request.set_tier_free_version();
|
||||
if let Err(err) = set
|
||||
.delete_object_version(&oi.bucket, &local_object, &delete_request, false)
|
||||
.await
|
||||
&& first_error.is_none()
|
||||
{
|
||||
first_error = Some(std::io::Error::other(err));
|
||||
}
|
||||
}
|
||||
let remaining = scan_exact_free_version_targets(&api, oi, &local_object).await?;
|
||||
if !remaining.is_empty() {
|
||||
return Err(first_error.unwrap_or_else(|| {
|
||||
std::io::Error::new(
|
||||
std::io::ErrorKind::WouldBlock,
|
||||
"tier free-version cleanup remained on at least one physical set",
|
||||
)
|
||||
}));
|
||||
}
|
||||
if let Some(err) = first_error {
|
||||
return Err(err);
|
||||
}
|
||||
Ok(true)
|
||||
}
|
||||
|
||||
#[cfg(all(test, feature = "test-util"))]
|
||||
async fn delete_free_version_remote_object(
|
||||
oi: &ObjectInfo,
|
||||
tier_config_mgr: &Arc<RwLock<TierConfigMgr>>,
|
||||
) -> Result<(), std::io::Error> {
|
||||
let (lease, version_id_exact) = acquire_free_version_tier_lease(oi, tier_config_mgr).await?;
|
||||
delete_free_version_remote_object_with_lease(oi, &lease, version_id_exact).await
|
||||
let version_id_exact = validate_transition_remote_version(oi)?;
|
||||
let identity = tier_destination_id_from_metadata(&oi.user_defined)?
|
||||
.ok_or_else(|| std::io::Error::other("tier free-version has no durable backend identity"))?;
|
||||
delete_object_from_remote_tier_idempotent_with_manager_and_identity(
|
||||
&oi.transitioned_object.name,
|
||||
&oi.transitioned_object.version_id,
|
||||
&oi.transitioned_object.tier,
|
||||
identity,
|
||||
tier_config_mgr,
|
||||
version_id_exact,
|
||||
)
|
||||
.await?;
|
||||
Ok(())
|
||||
}
|
||||
|
||||
#[allow(
|
||||
@@ -813,11 +618,8 @@ where
|
||||
F: FnOnce() -> Fut,
|
||||
Fut: std::future::Future<Output = T>,
|
||||
{
|
||||
let (lease, version_id_exact) = acquire_free_version_tier_lease(oi, tier_config_mgr).await?;
|
||||
delete_free_version_remote_object_with_lease(oi, &lease, version_id_exact).await?;
|
||||
let result = delete_local().await;
|
||||
drop(lease);
|
||||
Ok(result)
|
||||
delete_free_version_remote_object(oi, tier_config_mgr).await?;
|
||||
Ok(delete_local().await)
|
||||
}
|
||||
|
||||
struct NewerNoncurrentTask {
|
||||
@@ -888,10 +690,6 @@ impl ExpiryState {
|
||||
usize::try_from(self.stats.pending_tasks().max(0)).unwrap_or(usize::MAX)
|
||||
}
|
||||
|
||||
pub fn active_tasks(&self) -> usize {
|
||||
usize::try_from(self.stats.active_tasks().max(0)).unwrap_or(usize::MAX)
|
||||
}
|
||||
|
||||
fn send_expiry_task(&self, wrkr: Sender<Option<ExpiryOpType>>, task: ExpiryOpType) -> bool {
|
||||
let queued = wrkr.try_send(Some(task)).is_ok();
|
||||
if queued {
|
||||
@@ -1028,7 +826,7 @@ impl ExpiryState {
|
||||
}
|
||||
|
||||
pub async fn resize_workers(n: usize, api: Arc<ECStore>) {
|
||||
let expiry_state = api.ctx.expiry_state();
|
||||
let expiry_state = runtime_sources::expiry_state_handle();
|
||||
if n == expiry_state.read().await.tasks_tx.len() || n < 1 {
|
||||
return;
|
||||
}
|
||||
@@ -1069,7 +867,7 @@ impl ExpiryState {
|
||||
stats: Arc<ExpiryStats>,
|
||||
recovery_notify: Arc<Notify>,
|
||||
) {
|
||||
let cancel_token = api.ctx.background_cancel_token().unwrap_or_else(|| {
|
||||
let cancel_token = runtime_sources::background_services_cancel_token().unwrap_or_else(|| {
|
||||
static FALLBACK: std::sync::OnceLock<tokio_util::sync::CancellationToken> = std::sync::OnceLock::new();
|
||||
FALLBACK.get_or_init(tokio_util::sync::CancellationToken::new).clone()
|
||||
});
|
||||
@@ -1170,33 +968,119 @@ impl ExpiryState {
|
||||
else if v.as_any().is::<FreeVersionTask>() {
|
||||
let v = v.as_any().downcast_ref::<FreeVersionTask>().expect("FreeVersionTask downcast failed");
|
||||
let oi = v.0.clone();
|
||||
match cleanup_free_version_exact(api.clone(), &oi, &cancel_token).await {
|
||||
Ok(true) => {}
|
||||
Ok(false) => debug!(
|
||||
if let Err(err) = delete_free_version_remote_object(&oi, &api.tier_config_mgr()).await {
|
||||
recovery_notify.notify_one();
|
||||
debug!(
|
||||
bucket = %oi.bucket,
|
||||
object = %oi.name,
|
||||
remote_object = %oi.transitioned_object.name,
|
||||
remote_version_id = %oi.transitioned_object.version_id,
|
||||
tier = %oi.transitioned_object.tier,
|
||||
error = ?err,
|
||||
event = EVENT_LIFECYCLE_WORKER_STATE,
|
||||
component = LOG_COMPONENT_ECSTORE,
|
||||
subsystem = LOG_SUBSYSTEM_LIFECYCLE,
|
||||
reason = "remote_tier_delete_failed",
|
||||
"Lifecycle worker skipped remote tier delete"
|
||||
);
|
||||
continue;
|
||||
}
|
||||
|
||||
let local_object = encode_dir_object(&oi.name);
|
||||
let mut fi = FileInfo {
|
||||
name: local_object.clone(),
|
||||
version_id: oi.version_id,
|
||||
..Default::default()
|
||||
};
|
||||
// This removes an existing internal cleanup marker. Keeping
|
||||
// `deleted` false makes duplicate tasks return not-found
|
||||
// instead of creating an ordinary delete marker.
|
||||
fi.set_tier_free_version();
|
||||
|
||||
let mut deleted_locally = false;
|
||||
for pool in &api.pools {
|
||||
let set = pool.get_disks_by_key(&local_object);
|
||||
let ns_lock = match set.new_ns_lock(&oi.bucket, &local_object).await {
|
||||
Ok(lock) => lock,
|
||||
Err(err) => {
|
||||
recovery_notify.notify_one();
|
||||
debug!(
|
||||
event = EVENT_LIFECYCLE_WORKER_STATE,
|
||||
component = LOG_COMPONENT_ECSTORE,
|
||||
subsystem = LOG_SUBSYSTEM_LIFECYCLE,
|
||||
bucket = %oi.bucket,
|
||||
object = %oi.name,
|
||||
pool_index = pool.pool_idx,
|
||||
set_index = set.set_index,
|
||||
error = ?err,
|
||||
reason = "local_free_version_lock_failed",
|
||||
"Lifecycle worker failed to create local free-version cleanup lock"
|
||||
);
|
||||
continue;
|
||||
}
|
||||
};
|
||||
let _object_lock_guard =
|
||||
match ns_lock.get_write_lock_quiet(get_lock_acquire_timeout()).await {
|
||||
Ok(guard) => guard,
|
||||
Err(err) => {
|
||||
recovery_notify.notify_one();
|
||||
debug!(
|
||||
event = EVENT_LIFECYCLE_WORKER_STATE,
|
||||
component = LOG_COMPONENT_ECSTORE,
|
||||
subsystem = LOG_SUBSYSTEM_LIFECYCLE,
|
||||
bucket = %oi.bucket,
|
||||
object = %oi.name,
|
||||
pool_index = pool.pool_idx,
|
||||
set_index = set.set_index,
|
||||
error = ?err,
|
||||
reason = "local_free_version_lock_failed",
|
||||
"Lifecycle worker failed to acquire local free-version cleanup lock"
|
||||
);
|
||||
continue;
|
||||
}
|
||||
};
|
||||
match set
|
||||
.delete_object_version(&oi.bucket, &local_object, &fi, false)
|
||||
.await
|
||||
{
|
||||
Ok(()) => {
|
||||
deleted_locally = true;
|
||||
break;
|
||||
}
|
||||
Err(err) if is_err_version_not_found(&err) || is_err_object_not_found(&err) => continue,
|
||||
Err(err) => {
|
||||
recovery_notify.notify_one();
|
||||
debug!(
|
||||
event = EVENT_LIFECYCLE_WORKER_STATE,
|
||||
component = LOG_COMPONENT_ECSTORE,
|
||||
subsystem = LOG_SUBSYSTEM_LIFECYCLE,
|
||||
bucket = %oi.bucket,
|
||||
object = %oi.name,
|
||||
remote_object = %oi.transitioned_object.name,
|
||||
remote_version_id = %oi.transitioned_object.version_id,
|
||||
tier = %oi.transitioned_object.tier,
|
||||
error = ?err,
|
||||
reason = "local_free_version_delete_failed",
|
||||
"Lifecycle worker failed local free-version cleanup"
|
||||
);
|
||||
break;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
if !deleted_locally {
|
||||
debug!(
|
||||
event = EVENT_LIFECYCLE_WORKER_STATE,
|
||||
component = LOG_COMPONENT_ECSTORE,
|
||||
subsystem = LOG_SUBSYSTEM_LIFECYCLE,
|
||||
bucket = %oi.bucket,
|
||||
object = %oi.name,
|
||||
remote_object = %oi.transitioned_object.name,
|
||||
remote_version_id = %oi.transitioned_object.version_id,
|
||||
tier = %oi.transitioned_object.tier,
|
||||
reason = "local_free_version_missing",
|
||||
"Lifecycle worker found that the exact free-version was already absent"
|
||||
),
|
||||
Err(err) => {
|
||||
recovery_notify.notify_one();
|
||||
debug!(
|
||||
event = EVENT_LIFECYCLE_WORKER_STATE,
|
||||
component = LOG_COMPONENT_ECSTORE,
|
||||
subsystem = LOG_SUBSYSTEM_LIFECYCLE,
|
||||
bucket = %oi.bucket,
|
||||
object = %oi.name,
|
||||
remote_object = %oi.transitioned_object.name,
|
||||
remote_version_id = %oi.transitioned_object.version_id,
|
||||
tier = %oi.transitioned_object.tier,
|
||||
error = ?err,
|
||||
reason = "free_version_exact_cleanup_deferred",
|
||||
"Lifecycle worker retained the exact free-version for a fenced retry"
|
||||
);
|
||||
}
|
||||
"Lifecycle worker could not find transitioned free version locally"
|
||||
);
|
||||
}
|
||||
}
|
||||
else {
|
||||
@@ -1268,8 +1152,8 @@ fn set_recovered_free_version_enqueue_observer(
|
||||
RecoveredFreeVersionEnqueueObserverGuard
|
||||
}
|
||||
|
||||
pub async fn enqueue_recovered_free_version(api: &ECStore, oi: ObjectInfo) -> bool {
|
||||
let expiry_state = api.ctx.expiry_state();
|
||||
pub async fn enqueue_recovered_free_version(oi: ObjectInfo) -> bool {
|
||||
let expiry_state = runtime_sources::expiry_state_handle();
|
||||
let queued = enqueue_recovered_free_version_with_state(&expiry_state, oi).await;
|
||||
|
||||
#[cfg(test)]
|
||||
@@ -2696,8 +2580,8 @@ fn spawn_tier_free_version_recovery_once(api: Arc<ECStore>, started: &OnceLock<(
|
||||
}
|
||||
|
||||
Some(tokio::spawn(async move {
|
||||
let cancel_token = api.ctx.background_cancel_token().unwrap_or_default();
|
||||
let expiry_state = api.ctx.expiry_state();
|
||||
let cancel_token = runtime_sources::background_services_cancel_token().unwrap_or_default();
|
||||
let expiry_state = runtime_sources::expiry_state_handle();
|
||||
run_tier_free_version_recovery_loop(
|
||||
cancel_token,
|
||||
expiry_state,
|
||||
@@ -6345,18 +6229,9 @@ mod tests {
|
||||
rustfs_utils::crypto::hex(old_identity),
|
||||
);
|
||||
oi.user_defined = Arc::new(metadata.clone());
|
||||
let lease_observed_during_local_delete = Arc::new(std::sync::atomic::AtomicBool::new(false));
|
||||
delete_free_version_remote_object_then(&oi, &manager, {
|
||||
let local_delete_calls = Arc::clone(&local_delete_calls);
|
||||
let lease_observed_during_local_delete = Arc::clone(&lease_observed_during_local_delete);
|
||||
let manager = manager.clone();
|
||||
move || async move {
|
||||
assert_eq!(
|
||||
crate::services::tier::tier::TierConfigMgr::active_operation_lease_count(&manager, "WARM").await,
|
||||
1,
|
||||
"the identity-bound tier lease must span the exact local marker delete"
|
||||
);
|
||||
lease_observed_during_local_delete.store(true, Ordering::Relaxed);
|
||||
local_delete_calls.fetch_add(1, Ordering::Relaxed);
|
||||
}
|
||||
})
|
||||
@@ -6364,12 +6239,6 @@ mod tests {
|
||||
.expect("matching destination identity should allow idempotent remote cleanup");
|
||||
assert_eq!(old_backend.remove_count().await, 1);
|
||||
assert_eq!(local_delete_calls.load(Ordering::Relaxed), 1);
|
||||
assert!(lease_observed_during_local_delete.load(Ordering::Relaxed));
|
||||
assert_eq!(
|
||||
crate::services::tier::tier::TierConfigMgr::active_operation_lease_count(&manager, "WARM").await,
|
||||
0,
|
||||
"the tier lease should be released after the local marker delete completes"
|
||||
);
|
||||
|
||||
let mut single_prefix_metadata = HashMap::new();
|
||||
single_prefix_metadata.insert(
|
||||
@@ -6603,7 +6472,6 @@ mod tests {
|
||||
let state = ExpiryState::new();
|
||||
let mut state = state.write().await;
|
||||
let je = Jentry {
|
||||
persisted_version: 0,
|
||||
obj_name: "remote/object".to_string(),
|
||||
version_id: "remote-version".to_string(),
|
||||
tier_name: "WARM".to_string(),
|
||||
@@ -6612,7 +6480,6 @@ mod tests {
|
||||
version_state: rustfs_filemeta::TransitionVersionState::Exact,
|
||||
state: crate::bucket::lifecycle::tier_sweeper::TierDeleteJournalState::Committed,
|
||||
source: None,
|
||||
dispatch: None,
|
||||
};
|
||||
|
||||
let err = state
|
||||
@@ -6753,7 +6620,6 @@ mod tests {
|
||||
let state = ExpiryState::new_with_unconsumed_worker_channel(1);
|
||||
let mut state = state.write().await;
|
||||
let je = Jentry {
|
||||
persisted_version: 0,
|
||||
obj_name: "remote/object".to_string(),
|
||||
version_id: "remote-version".to_string(),
|
||||
tier_name: "WARM".to_string(),
|
||||
@@ -6762,7 +6628,6 @@ mod tests {
|
||||
version_state: rustfs_filemeta::TransitionVersionState::Exact,
|
||||
state: crate::bucket::lifecycle::tier_sweeper::TierDeleteJournalState::Committed,
|
||||
source: None,
|
||||
dispatch: None,
|
||||
};
|
||||
|
||||
state
|
||||
@@ -6894,7 +6759,7 @@ mod tests {
|
||||
};
|
||||
|
||||
assert!(
|
||||
super::enqueue_recovered_free_version(&ecstore, oi).await,
|
||||
super::enqueue_recovered_free_version(oi).await,
|
||||
"the resized production worker queue should accept the task"
|
||||
);
|
||||
stop_tx.send(None).await.expect("worker stop signal should be delivered");
|
||||
@@ -7010,12 +6875,12 @@ mod tests {
|
||||
.await
|
||||
.expect("free-version task should reach the worker");
|
||||
tokio::time::timeout(StdDuration::from_secs(30), async {
|
||||
while stats.active_tasks() == 0 {
|
||||
while remote_backend.remove_count().await == 0 {
|
||||
tokio::task::yield_now().await;
|
||||
}
|
||||
})
|
||||
.await
|
||||
.expect("worker should mark the cleanup task active before the lock assertion");
|
||||
.expect("worker should complete remote cleanup before taking the local lock");
|
||||
let completed_while_locked = tokio::time::timeout(StdDuration::from_millis(100), async {
|
||||
while stats.active_tasks() != 0 {
|
||||
tokio::task::yield_now().await;
|
||||
@@ -7024,12 +6889,7 @@ mod tests {
|
||||
.await;
|
||||
assert!(
|
||||
completed_while_locked.is_err(),
|
||||
"the cleanup task must wait while a competing object writer owns the namespace lock"
|
||||
);
|
||||
assert_eq!(
|
||||
remote_backend.remove_count().await,
|
||||
0,
|
||||
"the remote tuple must not be deleted before the all-physical namespace fence is acquired"
|
||||
"local cleanup must wait while a competing object writer owns the namespace lock"
|
||||
);
|
||||
for disk_path in &disk_paths {
|
||||
assert!(
|
||||
@@ -7040,13 +6900,6 @@ mod tests {
|
||||
}
|
||||
|
||||
drop(object_lock_guard);
|
||||
tokio::time::timeout(StdDuration::from_secs(30), async {
|
||||
while remote_backend.remove_count().await == 0 {
|
||||
tokio::task::yield_now().await;
|
||||
}
|
||||
})
|
||||
.await
|
||||
.expect("worker should delete the remote tuple after acquiring the released namespace fence");
|
||||
tx.send(None).await.expect("worker stop signal should be delivered");
|
||||
worker.await.expect("free-version worker should stop cleanly");
|
||||
|
||||
@@ -7142,7 +6995,6 @@ mod tests {
|
||||
.next()
|
||||
.expect("seeded free version should be recoverable");
|
||||
let stale_version_id = oi.version_id.expect("free version should have a concrete UUID");
|
||||
let ordinary_marker_mod_time = OffsetDateTime::now_utc();
|
||||
|
||||
for disk_path in &disk_paths {
|
||||
let metadata_path = disk_path.join(&bucket).join(object).join(STORAGE_FORMAT_FILE);
|
||||
@@ -7165,7 +7017,7 @@ mod tests {
|
||||
name: object.to_string(),
|
||||
version_id: Some(stale_version_id),
|
||||
deleted: true,
|
||||
mod_time: Some(ordinary_marker_mod_time),
|
||||
mod_time: Some(OffsetDateTime::now_utc()),
|
||||
..Default::default()
|
||||
})
|
||||
.expect("same-ID ordinary marker should replace the stale free version");
|
||||
@@ -7179,13 +7031,6 @@ mod tests {
|
||||
.expect("same-ID ordinary marker metadata should be written");
|
||||
}
|
||||
|
||||
assert!(
|
||||
!super::cleanup_free_version_exact(Arc::clone(&ecstore), &oi, &CancellationToken::new())
|
||||
.await
|
||||
.expect("a stale task whose local UUID now names an ordinary marker should be an idempotent no-op"),
|
||||
"the stale free-version task must not report local cleanup"
|
||||
);
|
||||
|
||||
let state = ExpiryState::new();
|
||||
let (stats, recovery_notify) = {
|
||||
let state = state.read().await;
|
||||
@@ -11677,7 +11522,7 @@ mod tests {
|
||||
|
||||
#[cfg(feature = "test-util")]
|
||||
#[tokio::test]
|
||||
async fn journal_replay_quarantines_legacy_unknown_version_state_before_backend_io() {
|
||||
async fn journal_replay_rejects_unknown_version_state_before_backend_io() {
|
||||
let (_disk_paths, ecstore) = setup_test_env().await;
|
||||
let (backend, _) = register_recovery_mock_tier(&ecstore).await;
|
||||
let identity = TierConfigMgr::acquire_operation_lease(&ecstore.tier_config_mgr(), "WARM")
|
||||
@@ -11685,7 +11530,6 @@ mod tests {
|
||||
.expect("mock tier lease should be available")
|
||||
.backend_identity();
|
||||
let je = Jentry {
|
||||
persisted_version: 0,
|
||||
obj_name: "remote/object".to_string(),
|
||||
version_id: "legacy-version".to_string(),
|
||||
tier_name: "WARM".to_string(),
|
||||
@@ -11694,28 +11538,25 @@ mod tests {
|
||||
version_state: rustfs_filemeta::TransitionVersionState::Unknown,
|
||||
state: crate::bucket::lifecycle::tier_sweeper::TierDeleteJournalState::Committed,
|
||||
source: None,
|
||||
dispatch: None,
|
||||
};
|
||||
|
||||
crate::bucket::lifecycle::tier_delete_journal::persist_tier_delete_journal_entry(ecstore.clone(), &je)
|
||||
.await
|
||||
.expect("legacy unknown journal should remain byte-compatible and persistable");
|
||||
let err = crate::bucket::lifecycle::tier_delete_journal::process_tier_delete_journal_entry(ecstore, &je)
|
||||
.await
|
||||
.expect_err("legacy unknown journal must be quarantined before backend IO");
|
||||
.expect_err("unknown journal state must fail before backend IO");
|
||||
|
||||
assert_eq!(err.kind(), std::io::ErrorKind::WouldBlock);
|
||||
assert_eq!(err.kind(), std::io::ErrorKind::InvalidData);
|
||||
assert_eq!(backend.remove_count().await, 0);
|
||||
}
|
||||
|
||||
#[cfg(feature = "test-util")]
|
||||
#[tokio::test]
|
||||
async fn rejected_upload_cleanup_retries_confirmed_exact_provider_token_without_legacy_journal() {
|
||||
async fn journal_replay_deletes_confirmed_exact_provider_token() {
|
||||
let (_disk_paths, ecstore) = setup_test_env().await;
|
||||
let (backend, _) = register_recovery_mock_tier(&ecstore).await;
|
||||
let lease = TierConfigMgr::acquire_operation_lease(&ecstore.tier_config_mgr(), "WARM")
|
||||
.await
|
||||
.expect("mock tier lease should be available");
|
||||
let identity = lease.backend_identity();
|
||||
backend
|
||||
.set_put_remote_version(Some("provider-version-token".to_string()))
|
||||
.await;
|
||||
@@ -11729,30 +11570,34 @@ mod tests {
|
||||
.expect("confirmed remote candidate should be seeded");
|
||||
backend.set_remove_failure(true);
|
||||
backend.set_reject_non_empty_remote_versions(true);
|
||||
let err = crate::set_disk::cleanup_rejected_transition_upload_durably(
|
||||
let je = Jentry {
|
||||
obj_name: "remote/object".to_string(),
|
||||
version_id: "provider-version-token".to_string(),
|
||||
tier_name: "WARM".to_string(),
|
||||
backend_identity: Some(identity),
|
||||
version_id_exact: true,
|
||||
version_state: rustfs_filemeta::TransitionVersionState::Exact,
|
||||
state: crate::bucket::lifecycle::tier_sweeper::TierDeleteJournalState::Committed,
|
||||
source: None,
|
||||
};
|
||||
|
||||
crate::set_disk::cleanup_rejected_transition_upload_durably(
|
||||
&lease,
|
||||
"remote/object",
|
||||
"provider-version-token",
|
||||
&je.obj_name,
|
||||
&je.version_id,
|
||||
true,
|
||||
Some(ecstore.clone()),
|
||||
)
|
||||
.await
|
||||
.expect_err("a failed immediate cleanup must remain owned by the caller's transition transaction");
|
||||
assert_eq!(err.kind(), std::io::ErrorKind::Other);
|
||||
assert!(backend.contains("remote/object").await);
|
||||
.expect("failed immediate cleanup should remain durable in the journal");
|
||||
assert!(backend.contains(&je.obj_name).await);
|
||||
|
||||
backend.set_remove_failure(false);
|
||||
crate::set_disk::cleanup_rejected_transition_upload_durably(
|
||||
&lease,
|
||||
"remote/object",
|
||||
"provider-version-token",
|
||||
true,
|
||||
Some(ecstore),
|
||||
)
|
||||
.await
|
||||
.expect("the transaction retry must delete the same confirmed candidate");
|
||||
crate::bucket::lifecycle::tier_delete_journal::process_tier_delete_journal_entry(ecstore, &je)
|
||||
.await
|
||||
.expect("identity-bound exact journal must retry confirmed candidate cleanup");
|
||||
|
||||
assert!(!backend.contains("remote/object").await);
|
||||
assert!(!backend.contains(&je.obj_name).await);
|
||||
assert_eq!(backend.exact_remove_count(), 2);
|
||||
assert_eq!(
|
||||
backend.remove_versions().await,
|
||||
@@ -11915,14 +11760,11 @@ mod tests {
|
||||
};
|
||||
let mut recovery_rx = recovery_rx.lock().await;
|
||||
assert!(
|
||||
super::enqueue_recovered_free_version(
|
||||
&ecstore,
|
||||
ObjectInfo {
|
||||
bucket: "prefill".to_string(),
|
||||
name: "prefill".to_string(),
|
||||
..Default::default()
|
||||
},
|
||||
)
|
||||
super::enqueue_recovered_free_version(ObjectInfo {
|
||||
bucket: "prefill".to_string(),
|
||||
name: "prefill".to_string(),
|
||||
..Default::default()
|
||||
})
|
||||
.await,
|
||||
"the production recovery queue should accept its first task"
|
||||
);
|
||||
@@ -12353,7 +12195,7 @@ mod tests {
|
||||
#[tokio::test]
|
||||
#[serial]
|
||||
async fn tier_free_version_recovery_continues_after_deleted_marker_bucket() {
|
||||
let (disk_paths, ecstore) = setup_test_env().await;
|
||||
let (_paths, ecstore) = setup_test_env().await;
|
||||
let suffix = Uuid::new_v4().simple();
|
||||
let earlier_bucket = format!("zzzz-recovery-{suffix}-a");
|
||||
let deleted_marker = format!("zzzz-recovery-{suffix}-m");
|
||||
@@ -12361,7 +12203,11 @@ mod tests {
|
||||
let later_object = "a-before-stale-marker";
|
||||
create_test_bucket(&ecstore, &earlier_bucket).await;
|
||||
create_test_bucket(&ecstore, &later_bucket).await;
|
||||
seed_recoverable_free_version(&disk_paths, &later_bucket, later_object, None, None).await;
|
||||
let mut reader = PutObjReader::from_vec(b"cursor reset probe".to_vec());
|
||||
ecstore
|
||||
.put_object(&later_bucket, later_object, &mut reader, &ObjectOptions::default())
|
||||
.await
|
||||
.expect("successor bucket object should be created");
|
||||
|
||||
let page = list_tier_free_versions(
|
||||
Arc::clone(&ecstore),
|
||||
@@ -12374,10 +12220,14 @@ mod tests {
|
||||
.expect("recovery should resume at the first bucket after a deleted marker bucket");
|
||||
|
||||
assert_eq!(page.buckets_scanned, 1, "the later bucket must not be skipped");
|
||||
assert_eq!(page.items.len(), 1, "the successor bucket's recoverable object must be returned");
|
||||
assert_eq!(page.items[0].bucket, later_bucket);
|
||||
assert_eq!(page.items[0].name, later_object);
|
||||
remove_seeded_free_version(&disk_paths, &later_bucket, later_object).await;
|
||||
assert_eq!(
|
||||
page.scanned_entries, 1,
|
||||
"the deleted bucket's object marker must not skip objects in the successor bucket"
|
||||
);
|
||||
ecstore
|
||||
.delete_object(&later_bucket, later_object, ObjectOptions::default())
|
||||
.await
|
||||
.expect("successor bucket object should be removed");
|
||||
for bucket in [&earlier_bucket, &later_bucket] {
|
||||
ecstore
|
||||
.delete_bucket(bucket, &DeleteBucketOptions::default())
|
||||
|
||||
@@ -33,7 +33,6 @@ const MANUAL_TRANSITION_CURSOR_MARKER_PROOF_MAX_SIZE: usize = 1024;
|
||||
#[derive(Debug, Clone, Copy, PartialEq, Eq)]
|
||||
pub(crate) enum DurableIlmRecordKind {
|
||||
TierDeleteJournal,
|
||||
TierDeleteDispatchManifest,
|
||||
TransitionTransaction,
|
||||
ManualTransitionJob,
|
||||
ManualTransitionScope,
|
||||
@@ -55,18 +54,6 @@ pub(crate) const TIER_DELETE_JOURNAL_NAMESPACE: DurableIlmNamespace = DurableIlm
|
||||
max_record_size: 64 * 1024,
|
||||
kind: DurableIlmRecordKind::TierDeleteJournal,
|
||||
};
|
||||
pub(crate) const TIER_DELETE_JOURNAL_V6_NAMESPACE: DurableIlmNamespace = DurableIlmNamespace {
|
||||
name: "tier-delete-journal-v6",
|
||||
prefix: "ilm/tier-delete-journal-v6/",
|
||||
max_record_size: 64 * 1024,
|
||||
kind: DurableIlmRecordKind::TierDeleteJournal,
|
||||
};
|
||||
pub(crate) const TIER_DELETE_DISPATCH_MANIFEST_NAMESPACE: DurableIlmNamespace = DurableIlmNamespace {
|
||||
name: "tier-delete-dispatch-manifest",
|
||||
prefix: tier_delete_journal::TIER_DELETE_DISPATCH_MANIFEST_PREFIX,
|
||||
max_record_size: tier_delete_journal::MAX_TIER_DELETE_DISPATCH_MANIFEST_SIZE,
|
||||
kind: DurableIlmRecordKind::TierDeleteDispatchManifest,
|
||||
};
|
||||
pub(crate) const TRANSITION_TRANSACTION_NAMESPACE: DurableIlmNamespace = DurableIlmNamespace {
|
||||
name: "transition-transaction",
|
||||
prefix: "ilm/transition-transactions/records",
|
||||
@@ -98,10 +85,8 @@ pub(crate) const MANUAL_TRANSITION_WORKER_RESULT_NAMESPACE: DurableIlmNamespace
|
||||
kind: DurableIlmRecordKind::ManualTransitionWorkerResult,
|
||||
};
|
||||
|
||||
pub(crate) const DURABLE_ILM_NAMESPACES: [DurableIlmNamespace; 8] = [
|
||||
pub(crate) const DURABLE_ILM_NAMESPACES: [DurableIlmNamespace; 6] = [
|
||||
TIER_DELETE_JOURNAL_NAMESPACE,
|
||||
TIER_DELETE_JOURNAL_V6_NAMESPACE,
|
||||
TIER_DELETE_DISPATCH_MANIFEST_NAMESPACE,
|
||||
TRANSITION_TRANSACTION_NAMESPACE,
|
||||
MANUAL_TRANSITION_JOB_NAMESPACE,
|
||||
MANUAL_TRANSITION_SCOPE_NAMESPACE,
|
||||
@@ -172,15 +157,6 @@ pub(crate) enum DurableIlmRecordCheckpoint {
|
||||
content_sha256: String,
|
||||
identity_sha256: String,
|
||||
committed: bool,
|
||||
#[serde(default, skip_serializing_if = "Option::is_none")]
|
||||
dispatch_identity_sha256: Option<String>,
|
||||
#[serde(default, skip_serializing_if = "Option::is_none")]
|
||||
state: Option<super::tier_sweeper::TierDeleteJournalState>,
|
||||
},
|
||||
TierDeleteDispatchManifest {
|
||||
content_sha256: String,
|
||||
identity_sha256: String,
|
||||
state: tier_delete_journal::TierDeleteDispatchManifestState,
|
||||
},
|
||||
TransitionTransaction {
|
||||
content_sha256: String,
|
||||
@@ -219,7 +195,6 @@ impl DurableIlmRecordCheckpoint {
|
||||
pub(crate) fn content_sha256(&self) -> &str {
|
||||
match self {
|
||||
Self::TierDeleteJournal { content_sha256, .. }
|
||||
| Self::TierDeleteDispatchManifest { content_sha256, .. }
|
||||
| Self::TransitionTransaction { content_sha256, .. }
|
||||
| Self::ManualTransitionJob { content_sha256, .. }
|
||||
| Self::ManualTransitionScope { content_sha256, .. }
|
||||
@@ -253,19 +228,6 @@ impl DurableIlmRecordCheckpoint {
|
||||
}
|
||||
|
||||
pub(crate) fn validate_successor(&self, next: &Self) -> Result<()> {
|
||||
for checkpoint in [self, next] {
|
||||
if let Self::TierDeleteJournal {
|
||||
committed,
|
||||
dispatch_identity_sha256,
|
||||
state,
|
||||
..
|
||||
} = checkpoint
|
||||
&& (state.is_some() != dispatch_identity_sha256.is_some()
|
||||
|| state.is_some_and(|state| *committed != (state == super::tier_sweeper::TierDeleteJournalState::Committed)))
|
||||
{
|
||||
return Err(Error::other("durable ILM tier delete journal checkpoint is invalid"));
|
||||
}
|
||||
}
|
||||
if self == next {
|
||||
if let Self::ManualTransitionJob {
|
||||
progress,
|
||||
@@ -282,64 +244,18 @@ impl DurableIlmRecordCheckpoint {
|
||||
let valid = match (self, next) {
|
||||
(
|
||||
Self::TierDeleteJournal {
|
||||
content_sha256: previous_content,
|
||||
identity_sha256: previous_identity,
|
||||
committed: previous_committed,
|
||||
dispatch_identity_sha256: previous_dispatch_identity,
|
||||
state: previous_state,
|
||||
..
|
||||
},
|
||||
Self::TierDeleteJournal {
|
||||
content_sha256: next_content,
|
||||
identity_sha256: next_identity,
|
||||
committed: next_committed,
|
||||
dispatch_identity_sha256: next_dispatch_identity,
|
||||
state: next_state,
|
||||
..
|
||||
},
|
||||
) => {
|
||||
use super::tier_sweeper::TierDeleteJournalState::{Committed, Dispatched, Prepared};
|
||||
|
||||
let dispatch_identity_is_monotonic = match (previous_dispatch_identity, next_dispatch_identity) {
|
||||
(Some(previous), Some(next)) => previous == next,
|
||||
(None, None) => true,
|
||||
// Old receipts did not record the v6 dispatch binding. A
|
||||
// byte-identical observation may adopt the stronger proof,
|
||||
// but an in-flight mutation must fail closed instead of
|
||||
// guessing which operation owned the journal.
|
||||
(None, Some(_)) => previous_content == next_content,
|
||||
(Some(_), None) => false,
|
||||
};
|
||||
let state_is_monotonic = match (previous_state, next_state) {
|
||||
(Some(previous), Some(next)) => {
|
||||
previous == next || matches!((previous, next), (Prepared, Dispatched) | (Dispatched, Committed))
|
||||
}
|
||||
(None, None) => previous_committed == next_committed || (!previous_committed && *next_committed),
|
||||
(None, Some(_)) => previous_content == next_content,
|
||||
(Some(_), None) => false,
|
||||
};
|
||||
previous_identity == next_identity && dispatch_identity_is_monotonic && state_is_monotonic
|
||||
}
|
||||
(
|
||||
Self::TierDeleteDispatchManifest {
|
||||
identity_sha256: previous_identity,
|
||||
state: previous_state,
|
||||
..
|
||||
},
|
||||
Self::TierDeleteDispatchManifest {
|
||||
identity_sha256: next_identity,
|
||||
state: next_state,
|
||||
..
|
||||
},
|
||||
) => {
|
||||
use tier_delete_journal::TierDeleteDispatchManifestState::{
|
||||
Aborted, Aborting, Completed, DispatchAuthorized, Preparing,
|
||||
};
|
||||
previous_identity == next_identity
|
||||
&& matches!(
|
||||
(previous_state, next_state),
|
||||
(Preparing, DispatchAuthorized | Aborting) | (Aborting, Aborted) | (DispatchAuthorized, Completed)
|
||||
)
|
||||
&& (previous_committed == next_committed || (!previous_committed && *next_committed))
|
||||
}
|
||||
(
|
||||
Self::TransitionTransaction {
|
||||
@@ -435,49 +351,6 @@ impl DurableIlmRecordCheckpoint {
|
||||
Err(Error::other("durable ILM record generation is not a monotonic successor"))
|
||||
}
|
||||
}
|
||||
|
||||
/// Whether `self` is an older generation of the same immutable record
|
||||
/// that can reach `terminal` through one or more valid state transitions.
|
||||
/// This is deliberately broader than `validate_successor`, which remains
|
||||
/// adjacent-only for receipt advancement. Terminal cleanup uses this only
|
||||
/// after the exact terminal ETag and terminal receipt were committed, to
|
||||
/// purge older object versions exposed by that deletion.
|
||||
pub(crate) fn is_predecessor_of_terminal(&self, terminal: &Self) -> bool {
|
||||
if self == terminal || self.validate_successor(terminal).is_ok() {
|
||||
return true;
|
||||
}
|
||||
match (self, terminal) {
|
||||
(
|
||||
Self::TierDeleteJournal {
|
||||
identity_sha256: previous_identity,
|
||||
dispatch_identity_sha256: previous_dispatch,
|
||||
state: Some(super::tier_sweeper::TierDeleteJournalState::Prepared),
|
||||
..
|
||||
},
|
||||
Self::TierDeleteJournal {
|
||||
identity_sha256: terminal_identity,
|
||||
dispatch_identity_sha256: terminal_dispatch,
|
||||
state: Some(super::tier_sweeper::TierDeleteJournalState::Committed),
|
||||
..
|
||||
},
|
||||
) => previous_identity == terminal_identity && previous_dispatch == terminal_dispatch,
|
||||
(
|
||||
Self::TierDeleteDispatchManifest {
|
||||
identity_sha256: previous_identity,
|
||||
state: tier_delete_journal::TierDeleteDispatchManifestState::Preparing,
|
||||
..
|
||||
},
|
||||
Self::TierDeleteDispatchManifest {
|
||||
identity_sha256: terminal_identity,
|
||||
state:
|
||||
tier_delete_journal::TierDeleteDispatchManifestState::Aborted
|
||||
| tier_delete_journal::TierDeleteDispatchManifestState::Completed,
|
||||
..
|
||||
},
|
||||
) => previous_identity == terminal_identity,
|
||||
_ => false,
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
fn transition_state_distance(
|
||||
@@ -877,19 +750,10 @@ pub(crate) fn validate_durable_ilm_record(path: &str, data: &[u8]) -> Result<Val
|
||||
if tier_delete_journal::tier_delete_journal_object_name(&entry) != path {
|
||||
return Err(Error::other("tier delete journal content does not match its path"));
|
||||
}
|
||||
let legacy_operation_id = path
|
||||
let operation_id = path
|
||||
.strip_prefix(namespace.prefix)
|
||||
.and_then(|suffix| suffix.strip_suffix(".json"))
|
||||
.ok_or_else(|| Error::other("tier delete journal path is invalid"))?;
|
||||
// Legacy v1-v5 paths already expose a 64-hex operation id and
|
||||
// must remain receipt-compatible. V6 uses an operation-scoped
|
||||
// nested path, so derive a fixed, path-unique receipt id instead
|
||||
// of embedding slashes in the receipt locator.
|
||||
let operation_id = if entry.persisted_version == 6 {
|
||||
hex_sha256(path.as_bytes(), ToOwned::to_owned)
|
||||
} else {
|
||||
legacy_operation_id.to_string()
|
||||
};
|
||||
let identity_sha256 = checkpoint_hash(&(
|
||||
&entry.obj_name,
|
||||
&entry.version_id,
|
||||
@@ -899,29 +763,13 @@ pub(crate) fn validate_durable_ilm_record(path: &str, data: &[u8]) -> Result<Val
|
||||
entry.version_state,
|
||||
&entry.source,
|
||||
))?;
|
||||
let dispatch_identity_sha256 = entry.dispatch.as_ref().map(checkpoint_hash).transpose()?;
|
||||
(
|
||||
"operation_id",
|
||||
operation_id,
|
||||
operation_id.to_string(),
|
||||
DurableIlmRecordCheckpoint::TierDeleteJournal {
|
||||
content_sha256,
|
||||
identity_sha256,
|
||||
committed: entry.state == super::tier_sweeper::TierDeleteJournalState::Committed,
|
||||
dispatch_identity_sha256,
|
||||
state: (entry.persisted_version == 6).then_some(entry.state),
|
||||
},
|
||||
)
|
||||
}
|
||||
DurableIlmRecordKind::TierDeleteDispatchManifest => {
|
||||
let (operation_id, identity_sha256, state) =
|
||||
tier_delete_journal::validate_tier_delete_dispatch_manifest_record(path, data)?;
|
||||
(
|
||||
"operation_id",
|
||||
hex_sha256(operation_id.as_bytes(), ToOwned::to_owned),
|
||||
DurableIlmRecordCheckpoint::TierDeleteDispatchManifest {
|
||||
content_sha256,
|
||||
identity_sha256,
|
||||
state,
|
||||
},
|
||||
)
|
||||
}
|
||||
@@ -1108,87 +956,6 @@ mod tests {
|
||||
}
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn tier_delete_dispatch_manifest_namespace_validates_monotonic_branches() {
|
||||
use tier_delete_journal::TierDeleteDispatchManifestState::{Aborted, Aborting, Completed, DispatchAuthorized, Preparing};
|
||||
|
||||
let operation_id = Uuid::new_v4();
|
||||
let checkpoint = |state| {
|
||||
let (path, data) = tier_delete_journal::test_tier_delete_dispatch_manifest_record(operation_id, state);
|
||||
let namespace = classify_durable_ilm_record(&path)
|
||||
.expect("dispatch manifest namespace should classify")
|
||||
.expect("dispatch manifest should be durable");
|
||||
assert_eq!(namespace, &TIER_DELETE_DISPATCH_MANIFEST_NAMESPACE);
|
||||
validate_durable_ilm_record(&path, &data)
|
||||
.expect("dispatch manifest should validate")
|
||||
.checkpoint
|
||||
};
|
||||
|
||||
let preparing = checkpoint(Preparing);
|
||||
let authorized = checkpoint(DispatchAuthorized);
|
||||
let completed = checkpoint(Completed);
|
||||
let aborting = checkpoint(Aborting);
|
||||
let aborted = checkpoint(Aborted);
|
||||
|
||||
preparing
|
||||
.validate_successor(&authorized)
|
||||
.expect("Preparing may become DispatchAuthorized");
|
||||
authorized
|
||||
.validate_successor(&completed)
|
||||
.expect("DispatchAuthorized may become Completed");
|
||||
preparing.validate_successor(&aborting).expect("Preparing may enter rollback");
|
||||
aborting.validate_successor(&aborted).expect("Aborting may become Aborted");
|
||||
assert!(authorized.validate_successor(&aborting).is_err());
|
||||
assert!(completed.validate_successor(&authorized).is_err());
|
||||
assert!(aborted.validate_successor(&preparing).is_err());
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn tier_delete_journal_checkpoint_binds_dispatch_and_full_state_monotonically() {
|
||||
use crate::bucket::lifecycle::tier_sweeper::TierDeleteJournalState::{Committed, Dispatched, Prepared};
|
||||
|
||||
let checkpoint = |content: &str, dispatch: Option<&str>, state| DurableIlmRecordCheckpoint::TierDeleteJournal {
|
||||
content_sha256: content.repeat(64),
|
||||
identity_sha256: "i".repeat(64),
|
||||
committed: state == Some(Committed),
|
||||
dispatch_identity_sha256: dispatch.map(|value| value.repeat(64)),
|
||||
state,
|
||||
};
|
||||
let prepared = checkpoint("a", Some("d"), Some(Prepared));
|
||||
let dispatched = checkpoint("b", Some("d"), Some(Dispatched));
|
||||
let committed = checkpoint("c", Some("d"), Some(Committed));
|
||||
prepared
|
||||
.validate_successor(&dispatched)
|
||||
.expect("Prepared may advance to Dispatched");
|
||||
dispatched
|
||||
.validate_successor(&committed)
|
||||
.expect("Dispatched may advance to Committed");
|
||||
assert!(prepared.validate_successor(&committed).is_err());
|
||||
assert!(dispatched.validate_successor(&prepared).is_err());
|
||||
|
||||
let rebound = checkpoint("b", Some("e"), Some(Dispatched));
|
||||
assert!(dispatched.validate_successor(&rebound).is_err());
|
||||
|
||||
let legacy: DurableIlmRecordCheckpoint = serde_json::from_value(serde_json::json!({
|
||||
"kind": "tier_delete_journal",
|
||||
"content_sha256": "a".repeat(64),
|
||||
"identity_sha256": "i".repeat(64),
|
||||
"committed": false
|
||||
}))
|
||||
.expect("legacy tier-delete checkpoint should remain decodable");
|
||||
legacy
|
||||
.validate_successor(&prepared)
|
||||
.expect("byte-identical legacy receipt may adopt the stronger v6 proof");
|
||||
let changed_legacy = DurableIlmRecordCheckpoint::TierDeleteJournal {
|
||||
content_sha256: "z".repeat(64),
|
||||
identity_sha256: "i".repeat(64),
|
||||
committed: false,
|
||||
dispatch_identity_sha256: None,
|
||||
state: None,
|
||||
};
|
||||
assert!(changed_legacy.validate_successor(&prepared).is_err());
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn manual_transition_job_checkpoint_compacts_legacy_progress_compatibly() {
|
||||
let options = super::super::bucket_lifecycle_ops::ManualTransitionRunOptions::default();
|
||||
|
||||
@@ -34,6 +34,6 @@ pub mod tier_sweeper;
|
||||
pub mod transition_transaction;
|
||||
|
||||
pub(crate) use durable_namespace::{
|
||||
DurableIlmRecordCheckpoint, ILM_META_PREFIX, TIER_DELETE_DISPATCH_MANIFEST_NAMESPACE, ValidatedDurableIlmRecord,
|
||||
classify_durable_ilm_record, validate_durable_ilm_record,
|
||||
DurableIlmRecordCheckpoint, ILM_META_PREFIX, ValidatedDurableIlmRecord, classify_durable_ilm_record,
|
||||
validate_durable_ilm_record,
|
||||
};
|
||||
|
||||
File diff suppressed because it is too large
Load Diff
@@ -172,8 +172,7 @@ pub(super) async fn recover_tier_free_versions_with_cancel(
|
||||
return Err(std::io::Error::other("free-version recovery limit must be greater than zero").into());
|
||||
}
|
||||
|
||||
let page =
|
||||
list_tier_free_versions(api.clone(), limit, bucket_marker.clone(), object_marker.clone(), cancel_token.clone()).await?;
|
||||
let page = list_tier_free_versions(api, limit, bucket_marker.clone(), object_marker.clone(), cancel_token.clone()).await?;
|
||||
let mut stats = FreeVersionRecoveryStats {
|
||||
scanned: 0,
|
||||
enqueued: 0,
|
||||
@@ -191,7 +190,7 @@ pub(super) async fn recover_tier_free_versions_with_cancel(
|
||||
return Err(tier_free_version_recovery_cancelled());
|
||||
}
|
||||
retry_cursor.visit(&oi);
|
||||
if !record_recovered_free_version_enqueue(&mut stats, enqueue_recovered_free_version(&api, oi).await) {
|
||||
if !record_recovered_free_version_enqueue(&mut stats, enqueue_recovered_free_version(oi).await) {
|
||||
let (bucket_marker, object_marker) = retry_cursor.retry_markers();
|
||||
stats.truncated = true;
|
||||
stats.next_bucket_marker = bucket_marker;
|
||||
|
||||
@@ -255,7 +255,6 @@ impl ObjSweeper {
|
||||
}
|
||||
if del_tier {
|
||||
return Some(Jentry {
|
||||
persisted_version: 0,
|
||||
obj_name: self.remote_object.clone(),
|
||||
version_id: self.transition_version_id.clone(),
|
||||
tier_name: self.transition_tier.clone(),
|
||||
@@ -267,7 +266,6 @@ impl ObjSweeper {
|
||||
version_state: self.transition_version_state,
|
||||
state: TierDeleteJournalState::Committed,
|
||||
source: None,
|
||||
dispatch: None,
|
||||
});
|
||||
}
|
||||
None
|
||||
@@ -300,19 +298,9 @@ impl ObjSweeper {
|
||||
#[derive(Debug, Clone, Copy, Serialize, Deserialize, PartialEq, Eq)]
|
||||
pub(crate) enum TierDeleteJournalState {
|
||||
Prepared,
|
||||
Dispatched,
|
||||
Committed,
|
||||
}
|
||||
|
||||
#[derive(Debug, Clone, Serialize, Deserialize, PartialEq, Eq)]
|
||||
#[serde(deny_unknown_fields)]
|
||||
pub(crate) struct TierDeleteDispatchBinding {
|
||||
pub(crate) operation_id: Uuid,
|
||||
pub(crate) manifest_object: String,
|
||||
pub(crate) journal_set_sha256: String,
|
||||
pub(crate) topology_generation: String,
|
||||
}
|
||||
|
||||
#[derive(Debug, Clone, Serialize, Deserialize, PartialEq, Eq)]
|
||||
#[serde(deny_unknown_fields)]
|
||||
pub(crate) struct TierDeleteSourceIdentity {
|
||||
@@ -354,10 +342,6 @@ impl TierDeleteSourceIdentity {
|
||||
#[derive(Debug, Clone)]
|
||||
#[allow(unused_assignments)]
|
||||
pub struct Jentry {
|
||||
/// On-disk format version when decoded. Newly constructed entries use 0;
|
||||
/// the encoder chooses their format from the durable ownership fields.
|
||||
/// Recovery uses this value to quarantine v1-v5 without rewriting them.
|
||||
pub(crate) persisted_version: u8,
|
||||
pub(crate) obj_name: String,
|
||||
pub(crate) version_id: String,
|
||||
pub(crate) tier_name: String,
|
||||
@@ -366,23 +350,6 @@ pub struct Jentry {
|
||||
pub(crate) version_state: rustfs_filemeta::TransitionVersionState,
|
||||
pub(crate) state: TierDeleteJournalState,
|
||||
pub(crate) source: Option<TierDeleteSourceIdentity>,
|
||||
pub(crate) dispatch: Option<TierDeleteDispatchBinding>,
|
||||
}
|
||||
|
||||
impl Jentry {
|
||||
/// Whether this prepared transaction is eligible to become the sole
|
||||
/// cleanup owner for its transitioned source. The caller may use this to
|
||||
/// decide whether to persist it, but must not set `skip_free_version`
|
||||
/// until persistence succeeds.
|
||||
pub(crate) fn can_replace_tier_free_version(&self) -> bool {
|
||||
self.state == TierDeleteJournalState::Prepared
|
||||
&& self.backend_identity.is_some()
|
||||
&& self.version_state != rustfs_filemeta::TransitionVersionState::Unknown
|
||||
&& self
|
||||
.source
|
||||
.as_ref()
|
||||
.is_some_and(TierDeleteSourceIdentity::has_stable_identity)
|
||||
}
|
||||
}
|
||||
|
||||
impl ExpiryOp for Jentry {
|
||||
@@ -650,7 +617,6 @@ pub fn transitioned_force_delete_journal_entry(
|
||||
}
|
||||
|
||||
Some(Jentry {
|
||||
persisted_version: 0,
|
||||
obj_name: transitioned.name.clone(),
|
||||
version_id: transitioned.version_id.clone(),
|
||||
tier_name: transitioned.tier.clone(),
|
||||
@@ -662,7 +628,6 @@ pub fn transitioned_force_delete_journal_entry(
|
||||
version_state: transition_version_state,
|
||||
state: TierDeleteJournalState::Committed,
|
||||
source: None,
|
||||
dispatch: None,
|
||||
})
|
||||
}
|
||||
|
||||
@@ -708,73 +673,17 @@ mod test {
|
||||
use rustfs_s3_client::signer_error::invalid_utf8_header_error;
|
||||
|
||||
use super::{
|
||||
CONFIRMED_TRANSITION_EMPTY_GUARD_DISPATCHES, ERR_REMOTE_DELETE_BREAKER_OPEN, ERR_REMOTE_DELETE_LIMITER_CLOSED, Jentry,
|
||||
RemoteDeleteBreaker, RemoteTierDeleteOutcome, TierDeleteJournalState, TierDeleteSourceIdentity,
|
||||
delete_confirmed_transition_candidate_exact_with_manager_and_identity, delete_object_from_remote_tier_idempotent,
|
||||
delete_object_from_remote_tier_idempotent_with_manager_and_identity, is_remote_tier_not_found_error,
|
||||
is_signer_header_error, lifecycle, set_remote_tier_delete_test_hook, should_record_remote_delete_failure,
|
||||
transitioned_delete_journal_entry, transitioned_force_delete_journal_entry,
|
||||
CONFIRMED_TRANSITION_EMPTY_GUARD_DISPATCHES, ERR_REMOTE_DELETE_BREAKER_OPEN, ERR_REMOTE_DELETE_LIMITER_CLOSED,
|
||||
RemoteDeleteBreaker, RemoteTierDeleteOutcome, delete_confirmed_transition_candidate_exact_with_manager_and_identity,
|
||||
delete_object_from_remote_tier_idempotent, delete_object_from_remote_tier_idempotent_with_manager_and_identity,
|
||||
is_remote_tier_not_found_error, is_signer_header_error, lifecycle, set_remote_tier_delete_test_hook,
|
||||
should_record_remote_delete_failure, transitioned_delete_journal_entry, transitioned_force_delete_journal_entry,
|
||||
};
|
||||
use crate::storage_api_contracts::lifecycle::TransitionedObject;
|
||||
use rustfs_filemeta::TransitionVersionState;
|
||||
use std::io::{Error, ErrorKind};
|
||||
use std::time::{Duration, Instant};
|
||||
|
||||
fn stable_prepared_journal() -> Jentry {
|
||||
Jentry {
|
||||
persisted_version: 0,
|
||||
obj_name: "remote/object".to_string(),
|
||||
version_id: "remote-version".to_string(),
|
||||
tier_name: "WARM".to_string(),
|
||||
backend_identity: Some([7; 32]),
|
||||
version_id_exact: true,
|
||||
version_state: TransitionVersionState::Exact,
|
||||
state: TierDeleteJournalState::Prepared,
|
||||
source: Some(TierDeleteSourceIdentity {
|
||||
bucket: "bucket".to_string(),
|
||||
object: "object".to_string(),
|
||||
version_id: Some(uuid::Uuid::new_v4().to_string()),
|
||||
versioned: true,
|
||||
version_suspended: false,
|
||||
data_dir: None,
|
||||
etag: None,
|
||||
mod_time: None,
|
||||
}),
|
||||
dispatch: None,
|
||||
}
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn only_stable_prepared_journal_can_replace_tier_free_version() {
|
||||
let stable = stable_prepared_journal();
|
||||
assert!(stable.can_replace_tier_free_version());
|
||||
|
||||
let mut committed = stable.clone();
|
||||
committed.state = TierDeleteJournalState::Committed;
|
||||
assert!(!committed.can_replace_tier_free_version());
|
||||
|
||||
let mut unbound = stable.clone();
|
||||
unbound.backend_identity = None;
|
||||
assert!(!unbound.can_replace_tier_free_version());
|
||||
|
||||
let mut unknown = stable.clone();
|
||||
unknown.version_state = TransitionVersionState::Unknown;
|
||||
assert!(!unknown.can_replace_tier_free_version());
|
||||
|
||||
let mut unstable = stable;
|
||||
unstable.source = Some(TierDeleteSourceIdentity {
|
||||
bucket: "bucket".to_string(),
|
||||
object: "object".to_string(),
|
||||
version_id: None,
|
||||
versioned: false,
|
||||
version_suspended: false,
|
||||
data_dir: None,
|
||||
etag: Some("etag-only".to_string()),
|
||||
mod_time: None,
|
||||
});
|
||||
assert!(!unstable.can_replace_tier_free_version());
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn signer_header_error_detection_matches_utf8_failures() {
|
||||
let err = Error::new(
|
||||
|
||||
@@ -412,14 +412,8 @@ pub(crate) fn require_bucket_metadata_sys_in(
|
||||
}
|
||||
|
||||
pub(crate) async fn object_store_in(ctx: &crate::runtime::instance::InstanceContext) -> Result<Arc<ECStore>> {
|
||||
object_store_if_initialized_in(ctx)
|
||||
.await
|
||||
.ok_or_else(|| Error::other("bucket metadata sys not initialized for this instance"))
|
||||
}
|
||||
|
||||
pub(crate) async fn object_store_if_initialized_in(ctx: &crate::runtime::instance::InstanceContext) -> Option<Arc<ECStore>> {
|
||||
let sys = ctx.bucket_metadata_sys().or_else(get_global_bucket_metadata_sys)?;
|
||||
Some(sys.read().await.api.clone())
|
||||
let sys = bucket_metadata_sys_of(ctx)?;
|
||||
Ok(sys.read().await.api.clone())
|
||||
}
|
||||
|
||||
pub(crate) async fn get_in(ctx: &crate::runtime::instance::InstanceContext, bucket: &str) -> Result<Arc<BucketMetadata>> {
|
||||
@@ -1136,16 +1130,6 @@ pub(crate) async fn has_authoritative_never_versioned_state(bucket: &str) -> Res
|
||||
bucket_meta_sys.has_authoritative_never_versioned_state(bucket).await
|
||||
}
|
||||
|
||||
pub(crate) async fn has_authoritative_never_versioned_state_in(
|
||||
ctx: &crate::runtime::instance::InstanceContext,
|
||||
bucket: &str,
|
||||
) -> Result<bool> {
|
||||
let bucket_meta_sys_lock = bucket_metadata_sys_of(ctx)?;
|
||||
let bucket_meta_sys = bucket_meta_sys_lock.read().await.clone();
|
||||
|
||||
bucket_meta_sys.has_authoritative_never_versioned_state(bucket).await
|
||||
}
|
||||
|
||||
pub async fn get_website_config(bucket: &str) -> Result<(WebsiteConfiguration, OffsetDateTime)> {
|
||||
let bucket_meta_sys_lock = get_bucket_metadata_sys()?;
|
||||
let bucket_meta_sys = bucket_meta_sys_lock.read().await;
|
||||
@@ -2528,169 +2512,11 @@ pub(crate) mod test_support {
|
||||
mod tests {
|
||||
use super::test_support::isolated_store_over_temp_disks;
|
||||
use super::*;
|
||||
use crate::bucket::metadata::{
|
||||
BUCKET_ACCELERATE_CONFIG, BUCKET_CORS_CONFIG, BUCKET_LIFECYCLE_CONFIG, BUCKET_LOGGING_CONFIG, BUCKET_NOTIFICATION_CONFIG,
|
||||
BUCKET_POLICY_CONFIG, BUCKET_PUBLIC_ACCESS_BLOCK_CONFIG, BUCKET_REPLICATION_CONFIG, BUCKET_REQUEST_PAYMENT_CONFIG,
|
||||
BUCKET_SSECONFIG, BUCKET_TAGGING_CONFIG, BUCKET_VERSIONING_CONFIG, BUCKET_WEBSITE_CONFIG, OBJECT_LOCK_CONFIG,
|
||||
};
|
||||
use crate::bucket::target::{BucketTarget, BucketTargetType, Credentials};
|
||||
use crate::config::com::read_config;
|
||||
use crate::storage_api_contracts::bucket::{BucketOperations as _, DeleteBucketOptions, MakeBucketOptions};
|
||||
use byteorder::{ByteOrder as _, LittleEndian};
|
||||
use serial_test::serial;
|
||||
use tokio::time::timeout;
|
||||
|
||||
const NEW_WRITER_REPLICATION_XML: &[u8] = br#"<ReplicationConfiguration xmlns="http://s3.amazonaws.com/doc/2006-03-01/"><Role>arn:aws:iam::111122223333:role/replication-role</Role><Rule><ID>rollback</ID><Priority>1</Priority><Filter><Prefix>documents/</Prefix></Filter><Status>Enabled</Status><Destination><Bucket>arn:aws:s3:::replica-bucket</Bucket></Destination><DeleteMarkerReplication><Status>Disabled</Status></DeleteMarkerReplication></Rule></ReplicationConfiguration>"#;
|
||||
|
||||
const NEW_WRITER_CONFIGS: [(&str, &[u8]); 14] = [
|
||||
(BUCKET_POLICY_CONFIG, br#"{"Version":"2012-10-17","Statement":[]}"#),
|
||||
(BUCKET_NOTIFICATION_CONFIG, br#"<NotificationConfiguration/>"#),
|
||||
(
|
||||
BUCKET_LIFECYCLE_CONFIG,
|
||||
br#"<LifecycleConfiguration><Rule><ID>expire</ID><Status>Enabled</Status><Filter><Prefix>logs/</Prefix></Filter><Expiration><Days>30</Days></Expiration></Rule></LifecycleConfiguration>"#,
|
||||
),
|
||||
(
|
||||
OBJECT_LOCK_CONFIG,
|
||||
br#"<ObjectLockConfiguration><ObjectLockEnabled>Enabled</ObjectLockEnabled><Rule><DefaultRetention><Mode>GOVERNANCE</Mode><Days>7</Days></DefaultRetention></Rule></ObjectLockConfiguration>"#,
|
||||
),
|
||||
(
|
||||
BUCKET_VERSIONING_CONFIG,
|
||||
br#"<VersioningConfiguration><Status>Enabled</Status></VersioningConfiguration>"#,
|
||||
),
|
||||
(
|
||||
BUCKET_SSECONFIG,
|
||||
br#"<ServerSideEncryptionConfiguration><Rule><ApplyServerSideEncryptionByDefault><SSEAlgorithm>AES256</SSEAlgorithm></ApplyServerSideEncryptionByDefault></Rule></ServerSideEncryptionConfiguration>"#,
|
||||
),
|
||||
(
|
||||
BUCKET_TAGGING_CONFIG,
|
||||
r#"<Tagging><TagSet><Tag><Key>environment</Key><Value>测试-🦀</Value></Tag></TagSet></Tagging>"#.as_bytes(),
|
||||
),
|
||||
(BUCKET_REPLICATION_CONFIG, NEW_WRITER_REPLICATION_XML),
|
||||
(
|
||||
BUCKET_CORS_CONFIG,
|
||||
br#"<CORSConfiguration><CORSRule><AllowedMethod>GET</AllowedMethod><AllowedOrigin>https://example.test</AllowedOrigin></CORSRule></CORSConfiguration>"#,
|
||||
),
|
||||
(BUCKET_LOGGING_CONFIG, br#"<BucketLoggingStatus/>"#),
|
||||
(
|
||||
BUCKET_WEBSITE_CONFIG,
|
||||
br#"<WebsiteConfiguration><IndexDocument><Suffix>index.html</Suffix></IndexDocument></WebsiteConfiguration>"#,
|
||||
),
|
||||
(
|
||||
BUCKET_ACCELERATE_CONFIG,
|
||||
br#"<AccelerateConfiguration><Status>Enabled</Status></AccelerateConfiguration>"#,
|
||||
),
|
||||
(
|
||||
BUCKET_REQUEST_PAYMENT_CONFIG,
|
||||
br#"<RequestPaymentConfiguration><Payer>Requester</Payer></RequestPaymentConfiguration>"#,
|
||||
),
|
||||
(
|
||||
BUCKET_PUBLIC_ACCESS_BLOCK_CONFIG,
|
||||
br#"<PublicAccessBlockConfiguration><BlockPublicAcls>true</BlockPublicAcls><IgnorePublicAcls>true</IgnorePublicAcls><BlockPublicPolicy>true</BlockPublicPolicy><RestrictPublicBuckets>false</RestrictPublicBuckets></PublicAccessBlockConfiguration>"#,
|
||||
),
|
||||
];
|
||||
|
||||
#[tokio::test]
|
||||
async fn g_d3_003_new_writer_replication_loads_without_fail_closed_state() {
|
||||
let (dirs, store) = isolated_store_over_temp_disks().await;
|
||||
let bucket = "rollback-new-replication";
|
||||
for dir in &dirs {
|
||||
std::fs::create_dir_all(dir.path().join(bucket)).expect("rollback fixture bucket should be created");
|
||||
}
|
||||
|
||||
let writer = BucketMetadataSys::new(store.clone());
|
||||
let mut metadata = BucketMetadata::new(bucket);
|
||||
metadata
|
||||
.update_config(BUCKET_REPLICATION_CONFIG, NEW_WRITER_REPLICATION_XML.to_vec())
|
||||
.expect("new-writer replication XML should be accepted before persistence");
|
||||
writer
|
||||
.persist_new_and_set(metadata)
|
||||
.await
|
||||
.expect("new-writer replication metadata should persist");
|
||||
|
||||
let old_reader = BucketMetadataSys::new(store);
|
||||
let (loaded, _) = old_reader
|
||||
.get_replication_config(bucket)
|
||||
.await
|
||||
.expect("old metadata_sys must not classify new-writer replication XML as invalid");
|
||||
assert_eq!(loaded.role, "arn:aws:iam::111122223333:role/replication-role");
|
||||
assert_eq!(loaded.rules.len(), 1);
|
||||
assert_eq!(loaded.rules[0].id.as_deref(), Some("rollback"));
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
async fn g_d3_004_new_writer_metadata_blob_keeps_legacy_header_and_configs() {
|
||||
let (dirs, store) = isolated_store_over_temp_disks().await;
|
||||
let bucket = "rollback-new-metadata";
|
||||
for dir in &dirs {
|
||||
std::fs::create_dir_all(dir.path().join(bucket)).expect("rollback fixture bucket should be created");
|
||||
}
|
||||
|
||||
let writer = BucketMetadataSys::new(store.clone());
|
||||
let mut metadata = BucketMetadata::new(bucket);
|
||||
for (config_file, bytes) in NEW_WRITER_CONFIGS {
|
||||
metadata
|
||||
.update_config(config_file, bytes.to_vec())
|
||||
.unwrap_or_else(|err| panic!("new-writer {config_file} fixture must be valid: {err}"));
|
||||
}
|
||||
writer
|
||||
.persist_new_and_set(metadata)
|
||||
.await
|
||||
.expect("new-writer metadata should persist");
|
||||
|
||||
let path = BucketMetadata::new(bucket).save_file_path();
|
||||
let blob = read_config(store.clone(), &path)
|
||||
.await
|
||||
.expect("persisted .metadata.bin should be readable");
|
||||
assert_eq!(
|
||||
LittleEndian::read_u16(&blob[0..2]),
|
||||
1,
|
||||
"bucket metadata format must stay rollback-readable"
|
||||
);
|
||||
assert_eq!(
|
||||
LittleEndian::read_u16(&blob[2..4]),
|
||||
1,
|
||||
"bucket metadata version must stay rollback-readable"
|
||||
);
|
||||
|
||||
let loaded = load_bucket_metadata(store, bucket)
|
||||
.await
|
||||
.expect("old read_bucket_metadata path must load the new-writer blob");
|
||||
let loaded_configs: [(&str, &[u8]); 14] = [
|
||||
(BUCKET_POLICY_CONFIG, &loaded.policy_config_json),
|
||||
(BUCKET_NOTIFICATION_CONFIG, &loaded.notification_config_xml),
|
||||
(BUCKET_LIFECYCLE_CONFIG, &loaded.lifecycle_config_xml),
|
||||
(OBJECT_LOCK_CONFIG, &loaded.object_lock_config_xml),
|
||||
(BUCKET_VERSIONING_CONFIG, &loaded.versioning_config_xml),
|
||||
(BUCKET_SSECONFIG, &loaded.encryption_config_xml),
|
||||
(BUCKET_TAGGING_CONFIG, &loaded.tagging_config_xml),
|
||||
(BUCKET_REPLICATION_CONFIG, &loaded.replication_config_xml),
|
||||
(BUCKET_CORS_CONFIG, &loaded.cors_config_xml),
|
||||
(BUCKET_LOGGING_CONFIG, &loaded.logging_config_xml),
|
||||
(BUCKET_WEBSITE_CONFIG, &loaded.website_config_xml),
|
||||
(BUCKET_ACCELERATE_CONFIG, &loaded.accelerate_config_xml),
|
||||
(BUCKET_REQUEST_PAYMENT_CONFIG, &loaded.request_payment_config_xml),
|
||||
(BUCKET_PUBLIC_ACCESS_BLOCK_CONFIG, &loaded.public_access_block_config_xml),
|
||||
];
|
||||
for ((expected_name, expected), (loaded_name, actual)) in NEW_WRITER_CONFIGS.into_iter().zip(loaded_configs) {
|
||||
assert_eq!(loaded_name, expected_name);
|
||||
assert_eq!(actual, expected, "old read_bucket_metadata changed {expected_name} bytes");
|
||||
}
|
||||
assert!(loaded.policy_config.is_some());
|
||||
assert!(loaded.notification_config.is_some());
|
||||
assert!(loaded.lifecycle_config.is_some());
|
||||
assert!(loaded.object_lock_config.is_some());
|
||||
assert!(loaded.versioning_config.is_some());
|
||||
assert!(loaded.sse_config.is_some());
|
||||
assert!(loaded.tagging_config.is_some());
|
||||
assert!(loaded.replication_config.is_some());
|
||||
assert!(loaded.cors_config.is_some());
|
||||
assert!(loaded.logging_config.is_some());
|
||||
assert!(loaded.website_config.is_some());
|
||||
assert!(loaded.accelerate_config.is_some());
|
||||
assert!(loaded.request_payment_config.is_some());
|
||||
assert!(loaded.public_access_block_config.is_some());
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
async fn malformed_delete_configs_are_not_treated_as_absent() {
|
||||
let (_dirs, ecstore) = isolated_store_over_temp_disks().await;
|
||||
|
||||
@@ -177,28 +177,6 @@ pub fn replication_write_may_pass_worm_gate(
|
||||
Ok(!(retention_locked && opts.replication_retention_timestamp.is_none()))
|
||||
}
|
||||
|
||||
/// Whether an authorized replication delete (`ObjectOptions::replication_request`)
|
||||
/// addressed to an explicit version may bypass GOVERNANCE retention on the
|
||||
/// local replica, exactly as an `x-amz-bypass-governance-retention` caller
|
||||
/// with the bypass permission would.
|
||||
///
|
||||
/// The source is authoritative for a replicated version purge (issue #6850):
|
||||
/// the same WORM deletion gate already ran there, and GOVERNANCE retention
|
||||
/// with an authorized bypass is the only lock state it can purge through.
|
||||
/// Requiring the bypass header again here makes the purge permanently
|
||||
/// undeliverable — replication senders never carry it — and the sites diverge
|
||||
/// forever. COMPLIANCE retention and legal hold stay blocking: the source
|
||||
/// gate can never purge through them, so a replication purge that meets one
|
||||
/// here is divergence or forgery and fails closed.
|
||||
///
|
||||
/// The trust judgment is the same one the write-path exemption uses:
|
||||
/// `replication_request` is only set once the receiving handler has
|
||||
/// authorized the caller for the replication action
|
||||
/// (`ReplicateDeleteAction`), never straight from request headers.
|
||||
pub fn replication_delete_may_bypass_governance(opts: &ObjectOptions) -> bool {
|
||||
opts.replication_request && opts.version_id.is_some()
|
||||
}
|
||||
|
||||
/// Check if an object is locked based on its metadata.
|
||||
/// This is a common function used by both lifecycle evaluation and deletion checks.
|
||||
///
|
||||
@@ -702,32 +680,6 @@ mod tests {
|
||||
assert!(err.to_string().contains("modification time"));
|
||||
}
|
||||
|
||||
/// The replicated-purge GOVERNANCE bypass (#6850) applies only to an
|
||||
/// authorized replication delete addressed to an explicit version: a
|
||||
/// local delete never gets it, and a replicated delete without a version
|
||||
/// id creates a delete marker rather than purging anything.
|
||||
#[test]
|
||||
fn replication_delete_bypasses_governance_only_for_authorized_version_purges() {
|
||||
let version_purge = ObjectOptions {
|
||||
replication_request: true,
|
||||
version_id: Some("6b6ffbc0-b0d3-4a86-8f6c-fe19163b8dcd".to_string()),
|
||||
..Default::default()
|
||||
};
|
||||
assert!(replication_delete_may_bypass_governance(&version_purge));
|
||||
|
||||
let local_version_delete = ObjectOptions {
|
||||
replication_request: false,
|
||||
..version_purge.clone()
|
||||
};
|
||||
assert!(!replication_delete_may_bypass_governance(&local_version_delete));
|
||||
|
||||
let replicated_marker_creation = ObjectOptions {
|
||||
version_id: None,
|
||||
..version_purge
|
||||
};
|
||||
assert!(!replication_delete_may_bypass_governance(&replicated_marker_creation));
|
||||
}
|
||||
|
||||
/// A local PutObjectRetention / PutObjectLegalHold "clear" persists the
|
||||
/// lock keys as empty strings (the MinIO on-disk shape, see
|
||||
/// `parse_object_lock_retention`); that is "no lock", not corruption, and
|
||||
|
||||
@@ -196,17 +196,13 @@ const METRIC_VERSION_IDENTITY_DRIFT_TOTAL: &str = "rustfs_replication_version_id
|
||||
/// after a restart is acceptable.
|
||||
static VERSION_IDENTITY_WARNED_ARNS: LazyLock<StdMutex<HashSet<String>>> = LazyLock::new(|| StdMutex::new(HashSet::new()));
|
||||
|
||||
/// Version purges the peer denied under object lock (#6850). A RustFS peer
|
||||
/// with the replicated-purge GOVERNANCE exemption
|
||||
/// (`replication_delete_may_bypass_governance`) no longer produces this for
|
||||
/// governance retention, but COMPLIANCE retention, legal hold, and targets
|
||||
/// without the exemption (older RustFS, MinIO, generic S3) still deny — and
|
||||
/// such a purge cannot succeed until the lock on the replica lapses, so
|
||||
/// retrying every heal cycle only burns bandwidth and failure counters.
|
||||
/// Entries suppress heal requeues for the backoff window; after it expires
|
||||
/// one probe runs again, so the purge still converges on its own once
|
||||
/// retention ends. In-process only: a restart costs at most one extra probe
|
||||
/// per entry.
|
||||
/// Version purges the peer denied under object lock (#6850). Replication
|
||||
/// carries no governance bypass, so such a purge cannot succeed until the
|
||||
/// lock on the replica lapses — retrying every heal cycle only burns
|
||||
/// bandwidth and failure counters. Entries suppress heal requeues for the
|
||||
/// backoff window; after it expires one probe runs again, so the purge still
|
||||
/// converges on its own once retention ends. In-process only: a restart
|
||||
/// costs at most one extra probe per entry.
|
||||
const OBJECT_LOCK_DENIED_PURGE_BACKOFF: std::time::Duration = std::time::Duration::from_secs(60 * 60);
|
||||
const OBJECT_LOCK_DENIED_PURGE_CACHE_MAX: usize = 4096;
|
||||
type ObjectLockDeniedPurgeKey = (String, String, String);
|
||||
@@ -2834,12 +2830,11 @@ async fn replicate_delete_to_target(dobj: &DeletedObjectReplicationInfo, tgt_cli
|
||||
let object_lock_denied = is_version_purge && is_object_lock_denied_delete(e.code.as_deref(), e.message.as_deref());
|
||||
if object_lock_denied {
|
||||
// Terminal for as long as the lock holds: the peer retains
|
||||
// this version under COMPLIANCE retention or legal hold, or
|
||||
// is a target without the replicated-purge GOVERNANCE
|
||||
// exemption (#6850), so the sites stay diverged until the
|
||||
// lock on the replica lapses. Surface it loudly instead of
|
||||
// letting a silent failed counter and a hot heal-retry loop
|
||||
// stand in for the divergence.
|
||||
// this version and replication carries no governance bypass
|
||||
// (#6850), so the sites stay diverged until the retention or
|
||||
// legal hold on the replica lapses. Surface it loudly instead
|
||||
// of letting a silent failed counter and a hot heal-retry
|
||||
// loop stand in for the divergence.
|
||||
record_object_lock_denied_purge(dobj, &tgt_client.arn);
|
||||
error!(
|
||||
event = EVENT_REPLICATION_PURGE_OBJECT_LOCK_DENIED,
|
||||
|
||||
@@ -73,7 +73,6 @@ pub fn check_valid_bucket_name_strict(bucket_name: &str) -> Result<()> {
|
||||
check_bucket_name_common(bucket_name, true)
|
||||
}
|
||||
|
||||
// RUSTFS_COMPAT_TODO(s3gate-metadata-xml): the s3s codec reads persisted XML during migration. Remove after every supported writer uses the gateway codec and every retained metadata object and backup archive is verified or rewritten.
|
||||
pub fn deserialize<T>(input: &[u8]) -> xml::DeResult<T>
|
||||
where
|
||||
T: for<'xml> xml::Deserialize<'xml>,
|
||||
|
||||
@@ -36,7 +36,6 @@ use rustfs_rio::{ChunkReaderBox, HttpChunkReader, HttpReader, HttpWriter};
|
||||
use sha2::{Digest, Sha256};
|
||||
use std::collections::HashMap;
|
||||
use std::future::Future;
|
||||
use std::io;
|
||||
use std::pin::Pin;
|
||||
use std::sync::{Arc, LazyLock, OnceLock};
|
||||
use std::task::{Context, Poll};
|
||||
@@ -106,13 +105,9 @@ struct PutFileCapabilityCacheState {
|
||||
cached: Option<PutFileCapabilityState>,
|
||||
generation: u64,
|
||||
in_flight: Option<PutFileCapabilityFlight>,
|
||||
rejected_server_epoch: Option<Uuid>,
|
||||
}
|
||||
|
||||
// The registry lock is released before taking an entry lock. Entry guards cover
|
||||
// only cache transitions, never a probe or await; poll-based writers must be
|
||||
// able to reject an epoch atomically with those transitions.
|
||||
type PutFileCapabilityCacheEntry = Arc<parking_lot::RwLock<PutFileCapabilityCacheState>>;
|
||||
type PutFileCapabilityCacheEntry = Arc<tokio::sync::RwLock<PutFileCapabilityCacheState>>;
|
||||
|
||||
static PUT_FILE_CAPABILITY_CACHE: LazyLock<parking_lot::RwLock<HashMap<String, PutFileCapabilityCacheEntry>>> =
|
||||
LazyLock::new(|| parking_lot::RwLock::new(HashMap::new()));
|
||||
@@ -124,7 +119,7 @@ fn put_file_capability_cache_entry(endpoint: &str) -> PutFileCapabilityCacheEntr
|
||||
PUT_FILE_CAPABILITY_CACHE
|
||||
.write()
|
||||
.entry(endpoint.to_owned())
|
||||
.or_insert_with(|| Arc::new(parking_lot::RwLock::new(PutFileCapabilityCacheState::default())))
|
||||
.or_insert_with(|| Arc::new(tokio::sync::RwLock::new(PutFileCapabilityCacheState::default())))
|
||||
.clone()
|
||||
}
|
||||
|
||||
@@ -139,23 +134,6 @@ fn fresh_put_file_capability(state: Option<PutFileCapabilityState>, now: Instant
|
||||
}
|
||||
}
|
||||
|
||||
fn reject_put_file_server_epoch(endpoint: &str, server_epoch: Uuid) {
|
||||
let entry = PUT_FILE_CAPABILITY_CACHE.read().get(endpoint).cloned();
|
||||
if let Some(entry) = entry {
|
||||
let mut state = entry.write();
|
||||
if matches!(state.cached, Some(PutFileCapabilityState::V1 { server_epoch: cached, .. }) if cached == server_epoch) {
|
||||
state.rejected_server_epoch = Some(server_epoch);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
fn usable_put_file_capability(state: &PutFileCapabilityCacheState, now: Instant) -> Option<Option<Uuid>> {
|
||||
match fresh_put_file_capability(state.cached, now)? {
|
||||
Some(server_epoch) if state.rejected_server_epoch == Some(server_epoch) => None,
|
||||
capability => Some(capability),
|
||||
}
|
||||
}
|
||||
|
||||
fn put_file_capability_status_is_legacy(status: u16) -> bool {
|
||||
status == 404
|
||||
}
|
||||
@@ -344,14 +322,13 @@ impl InternodeDataTransport for TcpHttpInternodeDataTransport {
|
||||
|
||||
async fn open_write(&self, request: WriteStreamRequest) -> Result<FileWriter> {
|
||||
let server_epoch = self.put_file_auth_capability(&request.endpoint).await?;
|
||||
let auth_scope = server_epoch.map(|server_epoch| (Uuid::new_v4(), server_epoch));
|
||||
let url = build_put_file_stream_url(&request, auth_scope);
|
||||
let endpoint = request.endpoint;
|
||||
let nonce = server_epoch.map(|_| Uuid::new_v4());
|
||||
let url = build_put_file_stream_url(&request, nonce.zip(server_epoch));
|
||||
let mut headers = json_headers();
|
||||
build_auth_headers(&url, &Method::PUT, &mut headers)?;
|
||||
let writer = HttpWriter::new(url.clone(), Method::PUT, headers).await?;
|
||||
match auth_scope {
|
||||
Some((nonce, server_epoch)) => Ok(Box::new(PutFileAuthWriter::new(writer, url, nonce, endpoint, server_epoch))),
|
||||
match nonce {
|
||||
Some(nonce) => Ok(Box::new(PutFileAuthWriter::new(writer, url, nonce))),
|
||||
None => Ok(Box::new(writer)),
|
||||
}
|
||||
}
|
||||
@@ -521,15 +498,15 @@ where
|
||||
{
|
||||
let entry = put_file_capability_cache_entry(endpoint);
|
||||
{
|
||||
let state = entry.read();
|
||||
if let Some(cached) = usable_put_file_capability(&state, Instant::now()) {
|
||||
let state = entry.read().await;
|
||||
if let Some(cached) = fresh_put_file_capability(state.cached, Instant::now()) {
|
||||
return Ok(cached);
|
||||
}
|
||||
}
|
||||
|
||||
let flight = {
|
||||
let mut state = entry.write();
|
||||
if let Some(cached) = usable_put_file_capability(&state, Instant::now()) {
|
||||
let mut state = entry.write().await;
|
||||
if let Some(cached) = fresh_put_file_capability(state.cached, Instant::now()) {
|
||||
return Ok(cached);
|
||||
}
|
||||
if let Some(flight) = state.in_flight.clone() {
|
||||
@@ -555,7 +532,7 @@ where
|
||||
.await;
|
||||
|
||||
{
|
||||
let mut state = entry.write();
|
||||
let mut state = entry.write().await;
|
||||
let is_current_flight = state
|
||||
.in_flight
|
||||
.as_ref()
|
||||
@@ -563,9 +540,6 @@ where
|
||||
if is_current_flight {
|
||||
match outcome {
|
||||
Ok(Some(server_epoch)) => {
|
||||
if state.rejected_server_epoch != Some(*server_epoch) {
|
||||
state.rejected_server_epoch = None;
|
||||
}
|
||||
state.cached = Some(PutFileCapabilityState::V1 {
|
||||
server_epoch: *server_epoch,
|
||||
revalidate_after: Instant::now() + PUT_FILE_V1_CAPABILITY_TTL,
|
||||
@@ -656,23 +630,17 @@ struct PutFileAuthWriter<W> {
|
||||
inner: W,
|
||||
url: String,
|
||||
nonce: Uuid,
|
||||
endpoint: String,
|
||||
server_epoch: Uuid,
|
||||
server_epoch_rejected: bool,
|
||||
hasher: Sha256,
|
||||
trailer: Option<Vec<u8>>,
|
||||
trailer_offset: usize,
|
||||
}
|
||||
|
||||
impl<W> PutFileAuthWriter<W> {
|
||||
fn new(inner: W, url: String, nonce: Uuid, endpoint: String, server_epoch: Uuid) -> Self {
|
||||
fn new(inner: W, url: String, nonce: Uuid) -> Self {
|
||||
Self {
|
||||
inner,
|
||||
url,
|
||||
nonce,
|
||||
endpoint,
|
||||
server_epoch,
|
||||
server_epoch_rejected: false,
|
||||
hasher: Sha256::new(),
|
||||
trailer: None,
|
||||
trailer_offset: 0,
|
||||
@@ -688,14 +656,6 @@ impl<W> PutFileAuthWriter<W> {
|
||||
Ok(())
|
||||
}
|
||||
|
||||
fn reject_server_epoch_on_conflict(&mut self, error: &io::Error) {
|
||||
if self.server_epoch_rejected || !io_error_has_put_file_epoch_conflict(error) {
|
||||
return;
|
||||
}
|
||||
reject_put_file_server_epoch(&self.endpoint, self.server_epoch);
|
||||
self.server_epoch_rejected = true;
|
||||
}
|
||||
|
||||
fn poll_write_trailer(&mut self, cx: &mut Context<'_>) -> Poll<std::io::Result<()>>
|
||||
where
|
||||
W: AsyncWrite + Unpin,
|
||||
@@ -713,10 +673,7 @@ impl<W> PutFileAuthWriter<W> {
|
||||
)));
|
||||
}
|
||||
Poll::Ready(Ok(written)) => written,
|
||||
Poll::Ready(Err(err)) => {
|
||||
self.reject_server_epoch_on_conflict(&err);
|
||||
return Poll::Ready(Err(err));
|
||||
}
|
||||
Poll::Ready(Err(err)) => return Poll::Ready(Err(err)),
|
||||
Poll::Pending => return Poll::Pending,
|
||||
};
|
||||
self.trailer_offset += written;
|
||||
@@ -725,15 +682,6 @@ impl<W> PutFileAuthWriter<W> {
|
||||
}
|
||||
}
|
||||
|
||||
fn io_error_has_put_file_epoch_conflict(error: &io::Error) -> bool {
|
||||
error
|
||||
.get_ref()
|
||||
.and_then(|source| source.downcast_ref::<rustfs_rio::InternodeHttpError>())
|
||||
.is_some_and(
|
||||
|error| matches!(error.kind(), rustfs_rio::InternodeHttpErrorKind::HttpStatus(status) if status.as_u16() == 409),
|
||||
)
|
||||
}
|
||||
|
||||
impl<W> AsyncWrite for PutFileAuthWriter<W>
|
||||
where
|
||||
W: AsyncWrite + Unpin,
|
||||
@@ -750,22 +698,12 @@ where
|
||||
self.hasher.update(&buf[..written]);
|
||||
Poll::Ready(Ok(written))
|
||||
}
|
||||
Poll::Ready(Err(err)) => {
|
||||
self.reject_server_epoch_on_conflict(&err);
|
||||
Poll::Ready(Err(err))
|
||||
}
|
||||
Poll::Pending => Poll::Pending,
|
||||
other => other,
|
||||
}
|
||||
}
|
||||
|
||||
fn poll_flush(mut self: Pin<&mut Self>, cx: &mut Context<'_>) -> Poll<std::io::Result<()>> {
|
||||
match Pin::new(&mut self.inner).poll_flush(cx) {
|
||||
Poll::Ready(Err(err)) => {
|
||||
self.reject_server_epoch_on_conflict(&err);
|
||||
Poll::Ready(Err(err))
|
||||
}
|
||||
other => other,
|
||||
}
|
||||
Pin::new(&mut self.inner).poll_flush(cx)
|
||||
}
|
||||
|
||||
fn poll_shutdown(mut self: Pin<&mut Self>, cx: &mut Context<'_>) -> Poll<std::io::Result<()>> {
|
||||
@@ -774,13 +712,7 @@ where
|
||||
Poll::Ready(Err(err)) => return Poll::Ready(Err(err)),
|
||||
Poll::Pending => return Poll::Pending,
|
||||
}
|
||||
match Pin::new(&mut self.inner).poll_shutdown(cx) {
|
||||
Poll::Ready(Err(err)) => {
|
||||
self.reject_server_epoch_on_conflict(&err);
|
||||
Poll::Ready(Err(err))
|
||||
}
|
||||
other => other,
|
||||
}
|
||||
Pin::new(&mut self.inner).poll_shutdown(cx)
|
||||
}
|
||||
}
|
||||
|
||||
@@ -908,6 +840,7 @@ mod tests {
|
||||
loop {
|
||||
let strong_count = entry
|
||||
.read()
|
||||
.await
|
||||
.in_flight
|
||||
.as_ref()
|
||||
.map(|flight| Arc::strong_count(&flight.outcome))
|
||||
@@ -925,50 +858,6 @@ mod tests {
|
||||
#[derive(Debug)]
|
||||
struct LegacyTestTransport;
|
||||
|
||||
#[derive(Clone, Copy, Debug)]
|
||||
enum PutFileFailurePhase {
|
||||
Write,
|
||||
Flush,
|
||||
Shutdown,
|
||||
}
|
||||
|
||||
struct PutFileFailureWriter {
|
||||
phase: PutFileFailurePhase,
|
||||
status: reqwest::StatusCode,
|
||||
}
|
||||
|
||||
impl PutFileFailureWriter {
|
||||
fn error(&self) -> io::Error {
|
||||
rustfs_rio::new_test_internode_http_io_error(rustfs_rio::InternodeHttpErrorKind::HttpStatus(self.status))
|
||||
}
|
||||
}
|
||||
|
||||
impl tokio::io::AsyncWrite for PutFileFailureWriter {
|
||||
fn poll_write(self: Pin<&mut Self>, _cx: &mut Context<'_>, buf: &[u8]) -> Poll<std::io::Result<usize>> {
|
||||
Poll::Ready(if matches!(self.phase, PutFileFailurePhase::Write) {
|
||||
Err(self.error())
|
||||
} else {
|
||||
Ok(buf.len())
|
||||
})
|
||||
}
|
||||
|
||||
fn poll_flush(self: Pin<&mut Self>, _cx: &mut Context<'_>) -> Poll<std::io::Result<()>> {
|
||||
Poll::Ready(if matches!(self.phase, PutFileFailurePhase::Flush) {
|
||||
Err(self.error())
|
||||
} else {
|
||||
Ok(())
|
||||
})
|
||||
}
|
||||
|
||||
fn poll_shutdown(self: Pin<&mut Self>, _cx: &mut Context<'_>) -> Poll<std::io::Result<()>> {
|
||||
Poll::Ready(if matches!(self.phase, PutFileFailurePhase::Shutdown) {
|
||||
Err(self.error())
|
||||
} else {
|
||||
Ok(())
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
#[async_trait::async_trait]
|
||||
impl InternodeDataTransport for LegacyTestTransport {
|
||||
async fn open_read(&self, _request: ReadStreamRequest) -> Result<FileReader> {
|
||||
@@ -1159,7 +1048,7 @@ mod tests {
|
||||
let v1_endpoint = format!("http://v1-{}.invalid", Uuid::new_v4());
|
||||
let v1_entry = put_file_capability_cache_entry(&v1_endpoint);
|
||||
let server_epoch = Uuid::new_v4();
|
||||
v1_entry.write().cached = Some(PutFileCapabilityState::V1 {
|
||||
v1_entry.write().await.cached = Some(PutFileCapabilityState::V1 {
|
||||
server_epoch,
|
||||
revalidate_after: Instant::now() + PUT_FILE_V1_CAPABILITY_TTL,
|
||||
});
|
||||
@@ -1178,7 +1067,7 @@ mod tests {
|
||||
Some(server_epoch)
|
||||
);
|
||||
assert!(!cache_probe_called.load(Ordering::SeqCst));
|
||||
v1_entry.write().cached = Some(PutFileCapabilityState::V1 {
|
||||
v1_entry.write().await.cached = Some(PutFileCapabilityState::V1 {
|
||||
server_epoch,
|
||||
revalidate_after: Instant::now(),
|
||||
});
|
||||
@@ -1197,7 +1086,8 @@ mod tests {
|
||||
|
||||
let legacy_endpoint = format!("http://legacy-{}.invalid", Uuid::new_v4());
|
||||
let legacy_entry = put_file_capability_cache_entry(&legacy_endpoint);
|
||||
legacy_entry.write().cached = Some(PutFileCapabilityState::LegacyUntil(Instant::now() + PUT_FILE_LEGACY_CAPABILITY_TTL));
|
||||
legacy_entry.write().await.cached =
|
||||
Some(PutFileCapabilityState::LegacyUntil(Instant::now() + PUT_FILE_LEGACY_CAPABILITY_TTL));
|
||||
assert!(
|
||||
transport
|
||||
.put_file_auth_capability(&legacy_endpoint)
|
||||
@@ -1208,7 +1098,7 @@ mod tests {
|
||||
|
||||
let expired_endpoint = format!("http://expired-legacy-{}.invalid", Uuid::new_v4());
|
||||
let expired_entry = put_file_capability_cache_entry(&expired_endpoint);
|
||||
expired_entry.write().cached = Some(PutFileCapabilityState::LegacyUntil(Instant::now()));
|
||||
expired_entry.write().await.cached = Some(PutFileCapabilityState::LegacyUntil(Instant::now()));
|
||||
let reprobed = std::sync::atomic::AtomicBool::new(false);
|
||||
assert_eq!(
|
||||
resolve_put_file_auth_capability(&expired_endpoint, || async {
|
||||
@@ -1459,7 +1349,7 @@ mod tests {
|
||||
};
|
||||
probe_started.notified().await;
|
||||
{
|
||||
let mut state = entry.write();
|
||||
let mut state = entry.write().await;
|
||||
state.generation = state.generation.checked_add(1).expect("test generation should advance");
|
||||
state.cached = Some(PutFileCapabilityState::V1 {
|
||||
server_epoch: newer_epoch,
|
||||
@@ -1472,7 +1362,10 @@ mod tests {
|
||||
task.await.expect("stale task should finish").expect("stale probe result"),
|
||||
Some(stale_epoch)
|
||||
);
|
||||
assert_eq!(fresh_put_file_capability(entry.read().cached, Instant::now()), Some(Some(newer_epoch)));
|
||||
assert_eq!(
|
||||
fresh_put_file_capability(entry.read().await.cached, Instant::now()),
|
||||
Some(Some(newer_epoch))
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
@@ -1505,8 +1398,6 @@ mod tests {
|
||||
|
||||
let _ = rustfs_credentials::set_global_rpc_secret("put-file-auth-writer-test-secret".to_string());
|
||||
let nonce = Uuid::parse_str("11111111-2222-4333-8444-555555555555").expect("nonce");
|
||||
let server_epoch = Uuid::parse_str("aaaaaaaa-bbbb-4ccc-8ddd-eeeeeeeeeeee").expect("server epoch");
|
||||
let endpoint = "http://node1:9000".to_string();
|
||||
let url = concat!(
|
||||
"http://node1:9000/rustfs/rpc/put_file_stream?disk=disk-a&volume=bucket&path=object%2Fpart.1",
|
||||
"&append=false&size=11&put_file_auth=digest-trailer-v1&put_file_nonce=11111111-2222-4333-8444-555555555555"
|
||||
@@ -1515,7 +1406,7 @@ mod tests {
|
||||
let mut sink = Vec::new();
|
||||
|
||||
{
|
||||
let mut writer = PutFileAuthWriter::new(&mut sink, url.clone(), nonce, endpoint, server_epoch);
|
||||
let mut writer = PutFileAuthWriter::new(&mut sink, url.clone(), nonce);
|
||||
writer.write_all(b"hello world").await.expect("body write should succeed");
|
||||
writer.shutdown().await.expect("shutdown should append auth trailer");
|
||||
let err = writer
|
||||
@@ -1533,143 +1424,6 @@ mod tests {
|
||||
assert_eq!(verified, expected_digest);
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
async fn put_file_auth_writer_reprobes_after_server_epoch_conflict() {
|
||||
use tokio::io::AsyncWriteExt;
|
||||
|
||||
let _ = rustfs_credentials::set_global_rpc_secret("put-file-epoch-conflict-test-secret".to_string());
|
||||
for status in [reqwest::StatusCode::CONFLICT, reqwest::StatusCode::BAD_REQUEST] {
|
||||
for (phase, trailer_write) in [
|
||||
(PutFileFailurePhase::Write, false),
|
||||
(PutFileFailurePhase::Write, true),
|
||||
(PutFileFailurePhase::Flush, false),
|
||||
(PutFileFailurePhase::Shutdown, false),
|
||||
] {
|
||||
let endpoint = format!("http://epoch-conflict-{}.invalid", Uuid::new_v4());
|
||||
let stale_epoch = Uuid::new_v4();
|
||||
let replacement_epoch = Uuid::new_v4();
|
||||
resolve_put_file_auth_capability(&endpoint, || async { Ok(Some(stale_epoch)) })
|
||||
.await
|
||||
.expect("initial capability should resolve");
|
||||
let mut writer = PutFileAuthWriter::new(
|
||||
PutFileFailureWriter { phase, status },
|
||||
format!("{endpoint}{PUT_FILE_AUTH_STREAM_PATH}"),
|
||||
Uuid::new_v4(),
|
||||
endpoint.clone(),
|
||||
stale_epoch,
|
||||
);
|
||||
let error = match (phase, trailer_write) {
|
||||
(PutFileFailurePhase::Write, false) => writer.write_all(b"body").await,
|
||||
(PutFileFailurePhase::Flush, _) => writer.flush().await,
|
||||
_ => writer.shutdown().await,
|
||||
}
|
||||
.expect_err("injected writer error must reach the caller");
|
||||
let conflict = status == reqwest::StatusCode::CONFLICT;
|
||||
assert_eq!(io_error_has_put_file_epoch_conflict(&error), conflict);
|
||||
|
||||
let probe_called = AtomicBool::new(false);
|
||||
let resolved = resolve_put_file_auth_capability(&endpoint, || async {
|
||||
probe_called.store(true, Ordering::SeqCst);
|
||||
Ok(Some(replacement_epoch))
|
||||
})
|
||||
.await
|
||||
.expect("capability should remain usable or be reprobed");
|
||||
assert_eq!(probe_called.load(Ordering::SeqCst), conflict, "phase={phase:?}, trailer={trailer_write}");
|
||||
assert_eq!(resolved, Some(if conflict { replacement_epoch } else { stale_epoch }));
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
async fn late_put_file_epoch_rejection_preserves_current_rejection() {
|
||||
let endpoint = format!("http://late-epoch-conflict-{}.invalid", Uuid::new_v4());
|
||||
let old_epoch = Uuid::new_v4();
|
||||
let current_epoch = Uuid::new_v4();
|
||||
let replacement_epoch = Uuid::new_v4();
|
||||
assert_eq!(
|
||||
resolve_put_file_auth_capability(&endpoint, || async { Ok(Some(old_epoch)) })
|
||||
.await
|
||||
.expect("initial epoch should be cached"),
|
||||
Some(old_epoch)
|
||||
);
|
||||
reject_put_file_server_epoch(&endpoint, old_epoch);
|
||||
assert_eq!(
|
||||
resolve_put_file_auth_capability(&endpoint, || async { Ok(Some(current_epoch)) })
|
||||
.await
|
||||
.expect("first restart should install a new epoch"),
|
||||
Some(current_epoch)
|
||||
);
|
||||
|
||||
reject_put_file_server_epoch(&endpoint, current_epoch);
|
||||
// A writer opened before the first restart can report its 409 after
|
||||
// a newer writer has already rejected the second server incarnation.
|
||||
reject_put_file_server_epoch(&endpoint, old_epoch);
|
||||
let probe_called = AtomicBool::new(false);
|
||||
let resolved = resolve_put_file_auth_capability(&endpoint, || async {
|
||||
probe_called.store(true, Ordering::SeqCst);
|
||||
Ok(Some(replacement_epoch))
|
||||
})
|
||||
.await
|
||||
.expect("late old-epoch rejection must preserve the current rejection");
|
||||
|
||||
assert!(probe_called.load(Ordering::SeqCst), "known-rejected current epoch must be reprobed");
|
||||
assert_eq!(resolved, Some(replacement_epoch));
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
async fn put_file_epoch_rejection_is_endpoint_and_epoch_scoped() {
|
||||
let endpoint = format!("http://scoped-epoch-{}.invalid", Uuid::new_v4());
|
||||
let other_endpoint = format!("http://other-epoch-{}.invalid", Uuid::new_v4());
|
||||
let current_epoch = Uuid::new_v4();
|
||||
for endpoint in [&endpoint, &other_endpoint] {
|
||||
resolve_put_file_auth_capability(endpoint, || async { Ok(Some(current_epoch)) })
|
||||
.await
|
||||
.expect("initial epoch should resolve");
|
||||
}
|
||||
reject_put_file_server_epoch(&endpoint, Uuid::new_v4());
|
||||
assert_eq!(
|
||||
resolve_put_file_auth_capability(&endpoint, || async { panic!("old writer must not invalidate a new epoch") })
|
||||
.await
|
||||
.expect("new epoch must remain cached"),
|
||||
Some(current_epoch)
|
||||
);
|
||||
reject_put_file_server_epoch(&endpoint, current_epoch);
|
||||
assert_eq!(
|
||||
resolve_put_file_auth_capability(&other_endpoint, || async { panic!("another endpoint must stay cached") })
|
||||
.await
|
||||
.expect("other endpoint must remain cached"),
|
||||
Some(current_epoch)
|
||||
);
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
async fn put_file_rejected_epoch_survives_failed_stale_and_downgrade_probes() {
|
||||
let endpoint = format!("http://rejected-probe-{}.invalid", Uuid::new_v4());
|
||||
let rejected_epoch = Uuid::new_v4();
|
||||
let replacement_epoch = Uuid::new_v4();
|
||||
resolve_put_file_auth_capability(&endpoint, || async { Ok(Some(rejected_epoch)) })
|
||||
.await
|
||||
.expect("initial epoch should resolve");
|
||||
reject_put_file_server_epoch(&endpoint, rejected_epoch);
|
||||
let failure = resolve_put_file_auth_capability(&endpoint, || async { Err(Error::other("injected probe failure")) })
|
||||
.await
|
||||
.expect_err("probe failure must be returned");
|
||||
assert!(failure.to_string().contains("injected probe failure"));
|
||||
let downgrade = resolve_put_file_auth_capability(&endpoint, || async { Ok(None) })
|
||||
.await
|
||||
.expect_err("rejection must not unpin authenticated v1");
|
||||
assert!(downgrade.to_string().contains("downgrade rejected"));
|
||||
resolve_put_file_auth_capability(&endpoint, || async { Ok(Some(rejected_epoch)) })
|
||||
.await
|
||||
.expect("a probe racing a restart can still return the old epoch");
|
||||
assert_eq!(
|
||||
resolve_put_file_auth_capability(&endpoint, || async { Ok(Some(replacement_epoch)) })
|
||||
.await
|
||||
.expect("same-epoch probe must not clear known rejection"),
|
||||
Some(replacement_epoch)
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn walk_dir_url_encodes_disk_ref() {
|
||||
let url = build_walk_dir_url(&WalkDirStreamRequest {
|
||||
|
||||
@@ -49,8 +49,8 @@ use rustfs_protos::proto_gen::node_service::{
|
||||
ScannerActivityRequest, ScannerActivityResponse, ScannerPublicationLeaseReleaseRequest, ScannerPublicationLeaseRequest,
|
||||
ScannerPublicationLeaseResponse, ServerInfoRequest, SignalServiceRequest, SignalServiceResponse, StartDecommissionRequest,
|
||||
StartProfilingRequest, StopRebalanceRequest, TierMutationAbortRequest, TierMutationCommitRequest,
|
||||
TierMutationControlResponse, TierMutationFailureClass, TierMutationPeerState, TierMutationPrepareRequest,
|
||||
node_service_client::NodeServiceClient, tier_mutation_control_service_client::TierMutationControlServiceClient,
|
||||
TierMutationControlResponse, TierMutationPeerState, TierMutationPrepareRequest, node_service_client::NodeServiceClient,
|
||||
tier_mutation_control_service_client::TierMutationControlServiceClient,
|
||||
};
|
||||
pub use rustfs_protos::{PEER_RESTDRY_RUN, PEER_RESTSIGNAL, PEER_RESTSUB_SYS};
|
||||
use rustfs_protos::{TierMutationRpcPhase, evict_failed_connection};
|
||||
@@ -462,31 +462,6 @@ pub struct PeerTierMutationOutcome {
|
||||
pub applied: bool,
|
||||
}
|
||||
|
||||
#[derive(Debug, thiserror::Error)]
|
||||
#[error("{message}")]
|
||||
struct TierMutationDefinitelyRejected {
|
||||
message: String,
|
||||
}
|
||||
|
||||
fn tier_mutation_definitely_rejected_error(message: String) -> Error {
|
||||
Error::other(TierMutationDefinitelyRejected { message })
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
pub(crate) fn test_tier_mutation_definitely_rejected_error(message: &str) -> Error {
|
||||
tier_mutation_definitely_rejected_error(message.to_string())
|
||||
}
|
||||
|
||||
pub(crate) fn tier_mutation_error_is_definitely_rejected(error: &Error) -> bool {
|
||||
matches!(
|
||||
error,
|
||||
Error::Io(io_error)
|
||||
if io_error
|
||||
.get_ref()
|
||||
.is_some_and(|source| source.downcast_ref::<TierMutationDefinitelyRejected>().is_some())
|
||||
)
|
||||
}
|
||||
|
||||
fn validate_tier_mutation_response_proof(
|
||||
version: u32,
|
||||
phase: TierMutationRpcPhase,
|
||||
@@ -494,16 +469,6 @@ fn validate_tier_mutation_response_proof(
|
||||
canonical_payload: &[u8],
|
||||
response: &TierMutationControlResponse,
|
||||
) -> Result<()> {
|
||||
if response.response_proof.len() > rustfs_protos::TIER_MUTATION_RPC_MAX_RESPONSE_PROOF_SIZE {
|
||||
return Err(Error::other("peer tier mutation response proof exceeds size limit"));
|
||||
}
|
||||
if response
|
||||
.error_info
|
||||
.as_ref()
|
||||
.is_some_and(|error| error.len() > rustfs_protos::TIER_MUTATION_RPC_MAX_ERROR_INFO_SIZE)
|
||||
{
|
||||
return Err(Error::other("peer tier mutation error response exceeds size limit"));
|
||||
}
|
||||
let canonical_response =
|
||||
rustfs_protos::canonical_tier_mutation_rpc_response_body(rustfs_protos::TierMutationRpcResponseProofInput {
|
||||
version,
|
||||
@@ -514,7 +479,6 @@ fn validate_tier_mutation_response_proof(
|
||||
state: response.state,
|
||||
applied: response.applied,
|
||||
error_info: response.error_info.as_deref(),
|
||||
failure_class: response.failure_class,
|
||||
})
|
||||
.map_err(|_| Error::other("tier mutation response length cannot be represented"))?;
|
||||
verify_tonic_rpc_response_proof(&canonical_response, &response.response_proof)
|
||||
@@ -536,9 +500,9 @@ fn validate_tier_mutation_payload_len(phase: TierMutationRpcPhase, payload_len:
|
||||
TierMutationRpcPhase::Commit => rustfs_protos::TIER_MUTATION_RPC_MAX_COMMIT_PAYLOAD_SIZE,
|
||||
TierMutationRpcPhase::Abort => {
|
||||
if payload_len == 0 {
|
||||
return Err(Error::other("tier mutation abort payload is empty"));
|
||||
return Ok(());
|
||||
}
|
||||
rustfs_protos::TIER_MUTATION_RPC_MAX_ABORT_PAYLOAD_SIZE
|
||||
return Err(Error::other("tier mutation abort payload must be empty"));
|
||||
}
|
||||
_ => return Err(Error::other("tier mutation rpc phase is unsupported")),
|
||||
};
|
||||
@@ -557,29 +521,8 @@ fn tier_mutation_phase_label(phase: TierMutationRpcPhase) -> &'static str {
|
||||
}
|
||||
}
|
||||
|
||||
fn tier_mutation_control_status_error(phase: TierMutationRpcPhase, requested_version: u32, status: tonic::Status) -> Error {
|
||||
let message = format!("peer tier mutation {} RPC failed: {status}", tier_mutation_phase_label(phase));
|
||||
let legacy_rejection = format!("unsupported tier mutation peer protocol version: {requested_version}");
|
||||
// RUSTFS_COMPAT_TODO(backlog-2097-tier-mutation-v4-error-text): retain this exact v3-server rejection classifier for mixed-version peers. Remove after every supported peer returns the signed v4 failure class.
|
||||
if requested_version == rustfs_protos::TIER_MUTATION_RPC_PROTOCOL_VERSION
|
||||
&& status.code() == tonic::Code::FailedPrecondition
|
||||
&& status.message().as_bytes() == legacy_rejection.as_bytes()
|
||||
{
|
||||
return tier_mutation_definitely_rejected_error(message);
|
||||
}
|
||||
Error::other(message)
|
||||
}
|
||||
|
||||
fn tier_mutation_failed_response_error(version: u32, failure_class: i32, error_info: Option<String>) -> Error {
|
||||
let message = error_info.unwrap_or_else(|| "peer tier mutation failed without an error".to_string());
|
||||
if version == rustfs_protos::TIER_MUTATION_RPC_PROTOCOL_VERSION
|
||||
&& TierMutationFailureClass::try_from(failure_class).ok() == Some(TierMutationFailureClass::PreDispatchRejected)
|
||||
{
|
||||
return tier_mutation_definitely_rejected_error(message);
|
||||
}
|
||||
// Missing/zero, unknown, and explicit Ambiguous are deliberately the same
|
||||
// fail-closed result: the coordinator must include this peer in Abort.
|
||||
Error::other(message)
|
||||
fn tier_mutation_control_status_error(phase: TierMutationRpcPhase, status: tonic::Status) -> Error {
|
||||
Error::other(format!("peer tier mutation {} RPC failed: {status}", tier_mutation_phase_label(phase)))
|
||||
}
|
||||
|
||||
impl PeerRestClient {
|
||||
@@ -1372,12 +1315,8 @@ impl PeerRestClient {
|
||||
.await
|
||||
}
|
||||
|
||||
pub async fn abort_tier_mutation(
|
||||
&self,
|
||||
mutation_id: Uuid,
|
||||
canonical_prepare_payload: Bytes,
|
||||
) -> Result<PeerTierMutationOutcome> {
|
||||
self.tier_mutation_control(TierMutationRpcPhase::Abort, mutation_id, canonical_prepare_payload)
|
||||
pub async fn abort_tier_mutation(&self, mutation_id: Uuid) -> Result<PeerTierMutationOutcome> {
|
||||
self.tier_mutation_control(TierMutationRpcPhase::Abort, mutation_id, Bytes::new())
|
||||
.await
|
||||
}
|
||||
|
||||
@@ -1411,7 +1350,7 @@ impl PeerRestClient {
|
||||
client
|
||||
.prepare_tier_mutation(request)
|
||||
.await
|
||||
.map_err(|status| tier_mutation_control_status_error(phase, version, status))?
|
||||
.map_err(|status| tier_mutation_control_status_error(phase, status))?
|
||||
.into_inner()
|
||||
}
|
||||
TierMutationRpcPhase::Commit => {
|
||||
@@ -1424,7 +1363,7 @@ impl PeerRestClient {
|
||||
client
|
||||
.commit_tier_mutation(request)
|
||||
.await
|
||||
.map_err(|status| tier_mutation_control_status_error(phase, version, status))?
|
||||
.map_err(|status| tier_mutation_control_status_error(phase, status))?
|
||||
.into_inner()
|
||||
}
|
||||
TierMutationRpcPhase::Abort => {
|
||||
@@ -1437,19 +1376,18 @@ impl PeerRestClient {
|
||||
client
|
||||
.abort_tier_mutation(request)
|
||||
.await
|
||||
.map_err(|status| tier_mutation_control_status_error(phase, version, status))?
|
||||
.map_err(|status| tier_mutation_control_status_error(phase, status))?
|
||||
.into_inner()
|
||||
}
|
||||
_ => return Err(Error::other("tier mutation rpc phase is unsupported")),
|
||||
};
|
||||
validate_tier_mutation_response_proof(version, phase, mutation_id, &canonical_payload, &response)?;
|
||||
if !response.success {
|
||||
return Err(tier_mutation_failed_response_error(version, response.failure_class, response.error_info));
|
||||
}
|
||||
if version == rustfs_protos::TIER_MUTATION_RPC_PROTOCOL_VERSION
|
||||
&& response.failure_class != TierMutationFailureClass::Unspecified as i32
|
||||
{
|
||||
return Err(Error::other("successful peer tier mutation response carried a failure class"));
|
||||
return Err(Error::other(
|
||||
response
|
||||
.error_info
|
||||
.unwrap_or_else(|| "peer tier mutation failed without an error".to_string()),
|
||||
));
|
||||
}
|
||||
let state = decode_tier_mutation_peer_state(response.state)?;
|
||||
Ok(PeerTierMutationOutcome {
|
||||
@@ -3394,7 +3332,6 @@ mod tests {
|
||||
state: i32,
|
||||
applied: bool,
|
||||
error_info: Option<&'a str>,
|
||||
failure_class: i32,
|
||||
}
|
||||
|
||||
fn signed_tier_mutation_response(input: TierMutationResponseFixture<'_>) -> TierMutationControlResponse {
|
||||
@@ -3408,7 +3345,6 @@ mod tests {
|
||||
state: input.state,
|
||||
applied: input.applied,
|
||||
error_info: input.error_info,
|
||||
failure_class: input.failure_class,
|
||||
})
|
||||
.expect("small tier mutation response should encode");
|
||||
let response_proof =
|
||||
@@ -3419,7 +3355,6 @@ mod tests {
|
||||
applied: input.applied,
|
||||
error_info: input.error_info.map(str::to_string),
|
||||
response_proof: response_proof.into(),
|
||||
failure_class: input.failure_class,
|
||||
}
|
||||
}
|
||||
|
||||
@@ -3437,7 +3372,6 @@ mod tests {
|
||||
state: TierMutationPeerState::Prepared as i32,
|
||||
applied: true,
|
||||
error_info: None,
|
||||
failure_class: TierMutationFailureClass::Unspecified as i32,
|
||||
});
|
||||
validate_tier_mutation_response_proof(
|
||||
rustfs_protos::TIER_MUTATION_RPC_PROTOCOL_VERSION,
|
||||
@@ -3462,10 +3396,6 @@ mod tests {
|
||||
applied: false,
|
||||
..response.clone()
|
||||
},
|
||||
TierMutationControlResponse {
|
||||
failure_class: TierMutationFailureClass::Ambiguous as i32,
|
||||
..response.clone()
|
||||
},
|
||||
] {
|
||||
let err = validate_tier_mutation_response_proof(
|
||||
rustfs_protos::TIER_MUTATION_RPC_PROTOCOL_VERSION,
|
||||
@@ -3489,44 +3419,6 @@ mod tests {
|
||||
assert!(err.to_string().contains("invalid tier mutation response proof"));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn tier_mutation_response_rejects_oversized_proof_and_error_before_verification() {
|
||||
let mutation_id = Uuid::new_v4();
|
||||
let payload = b"tier-mutation-prepare";
|
||||
let oversized_proof = TierMutationControlResponse {
|
||||
success: false,
|
||||
state: TierMutationPeerState::Unspecified as i32,
|
||||
applied: false,
|
||||
error_info: None,
|
||||
response_proof: vec![0; rustfs_protos::TIER_MUTATION_RPC_MAX_RESPONSE_PROOF_SIZE + 1].into(),
|
||||
failure_class: TierMutationFailureClass::Ambiguous as i32,
|
||||
};
|
||||
let err = validate_tier_mutation_response_proof(
|
||||
rustfs_protos::TIER_MUTATION_RPC_PROTOCOL_VERSION,
|
||||
TierMutationRpcPhase::Prepare,
|
||||
mutation_id,
|
||||
payload,
|
||||
&oversized_proof,
|
||||
)
|
||||
.expect_err("oversized proof must fail before cryptographic verification");
|
||||
assert!(err.to_string().contains("response proof exceeds size limit"));
|
||||
|
||||
let oversized_error = TierMutationControlResponse {
|
||||
response_proof: Bytes::new(),
|
||||
error_info: Some("e".repeat(rustfs_protos::TIER_MUTATION_RPC_MAX_ERROR_INFO_SIZE + 1)),
|
||||
..oversized_proof
|
||||
};
|
||||
let err = validate_tier_mutation_response_proof(
|
||||
rustfs_protos::TIER_MUTATION_RPC_PROTOCOL_VERSION,
|
||||
TierMutationRpcPhase::Prepare,
|
||||
mutation_id,
|
||||
payload,
|
||||
&oversized_error,
|
||||
)
|
||||
.expect_err("oversized error detail must fail before proof construction");
|
||||
assert!(err.to_string().contains("error response exceeds size limit"));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn tier_mutation_peer_state_decode_fails_closed() {
|
||||
assert_eq!(
|
||||
@@ -3571,17 +3463,8 @@ mod tests {
|
||||
)
|
||||
.is_err()
|
||||
);
|
||||
assert!(validate_tier_mutation_payload_len(TierMutationRpcPhase::Abort, 0).is_err());
|
||||
validate_tier_mutation_payload_len(TierMutationRpcPhase::Abort, 1).expect("non-empty abort payload should fit");
|
||||
validate_tier_mutation_payload_len(TierMutationRpcPhase::Abort, rustfs_protos::TIER_MUTATION_RPC_MAX_ABORT_PAYLOAD_SIZE)
|
||||
.expect("max abort payload should fit");
|
||||
assert!(
|
||||
validate_tier_mutation_payload_len(
|
||||
TierMutationRpcPhase::Abort,
|
||||
rustfs_protos::TIER_MUTATION_RPC_MAX_ABORT_PAYLOAD_SIZE + 1,
|
||||
)
|
||||
.is_err()
|
||||
);
|
||||
validate_tier_mutation_payload_len(TierMutationRpcPhase::Abort, 0).expect("empty abort payload should fit");
|
||||
assert!(validate_tier_mutation_payload_len(TierMutationRpcPhase::Abort, 1).is_err());
|
||||
}
|
||||
|
||||
#[test]
|
||||
@@ -3596,7 +3479,7 @@ mod tests {
|
||||
tonic::Status::deadline_exceeded("peer tier mutation control timed out"),
|
||||
tonic::Status::unavailable("peer tier mutation control unavailable"),
|
||||
] {
|
||||
let err = tier_mutation_control_status_error(phase, rustfs_protos::TIER_MUTATION_RPC_PROTOCOL_VERSION, status);
|
||||
let err = tier_mutation_control_status_error(phase, status);
|
||||
let rendered = err.to_string();
|
||||
assert!(rendered.contains(&format!("peer tier mutation {label} RPC failed")), "{rendered}");
|
||||
assert!(
|
||||
@@ -3610,61 +3493,6 @@ mod tests {
|
||||
}
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn tier_mutation_v4_to_v3_rejection_classification_requires_exact_status_and_message() {
|
||||
let version = rustfs_protos::TIER_MUTATION_RPC_PROTOCOL_VERSION;
|
||||
let exact = format!("unsupported tier mutation peer protocol version: {version}");
|
||||
let rejected = tier_mutation_control_status_error(
|
||||
TierMutationRpcPhase::Prepare,
|
||||
version,
|
||||
tonic::Status::failed_precondition(exact.clone()),
|
||||
);
|
||||
assert!(tier_mutation_error_is_definitely_rejected(&rejected));
|
||||
|
||||
for status in [
|
||||
tonic::Status::failed_precondition(format!("{exact}.")),
|
||||
tonic::Status::failed_precondition(format!("unsupported tier mutation peer protocol version: {}", version - 1)),
|
||||
tonic::Status::invalid_argument(exact.clone()),
|
||||
tonic::Status::unimplemented(exact),
|
||||
] {
|
||||
let ambiguous = tier_mutation_control_status_error(TierMutationRpcPhase::Prepare, version, status);
|
||||
assert!(
|
||||
!tier_mutation_error_is_definitely_rejected(&ambiguous),
|
||||
"near-text, wrong-code, and Unimplemented failures must remain ambiguous"
|
||||
);
|
||||
}
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn tier_mutation_v4_failure_class_is_typed_and_fails_closed() {
|
||||
let version = rustfs_protos::TIER_MUTATION_RPC_PROTOCOL_VERSION;
|
||||
let rejected = tier_mutation_failed_response_error(
|
||||
version,
|
||||
TierMutationFailureClass::PreDispatchRejected as i32,
|
||||
Some("rejected".to_string()),
|
||||
);
|
||||
assert!(tier_mutation_error_is_definitely_rejected(&rejected));
|
||||
|
||||
for failure_class in [
|
||||
TierMutationFailureClass::Unspecified as i32,
|
||||
TierMutationFailureClass::Ambiguous as i32,
|
||||
99,
|
||||
] {
|
||||
let ambiguous = tier_mutation_failed_response_error(version, failure_class, None);
|
||||
assert!(
|
||||
!tier_mutation_error_is_definitely_rejected(&ambiguous),
|
||||
"missing, unknown, and explicit ambiguous classes must trigger Abort fanout"
|
||||
);
|
||||
}
|
||||
|
||||
let v3_ignores_v4_class = tier_mutation_failed_response_error(
|
||||
rustfs_protos::TIER_MUTATION_RPC_PREVIOUS_PROTOCOL_VERSION,
|
||||
TierMutationFailureClass::PreDispatchRejected as i32,
|
||||
Some("legacy failure".to_string()),
|
||||
);
|
||||
assert!(!tier_mutation_error_is_definitely_rejected(&v3_ignores_v4_class));
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
async fn peer_rest_client_rejects_oversized_tier_prepare_before_dialing() {
|
||||
let client = test_peer_client();
|
||||
|
||||
+423
-5418
File diff suppressed because it is too large
Load Diff
File diff suppressed because it is too large
Load Diff
@@ -37,9 +37,8 @@ use crate::{
|
||||
runtime::instance::{InstanceContext, bootstrap_ctx},
|
||||
runtime::sources as runtime_sources,
|
||||
set_disk::{PreparedGetObjectMetadata, SetDisks},
|
||||
store::{
|
||||
RemoteTuplePublicationFence,
|
||||
init_format::{check_format_erasure_values, load_format_erasure_all, save_format_file, select_format_erasure_in_quorum},
|
||||
store::init_format::{
|
||||
check_format_erasure_values, load_format_erasure_all, save_format_file, select_format_erasure_in_quorum,
|
||||
},
|
||||
};
|
||||
use futures::{
|
||||
@@ -626,19 +625,6 @@ impl Sets {
|
||||
.put_object_with_old_current_size(bucket, object, data, opts)
|
||||
.await
|
||||
}
|
||||
|
||||
pub(crate) async fn put_object_with_old_current_size_for_data_movement(
|
||||
&self,
|
||||
bucket: &str,
|
||||
object: &str,
|
||||
data: &mut PutObjReader,
|
||||
opts: &ObjectOptions,
|
||||
publication_fence: RemoteTuplePublicationFence,
|
||||
) -> Result<(ObjectInfo, Option<crate::disk::OldCurrentSize>)> {
|
||||
self.get_disks_by_key(object)
|
||||
.put_object_with_old_current_size_for_data_movement(bucket, object, data, opts, publication_fence)
|
||||
.await
|
||||
}
|
||||
}
|
||||
|
||||
#[async_trait::async_trait]
|
||||
@@ -1365,20 +1351,11 @@ pub(crate) async fn make_local_two_set_sets_with_ctx(ctx: Arc<InstanceContext>)
|
||||
pub(crate) async fn make_local_two_set_sets_for_pool_with_ctx(
|
||||
ctx: Arc<InstanceContext>,
|
||||
pool_idx: usize,
|
||||
) -> (Vec<tempfile::TempDir>, Arc<Sets>) {
|
||||
make_local_two_set_sets_for_pool_with_drive_count_and_ctx(ctx, pool_idx, 2).await
|
||||
}
|
||||
|
||||
#[cfg(any(test, feature = "test-util"))]
|
||||
pub(crate) async fn make_local_two_set_sets_for_pool_with_drive_count_and_ctx(
|
||||
ctx: Arc<InstanceContext>,
|
||||
pool_idx: usize,
|
||||
set_drive_count: usize,
|
||||
) -> (Vec<tempfile::TempDir>, Arc<Sets>) {
|
||||
use crate::layout::endpoint::Endpoint;
|
||||
use rustfs_lock::client::local::LocalClient;
|
||||
|
||||
let format = FormatV3::new(2, set_drive_count);
|
||||
let format = FormatV3::new(2, 2);
|
||||
let mut temp_dirs = Vec::new();
|
||||
let mut all_endpoints = Vec::new();
|
||||
let mut disk_sets = Vec::new();
|
||||
@@ -1386,7 +1363,7 @@ pub(crate) async fn make_local_two_set_sets_for_pool_with_drive_count_and_ctx(
|
||||
for set_index in 0..2 {
|
||||
let mut endpoints = Vec::new();
|
||||
let mut disks = Vec::new();
|
||||
for disk_index in 0..set_drive_count {
|
||||
for disk_index in 0..2 {
|
||||
let temp_dir = tempfile::tempdir().expect("tempdir should be created");
|
||||
let mut endpoint = Endpoint::try_from(temp_dir.path().to_str().expect("tempdir path should be utf8"))
|
||||
.expect("endpoint should parse");
|
||||
@@ -1412,7 +1389,7 @@ pub(crate) async fn make_local_two_set_sets_for_pool_with_drive_count_and_ctx(
|
||||
endpoints.push(endpoint);
|
||||
disks.push(Some(disk));
|
||||
}
|
||||
let lockers = (0..set_drive_count)
|
||||
let lockers = (0..2)
|
||||
.map(|_| {
|
||||
Arc::new(LocalClient::with_manager(Arc::new(rustfs_lock::GlobalLockManager::Enabled(Arc::new(
|
||||
rustfs_lock::FastObjectLockManager::new(),
|
||||
@@ -1423,7 +1400,7 @@ pub(crate) async fn make_local_two_set_sets_for_pool_with_drive_count_and_ctx(
|
||||
SetDisks::new_with_instance_ctx(
|
||||
"test-owner".to_string(),
|
||||
Arc::new(RwLock::new(disks)),
|
||||
set_drive_count,
|
||||
2,
|
||||
1,
|
||||
set_index,
|
||||
pool_idx,
|
||||
@@ -1443,7 +1420,7 @@ pub(crate) async fn make_local_two_set_sets_for_pool_with_drive_count_and_ctx(
|
||||
endpoints: PoolEndpoints {
|
||||
legacy: false,
|
||||
set_count: 2,
|
||||
drives_per_set: set_drive_count,
|
||||
drives_per_set: 2,
|
||||
endpoints: Endpoints::from(all_endpoints),
|
||||
cmd_line: String::new(),
|
||||
platform: String::new(),
|
||||
@@ -1451,7 +1428,7 @@ pub(crate) async fn make_local_two_set_sets_for_pool_with_drive_count_and_ctx(
|
||||
format,
|
||||
parity_count: 1,
|
||||
set_count: 2,
|
||||
set_drive_count,
|
||||
set_drive_count: 2,
|
||||
default_parity_count: 1,
|
||||
distribution_algo: DistributionAlgoVersion::V1,
|
||||
exit_signal: None,
|
||||
|
||||
@@ -16,18 +16,17 @@
|
||||
|
||||
pub(crate) mod backpressure;
|
||||
|
||||
use crate::core::pools::{DecommissionCapacityOwner, decommission_capacity_mutation_id};
|
||||
use crate::error::{
|
||||
Error, Result, is_err_data_movement_overwrite, is_err_invalid_upload_id, is_err_object_not_found, is_err_version_not_found,
|
||||
};
|
||||
use crate::object_api::{GetObjectReader, ObjectInfo, ObjectOptions, PutObjReader};
|
||||
use crate::set_disk::{SetDisks, get_lock_acquire_timeout};
|
||||
use crate::storage_api_contracts::{
|
||||
multipart::CompletePart,
|
||||
multipart::{CompletePart, MultipartOperations as _},
|
||||
namespace::NamespaceLocking as _,
|
||||
object::{HTTPPreconditions, ObjectOperations as _},
|
||||
};
|
||||
use crate::store::{DecommissionFixedReadAnchor, ECStore, SourceCleanupMutationFence};
|
||||
use crate::store::{ECStore, ObjectLockDiagGuard, SourceCleanupMutationFence};
|
||||
use bytes::Bytes;
|
||||
use rustfs_filemeta::{FileInfo, FileInfoVersions, ObjectPartInfo};
|
||||
use rustfs_rio::{EtagResolvable, HashReader, HashReaderDetector, Index, TryGetIndex};
|
||||
@@ -161,99 +160,6 @@ pub fn mark_multipart_upload_completed(flag: &Arc<AtomicBool>) {
|
||||
flag.store(false, Ordering::Relaxed);
|
||||
}
|
||||
|
||||
#[cfg(all(test, feature = "test-util"))]
|
||||
struct DataMovementMultipartAbortBarrierState {
|
||||
bucket: String,
|
||||
object: String,
|
||||
arrived: tokio::sync::Notify,
|
||||
release: tokio::sync::Notify,
|
||||
}
|
||||
|
||||
#[cfg(all(test, feature = "test-util"))]
|
||||
pub(crate) struct DataMovementMultipartAbortBarrier {
|
||||
state: Arc<DataMovementMultipartAbortBarrierState>,
|
||||
}
|
||||
|
||||
#[cfg(all(test, feature = "test-util"))]
|
||||
static DATA_MOVEMENT_MULTIPART_ABORT_BARRIER: std::sync::OnceLock<
|
||||
std::sync::Mutex<Option<Arc<DataMovementMultipartAbortBarrierState>>>,
|
||||
> = std::sync::OnceLock::new();
|
||||
|
||||
#[cfg(all(test, feature = "test-util"))]
|
||||
impl DataMovementMultipartAbortBarrier {
|
||||
pub(crate) fn install(bucket: &str, object: &str) -> Self {
|
||||
let state = Arc::new(DataMovementMultipartAbortBarrierState {
|
||||
bucket: bucket.to_string(),
|
||||
object: object.to_string(),
|
||||
arrived: tokio::sync::Notify::new(),
|
||||
release: tokio::sync::Notify::new(),
|
||||
});
|
||||
let mut slot = DATA_MOVEMENT_MULTIPART_ABORT_BARRIER
|
||||
.get_or_init(|| std::sync::Mutex::new(None))
|
||||
.lock()
|
||||
.expect("data movement multipart abort barrier mutex should not poison");
|
||||
assert!(slot.is_none(), "data movement multipart abort barrier must be unique");
|
||||
*slot = Some(Arc::clone(&state));
|
||||
Self { state }
|
||||
}
|
||||
|
||||
pub(crate) async fn wait_until_paused(&self) {
|
||||
tokio::time::timeout(StdDuration::from_secs(30), self.state.arrived.notified())
|
||||
.await
|
||||
.expect("data movement multipart failure should reach abort cleanup");
|
||||
}
|
||||
}
|
||||
|
||||
#[cfg(all(test, feature = "test-util"))]
|
||||
impl Drop for DataMovementMultipartAbortBarrier {
|
||||
fn drop(&mut self) {
|
||||
self.state.release.notify_one();
|
||||
let mut slot = DATA_MOVEMENT_MULTIPART_ABORT_BARRIER
|
||||
.get_or_init(|| std::sync::Mutex::new(None))
|
||||
.lock()
|
||||
.expect("data movement multipart abort barrier mutex should not poison");
|
||||
if slot.as_ref().is_some_and(|state| Arc::ptr_eq(state, &self.state)) {
|
||||
*slot = None;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
#[cfg(all(test, feature = "test-util"))]
|
||||
async fn pause_data_movement_multipart_before_abort(bucket: &str, object: &str) {
|
||||
let barrier = DATA_MOVEMENT_MULTIPART_ABORT_BARRIER
|
||||
.get_or_init(|| std::sync::Mutex::new(None))
|
||||
.lock()
|
||||
.expect("data movement multipart abort barrier mutex should not poison")
|
||||
.as_ref()
|
||||
.filter(|barrier| barrier.bucket == bucket && barrier.object == object)
|
||||
.cloned();
|
||||
if let Some(barrier) = barrier {
|
||||
barrier.arrived.notify_one();
|
||||
barrier.release.notified().await;
|
||||
}
|
||||
}
|
||||
|
||||
fn data_movement_abort_opts(
|
||||
src_pool_idx: usize,
|
||||
expected_bucket_incarnation_id: Option<uuid::Uuid>,
|
||||
lock_lost_signal: Option<&Arc<rustfs_lock::distributed_lock::LockLostSignal>>,
|
||||
capacity_owner: Option<DecommissionCapacityOwner>,
|
||||
) -> ObjectOptions {
|
||||
let mut opts = ObjectOptions {
|
||||
data_movement: true,
|
||||
src_pool_idx,
|
||||
expected_bucket_incarnation_id,
|
||||
..Default::default()
|
||||
};
|
||||
if let Some(capacity_owner) = capacity_owner {
|
||||
capacity_owner.apply_to(&mut opts);
|
||||
}
|
||||
if let Some(signal) = lock_lost_signal {
|
||||
opts.add_namespace_lock_lost_signal(Arc::clone(signal));
|
||||
}
|
||||
opts
|
||||
}
|
||||
|
||||
fn insert_data_movement_checksum(user_defined: &mut HashMap<String, String>, object_info: &ObjectInfo) {
|
||||
rustfs_utils::http::remove_header_map(user_defined, rustfs_utils::http::SUFFIX_REPLICATION_SSEC_CRC);
|
||||
if let Some(checksum) = object_info.checksum.as_ref().filter(|checksum| !checksum.is_empty()) {
|
||||
@@ -286,7 +192,7 @@ fn data_movement_new_multipart_opts(object_info: &ObjectInfo, src_pool_idx: usiz
|
||||
preserve_etag: object_info.etag.clone(),
|
||||
src_pool_idx,
|
||||
data_movement: true,
|
||||
..ObjectOptions::with_capacity_expected_data_bytes(usize::try_from(object_info.size).ok())
|
||||
..Default::default()
|
||||
}
|
||||
}
|
||||
|
||||
@@ -457,7 +363,7 @@ fn data_movement_complete_multipart_opts(
|
||||
preserve_etag: object_info.etag.clone(),
|
||||
user_defined,
|
||||
src_pool_idx,
|
||||
..ObjectOptions::with_capacity_expected_data_bytes(usize::try_from(object_info.size).ok())
|
||||
..Default::default()
|
||||
})
|
||||
}
|
||||
|
||||
@@ -627,7 +533,6 @@ fn schedule_data_movement_multipart_abort_cleanup(
|
||||
bucket: String,
|
||||
object: String,
|
||||
upload_id: String,
|
||||
opts: ObjectOptions,
|
||||
op_label: &str,
|
||||
) {
|
||||
let op_label = op_label.to_string();
|
||||
@@ -635,32 +540,23 @@ fn schedule_data_movement_multipart_abort_cleanup(
|
||||
for attempt in 1..=DATA_MOVEMENT_MULTIPART_ABORT_RETRY_ATTEMPTS {
|
||||
tokio::time::sleep(StdDuration::from_secs(DATA_MOVEMENT_MULTIPART_ABORT_RETRY_DELAY_SECS)).await;
|
||||
|
||||
if store.pools.get(target_pool_idx).is_none() {
|
||||
let Some(pool) = store.pools.get(target_pool_idx).cloned() else {
|
||||
error!(
|
||||
"{op_label}: background abort_multipart_upload cleanup skipped for {bucket}/{object} upload {upload_id}: target pool {target_pool_idx} is out of range"
|
||||
);
|
||||
return;
|
||||
}
|
||||
|
||||
let mut cleanup_opts = opts.clone();
|
||||
let _multipart_mutation_fence = match DecommissionCapacityOwner::from_options(&cleanup_opts) {
|
||||
Some(owner) => match store.acquire_decommission_multipart_mutation_fence(owner).await {
|
||||
Ok(fence) => {
|
||||
fence.add_namespace_lock_fence(&mut cleanup_opts);
|
||||
Some(fence)
|
||||
}
|
||||
Err(err) => {
|
||||
error!(
|
||||
"{op_label}: background abort_multipart_upload cleanup could not fence {bucket}/{object} upload {upload_id} on attempt {attempt}: {err:?}"
|
||||
);
|
||||
continue;
|
||||
}
|
||||
},
|
||||
None => None,
|
||||
};
|
||||
|
||||
match store
|
||||
.abort_multipart_upload_for_data_movement(target_pool_idx, &bucket, &object, &upload_id, &cleanup_opts)
|
||||
match pool
|
||||
.abort_multipart_upload(
|
||||
&bucket,
|
||||
&object,
|
||||
&upload_id,
|
||||
&ObjectOptions {
|
||||
data_movement: true,
|
||||
..Default::default()
|
||||
},
|
||||
)
|
||||
.await
|
||||
{
|
||||
Ok(()) => {
|
||||
@@ -1438,43 +1334,27 @@ fn resolve_data_movement_overwrite_resume_result_for(
|
||||
Ok(matches!(err, Error::PreconditionFailed) && is_superseding_unversioned_data_movement_object(source, &target))
|
||||
}
|
||||
|
||||
#[derive(Clone, Copy)]
|
||||
struct DataMovementOverwriteCapacity {
|
||||
owner: Option<DecommissionCapacityOwner>,
|
||||
expected_data_bytes: Option<usize>,
|
||||
}
|
||||
|
||||
async fn should_treat_data_movement_overwrite_as_complete(
|
||||
store: &ECStore,
|
||||
pool_indices: (usize, usize),
|
||||
src_pool_idx: usize,
|
||||
target_pool_idx: usize,
|
||||
bucket: &str,
|
||||
object_info: &ObjectInfo,
|
||||
err: &Error,
|
||||
compare_part_checksums: bool,
|
||||
capacity: DataMovementOverwriteCapacity,
|
||||
) -> Result<bool> {
|
||||
if !should_check_data_movement_overwrite_resume(err) {
|
||||
return Ok(false);
|
||||
}
|
||||
let (src_pool_idx, target_pool_idx) = pool_indices;
|
||||
|
||||
let equivalent = resolve_data_movement_overwrite_resume_result_for(
|
||||
resolve_data_movement_overwrite_resume_result_for(
|
||||
err,
|
||||
find_data_movement_target_info(store, target_pool_idx, bucket, object_info).await,
|
||||
object_info,
|
||||
src_pool_idx,
|
||||
target_pool_idx,
|
||||
compare_part_checksums,
|
||||
)?;
|
||||
if equivalent && let Some(owner) = capacity.owner {
|
||||
let expected_data_bytes = capacity
|
||||
.expected_data_bytes
|
||||
.ok_or_else(|| Error::other("equivalent data-movement target cannot reconcile unknown committed data size"))?;
|
||||
store
|
||||
.reconcile_decommission_capacity_after_equivalent_target(owner, target_pool_idx, expected_data_bytes)
|
||||
.await?;
|
||||
}
|
||||
Ok(equivalent)
|
||||
)
|
||||
}
|
||||
|
||||
fn data_movement_part_stage_error(
|
||||
@@ -1515,10 +1395,9 @@ pub(crate) async fn migrate_decommission_object(
|
||||
rd: GetObjectReader,
|
||||
source_bucket_incarnation_id: Option<uuid::Uuid>,
|
||||
op_label: &str,
|
||||
capacity_owner: Option<DecommissionCapacityOwner>,
|
||||
) -> Result<()> {
|
||||
let source = rd.object_info.clone();
|
||||
let mutation_fence = store
|
||||
let _mutation_fence = store
|
||||
.acquire_decommission_object_mutation_fence(&bucket, &source.name)
|
||||
.await?;
|
||||
let current = find_data_movement_target_info(store.as_ref(), pool_idx, &bucket, &source)
|
||||
@@ -1536,8 +1415,7 @@ pub(crate) async fn migrate_decommission_object(
|
||||
source_bucket_incarnation_id,
|
||||
op_label,
|
||||
None,
|
||||
capacity_owner,
|
||||
Some(mutation_fence),
|
||||
Some(&_mutation_fence),
|
||||
)
|
||||
.await
|
||||
}
|
||||
@@ -1573,7 +1451,6 @@ pub(crate) async fn migrate_object_with_lock_lost_signal(
|
||||
op_label,
|
||||
lock_lost_signal,
|
||||
None,
|
||||
None,
|
||||
)
|
||||
.await
|
||||
}
|
||||
@@ -1587,117 +1464,24 @@ async fn migrate_object_inner(
|
||||
source_bucket_incarnation_id: Option<uuid::Uuid>,
|
||||
op_label: &str,
|
||||
lock_lost_signal: Option<Arc<rustfs_lock::distributed_lock::LockLostSignal>>,
|
||||
capacity_owner: Option<DecommissionCapacityOwner>,
|
||||
mutation_fence: Option<DecommissionFixedReadAnchor>,
|
||||
mutation_fence: Option<&ObjectLockDiagGuard>,
|
||||
) -> Result<()> {
|
||||
let mut mutation_fence = mutation_fence;
|
||||
let object_info = rd.object_info.clone();
|
||||
let capacity_owner = capacity_owner.map(|owner| {
|
||||
let version_id = object_info.version_id.map(|version_id| version_id.to_string());
|
||||
let mutation_id = owner.mutation_id.unwrap_or_else(|| {
|
||||
decommission_capacity_mutation_id(
|
||||
owner,
|
||||
&bucket,
|
||||
&object_info.name,
|
||||
version_id.as_deref(),
|
||||
object_info.delete_marker,
|
||||
object_info.mod_time,
|
||||
)
|
||||
});
|
||||
owner.with_mutation_id(mutation_id)
|
||||
});
|
||||
// Capture the exact source/tier identity before any client-paced read, but
|
||||
// defer both the tier lease and source/target write locks to the final
|
||||
// publication. Decommission already owns main's fixed-domain mutation
|
||||
// fence, so reacquiring that domain as a write lock would self-deadlock.
|
||||
let remote_tuple_publication_fence = store
|
||||
.acquire_remote_tuple_publication_fence(&bucket, pool_idx, &object_info, false)
|
||||
.await?;
|
||||
let has_part_checksums = object_info
|
||||
.parts
|
||||
.iter()
|
||||
.any(|part| part.checksums.as_ref().is_some_and(|checksums| !checksums.is_empty()));
|
||||
|
||||
let preserve_part_checksums = data_movement_part_checksum_writer_enabled();
|
||||
let capacity_expected_data_bytes = usize::try_from(object_info.size).ok();
|
||||
|
||||
if should_use_multipart_data_movement(&object_info, has_part_checksums) {
|
||||
// The decommission object fence already covers the source/target
|
||||
// namespace for this migration. Acquiring the synthetic multipart
|
||||
// fence while holding that read lock deadlocks local lock domains;
|
||||
// retain the extra fence only for callers without the outer fence.
|
||||
let multipart_mutation_fence = match (capacity_owner, mutation_fence.is_some()) {
|
||||
(Some(owner), false) => Some(store.acquire_decommission_multipart_mutation_fence(owner).await?),
|
||||
_ => None,
|
||||
};
|
||||
let mut new_multipart_opts = data_movement_new_multipart_opts(&object_info, pool_idx);
|
||||
if let Some(capacity_owner) = capacity_owner {
|
||||
capacity_owner.apply_to(&mut new_multipart_opts);
|
||||
}
|
||||
new_multipart_opts.expected_bucket_incarnation_id = source_bucket_incarnation_id;
|
||||
if let Some(signal) = lock_lost_signal.as_ref() {
|
||||
new_multipart_opts.add_namespace_lock_lost_signal(Arc::clone(signal));
|
||||
}
|
||||
if let Some(fence) = multipart_mutation_fence.as_ref() {
|
||||
fence.add_namespace_lock_fence(&mut new_multipart_opts);
|
||||
}
|
||||
if let Some(owner) = capacity_owner {
|
||||
let existing_target_pool_idx = store
|
||||
.select_data_movement_pool_idx(&bucket, &object_info.name, -1, &new_multipart_opts, false)
|
||||
.await?;
|
||||
if existing_target_pool_idx != pool_idx
|
||||
&& let Some(target) =
|
||||
find_data_movement_target_info(store.as_ref(), existing_target_pool_idx, &bucket, &object_info).await?
|
||||
&& is_equivalent_data_movement_object_identity(&object_info, &target, true, preserve_part_checksums)
|
||||
{
|
||||
let expected_data_bytes = capacity_expected_data_bytes
|
||||
.ok_or_else(|| Error::other("equivalent multipart target cannot reconcile unknown committed data size"))?;
|
||||
store
|
||||
.reconcile_decommission_capacity_after_equivalent_target(owner, existing_target_pool_idx, expected_data_bytes)
|
||||
.await?;
|
||||
info!(
|
||||
"{op_label}: multipart upload restart reconciled equivalent target for {}/{}",
|
||||
bucket.as_str(),
|
||||
object_info.name.as_str()
|
||||
);
|
||||
return Ok(());
|
||||
}
|
||||
let mut cleanup_opts =
|
||||
data_movement_abort_opts(pool_idx, source_bucket_incarnation_id, lock_lost_signal.as_ref(), capacity_owner);
|
||||
if let Some(anchor) = mutation_fence.as_ref() {
|
||||
anchor.guard().add_namespace_lock_fence(&mut cleanup_opts);
|
||||
}
|
||||
if let Some(fence) = multipart_mutation_fence.as_ref() {
|
||||
fence.add_namespace_lock_fence(&mut cleanup_opts);
|
||||
}
|
||||
for target_pool_idx in store.decommission_capacity_cleanup_target_indices(owner).await? {
|
||||
store
|
||||
.reconcile_multipart_uploads_for_data_movement(
|
||||
target_pool_idx,
|
||||
&bucket,
|
||||
&object_info.name,
|
||||
&data_movement_upload_identity(&object_info),
|
||||
&cleanup_opts,
|
||||
)
|
||||
.await
|
||||
.map_err(|err| {
|
||||
data_movement_stage_error(
|
||||
op_label,
|
||||
"reconcile_multipart_upload",
|
||||
bucket.as_str(),
|
||||
object_info.name.as_str(),
|
||||
err,
|
||||
)
|
||||
})?;
|
||||
}
|
||||
}
|
||||
let (res, target_pool_idx, expected_bucket_incarnation_id) = match store
|
||||
.handle_new_multipart_upload_with_pool_idx(
|
||||
&bucket,
|
||||
&object_info.name,
|
||||
&new_multipart_opts,
|
||||
mutation_fence.as_ref().map(DecommissionFixedReadAnchor::guard),
|
||||
)
|
||||
.handle_new_multipart_upload_with_pool_idx(&bucket, &object_info.name, &new_multipart_opts, mutation_fence)
|
||||
.await
|
||||
{
|
||||
Ok(res) => res,
|
||||
@@ -1748,15 +1532,9 @@ async fn migrate_object_inner(
|
||||
expected_bucket_incarnation_id,
|
||||
..Default::default()
|
||||
};
|
||||
if let Some(capacity_owner) = capacity_owner {
|
||||
capacity_owner.apply_to(&mut part_opts);
|
||||
}
|
||||
if let Some(signal) = lock_lost_signal.as_ref() {
|
||||
part_opts.add_namespace_lock_lost_signal(Arc::clone(signal));
|
||||
}
|
||||
if let Some(fence) = multipart_mutation_fence.as_ref() {
|
||||
fence.add_namespace_lock_fence(&mut part_opts);
|
||||
}
|
||||
let pi = match store
|
||||
.put_object_part_for_data_movement(
|
||||
target_pool_idx,
|
||||
@@ -1800,44 +1578,30 @@ async fn migrate_object_inner(
|
||||
err,
|
||||
)
|
||||
})?;
|
||||
if let Some(capacity_owner) = capacity_owner {
|
||||
capacity_owner.apply_to(&mut complete_multipart_opts);
|
||||
}
|
||||
complete_multipart_opts.expected_bucket_incarnation_id = expected_bucket_incarnation_id;
|
||||
if let Some(signal) = lock_lost_signal.as_ref() {
|
||||
complete_multipart_opts.add_namespace_lock_lost_signal(Arc::clone(signal));
|
||||
}
|
||||
if let Some(fence) = multipart_mutation_fence.as_ref() {
|
||||
fence.add_namespace_lock_fence(&mut complete_multipart_opts);
|
||||
}
|
||||
let remote_tuple_publication_fence = match mutation_fence.take() {
|
||||
Some(anchor) => remote_tuple_publication_fence.under_fixed_read_anchor(anchor)?,
|
||||
None => remote_tuple_publication_fence,
|
||||
};
|
||||
if let Err(err) = store
|
||||
.clone()
|
||||
.complete_multipart_upload_for_data_movement_with_publication_fence(
|
||||
target_pool_idx,
|
||||
.complete_multipart_upload_for_data_movement(
|
||||
(target_pool_idx, mutation_fence),
|
||||
&bucket,
|
||||
&object_info.name,
|
||||
&res.upload_id,
|
||||
parts,
|
||||
&complete_multipart_opts,
|
||||
remote_tuple_publication_fence,
|
||||
)
|
||||
.await
|
||||
{
|
||||
if should_treat_data_movement_overwrite_as_complete(
|
||||
store.as_ref(),
|
||||
(pool_idx, target_pool_idx),
|
||||
pool_idx,
|
||||
target_pool_idx,
|
||||
bucket.as_str(),
|
||||
&object_info,
|
||||
&err,
|
||||
preserve_part_checksums,
|
||||
DataMovementOverwriteCapacity {
|
||||
owner: capacity_owner,
|
||||
expected_data_bytes: capacity_expected_data_bytes,
|
||||
},
|
||||
)
|
||||
.await?
|
||||
{
|
||||
@@ -1865,37 +1629,31 @@ async fn migrate_object_inner(
|
||||
.await;
|
||||
|
||||
if multipart_result.is_ok() && should_abort_multipart_upload(&abort_multipart_flag) {
|
||||
let mut abort_opts =
|
||||
data_movement_abort_opts(pool_idx, expected_bucket_incarnation_id, lock_lost_signal.as_ref(), capacity_owner);
|
||||
if let Some(anchor) = mutation_fence.as_ref() {
|
||||
anchor.guard().add_namespace_lock_fence(&mut abort_opts);
|
||||
}
|
||||
if let Some(fence) = multipart_mutation_fence.as_ref() {
|
||||
fence.add_namespace_lock_fence(&mut abort_opts);
|
||||
}
|
||||
let abort_result = store
|
||||
.abort_multipart_upload_for_data_movement(
|
||||
target_pool_idx,
|
||||
&bucket,
|
||||
&object_info.name,
|
||||
&res.upload_id,
|
||||
&abort_opts,
|
||||
)
|
||||
.abort_multipart_upload_for_data_movement(target_pool_idx, &bucket, &object_info.name, &res.upload_id, &{
|
||||
let mut opts = ObjectOptions {
|
||||
data_movement: true,
|
||||
src_pool_idx: pool_idx,
|
||||
expected_bucket_incarnation_id,
|
||||
..Default::default()
|
||||
};
|
||||
if let Some(signal) = lock_lost_signal.as_ref() {
|
||||
opts.add_namespace_lock_lost_signal(Arc::clone(signal));
|
||||
}
|
||||
opts
|
||||
})
|
||||
.await;
|
||||
match abort_result {
|
||||
Ok(()) => return Ok(()),
|
||||
Err(abort_err) if is_err_invalid_upload_id(&abort_err) => {
|
||||
if should_treat_data_movement_overwrite_as_complete(
|
||||
store.as_ref(),
|
||||
(pool_idx, target_pool_idx),
|
||||
pool_idx,
|
||||
target_pool_idx,
|
||||
bucket.as_str(),
|
||||
&object_info,
|
||||
&abort_err,
|
||||
preserve_part_checksums,
|
||||
DataMovementOverwriteCapacity {
|
||||
owner: capacity_owner,
|
||||
expected_data_bytes: capacity_expected_data_bytes,
|
||||
},
|
||||
)
|
||||
.await?
|
||||
{
|
||||
@@ -1925,7 +1683,6 @@ async fn migrate_object_inner(
|
||||
bucket.clone(),
|
||||
object_info.name.clone(),
|
||||
res.upload_id.clone(),
|
||||
abort_opts,
|
||||
op_label,
|
||||
);
|
||||
return Err(data_movement_stage_error(
|
||||
@@ -1941,24 +1698,19 @@ async fn migrate_object_inner(
|
||||
|
||||
if let Err(primary_err) = multipart_result {
|
||||
if should_abort_multipart_upload(&abort_multipart_flag) {
|
||||
#[cfg(all(test, feature = "test-util"))]
|
||||
pause_data_movement_multipart_before_abort(&bucket, &object_info.name).await;
|
||||
let mut abort_opts =
|
||||
data_movement_abort_opts(pool_idx, expected_bucket_incarnation_id, lock_lost_signal.as_ref(), capacity_owner);
|
||||
if let Some(anchor) = mutation_fence.as_ref() {
|
||||
anchor.guard().add_namespace_lock_fence(&mut abort_opts);
|
||||
}
|
||||
if let Some(fence) = multipart_mutation_fence.as_ref() {
|
||||
fence.add_namespace_lock_fence(&mut abort_opts);
|
||||
}
|
||||
return match store
|
||||
.abort_multipart_upload_for_data_movement(
|
||||
target_pool_idx,
|
||||
&bucket,
|
||||
&object_info.name,
|
||||
&res.upload_id,
|
||||
&abort_opts,
|
||||
)
|
||||
.abort_multipart_upload_for_data_movement(target_pool_idx, &bucket, &object_info.name, &res.upload_id, &{
|
||||
let mut opts = ObjectOptions {
|
||||
data_movement: true,
|
||||
src_pool_idx: pool_idx,
|
||||
expected_bucket_incarnation_id,
|
||||
..Default::default()
|
||||
};
|
||||
if let Some(signal) = lock_lost_signal.as_ref() {
|
||||
opts.add_namespace_lock_lost_signal(Arc::clone(signal));
|
||||
}
|
||||
opts
|
||||
})
|
||||
.await
|
||||
{
|
||||
Ok(()) => Err(primary_err),
|
||||
@@ -1970,7 +1722,6 @@ async fn migrate_object_inner(
|
||||
bucket.clone(),
|
||||
object_info.name.clone(),
|
||||
res.upload_id.clone(),
|
||||
abort_opts,
|
||||
op_label,
|
||||
);
|
||||
Err(resolve_data_movement_abort_result(
|
||||
@@ -1993,39 +1744,23 @@ async fn migrate_object_inner(
|
||||
let mut data = data_movement_put_object_reader(bucket.as_str(), &object_info, rd, op_label)?;
|
||||
|
||||
let mut put_opts = data_movement_put_object_opts(&object_info, pool_idx);
|
||||
if let Some(capacity_owner) = capacity_owner {
|
||||
capacity_owner.apply_to(&mut put_opts);
|
||||
}
|
||||
put_opts.expected_bucket_incarnation_id = source_bucket_incarnation_id;
|
||||
if let Some(signal) = lock_lost_signal {
|
||||
put_opts.add_namespace_lock_lost_signal(signal);
|
||||
}
|
||||
let remote_tuple_publication_fence = match mutation_fence.take() {
|
||||
Some(anchor) => remote_tuple_publication_fence.under_fixed_read_anchor(anchor)?,
|
||||
None => remote_tuple_publication_fence,
|
||||
};
|
||||
let (target_pool_idx, put_result) = store
|
||||
.put_object_for_data_movement_with_publication_fence(
|
||||
&bucket,
|
||||
&object_info.name,
|
||||
&mut data,
|
||||
&put_opts,
|
||||
remote_tuple_publication_fence,
|
||||
)
|
||||
.put_object_for_data_movement(&bucket, &object_info.name, &mut data, &put_opts, mutation_fence)
|
||||
.await
|
||||
.map_err(|err| data_movement_stage_error(op_label, "prepare_put_object", &bucket, &object_info.name, err))?;
|
||||
if let Err(err) = put_result {
|
||||
if should_treat_data_movement_overwrite_as_complete(
|
||||
store.as_ref(),
|
||||
(pool_idx, target_pool_idx),
|
||||
pool_idx,
|
||||
target_pool_idx,
|
||||
bucket.as_str(),
|
||||
&object_info,
|
||||
&err,
|
||||
preserve_part_checksums,
|
||||
DataMovementOverwriteCapacity {
|
||||
owner: capacity_owner,
|
||||
expected_data_bytes: capacity_expected_data_bytes,
|
||||
},
|
||||
)
|
||||
.await?
|
||||
{
|
||||
|
||||
@@ -2964,7 +2964,6 @@ mod tests {
|
||||
decommission_cancelers: RwLock::new(Vec::new()),
|
||||
start_gate: TokioMutex::new(()),
|
||||
pool_meta_save_gate: TokioMutex::default(),
|
||||
decommission_capacity_entry_gate: TokioMutex::default(),
|
||||
ctx,
|
||||
bucket_fence_registry: Arc::default(),
|
||||
})
|
||||
|
||||
@@ -2435,25 +2435,6 @@ mod tests {
|
||||
assert_eq!(window.acc_time, 18_000);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn timed_action_slot_snapshot_skips_writer_owned_slot() {
|
||||
let slot = TimedActionSlot::default();
|
||||
slot.unix_sec.store(70, Ordering::Relaxed);
|
||||
slot.count.store(2, Ordering::Relaxed);
|
||||
slot.acc_time.store(18_000, Ordering::Relaxed);
|
||||
slot.version.store(2, Ordering::Release);
|
||||
assert_eq!(slot.snapshot(), Some((70, 2, 18_000)));
|
||||
|
||||
assert_eq!(slot.version.compare_exchange(2, 3, Ordering::AcqRel, Ordering::Relaxed), Ok(2));
|
||||
slot.unix_sec.store(71, Ordering::Relaxed);
|
||||
slot.count.store(1, Ordering::Relaxed);
|
||||
slot.acc_time.store(11_000, Ordering::Relaxed);
|
||||
assert_eq!(slot.snapshot(), None);
|
||||
|
||||
slot.version.store(4, Ordering::Release);
|
||||
assert_eq!(slot.snapshot(), Some((71, 1, 11_000)));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn disk_health_metrics_snapshot_exports_waiting_errors_and_operation_windows() {
|
||||
let metrics = DiskHealthMetricEpoch::default();
|
||||
|
||||
@@ -12,7 +12,6 @@
|
||||
// See the License for the specific language governing permissions and
|
||||
// limitations under the License.
|
||||
|
||||
use rustfs_io_metrics::internode_metrics::INTERNODE_OPERATION_PUT_FILE_STREAM;
|
||||
use rustfs_rio::{InternodeHttpError, InternodeHttpErrorKind};
|
||||
use std::error::Error as StdError;
|
||||
use std::hash::{Hash, Hasher};
|
||||
@@ -230,19 +229,6 @@ fn classify_internode_missing_error(error: &InternodeHttpError) -> Option<DiskEr
|
||||
None
|
||||
}
|
||||
|
||||
fn internode_write_error_is_retryable(error: &InternodeHttpError) -> bool {
|
||||
error.kind().is_retryable()
|
||||
|| (matches!(error.kind(), InternodeHttpErrorKind::HttpStatus(status) if status.as_u16() == 409)
|
||||
&& error.context().operation() == Some(INTERNODE_OPERATION_PUT_FILE_STREAM))
|
||||
}
|
||||
|
||||
fn io_error_contains_retryable_internode_write(error: &io::Error) -> bool {
|
||||
error
|
||||
.get_ref()
|
||||
.and_then(|source| source.downcast_ref::<InternodeHttpError>())
|
||||
.is_some_and(internode_write_error_is_retryable)
|
||||
}
|
||||
|
||||
/// Wrap a terminal shard-read failure without changing its typed
|
||||
/// classification. Timeout-like disk errors retain `TimedOut`; other errors
|
||||
/// retain their inner I/O kind or use `Other` when no more specific kind exists.
|
||||
@@ -350,7 +336,10 @@ impl DiskError {
|
||||
|
||||
pub fn is_retryable_internode_write_failure(&self) -> bool {
|
||||
match self {
|
||||
DiskError::Io(io_error) => io_error_contains_retryable_internode_write(io_error),
|
||||
DiskError::Io(io_error) => io_error
|
||||
.get_ref()
|
||||
.and_then(|source| source.downcast_ref::<InternodeHttpError>())
|
||||
.is_some_and(|err| err.kind().is_retryable()),
|
||||
_ => false,
|
||||
}
|
||||
}
|
||||
@@ -1251,68 +1240,6 @@ mod tests {
|
||||
assert!(!DiskError::FileNotFound.is_internode_http_status(429));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_put_file_server_epoch_conflict_is_retryable_write_failure() {
|
||||
let conflict = DiskError::from(rustfs_rio::new_test_internode_http_io_error(
|
||||
rustfs_rio::InternodeHttpErrorKind::HttpStatus(http::StatusCode::CONFLICT),
|
||||
));
|
||||
let bad_request = DiskError::from(rustfs_rio::new_test_internode_http_io_error(
|
||||
rustfs_rio::InternodeHttpErrorKind::HttpStatus(http::StatusCode::BAD_REQUEST),
|
||||
));
|
||||
|
||||
assert!(conflict.is_retryable_internode_write_failure());
|
||||
assert!(!bad_request.is_retryable_internode_write_failure());
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
async fn read_stream_conflict_is_not_a_retryable_put_file_failure() {
|
||||
use tokio::io::{AsyncReadExt, AsyncWriteExt};
|
||||
|
||||
tokio::time::timeout(std::time::Duration::from_secs(5), async {
|
||||
let listener = tokio::net::TcpListener::bind("127.0.0.1:0")
|
||||
.await
|
||||
.expect("bind isolated HTTP fixture");
|
||||
let address = listener.local_addr().expect("fixture address");
|
||||
let server = tokio::spawn(async move {
|
||||
let (mut stream, _) = listener.accept().await.expect("accept read request");
|
||||
let mut request = [0_u8; 4096];
|
||||
let mut read = 0;
|
||||
loop {
|
||||
let count = stream.read(&mut request[read..]).await.expect("read HTTP request");
|
||||
assert!(count > 0, "request ended before its complete headers");
|
||||
read += count;
|
||||
if request[..read].windows(4).any(|bytes| bytes == b"\r\n\r\n") {
|
||||
break;
|
||||
}
|
||||
assert!(read < request.len(), "fixture request headers exceed their budget");
|
||||
}
|
||||
stream
|
||||
.write_all(b"HTTP/1.1 409 Conflict\r\nContent-Length: 0\r\nConnection: close\r\n\r\n")
|
||||
.await
|
||||
.expect("send typed conflict response");
|
||||
});
|
||||
let error = match rustfs_rio::HttpReader::new(
|
||||
format!("http://{address}/rustfs/rpc/read_file_stream"),
|
||||
http::Method::GET,
|
||||
http::HeaderMap::new(),
|
||||
None,
|
||||
)
|
||||
.await
|
||||
{
|
||||
Ok(_) => panic!("HTTP 409 must fail the read"),
|
||||
Err(error) => DiskError::from(error),
|
||||
};
|
||||
server.await.expect("fixture task should complete");
|
||||
assert!(error.is_internode_http_status(409));
|
||||
assert!(
|
||||
!error.is_retryable_internode_write_failure(),
|
||||
"read-operation 409 must not trigger put-file retry"
|
||||
);
|
||||
})
|
||||
.await
|
||||
.expect("isolated read-conflict test must finish within its budget");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_internode_missing_errors_preserve_disk_error_types() {
|
||||
let file_missing = DiskError::from(rustfs_rio::new_test_remote_file_not_found_http_io_error());
|
||||
|
||||
@@ -7092,9 +7092,6 @@ impl LocalDisk {
|
||||
.await?
|
||||
{
|
||||
meta.name.push_str(SLASH_SEPARATOR);
|
||||
// Conservative listings verify physical prefixes. Never-versioned
|
||||
// buckets use the bounded fast path and reclaim residue after an
|
||||
// exact recursive listing proves that prefix empty.
|
||||
if opts.recursive
|
||||
|| opts.incl_deleted
|
||||
|| opts.skip_hidden_prefix_check
|
||||
@@ -17622,7 +17619,7 @@ mod test {
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
async fn test_scan_dir_nonrecursive_fast_path_preserves_probe_bound() {
|
||||
async fn test_scan_dir_nonrecursive_visible_prefix_probe_cost() {
|
||||
use rustfs_filemeta::MetacacheReader;
|
||||
use tempfile::tempdir;
|
||||
|
||||
@@ -17654,10 +17651,6 @@ mod test {
|
||||
expected_names.push(format!("{prefix}/"));
|
||||
}
|
||||
|
||||
fs::create_dir_all(bucket_dir.join("stale/nested/residue"))
|
||||
.await
|
||||
.expect("stale backing directory should be created");
|
||||
|
||||
async fn scan_prefixes(disk: &LocalDisk, bucket: &str, skip_hidden_prefix_check: bool) -> (Vec<String>, usize) {
|
||||
let probe_count = Arc::new(AtomicUsize::new(0));
|
||||
let (reader, mut writer) = tokio::io::duplex(64 * 1024);
|
||||
@@ -17700,11 +17693,8 @@ mod test {
|
||||
let (fast_path_names, fast_path_probes) = scan_prefixes(&disk, bucket, true).await;
|
||||
|
||||
assert_eq!(conservative_names, expected_names);
|
||||
let mut expected_fast_path_names = expected_names.clone();
|
||||
expected_fast_path_names.push("stale/".to_owned());
|
||||
assert_eq!(fast_path_names, expected_fast_path_names);
|
||||
let expected_probes = PREFIX_COUNT * 3 + 3;
|
||||
assert_eq!(conservative_probes, expected_probes);
|
||||
assert_eq!(fast_path_names, expected_names);
|
||||
assert_eq!(conservative_probes, PREFIX_COUNT * 3);
|
||||
assert_eq!(fast_path_probes, 0);
|
||||
}
|
||||
|
||||
|
||||
+33
-248
@@ -298,39 +298,6 @@ pub(crate) mod windows_rename_test_hooks {
|
||||
}
|
||||
}
|
||||
|
||||
/// Test-only hooks into the destination-parent walk of rename preparation.
|
||||
///
|
||||
/// The prune race lives between two syscalls inside
|
||||
/// [`mkdir_all_below_existing_base_std`], so only an injection at that exact
|
||||
/// point reproduces it deterministically. Hooks are keyed by the absolute path
|
||||
/// of the component just opened and queued per path: a retrying preparation
|
||||
/// visits the same component again, so a test models a pruner that keeps
|
||||
/// walking upward by queueing one hook per visit.
|
||||
#[cfg(all(test, unix))]
|
||||
pub(crate) mod prepare_rename_test_hooks {
|
||||
use super::*;
|
||||
|
||||
type Hook = Box<dyn FnOnce() + Send>;
|
||||
|
||||
static AFTER_COMPONENT_OPENED: LazyLock<Mutex<HashMap<PathBuf, VecDeque<Hook>>>> =
|
||||
LazyLock::new(|| Mutex::new(HashMap::new()));
|
||||
|
||||
pub(crate) fn queue_after_component_opened(path: &Path, hook: impl FnOnce() + Send + 'static) {
|
||||
AFTER_COMPONENT_OPENED
|
||||
.lock()
|
||||
.entry(path.to_path_buf())
|
||||
.or_default()
|
||||
.push_back(Box::new(hook));
|
||||
}
|
||||
|
||||
pub(crate) fn run_after_component_opened(path: &Path) {
|
||||
let hook = AFTER_COMPONENT_OPENED.lock().get_mut(path).and_then(VecDeque::pop_front);
|
||||
if let Some(hook) = hook {
|
||||
hook();
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// Fsync a directory so recently created or renamed entries survive power loss.
|
||||
/// No-op on non-Unix platforms where directories cannot be opened for syncing.
|
||||
pub fn fsync_dir_std(dir: impl AsRef<Path>) -> io::Result<()> {
|
||||
@@ -1938,8 +1905,8 @@ pub(crate) async fn rename_all_with_prepared_source(
|
||||
let base_dir = base_dir.clone();
|
||||
move || {
|
||||
validate_prepared_rename_source(&prepared_source, &src_file_path)?;
|
||||
let preparation = prepare_rename_with_retry(&src_file_path, &dst_file_path, &base_dir, &publication_root)?;
|
||||
rename_prepared(&src_file_path, &dst_file_path, &preparation)
|
||||
let (preparation, attempt) = prepare_rename_with_retry(&src_file_path, &dst_file_path, &base_dir, &publication_root)?;
|
||||
rename_prepared(&src_file_path, &dst_file_path, &preparation, attempt)
|
||||
}
|
||||
};
|
||||
let result = run_blocking_namespace_operation(lease, operation).await;
|
||||
@@ -2041,8 +2008,8 @@ async fn reliable_rename_inner_with_lease(
|
||||
let dst_file_path = dst_file_path.clone();
|
||||
let base_dir = base_dir.clone();
|
||||
move || {
|
||||
let preparation = prepare_rename_with_retry(&src_file_path, &dst_file_path, &base_dir, &publication_root)?;
|
||||
rename_prepared(&src_file_path, &dst_file_path, &preparation)
|
||||
let (preparation, attempt) = prepare_rename_with_retry(&src_file_path, &dst_file_path, &base_dir, &publication_root)?;
|
||||
rename_prepared(&src_file_path, &dst_file_path, &preparation, attempt)
|
||||
}
|
||||
};
|
||||
let result = run_blocking_namespace_operation(lease, operation).await;
|
||||
@@ -2266,13 +2233,12 @@ fn prepare_rename_with_retry(
|
||||
dst_file_path: &Path,
|
||||
base_dir: &Path,
|
||||
publication_root: &PublicationRoot,
|
||||
) -> io::Result<RenamePreparation> {
|
||||
let prune_budget = prepare_prune_budget(dst_file_path, base_dir);
|
||||
) -> io::Result<(RenamePreparation, usize)> {
|
||||
let mut attempt = 0;
|
||||
loop {
|
||||
match prepare_rename(src_file_path, dst_file_path, base_dir, publication_root) {
|
||||
Ok(preparation) => return Ok(preparation),
|
||||
Err(err) if should_retry_prepare(&err, attempt, prune_budget) => {
|
||||
Ok(preparation) => return Ok((preparation, attempt)),
|
||||
Err(err) if should_retry_rename(&err, attempt) => {
|
||||
attempt += 1;
|
||||
}
|
||||
Err(err) => return Err(err),
|
||||
@@ -2286,26 +2252,23 @@ fn prepare_rename_with_retry(
|
||||
dst_file_path: &Path,
|
||||
base_dir: &Path,
|
||||
publication_root: &PublicationRoot,
|
||||
) -> io::Result<RenamePreparation> {
|
||||
) -> io::Result<(RenamePreparation, usize)> {
|
||||
let source_parent = src_file_path
|
||||
.parent()
|
||||
.ok_or_else(|| io::Error::new(io::ErrorKind::InvalidInput, "rename source must have a parent directory"))?;
|
||||
let destination_parent = dst_file_path.parent();
|
||||
let prune_budget = prepare_prune_budget(dst_file_path, base_dir);
|
||||
// The destination walk and the source open below keep separate counters:
|
||||
// exhausting one must not deny the other its own retry.
|
||||
let prepare_destination_parent = || -> io::Result<Option<ExistingBaseDirectoryGuard>> {
|
||||
let mut attempt = 0;
|
||||
let mut attempt = 0;
|
||||
let prepare_destination_parent = |attempt: &mut usize| -> io::Result<Option<ExistingBaseDirectoryGuard>> {
|
||||
loop {
|
||||
let result = destination_parent
|
||||
.map(|parent| mkdir_all_below_existing_base_std(parent, base_dir, publication_root))
|
||||
.transpose();
|
||||
match result {
|
||||
Ok(parent_guard) => break Ok(parent_guard),
|
||||
Err(err) if should_retry_prepare(&err, attempt, prune_budget) => {
|
||||
Err(err) if should_retry_rename(&err, *attempt) => {
|
||||
#[cfg(test)]
|
||||
windows_rename_test_hooks::run_before_rename_retry(dst_file_path);
|
||||
attempt += 1;
|
||||
*attempt += 1;
|
||||
}
|
||||
Err(err) => break Err(err),
|
||||
}
|
||||
@@ -2318,7 +2281,7 @@ fn prepare_rename_with_retry(
|
||||
None => false,
|
||||
};
|
||||
let (source_parent_guard, parent_guard, source_identity_anchor, expected_source_identity) = if same_parent {
|
||||
let parent_guard = prepare_destination_parent()?;
|
||||
let parent_guard = prepare_destination_parent(&mut attempt)?;
|
||||
let source_parent_guard = parent_guard
|
||||
.as_ref()
|
||||
.ok_or_else(|| io::Error::new(io::ErrorKind::InvalidInput, "rename destination must have a parent directory"))?
|
||||
@@ -2330,17 +2293,16 @@ fn prepare_rename_with_retry(
|
||||
let source_parent_guard = lock_windows_directory_tree(source_parent, destination_parent, publication_root)?;
|
||||
let (source_identity_anchor, expected_source_identity) =
|
||||
open_windows_rename_source_identity(src_file_path, &source_parent_guard)?;
|
||||
let parent_guard = prepare_destination_parent()?;
|
||||
let parent_guard = prepare_destination_parent(&mut attempt)?;
|
||||
(source_parent_guard, parent_guard, source_identity_anchor, expected_source_identity)
|
||||
};
|
||||
let mut source_attempt = 0;
|
||||
let source = loop {
|
||||
match open_windows_rename_source(src_file_path, &source_parent_guard) {
|
||||
Ok(source) => break source,
|
||||
Err(err) if should_retry_rename(&err, source_attempt) => {
|
||||
Err(err) if should_retry_rename(&err, attempt) => {
|
||||
#[cfg(test)]
|
||||
windows_rename_test_hooks::run_before_rename_retry(dst_file_path);
|
||||
source_attempt += 1;
|
||||
attempt += 1;
|
||||
}
|
||||
Err(err) => return Err(err),
|
||||
}
|
||||
@@ -2353,11 +2315,14 @@ fn prepare_rename_with_retry(
|
||||
}
|
||||
drop(source_identity_anchor);
|
||||
|
||||
Ok(RenamePreparation {
|
||||
parent_guard,
|
||||
_source_parent_guard: source_parent_guard,
|
||||
source,
|
||||
})
|
||||
Ok((
|
||||
RenamePreparation {
|
||||
parent_guard,
|
||||
_source_parent_guard: source_parent_guard,
|
||||
source,
|
||||
},
|
||||
attempt,
|
||||
))
|
||||
}
|
||||
|
||||
#[cfg(not(windows))]
|
||||
@@ -2374,22 +2339,24 @@ fn prepare_rename(
|
||||
Ok(RenamePreparation { parent_guard })
|
||||
}
|
||||
|
||||
/// Publish a prepared rename. The retry budget starts fresh here: preparation
|
||||
/// keeps its own counter, so a chain rebuilt after a concurrent prune must not
|
||||
/// cost the rename its one retry.
|
||||
fn rename_prepared(_src_file_path: &Path, dst_file_path: &Path, preparation: &RenamePreparation) -> io::Result<()> {
|
||||
fn rename_prepared(
|
||||
_src_file_path: &Path,
|
||||
dst_file_path: &Path,
|
||||
preparation: &RenamePreparation,
|
||||
attempt: usize,
|
||||
) -> io::Result<()> {
|
||||
#[cfg(windows)]
|
||||
{
|
||||
let parent_guard = preparation
|
||||
.parent_guard
|
||||
.as_ref()
|
||||
.ok_or_else(|| io::Error::new(io::ErrorKind::InvalidInput, "rename destination must have a parent directory"))?;
|
||||
rename_windows_prepared(dst_file_path, parent_guard, &preparation.source, 0)
|
||||
rename_windows_prepared(dst_file_path, parent_guard, &preparation.source, attempt)
|
||||
}
|
||||
|
||||
#[cfg(not(windows))]
|
||||
{
|
||||
let mut attempt = 0;
|
||||
let mut attempt = attempt;
|
||||
loop {
|
||||
let rename_result = rename_into_existing_parent(_src_file_path, dst_file_path, preparation.parent_guard.as_ref());
|
||||
match rename_result {
|
||||
@@ -3789,8 +3756,6 @@ pub(crate) fn mkdir_all_below_existing_base_std(
|
||||
let mode = Mode::RWXU | Mode::RWXG | Mode::RWXO;
|
||||
let mut parents = vec![open(base_dir, flags, Mode::empty()).map_err(io::Error::from)?];
|
||||
|
||||
#[cfg(test)]
|
||||
let mut walked_path = base_dir.to_path_buf();
|
||||
for component in relative.components() {
|
||||
let Component::Normal(component) = component else {
|
||||
continue;
|
||||
@@ -3804,11 +3769,6 @@ pub(crate) fn mkdir_all_below_existing_base_std(
|
||||
Err(err) => return Err(err.into()),
|
||||
}
|
||||
parents.push(openat(parent, component, flags, Mode::empty()).map_err(io::Error::from)?);
|
||||
#[cfg(test)]
|
||||
{
|
||||
walked_path.push(component);
|
||||
prepare_rename_test_hooks::run_after_component_opened(&walked_path);
|
||||
}
|
||||
}
|
||||
|
||||
Ok(parents)
|
||||
@@ -3906,51 +3866,11 @@ fn warn_reliable_rename_failure(src_file_path: &Path, dst_file_path: &Path, base
|
||||
/// cleanup renames (e.g. `move_to_trash` on an already-removed tmp path) a
|
||||
/// pointless second syscall. This predicate is shared by the `rename_data`
|
||||
/// commit path via `rename_all`, so any relaxation here must keep genuine
|
||||
/// transient errors retryable. The *preparation* phase deliberately uses
|
||||
/// [`should_retry_prepare`] instead — see there for why `NotFound` is
|
||||
/// recoverable while the destination parent chain is still being built.
|
||||
/// transient errors retryable.
|
||||
fn should_retry_rename(err: &io::Error, attempt: usize) -> bool {
|
||||
attempt == 0 && err.kind() != io::ErrorKind::NotFound
|
||||
}
|
||||
|
||||
/// How many times rename preparation may retry a `NotFound`.
|
||||
///
|
||||
/// A pruning walk (`LocalDisk::delete_file`) removes empty ancestors
|
||||
/// monotonically upward and stops at the volume root, so it can invalidate
|
||||
/// each component *below* the base at most once. One attempt per such
|
||||
/// component therefore outlasts a pruning walk, and concurrent walks only
|
||||
/// steal an attempt by making that same upward progress. A destination whose
|
||||
/// parent *is* the base gets a budget of zero, keeping `NotFound` immediately
|
||||
/// terminal for speculative cleanup renames.
|
||||
fn prepare_prune_budget(dst_file_path: &Path, base_dir: &Path) -> usize {
|
||||
dst_file_path
|
||||
.parent()
|
||||
.and_then(|parent| parent.strip_prefix(base_dir).ok())
|
||||
.map(|relative| relative.components().count())
|
||||
.unwrap_or(0)
|
||||
}
|
||||
|
||||
/// Whether a failed rename *preparation* attempt (building the destination
|
||||
/// parent chain) should be retried.
|
||||
///
|
||||
/// Unlike [`should_retry_rename`], `NotFound` is recoverable here: a concurrent
|
||||
/// delete prunes now-empty parent directories, so it can unlink an intermediate
|
||||
/// destination component between this walk opening a directory and creating the
|
||||
/// next child inside it, which a handle-relative `mkdirat`/`openat` reports as
|
||||
/// `NotFound`. Each retry rebuilds the whole chain from the base directory,
|
||||
/// which no walk below it can remove; `prune_budget` bounds how far a pruner
|
||||
/// can push the walk back. A genuinely missing base directory fails identically
|
||||
/// on every attempt — the base is only ever opened, never created — so the
|
||||
/// missing-base contract holds at the cost of a few extra syscalls on an
|
||||
/// already-failing path.
|
||||
fn should_retry_prepare(err: &io::Error, attempt: usize, prune_budget: usize) -> bool {
|
||||
if err.kind() == io::ErrorKind::NotFound {
|
||||
attempt < prune_budget
|
||||
} else {
|
||||
attempt == 0
|
||||
}
|
||||
}
|
||||
|
||||
pub async fn reliable_mkdir_all(path: impl AsRef<Path>, base_dir: impl AsRef<Path>) -> io::Result<()> {
|
||||
let mut i = 0;
|
||||
|
||||
@@ -4481,141 +4401,6 @@ mod tests {
|
||||
assert!(!should_retry_rename(&denied, 1));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn prepare_retry_budget_covers_every_prunable_component() {
|
||||
// A pruner can invalidate each component below the base once, so the
|
||||
// budget must match the chain depth, not a fixed count.
|
||||
let not_found = io::Error::new(io::ErrorKind::NotFound, "pruned");
|
||||
assert!(should_retry_prepare(¬_found, 0, 2));
|
||||
assert!(should_retry_prepare(¬_found, 1, 2));
|
||||
assert!(!should_retry_prepare(¬_found, 2, 2));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn prepare_retry_keeps_other_errors_at_a_single_retry() {
|
||||
// Only a prune produces a recoverable NotFound; everything else keeps
|
||||
// the historical single retry so persistent failures stay cheap.
|
||||
let denied = io::Error::new(io::ErrorKind::PermissionDenied, "denied");
|
||||
assert!(should_retry_prepare(&denied, 0, 3));
|
||||
assert!(!should_retry_prepare(&denied, 1, 3));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn prepare_retry_budget_is_zero_when_the_parent_is_the_base() {
|
||||
// Speculative cleanup renames (move_to_trash) land directly in their
|
||||
// base, so a NotFound there is a missing base: terminal, not a prune.
|
||||
let base = Path::new("/vol");
|
||||
assert_eq!(prepare_prune_budget(Path::new("/vol/entry"), base), 0);
|
||||
assert_eq!(prepare_prune_budget(Path::new("/vol/data-movement/sha/id/xl.meta"), base), 3);
|
||||
let not_found = io::Error::new(io::ErrorKind::NotFound, "missing base");
|
||||
assert!(!should_retry_prepare(¬_found, 0, 0));
|
||||
}
|
||||
|
||||
#[cfg(unix)]
|
||||
#[tokio::test]
|
||||
async fn rename_all_survives_concurrent_empty_parent_prune() {
|
||||
// A multipart staging cleanup prunes the momentarily empty shared
|
||||
// `data-movement/` prefix while a concurrent upload publishes its
|
||||
// xl.meta below that same prefix. The writer's walk holds an fd to the
|
||||
// pruned component, so its next handle-relative mkdirat fails
|
||||
// NotFound; preparation must rebuild the chain and still publish.
|
||||
let temp_dir = tempdir().expect("create temp dir");
|
||||
let base = temp_dir.path().join("multipart-volume");
|
||||
let shared = base.join("data-movement");
|
||||
std::fs::create_dir_all(&shared).expect("create shared prefix");
|
||||
let src = temp_dir.path().join("staged.meta");
|
||||
std::fs::write(&src, b"payload").expect("write staged meta");
|
||||
let dst = shared.join("sha").join("upload-id").join("xl.meta");
|
||||
|
||||
let pruned = shared.clone();
|
||||
prepare_rename_test_hooks::queue_after_component_opened(&shared, move || {
|
||||
// The cleanup chain's empty-parent prune lands after the writer
|
||||
// opened the shared component but before it creates its child.
|
||||
std::fs::remove_dir(&pruned).expect("prune the empty shared prefix");
|
||||
});
|
||||
|
||||
rename_all(&src, &dst, &base)
|
||||
.await
|
||||
.expect("a concurrently pruned intermediate directory must not fail the publish");
|
||||
|
||||
assert_eq!(std::fs::read(&dst).expect("read published meta"), b"payload");
|
||||
assert!(!src.exists(), "publish must consume the staged source");
|
||||
}
|
||||
|
||||
#[cfg(unix)]
|
||||
#[tokio::test]
|
||||
async fn rename_all_survives_a_prune_walking_up_every_shared_component() {
|
||||
// The multipart data-movement chain has TWO shared components below the
|
||||
// volume (`data-movement/` and the per-object `<sha>/`), so one cleanup
|
||||
// walk pruning upward can invalidate the writer twice: once at <sha>,
|
||||
// then again at data-movement while the writer rebuilds. A budget that
|
||||
// covers only a single component would still break write quorum here.
|
||||
let temp_dir = tempdir().expect("create temp dir");
|
||||
let base = temp_dir.path().join("multipart-volume");
|
||||
let movement = base.join("data-movement");
|
||||
let sha = movement.join("sha");
|
||||
std::fs::create_dir_all(&sha).expect("create shared chain");
|
||||
let src = temp_dir.path().join("staged.meta");
|
||||
std::fs::write(&src, b"payload").expect("write staged meta");
|
||||
let dst = sha.join("upload-id").join("xl.meta");
|
||||
|
||||
// First visit of `data-movement` is the writer's initial walk, which the
|
||||
// pruner has not reached yet; it prunes on the writer's rebuild.
|
||||
prepare_rename_test_hooks::queue_after_component_opened(&movement, || {});
|
||||
let pruned_sha = sha.clone();
|
||||
prepare_rename_test_hooks::queue_after_component_opened(&sha, move || {
|
||||
std::fs::remove_dir(&pruned_sha).expect("prune the empty per-object prefix");
|
||||
});
|
||||
let pruned_movement = movement.clone();
|
||||
prepare_rename_test_hooks::queue_after_component_opened(&movement, move || {
|
||||
std::fs::remove_dir(&pruned_movement).expect("prune the empty data-movement prefix");
|
||||
});
|
||||
|
||||
rename_all(&src, &dst, &base)
|
||||
.await
|
||||
.expect("a prune walking up the whole shared chain must not fail the publish");
|
||||
|
||||
assert_eq!(std::fs::read(&dst).expect("read published meta"), b"payload");
|
||||
assert!(!src.exists(), "publish must consume the staged source");
|
||||
}
|
||||
|
||||
#[cfg(unix)]
|
||||
#[tokio::test]
|
||||
async fn rename_all_rejects_a_symlink_swapped_in_between_prepare_attempts() {
|
||||
// The retry must not become a traversal window: replacing the pruned
|
||||
// component with a symlink out of the volume before the rebuilt walk
|
||||
// reopens it must fail closed, exactly as a symlink staged before the
|
||||
// first attempt does.
|
||||
use std::os::unix::fs::symlink;
|
||||
|
||||
let temp_dir = tempdir().expect("create temp dir");
|
||||
let base = temp_dir.path().join("multipart-volume");
|
||||
let shared = base.join("data-movement");
|
||||
let outside = temp_dir.path().join("outside");
|
||||
std::fs::create_dir_all(&shared).expect("create shared prefix");
|
||||
std::fs::create_dir_all(&outside).expect("create outside target");
|
||||
let src = temp_dir.path().join("staged.meta");
|
||||
std::fs::write(&src, b"payload").expect("write staged meta");
|
||||
let dst = shared.join("sha").join("upload-id").join("xl.meta");
|
||||
|
||||
let swapped = shared.clone();
|
||||
let outside_target = outside.clone();
|
||||
prepare_rename_test_hooks::queue_after_component_opened(&shared, move || {
|
||||
std::fs::remove_dir(&swapped).expect("prune the shared prefix");
|
||||
symlink(&outside_target, &swapped).expect("replace the pruned component with a symlink");
|
||||
});
|
||||
|
||||
rename_all(&src, &dst, &base)
|
||||
.await
|
||||
.expect_err("a symlink swapped in between attempts must not be followed");
|
||||
|
||||
assert!(src.exists(), "rejected publish must preserve the staged source");
|
||||
assert!(
|
||||
!outside.join("sha").exists(),
|
||||
"the rebuilt walk must not create or publish through the replacement symlink"
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn is_dir_not_empty_error_recognizes_directory_not_empty_kind() {
|
||||
let err = io::Error::from(io::ErrorKind::DirectoryNotEmpty);
|
||||
|
||||
@@ -16,143 +16,10 @@ use rustfs_filemeta::{MetacacheReader, MetacacheWriter};
|
||||
use std::io::Cursor;
|
||||
use std::path::PathBuf;
|
||||
use std::sync::Arc;
|
||||
use std::sync::atomic::{AtomicBool, Ordering};
|
||||
use tokio::fs;
|
||||
use tokio::io::AsyncReadExt;
|
||||
use tokio::sync::RwLock;
|
||||
|
||||
/// Test-only lock client whose refresh path can be rejected independently of
|
||||
/// every other lock operation. The observed event is awaitable so lock-loss
|
||||
/// tests do not depend on sleeps or scheduler timing.
|
||||
#[derive(Debug)]
|
||||
pub(crate) struct RefreshLossLockClient {
|
||||
inner: rustfs_lock::LocalClient,
|
||||
reject_refresh: AtomicBool,
|
||||
rejected_refresh: AtomicBool,
|
||||
rejected_refresh_notify: tokio::sync::Notify,
|
||||
}
|
||||
|
||||
impl RefreshLossLockClient {
|
||||
pub(crate) fn with_manager(manager: Arc<rustfs_lock::GlobalLockManager>) -> Self {
|
||||
Self {
|
||||
inner: rustfs_lock::LocalClient::with_manager(manager),
|
||||
reject_refresh: AtomicBool::new(false),
|
||||
rejected_refresh: AtomicBool::new(false),
|
||||
rejected_refresh_notify: tokio::sync::Notify::new(),
|
||||
}
|
||||
}
|
||||
|
||||
pub(crate) fn reject_refreshes(&self) {
|
||||
self.reject_refresh.store(true, Ordering::Release);
|
||||
}
|
||||
|
||||
pub(crate) fn refreshes_rejected(&self) -> bool {
|
||||
self.rejected_refresh.load(Ordering::Acquire)
|
||||
}
|
||||
|
||||
pub(crate) async fn wait_for_rejected_refresh(
|
||||
&self,
|
||||
timeout: std::time::Duration,
|
||||
) -> std::result::Result<(), tokio::time::error::Elapsed> {
|
||||
tokio::time::timeout(timeout, async {
|
||||
loop {
|
||||
let notified = self.rejected_refresh_notify.notified();
|
||||
if self.refreshes_rejected() {
|
||||
return;
|
||||
}
|
||||
notified.await;
|
||||
}
|
||||
})
|
||||
.await
|
||||
}
|
||||
}
|
||||
|
||||
#[async_trait::async_trait]
|
||||
impl rustfs_lock::LockClient for RefreshLossLockClient {
|
||||
async fn acquire_lock(&self, request: &rustfs_lock::LockRequest) -> rustfs_lock::Result<rustfs_lock::LockResponse> {
|
||||
rustfs_lock::LockClient::acquire_lock(&self.inner, request).await
|
||||
}
|
||||
|
||||
async fn release(&self, lock_id: &rustfs_lock::LockId) -> rustfs_lock::Result<bool> {
|
||||
rustfs_lock::LockClient::release(&self.inner, lock_id).await
|
||||
}
|
||||
|
||||
async fn refresh(&self, lock_id: &rustfs_lock::LockId) -> rustfs_lock::Result<bool> {
|
||||
if self.reject_refresh.load(Ordering::Acquire) {
|
||||
self.rejected_refresh.store(true, Ordering::Release);
|
||||
self.rejected_refresh_notify.notify_waiters();
|
||||
return Ok(false);
|
||||
}
|
||||
rustfs_lock::LockClient::refresh(&self.inner, lock_id).await
|
||||
}
|
||||
|
||||
async fn force_release(&self, lock_id: &rustfs_lock::LockId) -> rustfs_lock::Result<bool> {
|
||||
rustfs_lock::LockClient::force_release(&self.inner, lock_id).await
|
||||
}
|
||||
|
||||
async fn check_status(&self, lock_id: &rustfs_lock::LockId) -> rustfs_lock::Result<Option<rustfs_lock::LockInfo>> {
|
||||
rustfs_lock::LockClient::check_status(&self.inner, lock_id).await
|
||||
}
|
||||
|
||||
async fn list_lock_leases(&self) -> Vec<rustfs_lock::LockLeaseInfo> {
|
||||
rustfs_lock::LockClient::list_lock_leases(&self.inner).await
|
||||
}
|
||||
|
||||
async fn get_stats(&self) -> rustfs_lock::Result<rustfs_lock::LockStats> {
|
||||
rustfs_lock::LockClient::get_stats(&self.inner).await
|
||||
}
|
||||
|
||||
async fn close(&self) -> rustfs_lock::Result<()> {
|
||||
rustfs_lock::LockClient::close(&self.inner).await
|
||||
}
|
||||
|
||||
async fn is_online(&self) -> bool {
|
||||
rustfs_lock::LockClient::is_online(&self.inner).await
|
||||
}
|
||||
|
||||
async fn is_local(&self) -> bool {
|
||||
rustfs_lock::LockClient::is_local(&self.inner).await
|
||||
}
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
async fn refresh_loss_lock_client_keeps_rejection_observable_for_late_waiters() {
|
||||
let manager = Arc::new(rustfs_lock::GlobalLockManager::Enabled(Arc::new(
|
||||
rustfs_lock::FastObjectLockManager::new(),
|
||||
)));
|
||||
let client = RefreshLossLockClient::with_manager(manager);
|
||||
let resource = rustfs_lock::ObjectKey::new("bucket", "object");
|
||||
let response = rustfs_lock::LockClient::acquire_lock(
|
||||
&client,
|
||||
&rustfs_lock::LockRequest::new(resource, rustfs_lock::LockType::Shared, "refresh-loss-harness"),
|
||||
)
|
||||
.await
|
||||
.expect("acquire should reach the inner local client");
|
||||
let lock_id = response.lock_info.expect("the inner local client should acquire the lock").id;
|
||||
assert_eq!(
|
||||
rustfs_lock::LockClient::list_lock_leases(&client).await.len(),
|
||||
1,
|
||||
"lease diagnostics must remain transparent through the refresh wrapper"
|
||||
);
|
||||
|
||||
client.reject_refreshes();
|
||||
assert!(
|
||||
!rustfs_lock::LockClient::refresh(&client, &lock_id)
|
||||
.await
|
||||
.expect("refresh should return a response")
|
||||
);
|
||||
client
|
||||
.wait_for_rejected_refresh(std::time::Duration::from_millis(50))
|
||||
.await
|
||||
.expect("a waiter registered after rejection must still observe the event");
|
||||
assert!(client.refreshes_rejected());
|
||||
assert!(
|
||||
rustfs_lock::LockClient::release(&client, &lock_id)
|
||||
.await
|
||||
.expect("release should reach the inner local client")
|
||||
);
|
||||
}
|
||||
|
||||
/// Returns the backing [`tempfile::TempDir`]s alongside the set so callers keep
|
||||
/// them alive for the test's duration and the directories are removed on drop.
|
||||
pub(crate) async fn make_local_set_disks(drive_count: usize, parity_count: usize) -> (Vec<tempfile::TempDir>, Arc<SetDisks>) {
|
||||
|
||||
@@ -46,7 +46,6 @@ type ShardReadFuture<'a> = Pin<Box<dyn Future<Output = (usize, ShardReadCost, Re
|
||||
type OwnedShardReadFuture<'a, R> =
|
||||
Pin<Box<dyn Future<Output = (usize, ShardReadCost, Result<Vec<u8>, Error>, Option<BitrotReader<R>>, bool)> + Send + 'a>>;
|
||||
pub(crate) type DeferredReaderReopener<R> = Arc<dyn Fn(usize) -> Option<BitrotReader<R>> + Send + Sync>;
|
||||
pub(crate) type DecodeOutcome = (usize, Option<std::io::Error>, bool);
|
||||
|
||||
type ShardIndexes = SmallVec<[usize; INLINE_SHARD_SLOTS]>;
|
||||
type ActiveReaders = SmallVec<[bool; INLINE_SHARD_SLOTS]>;
|
||||
@@ -2191,10 +2190,8 @@ impl Erasure {
|
||||
W: AsyncWrite + Send + Sync + Unpin,
|
||||
R: crate::erasure::coding::ShardSource,
|
||||
{
|
||||
let (written, error, _) = self
|
||||
.decode_inner(writer, readers, offset, length, total_length, None, Vec::new(), Vec::new())
|
||||
.await;
|
||||
(written, error)
|
||||
self.decode_inner(writer, readers, offset, length, total_length, None, Vec::new(), Vec::new())
|
||||
.await
|
||||
}
|
||||
|
||||
#[allow(dead_code, reason = "read-cost decode path asserted by this file's tests (backlog#1823)")]
|
||||
@@ -2211,10 +2208,8 @@ impl Erasure {
|
||||
W: AsyncWrite + Send + Sync + Unpin,
|
||||
R: crate::erasure::coding::ShardSource,
|
||||
{
|
||||
let (written, error, _) = self
|
||||
.decode_inner(writer, readers, offset, length, total_length, Some(read_costs), Vec::new(), Vec::new())
|
||||
.await;
|
||||
(written, error)
|
||||
self.decode_inner(writer, readers, offset, length, total_length, Some(read_costs), Vec::new(), Vec::new())
|
||||
.await
|
||||
}
|
||||
|
||||
/// GET decode entry point that also carries the deferred-parity stripe
|
||||
@@ -2267,37 +2262,6 @@ impl Erasure {
|
||||
deferred_handles: Vec<Option<DeferredReaderStripeHandle>>,
|
||||
deferred_reopeners: Vec<Option<DeferredReaderReopener<R>>>,
|
||||
) -> (usize, Option<std::io::Error>)
|
||||
where
|
||||
W: AsyncWrite + Send + Sync + Unpin,
|
||||
R: crate::erasure::coding::ShardSource,
|
||||
{
|
||||
let (written, error, _) = self
|
||||
.decode_inner(
|
||||
writer,
|
||||
readers,
|
||||
offset,
|
||||
length,
|
||||
total_length,
|
||||
read_costs,
|
||||
deferred_handles,
|
||||
deferred_reopeners,
|
||||
)
|
||||
.await;
|
||||
(written, error)
|
||||
}
|
||||
|
||||
#[allow(clippy::too_many_arguments)]
|
||||
pub(crate) async fn decode_with_stripe_handles_and_reopeners_with_diagnostics<W, R>(
|
||||
&self,
|
||||
writer: &mut W,
|
||||
readers: Vec<Option<BitrotReader<R>>>,
|
||||
offset: usize,
|
||||
length: usize,
|
||||
total_length: usize,
|
||||
read_costs: Option<Vec<ShardReadCost>>,
|
||||
deferred_handles: Vec<Option<DeferredReaderStripeHandle>>,
|
||||
deferred_reopeners: Vec<Option<DeferredReaderReopener<R>>>,
|
||||
) -> DecodeOutcome
|
||||
where
|
||||
W: AsyncWrite + Send + Sync + Unpin,
|
||||
R: crate::erasure::coding::ShardSource,
|
||||
@@ -2447,48 +2411,36 @@ impl Erasure {
|
||||
read_costs: Option<Vec<ShardReadCost>>,
|
||||
deferred_handles: Vec<Option<DeferredReaderStripeHandle>>,
|
||||
deferred_reopeners: Vec<Option<DeferredReaderReopener<R>>>,
|
||||
) -> DecodeOutcome
|
||||
) -> (usize, Option<std::io::Error>)
|
||||
where
|
||||
W: AsyncWrite + Send + Sync + Unpin,
|
||||
R: crate::erasure::coding::ShardSource,
|
||||
{
|
||||
if readers.len() != self.data_shards + self.parity_shards {
|
||||
record_get_object_pipeline_failure(GET_STAGE_RANGE, GetObjectFailureReason::RangeOrLengthInvalid);
|
||||
return (0, Some(io::Error::new(ErrorKind::InvalidInput, "Invalid number of readers")), false);
|
||||
return (0, Some(io::Error::new(ErrorKind::InvalidInput, "Invalid number of readers")));
|
||||
}
|
||||
|
||||
// block_size/data_shards come from on-disk metadata; a corrupt FileInfo with a
|
||||
// zero here must surface as an error, not a divide-by-zero panic on every GET.
|
||||
if self.block_size == 0 || self.data_shards == 0 {
|
||||
record_get_object_pipeline_failure(GET_STAGE_RANGE, GetObjectFailureReason::RangeOrLengthInvalid);
|
||||
return (
|
||||
0,
|
||||
Some(io::Error::new(ErrorKind::InvalidInput, "Invalid erasure coding parameters")),
|
||||
false,
|
||||
);
|
||||
return (0, Some(io::Error::new(ErrorKind::InvalidInput, "Invalid erasure coding parameters")));
|
||||
}
|
||||
|
||||
let Some(end_offset) = offset.checked_add(length) else {
|
||||
record_get_object_pipeline_failure(GET_STAGE_RANGE, GetObjectFailureReason::RangeOrLengthInvalid);
|
||||
return (
|
||||
0,
|
||||
Some(io::Error::new(ErrorKind::InvalidInput, "offset + length exceeds total length")),
|
||||
false,
|
||||
);
|
||||
return (0, Some(io::Error::new(ErrorKind::InvalidInput, "offset + length exceeds total length")));
|
||||
};
|
||||
if end_offset > total_length {
|
||||
record_get_object_pipeline_failure(GET_STAGE_RANGE, GetObjectFailureReason::RangeOrLengthInvalid);
|
||||
return (
|
||||
0,
|
||||
Some(io::Error::new(ErrorKind::InvalidInput, "offset + length exceeds total length")),
|
||||
false,
|
||||
);
|
||||
return (0, Some(io::Error::new(ErrorKind::InvalidInput, "offset + length exceeds total length")));
|
||||
}
|
||||
|
||||
let mut ret_err = None;
|
||||
|
||||
if length == 0 {
|
||||
return (0, ret_err, false);
|
||||
return (0, ret_err);
|
||||
}
|
||||
|
||||
let mut written = 0;
|
||||
@@ -2528,7 +2480,6 @@ impl Erasure {
|
||||
}
|
||||
};
|
||||
|
||||
let mut exact_quorum = false;
|
||||
if legacy_stripe_prefetch_enabled() {
|
||||
// Depth-1 stripe prefetch (backlog#930 HP-9 step 2): while the current
|
||||
// stripe is reconstructed and emitted, the next stripe's shard reads
|
||||
@@ -2571,7 +2522,6 @@ impl Erasure {
|
||||
let Some((mut shards, errs)) = current.take() else {
|
||||
break;
|
||||
};
|
||||
exact_quorum |= shards.iter().filter(|shard| shard.is_some()).count() == self.data_shards;
|
||||
|
||||
if idx + 1 < blocks.len() {
|
||||
// Overlap: read stripe idx+1 while reconstructing/emitting idx.
|
||||
@@ -2686,7 +2636,6 @@ impl Erasure {
|
||||
let stage_metrics_enabled = rustfs_io_metrics::get_stage_metrics_enabled();
|
||||
let stripe_read_stage_start = get_stage_timer_if_enabled(stage_metrics_enabled);
|
||||
let (mut shards, errs) = reader.read().await;
|
||||
exact_quorum |= shards.iter().filter(|shard| shard.is_some()).count() == self.data_shards;
|
||||
record_get_stage_duration_if_enabled(
|
||||
GET_OBJECT_PATH_LEGACY_DUPLEX,
|
||||
GET_STAGE_STRIPE_READ,
|
||||
@@ -2716,14 +2665,14 @@ impl Erasure {
|
||||
}
|
||||
|
||||
if ret_err.is_some() {
|
||||
return (written, ret_err, exact_quorum);
|
||||
return (written, ret_err);
|
||||
}
|
||||
|
||||
if written < length {
|
||||
ret_err = Some(Error::LessData.into());
|
||||
}
|
||||
|
||||
(written, ret_err, exact_quorum)
|
||||
(written, ret_err)
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
@@ -321,13 +321,6 @@ impl<'a> MultiWriter<'a> {
|
||||
}
|
||||
}
|
||||
|
||||
pub(super) fn take_retryable_internode_write_failure(&mut self) -> Option<Error> {
|
||||
self.errs
|
||||
.iter_mut()
|
||||
.find(|error| error.as_ref().is_some_and(Error::is_retryable_internode_write_failure))
|
||||
.and_then(Option::take)
|
||||
}
|
||||
|
||||
/// Effective budget for one shard operation: the smaller of the per-shard
|
||||
/// stall timeout and the time remaining until the object's absolute cap.
|
||||
/// Returns `None` when neither deadline is configured (wait indefinitely).
|
||||
|
||||
@@ -108,13 +108,6 @@ where
|
||||
(shards, errs)
|
||||
}
|
||||
|
||||
fn heal_writer_failure(writers: &mut MultiWriter<'_>, error: io::Error) -> Error {
|
||||
writers
|
||||
.take_retryable_internode_write_failure()
|
||||
.map(|error| Error::RemoteClientUnavailable(error.to_string()))
|
||||
.unwrap_or_else(|| error.into())
|
||||
}
|
||||
|
||||
impl super::Erasure {
|
||||
pub async fn heal<R>(
|
||||
&self,
|
||||
@@ -209,14 +202,10 @@ impl super::Erasure {
|
||||
.map(|s| Bytes::from(s.unwrap_or_default()))
|
||||
.collect::<Vec<_>>();
|
||||
|
||||
if let Err(error) = writers.write(shards).await {
|
||||
return Err(heal_writer_failure(&mut writers, error));
|
||||
}
|
||||
writers.write(shards).await?;
|
||||
}
|
||||
|
||||
if let Err(error) = writers.shutdown().await {
|
||||
return Err(heal_writer_failure(&mut writers, error));
|
||||
}
|
||||
writers.shutdown().await?;
|
||||
Ok(())
|
||||
}
|
||||
}
|
||||
@@ -257,35 +246,6 @@ mod tests {
|
||||
}
|
||||
}
|
||||
|
||||
struct InternodeFailureWriter {
|
||||
fail_on_write: bool,
|
||||
status: http::StatusCode,
|
||||
}
|
||||
|
||||
impl InternodeFailureWriter {
|
||||
fn error(&self) -> io::Error {
|
||||
rustfs_rio::new_test_internode_http_io_error(rustfs_rio::InternodeHttpErrorKind::HttpStatus(self.status))
|
||||
}
|
||||
}
|
||||
|
||||
impl AsyncWrite for InternodeFailureWriter {
|
||||
fn poll_write(self: Pin<&mut Self>, _cx: &mut Context<'_>, buf: &[u8]) -> Poll<io::Result<usize>> {
|
||||
Poll::Ready(if self.fail_on_write {
|
||||
Err(self.error())
|
||||
} else {
|
||||
Ok(buf.len())
|
||||
})
|
||||
}
|
||||
|
||||
fn poll_flush(self: Pin<&mut Self>, _cx: &mut Context<'_>) -> Poll<io::Result<()>> {
|
||||
Poll::Ready(Ok(()))
|
||||
}
|
||||
|
||||
fn poll_shutdown(self: Pin<&mut Self>, _cx: &mut Context<'_>) -> Poll<io::Result<()>> {
|
||||
Poll::Ready(Err(self.error()))
|
||||
}
|
||||
}
|
||||
|
||||
struct PendingReader;
|
||||
|
||||
impl AsyncRead for PendingReader {
|
||||
@@ -371,94 +331,6 @@ mod tests {
|
||||
assert!(writers.iter().all(Option::is_some));
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
async fn heal_maps_put_file_epoch_conflict_to_retryable_remote_unavailable() {
|
||||
for status in [http::StatusCode::CONFLICT, http::StatusCode::BAD_REQUEST] {
|
||||
for (fail_on_write, data) in [
|
||||
(false, b"".as_slice()),
|
||||
(false, b"payload".as_slice()),
|
||||
(true, b"payload".as_slice()),
|
||||
] {
|
||||
let erasure = Erasure::new(2, 1, 64);
|
||||
let encoded = erasure.encode_data(data).expect("source shards should encode");
|
||||
let readers = encoded
|
||||
.iter()
|
||||
.enumerate()
|
||||
.map(|(index, shard)| {
|
||||
(index < erasure.data_shards).then(|| {
|
||||
BitrotReader::new(Cursor::new(shard.to_vec()), erasure.shard_size(), HashAlgorithm::None, false)
|
||||
})
|
||||
})
|
||||
.collect::<Vec<_>>();
|
||||
let mut writers = (0..erasure.total_shard_count())
|
||||
.map(|index| {
|
||||
(index == erasure.data_shards).then(|| {
|
||||
BitrotWriterWrapper::new(
|
||||
CustomWriter::new_tokio_writer(InternodeFailureWriter { fail_on_write, status }),
|
||||
erasure.shard_size(),
|
||||
HashAlgorithm::None,
|
||||
)
|
||||
})
|
||||
})
|
||||
.collect::<Vec<_>>();
|
||||
let error = erasure
|
||||
.heal(&mut writers, readers, data.len(), &[])
|
||||
.await
|
||||
.expect_err("failed sole target must not satisfy heal write quorum");
|
||||
assert_eq!(
|
||||
matches!(error, Error::RemoteClientUnavailable(_)),
|
||||
status == http::StatusCode::CONFLICT,
|
||||
"status={status}, fail_on_write={fail_on_write}, len={}, error={error:?}",
|
||||
data.len()
|
||||
);
|
||||
assert!(writers.iter().all(Option::is_none), "failed target must not be committed");
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
async fn heal_epoch_conflict_does_not_abort_healthy_target() {
|
||||
for fail_on_write in [false, true] {
|
||||
let erasure = Erasure::new(2, 2, 64);
|
||||
let data = b"healthy target must retain exact reconstructed bytes";
|
||||
let encoded = erasure.encode_data(data).expect("source shards should encode");
|
||||
let readers = encoded
|
||||
.iter()
|
||||
.enumerate()
|
||||
.map(|(index, shard)| {
|
||||
(index < erasure.data_shards)
|
||||
.then(|| BitrotReader::new(Cursor::new(shard.to_vec()), erasure.shard_size(), HashAlgorithm::None, false))
|
||||
})
|
||||
.collect::<Vec<_>>();
|
||||
let mut writers = vec![
|
||||
None,
|
||||
None,
|
||||
Some(BitrotWriterWrapper::new(
|
||||
CustomWriter::new_tokio_writer(InternodeFailureWriter {
|
||||
fail_on_write,
|
||||
status: http::StatusCode::CONFLICT,
|
||||
}),
|
||||
erasure.shard_size(),
|
||||
HashAlgorithm::None,
|
||||
)),
|
||||
Some(inline_writer(erasure.shard_size())),
|
||||
];
|
||||
erasure
|
||||
.heal(&mut writers, readers, data.len(), &[])
|
||||
.await
|
||||
.expect("one healthy target must still satisfy the existing heal quorum");
|
||||
assert!(writers[2].is_none(), "conflicting target must be dropped");
|
||||
assert_eq!(
|
||||
writers[3]
|
||||
.take()
|
||||
.expect("healthy target remains")
|
||||
.into_inline_data()
|
||||
.expect("inline target data"),
|
||||
encoded[3].to_vec()
|
||||
);
|
||||
}
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
async fn heal_reconstructs_missing_parity_shard() {
|
||||
let erasure = Erasure::new(2, 2, 64);
|
||||
|
||||
@@ -278,17 +278,3 @@ fn reduce_errs_buckets_identical_other_messages_together() {
|
||||
assert_eq!(count, 3);
|
||||
assert_eq!(err, Some(DiskError::other("can not get client")));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn stable_io_context_buckets_by_cause_and_preserves_diagnostic_source() {
|
||||
let first = StorageError::other_with_context("tier mutation intent changed", "mutation-a");
|
||||
let second = StorageError::other_with_context("tier mutation intent changed", "mutation-b");
|
||||
|
||||
assert_eq!(first, second, "diagnostic identity must not split quorum buckets");
|
||||
let StorageError::Io(io_error) = first else {
|
||||
panic!("stable context must remain an io error");
|
||||
};
|
||||
assert_eq!(io_error.to_string(), "tier mutation intent changed");
|
||||
let context = io_error.get_ref().expect("stable context must remain downcastable");
|
||||
assert_eq!(context.source().expect("diagnostic source must be retained").to_string(), "mutation-a");
|
||||
}
|
||||
|
||||
@@ -23,36 +23,6 @@ use s3s::S3ErrorCode;
|
||||
pub type Error = StorageError;
|
||||
pub type Result<T> = core::result::Result<T, Error>;
|
||||
|
||||
/// Keeps high-cardinality diagnostic detail in the error source while making
|
||||
/// the rendered `io::Error` stable for quorum aggregation.
|
||||
#[derive(Debug)]
|
||||
struct StableIoContextError {
|
||||
message: &'static str,
|
||||
source: Box<dyn std::error::Error + Send + Sync>,
|
||||
}
|
||||
|
||||
impl std::fmt::Display for StableIoContextError {
|
||||
fn fmt(&self, formatter: &mut std::fmt::Formatter<'_>) -> std::fmt::Result {
|
||||
formatter.write_str(self.message)
|
||||
}
|
||||
}
|
||||
|
||||
impl std::error::Error for StableIoContextError {
|
||||
fn source(&self) -> Option<&(dyn std::error::Error + 'static)> {
|
||||
Some(self.source.as_ref())
|
||||
}
|
||||
}
|
||||
|
||||
pub(crate) fn stable_io_error<E>(message: &'static str, source: E) -> std::io::Error
|
||||
where
|
||||
E: Into<Box<dyn std::error::Error + Send + Sync>>,
|
||||
{
|
||||
std::io::Error::other(StableIoContextError {
|
||||
message,
|
||||
source: source.into(),
|
||||
})
|
||||
}
|
||||
|
||||
/// Storage layer error type covering disk, volume, bucket, object, multipart,
|
||||
/// erasure-coding, and operational error conditions.
|
||||
///
|
||||
@@ -213,18 +183,8 @@ pub enum StorageError {
|
||||
DecommissionNotStarted,
|
||||
#[error("Decommission already running")]
|
||||
DecommissionAlreadyRunning,
|
||||
#[error("Decommission capacity error: {0}")]
|
||||
DecommissionCapacity(String),
|
||||
#[error("decommission_capacity_blocked: Storage reached its minimum free drive threshold.: {message}")]
|
||||
DecommissionCapacityBlocked { message: String },
|
||||
#[error("Rebalance already running")]
|
||||
RebalanceAlreadyRunning,
|
||||
#[error("{operation}: stale pool metadata update rejected for pool {pool_index}; {reason}")]
|
||||
StalePoolMetadataUpdate {
|
||||
operation: String,
|
||||
pool_index: usize,
|
||||
reason: &'static str,
|
||||
},
|
||||
#[error("Operation canceled")]
|
||||
OperationCanceled,
|
||||
#[error("No heal required")]
|
||||
@@ -294,13 +254,6 @@ impl StorageError {
|
||||
StorageError::Io(std::io::Error::other(error))
|
||||
}
|
||||
|
||||
pub(crate) fn other_with_context<E>(message: &'static str, source: E) -> Self
|
||||
where
|
||||
E: Into<Box<dyn std::error::Error + Send + Sync>>,
|
||||
{
|
||||
StorageError::Io(stable_io_error(message, source))
|
||||
}
|
||||
|
||||
pub fn is_not_found(&self) -> bool {
|
||||
matches!(
|
||||
self,
|
||||
@@ -610,20 +563,7 @@ impl Clone for StorageError {
|
||||
StorageError::EntityTooLarge(a, b) => StorageError::EntityTooLarge(*a, *b),
|
||||
StorageError::DoneForNow => StorageError::DoneForNow,
|
||||
StorageError::DecommissionAlreadyRunning => StorageError::DecommissionAlreadyRunning,
|
||||
StorageError::DecommissionCapacity(message) => StorageError::DecommissionCapacity(message.clone()),
|
||||
StorageError::DecommissionCapacityBlocked { message } => StorageError::DecommissionCapacityBlocked {
|
||||
message: message.clone(),
|
||||
},
|
||||
StorageError::RebalanceAlreadyRunning => StorageError::RebalanceAlreadyRunning,
|
||||
StorageError::StalePoolMetadataUpdate {
|
||||
operation,
|
||||
pool_index,
|
||||
reason,
|
||||
} => StorageError::StalePoolMetadataUpdate {
|
||||
operation: operation.clone(),
|
||||
pool_index: *pool_index,
|
||||
reason,
|
||||
},
|
||||
StorageError::OperationCanceled => StorageError::OperationCanceled,
|
||||
StorageError::ErasureReadQuorum => StorageError::ErasureReadQuorum,
|
||||
StorageError::ErasureWriteQuorum => StorageError::ErasureWriteQuorum,
|
||||
@@ -726,10 +666,7 @@ impl StorageError {
|
||||
StorageError::InvalidPart(_, _, _) => StorageErrorCode::InvalidPart,
|
||||
StorageError::DoneForNow => StorageErrorCode::DoneForNow,
|
||||
StorageError::DecommissionAlreadyRunning => StorageErrorCode::DecommissionAlreadyRunning,
|
||||
StorageError::DecommissionCapacity(_) => StorageErrorCode::InvalidArgument,
|
||||
StorageError::DecommissionCapacityBlocked { .. } => StorageErrorCode::StorageFull,
|
||||
StorageError::RebalanceAlreadyRunning => StorageErrorCode::RebalanceAlreadyRunning,
|
||||
StorageError::StalePoolMetadataUpdate { .. } => StorageErrorCode::InvalidArgument,
|
||||
StorageError::OperationCanceled => StorageErrorCode::OperationCanceled,
|
||||
StorageError::ErasureReadQuorum => StorageErrorCode::ErasureReadQuorum,
|
||||
StorageError::ErasureWriteQuorum => StorageErrorCode::ErasureWriteQuorum,
|
||||
|
||||
@@ -301,23 +301,36 @@ pub enum LifecycleDeleteAllPhase {
|
||||
#[doc(hidden)]
|
||||
#[derive(Default)]
|
||||
pub struct LifecycleDeleteAllJournalState {
|
||||
prepared: HashMap<String, crate::bucket::lifecycle::tier_sweeper::Jentry>,
|
||||
mutation_started: bool,
|
||||
}
|
||||
|
||||
impl Debug for LifecycleDeleteAllJournalState {
|
||||
fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result {
|
||||
f.debug_struct("LifecycleDeleteAllJournalState")
|
||||
.field("prepared_count", &self.prepared.len())
|
||||
.field("mutation_started", &self.mutation_started)
|
||||
.finish()
|
||||
}
|
||||
}
|
||||
|
||||
impl LifecycleDeleteAllJournalState {
|
||||
pub(crate) fn contains(&self, name: &str) -> bool {
|
||||
self.prepared.contains_key(name)
|
||||
}
|
||||
|
||||
pub(crate) fn insert(&mut self, name: String, entry: crate::bucket::lifecycle::tier_sweeper::Jentry) {
|
||||
self.prepared.insert(name, entry);
|
||||
}
|
||||
|
||||
pub(crate) fn prepared_entries(&self) -> Vec<crate::bucket::lifecycle::tier_sweeper::Jentry> {
|
||||
self.prepared.values().cloned().collect()
|
||||
}
|
||||
|
||||
pub(crate) fn mark_mutation_started(&mut self) {
|
||||
self.mutation_started = true;
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
pub(crate) fn mutation_started(&self) -> bool {
|
||||
self.mutation_started
|
||||
}
|
||||
@@ -668,16 +681,6 @@ impl Drop for ScannerPublicationCommitScopeInner {
|
||||
}
|
||||
}
|
||||
|
||||
#[derive(Clone, Default)]
|
||||
#[doc(hidden)]
|
||||
pub struct DecommissionCapacityOptions {
|
||||
pub(crate) expected_data_bytes: Option<usize>,
|
||||
pub(crate) operation_id: Option<Uuid>,
|
||||
pub(crate) generation: Option<u64>,
|
||||
pub(crate) owner_nonce: Option<Uuid>,
|
||||
pub(crate) mutation_id: Option<Uuid>,
|
||||
}
|
||||
|
||||
#[derive(Default, Clone)]
|
||||
pub struct ObjectOptions {
|
||||
// Use the maximum parity (N/2), used when saving server configuration files
|
||||
@@ -693,12 +696,6 @@ pub struct ObjectOptions {
|
||||
pub lifecycle_delete_all: Option<LifecycleDeleteAllRequest>,
|
||||
#[doc(hidden)]
|
||||
pub lifecycle_delete_all_journal: Option<Arc<parking_lot::Mutex<LifecycleDeleteAllJournalState>>>,
|
||||
/// Whole-operation authorization created only by consuming a validated
|
||||
/// v6 dispatch-manifest permit. Clones share the authorization, not the
|
||||
/// one-shot permit itself.
|
||||
#[doc(hidden)]
|
||||
pub tier_delete_dispatch_authorization:
|
||||
Option<crate::bucket::lifecycle::tier_delete_journal::TierDeleteDispatchAuthorization>,
|
||||
/// RustFS-only compare-and-set condition checked under the object write lock.
|
||||
pub expected_current_version_id: Option<String>,
|
||||
/// Persisted bucket incarnation observed before authorization.
|
||||
@@ -728,12 +725,6 @@ pub struct ObjectOptions {
|
||||
|
||||
pub data_movement: bool,
|
||||
pub raw_data_movement_read: bool,
|
||||
/// Durable reservation identity carried only by decommission writes. Other
|
||||
/// data-movement users, including rebalance, leave it unset. Keep this
|
||||
/// context boxed because `ObjectOptions` is passed by value through deep
|
||||
/// storage futures.
|
||||
#[doc(hidden)]
|
||||
pub decommission_capacity: Option<Box<DecommissionCapacityOptions>>,
|
||||
/// Materialize the data-movement per-part checksum sidecar for APIs that
|
||||
/// return part checksums. Ordinary object reads leave it encoded.
|
||||
pub include_part_checksums: bool,
|
||||
@@ -797,36 +788,6 @@ pub struct ObjectOptions {
|
||||
/// Storage-owned journal writer used by the atomic delete path. This is
|
||||
/// populated only by the `ECStore` wrapper that holds the namespace locks.
|
||||
pub tier_delete_journal_api: Option<Arc<crate::store::ECStore>>,
|
||||
/// Internal staged-mutation admission supplied by `ECStore`; each local
|
||||
/// publish is fenced namespace-first and then by decommission capacity.
|
||||
#[doc(hidden)]
|
||||
pub decommission_capacity_admission: Option<Arc<crate::store::ECStore>>,
|
||||
}
|
||||
|
||||
impl ObjectOptions {
|
||||
pub(crate) fn with_capacity_expected_data_bytes(expected_data_bytes: Option<usize>) -> Self {
|
||||
Self {
|
||||
decommission_capacity: expected_data_bytes.map(|expected_data_bytes| {
|
||||
Box::new(DecommissionCapacityOptions {
|
||||
expected_data_bytes: Some(expected_data_bytes),
|
||||
..Default::default()
|
||||
})
|
||||
}),
|
||||
..Default::default()
|
||||
}
|
||||
}
|
||||
|
||||
pub(crate) fn capacity_expected_data_bytes(&self) -> Option<usize> {
|
||||
self.decommission_capacity
|
||||
.as_deref()
|
||||
.and_then(|capacity| capacity.expected_data_bytes)
|
||||
}
|
||||
|
||||
pub(crate) fn has_decommission_capacity_reservation(&self) -> bool {
|
||||
self.decommission_capacity
|
||||
.as_deref()
|
||||
.is_some_and(|capacity| capacity.operation_id.is_some())
|
||||
}
|
||||
}
|
||||
|
||||
impl std::fmt::Debug for ObjectOptions {
|
||||
@@ -840,7 +801,6 @@ impl std::fmt::Debug for ObjectOptions {
|
||||
.field("version_id", &self.version_id.is_some())
|
||||
.field("lifecycle_delete_all", &self.lifecycle_delete_all.is_some())
|
||||
.field("lifecycle_delete_all_journal", &self.lifecycle_delete_all_journal.is_some())
|
||||
.field("tier_delete_dispatch_authorization", &self.tier_delete_dispatch_authorization.is_some())
|
||||
.field("expected_current_version_id", &self.expected_current_version_id.is_some())
|
||||
.field("expected_bucket_incarnation_id", &self.expected_bucket_incarnation_id)
|
||||
.field("no_lock", &self.no_lock)
|
||||
@@ -966,7 +926,7 @@ impl ObjectOptions {
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
pub(crate) fn add_namespace_lock_fence(&mut self, fence: &NamespaceLockFence) {
|
||||
pub(crate) fn add_namespace_lock_fence_for_test(&mut self, fence: &NamespaceLockFence) {
|
||||
self.namespace_lock_fence
|
||||
.get_or_insert_with(NamespaceLockFence::new)
|
||||
.extend(fence);
|
||||
|
||||
@@ -368,19 +368,6 @@ impl InstanceContext {
|
||||
Arc::clone(&self.data_movement_generation_notify)
|
||||
}
|
||||
|
||||
pub(crate) fn observe_durable_data_movement_generation(&self, generation: u64) {
|
||||
if generation == 0 || self.data_movement_generation_exhausted.load(Ordering::Acquire) {
|
||||
return;
|
||||
}
|
||||
let previous = self.data_movement_generation.fetch_max(generation, Ordering::AcqRel);
|
||||
if generation == u64::MAX {
|
||||
self.data_movement_generation_exhausted.store(true, Ordering::Release);
|
||||
}
|
||||
if generation > previous {
|
||||
self.data_movement_generation_notify.notify_waiters();
|
||||
}
|
||||
}
|
||||
|
||||
pub(crate) fn scanner_publication_state_allowed(&self) -> bool {
|
||||
!self.data_movement_operation_epoch_exhausted()
|
||||
&& !self.data_movement_generation_exhausted()
|
||||
@@ -399,20 +386,6 @@ impl InstanceContext {
|
||||
}
|
||||
|
||||
pub(crate) fn advance_data_movement_operation_epoch(&self) -> u64 {
|
||||
let (previous, result) = self.advance_data_movement_operation_epoch_only();
|
||||
if result != previous {
|
||||
let _ = self.advance_data_movement_generation();
|
||||
}
|
||||
result
|
||||
}
|
||||
|
||||
pub(crate) fn advance_data_movement_operation_epoch_to_durable_generation(&self, generation: u64) -> u64 {
|
||||
let (_, result) = self.advance_data_movement_operation_epoch_only();
|
||||
self.observe_durable_data_movement_generation(generation);
|
||||
result
|
||||
}
|
||||
|
||||
fn advance_data_movement_operation_epoch_only(&self) -> (u64, u64) {
|
||||
self.scanner_publication_state
|
||||
.store(SCANNER_PUBLICATION_STATE_UNKNOWN, Ordering::Release);
|
||||
let previous = self.data_movement_operation_epoch.load(Ordering::Acquire);
|
||||
@@ -423,7 +396,10 @@ impl InstanceContext {
|
||||
if result == u64::MAX {
|
||||
self.data_movement_operation_epoch_exhausted.store(true, Ordering::Release);
|
||||
}
|
||||
(previous, result)
|
||||
if result != previous {
|
||||
let _ = self.advance_data_movement_generation();
|
||||
}
|
||||
result
|
||||
}
|
||||
|
||||
/// Advance the movement generation after a durable movement transition.
|
||||
|
||||
@@ -29,14 +29,10 @@ use rustfs_madmin::metrics::RealtimeMetrics;
|
||||
use rustfs_madmin::net::NetInfo;
|
||||
use rustfs_madmin::{ItemState, ServerProperties, StorageInfo};
|
||||
use rustfs_utils::XHost;
|
||||
use sha2::{Digest, Sha256};
|
||||
use std::collections::{BTreeMap, HashMap, hash_map::DefaultHasher};
|
||||
use std::future::Future;
|
||||
use std::hash::{Hash, Hasher};
|
||||
use std::sync::{
|
||||
Arc, Mutex, OnceLock,
|
||||
atomic::{AtomicBool, AtomicUsize, Ordering},
|
||||
};
|
||||
use std::sync::{Arc, Mutex, OnceLock};
|
||||
use std::time::{Duration, Instant, SystemTime};
|
||||
use tokio::time::{sleep, timeout};
|
||||
use tokio_util::sync::CancellationToken;
|
||||
@@ -56,20 +52,6 @@ const REMOTE_VERSION_STATE_PROBE_INTERVAL: Duration = Duration::from_secs(10);
|
||||
const REMOTE_VERSION_STATE_PROBE_TIMEOUT: Duration = Duration::from_secs(5);
|
||||
const REMOTE_VERSION_STATE_PROOF_TTL: Duration = Duration::from_secs(30);
|
||||
const CROSS_POOL_FENCE_SUPPORTED_VERSION: u32 = 2;
|
||||
const TIER_DELETE_JOURNAL_POLICY_SUPPORTED_VERSION: u32 = 3;
|
||||
type CrossPoolFencePolicyResult = Result<BTreeMap<String, Uuid>>;
|
||||
|
||||
fn cross_pool_fence_policy_results(
|
||||
peer_epochs: BTreeMap<String, Uuid>,
|
||||
minimum_version: u32,
|
||||
) -> (CrossPoolFencePolicyResult, CrossPoolFencePolicyResult) {
|
||||
let journal_result = if minimum_version >= TIER_DELETE_JOURNAL_POLICY_SUPPORTED_VERSION {
|
||||
Ok(peer_epochs.clone())
|
||||
} else {
|
||||
Err(Error::other("tier delete journal v6 policy capability version is unsupported"))
|
||||
};
|
||||
(Ok(peer_epochs), journal_result)
|
||||
}
|
||||
|
||||
#[derive(Clone, Debug)]
|
||||
pub struct ScannerPublicationLeaseGrant {
|
||||
@@ -125,91 +107,15 @@ struct FleetCapabilityProof {
|
||||
topology_fingerprint: String,
|
||||
peer_epochs: Arc<BTreeMap<String, Uuid>>,
|
||||
expires_at: Instant,
|
||||
generation: Arc<FleetCapabilityProofGeneration>,
|
||||
}
|
||||
|
||||
impl FleetCapabilityProof {
|
||||
fn new(topology_fingerprint: String, peer_epochs: Arc<BTreeMap<String, Uuid>>, expires_at: Instant) -> Self {
|
||||
Self {
|
||||
topology_fingerprint,
|
||||
peer_epochs,
|
||||
expires_at,
|
||||
generation: FleetCapabilityProofGeneration::fresh(),
|
||||
}
|
||||
}
|
||||
|
||||
fn token(&self) -> FleetCapabilityProofToken {
|
||||
FleetCapabilityProofToken {
|
||||
topology_fingerprint: self.topology_fingerprint.clone(),
|
||||
peer_epochs: self.peer_epochs.clone(),
|
||||
}
|
||||
}
|
||||
|
||||
#[cfg(any(test, feature = "test-util"))]
|
||||
fn with_fresh_generation(&self) -> Self {
|
||||
Self::new(self.topology_fingerprint.clone(), Arc::clone(&self.peer_epochs), self.expires_at)
|
||||
}
|
||||
}
|
||||
|
||||
/// Admission generation for effects that must not straddle a fleet-proof
|
||||
/// replacement. Revocation is deliberately non-blocking: it closes admission
|
||||
/// immediately, while the proof slot withholds the successor generation until
|
||||
/// every admitted operation has drained.
|
||||
#[derive(Default)]
|
||||
struct FleetCapabilityProofGeneration {
|
||||
accepting: AtomicBool,
|
||||
active: AtomicUsize,
|
||||
}
|
||||
|
||||
impl FleetCapabilityProofGeneration {
|
||||
fn fresh() -> Arc<Self> {
|
||||
Arc::new(Self {
|
||||
accepting: AtomicBool::new(true),
|
||||
active: AtomicUsize::new(0),
|
||||
})
|
||||
}
|
||||
|
||||
fn try_acquire(self: &Arc<Self>) -> Option<FleetCapabilityProofPermit> {
|
||||
if !self.accepting.load(Ordering::Acquire) {
|
||||
return None;
|
||||
}
|
||||
self.active.fetch_add(1, Ordering::AcqRel);
|
||||
if self.accepting.load(Ordering::Acquire) {
|
||||
Some(FleetCapabilityProofPermit {
|
||||
generation: Arc::clone(self),
|
||||
})
|
||||
} else {
|
||||
self.release();
|
||||
None
|
||||
}
|
||||
}
|
||||
|
||||
fn revoke(&self) {
|
||||
self.accepting.store(false, Ordering::Release);
|
||||
}
|
||||
|
||||
fn is_accepting(&self) -> bool {
|
||||
self.accepting.load(Ordering::Acquire)
|
||||
}
|
||||
|
||||
fn is_drained(&self) -> bool {
|
||||
self.active.load(Ordering::Acquire) == 0
|
||||
}
|
||||
|
||||
fn release(&self) {
|
||||
let previous = self.active.fetch_sub(1, Ordering::AcqRel);
|
||||
debug_assert!(previous > 0, "fleet capability permit count underflow");
|
||||
}
|
||||
}
|
||||
|
||||
struct FleetCapabilityProofPermit {
|
||||
generation: Arc<FleetCapabilityProofGeneration>,
|
||||
}
|
||||
|
||||
impl Drop for FleetCapabilityProofPermit {
|
||||
fn drop(&mut self) {
|
||||
self.generation.release();
|
||||
}
|
||||
}
|
||||
|
||||
#[derive(Clone, PartialEq, Eq)]
|
||||
@@ -221,7 +127,6 @@ struct FleetCapabilityProofToken {
|
||||
#[derive(Default)]
|
||||
struct FleetCapabilityProofState {
|
||||
proof: Option<FleetCapabilityProof>,
|
||||
draining_generation: Option<Arc<FleetCapabilityProofGeneration>>,
|
||||
topology_conflict: bool,
|
||||
}
|
||||
|
||||
@@ -231,17 +136,8 @@ pub(crate) struct RemoteVersionStateFleetProofToken(FleetCapabilityProofToken);
|
||||
#[derive(Clone, PartialEq, Eq)]
|
||||
pub struct CrossPoolFenceFleetProofToken(FleetCapabilityProofToken);
|
||||
|
||||
/// A point-in-time proof that every current storage member implements the v6
|
||||
/// dispatch-manifest policy. It intentionally has no `Clone` implementation:
|
||||
/// one acquisition authorizes one manifest construction attempt.
|
||||
pub(crate) struct TierDeleteJournalFleetProofToken {
|
||||
token: FleetCapabilityProofToken,
|
||||
_permit: FleetCapabilityProofPermit,
|
||||
}
|
||||
|
||||
static REMOTE_VERSION_STATE_FLEET_PROOF: OnceLock<std::sync::RwLock<FleetCapabilityProofState>> = OnceLock::new();
|
||||
static CROSS_POOL_FENCE_FLEET_PROOF: OnceLock<std::sync::RwLock<FleetCapabilityProofState>> = OnceLock::new();
|
||||
static TIER_DELETE_JOURNAL_FLEET_PROOF: OnceLock<std::sync::RwLock<FleetCapabilityProofState>> = OnceLock::new();
|
||||
static REMOTE_VERSION_STATE_PROBE_TOPOLOGY: OnceLock<String> = OnceLock::new();
|
||||
|
||||
fn cross_pool_fence_fleet_proof_slot() -> &'static std::sync::RwLock<FleetCapabilityProofState> {
|
||||
@@ -252,35 +148,8 @@ fn remote_version_state_fleet_proof_slot() -> &'static std::sync::RwLock<FleetCa
|
||||
REMOTE_VERSION_STATE_FLEET_PROOF.get_or_init(|| std::sync::RwLock::new(FleetCapabilityProofState::default()))
|
||||
}
|
||||
|
||||
fn tier_delete_journal_fleet_proof_slot() -> &'static std::sync::RwLock<FleetCapabilityProofState> {
|
||||
TIER_DELETE_JOURNAL_FLEET_PROOF.get_or_init(|| std::sync::RwLock::new(FleetCapabilityProofState::default()))
|
||||
}
|
||||
|
||||
fn revoke_fleet_capability_proof_state(state: &mut FleetCapabilityProofState) {
|
||||
if let Some(proof) = state.proof.take() {
|
||||
proof.generation.revoke();
|
||||
if !proof.generation.is_drained() {
|
||||
state.draining_generation = Some(proof.generation);
|
||||
}
|
||||
}
|
||||
if state
|
||||
.draining_generation
|
||||
.as_ref()
|
||||
.is_some_and(|generation| generation.is_drained())
|
||||
{
|
||||
state.draining_generation = None;
|
||||
}
|
||||
}
|
||||
|
||||
fn revoke_fleet_capability_proof(slot: &std::sync::RwLock<FleetCapabilityProofState>) {
|
||||
let mut state = slot.write().unwrap_or_else(std::sync::PoisonError::into_inner);
|
||||
revoke_fleet_capability_proof_state(&mut state);
|
||||
}
|
||||
|
||||
fn mark_fleet_capability_topology_conflict(slot: &std::sync::RwLock<FleetCapabilityProofState>) {
|
||||
let mut state = slot.write().unwrap_or_else(std::sync::PoisonError::into_inner);
|
||||
state.topology_conflict = true;
|
||||
revoke_fleet_capability_proof_state(&mut state);
|
||||
fn replace_fleet_capability_proof(slot: &std::sync::RwLock<FleetCapabilityProofState>, proof: Option<FleetCapabilityProof>) {
|
||||
slot.write().unwrap_or_else(std::sync::PoisonError::into_inner).proof = proof;
|
||||
}
|
||||
|
||||
fn publish_fleet_capability_probe_result(
|
||||
@@ -292,42 +161,21 @@ fn publish_fleet_capability_probe_result(
|
||||
match result {
|
||||
Ok(peer_epochs) => {
|
||||
let mut state = slot.write().unwrap_or_else(std::sync::PoisonError::into_inner);
|
||||
if let Some(current) = state
|
||||
let peer_epochs = state
|
||||
.proof
|
||||
.as_mut()
|
||||
.filter(|proof| proof.topology_fingerprint == topology_fingerprint && proof.peer_epochs.as_ref() == &peer_epochs)
|
||||
{
|
||||
current.expires_at = observed_at + REMOTE_VERSION_STATE_PROOF_TTL;
|
||||
return None;
|
||||
}
|
||||
|
||||
if let Some(previous) = state.proof.take() {
|
||||
previous.generation.revoke();
|
||||
if !previous.generation.is_drained() {
|
||||
state.draining_generation = Some(previous.generation);
|
||||
}
|
||||
}
|
||||
if state
|
||||
.draining_generation
|
||||
.as_ref()
|
||||
.is_some_and(|generation| generation.is_drained())
|
||||
{
|
||||
state.draining_generation = None;
|
||||
}
|
||||
if state.draining_generation.is_some() {
|
||||
return Some(Error::other(
|
||||
"fleet capability proof successor waits for the previous generation to drain",
|
||||
));
|
||||
}
|
||||
state.proof = Some(FleetCapabilityProof::new(
|
||||
topology_fingerprint.to_string(),
|
||||
Arc::new(peer_epochs),
|
||||
observed_at + REMOTE_VERSION_STATE_PROOF_TTL,
|
||||
));
|
||||
.filter(|proof| proof.topology_fingerprint == topology_fingerprint && proof.peer_epochs.as_ref() == &peer_epochs)
|
||||
.map(|proof| Arc::clone(&proof.peer_epochs))
|
||||
.unwrap_or_else(|| Arc::new(peer_epochs));
|
||||
state.proof = Some(FleetCapabilityProof {
|
||||
topology_fingerprint: topology_fingerprint.to_string(),
|
||||
peer_epochs,
|
||||
expires_at: observed_at + REMOTE_VERSION_STATE_PROOF_TTL,
|
||||
});
|
||||
None
|
||||
}
|
||||
Err(err) => {
|
||||
revoke_fleet_capability_proof(slot);
|
||||
replace_fleet_capability_proof(slot, None);
|
||||
Some(err)
|
||||
}
|
||||
}
|
||||
@@ -368,72 +216,7 @@ pub fn cross_pool_fence_fleet_proof_matches(proof: &CrossPoolFenceFleetProofToke
|
||||
fleet_capability_proof_matches(cross_pool_fence_fleet_proof_slot(), &proof.0)
|
||||
}
|
||||
|
||||
pub(crate) fn acquire_tier_delete_journal_fleet_proof() -> Option<TierDeleteJournalFleetProofToken> {
|
||||
let expected_topology = REMOTE_VERSION_STATE_PROBE_TOPOLOGY.get()?;
|
||||
let state = tier_delete_journal_fleet_proof_slot()
|
||||
.read()
|
||||
.unwrap_or_else(std::sync::PoisonError::into_inner);
|
||||
acquire_tier_delete_journal_fleet_proof_from(&state, expected_topology, Instant::now())
|
||||
}
|
||||
|
||||
fn acquire_tier_delete_journal_fleet_proof_from(
|
||||
state: &FleetCapabilityProofState,
|
||||
expected_topology: &str,
|
||||
now: Instant,
|
||||
) -> Option<TierDeleteJournalFleetProofToken> {
|
||||
let token = acquire_fleet_capability_proof_from(state, expected_topology, now)?;
|
||||
let permit = state.proof.as_ref()?.generation.try_acquire()?;
|
||||
Some(TierDeleteJournalFleetProofToken { token, _permit: permit })
|
||||
}
|
||||
|
||||
pub(crate) fn tier_delete_journal_fleet_proof_matches(proof: &TierDeleteJournalFleetProofToken) -> bool {
|
||||
let Some(expected_topology) = REMOTE_VERSION_STATE_PROBE_TOPOLOGY.get() else {
|
||||
return false;
|
||||
};
|
||||
let state = tier_delete_journal_fleet_proof_slot()
|
||||
.read()
|
||||
.unwrap_or_else(std::sync::PoisonError::into_inner);
|
||||
tier_delete_journal_fleet_proof_matches_at(&state, proof, expected_topology, Instant::now())
|
||||
}
|
||||
|
||||
fn tier_delete_journal_fleet_proof_matches_at(
|
||||
state: &FleetCapabilityProofState,
|
||||
proof: &TierDeleteJournalFleetProofToken,
|
||||
expected_topology: &str,
|
||||
now: Instant,
|
||||
) -> bool {
|
||||
proof._permit.generation.is_accepting()
|
||||
&& fleet_capability_proof_matches_at(state, &proof.token, expected_topology, now)
|
||||
&& state
|
||||
.proof
|
||||
.as_ref()
|
||||
.is_some_and(|current| Arc::ptr_eq(¤t.generation, &proof._permit.generation))
|
||||
}
|
||||
|
||||
pub(crate) fn tier_delete_journal_topology_generation(proof: &TierDeleteJournalFleetProofToken) -> String {
|
||||
stable_tier_delete_journal_topology_generation(&proof.token.topology_fingerprint)
|
||||
}
|
||||
|
||||
#[cfg(all(test, feature = "test-util"))]
|
||||
pub(crate) fn tier_delete_journal_fleet_proof_has_inflight_for_test() -> bool {
|
||||
let state = tier_delete_journal_fleet_proof_slot()
|
||||
.read()
|
||||
.unwrap_or_else(std::sync::PoisonError::into_inner);
|
||||
state.proof.as_ref().is_some_and(|proof| !proof.generation.is_drained())
|
||||
|| state
|
||||
.draining_generation
|
||||
.as_ref()
|
||||
.is_some_and(|generation| !generation.is_drained())
|
||||
}
|
||||
|
||||
fn stable_tier_delete_journal_topology_generation(topology_fingerprint: &str) -> String {
|
||||
let mut hasher = Sha256::new();
|
||||
hasher.update(b"rustfs-tier-delete-journal-topology-v1\0");
|
||||
hasher.update(topology_fingerprint.as_bytes());
|
||||
rustfs_utils::crypto::hex(hasher.finalize().as_slice())
|
||||
}
|
||||
|
||||
#[cfg(any(test, feature = "test-util"))]
|
||||
#[cfg(test)]
|
||||
pub(crate) fn install_cross_pool_fence_fleet_proof_for_test() {
|
||||
let topology = REMOTE_VERSION_STATE_PROBE_TOPOLOGY
|
||||
.get()
|
||||
@@ -443,39 +226,18 @@ pub(crate) fn install_cross_pool_fence_fleet_proof_for_test() {
|
||||
let mut state = cross_pool_fence_fleet_proof_slot()
|
||||
.write()
|
||||
.unwrap_or_else(std::sync::PoisonError::into_inner);
|
||||
let now = Instant::now();
|
||||
let proof = if !state.topology_conflict && fleet_capability_proof_valid_at(state.proof.as_ref(), &topology, now) {
|
||||
state.proof.clone()
|
||||
} else {
|
||||
Some(FleetCapabilityProof::new(
|
||||
topology,
|
||||
Arc::new(BTreeMap::new()),
|
||||
now + Duration::from_secs(60 * 60),
|
||||
))
|
||||
};
|
||||
state.topology_conflict = false;
|
||||
state.proof = proof.clone();
|
||||
drop(state);
|
||||
let mut journal_state = tier_delete_journal_fleet_proof_slot()
|
||||
.write()
|
||||
.unwrap_or_else(std::sync::PoisonError::into_inner);
|
||||
debug_assert!(
|
||||
journal_state
|
||||
.proof
|
||||
.as_ref()
|
||||
.is_none_or(|current| current.generation.is_drained())
|
||||
);
|
||||
journal_state.topology_conflict = false;
|
||||
journal_state.draining_generation = None;
|
||||
journal_state.proof = proof.as_ref().map(FleetCapabilityProof::with_fresh_generation);
|
||||
state.proof = Some(FleetCapabilityProof {
|
||||
topology_fingerprint: topology,
|
||||
peer_epochs: Arc::new(BTreeMap::new()),
|
||||
expires_at: Instant::now() + Duration::from_secs(60 * 60),
|
||||
});
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
pub(crate) struct CrossPoolFenceFleetProofGuard {
|
||||
previous_proof: Option<FleetCapabilityProof>,
|
||||
previous_topology_conflict: bool,
|
||||
previous_journal_proof: Option<FleetCapabilityProof>,
|
||||
previous_journal_topology_conflict: bool,
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
@@ -484,24 +246,8 @@ impl Drop for CrossPoolFenceFleetProofGuard {
|
||||
let mut state = cross_pool_fence_fleet_proof_slot()
|
||||
.write()
|
||||
.unwrap_or_else(std::sync::PoisonError::into_inner);
|
||||
state.proof = self
|
||||
.previous_proof
|
||||
.take()
|
||||
.as_ref()
|
||||
.map(FleetCapabilityProof::with_fresh_generation);
|
||||
state.draining_generation = None;
|
||||
state.proof = self.previous_proof.take();
|
||||
state.topology_conflict = self.previous_topology_conflict;
|
||||
drop(state);
|
||||
let mut journal_state = tier_delete_journal_fleet_proof_slot()
|
||||
.write()
|
||||
.unwrap_or_else(std::sync::PoisonError::into_inner);
|
||||
journal_state.proof = self
|
||||
.previous_journal_proof
|
||||
.take()
|
||||
.as_ref()
|
||||
.map(FleetCapabilityProof::with_fresh_generation);
|
||||
journal_state.draining_generation = None;
|
||||
journal_state.topology_conflict = self.previous_journal_topology_conflict;
|
||||
}
|
||||
}
|
||||
|
||||
@@ -512,29 +258,12 @@ pub(crate) fn without_cross_pool_fence_fleet_proof_for_test() -> CrossPoolFenceF
|
||||
let mut state = cross_pool_fence_fleet_proof_slot()
|
||||
.write()
|
||||
.unwrap_or_else(std::sync::PoisonError::into_inner);
|
||||
let mut journal_state = tier_delete_journal_fleet_proof_slot()
|
||||
.write()
|
||||
.unwrap_or_else(std::sync::PoisonError::into_inner);
|
||||
let guard = CrossPoolFenceFleetProofGuard {
|
||||
previous_proof: state.proof.clone(),
|
||||
previous_topology_conflict: state.topology_conflict,
|
||||
previous_journal_proof: journal_state.proof.clone(),
|
||||
previous_journal_topology_conflict: journal_state.topology_conflict,
|
||||
};
|
||||
if let Some(proof) = state.proof.take() {
|
||||
proof.generation.revoke();
|
||||
if !proof.generation.is_drained() {
|
||||
state.draining_generation = Some(proof.generation);
|
||||
}
|
||||
}
|
||||
state.proof = None;
|
||||
state.topology_conflict = true;
|
||||
if let Some(proof) = journal_state.proof.take() {
|
||||
proof.generation.revoke();
|
||||
if !proof.generation.is_drained() {
|
||||
journal_state.draining_generation = Some(proof.generation);
|
||||
}
|
||||
}
|
||||
journal_state.topology_conflict = true;
|
||||
guard
|
||||
}
|
||||
|
||||
@@ -546,33 +275,11 @@ pub fn rotate_cross_pool_fence_fleet_proof_for_test() -> bool {
|
||||
let Some(current) = state.proof.as_ref() else {
|
||||
return false;
|
||||
};
|
||||
let proof = FleetCapabilityProof::new(
|
||||
current.topology_fingerprint.clone(),
|
||||
Arc::new(current.peer_epochs.as_ref().clone()),
|
||||
current.expires_at,
|
||||
);
|
||||
state.proof = Some(proof.clone());
|
||||
drop(state);
|
||||
let mut journal_state = tier_delete_journal_fleet_proof_slot()
|
||||
.write()
|
||||
.unwrap_or_else(std::sync::PoisonError::into_inner);
|
||||
journal_state.topology_conflict = false;
|
||||
if let Some(previous) = journal_state.proof.take() {
|
||||
previous.generation.revoke();
|
||||
if !previous.generation.is_drained() {
|
||||
journal_state.draining_generation = Some(previous.generation);
|
||||
}
|
||||
}
|
||||
if journal_state
|
||||
.draining_generation
|
||||
.as_ref()
|
||||
.is_some_and(|generation| generation.is_drained())
|
||||
{
|
||||
journal_state.draining_generation = None;
|
||||
}
|
||||
if journal_state.draining_generation.is_none() {
|
||||
journal_state.proof = Some(proof.with_fresh_generation());
|
||||
}
|
||||
state.proof = Some(FleetCapabilityProof {
|
||||
topology_fingerprint: current.topology_fingerprint.clone(),
|
||||
peer_epochs: Arc::new(current.peer_epochs.as_ref().clone()),
|
||||
expires_at: current.expires_at,
|
||||
});
|
||||
true
|
||||
}
|
||||
|
||||
@@ -584,22 +291,15 @@ fn fleet_capability_proof_matches(
|
||||
return false;
|
||||
};
|
||||
let state = slot.read().unwrap_or_else(std::sync::PoisonError::into_inner);
|
||||
fleet_capability_proof_matches_at(&state, proof, expected_topology, Instant::now())
|
||||
}
|
||||
|
||||
fn fleet_capability_proof_matches_at(
|
||||
state: &FleetCapabilityProofState,
|
||||
proof: &FleetCapabilityProofToken,
|
||||
expected_topology: &str,
|
||||
now: Instant,
|
||||
) -> bool {
|
||||
!state.topology_conflict
|
||||
&& state.proof.as_ref().is_some_and(|current| {
|
||||
current.topology_fingerprint == expected_topology
|
||||
&& current.topology_fingerprint == proof.topology_fingerprint
|
||||
&& Arc::ptr_eq(¤t.peer_epochs, &proof.peer_epochs)
|
||||
&& now < current.expires_at
|
||||
})
|
||||
if state.topology_conflict {
|
||||
return false;
|
||||
}
|
||||
state.proof.as_ref().is_some_and(|current| {
|
||||
current.topology_fingerprint == *expected_topology
|
||||
&& current.topology_fingerprint == proof.topology_fingerprint
|
||||
&& Arc::ptr_eq(¤t.peer_epochs, &proof.peer_epochs)
|
||||
&& Instant::now() < current.expires_at
|
||||
})
|
||||
}
|
||||
|
||||
fn fleet_capability_proof_valid_at(proof: Option<&FleetCapabilityProof>, expected_topology: &str, now: Instant) -> bool {
|
||||
@@ -612,7 +312,7 @@ pub(crate) struct RemoteVersionStateFleetProofGuard;
|
||||
#[cfg(test)]
|
||||
impl Drop for RemoteVersionStateFleetProofGuard {
|
||||
fn drop(&mut self) {
|
||||
revoke_fleet_capability_proof(remote_version_state_fleet_proof_slot());
|
||||
replace_fleet_capability_proof(remote_version_state_fleet_proof_slot(), None);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -648,12 +348,10 @@ fn insert_remote_version_state_peer(peer_epochs: &mut BTreeMap<String, Uuid>, pe
|
||||
pub fn start_remote_version_state_fleet_probe(topology_fingerprint: String) {
|
||||
if REMOTE_VERSION_STATE_PROBE_TOPOLOGY.set(topology_fingerprint.clone()).is_err() {
|
||||
if REMOTE_VERSION_STATE_PROBE_TOPOLOGY.get() != Some(&topology_fingerprint) {
|
||||
for slot in [
|
||||
remote_version_state_fleet_proof_slot(),
|
||||
cross_pool_fence_fleet_proof_slot(),
|
||||
tier_delete_journal_fleet_proof_slot(),
|
||||
] {
|
||||
mark_fleet_capability_topology_conflict(slot);
|
||||
for slot in [remote_version_state_fleet_proof_slot(), cross_pool_fence_fleet_proof_slot()] {
|
||||
let mut state = slot.write().unwrap_or_else(std::sync::PoisonError::into_inner);
|
||||
state.topology_conflict = true;
|
||||
state.proof = None;
|
||||
}
|
||||
}
|
||||
return;
|
||||
@@ -675,7 +373,7 @@ pub fn start_remote_version_state_fleet_probe(topology_fingerprint: String) {
|
||||
}
|
||||
None => Err(Error::other("remote version state fleet capability notification system is unavailable")),
|
||||
};
|
||||
let fence_probe = match get_global_notification_sys() {
|
||||
let fence_result = match get_global_notification_sys() {
|
||||
Some(notification_sys) => timeout(
|
||||
REMOTE_VERSION_STATE_PROBE_TIMEOUT,
|
||||
notification_sys.probe_cross_pool_fence_fleet(&topology_fingerprint),
|
||||
@@ -684,21 +382,13 @@ pub fn start_remote_version_state_fleet_probe(topology_fingerprint: String) {
|
||||
.unwrap_or_else(|_| Err(Error::other("cross-pool fence fleet capability probe timed out"))),
|
||||
None => Err(Error::other("cross-pool fence fleet capability notification system is unavailable")),
|
||||
};
|
||||
let (fence_result, journal_result) = match fence_probe {
|
||||
Ok((peer_epochs, minimum_version)) => cross_pool_fence_policy_results(peer_epochs, minimum_version),
|
||||
Err(err) => {
|
||||
let message = err.to_string();
|
||||
(Err(Error::other(message.clone())), Err(Error::other(message)))
|
||||
}
|
||||
};
|
||||
let topology_conflict = remote_version_state_fleet_proof_slot()
|
||||
.read()
|
||||
.unwrap_or_else(std::sync::PoisonError::into_inner)
|
||||
.topology_conflict;
|
||||
if topology_conflict {
|
||||
revoke_fleet_capability_proof(remote_version_state_fleet_proof_slot());
|
||||
revoke_fleet_capability_proof(cross_pool_fence_fleet_proof_slot());
|
||||
revoke_fleet_capability_proof(tier_delete_journal_fleet_proof_slot());
|
||||
replace_fleet_capability_proof(remote_version_state_fleet_proof_slot(), None);
|
||||
replace_fleet_capability_proof(cross_pool_fence_fleet_proof_slot(), None);
|
||||
} else if let Some(err) = publish_fleet_capability_probe_result(
|
||||
remote_version_state_fleet_proof_slot(),
|
||||
&topology_fingerprint,
|
||||
@@ -719,25 +409,7 @@ pub fn start_remote_version_state_fleet_probe(topology_fingerprint: String) {
|
||||
event = EVENT_NOTIFICATION_CAPABILITY_PROBE,
|
||||
component = LOG_COMPONENT_ECSTORE,
|
||||
subsystem = LOG_SUBSYSTEM_NOTIFICATION,
|
||||
capability = "cross_pool_fence",
|
||||
state = "failed_closed",
|
||||
error = %err,
|
||||
"notification capability probe"
|
||||
);
|
||||
}
|
||||
if !topology_conflict
|
||||
&& let Some(err) = publish_fleet_capability_probe_result(
|
||||
tier_delete_journal_fleet_proof_slot(),
|
||||
&topology_fingerprint,
|
||||
journal_result,
|
||||
Instant::now(),
|
||||
)
|
||||
{
|
||||
debug!(
|
||||
event = EVENT_NOTIFICATION_CAPABILITY_PROBE,
|
||||
component = LOG_COMPONENT_ECSTORE,
|
||||
subsystem = LOG_SUBSYSTEM_NOTIFICATION,
|
||||
capability = "tier_delete_journal_v6_policy",
|
||||
capability = "cross_pool_fence_v2",
|
||||
state = "failed_closed",
|
||||
error = %err,
|
||||
"notification capability probe"
|
||||
@@ -811,7 +483,7 @@ impl NotificationSys {
|
||||
Ok(peer_epochs)
|
||||
}
|
||||
|
||||
async fn probe_cross_pool_fence_fleet(&self, topology_fingerprint: &str) -> Result<(BTreeMap<String, Uuid>, u32)> {
|
||||
async fn probe_cross_pool_fence_fleet(&self, topology_fingerprint: &str) -> Result<BTreeMap<String, Uuid>> {
|
||||
if self.peer_clients.len() != self.peer_topology_hosts.len() {
|
||||
return Err(Error::other("cross-pool fence capability fleet membership is incomplete"));
|
||||
}
|
||||
@@ -822,21 +494,14 @@ impl NotificationSys {
|
||||
client.probe_cross_pool_fence(topology_fingerprint.to_string()).await
|
||||
});
|
||||
let mut peer_epochs = BTreeMap::new();
|
||||
let mut minimum_version = u32::MAX;
|
||||
for result in join_all(probes).await {
|
||||
let (peer, version, epoch) = result?;
|
||||
if version < CROSS_POOL_FENCE_SUPPORTED_VERSION {
|
||||
return Err(Error::other("cross-pool fence capability version is unsupported"));
|
||||
}
|
||||
minimum_version = minimum_version.min(version);
|
||||
insert_remote_version_state_peer(&mut peer_epochs, peer, epoch)?;
|
||||
}
|
||||
// A single-node deployment has no remote member to lower the local
|
||||
// policy version advertised by this binary.
|
||||
if minimum_version == u32::MAX {
|
||||
minimum_version = TIER_DELETE_JOURNAL_POLICY_SUPPORTED_VERSION;
|
||||
}
|
||||
Ok((peer_epochs, minimum_version))
|
||||
Ok(peer_epochs)
|
||||
}
|
||||
}
|
||||
|
||||
@@ -2162,13 +1827,12 @@ impl NotificationSys {
|
||||
join_all(futures).await
|
||||
}
|
||||
|
||||
pub async fn abort_tier_mutation(&self, mutation_id: Uuid, canonical_prepare_payload: Bytes) -> Vec<NotificationPeerErr> {
|
||||
pub async fn abort_tier_mutation(&self, mutation_id: Uuid) -> Vec<NotificationPeerErr> {
|
||||
let mut futures = Vec::with_capacity(self.peer_clients.len());
|
||||
for client in self.peer_clients.iter().cloned() {
|
||||
let payload = canonical_prepare_payload.clone();
|
||||
futures.push(async move {
|
||||
if let Some(client) = client {
|
||||
notification_peer_result(client.host.to_string(), client.abort_tier_mutation(mutation_id, payload).await)
|
||||
notification_peer_result(client.host.to_string(), client.abort_tier_mutation(mutation_id).await)
|
||||
} else {
|
||||
unreachable_notification_peer_err()
|
||||
}
|
||||
@@ -2803,24 +2467,16 @@ fn aggregate_scanner_dirty_usage_acknowledgement_results(
|
||||
mod tests {
|
||||
use super::*;
|
||||
|
||||
#[test]
|
||||
fn cross_pool_v2_remains_generic_but_cannot_authorize_v6_journal() {
|
||||
let peers = BTreeMap::from([("node-b:9000".to_string(), Uuid::new_v4())]);
|
||||
let (generic_v2, journal_v2) = cross_pool_fence_policy_results(peers.clone(), 2);
|
||||
assert!(generic_v2.is_ok(), "v2 remains valid for existing cross-pool fencing");
|
||||
assert!(journal_v2.is_err(), "a mixed v2/v3 fleet must fail closed for journal-v6 deletion");
|
||||
|
||||
let (generic_v3, journal_v3) = cross_pool_fence_policy_results(peers, 3);
|
||||
assert!(generic_v3.is_ok());
|
||||
assert!(journal_v3.is_ok(), "an all-v3 fleet may authorize journal-v6 deletion");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn remote_version_state_fleet_proof_rejects_stale_or_mismatched_membership() {
|
||||
let now = Instant::now();
|
||||
let mut peer_epochs = BTreeMap::new();
|
||||
peer_epochs.insert("peer-a".to_string(), Uuid::new_v4());
|
||||
let proof = FleetCapabilityProof::new("topology-a".to_string(), Arc::new(peer_epochs), now + Duration::from_secs(1));
|
||||
let proof = FleetCapabilityProof {
|
||||
topology_fingerprint: "topology-a".to_string(),
|
||||
peer_epochs: Arc::new(peer_epochs),
|
||||
expires_at: now + Duration::from_secs(1),
|
||||
};
|
||||
|
||||
assert!(fleet_capability_proof_valid_at(Some(&proof), "topology-a", now));
|
||||
assert!(!fleet_capability_proof_valid_at(Some(&proof), "topology-b", now));
|
||||
@@ -2839,7 +2495,11 @@ mod tests {
|
||||
#[test]
|
||||
fn remote_version_state_fleet_proof_accepts_single_node_membership() {
|
||||
let now = Instant::now();
|
||||
let proof = FleetCapabilityProof::new("topology-a".to_string(), Arc::new(BTreeMap::new()), now + Duration::from_secs(1));
|
||||
let proof = FleetCapabilityProof {
|
||||
topology_fingerprint: "topology-a".to_string(),
|
||||
peer_epochs: Arc::new(BTreeMap::new()),
|
||||
expires_at: now + Duration::from_secs(1),
|
||||
};
|
||||
|
||||
assert!(fleet_capability_proof_valid_at(Some(&proof), "topology-a", now));
|
||||
}
|
||||
@@ -2847,97 +2507,21 @@ mod tests {
|
||||
#[test]
|
||||
fn remote_version_state_fleet_proof_token_changes_with_process_epoch() {
|
||||
let now = Instant::now();
|
||||
let proof = FleetCapabilityProof::new(
|
||||
"topology-a".to_string(),
|
||||
Arc::new(BTreeMap::from([("peer-a".to_string(), Uuid::new_v4())])),
|
||||
now + Duration::from_secs(1),
|
||||
);
|
||||
let proof = FleetCapabilityProof {
|
||||
topology_fingerprint: "topology-a".to_string(),
|
||||
peer_epochs: Arc::new(BTreeMap::from([("peer-a".to_string(), Uuid::new_v4())])),
|
||||
expires_at: now + Duration::from_secs(1),
|
||||
};
|
||||
let captured = proof.token();
|
||||
let restarted = FleetCapabilityProof::new(
|
||||
proof.topology_fingerprint.clone(),
|
||||
Arc::new(BTreeMap::from([("peer-a".to_string(), Uuid::new_v4())])),
|
||||
proof.expires_at,
|
||||
);
|
||||
let restarted = FleetCapabilityProof {
|
||||
topology_fingerprint: proof.topology_fingerprint.clone(),
|
||||
peer_epochs: Arc::new(BTreeMap::from([("peer-a".to_string(), Uuid::new_v4())])),
|
||||
expires_at: proof.expires_at,
|
||||
};
|
||||
|
||||
assert!(captured != restarted.token());
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn tier_delete_journal_generation_is_stable_across_members_and_process_restarts() {
|
||||
let topology = "topology-a";
|
||||
let now = Instant::now();
|
||||
let node_a_view = FleetCapabilityProof::new(
|
||||
topology.to_string(),
|
||||
Arc::new(BTreeMap::from([("node-b".to_string(), Uuid::new_v4())])),
|
||||
now + Duration::from_secs(1),
|
||||
);
|
||||
let node_b_view = FleetCapabilityProof::new(
|
||||
topology.to_string(),
|
||||
Arc::new(BTreeMap::from([("node-a".to_string(), Uuid::new_v4())])),
|
||||
now + Duration::from_secs(1),
|
||||
);
|
||||
let restarted_node_a_view = FleetCapabilityProof::new(
|
||||
topology.to_string(),
|
||||
Arc::new(BTreeMap::from([("node-b".to_string(), Uuid::new_v4())])),
|
||||
now + Duration::from_secs(1),
|
||||
);
|
||||
|
||||
let generations = [&node_a_view, &node_b_view, &restarted_node_a_view]
|
||||
.map(|proof| stable_tier_delete_journal_topology_generation(&proof.token().topology_fingerprint));
|
||||
assert_eq!(generations[0], generations[1]);
|
||||
assert_eq!(generations[0], generations[2]);
|
||||
assert_ne!(
|
||||
generations[0],
|
||||
stable_tier_delete_journal_topology_generation("topology-b"),
|
||||
"a real topology change must produce a different durable generation"
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn tier_delete_journal_restart_revokes_old_token_but_fresh_token_recovers_same_generation() {
|
||||
let slot = std::sync::RwLock::new(FleetCapabilityProofState::default());
|
||||
let now = Instant::now();
|
||||
let original_peers = BTreeMap::from([("node-b".to_string(), Uuid::new_v4())]);
|
||||
assert!(publish_fleet_capability_probe_result(&slot, "topology-a", Ok(original_peers), now).is_none());
|
||||
let original = slot
|
||||
.read()
|
||||
.expect("proof slot should not poison")
|
||||
.proof
|
||||
.as_ref()
|
||||
.expect("successful probe should publish proof")
|
||||
.token();
|
||||
let original_generation = stable_tier_delete_journal_topology_generation(&original.topology_fingerprint);
|
||||
|
||||
let restarted_peers = BTreeMap::from([("node-b".to_string(), Uuid::new_v4())]);
|
||||
assert!(
|
||||
publish_fleet_capability_probe_result(&slot, "topology-a", Ok(restarted_peers), now + Duration::from_millis(1))
|
||||
.is_none()
|
||||
);
|
||||
let state = slot.read().expect("proof slot should not poison");
|
||||
let fresh = state
|
||||
.proof
|
||||
.as_ref()
|
||||
.expect("restart probe should publish a fresh proof")
|
||||
.token();
|
||||
|
||||
assert!(!fleet_capability_proof_matches_at(
|
||||
&state,
|
||||
&original,
|
||||
"topology-a",
|
||||
now + Duration::from_millis(2)
|
||||
));
|
||||
assert!(fleet_capability_proof_matches_at(
|
||||
&state,
|
||||
&fresh,
|
||||
"topology-a",
|
||||
now + Duration::from_millis(2)
|
||||
));
|
||||
assert_eq!(
|
||||
original_generation,
|
||||
stable_tier_delete_journal_topology_generation(&fresh.topology_fingerprint)
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn remote_version_state_fleet_proof_renewal_preserves_only_same_epoch_token() {
|
||||
let slot = std::sync::RwLock::new(FleetCapabilityProofState::default());
|
||||
@@ -2977,106 +2561,15 @@ mod tests {
|
||||
assert!(!Arc::ptr_eq(&original.peer_epochs, &replaced.peer_epochs));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn tier_delete_journal_successor_waits_for_inflight_generation_to_drain() {
|
||||
let slot = std::sync::RwLock::new(FleetCapabilityProofState::default());
|
||||
let now = Instant::now();
|
||||
let original_peers = BTreeMap::from([("peer-a".to_string(), Uuid::new_v4())]);
|
||||
assert!(publish_fleet_capability_probe_result(&slot, "topology-a", Ok(original_peers), now).is_none());
|
||||
|
||||
let admitted = {
|
||||
let state = slot.read().expect("proof slot should not poison");
|
||||
acquire_tier_delete_journal_fleet_proof_from(&state, "topology-a", now)
|
||||
.expect("a fresh proof should admit one journal operation")
|
||||
};
|
||||
{
|
||||
let state = slot.read().expect("proof slot should not poison");
|
||||
assert!(
|
||||
tier_delete_journal_fleet_proof_matches_at(&state, &admitted, "topology-a", now),
|
||||
"a freshly admitted journal proof must remain current"
|
||||
);
|
||||
assert!(
|
||||
acquire_tier_delete_journal_fleet_proof_from(&state, "topology-a", now + REMOTE_VERSION_STATE_PROOF_TTL,)
|
||||
.is_none(),
|
||||
"TTL expiry must stop new admission"
|
||||
);
|
||||
assert!(
|
||||
!tier_delete_journal_fleet_proof_matches_at(
|
||||
&state,
|
||||
&admitted,
|
||||
"topology-a",
|
||||
now + REMOTE_VERSION_STATE_PROOF_TTL,
|
||||
),
|
||||
"TTL expiry must also stop an admitted proof at its next durable fence"
|
||||
);
|
||||
assert!(!admitted._permit.generation.is_drained());
|
||||
}
|
||||
|
||||
let restarted_peers = BTreeMap::from([("peer-a".to_string(), Uuid::new_v4())]);
|
||||
let blocked = publish_fleet_capability_probe_result(
|
||||
&slot,
|
||||
"topology-a",
|
||||
Ok(restarted_peers.clone()),
|
||||
now + Duration::from_millis(1),
|
||||
)
|
||||
.expect("a successor proof must wait for the admitted generation");
|
||||
assert!(blocked.to_string().contains("previous generation to drain"));
|
||||
{
|
||||
let state = slot.read().expect("proof slot should not poison");
|
||||
assert!(state.proof.is_none(), "new operations must remain closed while the predecessor drains");
|
||||
assert!(state.draining_generation.is_some());
|
||||
assert!(
|
||||
!tier_delete_journal_fleet_proof_matches_at(&state, &admitted, "topology-a", now + Duration::from_millis(1),),
|
||||
"a restarted peer must revoke an admitted proof before its next durable fence"
|
||||
);
|
||||
}
|
||||
|
||||
drop(admitted);
|
||||
assert!(
|
||||
publish_fleet_capability_probe_result(&slot, "topology-a", Ok(restarted_peers), now + Duration::from_millis(2),)
|
||||
.is_none(),
|
||||
"the successor may publish after the in-flight operation releases its permit"
|
||||
);
|
||||
let state = slot.read().expect("proof slot should not poison");
|
||||
assert!(state.proof.is_some());
|
||||
assert!(state.draining_generation.is_none());
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn tier_delete_journal_topology_conflict_revokes_admitted_generation() {
|
||||
let slot = std::sync::RwLock::new(FleetCapabilityProofState::default());
|
||||
let now = Instant::now();
|
||||
let peers = BTreeMap::from([("peer-a".to_string(), Uuid::new_v4())]);
|
||||
assert!(publish_fleet_capability_probe_result(&slot, "topology-a", Ok(peers), now).is_none());
|
||||
let admitted = {
|
||||
let state = slot.read().expect("proof slot should not poison");
|
||||
acquire_tier_delete_journal_fleet_proof_from(&state, "topology-a", now)
|
||||
.expect("a fresh proof should admit one journal operation")
|
||||
};
|
||||
|
||||
mark_fleet_capability_topology_conflict(&slot);
|
||||
|
||||
let state = slot.read().expect("proof slot should not poison");
|
||||
assert!(state.topology_conflict);
|
||||
assert!(state.proof.is_none());
|
||||
assert!(state.draining_generation.is_some());
|
||||
assert!(!admitted._permit.generation.is_accepting());
|
||||
assert!(
|
||||
!tier_delete_journal_fleet_proof_matches_at(&state, &admitted, "topology-a", now),
|
||||
"topology conflict must revoke an already admitted journal proof"
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn remote_version_state_fleet_proof_conflict_revokes_atomic_snapshot() {
|
||||
let now = Instant::now();
|
||||
let mut state = FleetCapabilityProofState {
|
||||
proof: Some(FleetCapabilityProof::new(
|
||||
"topology-a".to_string(),
|
||||
Arc::new(BTreeMap::new()),
|
||||
now + Duration::from_secs(1),
|
||||
)),
|
||||
draining_generation: None,
|
||||
proof: Some(FleetCapabilityProof {
|
||||
topology_fingerprint: "topology-a".to_string(),
|
||||
peer_epochs: Arc::new(BTreeMap::new()),
|
||||
expires_at: now + Duration::from_secs(1),
|
||||
}),
|
||||
topology_conflict: false,
|
||||
};
|
||||
assert!(acquire_fleet_capability_proof_from(&state, "topology-a", now).is_some());
|
||||
@@ -3786,7 +3279,7 @@ mod tests {
|
||||
assert_eq!(commit.len(), 1);
|
||||
assert!(commit[0].err.is_some());
|
||||
|
||||
let abort = sys.abort_tier_mutation(mutation_id, Bytes::from_static(b"prepare")).await;
|
||||
let abort = sys.abort_tier_mutation(mutation_id).await;
|
||||
assert_eq!(abort.len(), 1);
|
||||
assert!(abort[0].err.is_some());
|
||||
}
|
||||
|
||||
@@ -845,7 +845,7 @@ impl ECStore {
|
||||
|
||||
let mut pool_stats = Vec::with_capacity(self.pools.len());
|
||||
|
||||
let now = self.next_scanner_data_movement_update(OffsetDateTime::now_utc()).await;
|
||||
let now = OffsetDateTime::now_utc();
|
||||
|
||||
for disk_stat in disk_stats.iter() {
|
||||
let mut pool_stat = RebalanceStats {
|
||||
@@ -868,10 +868,8 @@ impl ECStore {
|
||||
pool_stats.push(pool_stat);
|
||||
}
|
||||
|
||||
let has_participating_pool = pool_stats.iter().any(|pool_stat| pool_stat.participating);
|
||||
let meta = RebalanceMeta {
|
||||
id: Uuid::new_v4().to_string(),
|
||||
stopped_at: (!has_participating_pool).then_some(now),
|
||||
percent_free_goal,
|
||||
pool_stats,
|
||||
..Default::default()
|
||||
@@ -965,18 +963,6 @@ impl ECStore {
|
||||
)));
|
||||
}
|
||||
if meta.stopped_at.is_some() {
|
||||
if !is_rebalance_conflicting_with_decommission(meta) {
|
||||
debug!(
|
||||
event = EVENT_REBALANCE_STATE,
|
||||
component = LOG_COMPONENT_ECSTORE,
|
||||
subsystem = LOG_SUBSYSTEM_REBALANCE,
|
||||
state = "start_skipped",
|
||||
reason = "not_started_terminal",
|
||||
rebalance_id = %expected_id,
|
||||
"Skipped rebalance start because metadata is already terminal"
|
||||
);
|
||||
return Ok(());
|
||||
}
|
||||
return Err(Error::other(format!("rebalance {expected_id} was stopped before start")));
|
||||
}
|
||||
}
|
||||
@@ -1228,11 +1214,11 @@ impl ECStore {
|
||||
};
|
||||
let movement_gate = self.ctx.data_movement_operation_gate();
|
||||
let _movement_guard = movement_gate.write().await;
|
||||
let stopped_at = self.next_scanner_data_movement_update(OffsetDateTime::now_utc()).await;
|
||||
let (previous_meta, meta_to_save) = {
|
||||
let mut rebalance_meta = self.rebalance_meta.write().await;
|
||||
let previous_meta = rebalance_meta.clone();
|
||||
let meta_to_save = stop_rebalance_meta_snapshot_for_id(rebalance_meta.as_mut(), stopped_at, expected_id)?;
|
||||
let meta_to_save =
|
||||
stop_rebalance_meta_snapshot_for_id(rebalance_meta.as_mut(), OffsetDateTime::now_utc(), expected_id)?;
|
||||
(previous_meta, meta_to_save)
|
||||
};
|
||||
|
||||
@@ -1264,10 +1250,14 @@ impl ECStore {
|
||||
.await?;
|
||||
let movement_gate = self.ctx.data_movement_operation_gate();
|
||||
let _movement_guard = movement_gate.write().await;
|
||||
let failed_at = self.next_scanner_data_movement_update(OffsetDateTime::now_utc()).await;
|
||||
let meta_to_save = {
|
||||
let mut rebalance_meta = self.rebalance_meta.write().await;
|
||||
rollback_rebalance_start_meta_snapshot_for_id(rebalance_meta.as_mut(), failed_at, expected_id, start_error)
|
||||
rollback_rebalance_start_meta_snapshot_for_id(
|
||||
rebalance_meta.as_mut(),
|
||||
OffsetDateTime::now_utc(),
|
||||
expected_id,
|
||||
start_error,
|
||||
)
|
||||
};
|
||||
|
||||
if let Some(meta_to_save) = meta_to_save {
|
||||
@@ -1329,19 +1319,14 @@ mod tests {
|
||||
use super::*;
|
||||
use crate::config::com::delete_config;
|
||||
use crate::core::pools::{
|
||||
DecommissionErasureLayout, DecommissionPoolCapacityInfo, POOL_META_NAME, PoolActivationDurableSaveBarrier,
|
||||
PoolActivationStartKind, PoolActivationStartProbe, PoolMetaWriteState, persist_pool_meta_identity_for_startup,
|
||||
set_decommission_capacity_info_overrides_for_test,
|
||||
POOL_META_NAME, PoolActivationDurableSaveBarrier, PoolActivationStartKind, PoolActivationStartProbe, PoolMetaWriteState,
|
||||
persist_pool_meta_identity_for_startup,
|
||||
};
|
||||
use crate::object_api::NamespaceLockFence;
|
||||
use crate::set_disk::{PutObjectCommitBarrier, PutObjectCommitPause, hermetic_set_disks_isolated};
|
||||
|
||||
async fn persist_initialized_identity_then_remove_pool_meta(store: &Arc<ECStore>) {
|
||||
let deployment_id = store
|
||||
.ctx
|
||||
.deployment_id()
|
||||
.expect("test store should have a deployment identity");
|
||||
let mut write_state = PoolMetaWriteState::for_startup(deployment_id, false);
|
||||
let mut write_state = PoolMetaWriteState::for_startup(store.id, false);
|
||||
persist_pool_meta_identity_for_startup(store.pools.clone(), &mut write_state, true)
|
||||
.await
|
||||
.expect("initialized pool metadata identity should persist");
|
||||
@@ -1417,62 +1402,6 @@ mod tests {
|
||||
assert!(cancel.is_cancelled());
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
#[serial_test::serial]
|
||||
async fn equal_free_ratio_admin_no_participant_rebalance_succeeds_and_persists_terminal_generation_after_restart() {
|
||||
let (_temp_dirs, store, restarted) =
|
||||
crate::services::rebalance::test_two_pool_stores_with_isolated_node_contexts(None).await;
|
||||
let movement_floor = OffsetDateTime::from_unix_timestamp(4_100_000_000).expect("future test timestamp should be valid");
|
||||
*store.rebalance_meta.write().await = Some(RebalanceMeta {
|
||||
id: "previous-terminal-rebalance".to_string(),
|
||||
stopped_at: Some(movement_floor),
|
||||
..Default::default()
|
||||
});
|
||||
set_rebalance_disk_stats_override_for_test(
|
||||
store.id,
|
||||
vec![
|
||||
DiskStat {
|
||||
total_space: 100,
|
||||
available_space: 50,
|
||||
},
|
||||
DiskStat {
|
||||
total_space: 100,
|
||||
available_space: 50,
|
||||
},
|
||||
],
|
||||
);
|
||||
|
||||
let rebalance_id = store
|
||||
.init_and_start_rebalance(vec!["equal-ratio-no-op".to_string()])
|
||||
.await
|
||||
.expect("equal free ratio admin rebalance should succeed as a terminal no-op");
|
||||
let stopped_at = {
|
||||
let local = store.rebalance_meta.read().await;
|
||||
let local = local.as_ref().expect("no-op rebalance metadata should remain available");
|
||||
assert_eq!(local.id, rebalance_id);
|
||||
assert!(local.pool_stats.iter().all(|pool_stat| !pool_stat.participating));
|
||||
let stopped_at = local.stopped_at.expect("no-op rebalance must persist a terminal timestamp");
|
||||
assert_eq!(stopped_at, movement_floor + time::Duration::nanoseconds(1));
|
||||
stopped_at
|
||||
};
|
||||
|
||||
let stopped_generation =
|
||||
u64::try_from(stopped_at.unix_timestamp_nanos()).expect("terminal timestamp should map to scanner generation");
|
||||
let live_status = store.scanner_data_movement_pause_status().await;
|
||||
assert!(!live_status.paused);
|
||||
assert_eq!(live_status.movement_generation, stopped_generation);
|
||||
|
||||
restarted
|
||||
.load_rebalance_meta()
|
||||
.await
|
||||
.expect("restarted store should load the persisted no-op rebalance metadata");
|
||||
let status = restarted.scanner_data_movement_pause_status().await;
|
||||
|
||||
assert!(!status.paused);
|
||||
assert_eq!(status.movement_generation, stopped_generation);
|
||||
assert_eq!(restarted.scanner_data_movement_generation(), stopped_generation);
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
#[serial_test::serial]
|
||||
async fn rebalance_activation_rejects_initialized_cluster_with_all_pool_meta_missing() {
|
||||
@@ -1752,15 +1681,26 @@ mod tests {
|
||||
];
|
||||
set_rebalance_disk_stats_override_for_test(rebalance_store.id, disk_stats.clone());
|
||||
set_rebalance_disk_stats_override_for_test(decommission_store.id, disk_stats);
|
||||
let layout = DecommissionErasureLayout { data: 1, parity: 0 };
|
||||
let capacity_snapshot = vec![
|
||||
DecommissionPoolCapacityInfo::for_test(0, layout, 0, 100, 100),
|
||||
DecommissionPoolCapacityInfo::for_test(1, layout, 200, 200, 0),
|
||||
];
|
||||
// Decommission start samples capacity before and inside its durable activation fence.
|
||||
set_decommission_capacity_info_overrides_for_test(
|
||||
crate::core::pools::set_decommission_space_info_override_for_test(
|
||||
decommission_store.id,
|
||||
vec![capacity_snapshot.clone(), capacity_snapshot],
|
||||
vec![
|
||||
(
|
||||
0,
|
||||
crate::core::pools::PoolSpaceInfo {
|
||||
free: 0,
|
||||
total: 100,
|
||||
used: 100,
|
||||
},
|
||||
),
|
||||
(
|
||||
1,
|
||||
crate::core::pools::PoolSpaceInfo {
|
||||
free: 200,
|
||||
total: 200,
|
||||
used: 0,
|
||||
},
|
||||
),
|
||||
],
|
||||
);
|
||||
let (first_object, competing_object, competing_kind) = match paused_kind {
|
||||
PoolActivationStartKind::Rebalance => {
|
||||
|
||||
@@ -356,7 +356,7 @@ impl ECStore {
|
||||
};
|
||||
run_guard.ensure_held("rebalance version migration")?;
|
||||
let result = migrate_entry_version(
|
||||
&RebalanceMigrationBackend::new(set.as_ref(), self.clone(), lock_lost_signal.clone()),
|
||||
&RebalanceMigrationBackend::new(set.as_ref(), self.as_ref(), lock_lost_signal.clone()),
|
||||
bucket.clone(),
|
||||
pool_index,
|
||||
version,
|
||||
@@ -1478,7 +1478,6 @@ mod tests {
|
||||
let (_temp_dirs, store, _unused_store) =
|
||||
crate::services::rebalance::test_two_pool_stores(Some(active_rebalance_meta(REBALANCE_ID))).await;
|
||||
prepare_rebalance_test_volumes(store.as_ref()).await;
|
||||
crate::services::tier::test_util::register_mock_tier(&store.tier_config_mgr(), "WARM").await;
|
||||
let source_set = store.pools[0].get_disks_by_key(object);
|
||||
let target_set = store.pools[1].get_disks_by_key(object);
|
||||
let version_id = uuid::Uuid::new_v4();
|
||||
@@ -1521,17 +1520,14 @@ mod tests {
|
||||
let entry = metacache_entry_from_source(source_set.as_ref(), bucket, object).await;
|
||||
let run_signal_fence = RebalanceRunSignalTestFence::install(REBALANCE_ID);
|
||||
let barrier = TieredMetadataCommitBarrier::install(bucket, object);
|
||||
let mut task = spawn_real_rebalance_entry(
|
||||
let task = spawn_real_rebalance_entry(
|
||||
Arc::clone(&store),
|
||||
Arc::clone(&source_set),
|
||||
entry,
|
||||
REBALANCE_ID,
|
||||
Arc::new(RebalanceBucketConfigs::default()),
|
||||
);
|
||||
tokio::select! {
|
||||
_ = barrier.wait_until_paused() => {}
|
||||
result = &mut task => panic!("rebalance exited before the tiered commit barrier: {result:?}"),
|
||||
}
|
||||
barrier.wait_until_paused().await;
|
||||
run_signal_fence.mark_lost();
|
||||
barrier.release();
|
||||
drop(barrier);
|
||||
|
||||
@@ -101,14 +101,14 @@ pub(crate) trait MigrationBackend: Send + Sync {
|
||||
|
||||
pub(crate) struct RebalanceMigrationBackend<'a> {
|
||||
source: &'a SetDisks,
|
||||
store: std::sync::Arc<ECStore>,
|
||||
store: &'a ECStore,
|
||||
lock_lost_signal: Option<std::sync::Arc<rustfs_lock::distributed_lock::LockLostSignal>>,
|
||||
}
|
||||
|
||||
impl<'a> RebalanceMigrationBackend<'a> {
|
||||
pub(crate) fn new(
|
||||
source: &'a SetDisks,
|
||||
store: std::sync::Arc<ECStore>,
|
||||
store: &'a ECStore,
|
||||
lock_lost_signal: Option<std::sync::Arc<rustfs_lock::distributed_lock::LockLostSignal>>,
|
||||
) -> Self {
|
||||
Self {
|
||||
|
||||
@@ -83,7 +83,6 @@ pub async fn test_store_with_persisted_rebalance_meta(
|
||||
decommission_cancelers: tokio::sync::RwLock::new(vec![None]),
|
||||
start_gate: tokio::sync::Mutex::new(()),
|
||||
pool_meta_save_gate: tokio::sync::Mutex::default(),
|
||||
decommission_capacity_entry_gate: tokio::sync::Mutex::default(),
|
||||
ctx,
|
||||
bucket_fence_registry: std::sync::Arc::default(),
|
||||
});
|
||||
@@ -98,7 +97,7 @@ pub(crate) async fn test_two_pool_stores(
|
||||
std::sync::Arc<crate::store::ECStore>,
|
||||
std::sync::Arc<crate::store::ECStore>,
|
||||
) {
|
||||
test_pool_stores_with_contexts(rebalance_meta, false, 2, 2).await
|
||||
test_two_pool_stores_with_contexts(rebalance_meta, false).await
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
@@ -109,7 +108,7 @@ pub(crate) async fn test_two_pool_stores_with_isolated_node_contexts(
|
||||
std::sync::Arc<crate::store::ECStore>,
|
||||
std::sync::Arc<crate::store::ECStore>,
|
||||
) {
|
||||
test_pool_stores_with_contexts(rebalance_meta, true, 2, 2).await
|
||||
test_two_pool_stores_with_contexts(rebalance_meta, true).await
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
@@ -124,58 +123,26 @@ pub(crate) async fn promote_test_pool_meta_to_v2(store: &std::sync::Arc<crate::s
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
pub(crate) async fn test_three_pool_stores_with_isolated_node_contexts(
|
||||
rebalance_meta: Option<RebalanceMeta>,
|
||||
) -> (
|
||||
Vec<tempfile::TempDir>,
|
||||
std::sync::Arc<crate::store::ECStore>,
|
||||
std::sync::Arc<crate::store::ECStore>,
|
||||
) {
|
||||
test_pool_stores_with_contexts(rebalance_meta, true, 3, 2).await
|
||||
}
|
||||
|
||||
#[cfg(all(test, feature = "test-util"))]
|
||||
pub(crate) async fn test_three_pool_stores_with_three_disk_sets_with_isolated_node_contexts(
|
||||
rebalance_meta: Option<RebalanceMeta>,
|
||||
) -> (
|
||||
Vec<tempfile::TempDir>,
|
||||
std::sync::Arc<crate::store::ECStore>,
|
||||
std::sync::Arc<crate::store::ECStore>,
|
||||
) {
|
||||
test_pool_stores_with_contexts(rebalance_meta, true, 3, 3).await
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
async fn test_pool_stores_with_contexts(
|
||||
async fn test_two_pool_stores_with_contexts(
|
||||
rebalance_meta: Option<RebalanceMeta>,
|
||||
isolate_node_contexts: bool,
|
||||
pool_count: usize,
|
||||
set_drive_count: usize,
|
||||
) -> (
|
||||
Vec<tempfile::TempDir>,
|
||||
std::sync::Arc<crate::store::ECStore>,
|
||||
std::sync::Arc<crate::store::ECStore>,
|
||||
) {
|
||||
crate::services::notification_sys::install_cross_pool_fence_fleet_proof_for_test();
|
||||
use crate::core::pools::{POOL_META_VERSION, PoolMeta, PoolMetaWriteState, persist_pool_meta_identity_for_startup};
|
||||
use crate::core::pools::PoolMeta;
|
||||
use crate::layout::endpoints::{EndpointServerPools, SetupType};
|
||||
|
||||
let ctx = std::sync::Arc::new(crate::runtime::instance::InstanceContext::new());
|
||||
ctx.update_erasure_type(SetupType::DistErasure).await;
|
||||
let deployment_id = uuid::Uuid::new_v4();
|
||||
ctx.set_deployment_id(deployment_id);
|
||||
let mut temp_dirs = Vec::new();
|
||||
let mut pools = Vec::with_capacity(pool_count);
|
||||
for pool_index in 0..pool_count {
|
||||
let (pool_temp_dirs, pool) = crate::core::sets::make_local_two_set_sets_for_pool_with_drive_count_and_ctx(
|
||||
std::sync::Arc::clone(&ctx),
|
||||
pool_index,
|
||||
set_drive_count,
|
||||
)
|
||||
.await;
|
||||
temp_dirs.extend(pool_temp_dirs);
|
||||
pools.push(pool);
|
||||
}
|
||||
let (mut temp_dirs, first_pool) =
|
||||
crate::core::sets::make_local_two_set_sets_for_pool_with_ctx(std::sync::Arc::clone(&ctx), 0).await;
|
||||
let (second_temp_dirs, second_pool) =
|
||||
crate::core::sets::make_local_two_set_sets_for_pool_with_ctx(std::sync::Arc::clone(&ctx), 1).await;
|
||||
temp_dirs.extend(second_temp_dirs);
|
||||
let pools = vec![first_pool, second_pool];
|
||||
{
|
||||
let local_disk_map = ctx.local_disk_map();
|
||||
let mut local_disk_map = local_disk_map.write().await;
|
||||
@@ -187,24 +154,11 @@ async fn test_pool_stores_with_contexts(
|
||||
}
|
||||
}
|
||||
}
|
||||
let mut pool_meta = PoolMeta::new(&pools, &PoolMeta::default());
|
||||
pool_meta.version = POOL_META_VERSION;
|
||||
let pool_meta = PoolMeta::new(&pools, &PoolMeta::default());
|
||||
pool_meta
|
||||
.save_for_startup(pools.clone())
|
||||
.await
|
||||
.expect("baseline pool metadata should be persisted");
|
||||
let mut pool_meta_write_state = PoolMetaWriteState::for_startup(deployment_id, true);
|
||||
persist_pool_meta_identity_for_startup(pools.clone(), &mut pool_meta_write_state, false)
|
||||
.await
|
||||
.expect("pending pool metadata identity should be persisted");
|
||||
let replica_state = pool_meta
|
||||
.load_no_lock_from_replicas_observing(pools.clone(), &mut pool_meta_write_state)
|
||||
.await
|
||||
.expect("baseline pool metadata should remain readable");
|
||||
pool_meta_write_state.observe_replicas(replica_state);
|
||||
persist_pool_meta_identity_for_startup(pools.clone(), &mut pool_meta_write_state, true)
|
||||
.await
|
||||
.expect("initialized pool metadata identity should be persisted");
|
||||
if let Some(meta) = rebalance_meta.as_ref() {
|
||||
meta.save(pools[0].clone())
|
||||
.await
|
||||
@@ -215,7 +169,6 @@ async fn test_pool_stores_with_contexts(
|
||||
let other_ctx = if isolate_node_contexts {
|
||||
let other_ctx = std::sync::Arc::new(crate::runtime::instance::InstanceContext::new());
|
||||
other_ctx.update_erasure_type(SetupType::DistErasure).await;
|
||||
other_ctx.set_deployment_id(deployment_id);
|
||||
*other_ctx.local_disk_map().write().await = ctx.local_disk_map().read().await.clone();
|
||||
other_ctx.set_endpoints(endpoint_pools.clone());
|
||||
other_ctx
|
||||
@@ -230,10 +183,9 @@ async fn test_pool_stores_with_contexts(
|
||||
peer_sys: crate::cluster::rpc::S3PeerSys::new_with_instance_ctx(&endpoint_pools, std::sync::Arc::clone(&store_ctx)),
|
||||
pool_meta: tokio::sync::RwLock::new(pool_meta.clone()),
|
||||
rebalance_meta: tokio::sync::RwLock::new(rebalance_meta.clone()),
|
||||
decommission_cancelers: tokio::sync::RwLock::new(vec![None; pool_count]),
|
||||
decommission_cancelers: tokio::sync::RwLock::new(vec![None, None]),
|
||||
start_gate: tokio::sync::Mutex::new(()),
|
||||
pool_meta_save_gate: tokio::sync::Mutex::new(pool_meta_write_state.independent_clone_for_test()),
|
||||
decommission_capacity_entry_gate: tokio::sync::Mutex::default(),
|
||||
pool_meta_save_gate: tokio::sync::Mutex::default(),
|
||||
ctx: store_ctx,
|
||||
bucket_fence_registry: std::sync::Arc::default(),
|
||||
})
|
||||
@@ -243,8 +195,6 @@ async fn test_pool_stores_with_contexts(
|
||||
if isolate_node_contexts {
|
||||
crate::bucket::metadata_sys::init_bucket_metadata_sys(std::sync::Arc::clone(&store), Vec::new()).await;
|
||||
crate::bucket::metadata_sys::init_bucket_metadata_sys(std::sync::Arc::clone(&other_store), Vec::new()).await;
|
||||
} else {
|
||||
crate::bucket::metadata_sys::init_bucket_metadata_sys(std::sync::Arc::clone(&store), Vec::new()).await;
|
||||
}
|
||||
(temp_dirs, store, other_store)
|
||||
}
|
||||
|
||||
@@ -3009,7 +3009,6 @@ fn test_store_with_rebalance_meta(meta: RebalanceMeta) -> Arc<crate::store::ECSt
|
||||
decommission_cancelers: tokio::sync::RwLock::new(Vec::new()),
|
||||
start_gate: tokio::sync::Mutex::new(()),
|
||||
pool_meta_save_gate: tokio::sync::Mutex::default(),
|
||||
decommission_capacity_entry_gate: tokio::sync::Mutex::default(),
|
||||
ctx: crate::runtime::instance::bootstrap_ctx(),
|
||||
bucket_fence_registry: std::sync::Arc::default(),
|
||||
})
|
||||
|
||||
@@ -161,7 +161,6 @@ impl ECStore {
|
||||
|
||||
let cancel_tx = CancellationToken::new();
|
||||
let rx = cancel_tx.clone();
|
||||
let activation_at = self.next_scanner_data_movement_update(OffsetDateTime::now_utc()).await;
|
||||
let activation_outcome;
|
||||
let candidate;
|
||||
let expected_cancel;
|
||||
@@ -186,8 +185,12 @@ impl ECStore {
|
||||
return Ok(false);
|
||||
}
|
||||
expected_cancel = meta.cancel.clone();
|
||||
(candidate, activation_outcome, must_persist) =
|
||||
stage_local_rebalance_worker_activation(meta, expected_id.as_ref(), cancel_tx.clone(), activation_at)?;
|
||||
(candidate, activation_outcome, must_persist) = stage_local_rebalance_worker_activation(
|
||||
meta,
|
||||
expected_id.as_ref(),
|
||||
cancel_tx.clone(),
|
||||
OffsetDateTime::now_utc(),
|
||||
)?;
|
||||
if let Err(err) = activation_fence.ensure_held() {
|
||||
cancel_tx.cancel();
|
||||
return Err(err);
|
||||
@@ -381,11 +384,11 @@ impl ECStore {
|
||||
tokio::select! {
|
||||
result = done_rx.recv() => {
|
||||
quit = true;
|
||||
let now = OffsetDateTime::now_utc();
|
||||
let terminal_event = classify_rebalance_terminal_event(result, now);
|
||||
msg = terminal_event.message().to_string();
|
||||
let movement_gate = store.ctx.data_movement_operation_gate();
|
||||
let movement_guard = movement_gate.write().await;
|
||||
let terminal_at = store.next_scanner_data_movement_update(OffsetDateTime::now_utc()).await;
|
||||
let terminal_event = classify_rebalance_terminal_event(result, terminal_at);
|
||||
msg = terminal_event.message().to_string();
|
||||
let previous_meta = store.rebalance_meta.read().await.clone();
|
||||
let terminal_state_present = {
|
||||
let mut rebalance_meta = store.rebalance_meta.write().await;
|
||||
@@ -402,7 +405,7 @@ impl ECStore {
|
||||
{
|
||||
pool_stat.info.stopping = false;
|
||||
pool_stat.info.status = RebalStatus::Failed;
|
||||
pool_stat.info.end_time = Some(terminal_at);
|
||||
pool_stat.info.end_time = Some(now);
|
||||
pool_stat.info.last_error = Some(
|
||||
pool_stat
|
||||
.cleanup_warnings
|
||||
@@ -430,7 +433,7 @@ impl ECStore {
|
||||
&mut pool_stat.info.end_time,
|
||||
&mut pool_stat.info.last_error,
|
||||
terminal_event,
|
||||
terminal_at,
|
||||
now,
|
||||
);
|
||||
}
|
||||
true
|
||||
@@ -832,10 +835,6 @@ impl ECStore {
|
||||
opt: RebalSaveOpt,
|
||||
expected_id: Option<&str>,
|
||||
) -> Result<()> {
|
||||
let now = match opt {
|
||||
RebalSaveOpt::Stats => OffsetDateTime::now_utc(),
|
||||
RebalSaveOpt::StoppedAt => self.next_scanner_data_movement_update(OffsetDateTime::now_utc()).await,
|
||||
};
|
||||
let meta_to_save = {
|
||||
let mut rebalance_meta = self.rebalance_meta.write().await;
|
||||
if let Some(expected_id) = expected_id {
|
||||
@@ -845,6 +844,7 @@ impl ECStore {
|
||||
return Ok(());
|
||||
};
|
||||
|
||||
let now = OffsetDateTime::now_utc();
|
||||
apply_rebalance_save_option(meta, pool_idx, opt, now);
|
||||
meta.clone()
|
||||
};
|
||||
|
||||
@@ -12,7 +12,7 @@
|
||||
// See the License for the specific language governing permissions and
|
||||
// limitations under the License.
|
||||
|
||||
#[cfg(any(test, feature = "test-util"))]
|
||||
#[cfg(feature = "test-util")]
|
||||
pub mod test_util;
|
||||
pub mod tier;
|
||||
pub mod tier_admin;
|
||||
|
||||
File diff suppressed because it is too large
Load Diff
@@ -12,7 +12,7 @@
|
||||
// See the License for the specific language governing permissions and
|
||||
// limitations under the License.
|
||||
|
||||
use std::sync::{Arc, LazyLock};
|
||||
use std::sync::Arc;
|
||||
|
||||
use rustfs_utils::crypto::{hex_sha256, is_sha256_checksum};
|
||||
use serde::{Deserialize, Serialize};
|
||||
@@ -32,34 +32,9 @@ pub(crate) const TIER_MUTATION_INTENT_SCHEMA: &str = "rustfs-tier-mutation-inten
|
||||
pub(crate) const MAX_TIER_MUTATION_INTENT_SIZE: usize = rustfs_protos::TIER_MUTATION_RPC_MAX_PREPARE_PAYLOAD_SIZE;
|
||||
pub(crate) const TIER_MUTATION_INTENT_RECORD_PREFIX: &str = "tier/mutation-intents/records";
|
||||
pub(crate) const TIER_COORDINATOR_MUTATION_INTENT_RECORD_PREFIX: &str = "tier/mutation-intents/coordinators";
|
||||
pub(crate) const TIER_MUTATION_MUTEX_SHARDS: usize = 64;
|
||||
const TIER_MUTATION_INTENT_ADVANCE_CAS_ATTEMPTS: usize = 3;
|
||||
pub(crate) type TierMutationDigest = [u8; 32];
|
||||
|
||||
static TIER_MUTATION_MUTEXES: LazyLock<[tokio::sync::Mutex<()>; TIER_MUTATION_MUTEX_SHARDS]> =
|
||||
LazyLock::new(|| std::array::from_fn(|_| tokio::sync::Mutex::new(())));
|
||||
|
||||
/// Serializes every local phase and recovery action for one mutation id while
|
||||
/// retaining bounded parallelism for unrelated mutations.
|
||||
pub(crate) async fn acquire_tier_mutation_mutex(mutation_id: Uuid) -> tokio::sync::MutexGuard<'static, ()> {
|
||||
TIER_MUTATION_MUTEXES[tier_mutation_mutex_shard_index(mutation_id)]
|
||||
.lock()
|
||||
.await
|
||||
}
|
||||
|
||||
fn tier_mutation_mutex_shard_index(mutation_id: Uuid) -> usize {
|
||||
let raw = mutation_id.as_u128();
|
||||
let mut mixed = (raw as u64) ^ ((raw >> 64) as u64);
|
||||
// MurmurHash3's 64-bit finalizer gives stable diffusion without allocating
|
||||
// or relying on RandomState, whose seed differs between processes.
|
||||
mixed ^= mixed >> 33;
|
||||
mixed = mixed.wrapping_mul(0xff51_afd7_ed55_8ccd);
|
||||
mixed ^= mixed >> 33;
|
||||
mixed = mixed.wrapping_mul(0xc4ce_b9fe_1a85_ec53);
|
||||
mixed ^= mixed >> 33;
|
||||
(mixed as usize) & (TIER_MUTATION_MUTEX_SHARDS - 1)
|
||||
}
|
||||
|
||||
pub(crate) type Result<T> = std::result::Result<T, TierMutationIntentError>;
|
||||
|
||||
#[derive(Debug, thiserror::Error)]
|
||||
@@ -290,30 +265,6 @@ impl TierMutationIntent {
|
||||
&& self.expires_at_unix_nanos == other.expires_at_unix_nanos
|
||||
}
|
||||
|
||||
/// Reconstruct the exact Prepared record that originally produced this
|
||||
/// intent. Abort RPCs are identity-bound to that payload; serializing an
|
||||
/// Aborted terminal record would both violate the wire contract and use a
|
||||
/// different revision if a missing peer has to persist a tombstone.
|
||||
pub(crate) fn original_prepared(&self) -> Result<Self> {
|
||||
if self.state == TierMutationIntentState::Prepared {
|
||||
self.validate()?;
|
||||
return Ok(self.clone());
|
||||
}
|
||||
let mut prepared = self.clone();
|
||||
prepared.revision =
|
||||
prepared
|
||||
.revision
|
||||
.checked_sub(1)
|
||||
.filter(|revision| *revision != 0)
|
||||
.ok_or(TierMutationIntentError::Corrupt(
|
||||
"terminal intent cannot reconstruct its prepared revision",
|
||||
))?;
|
||||
prepared.state = TierMutationIntentState::Prepared;
|
||||
prepared.committed_config_etag = None;
|
||||
prepared.validate()?;
|
||||
Ok(prepared)
|
||||
}
|
||||
|
||||
pub(crate) fn encode(&self) -> Result<Vec<u8>> {
|
||||
self.validate()?;
|
||||
let intent_bytes = serde_json::to_vec(self)?;
|
||||
@@ -488,16 +439,6 @@ where
|
||||
load_tier_mutation_intent_record_with_etag_at_prefix(api, TIER_MUTATION_INTENT_RECORD_PREFIX, mutation_id).await
|
||||
}
|
||||
|
||||
pub(crate) async fn load_tier_coordinator_mutation_intent_record_with_etag<S>(
|
||||
api: Arc<S>,
|
||||
mutation_id: Uuid,
|
||||
) -> EcstoreResult<(TierMutationIntent, String)>
|
||||
where
|
||||
S: EcstoreObjectIO,
|
||||
{
|
||||
load_tier_mutation_intent_record_with_etag_at_prefix(api, TIER_COORDINATOR_MUTATION_INTENT_RECORD_PREFIX, mutation_id).await
|
||||
}
|
||||
|
||||
async fn load_tier_mutation_intent_record_with_etag_at_prefix<S>(
|
||||
api: Arc<S>,
|
||||
prefix: &str,
|
||||
@@ -566,7 +507,6 @@ where
|
||||
.await
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
pub(crate) async fn delete_tier_mutation_intent_record<S>(api: Arc<S>, mutation_id: Uuid) -> EcstoreResult<()>
|
||||
where
|
||||
S: EcstoreObjectOperations,
|
||||
@@ -574,36 +514,13 @@ where
|
||||
delete_tier_mutation_intent_record_with_prefix(api, TIER_MUTATION_INTENT_RECORD_PREFIX, mutation_id).await
|
||||
}
|
||||
|
||||
pub(crate) async fn delete_tier_mutation_intent_record_if_current<S>(
|
||||
api: Arc<S>,
|
||||
mutation_id: Uuid,
|
||||
current_etag: &str,
|
||||
) -> EcstoreResult<()>
|
||||
pub(crate) async fn delete_tier_coordinator_mutation_intent_record<S>(api: Arc<S>, mutation_id: Uuid) -> EcstoreResult<()>
|
||||
where
|
||||
S: EcstoreObjectOperations,
|
||||
{
|
||||
delete_tier_mutation_intent_record_if_current_with_prefix(api, TIER_MUTATION_INTENT_RECORD_PREFIX, mutation_id, current_etag)
|
||||
.await
|
||||
delete_tier_mutation_intent_record_with_prefix(api, TIER_COORDINATOR_MUTATION_INTENT_RECORD_PREFIX, mutation_id).await
|
||||
}
|
||||
|
||||
pub(crate) async fn delete_tier_coordinator_mutation_intent_record_if_current<S>(
|
||||
api: Arc<S>,
|
||||
mutation_id: Uuid,
|
||||
current_etag: &str,
|
||||
) -> EcstoreResult<()>
|
||||
where
|
||||
S: EcstoreObjectOperations,
|
||||
{
|
||||
delete_tier_mutation_intent_record_if_current_with_prefix(
|
||||
api,
|
||||
TIER_COORDINATOR_MUTATION_INTENT_RECORD_PREFIX,
|
||||
mutation_id,
|
||||
current_etag,
|
||||
)
|
||||
.await
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
async fn delete_tier_mutation_intent_record_with_prefix<S>(api: Arc<S>, prefix: &str, mutation_id: Uuid) -> EcstoreResult<()>
|
||||
where
|
||||
S: EcstoreObjectOperations,
|
||||
@@ -616,40 +533,6 @@ where
|
||||
}
|
||||
}
|
||||
|
||||
async fn delete_tier_mutation_intent_record_if_current_with_prefix<S>(
|
||||
api: Arc<S>,
|
||||
prefix: &str,
|
||||
mutation_id: Uuid,
|
||||
current_etag: &str,
|
||||
) -> EcstoreResult<()>
|
||||
where
|
||||
S: EcstoreObjectOperations,
|
||||
{
|
||||
if current_etag.trim().is_empty() {
|
||||
return Err(Error::other("tier mutation intent current ETag is empty"));
|
||||
}
|
||||
let object =
|
||||
tier_mutation_intent_record_object_name_with_prefix(prefix, mutation_id).map_err(tier_mutation_intent_store_error)?;
|
||||
match api
|
||||
.delete_object(
|
||||
RUSTFS_META_BUCKET,
|
||||
&object,
|
||||
ObjectOptions {
|
||||
http_preconditions: Some(HTTPPreconditions {
|
||||
if_match: Some(current_etag.to_string()),
|
||||
..Default::default()
|
||||
}),
|
||||
..Default::default()
|
||||
},
|
||||
)
|
||||
.await
|
||||
{
|
||||
Ok(_) => Ok(()),
|
||||
Err(err) if err == Error::FileNotFound || matches!(err, Error::ObjectNotFound(_, _)) => Err(Error::ConfigNotFound),
|
||||
Err(err) => Err(err),
|
||||
}
|
||||
}
|
||||
|
||||
pub(crate) async fn advance_tier_mutation_intent_record_idempotent<S>(
|
||||
api: Arc<S>,
|
||||
mutation_id: Uuid,
|
||||
@@ -830,7 +713,6 @@ fn digest_is_empty(digest: &TierMutationDigest) -> bool {
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use super::*;
|
||||
use std::time::Duration;
|
||||
|
||||
const OLD_IDENTITY: TierDestinationId = [1; 32];
|
||||
const NEW_IDENTITY: TierDestinationId = [2; 32];
|
||||
@@ -854,61 +736,6 @@ mod tests {
|
||||
}
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn mutation_mutex_uses_exactly_64_stable_shards() {
|
||||
assert_eq!(TIER_MUTATION_MUTEX_SHARDS, 64);
|
||||
assert_eq!(TIER_MUTATION_MUTEXES.len(), TIER_MUTATION_MUTEX_SHARDS);
|
||||
|
||||
let mutation_id = Uuid::parse_str("36e2220e-9ad2-495b-b3bc-c4d2caf70a31").expect("fixture uuid should parse");
|
||||
let shard = tier_mutation_mutex_shard_index(mutation_id);
|
||||
assert!(shard < TIER_MUTATION_MUTEX_SHARDS);
|
||||
assert_eq!(shard, tier_mutation_mutex_shard_index(mutation_id));
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
async fn mutation_mutex_serializes_the_same_id() {
|
||||
let mutation_id = Uuid::new_v4();
|
||||
let first = acquire_tier_mutation_mutex(mutation_id).await;
|
||||
let (started_tx, started_rx) = tokio::sync::oneshot::channel();
|
||||
let (acquired_tx, mut acquired_rx) = tokio::sync::oneshot::channel();
|
||||
|
||||
let waiter = tokio::spawn(async move {
|
||||
started_tx.send(()).expect("test receiver should remain alive");
|
||||
let _second = acquire_tier_mutation_mutex(mutation_id).await;
|
||||
acquired_tx.send(()).expect("test receiver should remain alive");
|
||||
});
|
||||
started_rx.await.expect("waiter should start");
|
||||
assert!(
|
||||
tokio::time::timeout(Duration::from_millis(25), &mut acquired_rx)
|
||||
.await
|
||||
.is_err(),
|
||||
"the same mutation id must not enter concurrently"
|
||||
);
|
||||
|
||||
drop(first);
|
||||
tokio::time::timeout(Duration::from_secs(1), &mut acquired_rx)
|
||||
.await
|
||||
.expect("waiter should acquire after release")
|
||||
.expect("waiter should report acquisition");
|
||||
waiter.await.expect("waiter task should finish");
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
async fn mutation_mutex_allows_different_shards_to_progress() {
|
||||
let first_id = Uuid::new_v4();
|
||||
let first_shard = tier_mutation_mutex_shard_index(first_id);
|
||||
let second_id = (0..1024)
|
||||
.map(|_| Uuid::new_v4())
|
||||
.find(|candidate| tier_mutation_mutex_shard_index(*candidate) != first_shard)
|
||||
.expect("a distinct shard should be easy to find");
|
||||
let first = acquire_tier_mutation_mutex(first_id).await;
|
||||
|
||||
let _second = tokio::time::timeout(Duration::from_secs(1), acquire_tier_mutation_mutex(second_id))
|
||||
.await
|
||||
.expect("a different shard must not wait for the first mutation");
|
||||
drop(first);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn intent_round_trip_preserves_committed_state() {
|
||||
let mut intent = prepared_intent();
|
||||
@@ -1046,37 +873,6 @@ mod tests {
|
||||
));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn terminal_intent_reconstructs_original_prepared_abort_payload() {
|
||||
for terminal in [TierMutationIntentState::Aborted, TierMutationIntentState::Committed] {
|
||||
let original = prepared_intent();
|
||||
let mut intent = original.clone();
|
||||
let committed_etag = (terminal == TierMutationIntentState::Committed).then(|| "new-etag".to_string());
|
||||
intent
|
||||
.advance(terminal, committed_etag)
|
||||
.expect("terminal transition should succeed");
|
||||
|
||||
let reconstructed = intent
|
||||
.original_prepared()
|
||||
.expect("terminal record should recover prepared payload");
|
||||
assert_eq!(reconstructed, original);
|
||||
assert!(intent.same_identity_as(&reconstructed));
|
||||
}
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn terminal_intent_with_initial_revision_fails_prepared_reconstruction() {
|
||||
let mut corrupt = prepared_intent();
|
||||
corrupt.state = TierMutationIntentState::Aborted;
|
||||
|
||||
assert!(matches!(
|
||||
corrupt.original_prepared(),
|
||||
Err(TierMutationIntentError::Corrupt(
|
||||
"terminal intent cannot reconstruct its prepared revision"
|
||||
))
|
||||
));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn intent_validation_rejects_placeholder_identity() {
|
||||
let mut intent = prepared_intent();
|
||||
|
||||
@@ -15,13 +15,12 @@
|
||||
use std::sync::Arc;
|
||||
|
||||
use rustfs_protos::{TIER_MUTATION_RPC_PROTOCOL_VERSION, TierMutationRpcPhase};
|
||||
use time::OffsetDateTime;
|
||||
use uuid::Uuid;
|
||||
|
||||
use super::tier::{TierConfigMgr, tier_config_abort_matches, tier_config_commit_matches, tier_config_etag_matches};
|
||||
use super::tier_mutation_intent::{
|
||||
MAX_TIER_MUTATION_INTENT_SIZE, TierMutationIntent, TierMutationIntentState, acquire_tier_mutation_mutex,
|
||||
advance_tier_mutation_intent_record_idempotent, load_tier_mutation_intent_record, save_tier_mutation_intent_record_if_absent,
|
||||
MAX_TIER_MUTATION_INTENT_SIZE, TierMutationIntent, TierMutationIntentState, advance_tier_mutation_intent_record_idempotent,
|
||||
load_tier_mutation_intent_record, save_tier_mutation_intent_record_if_absent,
|
||||
};
|
||||
use crate::error::{Error, StorageError};
|
||||
use crate::store::ECStore;
|
||||
@@ -58,8 +57,6 @@ pub enum TierMutationPeerError {
|
||||
CommitProofMismatch,
|
||||
#[error("tier mutation peer abort proof does not match the persisted tier configuration")]
|
||||
AbortProofMismatch,
|
||||
#[error("tier mutation peer prepared intent has expired")]
|
||||
ExpiredIntent,
|
||||
#[error("tier mutation peer runtime error: {0}")]
|
||||
Runtime(#[source] AdminError),
|
||||
#[error("tier mutation peer store error: {0}")]
|
||||
@@ -82,7 +79,6 @@ pub async fn handle_tier_mutation_peer_request(
|
||||
canonical_payload: &[u8],
|
||||
) -> TierMutationPeerResult<TierMutationPeerOutcome> {
|
||||
validate_peer_request_envelope(protocol_version, mutation_id, canonical_payload)?;
|
||||
let _mutation_guard = acquire_tier_mutation_mutex(mutation_id).await;
|
||||
match phase {
|
||||
TierMutationRpcPhase::Prepare => handle_prepare(api, mutation_id, canonical_payload).await,
|
||||
TierMutationRpcPhase::Commit => handle_commit(api, mutation_id, canonical_payload).await,
|
||||
@@ -107,57 +103,43 @@ async fn handle_prepare(
|
||||
}
|
||||
let tier_config_mgr = api.tier_config_mgr();
|
||||
|
||||
for _ in 0..3 {
|
||||
let (stored, applied) = match load_tier_mutation_intent_record(api.clone(), mutation_id).await {
|
||||
Ok(existing) => {
|
||||
if !existing.same_identity_as(&intent) {
|
||||
return Err(TierMutationPeerError::ConflictingIntent);
|
||||
}
|
||||
(existing, false)
|
||||
}
|
||||
Err(Error::ConfigNotFound) => {
|
||||
let now = i64::try_from(OffsetDateTime::now_utc().unix_timestamp_nanos()).unwrap_or(i64::MAX);
|
||||
if intent.expires_at_unix_nanos <= now {
|
||||
return Err(TierMutationPeerError::ExpiredIntent);
|
||||
}
|
||||
match save_tier_mutation_intent_record_if_absent(api.clone(), &intent).await {
|
||||
Ok(()) => (intent.clone(), true),
|
||||
Err(Error::PreconditionFailed) => continue,
|
||||
Err(err) => return Err(err.into()),
|
||||
}
|
||||
}
|
||||
Err(err) => return Err(err.into()),
|
||||
};
|
||||
|
||||
match stored.state {
|
||||
TierMutationIntentState::Prepared => {
|
||||
TierConfigMgr::apply_prepared_mutation_intent_block(&tier_config_mgr, &stored)
|
||||
.await
|
||||
.map_err(TierMutationPeerError::Runtime)?;
|
||||
TierConfigMgr::wait_for_blocked_tier_operation_leases(&tier_config_mgr, &stored)
|
||||
.await
|
||||
.map_err(TierMutationPeerError::Runtime)?;
|
||||
}
|
||||
TierMutationIntentState::Committed => {
|
||||
TierConfigMgr::apply_committed_mutation_intent_block(&tier_config_mgr, &stored)
|
||||
.await
|
||||
.map_err(TierMutationPeerError::Runtime)?;
|
||||
}
|
||||
TierMutationIntentState::Aborted => {
|
||||
TierConfigMgr::clear_prepared_mutation_intent_block(&tier_config_mgr, mutation_id)
|
||||
.await
|
||||
.map_err(TierMutationPeerError::Runtime)?;
|
||||
TierConfigMgr::request_committed_mutation_refresh(&tier_config_mgr).await;
|
||||
}
|
||||
match save_tier_mutation_intent_record_if_absent(api.clone(), &intent).await {
|
||||
Ok(()) => {
|
||||
TierConfigMgr::apply_prepared_mutation_intent_block(&tier_config_mgr, &intent)
|
||||
.await
|
||||
.map_err(TierMutationPeerError::Runtime)?;
|
||||
Ok(TierMutationPeerOutcome {
|
||||
state: TierMutationPeerState::Prepared,
|
||||
applied: true,
|
||||
})
|
||||
}
|
||||
return Ok(TierMutationPeerOutcome {
|
||||
state: peer_state_from_intent(stored.state),
|
||||
applied,
|
||||
});
|
||||
Err(Error::PreconditionFailed) => {
|
||||
let existing = load_tier_mutation_intent_record(api, mutation_id).await?;
|
||||
if !existing.same_identity_as(&intent) {
|
||||
return Err(TierMutationPeerError::ConflictingIntent);
|
||||
}
|
||||
match existing.state {
|
||||
TierMutationIntentState::Prepared => {
|
||||
TierConfigMgr::apply_prepared_mutation_intent_block(&tier_config_mgr, &existing)
|
||||
.await
|
||||
.map_err(TierMutationPeerError::Runtime)?;
|
||||
}
|
||||
TierMutationIntentState::Committed => {
|
||||
TierConfigMgr::apply_committed_mutation_intent_block(&tier_config_mgr, &existing)
|
||||
.await
|
||||
.map_err(TierMutationPeerError::Runtime)?;
|
||||
}
|
||||
TierMutationIntentState::Aborted => {
|
||||
TierConfigMgr::request_committed_mutation_refresh(&tier_config_mgr).await;
|
||||
}
|
||||
}
|
||||
Ok(TierMutationPeerOutcome {
|
||||
state: peer_state_from_intent(existing.state),
|
||||
applied: false,
|
||||
})
|
||||
}
|
||||
Err(err) => Err(err.into()),
|
||||
}
|
||||
Err(TierMutationPeerError::Store(Error::other(
|
||||
"tier mutation prepare raced repeatedly with another decision",
|
||||
)))
|
||||
}
|
||||
|
||||
async fn handle_commit(
|
||||
@@ -219,106 +201,26 @@ async fn handle_abort(
|
||||
mutation_id: Uuid,
|
||||
canonical_payload: &[u8],
|
||||
) -> TierMutationPeerResult<TierMutationPeerOutcome> {
|
||||
let prepared = TierMutationIntent::decode(mutation_id, canonical_payload)
|
||||
.map_err(|err| TierMutationPeerError::InvalidPayload(err.to_string()))?;
|
||||
if prepared.state != TierMutationIntentState::Prepared {
|
||||
return Err(TierMutationPeerError::InvalidPayload(
|
||||
"abort payload must carry the original prepared intent".to_string(),
|
||||
));
|
||||
if !canonical_payload.is_empty() {
|
||||
return Err(TierMutationPeerError::InvalidPayload("abort payload must be empty".to_string()));
|
||||
}
|
||||
let mut tombstone = prepared.clone();
|
||||
tombstone
|
||||
.advance(TierMutationIntentState::Aborted, None)
|
||||
.map_err(|err| TierMutationPeerError::InvalidPayload(err.to_string()))?;
|
||||
|
||||
for _ in 0..3 {
|
||||
match load_tier_mutation_intent_record(api.clone(), mutation_id).await {
|
||||
Ok(existing) => {
|
||||
if !existing.same_identity_as(&prepared) {
|
||||
return Err(TierMutationPeerError::ConflictingIntent);
|
||||
}
|
||||
match existing.state {
|
||||
TierMutationIntentState::Committed => {
|
||||
return Ok(TierMutationPeerOutcome {
|
||||
state: TierMutationPeerState::Committed,
|
||||
applied: false,
|
||||
});
|
||||
}
|
||||
TierMutationIntentState::Aborted => {
|
||||
TierConfigMgr::clear_prepared_mutation_intent_block(&api.tier_config_mgr(), mutation_id)
|
||||
.await
|
||||
.map_err(TierMutationPeerError::Runtime)?;
|
||||
TierConfigMgr::request_committed_mutation_refresh(&api.tier_config_mgr()).await;
|
||||
return Ok(TierMutationPeerOutcome {
|
||||
state: TierMutationPeerState::Aborted,
|
||||
applied: false,
|
||||
});
|
||||
}
|
||||
TierMutationIntentState::Prepared => {}
|
||||
}
|
||||
if !tier_config_abort_matches(api.clone(), &prepared)
|
||||
.await
|
||||
.map_err(Error::other)?
|
||||
{
|
||||
return Err(TierMutationPeerError::AbortProofMismatch);
|
||||
}
|
||||
let advanced = advance_tier_mutation_intent_record_idempotent(
|
||||
api.clone(),
|
||||
mutation_id,
|
||||
TierMutationIntentState::Aborted,
|
||||
None,
|
||||
)
|
||||
.await;
|
||||
let (intent, applied) = match advanced {
|
||||
Ok(result) => result,
|
||||
Err(err) => match load_tier_mutation_intent_record(api.clone(), mutation_id).await {
|
||||
Ok(current)
|
||||
if current.same_identity_as(&prepared) && current.state != TierMutationIntentState::Prepared =>
|
||||
{
|
||||
(current, false)
|
||||
}
|
||||
_ => return Err(err.into()),
|
||||
},
|
||||
};
|
||||
if intent.state == TierMutationIntentState::Aborted {
|
||||
TierConfigMgr::request_committed_mutation_refresh(&api.tier_config_mgr()).await;
|
||||
TierConfigMgr::clear_prepared_mutation_intent_block(&api.tier_config_mgr(), mutation_id)
|
||||
.await
|
||||
.map_err(TierMutationPeerError::Runtime)?;
|
||||
}
|
||||
return Ok(TierMutationPeerOutcome {
|
||||
state: peer_state_from_intent(intent.state),
|
||||
applied,
|
||||
});
|
||||
}
|
||||
Err(Error::ConfigNotFound) => {
|
||||
if !tier_config_abort_matches(api.clone(), &prepared)
|
||||
.await
|
||||
.map_err(Error::other)?
|
||||
{
|
||||
return Err(TierMutationPeerError::AbortProofMismatch);
|
||||
}
|
||||
match save_tier_mutation_intent_record_if_absent(api.clone(), &tombstone).await {
|
||||
Ok(()) => {
|
||||
TierConfigMgr::clear_prepared_mutation_intent_block(&api.tier_config_mgr(), mutation_id)
|
||||
.await
|
||||
.map_err(TierMutationPeerError::Runtime)?;
|
||||
TierConfigMgr::request_committed_mutation_refresh(&api.tier_config_mgr()).await;
|
||||
return Ok(TierMutationPeerOutcome {
|
||||
state: TierMutationPeerState::Aborted,
|
||||
applied: true,
|
||||
});
|
||||
}
|
||||
Err(Error::PreconditionFailed) => continue,
|
||||
Err(err) => return Err(err.into()),
|
||||
}
|
||||
}
|
||||
Err(err) => return Err(err.into()),
|
||||
}
|
||||
let existing = load_tier_mutation_intent_record(api.clone(), mutation_id).await?;
|
||||
if existing.state == TierMutationIntentState::Prepared
|
||||
&& !tier_config_abort_matches(api.clone(), &existing)
|
||||
.await
|
||||
.map_err(Error::other)?
|
||||
{
|
||||
return Err(TierMutationPeerError::AbortProofMismatch);
|
||||
}
|
||||
Err(TierMutationPeerError::Store(Error::other(
|
||||
"tier mutation abort raced repeatedly with prepare",
|
||||
)))
|
||||
let (intent, applied) =
|
||||
advance_tier_mutation_intent_record_idempotent(api.clone(), mutation_id, TierMutationIntentState::Aborted, None).await?;
|
||||
if intent.state == TierMutationIntentState::Aborted {
|
||||
TierConfigMgr::request_committed_mutation_refresh(&api.tier_config_mgr()).await;
|
||||
}
|
||||
Ok(TierMutationPeerOutcome {
|
||||
state: peer_state_from_intent(intent.state),
|
||||
applied,
|
||||
})
|
||||
}
|
||||
|
||||
fn validate_peer_request_envelope(
|
||||
@@ -326,10 +228,7 @@ fn validate_peer_request_envelope(
|
||||
mutation_id: Uuid,
|
||||
canonical_payload: &[u8],
|
||||
) -> TierMutationPeerResult<()> {
|
||||
if !matches!(
|
||||
protocol_version,
|
||||
rustfs_protos::TIER_MUTATION_RPC_PREVIOUS_PROTOCOL_VERSION | TIER_MUTATION_RPC_PROTOCOL_VERSION
|
||||
) {
|
||||
if protocol_version != TIER_MUTATION_RPC_PROTOCOL_VERSION {
|
||||
return Err(TierMutationPeerError::UnsupportedProtocolVersion(protocol_version));
|
||||
}
|
||||
if mutation_id.is_nil() {
|
||||
@@ -377,8 +276,6 @@ mod tests {
|
||||
#[test]
|
||||
fn peer_request_envelope_fails_closed_on_old_version_nil_id_and_large_payload() {
|
||||
let mutation_id = Uuid::new_v4();
|
||||
validate_peer_request_envelope(rustfs_protos::TIER_MUTATION_RPC_PREVIOUS_PROTOCOL_VERSION, mutation_id, b"payload")
|
||||
.expect("v3 must remain accepted during the v4 rollout");
|
||||
assert!(matches!(
|
||||
validate_peer_request_envelope(TIER_MUTATION_RPC_PROTOCOL_VERSION + 1, mutation_id, b"payload"),
|
||||
Err(TierMutationPeerError::UnsupportedProtocolVersion(_))
|
||||
|
||||
@@ -41,22 +41,21 @@ use super::super::ENV_RUSTFS_GET_METADATA_TWO_PHASE_READ_PLAN_ENABLE;
|
||||
#[cfg(test)]
|
||||
use super::super::get_metadata_slowtail_fault_delay;
|
||||
use super::super::{
|
||||
Bytes, CHECK_PART_DISK_NOT_FOUND, DeleteOptions, DiskError, DiskStore, EVENT_SET_DISK_ORPHAN_PURGE_SKIPPED,
|
||||
EVENT_SET_DISK_RENAME_TAIL_DRAIN_FAILED, EVENT_SET_DISK_WRITE, Error, FileInfo, FileMeta, FileMetaShallowVersion,
|
||||
GetCodecStreamingFallbackReason, GetObjectMetadataCacheEntry, HTTPPreconditions, HashAlgorithm, HealAdmissionResult,
|
||||
HealChannelPriority, HealRequestSource, LOG_COMPONENT_ECSTORE, LOG_SUBSYSTEM_SET_DISK, MultipartWriteQuorumContext,
|
||||
OBJECT_OP_IGNORED_ERRS, ObjectOptions, ObjectPartInfo, OffsetDateTime, RUSTFS_META_BUCKET, RUSTFS_META_MULTIPART_BUCKET,
|
||||
RawFileInfo, ReadMultipleReq, ReadMultipleResp, ReadOptions, Result, SLASH_SEPARATOR, STORAGE_FORMAT_FILE, SetDisks,
|
||||
SnapshotLeaseToken, StorageError, UpdateMetadataOpts, Uuid, build_inline_bitrot_readers_from_refs,
|
||||
can_try_inline_data_shards_direct, capacity_scope_from_disks, codec_streaming_rollout_applies, coding,
|
||||
collect_inline_data_shard_fileinfos_by_index_or_reason, current_dirty_generation, debug, disk,
|
||||
file_info_is_valid_for_metadata, get_metadata_slowtail_fault_request, info, inline_erasure_shard_file_offset,
|
||||
Bytes, CHECK_PART_DISK_NOT_FOUND, DeleteOptions, DiskError, DiskStore, EVENT_SET_DISK_RENAME_TAIL_DRAIN_FAILED,
|
||||
EVENT_SET_DISK_WRITE, Error, FileInfo, FileMeta, FileMetaShallowVersion, GetCodecStreamingFallbackReason,
|
||||
GetObjectMetadataCacheEntry, HTTPPreconditions, HashAlgorithm, HealAdmissionResult, HealChannelPriority, HealRequestSource,
|
||||
LOG_COMPONENT_ECSTORE, LOG_SUBSYSTEM_SET_DISK, MultipartWriteQuorumContext, OBJECT_OP_IGNORED_ERRS, ObjectOptions,
|
||||
ObjectPartInfo, OffsetDateTime, RUSTFS_META_BUCKET, RUSTFS_META_MULTIPART_BUCKET, RawFileInfo, ReadMultipleReq,
|
||||
ReadMultipleResp, ReadOptions, Result, SLASH_SEPARATOR, STORAGE_FORMAT_FILE, SetDisks, SnapshotLeaseToken, StorageError,
|
||||
UpdateMetadataOpts, Uuid, build_inline_bitrot_readers_from_refs, can_try_inline_data_shards_direct,
|
||||
capacity_scope_from_disks, coding, collect_inline_data_shard_fileinfos_by_index_or_reason, current_dirty_generation, debug,
|
||||
disk, file_info_is_valid_for_metadata, get_metadata_slowtail_fault_request, info, inline_erasure_shard_file_offset,
|
||||
inline_erasure_shard_size, is_err_object_not_found, is_err_version_not_found, is_get_metadata_data_read_early_stop_enabled,
|
||||
is_get_metadata_early_stop_bounded_fanout_enabled, is_get_metadata_early_stop_enabled,
|
||||
is_get_metadata_non_inline_data_read_early_stop_enabled, is_object_dangling, is_version_early_stop_enabled,
|
||||
issue3031_diag_enabled, join_all, join_errs, log_multipart_write_quorum_failure, merge_file_meta_versions,
|
||||
object_fits_single_block, path_join_buf, record_global_dirty_scope, reduce_read_quorum_errs, reduce_write_quorum_errs,
|
||||
send_heal_request_with_admission, should_prevent_write, to_object_err, try_read_inline_data_shards_direct, warn,
|
||||
issue3031_diag_enabled, join_all, join_errs, log_multipart_write_quorum_failure, merge_file_meta_versions, path_join_buf,
|
||||
record_global_dirty_scope, reduce_read_quorum_errs, reduce_write_quorum_errs, send_heal_request_with_admission,
|
||||
should_prevent_write, to_object_err, try_read_inline_data_shards_direct, warn,
|
||||
};
|
||||
#[cfg(test)]
|
||||
use crate::bucket::lifecycle::lifecycle::TRANSITION_COMPLETE;
|
||||
@@ -453,29 +452,13 @@ use tokio::io::{AsyncRead, ReadBuf};
|
||||
use tokio::sync::{Mutex, RwLock, oneshot};
|
||||
use tokio::task::JoinSet;
|
||||
|
||||
struct AbortOnDropJoinHandle<T>(tokio::task::JoinHandle<T>);
|
||||
|
||||
impl<T> Future for AbortOnDropJoinHandle<T> {
|
||||
type Output = std::result::Result<T, tokio::task::JoinError>;
|
||||
|
||||
fn poll(mut self: Pin<&mut Self>, cx: &mut Context<'_>) -> Poll<Self::Output> {
|
||||
Pin::new(&mut self.0).poll(cx)
|
||||
}
|
||||
}
|
||||
|
||||
impl<T> Drop for AbortOnDropJoinHandle<T> {
|
||||
fn drop(&mut self) {
|
||||
self.0.abort();
|
||||
}
|
||||
}
|
||||
|
||||
pub(in crate::set_disk) const EVENT_SET_DISK_READ: &str = "set_disk_read";
|
||||
pub(in crate::set_disk) const ENV_RUSTFS_GET_DATA_BLOCKS_FIRST_READER_SETUP: &str = "RUSTFS_GET_DATA_BLOCKS_FIRST_READER_SETUP";
|
||||
const ENV_RUSTFS_GET_METADATA_READ_VERSION_COALESCE: &str = "RUSTFS_GET_METADATA_READ_VERSION_COALESCE";
|
||||
const ENV_RUSTFS_GET_METADATA_READ_VERSION_COALESCE_DELAY_MICROS: &str = "RUSTFS_GET_METADATA_READ_VERSION_COALESCE_DELAY_MICROS";
|
||||
const DEFAULT_GET_METADATA_READ_VERSION_COALESCE_DELAY_MICROS: u64 = 200;
|
||||
const METRIC_GET_METADATA_READ_VERSION_COALESCER_TOTAL: &str = "rustfs_get_metadata_read_version_coalescer_total";
|
||||
pub(crate) const ENV_RUSTFS_PUT_RENAME_EARLY_ACK_ENABLE: &str = "RUSTFS_PUT_RENAME_EARLY_ACK_ENABLE";
|
||||
pub(in crate::set_disk) const ENV_RUSTFS_PUT_RENAME_EARLY_ACK_ENABLE: &str = "RUSTFS_PUT_RENAME_EARLY_ACK_ENABLE";
|
||||
/// Default reader-setup strategy for the GET read path (rustfs/backlog#1215,
|
||||
/// #1159, #923).
|
||||
///
|
||||
@@ -1149,7 +1132,7 @@ fn data_read_early_stop_inline_candidate_miss_reason(candidate: &FileInfo) -> Op
|
||||
None
|
||||
}
|
||||
|
||||
pub(in crate::set_disk) fn non_inline_data_read_candidate_is_safe(candidate: &FileInfo) -> bool {
|
||||
fn non_inline_data_read_candidate_is_safe(candidate: &FileInfo) -> bool {
|
||||
if candidate.inline_data()
|
||||
|| candidate.is_compressed()
|
||||
|| candidate.is_remote()
|
||||
@@ -1164,16 +1147,6 @@ pub(in crate::set_disk) fn non_inline_data_read_candidate_is_safe(candidate: &Fi
|
||||
candidate.has_valid_erasure_geometry()
|
||||
}
|
||||
|
||||
pub(in crate::set_disk) fn late_materialization_candidate_is_safe(candidate: &FileInfo) -> bool {
|
||||
non_inline_data_read_candidate_is_safe(candidate)
|
||||
&& candidate.size > 512 * 1024
|
||||
&& object_fits_single_block(candidate.size, candidate.erasure.block_size)
|
||||
}
|
||||
|
||||
pub(in crate::set_disk) fn non_inline_data_read_early_stop_allowed(read_data: bool, bucket: &str, object: &str) -> bool {
|
||||
read_data && is_get_metadata_non_inline_data_read_early_stop_enabled() && !codec_streaming_rollout_applies(bucket, object)
|
||||
}
|
||||
|
||||
const NON_INLINE_SINGLE_PENDING_HEDGE_DELAY: Duration = Duration::from_millis(100);
|
||||
|
||||
fn data_read_inline_missing_shards_are_pending(
|
||||
@@ -2958,7 +2931,7 @@ impl SetDisks {
|
||||
read_data,
|
||||
healing,
|
||||
incl_free_versions,
|
||||
non_inline_data_read_early_stop_allowed(read_data, bucket, object),
|
||||
read_data && is_get_metadata_non_inline_data_read_early_stop_enabled(),
|
||||
default_parity_count,
|
||||
allow_coalescing,
|
||||
)
|
||||
@@ -3024,7 +2997,7 @@ impl SetDisks {
|
||||
let object = object.clone();
|
||||
let version_id = version_id.clone();
|
||||
let slowtail_fault = slowtail_fault.clone();
|
||||
AbortOnDropJoinHandle(tokio::spawn(async move {
|
||||
tokio::spawn(async move {
|
||||
let response_start = observe.then(Instant::now);
|
||||
let result = if let Some(disk) = disk {
|
||||
Self::record_read_version_call(&object, disk_index);
|
||||
@@ -3039,7 +3012,7 @@ impl SetDisks {
|
||||
};
|
||||
let elapsed = response_start.map(|start| start.elapsed());
|
||||
(result, elapsed)
|
||||
}))
|
||||
})
|
||||
});
|
||||
|
||||
// Wait for all futures to complete
|
||||
@@ -3733,34 +3706,16 @@ fn dangling_delete_grace() -> time::Duration {
|
||||
/// Result of scanning one disk's copy of a directory prefix while deciding
|
||||
/// whether an orphan (metadata-less) directory tree can be safely purged.
|
||||
enum OrphanDirScan {
|
||||
/// The subtree holds object metadata or uncommitted data, so it must not be
|
||||
/// purged.
|
||||
/// The subtree holds at least one regular file (object metadata or data), so
|
||||
/// it is a real object and must not be purged.
|
||||
HasData,
|
||||
/// The prefix contains only empty directories and/or UUID data directories
|
||||
/// carrying a committed delete marker.
|
||||
Purgeable {
|
||||
empty_dirs: Vec<String>,
|
||||
committed_files: Vec<String>,
|
||||
},
|
||||
/// The prefix exists on this disk and contains only nested empty directories.
|
||||
/// Carries every directory path in pre-order (parents before children).
|
||||
Empty(Vec<String>),
|
||||
/// The prefix does not exist on this disk.
|
||||
Missing,
|
||||
}
|
||||
|
||||
fn is_safe_orphan_dir_entry(entry: &str) -> bool {
|
||||
let component = entry.strip_suffix(SLASH_SEPARATOR).unwrap_or(entry);
|
||||
!component.is_empty()
|
||||
&& component != "."
|
||||
&& component != ".."
|
||||
&& !component.contains(SLASH_SEPARATOR)
|
||||
&& !component.contains('\\')
|
||||
}
|
||||
|
||||
fn is_committed_delete_marker(entry: &str) -> bool {
|
||||
entry
|
||||
.strip_prefix(DELETE_DATA_DIR_MARKER_PREFIX)
|
||||
.is_some_and(|transaction| Uuid::parse_str(transaction).is_ok_and(|uuid| !uuid.is_nil()))
|
||||
}
|
||||
|
||||
/// Outcome of a *post-quorum* `rename_data` commit, classifying whether the
|
||||
/// committed replicas converged so the caller can decide heal admission
|
||||
/// WITHOUT conflating "a version signature exists" with "this write needs
|
||||
@@ -6143,151 +6098,52 @@ impl SetDisks {
|
||||
}
|
||||
|
||||
/// Scan a single disk's copy of `prefix` and decide whether it is an orphan
|
||||
/// directory subtree. Only empty directories and UUID data directories with
|
||||
/// valid committed delete markers are purgeable; every child is still scanned.
|
||||
/// (metadata-less) directory subtree. Walks the tree iteratively and returns
|
||||
/// [`OrphanDirScan::HasData`] as soon as any regular file is found.
|
||||
async fn scan_orphan_dir(disk: &DiskStore, bucket: &str, prefix: &str) -> OrphanDirScan {
|
||||
let root = prefix.trim_end_matches(SLASH_SEPARATOR).to_string();
|
||||
let mut stack = vec![root.clone()];
|
||||
// Pre-order list of directories (a parent always precedes its descendants),
|
||||
// so reversing it yields a safe children-first removal order.
|
||||
let mut dirs: Vec<String> = Vec::new();
|
||||
let mut committed_files: Vec<String> = Vec::new();
|
||||
let mut existed = false;
|
||||
|
||||
while let Some(dir) = stack.pop() {
|
||||
let entries = match disk.list_dir("", bucket, &dir, 0).await {
|
||||
Ok(entries) => entries,
|
||||
Err(DiskError::FileNotFound | DiskError::VolumeNotFound) => {
|
||||
Err(_) => {
|
||||
// The root missing (or never existing) means there is nothing to
|
||||
// purge on this disk. A nested directory vanishing mid-scan is a
|
||||
// benign race, so skip it and keep walking.
|
||||
if dir == root {
|
||||
return OrphanDirScan::Missing;
|
||||
}
|
||||
// A nested directory vanishing mid-scan is a benign race.
|
||||
continue;
|
||||
}
|
||||
// Classification must fail closed: committed residue is safe to
|
||||
// remove only after every reachable child was inspected.
|
||||
Err(_) => return OrphanDirScan::HasData,
|
||||
};
|
||||
|
||||
existed = true;
|
||||
let mut child_dirs = Vec::new();
|
||||
let mut files = Vec::new();
|
||||
dirs.push(dir.clone());
|
||||
|
||||
for entry in entries {
|
||||
if !is_safe_orphan_dir_entry(&entry) {
|
||||
return OrphanDirScan::HasData;
|
||||
}
|
||||
match entry.strip_suffix(SLASH_SEPARATOR) {
|
||||
Some(child) => child_dirs.push(format!("{dir}{SLASH_SEPARATOR}{child}")),
|
||||
None => files.push(entry),
|
||||
// `read_dir` marks directories with a trailing slash; anything else
|
||||
// is a regular file, which means real object data lives here.
|
||||
Some(child) => stack.push(format!("{dir}{SLASH_SEPARATOR}{child}")),
|
||||
None => return OrphanDirScan::HasData,
|
||||
}
|
||||
}
|
||||
|
||||
if !files.is_empty() {
|
||||
let data_dir_name = dir.rsplit(SLASH_SEPARATOR).next().unwrap_or_default();
|
||||
let is_uuid_data_dir = Uuid::parse_str(data_dir_name).is_ok_and(|uuid| !uuid.is_nil());
|
||||
let has_committed_delete = files.iter().any(|entry| is_committed_delete_marker(entry));
|
||||
|
||||
if !is_uuid_data_dir || !has_committed_delete || files.iter().any(|entry| entry == STORAGE_FORMAT_FILE) {
|
||||
return OrphanDirScan::HasData;
|
||||
}
|
||||
|
||||
committed_files.extend(files.into_iter().map(|entry| path_join_buf(&[&dir, &entry])));
|
||||
dirs.push(dir);
|
||||
stack.extend(child_dirs);
|
||||
continue;
|
||||
}
|
||||
|
||||
dirs.push(dir);
|
||||
stack.extend(child_dirs);
|
||||
}
|
||||
|
||||
if existed {
|
||||
OrphanDirScan::Purgeable {
|
||||
empty_dirs: dirs,
|
||||
committed_files,
|
||||
}
|
||||
OrphanDirScan::Empty(dirs)
|
||||
} else {
|
||||
OrphanDirScan::Missing
|
||||
}
|
||||
}
|
||||
|
||||
async fn delete_purgeable_orphan_entries(
|
||||
disk: &DiskStore,
|
||||
bucket: &str,
|
||||
object: &str,
|
||||
mut empty_dirs: Vec<String>,
|
||||
committed_files: Vec<String>,
|
||||
) {
|
||||
// Keep every committed marker until all ordinary residue files are gone.
|
||||
// If any delete fails, a later request can still recognize and retry the
|
||||
// committed cleanup instead of stranding an unmarked partial residue.
|
||||
for delete_markers in [false, true] {
|
||||
for file in &committed_files {
|
||||
let is_marker = file.rsplit(SLASH_SEPARATOR).next().is_some_and(is_committed_delete_marker);
|
||||
if is_marker != delete_markers {
|
||||
continue;
|
||||
}
|
||||
if let Err(err) = disk
|
||||
.delete(
|
||||
bucket,
|
||||
file,
|
||||
DeleteOptions {
|
||||
recursive: false,
|
||||
immediate: true,
|
||||
..Default::default()
|
||||
},
|
||||
)
|
||||
.await
|
||||
{
|
||||
debug!(
|
||||
event = EVENT_SET_DISK_ORPHAN_PURGE_SKIPPED,
|
||||
component = LOG_COMPONENT_ECSTORE,
|
||||
subsystem = LOG_SUBSYSTEM_SET_DISK,
|
||||
bucket,
|
||||
object,
|
||||
path = file,
|
||||
error = ?err,
|
||||
"Orphan prefix purge skipped"
|
||||
);
|
||||
return;
|
||||
}
|
||||
}
|
||||
}
|
||||
empty_dirs.reverse();
|
||||
for dir in empty_dirs {
|
||||
if let Err(err) = disk
|
||||
.delete(
|
||||
bucket,
|
||||
&dir,
|
||||
DeleteOptions {
|
||||
recursive: false,
|
||||
immediate: true,
|
||||
..Default::default()
|
||||
},
|
||||
)
|
||||
.await
|
||||
{
|
||||
// Best effort: a sibling removal may have already cleared a shared
|
||||
// parent, or a concurrent writer repopulated the directory. Neither
|
||||
// is fatal to purging the orphan tree.
|
||||
debug!(
|
||||
event = EVENT_SET_DISK_ORPHAN_PURGE_SKIPPED,
|
||||
component = LOG_COMPONENT_ECSTORE,
|
||||
subsystem = LOG_SUBSYSTEM_SET_DISK,
|
||||
bucket,
|
||||
object,
|
||||
path = dir,
|
||||
error = ?err,
|
||||
"Orphan prefix purge skipped"
|
||||
);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// Purge an orphan directory prefix — a trailing-slash key that exists on disk
|
||||
/// as empty directories or committed delete residue, with no object metadata
|
||||
/// or uncommitted data on any disk of this set.
|
||||
/// as an empty directory tree with no object metadata on any disk of this set.
|
||||
/// Such prefixes are listable (see `scan_dir`) yet are not real objects, so the
|
||||
/// normal delete path returns NotFound and leaves them stranded (issue #4189).
|
||||
///
|
||||
@@ -6304,18 +6160,15 @@ impl SetDisks {
|
||||
// Phase 1: classify every online disk. Refuse to purge if ANY disk holds
|
||||
// object data under the prefix, so a degraded/healable object is never
|
||||
// destroyed.
|
||||
let mut per_disk_dirs: Vec<(usize, Vec<String>, Vec<String>)> = Vec::new();
|
||||
let mut per_disk_dirs: Vec<(usize, Vec<String>)> = Vec::new();
|
||||
let mut existed = false;
|
||||
for (i, disk) in disks.iter().enumerate() {
|
||||
let Some(disk) = disk else { continue };
|
||||
match Self::scan_orphan_dir(disk, bucket, object).await {
|
||||
OrphanDirScan::HasData => return Ok(false),
|
||||
OrphanDirScan::Purgeable {
|
||||
empty_dirs,
|
||||
committed_files,
|
||||
} => {
|
||||
OrphanDirScan::Empty(dirs) => {
|
||||
existed = true;
|
||||
per_disk_dirs.push((i, empty_dirs, committed_files));
|
||||
per_disk_dirs.push((i, dirs));
|
||||
}
|
||||
OrphanDirScan::Missing => {}
|
||||
}
|
||||
@@ -6325,14 +6178,32 @@ impl SetDisks {
|
||||
return Ok(false);
|
||||
}
|
||||
|
||||
// Phase 2: remove only the files classified as committed residue, then
|
||||
// remove directories children-first. Every directory delete is
|
||||
// non-recursive, so a directory that concurrently gained an object fails
|
||||
// with DirectoryNotEmpty and is skipped — a racing PutObject is never
|
||||
// clobbered.
|
||||
for (i, empty_dirs, committed_files) in per_disk_dirs {
|
||||
// Phase 2: remove the empty directories children-first on each disk. A
|
||||
// non-recursive delete performs an empty-only `rmdir`, so a directory that
|
||||
// concurrently gained an object fails with DirectoryNotEmpty and is skipped —
|
||||
// a racing PutObject is never clobbered.
|
||||
for (i, mut dirs) in per_disk_dirs {
|
||||
let Some(disk) = disks[i].as_ref() else { continue };
|
||||
Self::delete_purgeable_orphan_entries(disk, bucket, object, empty_dirs, committed_files).await;
|
||||
dirs.reverse();
|
||||
for dir in dirs {
|
||||
if let Err(err) = disk
|
||||
.delete(
|
||||
bucket,
|
||||
&dir,
|
||||
DeleteOptions {
|
||||
recursive: false,
|
||||
immediate: true,
|
||||
..Default::default()
|
||||
},
|
||||
)
|
||||
.await
|
||||
{
|
||||
// Best effort: a sibling removal may have already cleared a shared
|
||||
// parent, or a concurrent writer repopulated the directory. Neither
|
||||
// is fatal to purging the orphan tree.
|
||||
debug!(bucket, object, dir, error = ?err, "purge_orphan_dir_object: skipped non-empty/absent directory");
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
Ok(true)
|
||||
@@ -6951,7 +6822,7 @@ pub(in crate::set_disk) mod rename_fanout_barrier_phase {
|
||||
/// Cross-process/black-box fault injection (toxiproxy, blackhole peers, 2-pool)
|
||||
/// is a later cluster-harness block, not this one.
|
||||
#[cfg(test)]
|
||||
pub(crate) mod rename_fanout_barrier {
|
||||
pub(in crate::set_disk) mod rename_fanout_barrier {
|
||||
use std::collections::HashMap;
|
||||
use std::sync::atomic::{AtomicBool, AtomicUsize, Ordering};
|
||||
use std::sync::{Arc, Mutex, OnceLock};
|
||||
@@ -7151,37 +7022,6 @@ mod tests {
|
||||
use tempfile::TempDir;
|
||||
use tokio::io::AsyncReadExt;
|
||||
|
||||
#[test]
|
||||
fn orphan_dir_entries_must_be_single_relative_components() {
|
||||
for entry in ["part.1", "child/", "delete-data.00000000-0000-0000-0000-000000000001"] {
|
||||
assert!(is_safe_orphan_dir_entry(entry), "{entry:?} should be accepted");
|
||||
}
|
||||
for entry in ["", "/", ".", "..", "../", "child//", "a/b", r"a\b", "./"] {
|
||||
assert!(!is_safe_orphan_dir_entry(entry), "{entry:?} should be rejected");
|
||||
}
|
||||
}
|
||||
|
||||
#[test]
|
||||
#[serial_test::serial(codec_streaming_env)]
|
||||
fn non_inline_early_stop_is_mutually_exclusive_with_codec_rollout() {
|
||||
temp_env::with_vars(
|
||||
[
|
||||
(ENV_RUSTFS_GET_METADATA_TWO_PHASE_READ_PLAN_ENABLE, Some("true")),
|
||||
("RUSTFS_GET_CODEC_STREAMING_ROLLOUT", Some("on")),
|
||||
("RUSTFS_GET_CODEC_STREAMING_BODY_COMPAT_CONFIRMED", Some("true")),
|
||||
("RUSTFS_GET_CODEC_STREAMING_HEADER_COMPAT_CONFIRMED", Some("true")),
|
||||
],
|
||||
|| assert!(!non_inline_data_read_early_stop_allowed(true, "bucket", "object")),
|
||||
);
|
||||
temp_env::with_vars(
|
||||
[
|
||||
(ENV_RUSTFS_GET_METADATA_TWO_PHASE_READ_PLAN_ENABLE, Some("true")),
|
||||
("RUSTFS_GET_CODEC_STREAMING_ROLLOUT", Some("off")),
|
||||
],
|
||||
|| assert!(non_inline_data_read_early_stop_allowed(true, "bucket", "object")),
|
||||
);
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
async fn scanner_delete_owner_survives_waiter_cancellation() {
|
||||
let movement_gate = Arc::new(tokio::sync::RwLock::new(()));
|
||||
@@ -7339,96 +7179,6 @@ mod tests {
|
||||
(dir, disk)
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
async fn orphan_cleanup_preserves_object_published_after_scan() {
|
||||
let (dir, disk) = read_multiple_test_disk("bucket", &[]).await;
|
||||
let transaction = Uuid::new_v4();
|
||||
let residue = dir
|
||||
.path()
|
||||
.join("bucket")
|
||||
.join("pfx")
|
||||
.join("object")
|
||||
.join(Uuid::new_v4().to_string());
|
||||
tokio::fs::create_dir_all(&residue)
|
||||
.await
|
||||
.expect("committed data directory should be created");
|
||||
tokio::fs::write(residue.join("part.1"), b"stale")
|
||||
.await
|
||||
.expect("stale part should be written");
|
||||
tokio::fs::write(residue.join(format!("{DELETE_DATA_DIR_MARKER_PREFIX}{transaction}")), [])
|
||||
.await
|
||||
.expect("committed delete marker should be written");
|
||||
|
||||
let OrphanDirScan::Purgeable {
|
||||
empty_dirs,
|
||||
committed_files,
|
||||
} = SetDisks::scan_orphan_dir(&disk, "bucket", "pfx/").await
|
||||
else {
|
||||
panic!("committed residue should be classified as purgeable");
|
||||
};
|
||||
|
||||
let nested_object = residue.join("nested");
|
||||
tokio::fs::create_dir_all(&nested_object)
|
||||
.await
|
||||
.expect("concurrent object directory should be created");
|
||||
tokio::fs::write(nested_object.join(STORAGE_FORMAT_FILE), b"new metadata")
|
||||
.await
|
||||
.expect("concurrent object metadata should be written");
|
||||
|
||||
SetDisks::delete_purgeable_orphan_entries(&disk, "bucket", "pfx/", empty_dirs, committed_files).await;
|
||||
|
||||
assert!(
|
||||
nested_object.join(STORAGE_FORMAT_FILE).exists(),
|
||||
"an object published after classification must survive cleanup"
|
||||
);
|
||||
assert!(!residue.join("part.1").exists(), "classified stale data should be reclaimed");
|
||||
}
|
||||
|
||||
#[cfg(unix)]
|
||||
#[tokio::test]
|
||||
async fn orphan_cleanup_keeps_commit_marker_when_residue_delete_fails() {
|
||||
use std::os::unix::fs::PermissionsExt;
|
||||
|
||||
let (dir, disk) = read_multiple_test_disk("bucket", &[]).await;
|
||||
let marker_name = format!("{DELETE_DATA_DIR_MARKER_PREFIX}{}", Uuid::new_v4());
|
||||
let residue = dir
|
||||
.path()
|
||||
.join("bucket")
|
||||
.join("pfx")
|
||||
.join("object")
|
||||
.join(Uuid::new_v4().to_string());
|
||||
tokio::fs::create_dir_all(&residue)
|
||||
.await
|
||||
.expect("committed data directory should be created");
|
||||
tokio::fs::write(residue.join("part.1"), b"stale")
|
||||
.await
|
||||
.expect("stale part should be written");
|
||||
tokio::fs::write(residue.join(&marker_name), [])
|
||||
.await
|
||||
.expect("committed delete marker should be written");
|
||||
|
||||
let OrphanDirScan::Purgeable {
|
||||
empty_dirs,
|
||||
committed_files,
|
||||
} = SetDisks::scan_orphan_dir(&disk, "bucket", "pfx/").await
|
||||
else {
|
||||
panic!("committed residue should be classified as purgeable");
|
||||
};
|
||||
tokio::fs::set_permissions(&residue, std::fs::Permissions::from_mode(0o555))
|
||||
.await
|
||||
.expect("residue directory should become read-only");
|
||||
|
||||
SetDisks::delete_purgeable_orphan_entries(&disk, "bucket", "pfx/", empty_dirs, committed_files).await;
|
||||
|
||||
let part_remains = residue.join("part.1").exists();
|
||||
let marker_remains = residue.join(marker_name).exists();
|
||||
tokio::fs::set_permissions(&residue, std::fs::Permissions::from_mode(0o755))
|
||||
.await
|
||||
.expect("residue directory permissions should be restored");
|
||||
assert!(part_remains, "the injected residue delete failure should retain the part");
|
||||
assert!(marker_remains, "the commit marker must remain so a later cleanup can retry");
|
||||
}
|
||||
|
||||
async fn io_primitives_test_set(disks: Vec<Option<DiskStore>>, default_parity_count: usize) -> Arc<SetDisks> {
|
||||
let set_drive_count = disks.len();
|
||||
SetDisks::new(
|
||||
|
||||
@@ -42,8 +42,7 @@ use crate::bucket::lifecycle::lifecycle::TRANSITION_COMPLETE;
|
||||
use crate::bucket::metadata_sys;
|
||||
use crate::bucket::metadata_sys::ObjectLockConfigState;
|
||||
use crate::bucket::object_lock::objectlock_sys::{
|
||||
check_object_lock_for_deletion_with_state, check_retention_for_modification, replication_delete_may_bypass_governance,
|
||||
replication_write_may_pass_worm_gate,
|
||||
check_object_lock_for_deletion_with_state, check_retention_for_modification, replication_write_may_pass_worm_gate,
|
||||
};
|
||||
#[cfg(test)]
|
||||
use crate::bucket::replication::ReplicationState;
|
||||
@@ -329,7 +328,6 @@ const EVENT_SET_DISK_HEAL: &str = "set_disk_heal";
|
||||
const EVENT_SET_DISK_COMMIT_TAIL_SLOW: &str = "set_disk_commit_tail_slow";
|
||||
const EVENT_SET_DISK_RENAME_TAIL_DRAIN_FAILED: &str = "set_disk_rename_tail_drain_failed";
|
||||
const EVENT_SET_DISK_PUT_OBJECT_STAGE_SUMMARY: &str = "set_disk_put_object_stage_summary";
|
||||
const EVENT_SET_DISK_ORPHAN_PURGE_SKIPPED: &str = "set_disk_orphan_purge_skipped";
|
||||
const SET_DISK_COMMIT_TAIL_WARN_THRESHOLD_MS: u128 = 5_000;
|
||||
const ENV_RUSTFS_PUT_LARGE_BATCH_MIN_SIZE_BYTES: &str = "RUSTFS_PUT_LARGE_BATCH_MIN_SIZE_BYTES";
|
||||
const DEFAULT_RUSTFS_PUT_LARGE_BATCH_MIN_SIZE_BYTES: usize = 64 * 1024 * 1024;
|
||||
@@ -860,8 +858,6 @@ static OBJECT_LOCK_DIAG_ENABLED: OnceLock<bool> = OnceLock::new();
|
||||
mod core;
|
||||
#[cfg(test)]
|
||||
pub(crate) use core::io_primitives::disk_call_counters;
|
||||
#[cfg(all(test, feature = "test-util"))]
|
||||
pub(crate) use core::io_primitives::{ENV_RUSTFS_PUT_RENAME_EARLY_ACK_ENABLE, rename_fanout_barrier};
|
||||
mod ctx;
|
||||
mod metadata;
|
||||
mod ops;
|
||||
@@ -874,7 +870,7 @@ pub(crate) use ops::multipart::NewMultipartUploadCommitObservation;
|
||||
pub use ops::multipart::{MultipartCommitBarrier, MultipartCommitPause};
|
||||
#[cfg(test)]
|
||||
pub(crate) use ops::object::DeleteObjectCommitBarrier;
|
||||
#[cfg(any(test, feature = "test-util"))]
|
||||
#[cfg(feature = "test-util")]
|
||||
pub(crate) use ops::object::TransitionCleanupStoreBarrier as SetDiskTransitionCleanupStoreBarrier;
|
||||
pub(crate) use ops::object::body_cache_plaintext_len;
|
||||
#[cfg(all(test, feature = "test-util"))]
|
||||
@@ -898,11 +894,8 @@ struct OwnedGetObjectFileInfo {
|
||||
fi: FileInfo,
|
||||
parts_metadata: Vec<FileInfo>,
|
||||
online_disks: Vec<Option<DiskStore>>,
|
||||
late_metadata_fanout_disks: Option<Vec<Option<DiskStore>>>,
|
||||
}
|
||||
|
||||
type OwnedGetObjectFileInfoParts = (FileInfo, Vec<FileInfo>, Vec<Option<DiskStore>>, Option<Vec<Option<DiskStore>>>);
|
||||
|
||||
impl GetObjectFileInfo {
|
||||
fn owned(fi: FileInfo, parts_metadata: Vec<FileInfo>, online_disks: Vec<Option<DiskStore>>) -> Self {
|
||||
Self {
|
||||
@@ -910,24 +903,6 @@ impl GetObjectFileInfo {
|
||||
fi,
|
||||
parts_metadata,
|
||||
online_disks,
|
||||
late_metadata_fanout_disks: None,
|
||||
}),
|
||||
shared: None,
|
||||
}
|
||||
}
|
||||
|
||||
fn owned_with_late_metadata_fanout(
|
||||
fi: FileInfo,
|
||||
parts_metadata: Vec<FileInfo>,
|
||||
online_disks: Vec<Option<DiskStore>>,
|
||||
late_metadata_fanout_disks: Vec<Option<DiskStore>>,
|
||||
) -> Self {
|
||||
Self {
|
||||
owned: Some(OwnedGetObjectFileInfo {
|
||||
fi,
|
||||
parts_metadata,
|
||||
online_disks,
|
||||
late_metadata_fanout_disks: Some(late_metadata_fanout_disks),
|
||||
}),
|
||||
shared: None,
|
||||
}
|
||||
@@ -964,28 +939,19 @@ impl GetObjectFileInfo {
|
||||
}
|
||||
}
|
||||
|
||||
fn has_late_metadata_fanout(&self) -> bool {
|
||||
self.owned
|
||||
.as_ref()
|
||||
.is_some_and(|snapshot| snapshot.late_metadata_fanout_disks.is_some())
|
||||
}
|
||||
|
||||
fn into_owned(self) -> (FileInfo, Vec<FileInfo>, Vec<Option<DiskStore>>) {
|
||||
let (fi, parts_metadata, online_disks, _) = self.into_owned_with_late_metadata_fanout();
|
||||
(fi, parts_metadata, online_disks)
|
||||
}
|
||||
|
||||
fn into_owned_with_late_metadata_fanout(self) -> OwnedGetObjectFileInfoParts {
|
||||
match (self.owned, self.shared) {
|
||||
(Some(snapshot), None) => (
|
||||
snapshot.fi,
|
||||
snapshot.parts_metadata,
|
||||
snapshot.online_disks,
|
||||
snapshot.late_metadata_fanout_disks,
|
||||
),
|
||||
(Some(snapshot), None) => {
|
||||
let OwnedGetObjectFileInfo {
|
||||
fi,
|
||||
parts_metadata,
|
||||
online_disks,
|
||||
} = snapshot;
|
||||
(fi, parts_metadata, online_disks)
|
||||
}
|
||||
(None, Some(entry)) => match Arc::try_unwrap(entry) {
|
||||
Ok(entry) => (entry.fi, entry.parts_metadata, entry.online_disks, None),
|
||||
Err(entry) => (entry.fi.clone(), entry.parts_metadata.clone(), entry.online_disks.clone(), None),
|
||||
Ok(entry) => (entry.fi, entry.parts_metadata, entry.online_disks),
|
||||
Err(entry) => (entry.fi.clone(), entry.parts_metadata.clone(), entry.online_disks.clone()),
|
||||
},
|
||||
_ => unreachable!("GET metadata snapshot representation must be exclusive"),
|
||||
}
|
||||
@@ -1266,278 +1232,6 @@ mod prepared_get_object_metadata_tests {
|
||||
.await;
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
#[serial_test::serial(body_cache_hook)]
|
||||
async fn non_inline_two_phase_read_fetches_late_parity_after_two_selected_shards_fail() {
|
||||
let (dirs, set_disks) = make_local_set_disks(4, 2).await;
|
||||
let bucket = "non-inline-read-late-parity";
|
||||
let object = object_with_initial_data_shards(bucket, "late-parity-object");
|
||||
let payload = vec![0x5a; 1024 * 1024];
|
||||
let opts = ObjectOptions {
|
||||
no_lock: true,
|
||||
..Default::default()
|
||||
};
|
||||
|
||||
set_disks
|
||||
.make_bucket(bucket, &MakeBucketOptions::default())
|
||||
.await
|
||||
.expect("bucket should be created");
|
||||
let mut put_reader = PutObjReader::from_vec(payload.clone());
|
||||
set_disks
|
||||
.put_object(bucket, &object, &mut put_reader, &opts)
|
||||
.await
|
||||
.expect("object should be written");
|
||||
|
||||
let order = bounded_metadata_fanout_order(bucket, &object, 4, 2);
|
||||
let distribution = FileInfo::new(&[bucket, object.as_str()].join("/"), 2, 2).erasure.distribution;
|
||||
assert!(
|
||||
order.iter().take(2).all(|disk_index| distribution[*disk_index] <= 2),
|
||||
"the two failed selected shards must be data shards"
|
||||
);
|
||||
assert!(
|
||||
distribution[order[3]] > 2,
|
||||
"the metadata shard omitted by the plan must be healthy parity"
|
||||
);
|
||||
for disk_index in order.iter().take(2) {
|
||||
let object_dir = dirs[*disk_index].path().join(bucket).join(&object);
|
||||
let data_dir = std::fs::read_dir(&object_dir)
|
||||
.expect("object directory should be readable")
|
||||
.find_map(|entry| {
|
||||
let entry = entry.expect("object directory entry should be readable");
|
||||
entry
|
||||
.file_type()
|
||||
.expect("object directory entry type should be readable")
|
||||
.is_dir()
|
||||
.then(|| entry.path())
|
||||
})
|
||||
.expect("object data directory should exist");
|
||||
let part_path = data_dir.join("part.1");
|
||||
let mut shard = std::fs::read(&part_path).expect("selected data shard should be readable before corruption");
|
||||
shard[0] ^= 0xff;
|
||||
std::fs::write(part_path, shard).expect("selected data shard should be corrupted after metadata was written");
|
||||
}
|
||||
|
||||
temp_env::async_with_vars(
|
||||
[
|
||||
("RUSTFS_GET_METADATA_TWO_PHASE_READ_PLAN_ENABLE", Some("true")),
|
||||
("RUSTFS_GET_METADATA_EARLY_STOP_ENABLE", Some("true")),
|
||||
("RUSTFS_GET_METADATA_EARLY_STOP_BOUNDED_FANOUT", Some("true")),
|
||||
],
|
||||
async {
|
||||
let calls = disk_call_counters::observe(&object);
|
||||
let mut reader = set_disks
|
||||
.get_object_reader(bucket, &object, None, HeaderMap::new(), &opts)
|
||||
.await
|
||||
.expect("two-phase GET should recover using late parity");
|
||||
let mut restored = Vec::new();
|
||||
reader
|
||||
.stream
|
||||
.read_to_end(&mut restored)
|
||||
.await
|
||||
.expect("late parity should restore the exact GET body");
|
||||
assert_eq!(restored, payload);
|
||||
assert_eq!(calls.total(disk_call_counters::KIND_READ_VERSION), 7);
|
||||
},
|
||||
)
|
||||
.await;
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
#[serial_test::serial(body_cache_hook)]
|
||||
async fn non_inline_two_phase_read_fetches_late_parity_when_selected_parts_are_missing() {
|
||||
let (dirs, set_disks) = make_local_set_disks(4, 2).await;
|
||||
let bucket = "non-inline-read-late-parity-missing";
|
||||
let object = object_with_initial_data_shards(bucket, "late-parity-missing-object");
|
||||
let payload = vec![0x3c; 1024 * 1024];
|
||||
let opts = ObjectOptions {
|
||||
no_lock: true,
|
||||
..Default::default()
|
||||
};
|
||||
|
||||
set_disks
|
||||
.make_bucket(bucket, &MakeBucketOptions::default())
|
||||
.await
|
||||
.expect("bucket should be created");
|
||||
let mut put_reader = PutObjReader::from_vec(payload.clone());
|
||||
set_disks
|
||||
.put_object(bucket, &object, &mut put_reader, &opts)
|
||||
.await
|
||||
.expect("object should be written");
|
||||
|
||||
let order = bounded_metadata_fanout_order(bucket, &object, 4, 2);
|
||||
for disk_index in order.iter().take(2) {
|
||||
let object_dir = dirs[*disk_index].path().join(bucket).join(&object);
|
||||
let data_dir = std::fs::read_dir(&object_dir)
|
||||
.expect("object directory should be readable")
|
||||
.find_map(|entry| {
|
||||
let entry = entry.expect("object directory entry should be readable");
|
||||
entry
|
||||
.file_type()
|
||||
.expect("entry type should be readable")
|
||||
.is_dir()
|
||||
.then(|| entry.path())
|
||||
})
|
||||
.expect("object data directory should exist");
|
||||
std::fs::remove_file(data_dir.join("part.1")).expect("selected data shard should be removed");
|
||||
}
|
||||
|
||||
temp_env::async_with_vars(
|
||||
[
|
||||
("RUSTFS_GET_METADATA_TWO_PHASE_READ_PLAN_ENABLE", Some("true")),
|
||||
("RUSTFS_GET_METADATA_EARLY_STOP_ENABLE", Some("true")),
|
||||
("RUSTFS_GET_METADATA_EARLY_STOP_BOUNDED_FANOUT", Some("true")),
|
||||
],
|
||||
async {
|
||||
let calls = disk_call_counters::observe(&object);
|
||||
let mut reader = set_disks
|
||||
.get_object_reader(bucket, &object, None, HeaderMap::new(), &opts)
|
||||
.await
|
||||
.expect("two-phase GET should recover using late parity");
|
||||
let mut restored = Vec::new();
|
||||
reader
|
||||
.stream
|
||||
.read_to_end(&mut restored)
|
||||
.await
|
||||
.expect("late parity should restore the exact GET body");
|
||||
assert_eq!(restored, payload);
|
||||
assert_eq!(calls.total(disk_call_counters::KIND_READ_VERSION), 7);
|
||||
},
|
||||
)
|
||||
.await;
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
#[serial_test::serial(body_cache_hook)]
|
||||
async fn four_data_two_parity_two_phase_read_recovers_one_failed_data_shard() {
|
||||
let (dirs, set_disks) = make_local_set_disks(6, 2).await;
|
||||
let bucket = "four-data-two-parity-late-read";
|
||||
let object = object_with_initial_data_shards_for_geometry(bucket, "one-failed-data", 4, 2);
|
||||
let payload = vec![0x7a; 1024 * 1024];
|
||||
let opts = ObjectOptions {
|
||||
no_lock: true,
|
||||
..Default::default()
|
||||
};
|
||||
|
||||
set_disks
|
||||
.make_bucket(bucket, &MakeBucketOptions::default())
|
||||
.await
|
||||
.expect("bucket should be created");
|
||||
let mut put_reader = PutObjReader::from_vec(payload.clone());
|
||||
set_disks
|
||||
.put_object(bucket, &object, &mut put_reader, &opts)
|
||||
.await
|
||||
.expect("object should be written");
|
||||
|
||||
let order = bounded_metadata_fanout_order(bucket, &object, 6, 2);
|
||||
let distribution = FileInfo::new(&[bucket, object.as_str()].join("/"), 4, 2).erasure.distribution;
|
||||
let failed_disk = *order
|
||||
.iter()
|
||||
.take(4)
|
||||
.find(|disk_index| distribution[**disk_index] <= 4)
|
||||
.expect("initial fanout should include a data shard");
|
||||
assert!(
|
||||
order.iter().take(4).all(|disk_index| distribution[*disk_index] <= 4),
|
||||
"initial fanout should cover all four data shards"
|
||||
);
|
||||
assert!(distribution[order[5]] > 4, "the final deferred metadata shard should be parity");
|
||||
|
||||
let object_dir = dirs[failed_disk].path().join(bucket).join(&object);
|
||||
let data_dir = std::fs::read_dir(&object_dir)
|
||||
.expect("object directory should be readable")
|
||||
.find_map(|entry| {
|
||||
let entry = entry.expect("object directory entry should be readable");
|
||||
entry
|
||||
.file_type()
|
||||
.expect("object directory entry type should be readable")
|
||||
.is_dir()
|
||||
.then(|| entry.path())
|
||||
})
|
||||
.expect("object data directory should exist");
|
||||
let part_path = data_dir.join("part.1");
|
||||
let mut shard = std::fs::read(&part_path).expect("selected data shard should be readable before corruption");
|
||||
shard[0] ^= 0xff;
|
||||
std::fs::write(part_path, shard).expect("selected data shard should be corrupted after metadata was written");
|
||||
|
||||
temp_env::async_with_vars(
|
||||
[
|
||||
("RUSTFS_GET_METADATA_TWO_PHASE_READ_PLAN_ENABLE", Some("true")),
|
||||
("RUSTFS_GET_METADATA_EARLY_STOP_ENABLE", Some("true")),
|
||||
("RUSTFS_GET_METADATA_EARLY_STOP_BOUNDED_FANOUT", Some("true")),
|
||||
],
|
||||
async {
|
||||
let calls = disk_call_counters::observe(&object);
|
||||
let mut reader = set_disks
|
||||
.get_object_reader(bucket, &object, None, HeaderMap::new(), &opts)
|
||||
.await
|
||||
.expect("two-phase GET should recover with one failed data shard");
|
||||
let mut restored = Vec::new();
|
||||
reader
|
||||
.stream
|
||||
.read_to_end(&mut restored)
|
||||
.await
|
||||
.expect("late parity should restore the exact GET body");
|
||||
assert_eq!(restored, payload);
|
||||
assert_eq!(calls.total(disk_call_counters::KIND_READ_VERSION), 11);
|
||||
},
|
||||
)
|
||||
.await;
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
#[serial_test::serial(body_cache_hook)]
|
||||
async fn four_data_two_parity_two_phase_read_rejects_below_read_quorum() {
|
||||
let (dirs, set_disks) = make_local_set_disks(6, 2).await;
|
||||
let bucket = "four-data-two-parity-quorum-minus-one";
|
||||
let object = object_with_initial_data_shards_for_geometry(bucket, "quorum-minus-one", 4, 2);
|
||||
let payload = vec![0x4b; 1024 * 1024];
|
||||
let opts = ObjectOptions {
|
||||
no_lock: true,
|
||||
..Default::default()
|
||||
};
|
||||
|
||||
set_disks
|
||||
.make_bucket(bucket, &MakeBucketOptions::default())
|
||||
.await
|
||||
.expect("bucket should be created");
|
||||
let mut put_reader = PutObjReader::from_vec(payload);
|
||||
set_disks
|
||||
.put_object(bucket, &object, &mut put_reader, &opts)
|
||||
.await
|
||||
.expect("object should be written");
|
||||
|
||||
let order = bounded_metadata_fanout_order(bucket, &object, 6, 2);
|
||||
for disk_index in order.iter().take(3) {
|
||||
let object_dir = dirs[*disk_index].path().join(bucket).join(&object);
|
||||
let data_dir = std::fs::read_dir(&object_dir)
|
||||
.expect("object directory should be readable")
|
||||
.find_map(|entry| {
|
||||
let entry = entry.expect("object directory entry should be readable");
|
||||
entry
|
||||
.file_type()
|
||||
.expect("entry type should be readable")
|
||||
.is_dir()
|
||||
.then(|| entry.path())
|
||||
})
|
||||
.expect("object data directory should exist");
|
||||
std::fs::remove_file(data_dir.join("part.1")).expect("selected shard should be removed");
|
||||
}
|
||||
|
||||
temp_env::async_with_vars(
|
||||
[
|
||||
("RUSTFS_GET_METADATA_TWO_PHASE_READ_PLAN_ENABLE", Some("true")),
|
||||
("RUSTFS_GET_METADATA_EARLY_STOP_ENABLE", Some("true")),
|
||||
("RUSTFS_GET_METADATA_EARLY_STOP_BOUNDED_FANOUT", Some("true")),
|
||||
],
|
||||
async {
|
||||
let result = set_disks
|
||||
.get_object_reader(bucket, &object, None, HeaderMap::new(), &opts)
|
||||
.await;
|
||||
assert!(result.is_err(), "quorum-minus-one read must fail closed without exposing a body");
|
||||
},
|
||||
)
|
||||
.await;
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
#[serial_test::serial(body_cache_hook)]
|
||||
async fn non_inline_data_read_early_stop_keeps_reserve_on_unequal_layout() {
|
||||
@@ -1641,13 +1335,10 @@ mod prepared_get_object_metadata_tests {
|
||||
.await;
|
||||
}
|
||||
|
||||
#[test]
|
||||
#[tokio::test]
|
||||
#[serial_test::serial(body_cache_hook)]
|
||||
fn non_inline_data_read_early_stop_does_not_add_inline_fanout_on_unequal_layout() {
|
||||
let runtime = tokio::runtime::Builder::new_current_thread()
|
||||
.enable_all()
|
||||
.build()
|
||||
.expect("current-thread runtime should build");
|
||||
async fn non_inline_data_read_early_stop_does_not_add_inline_fanout_on_unequal_layout() {
|
||||
let (_dirs, set_disks) = make_local_set_disks(6, 2).await;
|
||||
let bucket = "inline-read-plan-unequal";
|
||||
let object = object_with_initial_data_shards_for_geometry(bucket, "inline-object", 6, 2);
|
||||
let payload = b"inline quorum payload".repeat(256);
|
||||
@@ -1656,65 +1347,55 @@ mod prepared_get_object_metadata_tests {
|
||||
..Default::default()
|
||||
};
|
||||
|
||||
let (_dirs, set_disks) = runtime.block_on(async {
|
||||
let (dirs, set_disks) = make_local_set_disks(6, 2).await;
|
||||
set_disks
|
||||
.make_bucket(bucket, &MakeBucketOptions::default())
|
||||
.await
|
||||
.expect("bucket should be created");
|
||||
let mut put_reader = PutObjReader::from_vec(payload.clone());
|
||||
set_disks
|
||||
.put_object(bucket, &object, &mut put_reader, &opts)
|
||||
.await
|
||||
.expect("inline object should be written");
|
||||
(dirs, set_disks)
|
||||
});
|
||||
set_disks
|
||||
.make_bucket(bucket, &MakeBucketOptions::default())
|
||||
.await
|
||||
.expect("bucket should be created");
|
||||
let mut put_reader = PutObjReader::from_vec(payload.clone());
|
||||
set_disks
|
||||
.put_object(bucket, &object, &mut put_reader, &opts)
|
||||
.await
|
||||
.expect("inline object should be written");
|
||||
|
||||
// disk_call_counters counts tasks that started running, so an
|
||||
// early-stop abort races the single-pending inline hedge into a ±1
|
||||
// count per read. The fanout lifecycle histogram records the
|
||||
// scheduling decision itself and stays deterministic under load.
|
||||
let recorder = CapturingRecorder::default();
|
||||
let previous_gate = rustfs_io_metrics::get_stage_metrics_enabled();
|
||||
rustfs_io_metrics::set_get_stage_metrics_enabled(true);
|
||||
metrics::with_local_recorder(&recorder, || {
|
||||
runtime.block_on(async {
|
||||
for enabled in [false, true] {
|
||||
temp_env::async_with_vars(
|
||||
[
|
||||
(
|
||||
"RUSTFS_GET_METADATA_TWO_PHASE_READ_PLAN_ENABLE",
|
||||
Some(if enabled { "true" } else { "false" }),
|
||||
),
|
||||
("RUSTFS_GET_METADATA_EARLY_STOP_ENABLE", Some("true")),
|
||||
("RUSTFS_GET_METADATA_EARLY_STOP_BOUNDED_FANOUT", Some("true")),
|
||||
],
|
||||
async {
|
||||
let mut reader = set_disks
|
||||
.get_object_reader(bucket, &object, None, HeaderMap::new(), &opts)
|
||||
.await
|
||||
.expect("inline GET reader should open");
|
||||
let mut restored = Vec::new();
|
||||
reader
|
||||
.stream
|
||||
.read_to_end(&mut restored)
|
||||
.await
|
||||
.expect("inline GET body should stream");
|
||||
assert_eq!(restored, payload);
|
||||
},
|
||||
)
|
||||
.await;
|
||||
}
|
||||
})
|
||||
});
|
||||
rustfs_io_metrics::set_get_stage_metrics_enabled(previous_gate);
|
||||
let read_once = |enabled: bool| {
|
||||
let set_disks = Arc::clone(&set_disks);
|
||||
let bucket = bucket.to_string();
|
||||
let object = object.clone();
|
||||
let payload = payload.clone();
|
||||
let opts = opts.clone();
|
||||
async move {
|
||||
temp_env::async_with_vars(
|
||||
[
|
||||
(
|
||||
"RUSTFS_GET_METADATA_TWO_PHASE_READ_PLAN_ENABLE",
|
||||
Some(if enabled { "true" } else { "false" }),
|
||||
),
|
||||
("RUSTFS_GET_METADATA_EARLY_STOP_ENABLE", Some("true")),
|
||||
("RUSTFS_GET_METADATA_EARLY_STOP_BOUNDED_FANOUT", Some("true")),
|
||||
],
|
||||
async {
|
||||
let calls = disk_call_counters::observe(&object);
|
||||
let mut reader = set_disks
|
||||
.get_object_reader(&bucket, &object, None, HeaderMap::new(), &opts)
|
||||
.await
|
||||
.expect("inline GET reader should open");
|
||||
let mut restored = Vec::new();
|
||||
reader
|
||||
.stream
|
||||
.read_to_end(&mut restored)
|
||||
.await
|
||||
.expect("inline GET body should stream");
|
||||
assert_eq!(restored, payload);
|
||||
calls.total(disk_call_counters::KIND_READ_VERSION)
|
||||
},
|
||||
)
|
||||
.await
|
||||
}
|
||||
};
|
||||
|
||||
let scheduled = recorder.histogram_values(
|
||||
"rustfs_io_get_object_metadata_fanout_scheduled",
|
||||
&[("path", GET_OBJECT_PATH_LEGACY_DUPLEX)],
|
||||
);
|
||||
assert_eq!(scheduled.len(), 2, "each GET should run exactly one metadata fanout");
|
||||
assert_eq!(scheduled[1], scheduled[0], "inline gate must not add reserve fanout");
|
||||
let gate_off_calls = read_once(false).await;
|
||||
let gate_on_calls = read_once(true).await;
|
||||
assert_eq!(gate_on_calls, gate_off_calls, "inline gate must not add reserve fanout");
|
||||
}
|
||||
|
||||
#[test]
|
||||
@@ -2659,15 +2340,6 @@ fn should_use_codec_streaming(config: GetCodecStreamingConfig, bucket: &str, obj
|
||||
is_optimization_enabled_for_request(config.enabled, config.rollout_pct, bucket, object)
|
||||
}
|
||||
|
||||
pub(in crate::set_disk) fn codec_streaming_rollout_applies(bucket: &str, object: &str) -> bool {
|
||||
let config = get_codec_streaming_config();
|
||||
config.enabled
|
||||
&& config.body_compat_confirmed
|
||||
&& config.header_compat_confirmed
|
||||
&& config.rollout.is_opted_in()
|
||||
&& should_use_codec_streaming(config, bucket, object)
|
||||
}
|
||||
|
||||
/// Should this specific request use metadata early-stop?
|
||||
#[allow(
|
||||
dead_code,
|
||||
@@ -5595,15 +5267,10 @@ async fn check_object_lock_delete(
|
||||
return Ok(());
|
||||
}
|
||||
|
||||
// An authorized replicated version purge already passed this gate on the
|
||||
// source with the bypass it carried there, so it clears GOVERNANCE
|
||||
// retention here without the header; COMPLIANCE and legal hold still
|
||||
// block below (see `replication_delete_may_bypass_governance`, #6850).
|
||||
let bypass_governance = opts
|
||||
.object_lock_delete
|
||||
.as_ref()
|
||||
.is_some_and(|delete_opts| delete_opts.bypass_governance)
|
||||
|| replication_delete_may_bypass_governance(opts);
|
||||
.is_some_and(|delete_opts| delete_opts.bypass_governance);
|
||||
let blocked = match opts.object_lock_config_snapshot.as_deref() {
|
||||
Some(snapshot) => check_object_lock_for_deletion_with_state(snapshot.state(), obj_info, bypass_governance)?.is_some(),
|
||||
None => {
|
||||
@@ -8791,111 +8458,6 @@ mod tests {
|
||||
assert!(root.join("bucket").exists(), "bucket volume should remain");
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
async fn purge_orphan_dir_object_removes_committed_delete_residue() {
|
||||
let (dir, disk) = make_single_local_disk().await;
|
||||
let root = dir.path();
|
||||
let data_dir = Uuid::new_v4();
|
||||
let transaction = Uuid::new_v4();
|
||||
let residue = root
|
||||
.join("bucket")
|
||||
.join("pfx")
|
||||
.join("nested")
|
||||
.join("object")
|
||||
.join(data_dir.to_string());
|
||||
fs::create_dir_all(&residue)
|
||||
.await
|
||||
.expect("committed delete residue should be created");
|
||||
fs::write(residue.join("part.1"), b"stale")
|
||||
.await
|
||||
.expect("stale part should be written");
|
||||
fs::write(
|
||||
residue.join(format!("{}{}", crate::disk::local::DELETE_DATA_DIR_MARKER_PREFIX, transaction)),
|
||||
[],
|
||||
)
|
||||
.await
|
||||
.expect("committed delete marker should be written");
|
||||
|
||||
let set = make_set_disks_with(vec![Some(disk)]).await;
|
||||
let purged = set
|
||||
.purge_orphan_dir_object("bucket", "pfx/")
|
||||
.await
|
||||
.expect("purge should succeed");
|
||||
|
||||
assert!(purged, "committed delete residue should be purgeable");
|
||||
assert!(!root.join("bucket").join("pfx").exists(), "prefix directory should be gone");
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
async fn purge_orphan_dir_object_preserves_uncommitted_data_residue() {
|
||||
let (dir, disk) = make_single_local_disk().await;
|
||||
let root = dir.path();
|
||||
let residue = root
|
||||
.join("bucket")
|
||||
.join("pfx")
|
||||
.join("object")
|
||||
.join(Uuid::new_v4().to_string());
|
||||
fs::create_dir_all(&residue)
|
||||
.await
|
||||
.expect("uncommitted data residue should be created");
|
||||
fs::write(residue.join("part.1"), b"possibly live")
|
||||
.await
|
||||
.expect("data part should be written");
|
||||
fs::write(residue.join("delete-data.not-a-uuid"), [])
|
||||
.await
|
||||
.expect("malformed marker should be written");
|
||||
|
||||
let set = make_set_disks_with(vec![Some(disk)]).await;
|
||||
let purged = set
|
||||
.purge_orphan_dir_object("bucket", "pfx/")
|
||||
.await
|
||||
.expect("scan should succeed");
|
||||
|
||||
assert!(!purged, "data without a valid committed marker must be preserved");
|
||||
assert!(residue.join("part.1").exists(), "possibly live data must remain");
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
async fn purge_orphan_dir_object_preserves_nested_object_below_committed_residue() {
|
||||
let (dir, disk) = make_single_local_disk().await;
|
||||
let root = dir.path();
|
||||
let transaction = Uuid::new_v4();
|
||||
let residue = root
|
||||
.join("bucket")
|
||||
.join("pfx")
|
||||
.join("object")
|
||||
.join(Uuid::new_v4().to_string());
|
||||
fs::create_dir_all(&residue)
|
||||
.await
|
||||
.expect("committed data directory should be created");
|
||||
fs::write(residue.join("part.1"), b"stale")
|
||||
.await
|
||||
.expect("stale part should be written");
|
||||
fs::write(
|
||||
residue.join(format!("{}{}", crate::disk::local::DELETE_DATA_DIR_MARKER_PREFIX, transaction)),
|
||||
[],
|
||||
)
|
||||
.await
|
||||
.expect("committed delete marker should be written");
|
||||
let nested_object = residue.join("nested");
|
||||
fs::create_dir_all(&nested_object)
|
||||
.await
|
||||
.expect("nested object directory should be created");
|
||||
fs::write(nested_object.join(STORAGE_FORMAT_FILE), b"meta")
|
||||
.await
|
||||
.expect("nested object metadata should be written");
|
||||
|
||||
let set = make_set_disks_with(vec![Some(disk)]).await;
|
||||
let purged = set
|
||||
.purge_orphan_dir_object("bucket", "pfx/")
|
||||
.await
|
||||
.expect("scan should succeed");
|
||||
|
||||
assert!(!purged, "nested object metadata must veto committed-residue cleanup");
|
||||
assert!(nested_object.join(STORAGE_FORMAT_FILE).exists(), "nested object metadata must remain");
|
||||
assert!(residue.join("part.1").exists(), "committed residue must remain when cleanup is vetoed");
|
||||
}
|
||||
|
||||
// issue #4189: a prefix that still anchors a real object must be left intact.
|
||||
#[tokio::test]
|
||||
async fn purge_orphan_dir_object_preserves_prefix_with_object() {
|
||||
@@ -11709,100 +11271,6 @@ mod tests {
|
||||
.expect("versioned delete marker creation should not delete the locked version");
|
||||
}
|
||||
|
||||
fn governance_retained_obj_info() -> ObjectInfo {
|
||||
let retain_until = OffsetDateTime::now_utc() + Duration::from_secs(60 * 60 * 24 * 60);
|
||||
let mut user_defined = HashMap::new();
|
||||
user_defined.insert(
|
||||
X_AMZ_OBJECT_LOCK_MODE.as_str().to_string(),
|
||||
s3s::dto::ObjectLockRetentionMode::GOVERNANCE.to_string(),
|
||||
);
|
||||
user_defined.insert(
|
||||
X_AMZ_OBJECT_LOCK_RETAIN_UNTIL_DATE.as_str().to_string(),
|
||||
retain_until.format(&time::format_description::well_known::Rfc3339).unwrap(),
|
||||
);
|
||||
ObjectInfo {
|
||||
user_defined: Arc::new(user_defined),
|
||||
..Default::default()
|
||||
}
|
||||
}
|
||||
|
||||
fn explicit_version_delete_opts(replication_request: bool) -> ObjectOptions {
|
||||
ObjectOptions {
|
||||
version_id: Some(Uuid::new_v4().to_string()),
|
||||
versioned: true,
|
||||
replication_request,
|
||||
object_lock_config_snapshot: Some(Arc::new(ObjectLockConfigSnapshot::new(ObjectLockConfigState::ConfirmedAbsent))),
|
||||
..Default::default()
|
||||
}
|
||||
}
|
||||
|
||||
// Issue #6850: a replicated version purge carries no bypass header, so the
|
||||
// GOVERNANCE gate must honor the source's already-judged bypass instead of
|
||||
// keeping the sites permanently diverged.
|
||||
#[tokio::test]
|
||||
async fn test_check_object_lock_delete_allows_replicated_governance_version_purge() {
|
||||
let obj_info = governance_retained_obj_info();
|
||||
let opts = explicit_version_delete_opts(true);
|
||||
|
||||
check_object_lock_delete(&bootstrap_ctx(), "bucket", "object", &obj_info, &opts)
|
||||
.await
|
||||
.expect("an authorized replicated version purge must pass GOVERNANCE retention (#6850)");
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
async fn test_check_object_lock_delete_blocks_plain_governance_version_delete_without_bypass() {
|
||||
let obj_info = governance_retained_obj_info();
|
||||
let opts = explicit_version_delete_opts(false);
|
||||
|
||||
let err = check_object_lock_delete(&bootstrap_ctx(), "bucket", "object", &obj_info, &opts)
|
||||
.await
|
||||
.expect_err("a plain client delete without the bypass header must stay blocked by GOVERNANCE retention");
|
||||
|
||||
assert!(matches!(err, StorageError::PrefixAccessDenied(_, _)));
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
async fn test_check_object_lock_delete_blocks_replicated_compliance_version_purge() {
|
||||
let retain_until = OffsetDateTime::now_utc() + Duration::from_secs(60 * 60 * 24 * 60);
|
||||
let mut user_defined = HashMap::new();
|
||||
user_defined.insert(
|
||||
X_AMZ_OBJECT_LOCK_MODE.as_str().to_string(),
|
||||
s3s::dto::ObjectLockRetentionMode::COMPLIANCE.to_string(),
|
||||
);
|
||||
user_defined.insert(
|
||||
X_AMZ_OBJECT_LOCK_RETAIN_UNTIL_DATE.as_str().to_string(),
|
||||
retain_until.format(&time::format_description::well_known::Rfc3339).unwrap(),
|
||||
);
|
||||
let obj_info = ObjectInfo {
|
||||
user_defined: Arc::new(user_defined),
|
||||
..Default::default()
|
||||
};
|
||||
let opts = explicit_version_delete_opts(true);
|
||||
|
||||
let err = check_object_lock_delete(&bootstrap_ctx(), "bucket", "object", &obj_info, &opts)
|
||||
.await
|
||||
.expect_err("the source gate can never purge through COMPLIANCE, so a replicated purge fails closed");
|
||||
|
||||
assert!(matches!(err, StorageError::PrefixAccessDenied(_, _)));
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
async fn test_check_object_lock_delete_blocks_replicated_legal_hold_version_purge() {
|
||||
let mut user_defined = HashMap::new();
|
||||
user_defined.insert(X_AMZ_OBJECT_LOCK_LEGAL_HOLD.as_str().to_string(), "ON".to_string());
|
||||
let obj_info = ObjectInfo {
|
||||
user_defined: Arc::new(user_defined),
|
||||
..Default::default()
|
||||
};
|
||||
let opts = explicit_version_delete_opts(true);
|
||||
|
||||
let err = check_object_lock_delete(&bootstrap_ctx(), "bucket", "object", &obj_info, &opts)
|
||||
.await
|
||||
.expect_err("the source gate can never purge through a legal hold, so a replicated purge fails closed");
|
||||
|
||||
assert!(matches!(err, StorageError::PrefixAccessDenied(_, _)));
|
||||
}
|
||||
|
||||
// backlog#929 (HP-8): the delete_objects per-object stat is gated on the
|
||||
// bucket object-lock configuration. Lock-enabled buckets (either legacy
|
||||
// lock_enabled flag or an enabled ObjectLockConfiguration) and unknown
|
||||
@@ -12782,7 +12250,6 @@ mod tests {
|
||||
0,
|
||||
true,
|
||||
false,
|
||||
false,
|
||||
GET_OBJECT_PATH_LEGACY_DUPLEX,
|
||||
GET_CODEC_STREAMING_OBJECT_CLASS_PLAIN_SINGLE_PART,
|
||||
metrics_size_bucket,
|
||||
@@ -12895,7 +12362,6 @@ mod tests {
|
||||
0,
|
||||
true,
|
||||
false,
|
||||
false,
|
||||
GET_OBJECT_PATH_LEGACY_DUPLEX,
|
||||
GET_CODEC_STREAMING_OBJECT_CLASS_PLAIN_SINGLE_PART,
|
||||
metrics_size_bucket,
|
||||
|
||||
@@ -36,20 +36,6 @@ const EVENT_HEAL_OBJECT_RENAME: &str = "heal_object_rename";
|
||||
const HEAL_RENAME_INCOMPLETE: &str = "heal rename incomplete";
|
||||
const READ_REPAIR_DATA_PHASE_TIMEOUT: std::time::Duration = std::time::Duration::from_secs(60 * 60);
|
||||
|
||||
fn heal_drive_state_for_error(error: &DiskError) -> DriveState {
|
||||
match error {
|
||||
DiskError::DiskNotFound | DiskError::RemoteClientUnavailable(_) => DriveState::Offline,
|
||||
DiskError::FaultyDisk | DiskError::FaultyRemoteDisk => DriveState::Faulty,
|
||||
DiskError::FileNotFound
|
||||
| DiskError::FileVersionNotFound
|
||||
| DiskError::VolumeNotFound
|
||||
| DiskError::PartMissingOrCorrupt
|
||||
| DiskError::OutdatedXLMeta => DriveState::Missing,
|
||||
DiskError::FileCorrupt => DriveState::Corrupt,
|
||||
_ => DriveState::Unknown(error.to_string()),
|
||||
}
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
static HEAL_RENAME_FAILURES: std::sync::Mutex<Vec<(String, String, usize)>> = std::sync::Mutex::new(Vec::new());
|
||||
|
||||
@@ -906,7 +892,16 @@ impl SetDisks {
|
||||
}
|
||||
|
||||
let drive_state = match reason {
|
||||
Some(err) => heal_drive_state_for_error(&err).to_string(),
|
||||
Some(err) => match err {
|
||||
DiskError::DiskNotFound => DriveState::Offline.to_string(),
|
||||
DiskError::FileNotFound
|
||||
| DiskError::FileVersionNotFound
|
||||
| DiskError::VolumeNotFound
|
||||
| DiskError::PartMissingOrCorrupt
|
||||
| DiskError::OutdatedXLMeta => DriveState::Missing.to_string(),
|
||||
DiskError::FileCorrupt => DriveState::Corrupt.to_string(),
|
||||
_ => DriveState::Unknown(err.to_string()).to_string(),
|
||||
},
|
||||
None => DriveState::Ok.to_string(),
|
||||
};
|
||||
result.before.drives.push(HealDriveInfo {
|
||||
@@ -2678,17 +2673,6 @@ mod heal_result_report_tests {
|
||||
assert!(!super::metadata_less_part_file("xl.meta"));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn unavailable_heal_errors_use_stable_drive_states() {
|
||||
for error in [DiskError::FaultyDisk, DiskError::FaultyRemoteDisk] {
|
||||
assert_eq!(super::heal_drive_state_for_error(&error).to_string(), DriveState::Faulty.to_string());
|
||||
}
|
||||
assert_eq!(
|
||||
super::heal_drive_state_for_error(&DiskError::RemoteClientUnavailable("peer restarting".to_string())).to_string(),
|
||||
DriveState::Offline.to_string()
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn read_repair_commit_fingerprint_tracks_commit_identity_only() {
|
||||
let data_dir = Uuid::parse_str("11111111-1111-1111-1111-111111111111").expect("data dir should parse");
|
||||
@@ -2812,20 +2796,9 @@ mod heal_result_report_tests {
|
||||
}
|
||||
|
||||
let mut reader = PutObjReader::from_vec(vec![0x5a; 1024 * 1024]);
|
||||
// This fixture reads and removes physical shards immediately after
|
||||
// PUT. A lock-owning PUT may quorum-ack before its rename tail
|
||||
// drains, so keep the setup on the full-fanout commit path.
|
||||
set.put_object(
|
||||
&bucket,
|
||||
object,
|
||||
&mut reader,
|
||||
&ObjectOptions {
|
||||
no_lock: true,
|
||||
..Default::default()
|
||||
},
|
||||
)
|
||||
.await
|
||||
.expect("source object should be written");
|
||||
set.put_object(&bucket, object, &mut reader, &ObjectOptions::default())
|
||||
.await
|
||||
.expect("source object should be written");
|
||||
let source = disks[2]
|
||||
.read_version("", &bucket, object, "", &ReadOptions::default())
|
||||
.await
|
||||
@@ -3092,20 +3065,9 @@ mod heal_result_report_tests {
|
||||
|
||||
let payload = vec![0x5a; 1024 * 1024];
|
||||
let mut reader = PutObjReader::from_vec(payload);
|
||||
// This fixture removes physical shards immediately after PUT. A
|
||||
// lock-owning PUT may quorum-ack before its rename tail drains, so
|
||||
// keep the isolated setup on the full-fanout commit path.
|
||||
set.put_object(
|
||||
&bucket,
|
||||
object,
|
||||
&mut reader,
|
||||
&ObjectOptions {
|
||||
no_lock: true,
|
||||
..Default::default()
|
||||
},
|
||||
)
|
||||
.await
|
||||
.expect("source object should be written");
|
||||
set.put_object(&bucket, object, &mut reader, &ObjectOptions::default())
|
||||
.await
|
||||
.expect("source object should be written");
|
||||
let source = disks[2]
|
||||
.read_version("", &bucket, object, "", &ReadOptions::default())
|
||||
.await
|
||||
@@ -3241,20 +3203,9 @@ mod heal_result_report_tests {
|
||||
}
|
||||
|
||||
let mut reader = PutObjReader::from_vec(vec![0x5a; 1024 * 1024]);
|
||||
// The target-evidence readback below asserts per-disk state right
|
||||
// after PUT. A lock-owning PUT may quorum-ack before its rename tail
|
||||
// drains, so keep the setup on the full-fanout commit path.
|
||||
set.put_object(
|
||||
bucket,
|
||||
object,
|
||||
&mut reader,
|
||||
&ObjectOptions {
|
||||
no_lock: true,
|
||||
..Default::default()
|
||||
},
|
||||
)
|
||||
.await
|
||||
.expect("source object should be written");
|
||||
set.put_object(bucket, object, &mut reader, &ObjectOptions::default())
|
||||
.await
|
||||
.expect("source object should be written");
|
||||
let source = disks[2]
|
||||
.read_version("", bucket, object, "", &ReadOptions::default())
|
||||
.await
|
||||
@@ -3301,9 +3252,6 @@ mod heal_result_report_tests {
|
||||
.await
|
||||
.expect("versioned bucket should be created");
|
||||
|
||||
// The per-version target-evidence readback below asserts per-disk
|
||||
// state right after PUT. A lock-owning PUT may quorum-ack before its
|
||||
// rename tail drains, so keep the setup on the full-fanout commit path.
|
||||
let mut old_reader = PutObjReader::from_vec(vec![0x5a; 1024 * 1024]);
|
||||
let old_info = set
|
||||
.put_object(
|
||||
@@ -3312,7 +3260,6 @@ mod heal_result_report_tests {
|
||||
&mut old_reader,
|
||||
&ObjectOptions {
|
||||
versioned: true,
|
||||
no_lock: true,
|
||||
..Default::default()
|
||||
},
|
||||
)
|
||||
@@ -3330,7 +3277,6 @@ mod heal_result_report_tests {
|
||||
&mut latest_reader,
|
||||
&ObjectOptions {
|
||||
versioned: true,
|
||||
no_lock: true,
|
||||
..Default::default()
|
||||
},
|
||||
)
|
||||
@@ -4352,20 +4298,9 @@ mod heal_result_report_tests {
|
||||
|
||||
const PAYLOAD_SIZE: usize = 1024 * 1024;
|
||||
let mut initial_reader = PutObjReader::from_vec(vec![0x11; PAYLOAD_SIZE]);
|
||||
// This fixture reads and removes physical shards immediately after
|
||||
// PUT. A lock-owning PUT may quorum-ack before its rename tail drains,
|
||||
// so keep the setup on the full-fanout commit path.
|
||||
set.put_object(
|
||||
bucket,
|
||||
object,
|
||||
&mut initial_reader,
|
||||
&ObjectOptions {
|
||||
no_lock: true,
|
||||
..Default::default()
|
||||
},
|
||||
)
|
||||
.await
|
||||
.expect("initial object should be written");
|
||||
set.put_object(bucket, object, &mut initial_reader, &ObjectOptions::default())
|
||||
.await
|
||||
.expect("initial object should be written");
|
||||
|
||||
let current = disks[2]
|
||||
.read_version("", bucket, object, "", &ReadOptions::default())
|
||||
@@ -4453,12 +4388,11 @@ mod heal_result_report_tests {
|
||||
// Give the heal something to rebuild on alternating rounds: remove a
|
||||
// shard of the current data dir right before the race.
|
||||
if round % 2 == 1 {
|
||||
// The previous round's lock-owning PUT may still be
|
||||
// draining its rename tail on this disk; shard damage is
|
||||
// best-effort here, so skip injection when it lags.
|
||||
if let Ok(current) = disks[2].read_version("", bucket, object, "", &ReadOptions::default()).await
|
||||
&& let Some(data_dir) = current.data_dir
|
||||
{
|
||||
let current = disks[2]
|
||||
.read_version("", bucket, object, "", &ReadOptions::default())
|
||||
.await
|
||||
.expect("current metadata should be readable");
|
||||
if let Some(data_dir) = current.data_dir {
|
||||
let shard = temp_dirs[3]
|
||||
.path()
|
||||
.join(bucket)
|
||||
|
||||
@@ -27,18 +27,18 @@ use super::super::MetadataCacheInvalidationProbe;
|
||||
#[cfg(test)]
|
||||
use super::super::capacity_scope_from_disks;
|
||||
use super::super::{
|
||||
AMZ_STORAGE_CLASS, Arc, Bytes, CompletePart, Cursor, DATA_MOVEMENT_MULTIPART_PREFIX, DiskError, DiskStore,
|
||||
EVENT_SET_DISK_MULTIPART, Error, FileInfo, GLOBAL_MIN_PART_SIZE, HashAlgorithm, HashMap, HashReader, HashSet,
|
||||
HealChannelPriority, Instant, LOG_COMPONENT_ECSTORE, LOG_SUBSYSTEM_SET_DISK, ListMultipartsInfo, ListPartsInfo,
|
||||
MAX_PARTS_COUNT, MULTIPART_WRITE_QUORUM_RENAME_PART, MULTIPART_WRITE_QUORUM_UPLOAD_METADATA,
|
||||
MULTIPART_WRITE_QUORUM_WRITER_SETUP, MultipartInfo, MultipartUploadResult, MultipartWriteQuorumContext, NamespaceLockFence,
|
||||
OBJECT_OP_IGNORED_ERRS, ObjectInfo, ObjectLockDiagGuard, ObjectOptions, ObjectPartInfo, OffsetDateTime, PartInfo,
|
||||
PutObjReader, RUSTFS_META_MULTIPART_BUCKET, RUSTFS_META_TMP_BUCKET, RUSTFS_MULTIPART_BUCKET_KEY, RUSTFS_MULTIPART_OBJECT_KEY,
|
||||
Result, SLASH_SEPARATOR, SUFFIX_ACTUAL_OBJECT_SIZE_CAP, SUFFIX_ACTUAL_SIZE, SUFFIX_BUCKET_INCARNATION_ID,
|
||||
SUFFIX_COMPRESSION_SIZE, SUFFIX_REPLICATION_SSEC_CRC, SUFFIX_RESTORE_OPERATION_ID, SetDisks, SmallWritePath, StorageError,
|
||||
Uuid, WriteLayout, check_object_lock_for_deletion_with_state, classify_multipart_part_write_path, coding,
|
||||
complete_multipart_part_error, complete_multipart_part_error_result, complete_part_checksum, completed_multipart_object_part,
|
||||
contains_key_str, create_bitrot_writer, debug, disk, error, get_complete_multipart_md5, get_header_map, get_str, insert_str,
|
||||
AMZ_STORAGE_CLASS, Arc, Bytes, CompletePart, Cursor, DiskError, DiskStore, EVENT_SET_DISK_MULTIPART, Error, FileInfo,
|
||||
GLOBAL_MIN_PART_SIZE, HashAlgorithm, HashMap, HashReader, HashSet, HealChannelPriority, Instant, LOG_COMPONENT_ECSTORE,
|
||||
LOG_SUBSYSTEM_SET_DISK, ListMultipartsInfo, ListPartsInfo, MAX_PARTS_COUNT, MULTIPART_WRITE_QUORUM_RENAME_PART,
|
||||
MULTIPART_WRITE_QUORUM_UPLOAD_METADATA, MULTIPART_WRITE_QUORUM_WRITER_SETUP, MultipartInfo, MultipartUploadResult,
|
||||
MultipartWriteQuorumContext, NamespaceLockFence, OBJECT_OP_IGNORED_ERRS, ObjectInfo, ObjectLockDiagGuard, ObjectOptions,
|
||||
ObjectPartInfo, OffsetDateTime, PartInfo, PutObjReader, RUSTFS_META_MULTIPART_BUCKET, RUSTFS_META_TMP_BUCKET,
|
||||
RUSTFS_MULTIPART_BUCKET_KEY, RUSTFS_MULTIPART_OBJECT_KEY, Result, SLASH_SEPARATOR, SUFFIX_ACTUAL_OBJECT_SIZE_CAP,
|
||||
SUFFIX_ACTUAL_SIZE, SUFFIX_BUCKET_INCARNATION_ID, SUFFIX_COMPRESSION_SIZE, SUFFIX_REPLICATION_SSEC_CRC,
|
||||
SUFFIX_RESTORE_OPERATION_ID, SetDisks, SmallWritePath, StorageError, Uuid, WriteLayout,
|
||||
check_object_lock_for_deletion_with_state, classify_multipart_part_write_path, coding, complete_multipart_part_error,
|
||||
complete_multipart_part_error_result, complete_part_checksum, completed_multipart_object_part, contains_key_str,
|
||||
create_bitrot_writer, debug, disk, error, get_complete_multipart_md5, get_header_map, get_str, insert_str,
|
||||
is_err_object_not_found, is_err_version_not_found, is_min_allowed_part_size, log_multipart_write_quorum_failure,
|
||||
parts_after_marker, path_join_buf, record_compression_total_memory, reduce_read_quorum_errs, reduce_write_quorum_errs,
|
||||
remove_header_map, resolve_write_layout, restore_commit_operation_id_from_metadata, should_persist_encryption_original_size,
|
||||
@@ -183,45 +183,6 @@ pub(crate) struct StaleMultipartCleanupGuard {
|
||||
lock_guard: ObjectLockDiagGuard,
|
||||
}
|
||||
|
||||
pub(crate) struct DataMovementMultipartAbortGuard {
|
||||
upload_path: String,
|
||||
write_quorum: Option<usize>,
|
||||
lock_guard: ObjectLockDiagGuard,
|
||||
}
|
||||
|
||||
impl DataMovementMultipartAbortGuard {
|
||||
pub(crate) fn add_namespace_lock_fence(&self, opts: &mut ObjectOptions) {
|
||||
opts.add_namespace_lock_guard(&self.lock_guard.guard);
|
||||
}
|
||||
|
||||
pub(crate) async fn delete(&self, set: &SetDisks, bucket: &str, object: &str, opts: &ObjectOptions) -> Result<()> {
|
||||
#[cfg(any(test, feature = "test-util"))]
|
||||
pause_multipart_commit(bucket, object, MultipartCommitPause::AbortBeforeDelete).await;
|
||||
fence_commit_on_lock_loss(Some(&self.lock_guard), "abort_multipart_upload_commit", &self.upload_path)?;
|
||||
if opts
|
||||
.namespace_lock_fence
|
||||
.as_ref()
|
||||
.is_some_and(NamespaceLockFence::is_lock_lost)
|
||||
{
|
||||
return Err(StorageError::NamespaceLockQuorumUnavailable {
|
||||
mode: "abort_multipart_upload_outer_lock",
|
||||
bucket: bucket.to_string(),
|
||||
object: object.to_string(),
|
||||
required: 1,
|
||||
achieved: 0,
|
||||
});
|
||||
}
|
||||
ensure_multipart_bucket_lifecycle_lock_held(bucket, object, opts)?;
|
||||
if let Some(write_quorum) = self.write_quorum {
|
||||
set.delete_all_with_quorum(RUSTFS_META_MULTIPART_BUCKET, &self.upload_path, write_quorum)
|
||||
.await?;
|
||||
#[cfg(any(test, feature = "test-util"))]
|
||||
pause_multipart_commit(bucket, object, MultipartCommitPause::AbortAfterDelete).await;
|
||||
}
|
||||
Ok(())
|
||||
}
|
||||
}
|
||||
|
||||
impl StaleMultipartCleanupGuard {
|
||||
pub(crate) fn file_info(&self) -> &FileInfo {
|
||||
&self.file_info
|
||||
@@ -244,15 +205,12 @@ impl StaleMultipartCleanupGuard {
|
||||
}
|
||||
|
||||
#[cfg(any(test, feature = "test-util"))]
|
||||
#[derive(Clone, Copy, Debug, PartialEq, Eq)]
|
||||
#[derive(Clone, Copy, PartialEq, Eq)]
|
||||
pub enum MultipartCommitPause {
|
||||
NewUploadBeforeLockLost,
|
||||
PutPartBeforeLockAcquire,
|
||||
PutPartBeforeLockLost,
|
||||
PutPartAfterCapacityAdmission,
|
||||
PutPartAfterRename,
|
||||
AbortBeforeDelete,
|
||||
AbortAfterDelete,
|
||||
BeforeLockLost,
|
||||
BeforeQuotaRename,
|
||||
BeforeTransactionEpochVerify,
|
||||
@@ -715,28 +673,12 @@ fn is_corrupt_upload_metadata_error(err: &DiskError) -> bool {
|
||||
)
|
||||
}
|
||||
|
||||
async fn multipart_upload_paths_on_disk(disk: DiskStore, bucket: &str, root_prefix: &str) -> disk::error::Result<Vec<String>> {
|
||||
async fn multipart_upload_paths_on_disk(disk: DiskStore, bucket: &str) -> disk::error::Result<Vec<String>> {
|
||||
if !disk.is_online().await {
|
||||
return Err(DiskError::DiskNotFound);
|
||||
}
|
||||
|
||||
if !root_prefix.is_empty() {
|
||||
let upload_dirs = match disk.list_dir(bucket, RUSTFS_META_MULTIPART_BUCKET, root_prefix, -1).await {
|
||||
Ok(entries) => entries,
|
||||
Err(DiskError::FileNotFound | DiskError::VolumeNotFound) => return Ok(Vec::new()),
|
||||
Err(err) => return Err(err),
|
||||
};
|
||||
return Ok(upload_dirs
|
||||
.into_iter()
|
||||
.filter_map(|upload_dir| {
|
||||
let upload_dir = upload_dir.trim_end_matches('/');
|
||||
(!upload_dir.is_empty() && upload_dir != "." && upload_dir != ".." && !upload_dir.contains(['/', '\\']))
|
||||
.then(|| format!("{root_prefix}/{upload_dir}"))
|
||||
})
|
||||
.collect());
|
||||
}
|
||||
|
||||
let sha_dirs = match disk.list_dir(bucket, RUSTFS_META_MULTIPART_BUCKET, root_prefix, -1).await {
|
||||
let sha_dirs = match disk.list_dir(bucket, RUSTFS_META_MULTIPART_BUCKET, "", -1).await {
|
||||
Ok(entries) => entries,
|
||||
Err(DiskError::FileNotFound | DiskError::VolumeNotFound) => return Ok(Vec::new()),
|
||||
Err(err) => return Err(err),
|
||||
@@ -836,7 +778,6 @@ impl SetDisks {
|
||||
&self,
|
||||
orig_bucket: &str,
|
||||
error_path: &str,
|
||||
root_prefix: &str,
|
||||
) -> Result<(Vec<Option<DiskStore>>, Vec<String>, usize)> {
|
||||
let disks = self.disks.read().await.clone();
|
||||
if disks.is_empty() {
|
||||
@@ -853,10 +794,9 @@ impl SetDisks {
|
||||
for (index, disk) in disks.iter().enumerate() {
|
||||
let disk = disk.clone();
|
||||
let orig_bucket = orig_bucket.to_string();
|
||||
let root_prefix = root_prefix.to_string();
|
||||
discovery_tasks.spawn(async move {
|
||||
let result = match disk {
|
||||
Some(disk) => multipart_upload_paths_on_disk(disk, &orig_bucket, &root_prefix).await,
|
||||
Some(disk) => multipart_upload_paths_on_disk(disk, &orig_bucket).await,
|
||||
None => Err(DiskError::DiskNotFound),
|
||||
};
|
||||
(index, result)
|
||||
@@ -892,64 +832,11 @@ impl SetDisks {
|
||||
|
||||
pub(crate) async fn first_multipart_upload_path_for_decommission(&self, bucket: &str) -> Result<Option<String>> {
|
||||
let (_, paths, _) = self
|
||||
.discover_multipart_upload_paths(bucket, RUSTFS_META_MULTIPART_BUCKET, "")
|
||||
.discover_multipart_upload_paths(bucket, RUSTFS_META_MULTIPART_BUCKET)
|
||||
.await?;
|
||||
Ok(paths.into_iter().next())
|
||||
}
|
||||
|
||||
pub(crate) async fn data_movement_multipart_upload_ids(
|
||||
&self,
|
||||
bucket: &str,
|
||||
object: &str,
|
||||
expected_incarnation_id: Option<Uuid>,
|
||||
upload_identity: &str,
|
||||
) -> Result<Vec<String>> {
|
||||
let expected_parent = format!("{DATA_MOVEMENT_MULTIPART_PREFIX}/{}", Self::get_multipart_sha_dir(bucket, object));
|
||||
let (_, candidate_paths, _) = self.discover_multipart_upload_paths(bucket, object, &expected_parent).await?;
|
||||
let mut upload_ids = Vec::new();
|
||||
for upload_path in candidate_paths {
|
||||
let Some((parent, raw_upload_id)) = upload_path.rsplit_once('/') else {
|
||||
continue;
|
||||
};
|
||||
if parent != expected_parent || raw_upload_id.is_empty() {
|
||||
continue;
|
||||
}
|
||||
let upload_id = runtime_sources::deployment_upload_id(raw_upload_id);
|
||||
let file_info = match self
|
||||
.check_multipart_upload_path_exists(bucket, object, &upload_id, &upload_path, false)
|
||||
.await
|
||||
{
|
||||
Ok((file_info, _)) => file_info,
|
||||
Err(err) if crate::error::is_err_invalid_upload_id(&err) || crate::error::is_err_object_not_found(&err) => {
|
||||
continue;
|
||||
}
|
||||
Err(err) => return Err(err),
|
||||
};
|
||||
if file_info.metadata.get(RUSTFS_MULTIPART_BUCKET_KEY).map(String::as_str) != Some(bucket)
|
||||
|| file_info.metadata.get(RUSTFS_MULTIPART_OBJECT_KEY).map(String::as_str) != Some(object)
|
||||
{
|
||||
return Err(Error::other("data movement multipart upload target metadata is inconsistent"));
|
||||
}
|
||||
if expected_incarnation_id
|
||||
.is_some_and(|expected| !multipart_bucket_incarnation_matches(&file_info.metadata, expected))
|
||||
{
|
||||
return Err(Error::other("data movement multipart upload bucket incarnation is inconsistent"));
|
||||
}
|
||||
let Some(actual_upload_identity) =
|
||||
rustfs_utils::http::get_consistent_str(&file_info.metadata, rustfs_utils::http::SUFFIX_DATA_MOVEMENT_UPLOAD)
|
||||
else {
|
||||
return Err(Error::other("data movement multipart upload identity is inconsistent"));
|
||||
};
|
||||
if actual_upload_identity != upload_identity {
|
||||
continue;
|
||||
}
|
||||
upload_ids.push(upload_id);
|
||||
}
|
||||
upload_ids.sort_unstable();
|
||||
upload_ids.dedup();
|
||||
Ok(upload_ids)
|
||||
}
|
||||
|
||||
async fn acquire_multipart_upload_read_lock(
|
||||
&self,
|
||||
op: &'static str,
|
||||
@@ -986,55 +873,6 @@ impl SetDisks {
|
||||
.map(Some)
|
||||
}
|
||||
|
||||
pub(crate) async fn lock_data_movement_multipart_abort(
|
||||
&self,
|
||||
bucket: &str,
|
||||
object: &str,
|
||||
upload_id: &str,
|
||||
expected_upload_identity: Option<&str>,
|
||||
opts: &ObjectOptions,
|
||||
) -> Result<Option<DataMovementMultipartAbortGuard>> {
|
||||
let upload_path = Self::get_multipart_upload_dir(bucket, object, upload_id, true);
|
||||
let lock_guard = self
|
||||
.acquire_write_lock_diag("abort_data_movement_multipart", RUSTFS_META_MULTIPART_BUCKET, &upload_path)
|
||||
.await?;
|
||||
let file_info = match self
|
||||
.check_multipart_upload_path_exists(bucket, object, upload_id, &upload_path, true)
|
||||
.await
|
||||
{
|
||||
Ok((file_info, _)) => file_info,
|
||||
Err(err) if crate::error::is_err_invalid_upload_id(&err) || crate::error::is_err_object_not_found(&err) => {
|
||||
return Ok(Some(DataMovementMultipartAbortGuard {
|
||||
upload_path,
|
||||
write_quorum: None,
|
||||
lock_guard,
|
||||
}));
|
||||
}
|
||||
Err(err) => return Err(err),
|
||||
};
|
||||
ensure_data_movement_upload_access(&file_info, bucket, object, upload_id, opts)?;
|
||||
let upload_identity =
|
||||
rustfs_utils::http::get_consistent_str(&file_info.metadata, rustfs_utils::http::SUFFIX_DATA_MOVEMENT_UPLOAD);
|
||||
if upload_identity.is_none() || expected_upload_identity.is_some_and(|expected| upload_identity != Some(expected)) {
|
||||
return Err(StorageError::InvalidUploadID(bucket.to_owned(), object.to_owned(), upload_id.to_owned()));
|
||||
}
|
||||
ensure_multipart_bucket_incarnation(
|
||||
&self.ctx,
|
||||
&file_info,
|
||||
bucket,
|
||||
object,
|
||||
upload_id,
|
||||
opts.expected_bucket_incarnation_id,
|
||||
)
|
||||
.await?;
|
||||
ensure_multipart_bucket_lifecycle_lock_held(bucket, object, opts)?;
|
||||
Ok(Some(DataMovementMultipartAbortGuard {
|
||||
upload_path,
|
||||
write_quorum: Some(file_info.write_quorum(self.default_write_quorum())),
|
||||
lock_guard,
|
||||
}))
|
||||
}
|
||||
|
||||
pub(super) async fn list_parts(
|
||||
disks: &[Option<DiskStore>],
|
||||
part_path: &str,
|
||||
@@ -1190,7 +1028,7 @@ impl SetDisks {
|
||||
max_uploads: usize,
|
||||
expected_incarnation_id: Option<Uuid>,
|
||||
) -> Result<ListMultipartsInfo> {
|
||||
let (disks, candidate_paths, discovery_quorum) = self.discover_multipart_upload_paths(bucket, prefix, "").await?;
|
||||
let (disks, candidate_paths, discovery_quorum) = self.discover_multipart_upload_paths(bucket, prefix).await?;
|
||||
let listed_uploads = stream::iter(candidate_paths)
|
||||
.map(|upload_path| {
|
||||
let disks = &disks;
|
||||
@@ -1786,27 +1624,7 @@ impl crate::storage_api_contracts::multipart::MultipartOperations for SetDisks {
|
||||
admitted_multipart_size(current_size, candidate_size, limit)?;
|
||||
}
|
||||
|
||||
let decommission_capacity_guard = if let Some(store) = opts.decommission_capacity_admission.as_ref() {
|
||||
Some(
|
||||
store
|
||||
.acquire_external_decommission_capacity_fence(&[self.pool_index], "mutation")
|
||||
.await?,
|
||||
)
|
||||
} else {
|
||||
None
|
||||
};
|
||||
#[cfg(test)]
|
||||
pause_multipart_commit(bucket, object, MultipartCommitPause::PutPartAfterCapacityAdmission).await;
|
||||
if decommission_capacity_guard.as_ref().is_some_and(|guard| guard.is_lock_lost()) {
|
||||
return Err(StorageError::NamespaceLockQuorumUnavailable {
|
||||
mode: "put_object_part_decommission_capacity",
|
||||
bucket: bucket.to_string(),
|
||||
object: object.to_string(),
|
||||
required: 1,
|
||||
achieved: 0,
|
||||
});
|
||||
}
|
||||
let rename_result = self
|
||||
let _ = self
|
||||
.rename_part(
|
||||
&shuffle_disks,
|
||||
RUSTFS_META_TMP_BUCKET,
|
||||
@@ -1823,9 +1641,7 @@ impl crate::storage_api_contracts::multipart::MultipartOperations for SetDisks {
|
||||
part_number: Some(part_id),
|
||||
}),
|
||||
)
|
||||
.await;
|
||||
drop(decommission_capacity_guard);
|
||||
let _ = rename_result?;
|
||||
.await?;
|
||||
#[cfg(test)]
|
||||
observe_multipart_commit(bucket, object, MultipartCommitPause::PutPartBeforeLockLost);
|
||||
|
||||
@@ -2296,9 +2112,6 @@ impl crate::storage_api_contracts::multipart::MultipartOperations for SetDisks {
|
||||
let upload_id_path = Self::get_multipart_upload_dir(bucket, object, upload_id, opts.data_movement);
|
||||
let range_seek_rollout_enabled = crate::object_api::legacy_encrypted_range_seek_enabled() && !opts.no_lock;
|
||||
let mut object_lock_guard = None;
|
||||
let mut decommission_object_lock_guard = None;
|
||||
let mut decommission_target_lock_covered = false;
|
||||
let mut decommission_capacity_guard = None;
|
||||
|
||||
if opts.http_preconditions.is_some() {
|
||||
if !opts.no_lock {
|
||||
@@ -2313,25 +2126,7 @@ impl crate::storage_api_contracts::multipart::MultipartOperations for SetDisks {
|
||||
}
|
||||
}
|
||||
|
||||
if let Some(store) = opts.decommission_capacity_admission.as_ref() {
|
||||
#[cfg(test)]
|
||||
{
|
||||
crate::core::pools::notify_decommission_external_object_commit_phase_started(store.id);
|
||||
crate::core::pools::wait_for_decommission_external_object_commit_phase_release(store.id).await;
|
||||
}
|
||||
let (object_guard, target_lock_covered, capacity_guard) = store
|
||||
.acquire_external_decommission_commit_guards(
|
||||
self.pool_index,
|
||||
bucket,
|
||||
object,
|
||||
opts.no_lock || object_lock_guard.is_some(),
|
||||
)
|
||||
.await?;
|
||||
decommission_object_lock_guard = object_guard;
|
||||
decommission_target_lock_covered = target_lock_covered;
|
||||
decommission_capacity_guard = capacity_guard;
|
||||
}
|
||||
if !opts.no_lock && object_lock_guard.is_none() && !decommission_target_lock_covered {
|
||||
if !opts.no_lock && object_lock_guard.is_none() {
|
||||
object_lock_guard = Some(
|
||||
self.acquire_write_lock_diag("complete_multipart_upload_commit", bucket, object)
|
||||
.await?,
|
||||
@@ -2933,11 +2728,7 @@ impl crate::storage_api_contracts::multipart::MultipartOperations for SetDisks {
|
||||
// so a lost lock leaves the upload intact and retryable.
|
||||
#[cfg(test)]
|
||||
pause_multipart_commit(bucket, object, MultipartCommitPause::BeforeLockLost).await;
|
||||
if object_lock_guard.as_ref().is_some_and(|guard| guard.is_lock_lost())
|
||||
|| decommission_object_lock_guard
|
||||
.as_ref()
|
||||
.is_some_and(|guard| guard.is_lock_lost())
|
||||
{
|
||||
if object_lock_guard.as_ref().is_some_and(|guard| guard.is_lock_lost()) {
|
||||
return Err(StorageError::NamespaceLockQuorumUnavailable {
|
||||
mode: "complete_multipart_upload_commit",
|
||||
bucket: bucket.to_string(),
|
||||
@@ -3014,11 +2805,7 @@ impl crate::storage_api_contracts::multipart::MultipartOperations for SetDisks {
|
||||
Err(err) => return Err(err),
|
||||
}
|
||||
}
|
||||
if object_lock_guard.as_ref().is_some_and(|guard| guard.is_lock_lost())
|
||||
|| decommission_object_lock_guard
|
||||
.as_ref()
|
||||
.is_some_and(|guard| guard.is_lock_lost())
|
||||
{
|
||||
if object_lock_guard.as_ref().is_some_and(|guard| guard.is_lock_lost()) {
|
||||
return Err(StorageError::NamespaceLockQuorumUnavailable {
|
||||
mode: "complete_multipart_upload_commit",
|
||||
bucket: bucket.to_string(),
|
||||
@@ -3051,19 +2838,6 @@ impl crate::storage_api_contracts::multipart::MultipartOperations for SetDisks {
|
||||
}
|
||||
ensure_multipart_bucket_lifecycle_lock_held(bucket, object, opts)?;
|
||||
|
||||
// Complete has already acquired the object and upload-id namespaces.
|
||||
// Recheck decommission capacity only after those locks, and retain the
|
||||
// guard through the durable rename below.
|
||||
if decommission_capacity_guard.is_none()
|
||||
&& let Some(store) = opts.decommission_capacity_admission.as_ref()
|
||||
{
|
||||
decommission_capacity_guard = Some(
|
||||
store
|
||||
.acquire_external_decommission_capacity_fence(&[self.pool_index], "mutation")
|
||||
.await?,
|
||||
);
|
||||
}
|
||||
|
||||
let transaction_fencing_proof = object_transaction_fencing_fleet_proof();
|
||||
if object_transaction_fencing_requested() && transaction_fencing_proof.is_none() {
|
||||
return Err(Error::other("object transaction fencing requires a live fleet capability proof"));
|
||||
@@ -3122,16 +2896,11 @@ impl crate::storage_api_contracts::multipart::MultipartOperations for SetDisks {
|
||||
let commit_is_versioned = opts.versioned || opts.version_suspended;
|
||||
let commit_capacity_scope_token = opts.capacity_scope_token;
|
||||
let commit_object_lock_guard = object_lock_guard.take();
|
||||
let commit_decommission_object_lock_guard = decommission_object_lock_guard.take();
|
||||
let commit_decommission_capacity_guard = decommission_capacity_guard.take();
|
||||
let commit_allows_early_ack = !(opts.data_movement && opts.has_decommission_capacity_reservation())
|
||||
&& (commit_object_lock_guard.is_some() || commit_decommission_object_lock_guard.is_some());
|
||||
let commit_allows_early_ack = commit_object_lock_guard.is_some();
|
||||
let detach_commit_owner = commit_allows_early_ack || upload_guard.is_some() || quota_mutation_fence;
|
||||
let commit = async move {
|
||||
let mut _object_lock_guard = commit_object_lock_guard;
|
||||
let mut _decommission_object_lock_guard = commit_decommission_object_lock_guard;
|
||||
let mut _upload_guard = upload_guard;
|
||||
let mut _decommission_capacity_guard = commit_decommission_capacity_guard;
|
||||
let mut quota_reservation = quota_reservation;
|
||||
let complete_tail_stage_start = rustfs_io_metrics::put_stage_metrics_enabled().then(Instant::now);
|
||||
|
||||
@@ -3150,9 +2919,6 @@ impl crate::storage_api_contracts::multipart::MultipartOperations for SetDisks {
|
||||
if quota_reservation.is_lock_lost()
|
||||
|| !quota_reservation.capability_proof_matches()
|
||||
|| _object_lock_guard.as_ref().is_some_and(|guard| guard.is_lock_lost())
|
||||
|| _decommission_object_lock_guard
|
||||
.as_ref()
|
||||
.is_some_and(|guard| guard.is_lock_lost())
|
||||
|| _upload_guard.as_ref().is_some_and(|guard| guard.is_lock_lost())
|
||||
|| commit_namespace_lock_fence
|
||||
.as_ref()
|
||||
@@ -3160,9 +2926,6 @@ impl crate::storage_api_contracts::multipart::MultipartOperations for SetDisks {
|
||||
|| commit_bucket_lifecycle_lock_fence
|
||||
.as_ref()
|
||||
.is_some_and(NamespaceLockFence::is_lock_lost)
|
||||
|| _decommission_capacity_guard
|
||||
.as_ref()
|
||||
.is_some_and(|guard| guard.is_lock_lost())
|
||||
{
|
||||
return Err(StorageError::NamespaceLockQuorumUnavailable {
|
||||
mode: "quota_reservation",
|
||||
@@ -3204,9 +2967,6 @@ impl crate::storage_api_contracts::multipart::MultipartOperations for SetDisks {
|
||||
if quota_reservation.is_lock_lost()
|
||||
|| !quota_reservation.capability_proof_matches()
|
||||
|| _object_lock_guard.as_ref().is_some_and(|guard| guard.is_lock_lost())
|
||||
|| _decommission_object_lock_guard
|
||||
.as_ref()
|
||||
.is_some_and(|guard| guard.is_lock_lost())
|
||||
|| _upload_guard.as_ref().is_some_and(|guard| guard.is_lock_lost())
|
||||
|| commit_namespace_lock_fence
|
||||
.as_ref()
|
||||
@@ -3214,9 +2974,6 @@ impl crate::storage_api_contracts::multipart::MultipartOperations for SetDisks {
|
||||
|| commit_bucket_lifecycle_lock_fence
|
||||
.as_ref()
|
||||
.is_some_and(NamespaceLockFence::is_lock_lost)
|
||||
|| _decommission_capacity_guard
|
||||
.as_ref()
|
||||
.is_some_and(|guard| guard.is_lock_lost())
|
||||
{
|
||||
return Err(StorageError::NamespaceLockQuorumUnavailable {
|
||||
mode: "quota_reservation",
|
||||
@@ -3280,8 +3037,6 @@ impl crate::storage_api_contracts::multipart::MultipartOperations for SetDisks {
|
||||
.map(|version_id| version_id.to_string());
|
||||
let object_lock_guard = _object_lock_guard.take();
|
||||
let upload_guard = _upload_guard.take();
|
||||
let decommission_object_lock_guard = _decommission_object_lock_guard.take();
|
||||
let decommission_capacity_guard = _decommission_capacity_guard.take();
|
||||
let cleanup_bucket = commit_bucket.clone();
|
||||
let cleanup_object = commit_object.clone();
|
||||
let heal_set = commit_set.clone();
|
||||
@@ -3299,12 +3054,7 @@ impl crate::storage_api_contracts::multipart::MultipartOperations for SetDisks {
|
||||
tokio::spawn(finish_rename_tail_heal(
|
||||
rename_tail_drain,
|
||||
guard_release_rx,
|
||||
(
|
||||
object_lock_guard,
|
||||
upload_guard,
|
||||
decommission_object_lock_guard,
|
||||
decommission_capacity_guard,
|
||||
),
|
||||
(object_lock_guard, upload_guard),
|
||||
request,
|
||||
move || async move {
|
||||
if quota_mutation_fence {
|
||||
@@ -3318,8 +3068,7 @@ impl crate::storage_api_contracts::multipart::MultipartOperations for SetDisks {
|
||||
.await;
|
||||
}
|
||||
},
|
||||
move |(object_lock_guard, upload_guard, decommission_object_lock_guard, decommission_capacity_guard),
|
||||
targets| async move {
|
||||
move |(object_lock_guard, upload_guard), targets| async move {
|
||||
drop(object_lock_guard);
|
||||
cleanup_set.cleanup_multipart_path(&cleanup_parts).await;
|
||||
cleanup_set
|
||||
@@ -3344,16 +3093,11 @@ impl crate::storage_api_contracts::multipart::MultipartOperations for SetDisks {
|
||||
);
|
||||
}
|
||||
drop(upload_guard);
|
||||
drop(decommission_object_lock_guard);
|
||||
drop(decommission_capacity_guard);
|
||||
},
|
||||
|request| async move { heal_set.submit_rename_tail_heal(request).await },
|
||||
));
|
||||
}
|
||||
}
|
||||
if !tail_owns_staging_cleanup {
|
||||
drop(_decommission_capacity_guard.take());
|
||||
}
|
||||
if quota_mutation_fence && !tail_owns_staging_cleanup {
|
||||
let _ = SetDisks::release_quota_mutation_fences(
|
||||
&commit_disks,
|
||||
@@ -5459,42 +5203,19 @@ mod tests {
|
||||
disk.make_volume(bucket).await.expect("bucket volume should be created");
|
||||
}
|
||||
let mut initial_reader = PutObjReader::from_vec(b"old multipart body".to_vec());
|
||||
// A lock-owning PUT may quorum-ack before its rename tail drains, and
|
||||
// cache priming refuses to publish while a straggler disk still reads
|
||||
// as an error; keep the setup on the full-fanout commit path.
|
||||
set_disks
|
||||
.put_object(
|
||||
bucket,
|
||||
object,
|
||||
&mut initial_reader,
|
||||
&ObjectOptions {
|
||||
no_lock: true,
|
||||
..Default::default()
|
||||
},
|
||||
)
|
||||
.put_object(bucket, object, &mut initial_reader, &ObjectOptions::default())
|
||||
.await
|
||||
.expect("initial object should be written");
|
||||
// The publish is also bounded by the cache TTL, so re-prime until the
|
||||
// current generation is observably cached instead of asserting on a
|
||||
// single read that a loaded host can stall past expiry.
|
||||
let retired_key = tokio::time::timeout(std::time::Duration::from_secs(30), async {
|
||||
loop {
|
||||
set_disks
|
||||
.get_object_fileinfo(bucket, object, &ObjectOptions::default(), true, false)
|
||||
.await
|
||||
.expect("initial metadata should resolve");
|
||||
let generation = set_disks
|
||||
.get_object_metadata_cache_generation(bucket, object)
|
||||
.expect("metadata cache generation should be active");
|
||||
let key = GetObjectMetadataCacheKey::new(bucket, object, generation);
|
||||
if set_disks.get_object_metadata_cache.get(&key).await.is_some() {
|
||||
return key;
|
||||
}
|
||||
tokio::task::yield_now().await;
|
||||
}
|
||||
})
|
||||
.await
|
||||
.expect("metadata priming should publish the current generation");
|
||||
set_disks
|
||||
.get_object_fileinfo(bucket, object, &ObjectOptions::default(), true, false)
|
||||
.await
|
||||
.expect("initial metadata should resolve");
|
||||
let generation = set_disks
|
||||
.get_object_metadata_cache_generation(bucket, object)
|
||||
.expect("metadata cache generation should be active");
|
||||
let retired_key = GetObjectMetadataCacheKey::new(bucket, object, generation);
|
||||
assert!(set_disks.get_object_metadata_cache.get(&retired_key).await.is_some());
|
||||
|
||||
let upload = set_disks
|
||||
.new_multipart_upload(bucket, object, &ObjectOptions::default())
|
||||
|
||||
File diff suppressed because it is too large
Load Diff
@@ -609,20 +609,7 @@ impl SetDisks {
|
||||
|
||||
// let online_disks: Vec<Option<DiskStore>> = op_online_disks.iter().filter(|v| v.is_some()).cloned().collect();
|
||||
|
||||
if !metadata_fanout_complete
|
||||
&& allow_early_stop
|
||||
&& non_inline_data_read_early_stop_allowed(read_data, bucket, object)
|
||||
&& late_materialization_candidate_is_safe(&fi)
|
||||
{
|
||||
Ok(GetObjectFileInfo::owned_with_late_metadata_fanout(
|
||||
fi,
|
||||
parts_metadata,
|
||||
op_online_disks,
|
||||
disks,
|
||||
))
|
||||
} else {
|
||||
Ok(GetObjectFileInfo::owned(fi, parts_metadata, op_online_disks))
|
||||
}
|
||||
Ok(GetObjectFileInfo::owned(fi, parts_metadata, op_online_disks))
|
||||
}
|
||||
|
||||
#[hotpath::measure(impl_type = "SetDisks")]
|
||||
@@ -832,7 +819,6 @@ impl SetDisks {
|
||||
pool_index: usize,
|
||||
skip_verify_bitrot: bool,
|
||||
prefer_data_blocks_first_reader_setup: bool,
|
||||
require_reconstruction_surplus: bool,
|
||||
metrics_path: &'static str,
|
||||
metrics_object_class: &'static str,
|
||||
metrics_size_bucket: &'static str,
|
||||
@@ -1097,9 +1083,6 @@ impl SetDisks {
|
||||
}
|
||||
|
||||
let nil_count = reader_setup.available_shards();
|
||||
if require_reconstruction_surplus && nil_count <= erasure.data_shards {
|
||||
return Err(Error::other("insufficient reconstruction surplus for two-phase read"));
|
||||
}
|
||||
if nil_count < erasure.data_shards {
|
||||
if let Some(read_err) = reduce_read_quorum_errs(&reader_setup.errors, OBJECT_OP_IGNORED_ERRS, erasure.data_shards)
|
||||
{
|
||||
@@ -1207,34 +1190,18 @@ impl SetDisks {
|
||||
let readers = reader_setup.readers;
|
||||
let deferred_stripe_handles = reader_setup.deferred_stripe_handles;
|
||||
let deferred_reopeners = reader_setup.deferred_reopeners;
|
||||
let (written, err, exact_quorum) = if require_reconstruction_surplus {
|
||||
erasure
|
||||
.decode_with_stripe_handles_and_reopeners_with_diagnostics(
|
||||
writer,
|
||||
readers,
|
||||
part_offset,
|
||||
part_length,
|
||||
part_size,
|
||||
read_costs,
|
||||
deferred_stripe_handles,
|
||||
deferred_reopeners,
|
||||
)
|
||||
.await
|
||||
} else {
|
||||
let (written, err) = erasure
|
||||
.decode_with_stripe_handles_and_reopeners(
|
||||
writer,
|
||||
readers,
|
||||
part_offset,
|
||||
part_length,
|
||||
part_size,
|
||||
read_costs,
|
||||
deferred_stripe_handles,
|
||||
deferred_reopeners,
|
||||
)
|
||||
.await;
|
||||
(written, err, false)
|
||||
};
|
||||
let (written, err) = erasure
|
||||
.decode_with_stripe_handles_and_reopeners(
|
||||
writer,
|
||||
readers,
|
||||
part_offset,
|
||||
part_length,
|
||||
part_size,
|
||||
read_costs,
|
||||
deferred_stripe_handles,
|
||||
deferred_reopeners,
|
||||
)
|
||||
.await;
|
||||
let decode_elapsed = decode_stage_start.elapsed();
|
||||
rustfs_io_metrics::record_get_object_decode_duration(decode_elapsed.as_secs_f64());
|
||||
rustfs_io_metrics::record_get_object_stage_duration_by_size(
|
||||
@@ -1244,9 +1211,6 @@ impl SetDisks {
|
||||
metrics_size_bucket,
|
||||
decode_elapsed.as_secs_f64(),
|
||||
);
|
||||
if exact_quorum && err.is_none() {
|
||||
return Err(Error::other("two-phase read completed with exact reconstruction quorum"));
|
||||
}
|
||||
if decode_elapsed >= SLOW_OBJECT_READ_LOG_THRESHOLD && err.is_none() {
|
||||
warn!(
|
||||
event = EVENT_SET_DISK_READ,
|
||||
@@ -1794,102 +1758,6 @@ fn multipart_reader_setup_prefetch_enabled(policy: GetObjectReadPolicy) -> bool
|
||||
policy.allows_multipart_setup_prefetch() && is_multipart_reader_setup_prefetch_enabled()
|
||||
}
|
||||
|
||||
pub(super) struct LateMetadataIdentity {
|
||||
volume: String,
|
||||
name: String,
|
||||
algorithm: String,
|
||||
block_size: usize,
|
||||
uses_legacy_checksum: bool,
|
||||
quorum_hash: [u8; 32],
|
||||
distribution: Vec<usize>,
|
||||
parity_blocks: usize,
|
||||
}
|
||||
|
||||
impl LateMetadataIdentity {
|
||||
pub(super) fn from_file_info(file_info: &FileInfo) -> Self {
|
||||
Self {
|
||||
volume: file_info.volume.clone(),
|
||||
name: file_info.name.clone(),
|
||||
algorithm: file_info.erasure.algorithm.clone(),
|
||||
block_size: file_info.erasure.block_size,
|
||||
uses_legacy_checksum: file_info.uses_legacy_checksum,
|
||||
quorum_hash: SetDisks::file_info_quorum_hash(file_info),
|
||||
distribution: file_info.erasure.distribution.clone(),
|
||||
parity_blocks: file_info.erasure.parity_blocks,
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
fn late_metadata_read_identity_matches(expected: &LateMetadataIdentity, actual: &FileInfo) -> bool {
|
||||
expected.volume == actual.volume
|
||||
&& expected.name == actual.name
|
||||
&& expected.algorithm == actual.erasure.algorithm
|
||||
&& expected.block_size == actual.erasure.block_size
|
||||
&& expected.uses_legacy_checksum == actual.uses_legacy_checksum
|
||||
&& expected.quorum_hash == SetDisks::file_info_quorum_hash(actual)
|
||||
}
|
||||
|
||||
fn late_metadata_shard_matches(expected: &LateMetadataIdentity, actual: &FileInfo, disk_index: usize) -> bool {
|
||||
expected
|
||||
.distribution
|
||||
.get(disk_index)
|
||||
.is_some_and(|mapped_index| *mapped_index == actual.erasure.index)
|
||||
&& late_metadata_read_identity_matches(expected, actual)
|
||||
}
|
||||
|
||||
impl SetDisks {
|
||||
pub(super) async fn refresh_late_metadata_fanout(
|
||||
fallback_disks: &[Option<DiskStore>],
|
||||
bucket: &str,
|
||||
object: &str,
|
||||
expected: &LateMetadataIdentity,
|
||||
metrics_path: &'static str,
|
||||
) -> Result<(FileInfo, Vec<FileInfo>, Vec<Option<DiskStore>>)> {
|
||||
let (mut parts_metadata, errs, diagnostics) = SetDisks::read_all_fileinfo_observed(
|
||||
fallback_disks,
|
||||
"",
|
||||
bucket,
|
||||
object,
|
||||
"",
|
||||
true,
|
||||
false,
|
||||
false,
|
||||
false,
|
||||
expected.parity_blocks,
|
||||
)
|
||||
.await?;
|
||||
diagnostics.record(metrics_path);
|
||||
|
||||
let (read_quorum, write_quorum) = SetDisks::object_quorum_from_meta(&parts_metadata, &errs, expected.parity_blocks)
|
||||
.map_err(|err| to_object_err(err.into(), vec![bucket, object]))?;
|
||||
let read_quorum =
|
||||
usize::try_from(read_quorum).map_err(|_| to_object_err(DiskError::ErasureReadQuorum.into(), vec![bucket, object]))?;
|
||||
let write_quorum = usize::try_from(write_quorum)
|
||||
.map_err(|_| to_object_err(DiskError::ErasureWriteQuorum.into(), vec![bucket, object]))?;
|
||||
if let Some(err) = reduce_read_quorum_errs(&errs, OBJECT_OP_IGNORED_ERRS, read_quorum) {
|
||||
return Err(to_object_err(err.into(), vec![bucket, object]));
|
||||
}
|
||||
|
||||
let (mut online_disks, full_fi, _) =
|
||||
SetDisks::select_valid_fileinfo(fallback_disks, &parts_metadata, &errs, "", read_quorum, write_quorum)?;
|
||||
if !late_metadata_read_identity_matches(expected, &full_fi) {
|
||||
return Err(to_object_err(DiskError::ErasureReadQuorum.into(), vec![bucket, object]));
|
||||
}
|
||||
|
||||
for (disk_index, (metadata, disk)) in parts_metadata.iter_mut().zip(online_disks.iter_mut()).enumerate() {
|
||||
if !late_metadata_shard_matches(expected, metadata, disk_index) {
|
||||
*metadata = FileInfo::default();
|
||||
*disk = None;
|
||||
}
|
||||
}
|
||||
if online_disks.iter().filter(|disk| disk.is_some()).count() < read_quorum {
|
||||
return Err(to_object_err(DiskError::ErasureReadQuorum.into(), vec![bucket, object]));
|
||||
}
|
||||
|
||||
Ok((full_fi, parts_metadata, online_disks))
|
||||
}
|
||||
}
|
||||
|
||||
/// Run one part's bitrot reader setup and measure its wall-clock duration.
|
||||
///
|
||||
/// Shared by the synchronous path and the prefetch task in
|
||||
@@ -2481,7 +2349,6 @@ mod metadata_cache_tests {
|
||||
0,
|
||||
false,
|
||||
false,
|
||||
false,
|
||||
GET_OBJECT_PATH_SET_DISK,
|
||||
"plain",
|
||||
"small",
|
||||
@@ -2513,7 +2380,6 @@ mod metadata_cache_tests {
|
||||
0,
|
||||
false,
|
||||
false,
|
||||
false,
|
||||
GET_OBJECT_PATH_SET_DISK,
|
||||
"plain",
|
||||
"small",
|
||||
@@ -2538,7 +2404,6 @@ mod metadata_cache_tests {
|
||||
0,
|
||||
false,
|
||||
false,
|
||||
false,
|
||||
GET_OBJECT_PATH_SET_DISK,
|
||||
"plain",
|
||||
"small",
|
||||
@@ -2561,7 +2426,6 @@ mod metadata_cache_tests {
|
||||
0,
|
||||
false,
|
||||
false,
|
||||
false,
|
||||
GET_OBJECT_PATH_SET_DISK,
|
||||
"plain",
|
||||
"small",
|
||||
@@ -2586,7 +2450,6 @@ mod metadata_cache_tests {
|
||||
0,
|
||||
false,
|
||||
false,
|
||||
false,
|
||||
GET_OBJECT_PATH_SET_DISK,
|
||||
"plain",
|
||||
"small",
|
||||
@@ -2625,7 +2488,6 @@ mod metadata_cache_tests {
|
||||
0,
|
||||
false,
|
||||
false,
|
||||
false,
|
||||
GET_OBJECT_PATH_SET_DISK,
|
||||
"plain",
|
||||
"empty",
|
||||
@@ -2659,7 +2521,6 @@ mod metadata_cache_tests {
|
||||
0,
|
||||
false,
|
||||
false,
|
||||
false,
|
||||
GET_OBJECT_PATH_SET_DISK,
|
||||
"plain",
|
||||
"small",
|
||||
@@ -5021,7 +4882,6 @@ mod tests {
|
||||
0,
|
||||
false,
|
||||
false,
|
||||
false,
|
||||
GET_OBJECT_PATH_SET_DISK,
|
||||
"test-object-class",
|
||||
"test-size-bucket",
|
||||
|
||||
@@ -157,15 +157,7 @@ impl SetDisks {
|
||||
.clone()
|
||||
.unwrap_or_else(|| get_raw_etag(obj_info.user_defined.as_ref()));
|
||||
let version_id = expected.version_id.map(|v| v.to_string());
|
||||
let (decommission_object_lock_guard, decommission_target_lock_covered, mut decommission_capacity_guard) =
|
||||
if let Some(store) = opts.decommission_capacity_admission.as_ref() {
|
||||
store
|
||||
.acquire_external_decommission_commit_guards(self.pool_index, bucket, object, opts.no_lock)
|
||||
.await?
|
||||
} else {
|
||||
(None, false, None)
|
||||
};
|
||||
let lock_guard = if !opts.no_lock && !decommission_target_lock_covered {
|
||||
let lock_guard = if !opts.no_lock {
|
||||
Some(
|
||||
self.acquire_write_lock_diag("restore_finalize_metadata", bucket, object)
|
||||
.await?,
|
||||
@@ -173,15 +165,6 @@ impl SetDisks {
|
||||
} else {
|
||||
None
|
||||
};
|
||||
if decommission_capacity_guard.is_none()
|
||||
&& let Some(store) = opts.decommission_capacity_admission.as_ref()
|
||||
{
|
||||
decommission_capacity_guard = Some(
|
||||
store
|
||||
.acquire_external_decommission_capacity_fence(&[self.pool_index], "mutation")
|
||||
.await?,
|
||||
);
|
||||
}
|
||||
let read_opts = ObjectOptions {
|
||||
version_id,
|
||||
versioned: opts.versioned,
|
||||
@@ -215,12 +198,7 @@ impl SetDisks {
|
||||
);
|
||||
self.invalidate_get_object_metadata_cache(bucket, object).await;
|
||||
ensure_restore_metadata_lock_held(bucket, object, opts, "restore_finalize_metadata")?;
|
||||
if lock_guard.as_ref().is_some_and(|guard| guard.is_lock_lost())
|
||||
|| decommission_object_lock_guard
|
||||
.as_ref()
|
||||
.is_some_and(|guard| guard.is_lock_lost())
|
||||
|| decommission_capacity_guard.as_ref().is_some_and(|guard| guard.is_lock_lost())
|
||||
{
|
||||
if lock_guard.as_ref().is_some_and(|guard| guard.is_lock_lost()) {
|
||||
return Err(Error::other("restore finalization lock lost before metadata update"));
|
||||
}
|
||||
self.update_object_meta_with_opts(
|
||||
@@ -255,15 +233,7 @@ impl SetDisks {
|
||||
.clone()
|
||||
.unwrap_or_else(|| get_raw_etag(obj_info.user_defined.as_ref()));
|
||||
let version_id = expected.version_id.map(|v| v.to_string());
|
||||
let (decommission_object_lock_guard, decommission_target_lock_covered, mut decommission_capacity_guard) =
|
||||
if let Some(store) = opts.decommission_capacity_admission.as_ref() {
|
||||
store
|
||||
.acquire_external_decommission_commit_guards(self.pool_index, bucket, object, opts.no_lock)
|
||||
.await?
|
||||
} else {
|
||||
(None, false, None)
|
||||
};
|
||||
let lock_guard = if !opts.no_lock && !decommission_target_lock_covered {
|
||||
let lock_guard = if !opts.no_lock {
|
||||
Some(
|
||||
self.acquire_write_lock_diag("restore_cleanup_metadata", bucket, object)
|
||||
.await?,
|
||||
@@ -271,15 +241,6 @@ impl SetDisks {
|
||||
} else {
|
||||
None
|
||||
};
|
||||
if decommission_capacity_guard.is_none()
|
||||
&& let Some(store) = opts.decommission_capacity_admission.as_ref()
|
||||
{
|
||||
decommission_capacity_guard = Some(
|
||||
store
|
||||
.acquire_external_decommission_capacity_fence(&[self.pool_index], "mutation")
|
||||
.await?,
|
||||
);
|
||||
}
|
||||
let read_opts = ObjectOptions {
|
||||
version_id,
|
||||
versioned: opts.versioned,
|
||||
@@ -308,12 +269,7 @@ impl SetDisks {
|
||||
&mut fi.metadata,
|
||||
rustfs_utils::http::metadata_compat::SUFFIX_RESTORE_OPERATION_ID,
|
||||
);
|
||||
if lock_guard.as_ref().is_some_and(|guard| guard.is_lock_lost())
|
||||
|| decommission_object_lock_guard
|
||||
.as_ref()
|
||||
.is_some_and(|guard| guard.is_lock_lost())
|
||||
|| decommission_capacity_guard.as_ref().is_some_and(|guard| guard.is_lock_lost())
|
||||
{
|
||||
if lock_guard.as_ref().is_some_and(|guard| guard.is_lock_lost()) {
|
||||
return Err(Error::other("restore cleanup lock lost before metadata update".to_string()));
|
||||
}
|
||||
self.invalidate_get_object_metadata_cache(bucket, object).await;
|
||||
|
||||
@@ -29,25 +29,6 @@ use std::future::Future;
|
||||
|
||||
const DELETED_BUCKETS_PREFIX: &str = ".deleted";
|
||||
const SCANNER_BUCKET_LIST_SET_CONCURRENCY: usize = 4;
|
||||
const EVENT_BUCKET_DELETE_BLOCKED: &str = "bucket_delete_blocked";
|
||||
|
||||
fn record_bucket_delete_blocker(bucket: &str, kind: BucketDeleteBlockerKind, residue: &BucketMetadataLessResidue) {
|
||||
metrics::counter!("rustfs_bucket_delete_blockers_total", "kind" => kind.as_str()).increment(1);
|
||||
debug!(
|
||||
event = EVENT_BUCKET_DELETE_BLOCKED,
|
||||
component = "ecstore",
|
||||
subsystem = "bucket",
|
||||
bucket,
|
||||
blocker = kind.as_str(),
|
||||
files = residue.files,
|
||||
uuid_data_dirs = residue.uuid_data_dirs,
|
||||
entries_scanned = residue.entries_scanned,
|
||||
diagnostic_bytes_read = residue.diagnostic_bytes_read,
|
||||
diagnostic_truncated = residue.diagnostic_truncated,
|
||||
sample = residue.sample.as_deref().unwrap_or("<none>"),
|
||||
"Bucket deletion was blocked by durable local state"
|
||||
);
|
||||
}
|
||||
|
||||
fn scanner_bucket_list_set_concurrency(set_count: usize) -> usize {
|
||||
set_count.clamp(1, SCANNER_BUCKET_LIST_SET_CONCURRENCY)
|
||||
@@ -175,7 +156,6 @@ where
|
||||
async fn bucket_delete_local_blocker(
|
||||
ctx: &crate::runtime::instance::InstanceContext,
|
||||
bucket: &str,
|
||||
budget: &mut BucketDeleteDiagnosticBudget,
|
||||
) -> Result<Option<StorageError>> {
|
||||
let local_disks = runtime_sources::local_disks_in(ctx).await;
|
||||
let mut residue = BucketMetadataLessResidue::default();
|
||||
@@ -184,30 +164,18 @@ async fn bucket_delete_local_blocker(
|
||||
let Some(bucket_path) = disk.get_bucket_path_for_io_if_local(bucket) else {
|
||||
continue;
|
||||
};
|
||||
let scan = scan_metadata_less_residue_with_budget(&bucket_path?, budget).await?;
|
||||
let scan = scan_metadata_less_residue(&bucket_path?).await?;
|
||||
if scan.xlmeta_found {
|
||||
record_bucket_delete_blocker(bucket, scan.xlmeta_blocker.unwrap_or(BucketDeleteBlockerKind::UnknownXlMeta), &scan);
|
||||
return Ok(Some(StorageError::BucketNotEmpty(bucket.to_string())));
|
||||
}
|
||||
residue.files = residue.files.saturating_add(scan.files);
|
||||
residue.uuid_data_dirs = residue.uuid_data_dirs.saturating_add(scan.uuid_data_dirs);
|
||||
residue.entries_scanned = residue.entries_scanned.saturating_add(scan.entries_scanned);
|
||||
residue.diagnostic_bytes_read = residue.diagnostic_bytes_read.saturating_add(scan.diagnostic_bytes_read);
|
||||
if residue.sample.is_none() {
|
||||
residue.sample = scan.sample.clone();
|
||||
}
|
||||
if scan.diagnostic_truncated {
|
||||
residue.diagnostic_truncated = true;
|
||||
record_bucket_delete_blocker(bucket, BucketDeleteBlockerKind::DiagnosticBudgetExceeded, &residue);
|
||||
return Ok(Some(StorageError::BucketNotEmptyWithDetails {
|
||||
bucket: bucket.to_string(),
|
||||
details: residue.describe(),
|
||||
}));
|
||||
residue.sample = scan.sample;
|
||||
}
|
||||
}
|
||||
|
||||
if residue.has_residue_without_xlmeta() {
|
||||
record_bucket_delete_blocker(bucket, BucketDeleteBlockerKind::OrphanDirectory, &residue);
|
||||
return Ok(Some(StorageError::BucketNotEmptyWithDetails {
|
||||
bucket: bucket.to_string(),
|
||||
details: residue.describe(),
|
||||
@@ -815,14 +783,12 @@ impl ECStore {
|
||||
}
|
||||
}
|
||||
};
|
||||
let mut diagnostic_budget = None;
|
||||
|
||||
if bucket_exists {
|
||||
validate_table_bucket_delete_guard(&self.ctx, bucket).await?;
|
||||
|
||||
if !opts.force {
|
||||
let budget = diagnostic_budget.get_or_insert_with(BucketDeleteDiagnosticBudget::new);
|
||||
if let Some(blocker) = bucket_delete_local_blocker(&self.ctx, bucket, budget).await? {
|
||||
if let Some(blocker) = bucket_delete_local_blocker(&self.ctx, bucket).await? {
|
||||
return Err(blocker);
|
||||
}
|
||||
delete_opts.force_if_empty = true;
|
||||
@@ -861,12 +827,7 @@ impl ECStore {
|
||||
{
|
||||
if delete_opts.force_if_empty
|
||||
&& matches!(&err, StorageError::BucketNotEmpty(_))
|
||||
&& let Some(blocker) = bucket_delete_local_blocker(
|
||||
&self.ctx,
|
||||
bucket,
|
||||
diagnostic_budget.get_or_insert_with(BucketDeleteDiagnosticBudget::new),
|
||||
)
|
||||
.await?
|
||||
&& let Some(blocker) = bucket_delete_local_blocker(&self.ctx, bucket).await?
|
||||
{
|
||||
return Err(blocker);
|
||||
}
|
||||
@@ -895,17 +856,15 @@ impl ECStore {
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use super::{
|
||||
BUCKET_DELETE_DIAGNOSTIC_MAX_ENTRIES, BUCKET_DELETE_XLMETA_DIAGNOSTIC_MAX_BYTES, BucketDeleteBlockerKind,
|
||||
BucketDeleteDiagnosticBudget, SCANNER_BUCKET_LIST_SET_CONCURRENCY, await_bucket_namespace_operation,
|
||||
bucket_delete_metadata_cleanup_prefixes, bucket_deleted_marker_prefix, bucket_deleted_marker_volume,
|
||||
run_bucket_usage_cleanup, run_physical_bucket_deletion, scan_metadata_less_residue,
|
||||
scan_metadata_less_residue_with_budget, scanner_bucket_list_set_concurrency, should_override_created_from_metadata,
|
||||
SCANNER_BUCKET_LIST_SET_CONCURRENCY, await_bucket_namespace_operation, bucket_delete_metadata_cleanup_prefixes,
|
||||
bucket_deleted_marker_prefix, bucket_deleted_marker_volume, run_bucket_usage_cleanup, run_physical_bucket_deletion,
|
||||
scan_metadata_less_residue, scanner_bucket_list_set_concurrency, should_override_created_from_metadata,
|
||||
validate_table_bucket_delete_allowed,
|
||||
};
|
||||
use crate::bucket::metadata::table_bucket_catalog_metadata_prefix;
|
||||
use crate::bucket::metadata_sys;
|
||||
use crate::cluster::rpc::peer_s3_client::install_delete_bucket_empty_scan_barrier;
|
||||
use crate::disk::{BUCKET_META_PREFIX, RUSTFS_META_BUCKET, STORAGE_FORMAT_FILE};
|
||||
use crate::disk::{BUCKET_META_PREFIX, RUSTFS_META_BUCKET};
|
||||
use crate::error::StorageError;
|
||||
use crate::object_api::{ObjectOptions, PutObjReader};
|
||||
use crate::runtime::instance::InstanceContext;
|
||||
@@ -920,7 +879,6 @@ mod tests {
|
||||
layout::endpoints::{EndpointServerPools, Endpoints, PoolEndpoints},
|
||||
};
|
||||
use rustfs_data_usage::{BucketUsageInfo, DATA_USAGE_OBJECT_NAME, DataUsageInfo};
|
||||
use rustfs_filemeta::{FileInfo, FileMeta, TRANSITION_COMPLETE};
|
||||
use rustfs_lock::{LocalClient, LockRequest, LockType, NamespaceLock, ObjectKey};
|
||||
use serial_test::serial;
|
||||
use std::path::{Path, PathBuf};
|
||||
@@ -934,88 +892,6 @@ mod tests {
|
||||
|
||||
static BUCKET_DELETE_TEST_ENV: OnceCell<(Vec<PathBuf>, Arc<ECStore>)> = OnceCell::const_new();
|
||||
|
||||
#[tokio::test(start_paused = true)]
|
||||
async fn bucket_delete_diagnostic_budget_starts_with_first_scan_io_and_latches_once() {
|
||||
let mut budget = BucketDeleteDiagnosticBudget::with_limits(8, Duration::from_millis(100));
|
||||
tokio::time::advance(Duration::from_secs(10)).await;
|
||||
|
||||
let first_polled = Arc::new(AtomicBool::new(false));
|
||||
let first_polled_for_io = first_polled.clone();
|
||||
let first = budget
|
||||
.run_io(async move {
|
||||
first_polled_for_io.store(true, Ordering::SeqCst);
|
||||
Ok::<_, std::io::Error>(7_u8)
|
||||
})
|
||||
.await
|
||||
.expect("the first diagnostic IO should succeed");
|
||||
assert_eq!(first, Some(7));
|
||||
assert!(first_polled.load(Ordering::SeqCst));
|
||||
|
||||
tokio::time::advance(Duration::from_millis(101)).await;
|
||||
let expired_polled = Arc::new(AtomicBool::new(false));
|
||||
let expired_polled_for_io = expired_polled.clone();
|
||||
let expired = budget
|
||||
.run_io(async move {
|
||||
expired_polled_for_io.store(true, Ordering::SeqCst);
|
||||
Ok::<_, std::io::Error>(9_u8)
|
||||
})
|
||||
.await
|
||||
.expect("an expired diagnostic budget should not become an IO error");
|
||||
assert_eq!(expired, None);
|
||||
assert!(
|
||||
!expired_polled.load(Ordering::SeqCst),
|
||||
"the deadline must remain latched after the first scan IO"
|
||||
);
|
||||
}
|
||||
|
||||
#[tokio::test(start_paused = true)]
|
||||
async fn bucket_delete_diagnostic_budget_times_out_its_first_pending_io() {
|
||||
let mut budget = BucketDeleteDiagnosticBudget::with_limits(8, Duration::from_millis(100));
|
||||
let io_polled = Arc::new(AtomicBool::new(false));
|
||||
let io_polled_for_future = io_polled.clone();
|
||||
|
||||
let result = budget
|
||||
.run_io(std::future::poll_fn(move |_cx| {
|
||||
io_polled_for_future.store(true, Ordering::SeqCst);
|
||||
std::task::Poll::<std::io::Result<()>>::Pending
|
||||
}))
|
||||
.await
|
||||
.expect("a diagnostic timeout should fail closed without an IO error");
|
||||
|
||||
assert_eq!(result, None);
|
||||
assert!(
|
||||
io_polled.load(Ordering::SeqCst),
|
||||
"the first diagnostic IO must be polled before its timeout"
|
||||
);
|
||||
}
|
||||
|
||||
#[tokio::test(start_paused = true)]
|
||||
async fn delayed_metadata_less_scans_still_detect_orphans_and_xlmeta() {
|
||||
let root = tempfile::tempdir().expect("temporary delayed-scan roots should be created");
|
||||
let orphan_root = root.path().join("orphan-root");
|
||||
let xlmeta_root = root.path().join("xlmeta-root");
|
||||
std::fs::create_dir_all(&orphan_root).expect("orphan root should be created");
|
||||
std::fs::create_dir_all(&xlmeta_root).expect("xlmeta root should be created");
|
||||
std::fs::write(orphan_root.join("orphan-part"), b"orphan").expect("orphan fixture should be written");
|
||||
std::fs::write(xlmeta_root.join(STORAGE_FORMAT_FILE), b"invalid-xlmeta").expect("xl.meta fixture should be written");
|
||||
|
||||
let mut orphan_budget = BucketDeleteDiagnosticBudget::with_limits(16, Duration::from_secs(5));
|
||||
let mut xlmeta_budget = BucketDeleteDiagnosticBudget::with_limits(16, Duration::from_secs(5));
|
||||
tokio::time::advance(Duration::from_secs(60)).await;
|
||||
|
||||
let orphan = scan_metadata_less_residue_with_budget(&orphan_root, &mut orphan_budget)
|
||||
.await
|
||||
.expect("delayed orphan scan should complete");
|
||||
assert!(orphan.has_residue_without_xlmeta());
|
||||
assert!(!orphan.diagnostic_truncated);
|
||||
|
||||
let xlmeta = scan_metadata_less_residue_with_budget(&xlmeta_root, &mut xlmeta_budget)
|
||||
.await
|
||||
.expect("delayed xl.meta scan should complete");
|
||||
assert!(xlmeta.xlmeta_found);
|
||||
assert!(!xlmeta.diagnostic_truncated);
|
||||
}
|
||||
|
||||
#[tokio::test(start_paused = true)]
|
||||
async fn bucket_namespace_operation_fails_closed_after_lease_expiry() {
|
||||
let ttl = Duration::from_millis(20);
|
||||
@@ -1437,175 +1313,6 @@ mod tests {
|
||||
assert!(sample.ends_with("/part.1"));
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
async fn metadata_less_residue_scan_shares_one_entry_budget_across_roots() {
|
||||
let root = tempfile::tempdir().expect("temporary diagnostic roots should be created");
|
||||
let first_root = root.path().join("disk-a");
|
||||
let second_root = root.path().join("disk-b");
|
||||
tokio::fs::create_dir_all(&first_root)
|
||||
.await
|
||||
.expect("first diagnostic root should be created");
|
||||
tokio::fs::create_dir_all(&second_root)
|
||||
.await
|
||||
.expect("second diagnostic root should be created");
|
||||
for index in 0..3 {
|
||||
std::fs::write(first_root.join(format!("first-{index}")), b"").expect("first-root fixture should be written");
|
||||
std::fs::write(second_root.join(format!("second-{index}")), b"").expect("second-root fixture should be written");
|
||||
}
|
||||
|
||||
let mut budget = BucketDeleteDiagnosticBudget::with_limits(4, Duration::from_secs(5));
|
||||
let first = scan_metadata_less_residue_with_budget(&first_root, &mut budget)
|
||||
.await
|
||||
.expect("first root should fit the shared budget");
|
||||
assert!(!first.diagnostic_truncated);
|
||||
let second = scan_metadata_less_residue_with_budget(&second_root, &mut budget)
|
||||
.await
|
||||
.expect("second root should stop at the remaining shared budget");
|
||||
assert!(second.diagnostic_truncated);
|
||||
assert!(first.entries_scanned + second.entries_scanned <= 4);
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
async fn metadata_less_residue_scan_honors_an_expired_request_deadline() {
|
||||
let root = tempfile::tempdir().expect("temporary diagnostic root should be created");
|
||||
std::fs::write(root.path().join("orphan"), b"").expect("deadline fixture should be written");
|
||||
let mut budget = BucketDeleteDiagnosticBudget::with_limits(8, Duration::ZERO);
|
||||
|
||||
let scan = scan_metadata_less_residue_with_budget(root.path(), &mut budget)
|
||||
.await
|
||||
.expect("an expired diagnostic budget should fail closed without an IO error");
|
||||
|
||||
assert!(scan.diagnostic_truncated);
|
||||
assert_eq!(scan.entries_scanned, 0);
|
||||
assert_eq!(scan.diagnostic_bytes_read, 0);
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
async fn metadata_less_residue_scan_stops_at_diagnostic_budget() {
|
||||
let root = tempfile::tempdir().expect("temporary bucket root should be created");
|
||||
let bucket_path = root.path().join("bucket");
|
||||
tokio::fs::create_dir_all(&bucket_path)
|
||||
.await
|
||||
.expect("budget fixture directory should be created");
|
||||
for index in 0..(BUCKET_DELETE_DIAGNOSTIC_MAX_ENTRIES + 32) {
|
||||
std::fs::write(bucket_path.join(format!("orphan-{index:05}")), b"").expect("budget fixture file should be created");
|
||||
}
|
||||
|
||||
let residue = scan_metadata_less_residue(&bucket_path)
|
||||
.await
|
||||
.expect("budgeted residue scan should fail closed without an IO error");
|
||||
assert!(residue.diagnostic_truncated);
|
||||
assert!(residue.has_residue_without_xlmeta());
|
||||
assert!(!residue.xlmeta_found);
|
||||
assert!(residue.entries_scanned <= BUCKET_DELETE_DIAGNOSTIC_MAX_ENTRIES);
|
||||
assert!(residue.files <= BUCKET_DELETE_DIAGNOSTIC_MAX_ENTRIES);
|
||||
assert_eq!(residue.diagnostic_bytes_read, 0);
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
async fn bucket_residue_scan_distinguishes_visible_and_tier_free_xlmeta() {
|
||||
let root = tempfile::tempdir().expect("temporary bucket root should be created");
|
||||
let bucket_path = root.path().join("bucket");
|
||||
let visible_path = bucket_path.join("visible").join(STORAGE_FORMAT_FILE);
|
||||
tokio::fs::create_dir_all(visible_path.parent().expect("visible xl.meta should have a parent"))
|
||||
.await
|
||||
.expect("visible object directory should be created");
|
||||
let mut visible = FileMeta::new();
|
||||
visible
|
||||
.add_version(FileInfo {
|
||||
version_id: Some(Uuid::new_v4()),
|
||||
mod_time: Some(OffsetDateTime::now_utc()),
|
||||
..Default::default()
|
||||
})
|
||||
.expect("visible version should encode");
|
||||
tokio::fs::write(&visible_path, visible.marshal_msg().expect("visible xl.meta should marshal"))
|
||||
.await
|
||||
.expect("visible xl.meta should be written");
|
||||
|
||||
let visible_scan = scan_metadata_less_residue(&bucket_path)
|
||||
.await
|
||||
.expect("visible xl.meta scan should succeed");
|
||||
assert_eq!(visible_scan.xlmeta_blocker, Some(BucketDeleteBlockerKind::VisibleVersion));
|
||||
|
||||
tokio::fs::remove_dir_all(bucket_path.join("visible"))
|
||||
.await
|
||||
.expect("visible fixture should be removed");
|
||||
let free_path = bucket_path.join("free").join(STORAGE_FORMAT_FILE);
|
||||
tokio::fs::create_dir_all(free_path.parent().expect("free xl.meta should have a parent"))
|
||||
.await
|
||||
.expect("free-version object directory should be created");
|
||||
let source_version_id = Uuid::new_v4();
|
||||
let mut free = FileMeta::new();
|
||||
free.add_version(FileInfo {
|
||||
version_id: Some(source_version_id),
|
||||
transition_status: TRANSITION_COMPLETE.to_string(),
|
||||
transitioned_objname: "remote/object".to_string(),
|
||||
transition_version_id: Some(Uuid::new_v4()),
|
||||
transition_tier: "WARM".to_string(),
|
||||
mod_time: Some(OffsetDateTime::now_utc()),
|
||||
..Default::default()
|
||||
})
|
||||
.expect("transitioned source should encode");
|
||||
let mut delete = FileInfo {
|
||||
version_id: Some(source_version_id),
|
||||
mod_time: Some(OffsetDateTime::now_utc()),
|
||||
..Default::default()
|
||||
};
|
||||
delete.set_tier_free_version_id(&Uuid::new_v4().to_string());
|
||||
free.delete_version(&delete)
|
||||
.expect("transitioned source delete should create a free-version");
|
||||
tokio::fs::write(&free_path, free.marshal_msg().expect("free-version xl.meta should marshal"))
|
||||
.await
|
||||
.expect("free-version xl.meta should be written");
|
||||
|
||||
let free_scan = scan_metadata_less_residue(&bucket_path)
|
||||
.await
|
||||
.expect("free-version xl.meta scan should succeed");
|
||||
assert_eq!(free_scan.xlmeta_blocker, Some(BucketDeleteBlockerKind::TierFreeVersion));
|
||||
|
||||
tokio::fs::remove_dir_all(bucket_path.join("free"))
|
||||
.await
|
||||
.expect("free-version fixture should be removed");
|
||||
|
||||
let exact_limit_path = bucket_path.join("exact-limit").join(STORAGE_FORMAT_FILE);
|
||||
tokio::fs::create_dir_all(exact_limit_path.parent().expect("exact-limit xl.meta should have a parent"))
|
||||
.await
|
||||
.expect("exact-limit object directory should be created");
|
||||
let exact_limit = tokio::fs::File::create(&exact_limit_path)
|
||||
.await
|
||||
.expect("exact-limit xl.meta should be created");
|
||||
exact_limit
|
||||
.set_len(BUCKET_DELETE_XLMETA_DIAGNOSTIC_MAX_BYTES)
|
||||
.await
|
||||
.expect("exact-limit xl.meta should be extended without allocating its contents");
|
||||
let exact_limit_scan = scan_metadata_less_residue(&bucket_path)
|
||||
.await
|
||||
.expect("exact-limit xl.meta scan should remain fail closed");
|
||||
assert_eq!(exact_limit_scan.xlmeta_blocker, Some(BucketDeleteBlockerKind::UnknownXlMeta));
|
||||
assert_eq!(exact_limit_scan.diagnostic_bytes_read, BUCKET_DELETE_XLMETA_DIAGNOSTIC_MAX_BYTES);
|
||||
tokio::fs::remove_dir_all(bucket_path.join("exact-limit"))
|
||||
.await
|
||||
.expect("exact-limit fixture should be removed");
|
||||
|
||||
let oversized_path = bucket_path.join("oversized").join(STORAGE_FORMAT_FILE);
|
||||
tokio::fs::create_dir_all(oversized_path.parent().expect("oversized xl.meta should have a parent"))
|
||||
.await
|
||||
.expect("oversized object directory should be created");
|
||||
let oversized = tokio::fs::File::create(&oversized_path)
|
||||
.await
|
||||
.expect("oversized xl.meta should be created");
|
||||
oversized
|
||||
.set_len(BUCKET_DELETE_XLMETA_DIAGNOSTIC_MAX_BYTES + 1)
|
||||
.await
|
||||
.expect("oversized xl.meta should be extended without allocating its contents");
|
||||
let oversized_scan = scan_metadata_less_residue(&bucket_path)
|
||||
.await
|
||||
.expect("oversized xl.meta scan should remain fail closed");
|
||||
assert_eq!(oversized_scan.xlmeta_blocker, Some(BucketDeleteBlockerKind::UnknownXlMeta));
|
||||
assert_eq!(oversized_scan.diagnostic_bytes_read, 0);
|
||||
assert!(oversized_scan.diagnostic_bytes_read <= BUCKET_DELETE_XLMETA_DIAGNOSTIC_MAX_BYTES);
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
#[serial]
|
||||
async fn scanner_bucket_listing_unions_every_erasure_set() {
|
||||
|
||||
@@ -36,10 +36,6 @@ fn invalid_heal_pool_index(pool_idx: usize, pool_count: usize) -> Error {
|
||||
)
|
||||
}
|
||||
|
||||
fn is_pool_meta_object(bucket: &str, object: &str) -> bool {
|
||||
bucket == RUSTFS_META_BUCKET && object == POOL_META_NAME
|
||||
}
|
||||
|
||||
#[derive(Debug, Clone, Copy)]
|
||||
enum HealFormatPoolSkip {
|
||||
Completed,
|
||||
@@ -474,29 +470,8 @@ impl ECStore {
|
||||
let object = encode_dir_object(object);
|
||||
|
||||
let pools = self.get_pools_for_heal_object(opts)?;
|
||||
if let Some(set_idx) = opts.set {
|
||||
for pool in &pools {
|
||||
if set_idx >= pool.disk_set.len() {
|
||||
let err = StorageError::InvalidArgument(
|
||||
"heal".to_string(),
|
||||
"set".to_string(),
|
||||
format!(
|
||||
"invalid heal set index {set_idx} for pool {} with {} sets",
|
||||
pool.pool_idx,
|
||||
pool.disk_set.len()
|
||||
),
|
||||
);
|
||||
if opts.pool.is_some() {
|
||||
return Err(err);
|
||||
}
|
||||
return Ok((HealResultItem::default(), Some(err)));
|
||||
}
|
||||
}
|
||||
}
|
||||
#[cfg(test)]
|
||||
let store_id = self.id;
|
||||
|
||||
let mut heal_pools = Vec::with_capacity(pools.len());
|
||||
let mut futures = Vec::with_capacity(pools.len());
|
||||
for pool in pools.iter() {
|
||||
let suspended_complete = {
|
||||
let pool_meta = self.pool_meta.read().await;
|
||||
@@ -524,58 +499,9 @@ impl ECStore {
|
||||
}
|
||||
continue;
|
||||
}
|
||||
heal_pools.push(Arc::clone(pool));
|
||||
futures.push(pool.heal_object(bucket, &object, version_id, opts));
|
||||
}
|
||||
let results = if is_pool_meta_object(bucket, &object) && !opts.no_lock && !heal_pools.is_empty() {
|
||||
let target_pool_indices = heal_pools.iter().map(|pool| pool.pool_idx).collect::<Vec<_>>();
|
||||
match self.acquire_pool_meta_object_heal_fence(&target_pool_indices).await {
|
||||
Ok((pool_meta_guard, admissions)) => {
|
||||
let fixed_set = self.pools.first().and_then(|pool| pool.disk_set.first()).cloned();
|
||||
let futures = heal_pools.iter().zip(admissions).map(|(pool, admission)| {
|
||||
let pool = Arc::clone(pool);
|
||||
let pool_object = object.clone();
|
||||
let fixed_set = fixed_set.clone();
|
||||
let mut opts = *opts;
|
||||
async move {
|
||||
admission?;
|
||||
let fixed_set = fixed_set.ok_or_else(|| Error::other("pool metadata heal requires a fixed set"))?;
|
||||
let target_set = pool.get_disks_for_heal_object(&pool_object, &opts)?;
|
||||
opts.no_lock = fixed_set.shares_namespace_lock_domain(&target_set).await;
|
||||
#[cfg(test)]
|
||||
if !opts.no_lock {
|
||||
crate::core::pools::notify_decommission_external_heal_target_lock_attempted();
|
||||
}
|
||||
#[cfg(test)]
|
||||
crate::core::pools::notify_decommission_external_heal_operation_started(store_id);
|
||||
pool.heal_object(bucket, &pool_object, version_id, &opts).await
|
||||
}
|
||||
});
|
||||
let results = join_all(futures).await;
|
||||
drop(pool_meta_guard);
|
||||
results
|
||||
}
|
||||
Err(err) => (0..heal_pools.len()).map(|_| Err(err.clone())).collect(),
|
||||
}
|
||||
} else {
|
||||
let mut futures = Vec::with_capacity(heal_pools.len());
|
||||
for pool in heal_pools {
|
||||
let pool_idx = pool.pool_idx;
|
||||
let pool_object = object.clone();
|
||||
let opts = *opts;
|
||||
futures.push(self.run_external_decommission_capacity_heal(
|
||||
pool_idx,
|
||||
bucket,
|
||||
&object,
|
||||
opts,
|
||||
move |opts| async move {
|
||||
#[cfg(test)]
|
||||
crate::core::pools::notify_decommission_external_heal_operation_started(store_id);
|
||||
pool.heal_object(bucket, &pool_object, version_id, &opts).await
|
||||
},
|
||||
));
|
||||
}
|
||||
join_all(futures).await
|
||||
};
|
||||
let results = join_all(futures).await;
|
||||
|
||||
let mut errs = Vec::with_capacity(self.pools.len());
|
||||
let mut ress = Vec::with_capacity(self.pools.len());
|
||||
@@ -668,18 +594,14 @@ mod tests {
|
||||
use crate::cluster::rpc::PeerS3Client;
|
||||
use crate::config::com::{delete_config, read_config_no_lock_preserve_empty_with_metadata, save_config};
|
||||
use crate::core::pools::{
|
||||
DecommissionCapacityLockOrderBarrier, DecommissionErasureLayout, DecommissionPoolCapacityInfo, POOL_META_IDENTITY_NAME,
|
||||
PoolDecommissionInfo, PoolMetaReplicaState, PoolStatus, initialized_pool_meta_identity_for_test,
|
||||
set_decommission_capacity_info_overrides_for_test,
|
||||
POOL_META_IDENTITY_NAME, PoolDecommissionInfo, PoolMetaReplicaState, PoolStatus, initialized_pool_meta_identity_for_test,
|
||||
};
|
||||
use crate::core::sets::HealFormatAfterSaveBarrier;
|
||||
use crate::disk::error::Result as DiskResult;
|
||||
use crate::disk::{DeleteOptions, DiskOption, DiskStore, FORMAT_CONFIG_FILE, format::FormatV3, new_disk};
|
||||
use crate::disk::{DeleteOptions, DiskOption, FORMAT_CONFIG_FILE, format::FormatV3, new_disk};
|
||||
use crate::layout::endpoints::{EndpointServerPools, Endpoints, PoolEndpoints};
|
||||
use crate::runtime::instance::InstanceContext;
|
||||
use crate::services::rebalance::{
|
||||
RebalanceInfo, RebalanceStats, test_three_pool_stores_with_isolated_node_contexts, test_two_pool_stores,
|
||||
};
|
||||
use crate::services::rebalance::{RebalanceInfo, RebalanceStats};
|
||||
use crate::storage_api_contracts::bucket::{
|
||||
BucketInfo, BucketOperations, BucketOptions, DeleteBucketOptions, MakeBucketOptions,
|
||||
};
|
||||
@@ -821,36 +743,11 @@ mod tests {
|
||||
decommission_cancelers: RwLock::new(Vec::new()),
|
||||
start_gate: Mutex::new(()),
|
||||
pool_meta_save_gate: Mutex::default(),
|
||||
decommission_capacity_entry_gate: Mutex::default(),
|
||||
ctx: crate::runtime::instance::bootstrap_ctx(),
|
||||
bucket_fence_registry: std::sync::Arc::default(),
|
||||
}
|
||||
}
|
||||
|
||||
async fn remove_pool_meta_shard(store: &ECStore, pool_idx: usize) -> DiskStore {
|
||||
let target_set = store.pools[pool_idx].get_disks_by_key(POOL_META_NAME);
|
||||
let missing_disk = target_set.disks.read().await[0]
|
||||
.clone()
|
||||
.expect("pool metadata fixture disk should be online");
|
||||
missing_disk
|
||||
.delete(
|
||||
RUSTFS_META_BUCKET,
|
||||
POOL_META_NAME,
|
||||
DeleteOptions {
|
||||
recursive: true,
|
||||
immediate: true,
|
||||
..Default::default()
|
||||
},
|
||||
)
|
||||
.await
|
||||
.expect("one pool metadata shard should be removable");
|
||||
assert!(
|
||||
missing_disk.read_xl(RUSTFS_META_BUCKET, POOL_META_NAME, false).await.is_err(),
|
||||
"pool metadata fixture must start with one missing shard"
|
||||
);
|
||||
missing_disk
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
async fn heal_erasure_set_scopes_follow_requested_pool_and_set() {
|
||||
let store = minimal_heal_store().await;
|
||||
@@ -1310,473 +1207,6 @@ mod tests {
|
||||
);
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
#[serial_test::serial]
|
||||
async fn pool_meta_read_repair_reuses_its_write_fence() {
|
||||
let (_temp_dirs, store, _other_store) = test_two_pool_stores(None).await;
|
||||
let missing_disk = remove_pool_meta_shard(&store, 0).await;
|
||||
|
||||
let (result, err) = tokio::time::timeout(
|
||||
std::time::Duration::from_secs(30),
|
||||
store.handle_heal_object(
|
||||
RUSTFS_META_BUCKET,
|
||||
POOL_META_NAME,
|
||||
"",
|
||||
&HealOpts {
|
||||
read_repair: true,
|
||||
pool: Some(0),
|
||||
..Default::default()
|
||||
},
|
||||
),
|
||||
)
|
||||
.await
|
||||
.expect("pool metadata read repair must not wait on its own capacity fence")
|
||||
.expect("pool metadata read repair should complete");
|
||||
|
||||
assert!(err.is_none(), "pool metadata read repair should succeed: {err:?}");
|
||||
assert_eq!(result.object, POOL_META_NAME);
|
||||
assert!(
|
||||
missing_disk.read_xl(RUSTFS_META_BUCKET, POOL_META_NAME, false).await.is_ok(),
|
||||
"pool metadata read repair should restore the missing shard"
|
||||
);
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
#[serial_test::serial]
|
||||
async fn pool_meta_heal_preserves_typed_lock_timeout() {
|
||||
let (_temp_dirs, store, _other_store) = test_two_pool_stores(None).await;
|
||||
let lock = store.pools[0]
|
||||
.new_ns_lock(RUSTFS_META_BUCKET, POOL_META_NAME)
|
||||
.await
|
||||
.expect("pool metadata lock should be created");
|
||||
let guard = lock
|
||||
.get_write_lock(get_lock_acquire_timeout())
|
||||
.await
|
||||
.expect("pool metadata lock should be acquired");
|
||||
|
||||
let (_, err) = temp_env::async_with_vars(
|
||||
[(rustfs_config::ENV_OBJECT_LOCK_ACQUIRE_TIMEOUT, Some("1"))],
|
||||
store.handle_heal_object(
|
||||
RUSTFS_META_BUCKET,
|
||||
POOL_META_NAME,
|
||||
"",
|
||||
&HealOpts {
|
||||
pool: Some(0),
|
||||
..Default::default()
|
||||
},
|
||||
),
|
||||
)
|
||||
.await
|
||||
.expect("pool metadata lock timeout should be mapped into the heal result");
|
||||
|
||||
assert!(
|
||||
matches!(err, Some(Error::Lock(rustfs_lock::LockError::Timeout { .. }))),
|
||||
"pool metadata heal must preserve the recoverable lock timeout: {err:?}"
|
||||
);
|
||||
drop(guard);
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
#[serial_test::serial]
|
||||
async fn pool_meta_neighbor_keeps_ordinary_object_locking() {
|
||||
let (_temp_dirs, store, _other_store) = test_two_pool_stores(None).await;
|
||||
let object = "pool.bin.backup";
|
||||
save_config(store.pools[0].clone(), object, b"neighbor metadata".to_vec())
|
||||
.await
|
||||
.expect("neighbor metadata fixture should be written");
|
||||
|
||||
let target_set = store.pools[0].get_disks_by_key(object);
|
||||
let lock = target_set
|
||||
.new_ns_lock(RUSTFS_META_BUCKET, object)
|
||||
.await
|
||||
.expect("neighbor metadata lock should be created");
|
||||
let guard = lock
|
||||
.get_write_lock(get_lock_acquire_timeout())
|
||||
.await
|
||||
.expect("neighbor metadata lock should be acquired");
|
||||
let barrier = DecommissionCapacityLockOrderBarrier::install(store.id, store.id);
|
||||
let heal_store = Arc::clone(&store);
|
||||
let mut heal = tokio::spawn(async move {
|
||||
heal_store
|
||||
.handle_heal_object(
|
||||
RUSTFS_META_BUCKET,
|
||||
object,
|
||||
"",
|
||||
&HealOpts {
|
||||
pool: Some(0),
|
||||
..Default::default()
|
||||
},
|
||||
)
|
||||
.await
|
||||
});
|
||||
|
||||
barrier.wait_until_external_heal_target_lock_attempted().await;
|
||||
tokio::task::yield_now().await;
|
||||
assert!(!heal.is_finished(), "neighbor metadata heal must retain ordinary object locking");
|
||||
drop(guard);
|
||||
|
||||
let (result, err) = tokio::time::timeout(std::time::Duration::from_secs(30), &mut heal)
|
||||
.await
|
||||
.expect("neighbor metadata heal should finish after the object lock is released")
|
||||
.expect("neighbor metadata heal task should not panic")
|
||||
.expect("neighbor metadata heal should complete");
|
||||
assert!(err.is_none(), "neighbor metadata heal should succeed: {err:?}");
|
||||
assert_eq!(result.object, object);
|
||||
drop(barrier);
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
#[serial_test::serial]
|
||||
async fn targeted_heal_is_blocked_by_exact_fit_decommission_reservation() {
|
||||
let (_temp_dirs, store, _other_store) = test_two_pool_stores(None).await;
|
||||
let bucket = format!("heal-capacity-{}", Uuid::new_v4().simple());
|
||||
let object = "targeted-heal.bin";
|
||||
store
|
||||
.make_bucket(&bucket, &MakeBucketOptions::default())
|
||||
.await
|
||||
.expect("heal capacity bucket should be created");
|
||||
|
||||
let target_set = store.pools[1].get_disks(0);
|
||||
let mut reader = PutObjReader::from_vec(b"targeted heal body".to_vec());
|
||||
target_set
|
||||
.put_object(&bucket, object, &mut reader, &ObjectOptions::default())
|
||||
.await
|
||||
.expect("targeted heal fixture should be written");
|
||||
let missing_disk = target_set.disks.read().await[0]
|
||||
.clone()
|
||||
.expect("targeted heal fixture disk should be online");
|
||||
missing_disk
|
||||
.delete(
|
||||
&bucket,
|
||||
object,
|
||||
DeleteOptions {
|
||||
recursive: true,
|
||||
immediate: true,
|
||||
..Default::default()
|
||||
},
|
||||
)
|
||||
.await
|
||||
.expect("targeted heal fixture shard should be removed");
|
||||
assert!(
|
||||
missing_disk.read_xl(&bucket, object, false).await.is_err(),
|
||||
"targeted heal fixture must start with one missing metadata copy"
|
||||
);
|
||||
|
||||
let layout = DecommissionErasureLayout { data: 1, parity: 0 };
|
||||
set_decommission_capacity_info_overrides_for_test(
|
||||
store.id,
|
||||
vec![vec![
|
||||
DecommissionPoolCapacityInfo::for_test(0, layout, 0, 30, 30),
|
||||
DecommissionPoolCapacityInfo::for_test(1, layout, 60, 60, 0),
|
||||
]],
|
||||
);
|
||||
store
|
||||
.save_current_pool_meta_for_decommission_start(&[0], Vec::new())
|
||||
.await
|
||||
.expect("the exact-fit decommission reservation should activate");
|
||||
{
|
||||
let pool_meta = store.pool_meta.read().await;
|
||||
let reservation = pool_meta.pools[0]
|
||||
.decommission
|
||||
.as_ref()
|
||||
.and_then(|info| info.capacity_reservation.as_ref())
|
||||
.expect("the active decommission reservation should be durable");
|
||||
assert_eq!(reservation.targets.len(), 1);
|
||||
assert_eq!(reservation.targets[0].pool_index, 1);
|
||||
assert_eq!(reservation.targets[0].reserved_physical_bytes, 60);
|
||||
}
|
||||
|
||||
let (_, err) = store
|
||||
.handle_heal_object(
|
||||
&bucket,
|
||||
object,
|
||||
"",
|
||||
&HealOpts {
|
||||
pool: Some(1),
|
||||
set: Some(0),
|
||||
..Default::default()
|
||||
},
|
||||
)
|
||||
.await
|
||||
.expect("targeted heal should return a mapped capacity result");
|
||||
assert!(matches!(err, Some(Error::SlowDown)), "targeted heal must be capacity-blocked: {err:?}");
|
||||
assert!(
|
||||
missing_disk.read_xl(&bucket, object, false).await.is_err(),
|
||||
"capacity-blocked targeted heal must not rewrite the missing shard"
|
||||
);
|
||||
|
||||
let (_, err) = tokio::time::timeout(
|
||||
std::time::Duration::from_secs(30),
|
||||
store.handle_heal_object(
|
||||
RUSTFS_META_BUCKET,
|
||||
POOL_META_NAME,
|
||||
"",
|
||||
&HealOpts {
|
||||
pool: Some(1),
|
||||
..Default::default()
|
||||
},
|
||||
),
|
||||
)
|
||||
.await
|
||||
.expect("pool metadata admission should not recurse on its namespace lock")
|
||||
.expect("pool metadata capacity rejection should be mapped");
|
||||
assert!(
|
||||
matches!(err, Some(Error::SlowDown)),
|
||||
"pool metadata heal must preserve target reservation admission: {err:?}"
|
||||
);
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
#[serial_test::serial]
|
||||
async fn pool_meta_heal_keeps_per_target_capacity_admission() {
|
||||
let (_temp_dirs, store, _other_store) = test_three_pool_stores_with_isolated_node_contexts(None).await;
|
||||
let layout = DecommissionErasureLayout { data: 1, parity: 0 };
|
||||
set_decommission_capacity_info_overrides_for_test(
|
||||
store.id,
|
||||
vec![vec![
|
||||
DecommissionPoolCapacityInfo::for_test(0, layout, 0, 30, 30),
|
||||
DecommissionPoolCapacityInfo::for_test(1, layout, 60, 60, 0),
|
||||
DecommissionPoolCapacityInfo::for_test(2, layout, 60, 60, 0),
|
||||
]],
|
||||
);
|
||||
store
|
||||
.save_current_pool_meta_for_decommission_start(&[0], Vec::new())
|
||||
.await
|
||||
.expect("pool metadata reservation should activate");
|
||||
{
|
||||
let pool_meta = store.pool_meta.read().await;
|
||||
let reservation = pool_meta.pools[0]
|
||||
.decommission
|
||||
.as_ref()
|
||||
.and_then(|info| info.capacity_reservation.as_ref())
|
||||
.expect("pool metadata reservation should be durable");
|
||||
assert_eq!(reservation.targets.len(), 1);
|
||||
assert_eq!(reservation.targets[0].pool_index, 1);
|
||||
}
|
||||
|
||||
let missing_disk = remove_pool_meta_shard(&store, 2).await;
|
||||
|
||||
let (result, err) = tokio::time::timeout(
|
||||
std::time::Duration::from_secs(30),
|
||||
store.handle_heal_object(RUSTFS_META_BUCKET, POOL_META_NAME, "", &HealOpts::default()),
|
||||
)
|
||||
.await
|
||||
.expect("unscoped pool metadata heal should complete")
|
||||
.expect("unscoped pool metadata heal should return a mapped result");
|
||||
|
||||
assert!(err.is_none(), "an admitted target should let unscoped metadata heal succeed: {err:?}");
|
||||
assert_eq!(result.object, POOL_META_NAME);
|
||||
assert!(
|
||||
missing_disk.read_xl(RUSTFS_META_BUCKET, POOL_META_NAME, false).await.is_ok(),
|
||||
"unscoped metadata heal should repair the admitted target while another target is reserved"
|
||||
);
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
#[serial_test::serial]
|
||||
async fn targeted_heal_keeps_target_lock_for_different_lock_domain() {
|
||||
let (_temp_dirs, store, other_store) = test_three_pool_stores_with_isolated_node_contexts(None).await;
|
||||
let bucket = format!("heal-lock-domain-{}", Uuid::new_v4().simple());
|
||||
let object = "different-domain-heal.bin";
|
||||
store
|
||||
.make_bucket(&bucket, &MakeBucketOptions::default())
|
||||
.await
|
||||
.expect("heal lock-domain bucket should be created");
|
||||
|
||||
let target_set = other_store.pools[1].get_disks(0);
|
||||
let mut reader = PutObjReader::from_vec(b"different lock domain body".to_vec());
|
||||
target_set
|
||||
.put_object(&bucket, object, &mut reader, &ObjectOptions::default())
|
||||
.await
|
||||
.expect("heal lock-domain fixture should be written");
|
||||
let missing_disk = target_set.disks.read().await[0]
|
||||
.clone()
|
||||
.expect("heal lock-domain fixture disk should be online");
|
||||
missing_disk
|
||||
.delete(
|
||||
&bucket,
|
||||
object,
|
||||
DeleteOptions {
|
||||
recursive: true,
|
||||
immediate: true,
|
||||
..Default::default()
|
||||
},
|
||||
)
|
||||
.await
|
||||
.expect("heal lock-domain fixture shard should be removed");
|
||||
assert!(
|
||||
missing_disk.read_xl(&bucket, object, false).await.is_err(),
|
||||
"heal lock-domain fixture must start with one missing metadata copy"
|
||||
);
|
||||
|
||||
let layout = DecommissionErasureLayout { data: 1, parity: 0 };
|
||||
set_decommission_capacity_info_overrides_for_test(
|
||||
store.id,
|
||||
vec![vec![
|
||||
DecommissionPoolCapacityInfo::for_test(0, layout, 0, 30, 30),
|
||||
DecommissionPoolCapacityInfo::for_test(1, layout, 0, 100, 100),
|
||||
DecommissionPoolCapacityInfo::for_test(2, layout, 60, 60, 0),
|
||||
]],
|
||||
);
|
||||
store
|
||||
.save_current_pool_meta_for_decommission_start(&[0], Vec::new())
|
||||
.await
|
||||
.expect("heal lock-domain reservation should activate");
|
||||
{
|
||||
let pool_meta = store.pool_meta.read().await;
|
||||
let reservation = pool_meta.pools[0]
|
||||
.decommission
|
||||
.as_ref()
|
||||
.and_then(|info| info.capacity_reservation.as_ref())
|
||||
.expect("heal lock-domain reservation should be durable");
|
||||
assert_eq!(reservation.targets.len(), 1);
|
||||
assert_eq!(reservation.targets[0].pool_index, 2);
|
||||
}
|
||||
let pool_meta = store.pool_meta.read().await.clone();
|
||||
*other_store.pool_meta.write().await = pool_meta;
|
||||
|
||||
let fixed_set = other_store.pools[0].disk_set[0].clone();
|
||||
assert!(
|
||||
!fixed_set.shares_namespace_lock_domain(&target_set).await,
|
||||
"heal fixture must use different fixed and target lock domains"
|
||||
);
|
||||
let barrier = DecommissionCapacityLockOrderBarrier::install(store.id, other_store.id);
|
||||
let heal_store = Arc::clone(&other_store);
|
||||
let heal_bucket = bucket.clone();
|
||||
let heal_object = object.to_string();
|
||||
let mut heal = tokio::spawn(async move {
|
||||
heal_store
|
||||
.handle_heal_object(
|
||||
&heal_bucket,
|
||||
&heal_object,
|
||||
"",
|
||||
&HealOpts {
|
||||
pool: Some(1),
|
||||
set: Some(0),
|
||||
..Default::default()
|
||||
},
|
||||
)
|
||||
.await
|
||||
});
|
||||
|
||||
barrier.wait_until_external_capacity_released().await;
|
||||
barrier.wait_until_external_heal_target_lock_attempted().await;
|
||||
|
||||
let (_, err) = tokio::time::timeout(std::time::Duration::from_secs(30), &mut heal)
|
||||
.await
|
||||
.expect("targeted heal should finish after target lock release")
|
||||
.expect("targeted heal task should not panic")
|
||||
.expect("targeted heal should complete");
|
||||
assert!(err.is_none(), "targeted heal should repair after target lock release: {err:?}");
|
||||
assert!(
|
||||
missing_disk.read_xl(&bucket, object, false).await.is_ok(),
|
||||
"targeted heal should rewrite the missing shard after the target lock is released"
|
||||
);
|
||||
drop(barrier);
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
#[serial_test::serial]
|
||||
async fn targeted_heal_reuses_shared_target_lock_without_reentrant_lock() {
|
||||
let (_temp_dirs, store, _other_store) = test_three_pool_stores_with_isolated_node_contexts(None).await;
|
||||
let bucket = format!("heal-shared-lock-domain-{}", Uuid::new_v4().simple());
|
||||
let object = "shared-domain-heal.bin";
|
||||
store
|
||||
.make_bucket(&bucket, &MakeBucketOptions::default())
|
||||
.await
|
||||
.expect("shared heal lock-domain bucket should be created");
|
||||
|
||||
let target_set = store.pools[0].get_disks(0);
|
||||
let mut reader = PutObjReader::from_vec(b"shared lock domain body".to_vec());
|
||||
target_set
|
||||
.put_object(&bucket, object, &mut reader, &ObjectOptions::default())
|
||||
.await
|
||||
.expect("shared heal lock-domain fixture should be written");
|
||||
let missing_disk = target_set.disks.read().await[0]
|
||||
.clone()
|
||||
.expect("shared heal lock-domain fixture disk should be online");
|
||||
missing_disk
|
||||
.delete(
|
||||
&bucket,
|
||||
object,
|
||||
DeleteOptions {
|
||||
recursive: true,
|
||||
immediate: true,
|
||||
..Default::default()
|
||||
},
|
||||
)
|
||||
.await
|
||||
.expect("shared heal lock-domain fixture shard should be removed");
|
||||
assert!(
|
||||
missing_disk.read_xl(&bucket, object, false).await.is_err(),
|
||||
"shared heal lock-domain fixture must start with one missing metadata copy"
|
||||
);
|
||||
|
||||
let layout = DecommissionErasureLayout { data: 1, parity: 0 };
|
||||
set_decommission_capacity_info_overrides_for_test(
|
||||
store.id,
|
||||
vec![vec![
|
||||
DecommissionPoolCapacityInfo::for_test(0, layout, 0, 100, 100),
|
||||
DecommissionPoolCapacityInfo::for_test(1, layout, 60, 60, 0),
|
||||
DecommissionPoolCapacityInfo::for_test(2, layout, 0, 30, 30),
|
||||
]],
|
||||
);
|
||||
store
|
||||
.save_current_pool_meta_for_decommission_start(&[2], Vec::new())
|
||||
.await
|
||||
.expect("shared heal lock-domain reservation should activate");
|
||||
{
|
||||
let pool_meta = store.pool_meta.read().await;
|
||||
let reservation = pool_meta.pools[2]
|
||||
.decommission
|
||||
.as_ref()
|
||||
.and_then(|info| info.capacity_reservation.as_ref())
|
||||
.expect("shared heal lock-domain reservation should be durable");
|
||||
assert_eq!(reservation.targets.len(), 1);
|
||||
assert_eq!(reservation.targets[0].pool_index, 1);
|
||||
assert_eq!(reservation.targets[0].reserved_physical_bytes, 60);
|
||||
}
|
||||
|
||||
let fixed_set = store.pools[0].disk_set[0].clone();
|
||||
assert!(
|
||||
fixed_set.shares_namespace_lock_domain(&target_set).await,
|
||||
"shared heal fixture must use one fixed and target lock domain"
|
||||
);
|
||||
|
||||
let barrier = DecommissionCapacityLockOrderBarrier::install(store.id, store.id);
|
||||
let heal_store = Arc::clone(&store);
|
||||
let heal_bucket = bucket.clone();
|
||||
let heal_object = object.to_string();
|
||||
let mut heal = tokio::spawn(async move {
|
||||
heal_store
|
||||
.handle_heal_object(
|
||||
&heal_bucket,
|
||||
&heal_object,
|
||||
"",
|
||||
&HealOpts {
|
||||
pool: Some(0),
|
||||
set: Some(0),
|
||||
..Default::default()
|
||||
},
|
||||
)
|
||||
.await
|
||||
});
|
||||
|
||||
barrier.wait_until_external_capacity_released().await;
|
||||
barrier.wait_until_external_heal_operation_started().await;
|
||||
let (_, err) = tokio::time::timeout(std::time::Duration::from_secs(5), &mut heal)
|
||||
.await
|
||||
.expect("shared-domain heal must not reenter the fixed namespace lock")
|
||||
.expect("shared-domain heal task should not panic")
|
||||
.expect("shared-domain heal should complete");
|
||||
assert!(err.is_none(), "shared-domain heal should repair after admission: {err:?}");
|
||||
assert!(
|
||||
missing_disk.read_xl(&bucket, object, false).await.is_ok(),
|
||||
"shared-domain heal should rewrite the missing shard"
|
||||
);
|
||||
drop(barrier);
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
async fn scoped_heal_object_defers_when_requested_pool_is_suspended() {
|
||||
let mut store = minimal_heal_store().await;
|
||||
@@ -2267,10 +1697,8 @@ mod tests {
|
||||
.expect("quorum boundary heal should return a mapped result");
|
||||
*store.pools[0].disk_set[0].disks.write().await = original_quorum_disks;
|
||||
assert!(
|
||||
quorum_err.as_ref().is_some_and(|err| err
|
||||
.to_string()
|
||||
.contains("pool metadata writes remain blocked after a recovery-required replica state")),
|
||||
"heal must fail closed when capacity admission cannot verify pool metadata, got {quorum_err:?}"
|
||||
matches!(quorum_err, Some(Error::ErasureReadQuorum)),
|
||||
"quorum-boundary heal must preserve quorum error, got {quorum_err:?}"
|
||||
);
|
||||
shutdown.cancel();
|
||||
}
|
||||
@@ -2380,7 +1808,6 @@ mod tests {
|
||||
decommission_cancelers: RwLock::new(Vec::new()),
|
||||
start_gate: Mutex::new(()),
|
||||
pool_meta_save_gate: Mutex::default(),
|
||||
decommission_capacity_entry_gate: Mutex::default(),
|
||||
ctx: crate::runtime::instance::bootstrap_ctx(),
|
||||
bucket_fence_registry: std::sync::Arc::default(),
|
||||
};
|
||||
|
||||
+449
-5984
File diff suppressed because it is too large
Load Diff
@@ -20,9 +20,7 @@ fn to_filemeta_err(err: Error) -> rustfs_filemeta::Error {
|
||||
err.narrow_to_filemeta().unwrap_or_else(rustfs_filemeta::Error::other)
|
||||
}
|
||||
|
||||
use crate::bucket::metadata_sys::{
|
||||
get_versioning_config, has_authoritative_never_versioned_state, has_authoritative_never_versioned_state_in,
|
||||
};
|
||||
use crate::bucket::metadata_sys::{get_versioning_config, has_authoritative_never_versioned_state};
|
||||
use crate::bucket::utils::check_list_objs_args;
|
||||
use crate::bucket::versioning::VersioningApi;
|
||||
use crate::cache_value::metacache_set::{FallbackClaimTracker, ListPathRawOptions, list_path_raw_with_claim_tracker};
|
||||
@@ -316,25 +314,6 @@ async fn can_skip_hidden_prefix_check(options: &ListPathOptions) -> bool {
|
||||
.unwrap_or(false)
|
||||
}
|
||||
|
||||
fn should_purge_empty_directory_listing(
|
||||
prefix: &str,
|
||||
marker: Option<&str>,
|
||||
delimiter: Option<&str>,
|
||||
max_keys: i32,
|
||||
incl_deleted: bool,
|
||||
result: &ListObjectsInfo,
|
||||
) -> bool {
|
||||
!prefix.is_empty()
|
||||
&& prefix.ends_with(SLASH_SEPARATOR)
|
||||
&& marker.is_none()
|
||||
&& delimiter.is_none_or(str::is_empty)
|
||||
&& max_keys == 1
|
||||
&& !incl_deleted
|
||||
&& !result.is_truncated
|
||||
&& result.objects.is_empty()
|
||||
&& result.prefixes.is_empty()
|
||||
}
|
||||
|
||||
const MARKER_TAG_VERSION: &str = "v2";
|
||||
const LEGACY_MARKER_TAG_VERSIONS: &[&str] = &["v1", MARKER_TAG_VERSION];
|
||||
const LIST_CACHE_MARKER_PREFIX: &str = "[rustfs_cache:";
|
||||
@@ -2522,7 +2501,6 @@ fn list_metadata_resolution_params(
|
||||
listing_quorum: usize,
|
||||
latest_object_quorum: usize,
|
||||
versioned: bool,
|
||||
write_quorum_slack: usize,
|
||||
) -> MetadataResolutionParams {
|
||||
let quorum = if versioned {
|
||||
listing_quorum
|
||||
@@ -2533,7 +2511,6 @@ fn list_metadata_resolution_params(
|
||||
dir_quorum: quorum,
|
||||
obj_quorum: quorum,
|
||||
bucket,
|
||||
write_quorum_slack,
|
||||
..Default::default()
|
||||
};
|
||||
|
||||
@@ -2834,10 +2811,7 @@ fn cached_entry_needs_supplement(
|
||||
|
||||
let mut selected_object_versions = 0;
|
||||
for version in cached.versions.iter() {
|
||||
let required_quorum = version
|
||||
.write_quorum(resolver.obj_quorum)
|
||||
.saturating_sub(resolver.write_quorum_slack)
|
||||
.max(resolver.obj_quorum);
|
||||
let required_quorum = version.write_quorum(resolver.obj_quorum).max(resolver.obj_quorum);
|
||||
if version_requires_supplement(required_quorum, reader_disks, selected_object_versions, resolver.requested_versions) {
|
||||
return true;
|
||||
}
|
||||
@@ -2862,15 +2836,8 @@ fn listing_entries_supplement_target(
|
||||
return None;
|
||||
}
|
||||
|
||||
for (idx, entry) in entries.0.iter().enumerate() {
|
||||
let Some(entry) = entry.as_ref().filter(|entry| entry.is_object()) else {
|
||||
continue;
|
||||
};
|
||||
if entries.0[..idx]
|
||||
.iter()
|
||||
.flatten()
|
||||
.any(|previous| previous.name == entry.name && previous.is_object())
|
||||
{
|
||||
for entry in entries.0.iter().flatten() {
|
||||
if entry.is_dir() {
|
||||
continue;
|
||||
}
|
||||
|
||||
@@ -2883,15 +2850,6 @@ fn listing_entries_supplement_target(
|
||||
.is_some_and(|candidate| candidate.name == entry.name && candidate.is_object())
|
||||
})
|
||||
.count();
|
||||
let entries_disagree = entries.0.iter().flatten().any(|candidate| {
|
||||
candidate.name == entry.name && candidate.is_object() && !entry.matches(Some(candidate), resolver.strict).1
|
||||
});
|
||||
// A split primary sample can fall back to an older version even when a
|
||||
// newer version reaches write quorum only after the fallback disks join.
|
||||
if entries_disagree {
|
||||
return Some(entry.name.clone());
|
||||
}
|
||||
|
||||
let mut entry = entry.clone();
|
||||
if let Ok(cached) = entry.xl_meta()
|
||||
&& cached_entry_needs_supplement(&cached, reader_disks, resolver, enforce_write_quorum)
|
||||
@@ -2988,10 +2946,7 @@ fn resolve_agreed_listing_entry(
|
||||
let mut needs_supplement = false;
|
||||
|
||||
for (idx, version) in cached.versions.iter().enumerate() {
|
||||
let required_quorum = version
|
||||
.write_quorum(resolver.obj_quorum)
|
||||
.saturating_sub(resolver.write_quorum_slack)
|
||||
.max(resolver.obj_quorum);
|
||||
let required_quorum = version.write_quorum(resolver.obj_quorum).max(resolver.obj_quorum);
|
||||
if reader_disks < required_quorum {
|
||||
needs_supplement |=
|
||||
version_requires_supplement(required_quorum, reader_disks, selected_object_versions, resolver.requested_versions);
|
||||
@@ -3045,51 +3000,21 @@ fn latest_listing_object_quorum(
|
||||
drive_count: usize,
|
||||
parity_count: usize,
|
||||
enforce_write_quorum: bool,
|
||||
unreachable_disks: usize,
|
||||
) -> usize {
|
||||
latest_listing_required_object_quorum(listing_quorum, drive_count, parity_count, enforce_write_quorum, unreachable_disks)
|
||||
latest_listing_required_object_quorum(listing_quorum, drive_count, parity_count, enforce_write_quorum)
|
||||
}
|
||||
|
||||
/// Object quorum a listed "latest" version must reach among the drives the
|
||||
/// listing can actually consult.
|
||||
///
|
||||
/// An object legally committed at write quorum can have up to
|
||||
/// `unreachable_disks` of its metadata copies on drives that are offline for
|
||||
/// this listing, so the write-quorum requirement is relaxed by that amount.
|
||||
/// The result is floored at the erasure read quorum (data drives): below that
|
||||
/// the object could not be read back either, and a quorum-deleted object
|
||||
/// leaves at most `drive_count - write_quorum < read_quorum` stale copies, so
|
||||
/// the floor also keeps deleted objects from resurfacing.
|
||||
///
|
||||
/// Trade-off: while a drive is offline, a torn overwrite that reached only
|
||||
/// `write_quorum - 1` drives becomes indistinguishable from a committed write
|
||||
/// whose missing copy sits on the offline drive, so it can be listed as
|
||||
/// latest. GET at read quorum serves that same version in that state, so the
|
||||
/// listing stays consistent with reads instead of hiding readable objects.
|
||||
fn latest_listing_required_object_quorum(
|
||||
listing_quorum: usize,
|
||||
drive_count: usize,
|
||||
parity_count: usize,
|
||||
enforce_write_quorum: bool,
|
||||
unreachable_disks: usize,
|
||||
) -> usize {
|
||||
if !enforce_write_quorum {
|
||||
return listing_quorum;
|
||||
}
|
||||
|
||||
let read_quorum = drive_count.saturating_sub(parity_count);
|
||||
write_quorum_for_drive_count(drive_count, parity_count)
|
||||
.saturating_sub(unreachable_disks)
|
||||
.max(read_quorum)
|
||||
.max(listing_quorum)
|
||||
}
|
||||
|
||||
fn latest_listing_write_quorum_slack(enforce_write_quorum: bool, drive_count: usize, online_disks: usize) -> usize {
|
||||
if !enforce_write_quorum {
|
||||
return 0;
|
||||
}
|
||||
|
||||
drive_count.saturating_sub(online_disks)
|
||||
write_quorum_for_drive_count(drive_count, parity_count).max(listing_quorum)
|
||||
}
|
||||
|
||||
fn enforce_latest_listing_write_quorum(strict_latest: bool, ask_disks: &str) -> bool {
|
||||
@@ -3847,19 +3772,6 @@ impl ECStore {
|
||||
.list_objects_from_opt_in_key_only_provider(&opts, mode, max_keys, incl_deleted)
|
||||
.await?
|
||||
{
|
||||
if should_purge_empty_directory_listing(
|
||||
prefix,
|
||||
opts.marker.as_deref(),
|
||||
delimiter.as_deref(),
|
||||
max_keys,
|
||||
incl_deleted,
|
||||
&result,
|
||||
) && has_authoritative_never_versioned_state_in(&self.ctx, bucket)
|
||||
.await
|
||||
.unwrap_or(false)
|
||||
{
|
||||
self.purge_orphan_dir_object(bucket, prefix).await;
|
||||
}
|
||||
return Ok(result);
|
||||
}
|
||||
|
||||
@@ -3890,7 +3802,6 @@ impl ECStore {
|
||||
};
|
||||
|
||||
let mut list_result = self
|
||||
.clone()
|
||||
.list_path(&opts)
|
||||
.await
|
||||
.unwrap_or_else(|err| MetaCacheEntriesSortedResult {
|
||||
@@ -3909,7 +3820,7 @@ impl ECStore {
|
||||
}
|
||||
|
||||
if let Some(result) = list_result.entries.as_mut() {
|
||||
result.forward_past(opts.marker.clone());
|
||||
result.forward_past(opts.marker);
|
||||
}
|
||||
|
||||
// contextCanceled
|
||||
@@ -3937,26 +3848,12 @@ impl ECStore {
|
||||
);
|
||||
let _ = next_version_idmarker;
|
||||
|
||||
let result = ListObjectsInfo {
|
||||
Ok(ListObjectsInfo {
|
||||
is_truncated,
|
||||
next_marker,
|
||||
objects,
|
||||
prefixes,
|
||||
};
|
||||
if should_purge_empty_directory_listing(
|
||||
prefix,
|
||||
opts.marker.as_deref(),
|
||||
delimiter.as_deref(),
|
||||
max_keys,
|
||||
incl_deleted,
|
||||
&result,
|
||||
) && has_authoritative_never_versioned_state_in(&self.ctx, bucket)
|
||||
.await
|
||||
.unwrap_or(false)
|
||||
{
|
||||
self.purge_orphan_dir_object(bucket, prefix).await;
|
||||
}
|
||||
Ok(result)
|
||||
})
|
||||
}
|
||||
|
||||
pub async fn inner_list_object_versions(
|
||||
@@ -4409,9 +4306,6 @@ impl ECStore {
|
||||
for eset in self.pools.iter() {
|
||||
for set in eset.disk_set.iter() {
|
||||
let (mut disks, infos, _) = set.get_online_disks_with_healing_and_info(true).await;
|
||||
// Captured before any quorum-based filtering: only genuinely
|
||||
// unreachable drives may relax the write-quorum requirement.
|
||||
let online_disks = disks.len();
|
||||
let opts = opts.clone();
|
||||
|
||||
let (sender, list_out_rx) = mpsc::channel::<MetaCacheEntry>(1);
|
||||
@@ -4437,14 +4331,11 @@ impl ECStore {
|
||||
let listing_quorum = listing_quorum_from_ask_disks(ask_disks);
|
||||
let enforce_write_quorum = enforce_latest_listing_write_quorum(opts.latest_only, &opts.ask_disks);
|
||||
let write_quorum_parity = set.default_parity_count;
|
||||
let write_quorum_slack =
|
||||
latest_listing_write_quorum_slack(enforce_write_quorum, set.set_drive_count, online_disks);
|
||||
let required_obj_quorum = latest_listing_required_object_quorum(
|
||||
listing_quorum,
|
||||
set.set_drive_count,
|
||||
write_quorum_parity,
|
||||
enforce_write_quorum,
|
||||
write_quorum_slack,
|
||||
);
|
||||
ask_disks = expand_ask_disks_for_object_quorum(ask_disks, disks.len(), required_obj_quorum);
|
||||
let fallback_disks = {
|
||||
@@ -4466,17 +4357,11 @@ impl ECStore {
|
||||
set.set_drive_count,
|
||||
write_quorum_parity,
|
||||
enforce_write_quorum,
|
||||
write_quorum_slack,
|
||||
);
|
||||
let raw_min_disks = latest_listing_raw_min_disks(listing_quorum, obj_quorum, enforce_write_quorum);
|
||||
|
||||
let resolver = list_metadata_resolution_params(
|
||||
bucket.to_owned(),
|
||||
listing_quorum,
|
||||
obj_quorum,
|
||||
!opts.latest_only,
|
||||
write_quorum_slack,
|
||||
);
|
||||
let resolver =
|
||||
list_metadata_resolution_params(bucket.to_owned(), listing_quorum, obj_quorum, !opts.latest_only);
|
||||
let agreed_resolver = resolver.clone();
|
||||
let partial_resolver = resolver.clone();
|
||||
let reader_disks = disks.len();
|
||||
@@ -5663,9 +5548,6 @@ impl Sets {
|
||||
|
||||
for set in &self.disk_set {
|
||||
let (mut disks, infos, _) = set.get_online_disks_with_healing_and_info(true).await;
|
||||
// Captured before any quorum-based filtering: only genuinely
|
||||
// unreachable drives may relax the write-quorum requirement.
|
||||
let online_disks = disks.len();
|
||||
let opts = opts.clone();
|
||||
let (sender, list_out_rx) = mpsc::channel::<MetaCacheEntry>(1);
|
||||
inputs.push(list_out_rx);
|
||||
@@ -5691,14 +5573,11 @@ impl Sets {
|
||||
let listing_quorum = listing_quorum_from_ask_disks(ask_disks);
|
||||
let enforce_write_quorum = enforce_latest_listing_write_quorum(opts.latest_only, &opts.ask_disks);
|
||||
let write_quorum_parity = set.default_parity_count;
|
||||
let write_quorum_slack =
|
||||
latest_listing_write_quorum_slack(enforce_write_quorum, set.set_drive_count, online_disks);
|
||||
let required_obj_quorum = latest_listing_required_object_quorum(
|
||||
listing_quorum,
|
||||
set.set_drive_count,
|
||||
write_quorum_parity,
|
||||
enforce_write_quorum,
|
||||
write_quorum_slack,
|
||||
);
|
||||
ask_disks = expand_ask_disks_for_object_quorum(ask_disks, disks.len(), required_obj_quorum);
|
||||
let fallback_disks = if let Some(asked_disks) = positive_ask_disks(ask_disks)
|
||||
@@ -5713,21 +5592,10 @@ impl Sets {
|
||||
let fallback_disks = Arc::new(fallback_disks);
|
||||
let claim_tracker = FallbackClaimTracker::default();
|
||||
|
||||
let obj_quorum = latest_listing_object_quorum(
|
||||
listing_quorum,
|
||||
set.set_drive_count,
|
||||
write_quorum_parity,
|
||||
enforce_write_quorum,
|
||||
write_quorum_slack,
|
||||
);
|
||||
let obj_quorum =
|
||||
latest_listing_object_quorum(listing_quorum, set.set_drive_count, write_quorum_parity, enforce_write_quorum);
|
||||
let raw_min_disks = latest_listing_raw_min_disks(listing_quorum, obj_quorum, enforce_write_quorum);
|
||||
let resolver = list_metadata_resolution_params(
|
||||
bucket.to_owned(),
|
||||
listing_quorum,
|
||||
obj_quorum,
|
||||
!opts.latest_only,
|
||||
write_quorum_slack,
|
||||
);
|
||||
let resolver = list_metadata_resolution_params(bucket.to_owned(), listing_quorum, obj_quorum, !opts.latest_only);
|
||||
let agreed_resolver = resolver.clone();
|
||||
let partial_resolver = resolver.clone();
|
||||
let reader_disks = disks.len();
|
||||
@@ -6649,9 +6517,6 @@ impl SetDisks {
|
||||
let list_path_started = std::time::Instant::now();
|
||||
|
||||
let (mut disks, infos, _) = self.get_online_disks_with_healing_and_info(true).await;
|
||||
// Captured before any quorum-based filtering: only genuinely
|
||||
// unreachable drives may relax the write-quorum requirement.
|
||||
let online_disks = disks.len();
|
||||
|
||||
let mut ask_disks = get_list_quorum(&opts.ask_disks, self.set_drive_count as i32);
|
||||
if ask_disks == -1 {
|
||||
@@ -6675,13 +6540,11 @@ impl SetDisks {
|
||||
|
||||
let enforce_write_quorum = enforce_latest_listing_write_quorum(!opts.versioned, &opts.ask_disks);
|
||||
let write_quorum_parity = self.default_parity_count;
|
||||
let write_quorum_slack = latest_listing_write_quorum_slack(enforce_write_quorum, self.set_drive_count, online_disks);
|
||||
let required_obj_quorum = latest_listing_required_object_quorum(
|
||||
listing_quorum,
|
||||
self.set_drive_count,
|
||||
write_quorum_parity,
|
||||
enforce_write_quorum,
|
||||
write_quorum_slack,
|
||||
);
|
||||
ask_disks = expand_ask_disks_for_object_quorum(ask_disks, disks.len(), required_obj_quorum);
|
||||
let mut fallback_disks = Vec::new();
|
||||
@@ -6699,21 +6562,10 @@ impl SetDisks {
|
||||
|
||||
let bucket = opts.bucket.clone();
|
||||
let base_dir = opts.base_dir.clone();
|
||||
let latest_object_quorum = latest_listing_object_quorum(
|
||||
listing_quorum,
|
||||
self.set_drive_count,
|
||||
write_quorum_parity,
|
||||
enforce_write_quorum,
|
||||
write_quorum_slack,
|
||||
);
|
||||
let latest_object_quorum =
|
||||
latest_listing_object_quorum(listing_quorum, self.set_drive_count, write_quorum_parity, enforce_write_quorum);
|
||||
let raw_min_disks = latest_listing_raw_min_disks(listing_quorum, latest_object_quorum, enforce_write_quorum);
|
||||
let resolver = list_metadata_resolution_params(
|
||||
bucket.clone(),
|
||||
listing_quorum,
|
||||
latest_object_quorum,
|
||||
opts.versioned,
|
||||
write_quorum_slack,
|
||||
);
|
||||
let resolver = list_metadata_resolution_params(bucket.clone(), listing_quorum, latest_object_quorum, opts.versioned);
|
||||
let agreed_resolver = resolver.clone();
|
||||
let partial_resolver = resolver.clone();
|
||||
let reader_disks = disks.len();
|
||||
@@ -6725,7 +6577,6 @@ impl SetDisks {
|
||||
asked_disks = ask_disks,
|
||||
listing_quorum = listing_quorum,
|
||||
latest_object_quorum = latest_object_quorum,
|
||||
write_quorum_slack = write_quorum_slack,
|
||||
raw_min_disks = raw_min_disks,
|
||||
fallback_disks = fallback_disks.len(),
|
||||
limit = opts.limit,
|
||||
@@ -6967,36 +6818,34 @@ mod test {
|
||||
LIST_CURSOR_GENERATION_LIVE, LIST_OBJECTS_INDEX_PROVIDER_PERSISTENT_KEY_ONLY,
|
||||
LIST_OBJECTS_INDEX_PROVIDER_WALKER_KEY_ONLY, ListIndexFallbackReason, ListIndexLifecycle, ListIndexLifecycleState,
|
||||
ListIndexSourceDecision, ListMetadataAuthority, ListMetadataIndexHealth, ListObjectsIndexProviderKind,
|
||||
ListObjectsIndexProviderState, ListObjectsInfo, ListPathOptions, ListPathRawOptions, ListSourceMode,
|
||||
ListingEntryResolution, ListingSupplement, ListingSupplementOptions, MAX_OBJECT_LIST, NamespaceMutationJournalBackend,
|
||||
ListObjectsIndexProviderState, ListPathOptions, ListPathRawOptions, ListSourceMode, ListingEntryResolution,
|
||||
ListingSupplement, ListingSupplementOptions, MAX_OBJECT_LIST, NamespaceMutationJournalBackend,
|
||||
NamespaceMutationJournalSnapshot, NamespaceMutationJournalStatus, PERSISTENT_KEY_ONLY_INDEX_BUCKET_HEADER,
|
||||
PERSISTENT_KEY_ONLY_INDEX_CHECKPOINT_HEADER, PERSISTENT_KEY_ONLY_INDEX_FORMAT_VERSION,
|
||||
PERSISTENT_KEY_ONLY_INDEX_GENERATION_HEADER, PERSISTENT_KEY_ONLY_INDEX_HEADER, PersistentKeyOnlyIndex,
|
||||
PersistentListMetadataObject, RUSTFS_META_BUCKET, VerifiedIndexCandidateStats, VersionMarker,
|
||||
cached_entry_needs_supplement, current_list_objects_mutation_sequence, encode_persistent_list_metadata_object,
|
||||
enforce_latest_listing_write_quorum, expand_ask_disks_for_object_quorum, fallback_entries_for_object, gather_results,
|
||||
latest_listing_allow_agreed_objects, latest_listing_object_quorum, latest_listing_raw_min_disks,
|
||||
latest_listing_required_object_quorum, latest_listing_write_quorum_slack, list_marker_key, list_merged_entry_channel,
|
||||
list_metadata_resolution_params, list_objects_from_metadata_snapshot_candidates,
|
||||
current_list_objects_mutation_sequence, encode_persistent_list_metadata_object, enforce_latest_listing_write_quorum,
|
||||
expand_ask_disks_for_object_quorum, fallback_entries_for_object, gather_results, latest_listing_allow_agreed_objects,
|
||||
latest_listing_object_quorum, latest_listing_raw_min_disks, latest_listing_required_object_quorum, list_marker_key,
|
||||
list_merged_entry_channel, list_metadata_resolution_params, list_objects_from_metadata_snapshot_candidates,
|
||||
list_objects_from_verified_index_candidates, list_objects_from_verified_index_candidates_with_optional_stats,
|
||||
list_objects_from_verified_index_candidates_with_stats, list_objects_index_mode_from_env,
|
||||
list_objects_index_provider_from_env, list_objects_index_provider_state_from_env,
|
||||
list_objects_metadata_fast_guardrails_from_env, list_objects_paginate, list_objects_quorum_from_env,
|
||||
listing_entries_supplement_target, load_namespace_mutation_journal_state, load_persistent_key_only_index,
|
||||
max_keys_plus_one, merge_entry_channels, namespace_mutation_journal_chaos_bucket_from_env,
|
||||
namespace_mutation_journal_chaos_config_from_env, namespace_mutation_journal_chaos_enabled_from_env,
|
||||
namespace_mutation_journal_chaos_sequence_from_env, namespace_mutation_journal_chaos_status_from_env,
|
||||
normalize_list_quorum, observe_list_objects_mutations_with_store, parse_namespace_mutation_journal_state,
|
||||
parse_persistent_key_only_index, parse_persistent_list_metadata_object, parse_version_marker,
|
||||
persist_observed_list_objects_mutation, persistent_key_only_index_has_complete_metadata_snapshot,
|
||||
load_namespace_mutation_journal_state, load_persistent_key_only_index, max_keys_plus_one, merge_entry_channels,
|
||||
namespace_mutation_journal_chaos_bucket_from_env, namespace_mutation_journal_chaos_config_from_env,
|
||||
namespace_mutation_journal_chaos_enabled_from_env, namespace_mutation_journal_chaos_sequence_from_env,
|
||||
namespace_mutation_journal_chaos_status_from_env, normalize_list_quorum, observe_list_objects_mutations_with_store,
|
||||
parse_namespace_mutation_journal_state, parse_persistent_key_only_index, parse_persistent_list_metadata_object,
|
||||
parse_version_marker, persist_observed_list_objects_mutation, persistent_key_only_index_has_complete_metadata_snapshot,
|
||||
persistent_key_only_index_health, persistent_key_only_index_matches_provider,
|
||||
reset_list_objects_mutation_sequences_for_test, resolve_agreed_listing_entry, resolve_listing_entries,
|
||||
resolve_listing_entries_with_supplement, scanner_namespace_mutation_generation, select_list_index_provider_source_mode,
|
||||
select_list_index_source_mode, send_or_cancel, should_purge_empty_directory_listing, version_marker_for_entries,
|
||||
walk_result_from_set_errors, write_namespace_mutation_journal_state, write_persistent_key_only_index_with_metadata,
|
||||
scanner_namespace_mutation_generation, select_list_index_provider_source_mode, select_list_index_source_mode,
|
||||
send_or_cancel, version_marker_for_entries, walk_result_from_set_errors, write_namespace_mutation_journal_state,
|
||||
write_persistent_key_only_index_with_metadata,
|
||||
};
|
||||
use crate::cache_value::metacache_set::{FallbackClaimTracker, TestReaderBehavior, list_path_raw};
|
||||
use crate::disk::{DiskAPI, DiskOption, STORAGE_FORMAT_FILE, endpoint::Endpoint, error::DiskError, new_disk};
|
||||
use crate::disk::{DiskAPI, DiskOption, endpoint::Endpoint, error::DiskError, new_disk};
|
||||
use crate::error::StorageError;
|
||||
use crate::object_api::ObjectInfo;
|
||||
use rustfs_filemeta::{
|
||||
@@ -7319,32 +7168,6 @@ mod test {
|
||||
}
|
||||
}
|
||||
|
||||
fn test_object_with_delete_marker_meta_entry(
|
||||
name: &str,
|
||||
object_mod_time: time::OffsetDateTime,
|
||||
delete_mod_time: time::OffsetDateTime,
|
||||
) -> MetaCacheEntry {
|
||||
let mut object = test_object_meta_entry_with_erasure_versions(name, &[(object_mod_time, "object-etag", 4, 2)]);
|
||||
let delete = test_delete_marker_meta_entry(name, delete_mod_time);
|
||||
let mut metadata = object.cached.take().expect("test object metadata should be cached");
|
||||
let delete_version = delete
|
||||
.cached
|
||||
.expect("test delete marker metadata should be cached")
|
||||
.versions
|
||||
.into_iter()
|
||||
.next()
|
||||
.expect("test delete marker should contain one version");
|
||||
metadata.versions.insert(0, delete_version);
|
||||
let encoded = metadata.marshal_msg().expect("test metadata should marshal");
|
||||
|
||||
MetaCacheEntry {
|
||||
name: name.to_owned(),
|
||||
metadata: encoded,
|
||||
cached: Some(metadata),
|
||||
reusable: false,
|
||||
}
|
||||
}
|
||||
|
||||
fn test_dir_meta_entry(name: &str) -> MetaCacheEntry {
|
||||
MetaCacheEntry {
|
||||
name: name.to_owned(),
|
||||
@@ -8842,80 +8665,6 @@ mod test {
|
||||
assert_eq!(scanner_namespace_mutation_generation(), 2);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn empty_directory_listing_purge_requires_complete_exact_recursive_request() {
|
||||
let empty = ListObjectsInfo::default();
|
||||
assert!(should_purge_empty_directory_listing("ghost/", None, None, 1, false, &empty));
|
||||
assert!(should_purge_empty_directory_listing("ghost/", None, Some(""), 1, false, &empty));
|
||||
assert!(!should_purge_empty_directory_listing("ghost", None, None, 1, false, &empty));
|
||||
assert!(!should_purge_empty_directory_listing("ghost/", Some("marker"), None, 1, false, &empty));
|
||||
assert!(!should_purge_empty_directory_listing("ghost/", None, Some("/"), 1, false, &empty));
|
||||
assert!(!should_purge_empty_directory_listing("ghost/", None, None, 0, false, &empty));
|
||||
assert!(!should_purge_empty_directory_listing("ghost/", None, None, 2, false, &empty));
|
||||
assert!(!should_purge_empty_directory_listing("ghost/", None, None, 1, true, &empty));
|
||||
|
||||
let mut live = ListObjectsInfo::default();
|
||||
live.objects.push(ObjectInfo::default());
|
||||
assert!(!should_purge_empty_directory_listing("ghost/", None, None, 1, false, &live));
|
||||
|
||||
let truncated = ListObjectsInfo {
|
||||
is_truncated: true,
|
||||
..Default::default()
|
||||
};
|
||||
assert!(!should_purge_empty_directory_listing("ghost/", None, None, 1, false, &truncated));
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
async fn empty_recursive_listing_purges_committed_delete_residue() {
|
||||
use crate::bucket::metadata_sys::{init_bucket_metadata_sys, test_support::isolated_store_over_temp_disks};
|
||||
use crate::storage_api_contracts::bucket::{BucketOperations as _, MakeBucketOptions};
|
||||
|
||||
let (dirs, store) = isolated_store_over_temp_disks().await;
|
||||
let bucket = "listing-purge-bucket";
|
||||
init_bucket_metadata_sys(store.clone(), Vec::new()).await;
|
||||
store
|
||||
.make_bucket(bucket, &MakeBucketOptions::default())
|
||||
.await
|
||||
.expect("bucket should be created with authoritative metadata");
|
||||
let data_dir = uuid::Uuid::new_v4();
|
||||
let transaction = uuid::Uuid::new_v4();
|
||||
for dir in &dirs {
|
||||
let residue = dir
|
||||
.path()
|
||||
.join(bucket)
|
||||
.join("ghost")
|
||||
.join("nested")
|
||||
.join("object")
|
||||
.join(data_dir.to_string());
|
||||
tokio::fs::create_dir_all(&residue)
|
||||
.await
|
||||
.expect("committed delete residue should be created");
|
||||
tokio::fs::write(residue.join("part.1"), b"stale")
|
||||
.await
|
||||
.expect("stale part should be written");
|
||||
tokio::fs::write(
|
||||
residue.join(format!("{}{}", crate::disk::local::DELETE_DATA_DIR_MARKER_PREFIX, transaction)),
|
||||
[],
|
||||
)
|
||||
.await
|
||||
.expect("committed delete marker should be written");
|
||||
}
|
||||
|
||||
let result = store
|
||||
.list_objects_generic(bucket, "ghost/", None, None, 1, false)
|
||||
.await
|
||||
.expect("empty recursive listing should succeed");
|
||||
|
||||
assert!(result.objects.is_empty());
|
||||
assert!(result.prefixes.is_empty());
|
||||
for dir in &dirs {
|
||||
assert!(
|
||||
!dir.path().join(bucket).join("ghost").exists(),
|
||||
"the empty listing should reclaim its committed delete residue"
|
||||
);
|
||||
}
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn list_objects_index_provider_state_uses_lifecycle_active_generation() {
|
||||
let provider = ListObjectsIndexProviderState::walker_key_only();
|
||||
@@ -9198,7 +8947,7 @@ mod test {
|
||||
|
||||
#[test]
|
||||
fn list_metadata_resolution_params_limits_plain_listing_to_latest_version() {
|
||||
let resolver = list_metadata_resolution_params("bucket".to_string(), 2, 3, false, 0);
|
||||
let resolver = list_metadata_resolution_params("bucket".to_string(), 2, 3, false);
|
||||
|
||||
assert_eq!(resolver.dir_quorum, 3);
|
||||
assert_eq!(resolver.obj_quorum, 3);
|
||||
@@ -9208,7 +8957,7 @@ mod test {
|
||||
|
||||
#[test]
|
||||
fn list_metadata_resolution_params_keeps_all_versions_for_version_listing() {
|
||||
let resolver = list_metadata_resolution_params("bucket".to_string(), 3, 5, true, 0);
|
||||
let resolver = list_metadata_resolution_params("bucket".to_string(), 3, 5, true);
|
||||
|
||||
assert_eq!(resolver.dir_quorum, 3);
|
||||
assert_eq!(resolver.obj_quorum, 3);
|
||||
@@ -9218,22 +8967,22 @@ mod test {
|
||||
|
||||
#[test]
|
||||
fn latest_listing_object_quorum_uses_write_quorum_for_strict_latest_listing() {
|
||||
let required_quorum = latest_listing_required_object_quorum(2, 8, 4, true, 0);
|
||||
let required_quorum = latest_listing_required_object_quorum(2, 8, 4, true);
|
||||
let ask_disks = expand_ask_disks_for_object_quorum(4, 8, required_quorum);
|
||||
|
||||
assert_eq!(required_quorum, 5);
|
||||
assert_eq!(ask_disks, 5);
|
||||
assert_eq!(latest_listing_object_quorum(2, 8, 4, true, 0), 5);
|
||||
assert_eq!(latest_listing_object_quorum(2, 8, 4, true), 5);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn latest_listing_object_quorum_calculates_low_parity_write_quorum() {
|
||||
let required_quorum = latest_listing_required_object_quorum(2, 8, 1, true, 0);
|
||||
let required_quorum = latest_listing_required_object_quorum(2, 8, 1, true);
|
||||
let ask_disks = expand_ask_disks_for_object_quorum(4, 8, required_quorum);
|
||||
|
||||
assert_eq!(required_quorum, 7);
|
||||
assert_eq!(ask_disks, 7);
|
||||
assert_eq!(latest_listing_object_quorum(2, 8, 1, true, 0), 7);
|
||||
assert_eq!(latest_listing_object_quorum(2, 8, 1, true), 7);
|
||||
}
|
||||
|
||||
#[test]
|
||||
@@ -9242,196 +8991,21 @@ mod test {
|
||||
assert!(!enforce_latest_listing_write_quorum(true, "disk"));
|
||||
assert!(enforce_latest_listing_write_quorum(true, "optimal"));
|
||||
assert!(!enforce_latest_listing_write_quorum(false, "optimal"));
|
||||
assert_eq!(latest_listing_required_object_quorum(1, 4, 2, false, 0), 1);
|
||||
assert_eq!(latest_listing_object_quorum(1, 4, 2, false, 0), 1);
|
||||
assert_eq!(latest_listing_required_object_quorum(2, 4, 2, false, 0), 2);
|
||||
assert_eq!(latest_listing_object_quorum(2, 4, 2, false, 0), 2);
|
||||
assert_eq!(latest_listing_required_object_quorum(1, 4, 2, false), 1);
|
||||
assert_eq!(latest_listing_object_quorum(1, 4, 2, false), 1);
|
||||
assert_eq!(latest_listing_required_object_quorum(2, 4, 2, false), 2);
|
||||
assert_eq!(latest_listing_object_quorum(2, 4, 2, false), 2);
|
||||
assert_eq!(expand_ask_disks_for_object_quorum(2, 4, 2), 2);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn latest_listing_object_quorum_relaxes_write_quorum_by_unreachable_drives() {
|
||||
// 4-drive set, EC 2+2: with one drive offline an object committed at
|
||||
// write quorum 3 can only ever show 2 metadata copies to the listing.
|
||||
let slack = latest_listing_write_quorum_slack(true, 4, 3);
|
||||
assert_eq!(slack, 1);
|
||||
assert_eq!(latest_listing_required_object_quorum(2, 4, 2, true, slack), 2);
|
||||
assert_eq!(latest_listing_required_object_quorum(2, 4, 2, true, 0), 3);
|
||||
|
||||
// The relaxed quorum never drops below the erasure read quorum, so a
|
||||
// quorum-deleted object (at most one stale copy) stays hidden even
|
||||
// with half the set unreachable.
|
||||
let slack = latest_listing_write_quorum_slack(true, 4, 2);
|
||||
assert_eq!(slack, 2);
|
||||
assert_eq!(latest_listing_required_object_quorum(1, 4, 2, true, slack), 2);
|
||||
|
||||
assert_eq!(latest_listing_write_quorum_slack(false, 4, 2), 0);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn latest_listing_resolves_degraded_object_when_a_set_drive_is_unreachable() {
|
||||
let mod_time = time::OffsetDateTime::from_unix_timestamp(1_705_312_300).expect("valid timestamp");
|
||||
let entry = test_object_meta_entry_with_erasure_versions("object", &[(mod_time, "etag", 2, 2)]);
|
||||
let entries = || MetaCacheEntries(vec![Some(entry.clone()), Some(entry.clone()), None]);
|
||||
|
||||
let slack = latest_listing_write_quorum_slack(true, 4, 3);
|
||||
let obj_quorum = latest_listing_object_quorum(2, 4, 2, true, slack);
|
||||
assert_eq!(obj_quorum, 2);
|
||||
let resolver = list_metadata_resolution_params("bucket".to_string(), 2, obj_quorum, false, slack);
|
||||
let resolved = resolve_listing_entries(entries(), resolver, true)
|
||||
.expect("object committed at write quorum should stay listable with one holder drive offline");
|
||||
assert_eq!(resolved.name, "object");
|
||||
|
||||
// Without the unreachable-drive slack the same sample is dropped even
|
||||
// though the object still satisfies read quorum for GET.
|
||||
let strict_quorum = latest_listing_object_quorum(2, 4, 2, true, 0);
|
||||
let strict_resolver = list_metadata_resolution_params("bucket".to_string(), 2, strict_quorum, false, 0);
|
||||
assert!(resolve_listing_entries(entries(), strict_resolver, true).is_none());
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn agreed_listing_entry_relaxes_write_quorum_by_unreachable_drives() {
|
||||
let mod_time = time::OffsetDateTime::from_unix_timestamp(1_705_312_300).expect("valid timestamp");
|
||||
let entry = test_object_meta_entry_with_erasure_versions("object", &[(mod_time, "etag", 2, 2)]);
|
||||
let mut resolver = list_metadata_resolution_params("bucket".to_string(), 2, 2, false, 1);
|
||||
|
||||
assert!(matches!(
|
||||
resolve_agreed_listing_entry(entry.clone(), 2, resolver.clone(), true),
|
||||
ListingEntryResolution::Resolved(_)
|
||||
));
|
||||
|
||||
resolver.write_quorum_slack = 0;
|
||||
assert!(matches!(
|
||||
resolve_agreed_listing_entry(entry, 2, resolver, true),
|
||||
ListingEntryResolution::NeedsSupplement(_, _)
|
||||
));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn cached_entry_supplement_check_honors_write_quorum_slack() {
|
||||
let mod_time = time::OffsetDateTime::from_unix_timestamp(1_705_312_300).expect("valid timestamp");
|
||||
let mut entry = test_object_meta_entry_with_erasure_versions("object", &[(mod_time, "etag", 2, 2)]);
|
||||
let cached = entry.xl_meta().expect("test entry should decode");
|
||||
let mut resolver = list_metadata_resolution_params("bucket".to_string(), 2, 2, false, 1);
|
||||
|
||||
assert!(!cached_entry_needs_supplement(&cached, 2, &resolver, true));
|
||||
|
||||
resolver.write_quorum_slack = 0;
|
||||
assert!(cached_entry_needs_supplement(&cached, 2, &resolver, true));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn latest_listing_object_quorum_requires_write_quorum_when_degraded_cannot_satisfy_it() {
|
||||
let required_quorum = latest_listing_required_object_quorum(2, 8, 4, true, 0);
|
||||
let required_quorum = latest_listing_required_object_quorum(2, 8, 4, true);
|
||||
let ask_disks = expand_ask_disks_for_object_quorum(4, 4, required_quorum);
|
||||
|
||||
assert_eq!(required_quorum, 5);
|
||||
assert_eq!(ask_disks, 4);
|
||||
assert_eq!(latest_listing_object_quorum(2, 8, 4, true, 0), 5);
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
async fn latest_listing_supplements_a_split_delete_marker_sample() {
|
||||
let object_mod_time = time::OffsetDateTime::from_unix_timestamp(1_705_312_300).expect("valid timestamp");
|
||||
let delete_mod_time = time::OffsetDateTime::from_unix_timestamp(1_705_312_400).expect("valid timestamp");
|
||||
let stale = test_object_meta_entry_with_erasure_versions("object", &[(object_mod_time, "object-etag", 4, 2)]);
|
||||
let deleted = test_object_with_delete_marker_meta_entry("object", object_mod_time, delete_mod_time);
|
||||
let entries = MetaCacheEntries(vec![
|
||||
Some(stale.clone()),
|
||||
Some(stale.clone()),
|
||||
Some(deleted.clone()),
|
||||
Some(deleted.clone()),
|
||||
]);
|
||||
let resolver = list_metadata_resolution_params("bucket".to_string(), 2, 4, false, 0);
|
||||
|
||||
assert_eq!(listing_entries_supplement_target(&entries, &resolver, true).as_deref(), Some("object"));
|
||||
assert_eq!(listing_entries_supplement_target(&entries, &resolver, false), None);
|
||||
|
||||
let mut primary = resolve_listing_entries(MetaCacheEntries(entries.0.clone()), resolver.clone(), true)
|
||||
.expect("the partial sample should fall back to the stale object version");
|
||||
assert!(!primary.is_latest_delete_marker());
|
||||
|
||||
let mut fallback_disks = Vec::new();
|
||||
let mut fallback_tempdirs = Vec::new();
|
||||
for _ in 0..2 {
|
||||
let tempdir = tempfile::tempdir().expect("fallback tempdir should be created");
|
||||
let endpoint =
|
||||
Endpoint::try_from(tempdir.path().to_str().expect("fallback path should be utf8")).expect("valid endpoint");
|
||||
let disk = new_disk(
|
||||
&endpoint,
|
||||
&DiskOption {
|
||||
cleanup: false,
|
||||
health_check: false,
|
||||
},
|
||||
)
|
||||
.await
|
||||
.expect("fallback disk should be created");
|
||||
disk.make_volume("bucket").await.expect("fallback bucket should be created");
|
||||
disk.write_all(
|
||||
"bucket",
|
||||
&format!("object/{STORAGE_FORMAT_FILE}"),
|
||||
bytes::Bytes::copy_from_slice(&deleted.metadata),
|
||||
)
|
||||
.await
|
||||
.expect("fallback metadata should be written");
|
||||
fallback_disks.push(disk);
|
||||
fallback_tempdirs.push(tempdir);
|
||||
}
|
||||
let supplement = ListingSupplement::new(
|
||||
ListingSupplementOptions {
|
||||
bucket: "bucket".to_string(),
|
||||
path: String::new(),
|
||||
recursive: true,
|
||||
incl_deleted: false,
|
||||
skip_hidden_prefix_check: false,
|
||||
filter_prefix: None,
|
||||
forward_to: None,
|
||||
per_disk_limit: 100,
|
||||
skip_total_timeout: true,
|
||||
walkdir_timeout: None,
|
||||
walkdir_stall_timeout: None,
|
||||
},
|
||||
Arc::new(fallback_disks),
|
||||
FallbackClaimTracker::default(),
|
||||
);
|
||||
|
||||
let mut supplemented = resolve_listing_entries_with_supplement(entries, resolver, true, supplement)
|
||||
.await
|
||||
.expect("the supplemented sample should resolve the committed delete marker");
|
||||
assert!(supplemented.is_latest_delete_marker());
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn latest_listing_supplement_keeps_a_subquorum_delete_marker_hidden() {
|
||||
let object_mod_time = time::OffsetDateTime::from_unix_timestamp(1_705_312_300).expect("valid timestamp");
|
||||
let delete_mod_time = time::OffsetDateTime::from_unix_timestamp(1_705_312_400).expect("valid timestamp");
|
||||
let stale = test_object_meta_entry_with_erasure_versions("object", &[(object_mod_time, "object-etag", 4, 2)]);
|
||||
let deleted = test_object_with_delete_marker_meta_entry("object", object_mod_time, delete_mod_time);
|
||||
let resolver = list_metadata_resolution_params("bucket".to_string(), 2, 4, false, 0);
|
||||
let entries = MetaCacheEntries(vec![
|
||||
Some(stale.clone()),
|
||||
Some(stale.clone()),
|
||||
Some(stale.clone()),
|
||||
Some(deleted.clone()),
|
||||
]);
|
||||
|
||||
assert_eq!(listing_entries_supplement_target(&entries, &resolver, true).as_deref(), Some("object"));
|
||||
|
||||
let mut resolved = resolve_listing_entries(
|
||||
MetaCacheEntries(vec![
|
||||
Some(stale.clone()),
|
||||
Some(stale.clone()),
|
||||
Some(stale),
|
||||
Some(deleted.clone()),
|
||||
Some(deleted.clone()),
|
||||
Some(deleted),
|
||||
]),
|
||||
resolver,
|
||||
true,
|
||||
)
|
||||
.expect("the previous object version should remain visible below delete-marker write quorum");
|
||||
|
||||
assert!(!resolved.is_latest_delete_marker());
|
||||
assert_eq!(latest_listing_object_quorum(2, 8, 4, true), 5);
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
@@ -9508,7 +9082,7 @@ mod test {
|
||||
&[(old_mod_time, "old-etag", 4, 4), (new_mod_time, "new-etag", 7, 1)],
|
||||
);
|
||||
let fallback_old_entry = test_object_meta_entry_with_erasure_versions("object", &[(old_mod_time, "old-etag", 4, 4)]);
|
||||
let resolver = list_metadata_resolution_params("bucket".to_string(), 3, 5, false, 0);
|
||||
let resolver = list_metadata_resolution_params("bucket".to_string(), 3, 5, false);
|
||||
let seen = Arc::new(Mutex::new(Vec::new()));
|
||||
let seen_clone = seen.clone();
|
||||
|
||||
@@ -9563,7 +9137,7 @@ mod test {
|
||||
"object",
|
||||
&[(old_mod_time, "old-etag", 4, 4), (new_mod_time, "new-etag", 7, 1)],
|
||||
);
|
||||
let resolver = list_metadata_resolution_params("bucket".to_string(), 3, 5, false, 0);
|
||||
let resolver = list_metadata_resolution_params("bucket".to_string(), 3, 5, false);
|
||||
let seen = Arc::new(Mutex::new(Vec::new()));
|
||||
let seen_clone = seen.clone();
|
||||
|
||||
@@ -9607,7 +9181,7 @@ mod test {
|
||||
let new_mod_time = time::OffsetDateTime::from_unix_timestamp(1_705_312_400).expect("valid timestamp");
|
||||
let entry = test_object_meta_entry_with_erasure_versions("object", &[(new_mod_time, "new-etag", 7, 1)]);
|
||||
let fallback_entry = entry.clone();
|
||||
let resolver = list_metadata_resolution_params("bucket".to_string(), 3, 5, false, 0);
|
||||
let resolver = list_metadata_resolution_params("bucket".to_string(), 3, 5, false);
|
||||
let seen = Arc::new(Mutex::new(Vec::new()));
|
||||
let seen_clone = seen.clone();
|
||||
|
||||
|
||||
+25
-1037
File diff suppressed because it is too large
Load Diff
@@ -13,7 +13,6 @@
|
||||
// limitations under the License.
|
||||
|
||||
use super::*;
|
||||
use crate::core::pools::{DecommissionCapacityOwner, ensure_decommission_capacity_mutation_id};
|
||||
use crate::multipart_listing::paginate_multipart_listing;
|
||||
use crate::set_disk::get_lock_acquire_timeout;
|
||||
use crate::storage_api_contracts::multipart::MultipartOperations as _;
|
||||
@@ -197,21 +196,6 @@ async fn list_pool_multipart_uploads_for_incarnation(
|
||||
}
|
||||
|
||||
impl ECStore {
|
||||
pub(crate) async fn acquire_decommission_multipart_mutation_fence(
|
||||
&self,
|
||||
owner: DecommissionCapacityOwner,
|
||||
) -> Result<ObjectLockDiagGuard> {
|
||||
let mutation_id = owner
|
||||
.mutation_id
|
||||
.ok_or_else(|| Error::other("decommission multipart mutation identity is missing"))?;
|
||||
let object = format!(
|
||||
"decommission-multipart/{}/{}/{}/{}",
|
||||
owner.source_pool_index, owner.operation_id, owner.generation, mutation_id
|
||||
);
|
||||
self.acquire_object_write_lock("decommission_multipart_mutation", crate::disk::RUSTFS_META_MULTIPART_BUCKET, &object)
|
||||
.await
|
||||
}
|
||||
|
||||
async fn existing_multipart_pool_order(&self) -> Vec<usize> {
|
||||
// A draining source must not hide a valid UploadID in an active target,
|
||||
// while physical order within each phase preserves fail-closed errors.
|
||||
@@ -231,24 +215,6 @@ impl ECStore {
|
||||
active
|
||||
}
|
||||
|
||||
async fn multipart_upload_pool_idx(
|
||||
&self,
|
||||
bucket: &str,
|
||||
object: &str,
|
||||
upload_id: &str,
|
||||
opts: &ObjectOptions,
|
||||
) -> Result<usize> {
|
||||
for pool_idx in self.existing_multipart_pool_order().await {
|
||||
match self.pools[pool_idx].get_multipart_info(bucket, object, upload_id, opts).await {
|
||||
Ok(_) => return Ok(pool_idx),
|
||||
Err(err) if is_err_invalid_upload_id(&err) => continue,
|
||||
Err(err) => return Err(err),
|
||||
}
|
||||
}
|
||||
|
||||
Err(StorageError::InvalidUploadID(bucket.to_owned(), object.to_owned(), upload_id.to_owned()))
|
||||
}
|
||||
|
||||
#[allow(clippy::too_many_arguments)]
|
||||
pub async fn list_multipart_uploads_for_bucket_incarnation(
|
||||
&self,
|
||||
@@ -467,10 +433,8 @@ impl ECStore {
|
||||
if self.single_pool() {
|
||||
self.apply_decommission_target_mutation_fence(0, object, &mut opts, mutation_fence)
|
||||
.await;
|
||||
return self
|
||||
.run_decommission_capacity_admitted_mutation(0, None, None, || async {
|
||||
self.pools[0].new_multipart_upload(bucket, object, &opts).await
|
||||
})
|
||||
return self.pools[0]
|
||||
.new_multipart_upload(bucket, object, &opts)
|
||||
.await
|
||||
.map(|res| (res, 0, opts.expected_bucket_incarnation_id));
|
||||
}
|
||||
@@ -486,14 +450,7 @@ impl ECStore {
|
||||
}
|
||||
self.apply_decommission_target_mutation_fence(idx, object, &mut opts, mutation_fence)
|
||||
.await;
|
||||
let res = self
|
||||
.run_decommission_capacity_temporary_mutation(
|
||||
idx,
|
||||
DecommissionCapacityOwner::from_options(&opts),
|
||||
None,
|
||||
|| async { self.pools[idx].new_multipart_upload(bucket, object, &opts).await },
|
||||
)
|
||||
.await?;
|
||||
let res = self.pools[idx].new_multipart_upload(bucket, object, &opts).await?;
|
||||
return Ok((res, idx, opts.expected_bucket_incarnation_id));
|
||||
}
|
||||
|
||||
@@ -518,19 +475,8 @@ impl ECStore {
|
||||
if !res.uploads.is_empty() {
|
||||
self.apply_decommission_target_mutation_fence(idx, object, &mut opts, mutation_fence)
|
||||
.await;
|
||||
let expected_bucket_incarnation_id = opts.expected_bucket_incarnation_id;
|
||||
let lock_object = encode_dir_object(object);
|
||||
let res = self
|
||||
.run_external_decommission_capacity_object_mutation(
|
||||
idx,
|
||||
bucket,
|
||||
&lock_object,
|
||||
object,
|
||||
opts,
|
||||
|opts| async move { self.pools[idx].new_multipart_upload(bucket, object, &opts).await },
|
||||
)
|
||||
.await?;
|
||||
return Ok((res, idx, expected_bucket_incarnation_id));
|
||||
let res = self.pools[idx].new_multipart_upload(bucket, object, &opts).await?;
|
||||
return Ok((res, idx, opts.expected_bucket_incarnation_id));
|
||||
}
|
||||
}
|
||||
let idx = self.get_pool_idx(bucket, object, -1).await?;
|
||||
@@ -544,23 +490,8 @@ impl ECStore {
|
||||
|
||||
self.apply_decommission_target_mutation_fence(idx, object, &mut opts, mutation_fence)
|
||||
.await;
|
||||
let expected_bucket_incarnation_id = opts.expected_bucket_incarnation_id;
|
||||
let res = if opts.data_movement {
|
||||
self.run_decommission_capacity_temporary_mutation(
|
||||
idx,
|
||||
DecommissionCapacityOwner::from_options(&opts),
|
||||
None,
|
||||
|| async { self.pools[idx].new_multipart_upload(bucket, object, &opts).await },
|
||||
)
|
||||
.await?
|
||||
} else {
|
||||
let lock_object = encode_dir_object(object);
|
||||
self.run_external_decommission_capacity_object_mutation(idx, bucket, &lock_object, object, opts, |opts| async move {
|
||||
self.pools[idx].new_multipart_upload(bucket, object, &opts).await
|
||||
})
|
||||
.await?
|
||||
};
|
||||
Ok((res, idx, expected_bucket_incarnation_id))
|
||||
let res = self.pools[idx].new_multipart_upload(bucket, object, &opts).await?;
|
||||
Ok((res, idx, opts.expected_bucket_incarnation_id))
|
||||
}
|
||||
|
||||
#[instrument(skip(self))]
|
||||
@@ -598,8 +529,7 @@ impl ECStore {
|
||||
opts: &ObjectOptions,
|
||||
) -> Result<PartInfo> {
|
||||
check_put_object_part_args(bucket, object, upload_id)?;
|
||||
let (mut opts, _bucket_lifecycle_guard) = self.guard_multipart_bucket_incarnation(bucket, opts).await?;
|
||||
opts.decommission_capacity_admission = crate::bucket::metadata_sys::object_store_if_initialized_in(&self.ctx).await;
|
||||
let (opts, _bucket_lifecycle_guard) = self.guard_multipart_bucket_incarnation(bucket, opts).await?;
|
||||
let opts = &opts;
|
||||
|
||||
if self.single_pool() {
|
||||
@@ -608,10 +538,26 @@ impl ECStore {
|
||||
.await;
|
||||
}
|
||||
|
||||
let pool_idx = self.multipart_upload_pool_idx(bucket, object, upload_id, opts).await?;
|
||||
self.pools[pool_idx]
|
||||
.put_object_part(bucket, object, upload_id, part_id, data, opts)
|
||||
.await
|
||||
for pool_idx in self.existing_multipart_pool_order().await {
|
||||
let pool = &self.pools[pool_idx];
|
||||
let err = match pool.put_object_part(bucket, object, upload_id, part_id, data, opts).await {
|
||||
Ok(res) => return Ok(res),
|
||||
Err(err) => {
|
||||
if is_err_invalid_upload_id(&err) {
|
||||
None
|
||||
} else {
|
||||
Some(err)
|
||||
}
|
||||
}
|
||||
};
|
||||
|
||||
if let Some(err) = err {
|
||||
error!("put_object_part err: {:?}", err);
|
||||
return Err(err);
|
||||
}
|
||||
}
|
||||
|
||||
Err(StorageError::InvalidUploadID(bucket.to_owned(), object.to_owned(), upload_id.to_owned()))
|
||||
}
|
||||
|
||||
pub(crate) async fn put_object_part_for_data_movement(
|
||||
@@ -630,24 +576,12 @@ impl ECStore {
|
||||
if !opts.data_movement {
|
||||
return Err(Error::other("targeted multipart upload requires data_movement options"));
|
||||
}
|
||||
let (mut opts, _bucket_lifecycle_guard) = self.guard_multipart_bucket_incarnation(bucket, opts).await?;
|
||||
ensure_decommission_capacity_mutation_id(bucket, object, &mut opts);
|
||||
let pool = self.pools.get(target_pool_idx).ok_or_else(|| {
|
||||
Error::InvalidArgument("data-movement".to_string(), "target-pool".to_string(), target_pool_idx.to_string())
|
||||
})?;
|
||||
let expected_data_bytes = usize::try_from(data.size()).ok();
|
||||
self.run_decommission_capacity_temporary_mutation_with_capacity_lease(
|
||||
target_pool_idx,
|
||||
DecommissionCapacityOwner::from_options(&opts),
|
||||
expected_data_bytes,
|
||||
|capacity_lease| async move {
|
||||
if let Some(capacity_lease) = capacity_lease {
|
||||
opts.add_namespace_lock_lost_signal(capacity_lease);
|
||||
}
|
||||
pool.put_object_part(bucket, object, upload_id, part_id, data, &opts).await
|
||||
},
|
||||
)
|
||||
.await
|
||||
let (opts, _bucket_lifecycle_guard) = self.guard_multipart_bucket_incarnation(bucket, opts).await?;
|
||||
let pool = self
|
||||
.pools
|
||||
.get(target_pool_idx)
|
||||
.ok_or_else(|| Error::other(format!("data movement target pool {target_pool_idx} is out of range")))?;
|
||||
pool.put_object_part(bucket, object, upload_id, part_id, data, &opts).await
|
||||
}
|
||||
|
||||
#[instrument(skip(self))]
|
||||
@@ -728,95 +662,16 @@ impl ECStore {
|
||||
upload_id: &str,
|
||||
opts: &ObjectOptions,
|
||||
) -> Result<()> {
|
||||
self.abort_multipart_uploads_for_data_movement(target_pool_idx, bucket, object, &[upload_id.to_owned()], None, opts)
|
||||
.await
|
||||
}
|
||||
|
||||
pub(crate) async fn reconcile_multipart_uploads_for_data_movement(
|
||||
&self,
|
||||
target_pool_idx: usize,
|
||||
bucket: &str,
|
||||
object: &str,
|
||||
upload_identity: &str,
|
||||
opts: &ObjectOptions,
|
||||
) -> Result<()> {
|
||||
let pool = self
|
||||
.pools
|
||||
.get(target_pool_idx)
|
||||
.ok_or_else(|| Error::other(format!("data movement target pool {target_pool_idx} is out of range")))?;
|
||||
let owner = DecommissionCapacityOwner::from_options(opts);
|
||||
let has_capacity_state = match owner {
|
||||
Some(owner) => {
|
||||
self.has_decommission_capacity_temporary_mutation_state(target_pool_idx, owner)
|
||||
.await
|
||||
}
|
||||
None => false,
|
||||
};
|
||||
if !has_capacity_state {
|
||||
return Ok(());
|
||||
}
|
||||
let set = pool.get_disks_by_key(object);
|
||||
let upload_ids = set
|
||||
.data_movement_multipart_upload_ids(bucket, object, opts.expected_bucket_incarnation_id, upload_identity)
|
||||
.await?;
|
||||
self.abort_multipart_uploads_for_data_movement(target_pool_idx, bucket, object, &upload_ids, Some(upload_identity), opts)
|
||||
.await
|
||||
}
|
||||
|
||||
async fn abort_multipart_uploads_for_data_movement(
|
||||
&self,
|
||||
target_pool_idx: usize,
|
||||
bucket: &str,
|
||||
object: &str,
|
||||
upload_ids: &[String],
|
||||
expected_upload_identity: Option<&str>,
|
||||
opts: &ObjectOptions,
|
||||
) -> Result<()> {
|
||||
check_new_multipart_args(bucket, object)?;
|
||||
for upload_id in upload_ids {
|
||||
check_abort_multipart_args(bucket, object, upload_id)?;
|
||||
}
|
||||
check_abort_multipart_args(bucket, object, upload_id)?;
|
||||
if !opts.data_movement {
|
||||
return Err(Error::other("targeted multipart abort requires data_movement options"));
|
||||
}
|
||||
let (mut opts, _bucket_lifecycle_guard) = self.guard_multipart_bucket_incarnation(bucket, opts).await?;
|
||||
ensure_decommission_capacity_mutation_id(bucket, object, &mut opts);
|
||||
let (opts, _bucket_lifecycle_guard) = self.guard_multipart_bucket_incarnation(bucket, opts).await?;
|
||||
let pool = self
|
||||
.pools
|
||||
.get(target_pool_idx)
|
||||
.ok_or_else(|| Error::other(format!("data movement target pool {target_pool_idx} is out of range")))?;
|
||||
let set = pool.get_disks_by_key(object);
|
||||
let mut guards = Vec::with_capacity(upload_ids.len());
|
||||
for upload_id in upload_ids {
|
||||
if let Some(guard) = set
|
||||
.lock_data_movement_multipart_abort(bucket, object, upload_id, expected_upload_identity, &opts)
|
||||
.await?
|
||||
{
|
||||
guard.add_namespace_lock_fence(&mut opts);
|
||||
guards.push(guard);
|
||||
}
|
||||
}
|
||||
opts.no_lock = true;
|
||||
let capacity_owner = DecommissionCapacityOwner::from_options(&opts);
|
||||
// Keep every upload namespace guard alive through the final capacity progress save.
|
||||
let result = self
|
||||
.run_decommission_capacity_temporary_release_with_capacity_lease(target_pool_idx, capacity_owner, |capacity_lease| {
|
||||
let mut delete_opts = opts.clone();
|
||||
let guards = &guards;
|
||||
let set = &set;
|
||||
async move {
|
||||
if let Some(capacity_lease) = capacity_lease {
|
||||
delete_opts.add_namespace_lock_lost_signal(capacity_lease);
|
||||
}
|
||||
for guard in guards {
|
||||
guard.delete(set, bucket, object, &delete_opts).await?;
|
||||
}
|
||||
Ok(())
|
||||
}
|
||||
})
|
||||
.await;
|
||||
drop(guards);
|
||||
result
|
||||
pool.abort_multipart_upload(bucket, object, upload_id, &opts).await
|
||||
}
|
||||
|
||||
#[instrument(skip(self))]
|
||||
@@ -829,8 +684,7 @@ impl ECStore {
|
||||
opts: &ObjectOptions,
|
||||
) -> Result<ObjectInfo> {
|
||||
check_complete_multipart_args(bucket, object, upload_id)?;
|
||||
let (mut opts, _bucket_lifecycle_guard) = self.guard_multipart_bucket_incarnation(bucket, opts).await?;
|
||||
opts.decommission_capacity_admission = crate::bucket::metadata_sys::object_store_if_initialized_in(&self.ctx).await;
|
||||
let (opts, _bucket_lifecycle_guard) = self.guard_multipart_bucket_incarnation(bucket, opts).await?;
|
||||
let opts = &opts;
|
||||
|
||||
if self.single_pool() {
|
||||
@@ -840,13 +694,29 @@ impl ECStore {
|
||||
.await;
|
||||
}
|
||||
|
||||
let pool_idx = self.multipart_upload_pool_idx(bucket, object, upload_id, opts).await?;
|
||||
let pool = self.pools[pool_idx].clone();
|
||||
pool.complete_multipart_upload(bucket, object, upload_id, uploaded_parts, opts)
|
||||
.await
|
||||
for pool_idx in self.existing_multipart_pool_order().await {
|
||||
let pool = &self.pools[pool_idx];
|
||||
|
||||
let pool = pool.clone();
|
||||
let err = match pool
|
||||
.complete_multipart_upload(bucket, object, upload_id, uploaded_parts.clone(), opts)
|
||||
.await
|
||||
{
|
||||
Ok(res) => return Ok(res),
|
||||
Err(err) => {
|
||||
//
|
||||
if is_err_invalid_upload_id(&err) { None } else { Some(err) }
|
||||
}
|
||||
};
|
||||
|
||||
if let Some(er) = err {
|
||||
return Err(er);
|
||||
}
|
||||
}
|
||||
|
||||
Err(StorageError::InvalidUploadID(bucket.to_owned(), object.to_owned(), upload_id.to_owned()))
|
||||
}
|
||||
|
||||
#[cfg(all(test, feature = "test-util"))]
|
||||
pub(crate) async fn complete_multipart_upload_for_data_movement(
|
||||
self: Arc<Self>,
|
||||
target: (usize, Option<&ObjectLockDiagGuard>),
|
||||
@@ -855,44 +725,6 @@ impl ECStore {
|
||||
upload_id: &str,
|
||||
uploaded_parts: Vec<CompletePart>,
|
||||
opts: &ObjectOptions,
|
||||
) -> Result<ObjectInfo> {
|
||||
self.complete_multipart_upload_for_data_movement_inner(target, bucket, object, upload_id, uploaded_parts, opts, None)
|
||||
.await
|
||||
}
|
||||
|
||||
#[allow(clippy::too_many_arguments)]
|
||||
pub(crate) async fn complete_multipart_upload_for_data_movement_with_publication_fence(
|
||||
self: Arc<Self>,
|
||||
target_pool_idx: usize,
|
||||
bucket: &str,
|
||||
object: &str,
|
||||
upload_id: &str,
|
||||
uploaded_parts: Vec<CompletePart>,
|
||||
opts: &ObjectOptions,
|
||||
publication_fence: RemoteTuplePublicationFence,
|
||||
) -> Result<ObjectInfo> {
|
||||
self.complete_multipart_upload_for_data_movement_inner(
|
||||
(target_pool_idx, None),
|
||||
bucket,
|
||||
object,
|
||||
upload_id,
|
||||
uploaded_parts,
|
||||
opts,
|
||||
Some(publication_fence),
|
||||
)
|
||||
.await
|
||||
}
|
||||
|
||||
#[allow(clippy::too_many_arguments)]
|
||||
async fn complete_multipart_upload_for_data_movement_inner(
|
||||
self: Arc<Self>,
|
||||
target: (usize, Option<&ObjectLockDiagGuard>),
|
||||
bucket: &str,
|
||||
object: &str,
|
||||
upload_id: &str,
|
||||
uploaded_parts: Vec<CompletePart>,
|
||||
opts: &ObjectOptions,
|
||||
publication_fence: Option<RemoteTuplePublicationFence>,
|
||||
) -> Result<ObjectInfo> {
|
||||
let (target_pool_idx, mutation_fence) = target;
|
||||
check_complete_multipart_args(bucket, object, upload_id)?;
|
||||
@@ -900,7 +732,6 @@ impl ECStore {
|
||||
return Err(Error::other("targeted multipart completion requires data_movement options"));
|
||||
}
|
||||
let (mut opts, _bucket_lifecycle_guard) = self.guard_multipart_bucket_incarnation(bucket, opts).await?;
|
||||
ensure_decommission_capacity_mutation_id(bucket, object, &mut opts);
|
||||
if opts.overwrites_existing_version() && !is_meta_bucketname(bucket) {
|
||||
let expected_incarnation_id = opts
|
||||
.expected_bucket_incarnation_id
|
||||
@@ -924,36 +755,8 @@ impl ECStore {
|
||||
snapshot.add_lock_fences(&mut opts);
|
||||
opts.object_lock_config_snapshot = Some(snapshot);
|
||||
}
|
||||
let fixed_read_anchor = publication_fence
|
||||
.as_ref()
|
||||
.and_then(RemoteTuplePublicationFence::fixed_read_anchor_guard);
|
||||
self.apply_decommission_target_mutation_fence(target_pool_idx, object, &mut opts, mutation_fence.or(fixed_read_anchor))
|
||||
self.apply_decommission_target_mutation_fence(target_pool_idx, object, &mut opts, mutation_fence)
|
||||
.await;
|
||||
// NewMultipart/UploadPart are staging only. Acquire and consume the
|
||||
// non-cloneable publication capability immediately before Complete,
|
||||
// then retain its guards until Complete has drained the commit path.
|
||||
let publication_object = encode_dir_object(object);
|
||||
let publication_guard = match publication_fence {
|
||||
Some(publication_fence) => {
|
||||
let guard = publication_fence
|
||||
.into_commit_guard(target_pool_idx, bucket, &publication_object)
|
||||
.await?;
|
||||
guard.add_namespace_lock_fence(&mut opts);
|
||||
opts.no_lock = true;
|
||||
Some(guard)
|
||||
}
|
||||
None => {
|
||||
if rustfs_utils::http::metadata_compat::contains_key_str(
|
||||
&opts.user_defined,
|
||||
rustfs_utils::http::SUFFIX_TRANSITION_STATUS,
|
||||
) {
|
||||
return Err(Error::other(
|
||||
"data movement multipart completion cannot publish transition ownership without a publication capability",
|
||||
));
|
||||
}
|
||||
None
|
||||
}
|
||||
};
|
||||
#[cfg(test)]
|
||||
pause_data_movement_multipart_before_selected_completion(bucket).await;
|
||||
let pool = self
|
||||
@@ -961,25 +764,12 @@ impl ECStore {
|
||||
.get(target_pool_idx)
|
||||
.ok_or_else(|| Error::other(format!("data movement target pool {target_pool_idx} is out of range")))?
|
||||
.clone();
|
||||
// Data movement already owns the pool-meta write lease. Forward its
|
||||
// loss signal into SetDisks so commit fencing observes the same lease
|
||||
// without trying to reacquire the namespace.
|
||||
let result = self
|
||||
.run_decommission_capacity_admitted_mutation_with_capacity_lease(
|
||||
target_pool_idx,
|
||||
DecommissionCapacityOwner::from_options(&opts),
|
||||
opts.capacity_expected_data_bytes(),
|
||||
|capacity_lease| async move {
|
||||
if let Some(capacity_lease) = capacity_lease {
|
||||
opts.add_namespace_lock_lost_signal(capacity_lease);
|
||||
}
|
||||
pool.complete_multipart_upload(bucket, object, upload_id, uploaded_parts, &opts)
|
||||
.await
|
||||
},
|
||||
)
|
||||
.await;
|
||||
drop(publication_guard);
|
||||
let result = enqueue_transition_after_write(result, LcEventSrc::S3CompleteMultipartUpload).await;
|
||||
let result = enqueue_transition_after_write(
|
||||
pool.complete_multipart_upload(bucket, object, upload_id, uploaded_parts, &opts)
|
||||
.await,
|
||||
LcEventSrc::S3CompleteMultipartUpload,
|
||||
)
|
||||
.await;
|
||||
if result.is_ok() {
|
||||
list_objects::observe_list_objects_mutation(self.as_ref(), bucket).await;
|
||||
}
|
||||
@@ -1171,7 +961,6 @@ mod tests {
|
||||
decommission_cancelers: RwLock::new(Vec::new()),
|
||||
start_gate: Mutex::new(()),
|
||||
pool_meta_save_gate: Mutex::default(),
|
||||
decommission_capacity_entry_gate: Mutex::default(),
|
||||
ctx: crate::runtime::instance::bootstrap_ctx(),
|
||||
bucket_fence_registry: std::sync::Arc::default(),
|
||||
}
|
||||
|
||||
+414
-2166
File diff suppressed because it is too large
Load Diff
Some files were not shown because too many files have changed in this diff Show More
Reference in New Issue
Block a user