mirror of
https://github.com/rustfs/rustfs.git
synced 2026-09-04 19:25:40 +00:00
Compare commits
123 Commits
| Author | SHA1 | Date | |
|---|---|---|---|
| 40a2470feb | |||
| 7dcfdb3320 | |||
| 397dbcf102 | |||
| 99073938ae | |||
| d22991f33b | |||
| 194c8643c0 | |||
| 5720c5c748 | |||
| 1941189499 | |||
| bba934723a | |||
| cebe57a2f0 | |||
| 6a8a8a1eaf | |||
| 833cc51534 | |||
| 43450df589 | |||
| 394394cdfc | |||
| af896dc427 | |||
| 297ff4688c | |||
| b9b2aa0b76 | |||
| 1dcdfe4817 | |||
| 6e26769265 | |||
| bd66fa9dca | |||
| 1aea7541c8 | |||
| 9e6d34785b | |||
| 03aecc5c3e | |||
| 45fe54e389 | |||
| 2ed5c297ac | |||
| 23ab078c56 | |||
| 80c629bfe0 | |||
| 47304cc68d | |||
| cee84561e7 | |||
| b09ce8e6b5 | |||
| a45951260a | |||
| a41134eb8a | |||
| 35ce8cdb80 | |||
| 0b1a588da5 | |||
| f6c6736a01 | |||
| ab44ae7e83 | |||
| b0256e3453 | |||
| 14a77f9d79 | |||
| 436a1be899 | |||
| 1ea1dfa0a1 | |||
| c45a8c35c4 | |||
| 4932d1dedf | |||
| 041af14143 | |||
| e44007012b | |||
| e3ca1ca54c | |||
| ec0a65703a | |||
| 0d1e40ee73 | |||
| 281e40f1cc | |||
| 7541bb2c5d | |||
| 25dd879cf4 | |||
| af1ebbfb8e | |||
| 3e3eb4d8d5 | |||
| 48b6548988 | |||
| 655f6ae452 | |||
| 61821a6f3e | |||
| 896781a52b | |||
| 612dd38fea | |||
| ff28b79088 | |||
| 35456bcede | |||
| 9d4ccb7884 | |||
| 9a22cb85f3 | |||
| 6c67086d0b | |||
| ea01cd339c | |||
| bb37841362 | |||
| 59a7194d7f | |||
| f647ada320 | |||
| 589a954478 | |||
| 1d606e1cf6 | |||
| dc2e25b48c | |||
| d690f5d60d | |||
| 3eca80e37d | |||
| 45a2ccb734 | |||
| c876df53f5 | |||
| 769da6d81f | |||
| ca46ae9e56 | |||
| 7df0920c80 | |||
| c4ac11d22e | |||
| 602ed2cbcd | |||
| b6c3108e53 | |||
| 8ecd8f2520 | |||
| e6234d3714 | |||
| 042a0c3014 | |||
| 87333f7b24 | |||
| 9945c67f7e | |||
| fca1514aac | |||
| 47ad69b691 | |||
| 489408c0b0 | |||
| 1b3744a1da | |||
| 9244eb36ed | |||
| 442298d5f7 | |||
| be7d35d441 | |||
| ec1cd606d3 | |||
| 16af688a7a | |||
| 37b23a16da | |||
| 006e9b7d28 | |||
| d214c27583 | |||
| 8fd364a99c | |||
| c2d8488728 | |||
| 1370434f3a | |||
| 5dde2c188c | |||
| 2f9c75d04f | |||
| 9ee7b1221d | |||
| fcc3c7fb6b | |||
| 01dc55ee5b | |||
| 3d24526704 | |||
| 51532e19fb | |||
| 931ff60182 | |||
| 07212c4e26 | |||
| 4932af080b | |||
| d6f9a7c462 | |||
| 7345b49cf6 | |||
| 4753e35035 | |||
| 96239fc034 | |||
| b428875bed | |||
| cf362282f0 | |||
| b2a2e637a5 | |||
| 0c18012442 | |||
| ee39e4fccb | |||
| 90ab2e24c3 | |||
| 21e5b3dc64 | |||
| 1e8c8d4cd5 | |||
| ff3ad30f0c | |||
| 47a3f5ef01 |
@@ -48,6 +48,7 @@ Update this file only when an advisory adds or changes a reusable lesson, affect
|
||||
|
||||
### S3 object actions, copy, multipart, and upload policy validation
|
||||
|
||||
- `GHSA-g8w9-qw9q-fghr`: a valid presigned `PutObject` accepted extra `x-amz-tagging`, website redirect, and storage-class headers omitted from `SignedHeaders`. Lesson: a presigned URL is a bounded capability; reject `x-amz-*` headers that are not cryptographically bound by the signature so unsigned metadata cannot change authorization, lifecycle, redirect, cost, or durability semantics.
|
||||
- `GHSA-3ppv-fx5m-m749`: explicit `versionId` reads and copy sources authorized `s3:GetObject` instead of `s3:GetObjectVersion`. Lesson: version-specific object access must select version-specific actions for direct reads, `CopyObject`, and `UploadPartCopy`, with tests proving the backend is not reached on denial.
|
||||
- `GHSA-x298-9x87-fvjq`: anonymous `ListObjectVersions` fell back to `ListBucket` and returned before public-access-block gates. Lesson: compatibility fallbacks must converge on the same post-authorization checks as direct grants, especially `RestrictPublicBuckets` and anonymous data-plane denies.
|
||||
- `GHSA-mx42-j6wv-px98`: `UploadPartCopy` missed source authorization and allowed cross-bucket object exfiltration. Lesson: multipart copy must enforce the same source and destination contract as `CopyObject`.
|
||||
@@ -119,7 +120,7 @@ Use these targeted searches when a diff touches security-sensitive code:
|
||||
```bash
|
||||
rg -n "validate_admin_request|check_permissions|AdminAction::|deny_only|is_allowed" rustfs crates
|
||||
rg -n "authorize_operation|FtpsDriver|SftpDriver|RETR|MKD|SIZE|MDTM|CreateBucket|GetObject|HeadObject" crates/protocols rustfs
|
||||
rg -n "UploadPartCopy|upload_part_copy|CompleteMultipart|PostObject|content-length-range|starts-with" rustfs crates
|
||||
rg -n "UploadPartCopy|upload_part_copy|CompleteMultipart|PostObject|presign|SignedHeaders|content-length-range|starts-with" rustfs crates
|
||||
rg -n "ListBucketVersions|GetObjectVersion|versionId|VersionId|ExistingObjectTag|ForAllValues|ForAnyValue|POLICY_PLUGIN|opa" rustfs crates
|
||||
rg -n "normalize_extract_entry_key|Snowball|auto-extract|PathBuf::join|canonicalize|\\.\\.|x-forwarded-for|x-real-ip|SourceIp" rustfs crates
|
||||
rg -n "DEFAULT_SECRET|DEFAULT_ACCESS|TEST_PRIVATE_KEY|rustfs rpc|RUSTFS_RPC_SECRET" rustfs crates
|
||||
@@ -136,6 +137,7 @@ rg -n "deny_unknown_fields|serde.default|as u32|as usize|as i32" rustfs crates
|
||||
- Protocol frontend authz fixes: include denied `RETR`, `SIZE`/`MDTM`, `MKD`, bucket probe, and sibling allowed-operation cases, and assert denied paths do not reach the storage backend.
|
||||
- IAM fixes: include import/update/list service-account cases with attacker-controlled parent, claims, access key, secret key, and policy.
|
||||
- Copy/upload fixes: include cross-bucket, cross-user, source-denied, destination-denied, copy-source-condition, and multipart completion cases.
|
||||
- Presigned upload fixes: include a valid presign with extra unsigned tagging, redirect, and storage-class headers; require rejection before storage access, and verify explicitly signed equivalents still work.
|
||||
- Version-action fixes: include historical UUID, explicit current version, `null`, range, partNumber, presigned, STS/session, service-account, anonymous bucket-policy, copy source, and multipart-copy source cases.
|
||||
- Policy-condition fixes: include reserved-key header collisions, missing keys, partially overlapping multi-value sets, plugin mode, and built-in policy mode.
|
||||
- Path fixes: include encoded traversal, absolute path, nested traversal, archive entries with `..`, valid object keys that resemble traversal text but should be rejected, and canonical bucket/prefix boundary checks.
|
||||
|
||||
@@ -1,2 +1,2 @@
|
||||
sha256-darwin=d6aa36cfaae2c4d8590482c7e47138c5965b335b34a75f50d11ffc3366e9021e
|
||||
sha256-linux=96db8060fce98addda4f69092d297ca236bec4892d820617a26a261eedac61b0
|
||||
sha256-darwin=ef914ec0b8daa9c2c5e52f501d339914662f42d6f6ed9d33877d56b97adf16f9
|
||||
sha256-linux=a8a816d7bb0e7cb5632b1863b33794bcb9fc7e765f150aa5e1bf16518e28dfb4
|
||||
|
||||
@@ -1 +1 @@
|
||||
sha256=9b9bc336b43b70d0e06e0adb5455bf035bb18945d85d60936eb6fe4d48e0e680
|
||||
sha256=51da41c54167602f2bd6c45921b39a44562bf3cfcdf468d992bb992c62cad7fd
|
||||
|
||||
@@ -1 +1 @@
|
||||
sha256=294350518743cac8d7c41880a2835216e4b697908d7b0b1bc92b62816d94c59d
|
||||
sha256=dbebfbab9b9efd4eff31211e69dd32235dc00e207f2ab0dd919a1b2ac9e724c2
|
||||
|
||||
@@ -23,4 +23,4 @@ coverage: core-deps ## Workspace line coverage (cargo-llvm-cov + nextest; slow,
|
||||
@mkdir -p target/llvm-cov
|
||||
cargo llvm-cov report --lcov --output-path target/llvm-cov/lcov.info
|
||||
cargo llvm-cov report --json --output-path target/llvm-cov/coverage.json
|
||||
python3 scripts/coverage_per_crate.py target/llvm-cov/coverage.json
|
||||
$(RUSTFS_PYTHON_BIN) scripts/coverage_per_crate.py target/llvm-cov/coverage.json
|
||||
|
||||
@@ -88,7 +88,7 @@ offline-enrollment-e2e-check: core-deps ## Build and exercise the dedicated offl
|
||||
.PHONY: test-wiring-check
|
||||
test-wiring-check: ## Check tests stay registered and selected by their intended runners
|
||||
@echo "🧪 Checking test wiring..."
|
||||
python3 ./scripts/check_test_wiring.py
|
||||
$(RUSTFS_PYTHON_BIN) ./scripts/check_test_wiring.py
|
||||
|
||||
.PHONY: log-analyzer-rules-check
|
||||
log-analyzer-rules-check: core-deps ## Check log-analyzer rule anchors still exist verbatim in source
|
||||
|
||||
@@ -35,13 +35,14 @@ script-tests: ## Run shell script tests
|
||||
./scripts/test_pinned_paired_abba_bench.sh
|
||||
./scripts/test_manual_transition_runbooks.sh
|
||||
./scripts/test_fuzz_runner.sh
|
||||
./scripts/test_python_bin.sh
|
||||
./scripts/check_embedded_secrets.sh --self-test
|
||||
python3 ./scripts/check_test_wiring.py --self-test
|
||||
python3 ./scripts/check_security_coverage.py --self-test
|
||||
python3 ./scripts/check_scheduled_validation_freshness.py --self-test
|
||||
python3 ./scripts/s3-tests/test_report_compat.py
|
||||
$(RUSTFS_PYTHON_BIN) ./scripts/check_test_wiring.py --self-test
|
||||
$(RUSTFS_PYTHON_BIN) ./scripts/check_security_coverage.py --self-test
|
||||
$(RUSTFS_PYTHON_BIN) ./scripts/check_scheduled_validation_freshness.py --self-test
|
||||
$(RUSTFS_PYTHON_BIN) ./scripts/s3-tests/test_report_compat.py
|
||||
bash -n ./scripts/validate_object_data_cache_cold_stampede.sh
|
||||
python3 ./scripts/check_object_data_cache_follower_samples.py --self-test
|
||||
$(RUSTFS_PYTHON_BIN) ./scripts/check_object_data_cache_follower_samples.py --self-test
|
||||
./scripts/validate_object_data_cache_cold_stampede.sh --self-test
|
||||
|
||||
.PHONY: test
|
||||
|
||||
+81
-15
@@ -46,6 +46,11 @@ e2e-reliability = { max-threads = 1 }
|
||||
e2e-inline-boundaries = { max-threads = 1 }
|
||||
e2e-cluster-nightly = { max-threads = 1 }
|
||||
|
||||
# Deep async storage futures are composed into tests across several crates.
|
||||
# Keep the test stack bounded but above libtest's 2 MiB default.
|
||||
[scripts.setup.ecstore-base-stack]
|
||||
command = ['sh', '-c', 'echo RUST_MIN_STACK=4194304 >> "$NEXTEST_ENV"']
|
||||
|
||||
# These exact regression scenarios build deep async storage futures that exceed
|
||||
# libtest's 2 MiB spawned-thread stack on Linux. Give only their test processes
|
||||
# the same 32 MiB stack already used by the crate's dedicated large-stack tests.
|
||||
@@ -60,9 +65,13 @@ command = ['sh', '-c', 'echo RUST_MIN_STACK=33554432 >> "$NEXTEST_ENV"']
|
||||
|
||||
# --- default profile (local): serialize the flaky groups, never retry --------
|
||||
[[profile.default.scripts]]
|
||||
filter = 'package(rustfs-ecstore) & test(/^(bucket::lifecycle::bucket_lifecycle_ops::tests::manual_transition_worker_result_recovery_marks_unknown_for_corrupt_marker|services::rebalance::entry::tests::real_rebalance_run_fence_loss_blocks_multipart_publication|store::init::tests::(decommission_entry_(allows_free_version_consumed_before_source_lock|rejects_subquorum_free_version_conflict_and_retains_source|skips_cleanup_only_marker_when_free_version_is_present)|prepared_tier_delete_recovery_(checks_later_pool_then_commits_after_source_removal|finds_directory_source_on_encoded_set|retains_journal_on_source_metadata_error)|tier_mutation_peer_handler_applies_prepare_commit_and_abort_idempotently|transition_response_loss_persists_unknown_outcome_for_provider_recovery|transition_transaction_recovery_(drops_record_after_confirmed_local_commit|keeps_cleanup_pending_local_commit)))$/)'
|
||||
filter = 'package(rustfs-ecstore) & test(/^(bucket::lifecycle::bucket_lifecycle_ops::tests::manual_transition_worker_result_recovery_marks_unknown_for_corrupt_marker|services::rebalance::entry::tests::real_rebalance_run_fence_loss_blocks_multipart_publication|store::init::tests::(batch_transitioned_delete_uses_free_version_per_item|decommission_entry_(allows_free_version_consumed_before_source_lock|rejects_subquorum_free_version_conflict_and_retains_source|skips_cleanup_only_marker_when_free_version_is_present)|dispatched_tier_delete_recovery_(checks_later_pool_then_commits_after_source_removal|finds_directory_source_on_encoded_set|retains_journal_on_source_metadata_error)|force_tier_remove_blocks_on_physical_free_version_hidden_by_other_pool|legacy_unknown_transition_delete_falls_back_for_single_batch_and_blocks_prefix|multi_pool_(recursive_prefix_rejects_legacy_or_hidden_merge_loser_before_delete|same_remote_tuple_(batch|single)_delete_waits_for_all_sources|same_tuple_recursive_prefix_uses_one_journal_owner|transitioned_delete_persists_one_free_version_per_remote_tuple)|recursive_prefix_partial_(pool|set)_failure_keeps_prepared_cleanup_owners|restored_transitioned_delete_uses_free_version_as_cleanup_owner|stable_transitioned_recursive_prefix_delete_uses_journal_owners|suspended_null_transition_delete_uses_free_version_as_sole_owner|tier_mutation_peer_handler_applies_prepare_commit_and_abort_idempotently|transition_response_loss_persists_unknown_outcome_for_provider_recovery|transition_transaction_recovery_(drops_record_after_confirmed_local_commit|keeps_cleanup_pending_local_commit)|transitioned_delete_(free_version_replays_after_store_restart|local_quorum_failure_rolls_back_without_cleanup_owner|uses_free_version_as_cleanup_owner)|versioned_delete_marker_keeps_transitioned_source_and_remote_object|versioned_explicit_transition_delete_preserves_other_version_then_allows_bucket_delete))$/)'
|
||||
setup = 'ecstore-large-stack'
|
||||
|
||||
[[profile.default.scripts]]
|
||||
filter = 'package(rustfs-ecstore) | package(rustfs-s3select-api) | package(rustfs-scanner) | (package(rustfs) & test(/^(app::multipart_usecase::tests::concurrent_completions_share_durable_bucket_quota_reservations|app::object::delete::tests::compressed_delete_requests_update_observed_usage_without_releasing_quota_floor|storage::access::tests::(delete_object_access_captures_authorized_bucket_incarnation|copy_operations_reject_recreated_source_bucket_after_authorization|request_slot_keeps_bucket_policy_bound_to_its_store))$/))'
|
||||
setup = 'ecstore-base-stack'
|
||||
|
||||
[[profile.default.scripts]]
|
||||
filter = 'binary(lifecycle_integration_test) | (package(rustfs) & test(/^app::lifecycle_transition_api_test::/))'
|
||||
setup = 'lifecycle-large-stack'
|
||||
@@ -80,6 +89,29 @@ test-group = 'ecstore-serial-flaky'
|
||||
filter = 'package(rustfs-ecstore) & test(/^set_disk::ops::multipart::tests::crash_consistency::/)'
|
||||
test-group = 'ecstore-serial-flaky'
|
||||
|
||||
# Serialize the heal result-report tests. Every test in the module builds a
|
||||
# real-disk (TempDir-backed) hermetic erasure set and drives MiB-scale writes
|
||||
# plus deep-scan heal — the same load-sensitive cross-disk IO shape as the
|
||||
# crash_consistency scenarios above. Under a heavily parallel run a single
|
||||
# disk's IO can fail while write quorum still holds, which flips per-disk
|
||||
# readback and aggregate-outcome assertions nondeterministically (different
|
||||
# tests each round; all pass standalone). Preventive serialization only, no
|
||||
# retries. The matching ci-profile override is after [profile.ci].
|
||||
[[profile.default.overrides]]
|
||||
filter = 'package(rustfs-ecstore) & test(/^set_disk::ops::heal::heal_result_report_tests::/)'
|
||||
test-group = 'ecstore-serial-flaky'
|
||||
|
||||
# Serialize the metadata-cache generation-retirement pair. Both carry
|
||||
# #[serial(metadata_cache_invalidation_probe)] — a no-op across nextest's
|
||||
# process boundary — and assert get_object_metadata_cache generation
|
||||
# semantics on a 4-disk hermetic set, the same load-sensitive shape that
|
||||
# forced the transition matrix tests into this group. Preventive
|
||||
# serialization only, no retries. The matching ci-profile override is after
|
||||
# [profile.ci].
|
||||
[[profile.default.overrides]]
|
||||
filter = 'package(rustfs-ecstore) & test(retires_cached_snapshot)'
|
||||
test-group = 'ecstore-serial-flaky'
|
||||
|
||||
# The production-handler relocation regression builds an isolated 8-disk,
|
||||
# 2-pool store and commits a 72 MiB multipart object. Keep that cross-disk IO
|
||||
# from overlapping the ecstore commit fixtures above.
|
||||
@@ -100,12 +132,29 @@ test-group = 'embedded-test-ports'
|
||||
filter = 'package(rustfs-ecstore) & test(manual_transition_page_checkpoint_persists_durable_job_progress)'
|
||||
test-group = 'ecstore-serial-flaky'
|
||||
|
||||
# Serialize the transition matrix tests. They build a 4-disk hermetic erasure
|
||||
# set, populate the get_object_metadata_cache, and assert generation lifecycle
|
||||
# semantics. serial_test's #[serial] has no effect across nextest's process
|
||||
# boundary, so concurrent execution races the shared metadata-cache generation
|
||||
# counter and causes spurious "metadata read should publish the generation"
|
||||
# panics. Preventive serialization, no retries.
|
||||
[[profile.default.overrides]]
|
||||
filter = 'package(rustfs-ecstore) & test(set_disk::transition_matrix_tests::)'
|
||||
test-group = 'ecstore-serial-flaky'
|
||||
|
||||
# The durable ILM decommission regressions build isolated multi-pool stores and
|
||||
# deliberately take source or target disks offline while checking fencing.
|
||||
[[profile.default.overrides]]
|
||||
filter = 'package(rustfs-ecstore) & (test(decommission_migrates_and_verifies_registered_durable_ilm_records) | test(decommission_durable_ilm_target_read_error_is_not_masked_by_peer_success) | test(decommission_durable_ilm_terminal_receipt_recovers_failed_source_cleanup) | test(decommission_durable_ilm_receipt_pagination_fails_closed_on_second_page) | test(decommission_durable_ilm_recovery_keeps_multiple_active_sources))'
|
||||
test-group = 'ecstore-serial-flaky'
|
||||
|
||||
# Decommission entry and marker/barrier tests share process-wide fault hooks and
|
||||
# deterministic commit barriers. Keep the whole init decommission family in one
|
||||
# nextest group; serial_test alone cannot isolate separate test processes.
|
||||
[[profile.default.overrides]]
|
||||
filter = 'package(rustfs-ecstore) & test(/^store::init::tests::(decommission_|suspended_.*decommission)$/)'
|
||||
test-group = 'ecstore-serial-flaky'
|
||||
|
||||
# Serialize the bucket-incarnation / lifecycle-fence tests. They drive
|
||||
# init_bucket_metadata_sys and bucket_metadata_sys_of, i.e. process-global
|
||||
# OnceLock state that serial_test's #[serial] cannot protect across nextest's
|
||||
@@ -157,9 +206,13 @@ fail-fast = false
|
||||
path = "junit.xml"
|
||||
|
||||
[[profile.ci.scripts]]
|
||||
filter = 'package(rustfs-ecstore) & test(/^(bucket::lifecycle::bucket_lifecycle_ops::tests::manual_transition_worker_result_recovery_marks_unknown_for_corrupt_marker|services::rebalance::entry::tests::real_rebalance_run_fence_loss_blocks_multipart_publication|store::init::tests::(decommission_entry_(allows_free_version_consumed_before_source_lock|rejects_subquorum_free_version_conflict_and_retains_source|skips_cleanup_only_marker_when_free_version_is_present)|prepared_tier_delete_recovery_(checks_later_pool_then_commits_after_source_removal|finds_directory_source_on_encoded_set|retains_journal_on_source_metadata_error)|tier_mutation_peer_handler_applies_prepare_commit_and_abort_idempotently|transition_response_loss_persists_unknown_outcome_for_provider_recovery|transition_transaction_recovery_(drops_record_after_confirmed_local_commit|keeps_cleanup_pending_local_commit)))$/)'
|
||||
filter = 'package(rustfs-ecstore) & test(/^(bucket::lifecycle::bucket_lifecycle_ops::tests::manual_transition_worker_result_recovery_marks_unknown_for_corrupt_marker|services::rebalance::entry::tests::real_rebalance_run_fence_loss_blocks_multipart_publication|store::init::tests::(batch_transitioned_delete_uses_free_version_per_item|decommission_entry_(allows_free_version_consumed_before_source_lock|rejects_subquorum_free_version_conflict_and_retains_source|skips_cleanup_only_marker_when_free_version_is_present)|dispatched_tier_delete_recovery_(checks_later_pool_then_commits_after_source_removal|finds_directory_source_on_encoded_set|retains_journal_on_source_metadata_error)|force_tier_remove_blocks_on_physical_free_version_hidden_by_other_pool|legacy_unknown_transition_delete_falls_back_for_single_batch_and_blocks_prefix|multi_pool_(recursive_prefix_rejects_legacy_or_hidden_merge_loser_before_delete|same_remote_tuple_(batch|single)_delete_waits_for_all_sources|same_tuple_recursive_prefix_uses_one_journal_owner|transitioned_delete_persists_one_free_version_per_remote_tuple)|recursive_prefix_partial_(pool|set)_failure_keeps_prepared_cleanup_owners|restored_transitioned_delete_uses_free_version_as_cleanup_owner|stable_transitioned_recursive_prefix_delete_uses_journal_owners|suspended_null_transition_delete_uses_free_version_as_sole_owner|tier_mutation_peer_handler_applies_prepare_commit_and_abort_idempotently|transition_response_loss_persists_unknown_outcome_for_provider_recovery|transition_transaction_recovery_(drops_record_after_confirmed_local_commit|keeps_cleanup_pending_local_commit)|transitioned_delete_(free_version_replays_after_store_restart|local_quorum_failure_rolls_back_without_cleanup_owner|uses_free_version_as_cleanup_owner)|versioned_delete_marker_keeps_transitioned_source_and_remote_object|versioned_explicit_transition_delete_preserves_other_version_then_allows_bucket_delete))$/)'
|
||||
setup = 'ecstore-large-stack'
|
||||
|
||||
[[profile.ci.scripts]]
|
||||
filter = 'package(rustfs-ecstore) | package(rustfs-s3select-api) | package(rustfs-scanner) | (package(rustfs) & test(/^(app::multipart_usecase::tests::concurrent_completions_share_durable_bucket_quota_reservations|app::object::delete::tests::compressed_delete_requests_update_observed_usage_without_releasing_quota_floor|storage::access::tests::(delete_object_access_captures_authorized_bucket_incarnation|copy_operations_reject_recreated_source_bucket_after_authorization|request_slot_keeps_bucket_policy_bound_to_its_store))$/))'
|
||||
setup = 'ecstore-base-stack'
|
||||
|
||||
[[profile.ci.scripts]]
|
||||
filter = 'binary(lifecycle_integration_test) | (package(rustfs) & test(/^app::lifecycle_transition_api_test::/))'
|
||||
setup = 'lifecycle-large-stack'
|
||||
@@ -220,6 +273,20 @@ test-group = 'e2e-reliability'
|
||||
filter = 'package(rustfs-ecstore) & test(/^set_disk::ops::multipart::tests::crash_consistency::/)'
|
||||
test-group = 'ecstore-serial-flaky'
|
||||
|
||||
# Serialize the heal result-report tests under the ci profile too (see the
|
||||
# matching default-profile override near the top). Not a quarantine: no
|
||||
# retries, just serialized real-disk heal IO.
|
||||
[[profile.ci.overrides]]
|
||||
filter = 'package(rustfs-ecstore) & test(/^set_disk::ops::heal::heal_result_report_tests::/)'
|
||||
test-group = 'ecstore-serial-flaky'
|
||||
|
||||
# Serialize the metadata-cache generation-retirement pair under the ci
|
||||
# profile too (see the matching default-profile override near the top). Not a
|
||||
# quarantine: no retries.
|
||||
[[profile.ci.overrides]]
|
||||
filter = 'package(rustfs-ecstore) & test(retires_cached_snapshot)'
|
||||
test-group = 'ecstore-serial-flaky'
|
||||
|
||||
# Match the default-profile embedded test isolation without quarantining or
|
||||
# retrying failures in CI.
|
||||
[[profile.ci.overrides]]
|
||||
@@ -232,10 +299,20 @@ test-group = 'embedded-test-ports'
|
||||
filter = 'package(rustfs-ecstore) & test(manual_transition_page_checkpoint_persists_durable_job_progress)'
|
||||
test-group = 'ecstore-serial-flaky'
|
||||
|
||||
# Serialize the transition matrix tests under the ci profile too (see the
|
||||
# matching default-profile override near the top). No retries.
|
||||
[[profile.ci.overrides]]
|
||||
filter = 'package(rustfs-ecstore) & test(set_disk::transition_matrix_tests::)'
|
||||
test-group = 'ecstore-serial-flaky'
|
||||
|
||||
[[profile.ci.overrides]]
|
||||
filter = 'package(rustfs-ecstore) & (test(decommission_migrates_and_verifies_registered_durable_ilm_records) | test(decommission_durable_ilm_target_read_error_is_not_masked_by_peer_success) | test(decommission_durable_ilm_terminal_receipt_recovers_failed_source_cleanup) | test(decommission_durable_ilm_receipt_pagination_fails_closed_on_second_page) | test(decommission_durable_ilm_recovery_keeps_multiple_active_sources))'
|
||||
test-group = 'ecstore-serial-flaky'
|
||||
|
||||
[[profile.ci.overrides]]
|
||||
filter = 'package(rustfs-ecstore) & test(/^store::init::tests::(decommission_|suspended_.*decommission)$/)'
|
||||
test-group = 'ecstore-serial-flaky'
|
||||
|
||||
# Serialize the bucket-incarnation / lifecycle-fence tests under the ci profile
|
||||
# too (see the matching default-profile override near the top). No retries.
|
||||
[[profile.ci.overrides]]
|
||||
@@ -400,7 +477,7 @@ path = "junit.xml"
|
||||
[profile.e2e-nightly]
|
||||
default-filter = """
|
||||
package(e2e_test)
|
||||
& test(/^(admin_timeout_regression_test|cluster_concurrency_test|cluster_multidrive_pool_test|heal_erasure_disk_rebuild_test|namespace_lock_quorum_test|object_lambda_test|stale_multipart_cleanup_cluster_test)::/)
|
||||
& test(/^(admin_timeout_regression_test|cluster_concurrency_test|cluster_multidrive_pool_test|degraded_listing_availability_test|heal_erasure_disk_rebuild_test|namespace_lock_quorum_test|object_lambda_test|stale_multipart_cleanup_cluster_test)::/)
|
||||
"""
|
||||
fail-fast = false
|
||||
|
||||
@@ -452,23 +529,12 @@ path = "junit.xml"
|
||||
# parallel-safe — the same property e2e-smoke relies on. The exceptions are the
|
||||
# 4-disk reliability / degraded-read fault-injection tests and the fixed-port
|
||||
# Vault tests, both serialized below.
|
||||
# KNOWN-FAILURE EXCLUSIONS (characterization run 29381309848, 2026-07-15:
|
||||
# 341 ran / 32 failed on the suites' first automated run ever). Deterministic
|
||||
# product failures cannot be quarantined away with retries, so each family is
|
||||
# excluded here with its tracking issue, under the same discipline as the
|
||||
# ci-profile quarantine (docs/testing/README.md): every entry MUST cite one
|
||||
# OPEN issue, and the fixing PR MUST delete the exclusion. The passing
|
||||
# negative-path siblings of each family stay in as regression guards.
|
||||
# * rustfs#4843 — over-limit archive entry paths hard-reject the whole
|
||||
# archive even under ignore-errors semantics.
|
||||
[profile.e2e-full]
|
||||
default-filter = """
|
||||
package(e2e_test)
|
||||
& !test(/^protocols::/)
|
||||
& !test(/^(admin_timeout_regression_test|cluster_concurrency_test|cluster_multidrive_pool_test|heal_erasure_disk_rebuild_test|namespace_lock_quorum_test|object_lambda_test|stale_multipart_cleanup_cluster_test)::/)
|
||||
& !test(/^(admin_timeout_regression_test|cluster_concurrency_test|cluster_multidrive_pool_test|degraded_listing_availability_test|heal_erasure_disk_rebuild_test|namespace_lock_quorum_test|object_lambda_test|stale_multipart_cleanup_cluster_test)::/)
|
||||
& !test(/^replication_extension_test::/)
|
||||
& !test(/^multipart_auth_test::test_signed_put_object_extract_skips_invalid_entry_when_ignore_errors_enabled$/)
|
||||
& !test(/^snowball_auto_extract_test::tests::snowball_auto_extract_(ignores_invalid_entries_when_requested|supports_standard_headers_with_combined_extract_options)$/)
|
||||
"""
|
||||
fail-fast = false
|
||||
|
||||
|
||||
@@ -24,8 +24,11 @@ on:
|
||||
- '.github/actions/**'
|
||||
- '.github/workflows/**'
|
||||
- 'scripts/release/create_or_update_release.sh'
|
||||
- 'scripts/release/package_versions.sh'
|
||||
- 'scripts/test_package_versions.sh'
|
||||
- 'scripts/security/check_performance_ab_workflow.sh'
|
||||
- 'scripts/security/check_preview_release_workflow.sh'
|
||||
- 'scripts/security/check_tier_artifact_workflow.sh'
|
||||
- 'scripts/security/check_workflow_pins.sh'
|
||||
pull_request:
|
||||
types: [ opened, synchronize, reopened, closed ]
|
||||
@@ -37,8 +40,11 @@ on:
|
||||
- '.github/actions/**'
|
||||
- '.github/workflows/**'
|
||||
- 'scripts/release/create_or_update_release.sh'
|
||||
- 'scripts/release/package_versions.sh'
|
||||
- 'scripts/test_package_versions.sh'
|
||||
- 'scripts/security/check_performance_ab_workflow.sh'
|
||||
- 'scripts/security/check_preview_release_workflow.sh'
|
||||
- 'scripts/security/check_tier_artifact_workflow.sh'
|
||||
- 'scripts/security/check_workflow_pins.sh'
|
||||
schedule:
|
||||
# Daily, not weekly. This schedule exists to catch RustSec advisories
|
||||
@@ -146,6 +152,12 @@ jobs:
|
||||
- name: Check performance A/B workflow trust boundary
|
||||
run: ./scripts/security/check_performance_ab_workflow.sh
|
||||
|
||||
- name: Check tier evidence workflow isolation
|
||||
run: ./scripts/security/check_tier_artifact_workflow.sh
|
||||
|
||||
- name: Check package version contract
|
||||
run: ./scripts/test_package_versions.sh
|
||||
|
||||
dependency-review:
|
||||
name: Dependency Review
|
||||
runs-on: ubuntu-latest
|
||||
|
||||
@@ -244,7 +244,7 @@ jobs:
|
||||
needs: [ build-check, prepare-platform-matrix ]
|
||||
if: needs.build-check.outputs.should_build == 'true' && needs.prepare-platform-matrix.result == 'success'
|
||||
runs-on: ${{ matrix.os }}
|
||||
timeout-minutes: 150
|
||||
timeout-minutes: 180
|
||||
env:
|
||||
FORCE_JAVASCRIPT_ACTIONS_TO_NODE24: "true"
|
||||
# Release binaries ship without dial9 telemetry and therefore do not need
|
||||
@@ -408,9 +408,9 @@ jobs:
|
||||
|
||||
if [[ "${{ matrix.cross }}" == "true" ]]; then
|
||||
# All cross targets in the matrix are Linux; zigbuild handles them.
|
||||
cargo zigbuild --release --target ${{ matrix.target }} -p rustfs --bins
|
||||
cargo zigbuild --release --target ${{ matrix.target }} -p rustfs --bin rustfs
|
||||
else
|
||||
cargo build --release --target ${{ matrix.target }} -p rustfs --bins
|
||||
cargo build --release --target ${{ matrix.target }} -p rustfs --bin rustfs
|
||||
fi
|
||||
|
||||
- name: Create release package
|
||||
|
||||
@@ -49,8 +49,20 @@ env:
|
||||
UPGRADE_SOURCE_SHA256: 7c789386bf85278f865b8e0d359bf4edb84d5aa408cc3fa54a18c25ca74cd6e7
|
||||
|
||||
jobs:
|
||||
direct-upgrade:
|
||||
name: Direct upgrade from rc.2
|
||||
upgrade:
|
||||
name: ${{ matrix.name }}
|
||||
strategy:
|
||||
fail-fast: false
|
||||
matrix:
|
||||
include:
|
||||
- name: Direct upgrade from rc.2
|
||||
cache_key: e2e-direct-upgrade
|
||||
test: direct_upgrade_from_rc2_preserves_object_contracts
|
||||
artifact: direct-upgrade
|
||||
- name: Mixed-version rolling upgrade from rc.2
|
||||
cache_key: e2e-mixed-version-upgrade
|
||||
test: rolling_upgrade_from_rc2_preserves_mixed_version_contracts
|
||||
artifact: mixed-version-upgrade
|
||||
runs-on: ubuntu-latest
|
||||
timeout-minutes: 60
|
||||
env:
|
||||
@@ -64,7 +76,7 @@ jobs:
|
||||
- name: Setup Rust environment
|
||||
uses: ./.github/actions/setup
|
||||
with:
|
||||
cache-shared-key: e2e-direct-upgrade
|
||||
cache-shared-key: ${{ matrix.cache_key }}
|
||||
cache-save-if: ${{ github.ref == 'refs/heads/main' }}
|
||||
install-build-packaging-tools: "false"
|
||||
|
||||
@@ -89,17 +101,17 @@ jobs:
|
||||
cargo build --locked -p rustfs --bin rustfs
|
||||
: > target/debug/rustfs.features
|
||||
|
||||
- name: Run direct-upgrade compatibility test
|
||||
- name: Run upgrade compatibility test
|
||||
run: |
|
||||
cargo test --locked -p e2e_test \
|
||||
upgrade_compatibility_test::direct_upgrade_from_rc2_preserves_object_contracts \
|
||||
"upgrade_compatibility_test::${{ matrix.test }}" \
|
||||
-- --ignored --exact --nocapture
|
||||
|
||||
- name: Upload server logs
|
||||
if: always()
|
||||
uses: actions/upload-artifact@b7c566a772e6b6bfb58ed0dc250532a479d7789f # v6
|
||||
with:
|
||||
name: direct-upgrade-server-logs-${{ github.run_number }}
|
||||
name: ${{ matrix.artifact }}-server-logs-${{ github.run_number }}
|
||||
path: ${{ runner.temp }}/rustfs-upgrade-logs
|
||||
if-no-files-found: warn
|
||||
retention-days: 14
|
||||
|
||||
+160
-90
@@ -21,10 +21,10 @@
|
||||
# - workflow_run: automatically package after "Build and Release" completes
|
||||
# for a release tag (the mac/windows/linux binaries are already uploaded
|
||||
# to the GitHub release before packaging starts)
|
||||
# - workflow_dispatch: manual fallback (backfill / re-run) with optional tag/run_id
|
||||
# - workflow_dispatch: manual fallback with a release tag and/or exact build run ID
|
||||
#
|
||||
# Flow:
|
||||
# 1. Resolve the triggering Build workflow run for the release tag
|
||||
# 1. Resolve and validate the selected Build workflow run and source identity
|
||||
# 2. Download Linux binaries (x86_64-gnu, aarch64-gnu) from build artifacts
|
||||
# 3. Build DEB packages for amd64 and arm64
|
||||
# 4. Build RPM packages for x86_64 and aarch64
|
||||
@@ -51,7 +51,7 @@ on:
|
||||
required: false
|
||||
type: string
|
||||
build_run_id:
|
||||
description: "Build workflow run ID (overrides tag lookup)"
|
||||
description: "Build workflow run ID (when combined with tag, both must identify the same release commit)"
|
||||
required: false
|
||||
type: string
|
||||
|
||||
@@ -82,6 +82,9 @@ jobs:
|
||||
version: ${{ steps.resolve.outputs.version }}
|
||||
build_type: ${{ steps.resolve.outputs.build_type }}
|
||||
build_run_id: ${{ steps.resolve.outputs.build_run_id }}
|
||||
build_run_number: ${{ steps.resolve.outputs.build_run_number }}
|
||||
head_sha: ${{ steps.resolve.outputs.head_sha }}
|
||||
dev_sequence: ${{ steps.resolve.outputs.dev_sequence }}
|
||||
tag: ${{ steps.resolve.outputs.tag }}
|
||||
steps:
|
||||
- name: Resolve build run
|
||||
@@ -89,90 +92,129 @@ jobs:
|
||||
shell: bash
|
||||
env:
|
||||
GH_TOKEN: ${{ github.token }}
|
||||
EVENT_NAME: ${{ github.event_name }}
|
||||
REPOSITORY: ${{ github.repository }}
|
||||
INPUT_TAG: ${{ github.event.inputs.tag }}
|
||||
INPUT_RUN_ID: ${{ github.event.inputs.build_run_id }}
|
||||
run: |
|
||||
set -euo pipefail
|
||||
|
||||
# Determine tag
|
||||
if [[ "${{ github.event_name }}" == "workflow_run" ]]; then
|
||||
TAG="${HEAD_BRANCH}"
|
||||
elif [[ -n "$INPUT_TAG" ]]; then
|
||||
TAG="$INPUT_TAG"
|
||||
fail() {
|
||||
echo "❌ $1" >&2
|
||||
exit 1
|
||||
}
|
||||
|
||||
TAG=""
|
||||
BUILD_RUN_ID=""
|
||||
case "$EVENT_NAME" in
|
||||
workflow_run)
|
||||
TAG="$HEAD_BRANCH"
|
||||
BUILD_RUN_ID="$WORKFLOW_RUN_ID"
|
||||
;;
|
||||
workflow_dispatch)
|
||||
TAG="$INPUT_TAG"
|
||||
BUILD_RUN_ID="$INPUT_RUN_ID"
|
||||
;;
|
||||
*) fail "unsupported event: $EVENT_NAME" ;;
|
||||
esac
|
||||
|
||||
# Validate and classify tags before using them in API paths or logs.
|
||||
semver_core='(0|[1-9][0-9]*)\.(0|[1-9][0-9]*)\.(0|[1-9][0-9]*)'
|
||||
prerelease_id='(alpha|beta|rc)\.(0|[1-9][0-9]*)'
|
||||
if [[ -n "$TAG" ]]; then
|
||||
if [[ "$TAG" =~ ^${semver_core}-${prerelease_id}-preview\.(0|[1-9][0-9]*)$ ]]; then
|
||||
BUILD_TYPE=preview
|
||||
elif [[ "$TAG" =~ ^${semver_core}-${prerelease_id}$ ]]; then
|
||||
BUILD_TYPE=prerelease
|
||||
elif [[ "$TAG" =~ ^${semver_core}$ ]]; then
|
||||
BUILD_TYPE=release
|
||||
else
|
||||
fail "tag is not a supported strict package version"
|
||||
fi
|
||||
else
|
||||
TAG=""
|
||||
BUILD_TYPE=development
|
||||
fi
|
||||
|
||||
echo "Tag: ${TAG:-<none>}"
|
||||
|
||||
# Determine build run ID
|
||||
BUILD_RUN_ID=""
|
||||
|
||||
if [[ -n "$INPUT_RUN_ID" ]]; then
|
||||
# Explicit run ID takes priority
|
||||
BUILD_RUN_ID="$INPUT_RUN_ID"
|
||||
echo "Using explicit build run ID: $BUILD_RUN_ID"
|
||||
|
||||
elif [[ "${{ github.event_name }}" == "workflow_run" ]]; then
|
||||
# Use the Build and Release run that triggered this workflow
|
||||
BUILD_RUN_ID="${WORKFLOW_RUN_ID}"
|
||||
echo "Using triggering workflow run: $BUILD_RUN_ID"
|
||||
|
||||
if [[ -n "$BUILD_RUN_ID" ]]; then
|
||||
[[ "$BUILD_RUN_ID" =~ ^[1-9][0-9]*$ ]] || fail "build run ID must be a positive decimal integer"
|
||||
echo "Using selected build run: $BUILD_RUN_ID"
|
||||
elif [[ -n "$TAG" ]]; then
|
||||
# Find the build run that produced this tag
|
||||
echo "Looking for build run for tag: $TAG"
|
||||
BUILD_RUN_ID=$(gh api \
|
||||
"repos/${{ github.repository }}/actions/workflows/build.yml/runs?branch=${TAG}&status=success&per_page=1" \
|
||||
--jq '.workflow_runs[0].id' 2>/dev/null || echo "")
|
||||
BUILD_RUN_ID=$(gh api --method GET \
|
||||
"repos/${REPOSITORY}/actions/workflows/build.yml/runs" \
|
||||
-f branch="$TAG" -f status=success -F per_page=1 \
|
||||
--jq '.workflow_runs[0].id // empty' 2>/dev/null || true)
|
||||
|
||||
if [[ -z "$BUILD_RUN_ID" || "$BUILD_RUN_ID" == "null" ]]; then
|
||||
# Tag might not be a branch; try event=push with head_branch matching
|
||||
BUILD_RUN_ID=$(gh api \
|
||||
"repos/${{ github.repository }}/actions/workflows/build.yml/runs?event=push&status=success&per_page=100" \
|
||||
--jq ".workflow_runs[] | select(.head_branch == \"$TAG\") | .id" 2>/dev/null | head -1 || echo "")
|
||||
fi
|
||||
|
||||
if [[ -z "$BUILD_RUN_ID" || "$BUILD_RUN_ID" == "null" ]]; then
|
||||
echo "❌ No successful build run found for tag: $TAG"
|
||||
exit 1
|
||||
if [[ -z "$BUILD_RUN_ID" ]]; then
|
||||
BUILD_RUN_ID=$(gh api --method GET \
|
||||
"repos/${REPOSITORY}/actions/workflows/build.yml/runs" \
|
||||
-f event=push -f status=success -F per_page=100 2>/dev/null |
|
||||
jq -r --arg tag "$TAG" \
|
||||
'[.workflow_runs[] | select(.head_branch == $tag)][0].id // empty' || true)
|
||||
fi
|
||||
[[ "$BUILD_RUN_ID" =~ ^[1-9][0-9]*$ ]] || fail "no successful build run found for tag"
|
||||
echo "Found build run: $BUILD_RUN_ID"
|
||||
|
||||
else
|
||||
# No tag — latest successful main build
|
||||
echo "No tag specified, looking for latest main build"
|
||||
BUILD_RUN_ID=$(gh api \
|
||||
"repos/${{ github.repository }}/actions/workflows/build.yml/runs?branch=main&status=success&per_page=1" \
|
||||
--jq '.workflow_runs[0].id' 2>/dev/null || echo "")
|
||||
|
||||
if [[ -z "$BUILD_RUN_ID" || "$BUILD_RUN_ID" == "null" ]]; then
|
||||
echo "❌ No successful main build found"
|
||||
exit 1
|
||||
fi
|
||||
BUILD_RUN_ID=$(gh api --method GET \
|
||||
"repos/${REPOSITORY}/actions/workflows/build.yml/runs" \
|
||||
-f branch=main -f status=success -F per_page=1 \
|
||||
--jq '.workflow_runs[0].id // empty' 2>/dev/null || true)
|
||||
[[ "$BUILD_RUN_ID" =~ ^[1-9][0-9]*$ ]] || fail "no successful main build found"
|
||||
echo "Latest main build: $BUILD_RUN_ID"
|
||||
fi
|
||||
|
||||
# Determine version and build type
|
||||
# Fetch once and use the same immutable run metadata for identity,
|
||||
# ordering, workflow provenance, and release-channel validation.
|
||||
RUN_JSON=$(gh api "repos/${REPOSITORY}/actions/runs/${BUILD_RUN_ID}") ||
|
||||
fail "cannot read selected build run"
|
||||
RUN_ID=$(jq -r '.id // empty' <<<"$RUN_JSON")
|
||||
RUN_NUMBER=$(jq -r '.run_number // empty' <<<"$RUN_JSON")
|
||||
RUN_STATUS=$(jq -r '.status // empty' <<<"$RUN_JSON")
|
||||
RUN_CONCLUSION=$(jq -r '.conclusion // empty' <<<"$RUN_JSON")
|
||||
RUN_PATH=$(jq -r '.path // empty' <<<"$RUN_JSON")
|
||||
HEAD_SHA=$(jq -r '.head_sha // empty' <<<"$RUN_JSON")
|
||||
RUN_HEAD_BRANCH=$(jq -r '.head_branch // empty' <<<"$RUN_JSON")
|
||||
|
||||
[[ "$RUN_ID" == "$BUILD_RUN_ID" ]] || fail "run metadata ID mismatch"
|
||||
[[ "$RUN_NUMBER" =~ ^[1-9][0-9]*$ ]] || fail "build run number must be a positive decimal integer"
|
||||
[[ "$RUN_STATUS" == completed && "$RUN_CONCLUSION" == success ]] || fail "selected build run is not successful"
|
||||
[[ "$RUN_PATH" == .github/workflows/build.yml ]] || fail "selected run is not Build and Release"
|
||||
[[ "$HEAD_SHA" =~ ^[0-9a-f]{40}$ ]] || fail "selected build run has an invalid head SHA"
|
||||
[[ "$RUN_HEAD_BRANCH" != *$'\n'* && -n "$RUN_HEAD_BRANCH" ]] || fail "selected build run has an invalid head branch"
|
||||
|
||||
if [[ -n "$TAG" ]]; then
|
||||
[[ "$RUN_HEAD_BRANCH" == "$TAG" ]] || fail "tag and build run head branch do not match"
|
||||
|
||||
TAG_REF_JSON=$(gh api "repos/${REPOSITORY}/git/ref/tags/${TAG}") ||
|
||||
fail "cannot resolve release tag ref"
|
||||
TAG_OBJECT_TYPE=$(jq -r '.object.type // empty' <<<"$TAG_REF_JSON")
|
||||
TAG_OBJECT_SHA=$(jq -r '.object.sha // empty' <<<"$TAG_REF_JSON")
|
||||
depth=0
|
||||
while [[ "$TAG_OBJECT_TYPE" == tag && $depth -lt 5 ]]; do
|
||||
TAG_OBJECT_JSON=$(gh api "repos/${REPOSITORY}/git/tags/${TAG_OBJECT_SHA}") ||
|
||||
fail "cannot peel annotated release tag"
|
||||
TAG_OBJECT_TYPE=$(jq -r '.object.type // empty' <<<"$TAG_OBJECT_JSON")
|
||||
TAG_OBJECT_SHA=$(jq -r '.object.sha // empty' <<<"$TAG_OBJECT_JSON")
|
||||
depth=$((depth + 1))
|
||||
done
|
||||
[[ "$TAG_OBJECT_TYPE" == commit && "$TAG_OBJECT_SHA" =~ ^[0-9a-f]{40}$ ]] ||
|
||||
fail "release tag does not resolve to a commit"
|
||||
[[ "$TAG_OBJECT_SHA" == "$HEAD_SHA" ]] || fail "release tag commit and build run head SHA do not match"
|
||||
VERSION="$TAG"
|
||||
if [[ "$TAG" == *"-preview"* ]]; then
|
||||
BUILD_TYPE="preview"
|
||||
elif [[ "$TAG" == *"alpha"* || "$TAG" == *"beta"* || "$TAG" == *"rc"* ]]; then
|
||||
BUILD_TYPE="prerelease"
|
||||
else
|
||||
BUILD_TYPE="release"
|
||||
fi
|
||||
DEV_SEQUENCE=""
|
||||
else
|
||||
SHORT_SHA=$(gh api "repos/${{ github.repository }}/actions/runs/${BUILD_RUN_ID}" \
|
||||
--jq '.head_sha' 2>/dev/null | head -c 7)
|
||||
VERSION="dev-${SHORT_SHA}"
|
||||
BUILD_TYPE="development"
|
||||
VERSION="dev-${HEAD_SHA}"
|
||||
DEV_SEQUENCE="$RUN_NUMBER"
|
||||
fi
|
||||
|
||||
{
|
||||
echo "version=$VERSION"
|
||||
echo "build_type=$BUILD_TYPE"
|
||||
echo "build_run_id=$BUILD_RUN_ID"
|
||||
echo "build_run_number=$RUN_NUMBER"
|
||||
echo "head_sha=$HEAD_SHA"
|
||||
echo "dev_sequence=$DEV_SEQUENCE"
|
||||
echo "tag=${TAG}"
|
||||
} >> "$GITHUB_OUTPUT"
|
||||
|
||||
@@ -180,6 +222,7 @@ jobs:
|
||||
echo " Version: $VERSION"
|
||||
echo " Build type: $BUILD_TYPE"
|
||||
echo " Build run ID: $BUILD_RUN_ID"
|
||||
echo " Build run number: $RUN_NUMBER"
|
||||
|
||||
# Build DEB and RPM packages for each architecture
|
||||
package:
|
||||
@@ -206,6 +249,22 @@ jobs:
|
||||
with:
|
||||
persist-credentials: false
|
||||
|
||||
- name: Normalize package metadata
|
||||
id: versions
|
||||
shell: bash
|
||||
env:
|
||||
BUILD_TYPE: ${{ needs.resolve.outputs.build_type }}
|
||||
SOURCE_VERSION: ${{ needs.resolve.outputs.version }}
|
||||
DEV_SEQUENCE: ${{ needs.resolve.outputs.dev_sequence }}
|
||||
DEB_ARCH: ${{ matrix.deb_arch }}
|
||||
RPM_ARCH: ${{ matrix.rpm_arch }}
|
||||
run: |
|
||||
set -euo pipefail
|
||||
|
||||
normalized=$(./scripts/release/package_versions.sh \
|
||||
"$BUILD_TYPE" "$SOURCE_VERSION" "$DEV_SEQUENCE" "$DEB_ARCH" "$RPM_ARCH")
|
||||
printf '%s\n' "$normalized" >> "$GITHUB_OUTPUT"
|
||||
|
||||
- name: Download binary artifact from build run
|
||||
uses: actions/download-artifact@37930b1c2abaa49bbe596cd826c3c89aef350131 # v7
|
||||
with:
|
||||
@@ -245,18 +304,16 @@ jobs:
|
||||
- name: Build DEB package
|
||||
id: deb
|
||||
shell: bash
|
||||
env:
|
||||
DEB_VERSION: ${{ steps.versions.outputs.deb_version }}
|
||||
DEB_ARCH: ${{ matrix.deb_arch }}
|
||||
DEB_FILE: ${{ steps.versions.outputs.deb_file }}
|
||||
run: |
|
||||
set -euo pipefail
|
||||
|
||||
VERSION="${{ needs.resolve.outputs.version }}"
|
||||
DEB_ARCH="${{ matrix.deb_arch }}"
|
||||
# DEB version: replace - with ~ (1.0.0-beta.12 -> 1.0.0~beta.12)
|
||||
# Use a variable for ~ to prevent tilde expansion by bash
|
||||
TILDE='~'
|
||||
DEB_VERSION="${VERSION/-/$TILDE}"
|
||||
PKG_DIR="rustfs_${DEB_VERSION}_${DEB_ARCH}"
|
||||
PKG_DIR="${DEB_FILE%.deb}"
|
||||
|
||||
echo "Building DEB: ${PKG_DIR}.deb"
|
||||
echo "Building DEB: ${DEB_FILE}"
|
||||
|
||||
mkdir -p "${PKG_DIR}/DEBIAN"
|
||||
mkdir -p "${PKG_DIR}/usr/bin"
|
||||
@@ -333,9 +390,12 @@ jobs:
|
||||
cp LICENSE "${PKG_DIR}/usr/share/doc/rustfs/"
|
||||
cp README.md "${PKG_DIR}/usr/share/doc/rustfs/"
|
||||
|
||||
fakeroot dpkg-deb --build "${PKG_DIR}"
|
||||
fakeroot dpkg-deb --build "${PKG_DIR}" "$DEB_FILE"
|
||||
|
||||
DEB_FILE="${PKG_DIR}.deb"
|
||||
[[ $(dpkg-deb -f "$DEB_FILE" Package) == rustfs ]]
|
||||
[[ $(dpkg-deb -f "$DEB_FILE" Version) == "$DEB_VERSION" ]]
|
||||
[[ $(dpkg-deb -f "$DEB_FILE" Architecture) == "$DEB_ARCH" ]]
|
||||
dpkg-deb --fsys-tarfile "$DEB_FILE" | tar -tf - | grep -Fx './usr/bin/rustfs' >/dev/null
|
||||
stat --printf='%n %s bytes\n' "$DEB_FILE"
|
||||
echo "deb_file=$DEB_FILE" >> "$GITHUB_OUTPUT"
|
||||
echo "✅ DEB built: $DEB_FILE"
|
||||
@@ -343,16 +403,19 @@ jobs:
|
||||
- name: Build RPM package
|
||||
id: rpm
|
||||
shell: bash
|
||||
env:
|
||||
RPM_VERSION: ${{ steps.versions.outputs.rpm_version }}
|
||||
RPM_RELEASE: ${{ steps.versions.outputs.rpm_release }}
|
||||
RPM_ARCH: ${{ matrix.rpm_arch }}
|
||||
RPM_FILE: ${{ steps.versions.outputs.rpm_file }}
|
||||
run: |
|
||||
set -euo pipefail
|
||||
|
||||
VERSION="${{ needs.resolve.outputs.version }}"
|
||||
RPM_ARCH="${{ matrix.rpm_arch }}"
|
||||
|
||||
echo "Building RPM for ${RPM_ARCH}"
|
||||
|
||||
sudo apt-get update && sudo apt-get install -y ruby ruby-dev build-essential
|
||||
sudo apt-get update && sudo apt-get install -y ruby ruby-dev build-essential rpm
|
||||
sudo gem install fpm
|
||||
./scripts/test_package_versions.sh --require-package-managers
|
||||
|
||||
# Create config file for fpm (DEB build creates it in its package dir structure,
|
||||
# but fpm needs the file to exist before packaging)
|
||||
@@ -367,8 +430,10 @@ jobs:
|
||||
|
||||
fpm -s dir -t rpm \
|
||||
--name rustfs \
|
||||
--version "$VERSION" \
|
||||
--version "$RPM_VERSION" \
|
||||
--iteration "$RPM_RELEASE" \
|
||||
--architecture "$RPM_ARCH" \
|
||||
--package "$RPM_FILE" \
|
||||
--depends "glibc >= 2.31" \
|
||||
--maintainer "RustFS Team <support@rustfs.com>" \
|
||||
--description "High-performance distributed object storage" \
|
||||
@@ -410,13 +475,15 @@ jobs:
|
||||
LICENSE=/usr/share/doc/rustfs/LICENSE \
|
||||
README.md=/usr/share/doc/rustfs/README.md
|
||||
|
||||
RPM_FILE=$(find . -maxdepth 1 -type f -name 'rustfs-*.rpm' -print | head -1)
|
||||
RPM_FILE="${RPM_FILE#./}"
|
||||
if [[ -z "$RPM_FILE" ]]; then
|
||||
if [[ ! -f "$RPM_FILE" ]]; then
|
||||
echo "❌ RPM build failed"
|
||||
exit 1
|
||||
fi
|
||||
|
||||
RPM_METADATA=$(rpm -qp --qf '%{NAME}\n%{VERSION}\n%{RELEASE}\n%{ARCH}\n' "$RPM_FILE")
|
||||
EXPECTED_METADATA=$(printf 'rustfs\n%s\n%s\n%s' "$RPM_VERSION" "$RPM_RELEASE" "$RPM_ARCH")
|
||||
[[ "$RPM_METADATA" == "$EXPECTED_METADATA" ]]
|
||||
rpm -qpl "$RPM_FILE" | grep -Fx '/usr/bin/rustfs' >/dev/null
|
||||
stat --printf='%n %s bytes\n' "$RPM_FILE"
|
||||
echo "rpm_file=$RPM_FILE" >> "$GITHUB_OUTPUT"
|
||||
echo "✅ RPM built: $RPM_FILE"
|
||||
@@ -438,6 +505,9 @@ jobs:
|
||||
R2_ENDPOINT: ${{ secrets.R2_ENDPOINT }}
|
||||
R2_BUCKET: ${{ secrets.R2_BUCKET }}
|
||||
AWS_EC2_METADATA_DISABLED: true
|
||||
BUILD_TYPE: ${{ needs.resolve.outputs.build_type }}
|
||||
DEB_FILE: ${{ steps.deb.outputs.deb_file }}
|
||||
RPM_FILE: ${{ steps.rpm.outputs.rpm_file }}
|
||||
shell: bash
|
||||
run: |
|
||||
set -euo pipefail
|
||||
@@ -455,7 +525,6 @@ jobs:
|
||||
export AWS_SECRET_ACCESS_KEY="$R2_SECRET_ACCESS_KEY"
|
||||
export AWS_DEFAULT_REGION="auto"
|
||||
|
||||
BUILD_TYPE="${{ needs.resolve.outputs.build_type }}"
|
||||
if [[ "$BUILD_TYPE" == "development" ]]; then
|
||||
R2_PREFIX="artifacts/rustfs/packages/dev"
|
||||
else
|
||||
@@ -465,9 +534,6 @@ jobs:
|
||||
|
||||
echo "📤 Uploading to $R2_PATH"
|
||||
|
||||
DEB_FILE="${{ steps.deb.outputs.deb_file }}"
|
||||
RPM_FILE="${{ steps.rpm.outputs.rpm_file }}"
|
||||
|
||||
for f in "$DEB_FILE" "$RPM_FILE"; do
|
||||
if [[ -n "$f" && -f "$f" ]]; then
|
||||
echo "Uploading: $f"
|
||||
@@ -493,14 +559,13 @@ jobs:
|
||||
if: needs.resolve.outputs.tag != ''
|
||||
env:
|
||||
GH_TOKEN: ${{ github.token }}
|
||||
TAG: ${{ needs.resolve.outputs.tag }}
|
||||
DEB_FILE: ${{ steps.deb.outputs.deb_file }}
|
||||
RPM_FILE: ${{ steps.rpm.outputs.rpm_file }}
|
||||
shell: bash
|
||||
run: |
|
||||
set -euo pipefail
|
||||
|
||||
TAG="${{ needs.resolve.outputs.tag }}"
|
||||
DEB_FILE="${{ steps.deb.outputs.deb_file }}"
|
||||
RPM_FILE="${{ steps.rpm.outputs.rpm_file }}"
|
||||
|
||||
# Upload the packages, then refresh the release checksums so the new
|
||||
# assets are covered, matching the binary release flow.
|
||||
for f in "$DEB_FILE" "$RPM_FILE"; do
|
||||
@@ -552,14 +617,19 @@ jobs:
|
||||
steps:
|
||||
- name: Print summary
|
||||
shell: bash
|
||||
env:
|
||||
SUMMARY_VERSION: ${{ needs.resolve.outputs.version }}
|
||||
SUMMARY_BUILD_TYPE: ${{ needs.resolve.outputs.build_type }}
|
||||
SUMMARY_BUILD_RUN_ID: ${{ needs.resolve.outputs.build_run_id }}
|
||||
SUMMARY_PACKAGE_STATUS: ${{ needs.package.result }}
|
||||
run: |
|
||||
{
|
||||
echo "## 📦 Package Summary"
|
||||
echo ""
|
||||
echo "| Item | Value |"
|
||||
echo "|------|-------|"
|
||||
echo "| Version | \`${{ needs.resolve.outputs.version }}\` |"
|
||||
echo "| Build Type | ${{ needs.resolve.outputs.build_type }} |"
|
||||
echo "| Build Run | #${{ needs.resolve.outputs.build_run_id }} |"
|
||||
echo "| Package Status | ${{ needs.package.result }} |"
|
||||
echo "| Version | \`${SUMMARY_VERSION}\` |"
|
||||
echo "| Build Type | ${SUMMARY_BUILD_TYPE} |"
|
||||
echo "| Build Run | #${SUMMARY_BUILD_RUN_ID} |"
|
||||
echo "| Package Status | ${SUMMARY_PACKAGE_STATUS} |"
|
||||
} >> "$GITHUB_STEP_SUMMARY"
|
||||
|
||||
@@ -0,0 +1,74 @@
|
||||
# Copyright 2024 RustFS Team
|
||||
#
|
||||
# Licensed under the Apache License, Version 2.0 (the "License");
|
||||
# you may not use this file except in compliance with the License.
|
||||
# You may obtain a copy of the License at
|
||||
#
|
||||
# http://www.apache.org/licenses/LICENSE-2.0
|
||||
#
|
||||
# Unless required by applicable law or agreed to in writing, software
|
||||
# distributed under the License is distributed on an "AS IS" BASIS,
|
||||
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||
# See the License for the specific language governing permissions and
|
||||
# limitations under the License.
|
||||
|
||||
# Functional chain driver: runs the nine functional suites in a fixed order
|
||||
# (upgrade -> s3 -> kms -> tier -> storage -> heal -> pool -> security, with
|
||||
# performance on its own runner in parallel) and guarantees the chain keeps
|
||||
# moving even when individual suites fail.
|
||||
#
|
||||
# Each suite workflow can still be dispatched standalone (workflow_dispatch);
|
||||
# only chain-triggered runs forward to the next suite via repository_dispatch,
|
||||
# so a standalone run never drags the rest of the chain behind it.
|
||||
#
|
||||
# Why not workflow_run chaining: GitHub does not guarantee delivery of
|
||||
# workflow_run events (they are fire-and-forget), and the head-SHA filter made
|
||||
# newly added suites (storage) unable to trigger at all. Explicit
|
||||
# repository_dispatch handoffs are verifiable and re-drivable.
|
||||
|
||||
name: RustFS Functional Chain
|
||||
|
||||
on:
|
||||
workflow_dispatch:
|
||||
workflow_run:
|
||||
# Entry point: start the chain after the nightly build completes. The
|
||||
# build's own conclusion does not gate the chain; each suite reports its
|
||||
# own result to rustfs/backlog and the dashboard.
|
||||
workflows: ["Nightly GNU Build"]
|
||||
types: [completed]
|
||||
|
||||
permissions:
|
||||
contents: read
|
||||
|
||||
jobs:
|
||||
start-chain:
|
||||
name: Start functional chain (upgrade first)
|
||||
runs-on: ubuntu-latest
|
||||
timeout-minutes: 10
|
||||
if: ${{ github.event_name == 'workflow_dispatch' || (github.event_name == 'workflow_run' && github.event.workflow_run.event == 'schedule') }}
|
||||
steps:
|
||||
- name: Dispatch first suite (upgrade)
|
||||
env:
|
||||
GH_TOKEN: ${{ secrets.PF_TESTING_GH_TOKEN }}
|
||||
run: |
|
||||
set -euo pipefail
|
||||
if [ -z "${GH_TOKEN:-}" ]; then
|
||||
echo "PF_TESTING_GH_TOKEN is not configured; cannot start the functional chain" >&2
|
||||
exit 1
|
||||
fi
|
||||
gh api --method POST repos/rustfs/rustfs/dispatches \
|
||||
-f event_type='rustfs-chain-upgrade' \
|
||||
-F 'client_payload[from_suite]=nightly-build'
|
||||
|
||||
- name: Dispatch performance suite (parallel, own runner)
|
||||
env:
|
||||
GH_TOKEN: ${{ secrets.PF_TESTING_GH_TOKEN }}
|
||||
run: |
|
||||
set -euo pipefail
|
||||
if [ -z "${GH_TOKEN:-}" ]; then
|
||||
echo "PF_TESTING_GH_TOKEN is not configured; cannot dispatch performance" >&2
|
||||
exit 1
|
||||
fi
|
||||
gh api --method POST repos/rustfs/rustfs/dispatches \
|
||||
-f event_type='rustfs-chain-performance' \
|
||||
-F 'client_payload[from_suite]=nightly-build'
|
||||
@@ -23,6 +23,11 @@ on:
|
||||
description: 'Reset the nodes after the test (DESTROYS test data/config)'
|
||||
type: boolean
|
||||
default: true
|
||||
repository_dispatch:
|
||||
# Chain handoff: dispatched when the storage suite finishes. Heal runs
|
||||
# exactly once per chain; the pool expansion workflow no longer embeds
|
||||
# its own heal pass.
|
||||
types: [rustfs-chain-heal]
|
||||
|
||||
permissions:
|
||||
contents: read
|
||||
@@ -43,24 +48,39 @@ env:
|
||||
RUSTFS_API_ENDPOINT: ${{ secrets.RUSTFS_API_ENDPOINT || vars.RUSTFS_API_ENDPOINT || vars.RUSTFS_RC_ENDPOINT }}
|
||||
RUSTFS_NODES: ${{ secrets.RUSTFS_NODES || vars.RUSTFS_NODES }}
|
||||
RUSTFS_SSH_USER: ${{ secrets.RUSTFS_SSH_USER || vars.RUSTFS_SSH_USER }}
|
||||
PF_TESTING_GH_TOKEN: ${{ secrets.PF_TESTING_GH_TOKEN }}
|
||||
RUSTFS_NIGHTLY_PACKAGE_URL: ${{ vars.RUSTFS_NIGHTLY_PACKAGE_URL || 'https://dl.rustfs.com/artifacts/rustfs/packages/nightly/rustfs-nightly-latest.deb' }}
|
||||
|
||||
jobs:
|
||||
heal-test:
|
||||
runs-on: smoke-testing
|
||||
# Requirement: a failing suite must not fail the workflow; failures
|
||||
# are filed to rustfs/backlog and the chain continues.
|
||||
continue-on-error: true
|
||||
timeout-minutes: 480
|
||||
# Manual-only standalone run. Nightly chain already runs heal in
|
||||
# rustfs-pool-expand-test.yml to avoid duplicate heal executions.
|
||||
if: ${{ github.event_name == 'workflow_dispatch' }}
|
||||
# Standalone manual run, or one link of the nightly functional chain
|
||||
# (storage -> heal -> pool). Pool expansion no longer re-runs heal.
|
||||
if: ${{ github.event_name == 'workflow_dispatch' || github.event_name == 'repository_dispatch' }}
|
||||
steps:
|
||||
- name: Checkout auto-testing scripts
|
||||
uses: actions/checkout@9c091bb21b7c1c1d1991bb908d89e4e9dddfe3e0 # v7
|
||||
with:
|
||||
repository: rustfs/auto-testing
|
||||
ref: main
|
||||
path: auto-testing
|
||||
persist-credentials: false
|
||||
token: ${{ secrets.PF_TESTING_GH_TOKEN }}
|
||||
# auto-testing is private: clone it with the dedicated PF token (not
|
||||
# GITHUB_TOKEN) and retry transient GitHub/network failures.
|
||||
- name: Checkout auto-testing scripts (with retry)
|
||||
env:
|
||||
GH_TOKEN: ${{ secrets.PF_TESTING_GH_TOKEN }}
|
||||
run: |
|
||||
set -euo pipefail
|
||||
rm -rf auto-testing
|
||||
for attempt in 1 2 3 4 5; do
|
||||
if gh repo clone rustfs/auto-testing auto-testing -- --depth 1 --quiet; then
|
||||
echo "auto-testing cloned (attempt ${attempt})"
|
||||
exit 0
|
||||
fi
|
||||
rm -rf auto-testing
|
||||
echo "clone attempt ${attempt} failed; retrying in $((attempt * 15))s" >&2
|
||||
sleep $((attempt * 15))
|
||||
done
|
||||
echo "ERROR: unable to clone rustfs/auto-testing after 5 attempts" >&2
|
||||
exit 1
|
||||
|
||||
- name: Show environment
|
||||
run: |
|
||||
@@ -70,11 +90,24 @@ jobs:
|
||||
warp --version || true
|
||||
df -h /data | tail -1
|
||||
|
||||
- name: Reset test environment (before)
|
||||
- name: Cleanup environment (before)
|
||||
if: ${{ inputs.cleanup_before != 'false' }}
|
||||
run: |
|
||||
chmod +x auto-testing/rustfs_heal_test.sh
|
||||
./auto-testing/rustfs_heal_test.sh --reset -y
|
||||
set -euo pipefail
|
||||
read -r -a NODES <<< "${RUSTFS_NODES:-vm000 vm001 vm002}"
|
||||
SSH_USER="${RUSTFS_SSH_USER:-azureuser}"
|
||||
for node in "${NODES[@]}"; do
|
||||
ssh -o BatchMode=yes -o ConnectTimeout=10 -o StrictHostKeyChecking=accept-new "${SSH_USER}@${node}" '
|
||||
set -euo pipefail
|
||||
SUDO=""; [ "$(id -u)" -ne 0 ] && SUDO="sudo -n"
|
||||
${SUDO} systemctl stop rustfs 2>/dev/null || true
|
||||
if ${SUDO} dpkg -l rustfs 2>/dev/null | grep -q "^ii"; then
|
||||
${SUDO} dpkg -P rustfs
|
||||
fi
|
||||
for i in 1 2 3 4; do ${SUDO} rm -rf /data/rustfs${i}/mnmd; done
|
||||
${SUDO} rm -rf /var/log/rustfs /var/lib/rustfs/kms /var/lib/rustfs/kms-backup
|
||||
'
|
||||
done
|
||||
|
||||
- name: Install RustFS package & start cluster
|
||||
run: |
|
||||
@@ -97,14 +130,129 @@ jobs:
|
||||
./auto-testing/rustfs_heal_test.sh "${ARGS[@]}"
|
||||
|
||||
- name: Run heal test (write -> outage -> heal -> verify)
|
||||
id: test
|
||||
run: |
|
||||
./auto-testing/rustfs_heal_test.sh \
|
||||
--steps "3,4,5,6,7" -y \
|
||||
--endpoint "${{ env.RUSTFS_API_ENDPOINT }}" \
|
||||
--stop-node-gb "${{ inputs.stop_node_gb }}" \
|
||||
--warp-stop-gb "${{ inputs.warp_stop_gb }}" \
|
||||
--stop-node-gb "${{ inputs.stop_node_gb || '15' }}" \
|
||||
--warp-stop-gb "${{ inputs.warp_stop_gb || '40' }}" \
|
||||
--log-file /tmp/rustfs-heal-test.log
|
||||
|
||||
- name: Generate report
|
||||
if: always()
|
||||
env:
|
||||
LOG_FILE: /tmp/rustfs-heal-test.log
|
||||
REPORT_FILE: /tmp/rustfs-heal-report.md
|
||||
run: |
|
||||
set -euo pipefail
|
||||
PACKAGE_URL='${{ inputs.package_url }}'
|
||||
if [ -n "${PACKAGE_URL}" ]; then
|
||||
PACKAGE_SOURCE="${PACKAGE_URL}"
|
||||
else
|
||||
PACKAGE_SOURCE="${RUSTFS_NIGHTLY_PACKAGE_URL}"
|
||||
fi
|
||||
{
|
||||
echo "# RustFS heal test report"
|
||||
echo ""
|
||||
echo "- Run: ${{ github.server_url }}/${{ github.repository }}/actions/runs/${{ github.run_id }}"
|
||||
echo "- Trigger: ${{ github.event_name }}"
|
||||
echo "- Package: ${PACKAGE_SOURCE}"
|
||||
echo "- Test Step Outcome: ${{ steps.test.outcome }}"
|
||||
echo ""
|
||||
echo "## Log tail"
|
||||
echo '```text'
|
||||
tail -n 200 "${LOG_FILE}" || true
|
||||
echo '```'
|
||||
} | tee "${REPORT_FILE}"
|
||||
cat "${REPORT_FILE}" >> "${GITHUB_STEP_SUMMARY}"
|
||||
|
||||
- name: Upload functional report to dashboard
|
||||
if: always()
|
||||
continue-on-error: true
|
||||
env:
|
||||
GH_TOKEN: ${{ env.PF_TESTING_GH_TOKEN }}
|
||||
REPORT_FILE: /tmp/rustfs-heal-report.md
|
||||
SUITE: heal
|
||||
run: |
|
||||
set -euo pipefail
|
||||
if [ -z "${GH_TOKEN:-}" ]; then
|
||||
echo "PF_TESTING_GH_TOKEN is not configured; skipping dashboard upload"
|
||||
exit 0
|
||||
fi
|
||||
DATE="$(date -u +%Y-%m-%d)"
|
||||
REPORT_PATH="functional-reports/${SUITE}/${DATE}.md"
|
||||
CONTENT="$(python3 -c 'import base64,sys;print(base64.b64encode(open(sys.argv[1],"rb").read()).decode())' "${REPORT_FILE}")"
|
||||
SHA="$(gh api "repos/rustfs/dashboard/contents/${REPORT_PATH}" -q '.sha' 2>/dev/null || true)"
|
||||
if [ -n "${SHA}" ]; then
|
||||
jq -n --arg msg "report(${SUITE}): ${DATE}" --arg content "${CONTENT}" --arg sha "${SHA}" \
|
||||
'{message:$msg, content:$content, sha:$sha}' \
|
||||
| gh api --method PUT "repos/rustfs/dashboard/contents/${REPORT_PATH}" --input - >/dev/null
|
||||
else
|
||||
jq -n --arg msg "report(${SUITE}): ${DATE}" --arg content "${CONTENT}" \
|
||||
'{message:$msg, content:$content}' \
|
||||
| gh api --method PUT "repos/rustfs/dashboard/contents/${REPORT_PATH}" --input - >/dev/null
|
||||
fi
|
||||
|
||||
- name: File failure issue in rustfs/backlog
|
||||
if: ${{ always() && (failure() || steps.test.outcome == 'failure' || steps.test.outcome == 'cancelled') }}
|
||||
continue-on-error: true
|
||||
env:
|
||||
GH_TOKEN: ${{ secrets.PF_TESTING_GH_TOKEN }}
|
||||
SUITE: 'heal'
|
||||
SUITE_LABEL: 'Heal'
|
||||
RUN_URL: ${{ github.server_url }}/${{ github.repository }}/actions/runs/${{ github.run_id }}
|
||||
REPORT_FILE: '/tmp/rustfs-heal-report.md'
|
||||
LOG_FILE: '/tmp/rustfs-heal-test.log'
|
||||
run: |
|
||||
set -euo pipefail
|
||||
if [ -z "${GH_TOKEN:-}" ]; then
|
||||
echo "PF_TESTING_GH_TOKEN is not configured; skipping backlog issue"
|
||||
exit 0
|
||||
fi
|
||||
TITLE="[functional][${SUITE}] ${SUITE_LABEL} suite failed (run ${GITHUB_RUN_ID})"
|
||||
EXISTING="$(gh issue list -R rustfs/backlog --state all \
|
||||
--search "in:title \"run ${GITHUB_RUN_ID}\"" \
|
||||
--json number --jq '.[].number' || true)"
|
||||
if [ -n "${EXISTING}" ]; then
|
||||
echo "backlog issue already exists for run ${GITHUB_RUN_ID}; skipping"
|
||||
exit 0
|
||||
fi
|
||||
redact() {
|
||||
sed -E \
|
||||
-e 's/(RUSTFS_(ACCESS_KEY|SECRET_KEY)[=: ]+)[^[:space:]]+/\1[REDACTED]/Ig' \
|
||||
-e 's/(Authorization:).*/\1 [REDACTED]/Ig' \
|
||||
-e 's/(X-Amz-Signature=)[^&[:space:]]+/\1[REDACTED]/Ig' \
|
||||
-e 's/^.*(password|secret|token)[=: ].*/[REDACTED SENSITIVE LINE]/Ig'
|
||||
}
|
||||
BODY_FILE="$(mktemp)"
|
||||
{
|
||||
echo "The **${SUITE_LABEL}** functional suite failed."
|
||||
echo ""
|
||||
echo "- Suite: \`${SUITE}\`"
|
||||
echo "- Run: ${RUN_URL}"
|
||||
echo "- Trigger: ${GITHUB_EVENT_NAME}"
|
||||
echo "- Date: $(date -u +%Y-%m-%d)"
|
||||
echo ""
|
||||
echo "## Report (errors and symptoms)"
|
||||
echo ""
|
||||
if [ -s "${REPORT_FILE}" ]; then
|
||||
redact < "${REPORT_FILE}"
|
||||
elif [ -s "${LOG_FILE:-}" ]; then
|
||||
echo "(report file missing; log tail below)"
|
||||
echo ""
|
||||
tail -n 200 "${LOG_FILE}" | redact
|
||||
else
|
||||
echo "(no report or log file was produced)"
|
||||
fi
|
||||
} | head -c 55000 > "${BODY_FILE}"
|
||||
gh label create functional-test -R rustfs/backlog --color d73a4a 2>/dev/null || true
|
||||
if ! gh issue create -R rustfs/backlog --title "${TITLE}" \
|
||||
--body-file "${BODY_FILE}" --label functional-test; then
|
||||
gh issue create -R rustfs/backlog --title "${TITLE}" --body-file "${BODY_FILE}"
|
||||
fi
|
||||
echo "filed backlog issue for suite ${SUITE}"
|
||||
|
||||
- name: Upload test logs
|
||||
if: always()
|
||||
uses: actions/upload-artifact@b7c566a772e6b6bfb58ed0dc250532a479d7789f # v6
|
||||
@@ -115,10 +263,73 @@ jobs:
|
||||
/tmp/rustfs-warp.*.log
|
||||
if-no-files-found: warn
|
||||
|
||||
- name: Reset test environment (after)
|
||||
- name: Cleanup environment (after)
|
||||
if: ${{ always() && inputs.cleanup_after != 'false' }}
|
||||
run: |
|
||||
./auto-testing/rustfs_heal_test.sh --reset -y
|
||||
set -euo pipefail
|
||||
read -r -a NODES <<< "${RUSTFS_NODES:-vm000 vm001 vm002}"
|
||||
SSH_USER="${RUSTFS_SSH_USER:-azureuser}"
|
||||
for node in "${NODES[@]}"; do
|
||||
ssh -o BatchMode=yes -o ConnectTimeout=10 -o StrictHostKeyChecking=accept-new "${SSH_USER}@${node}" '
|
||||
set -euo pipefail
|
||||
SUDO=""; [ "$(id -u)" -ne 0 ] && SUDO="sudo -n"
|
||||
${SUDO} systemctl stop rustfs 2>/dev/null || true
|
||||
if ${SUDO} dpkg -l rustfs 2>/dev/null | grep -q "^ii"; then
|
||||
${SUDO} dpkg -P rustfs
|
||||
fi
|
||||
for i in 1 2 3 4; do ${SUDO} rm -rf /data/rustfs${i}/mnmd; done
|
||||
${SUDO} rm -rf /var/log/rustfs /var/lib/rustfs/kms /var/lib/rustfs/kms-backup
|
||||
'
|
||||
done
|
||||
|
||||
- name: "Continue functional chain (next: Pool expansion)"
|
||||
# Only chain-triggered runs forward to the next suite; standalone
|
||||
# workflow_dispatch runs stop after their own cleanup. A failed
|
||||
# handoff must never pass silently: it retries, then files an alert
|
||||
# issue in rustfs/backlog so a stalled chain is visible.
|
||||
if: ${{ always() && github.event_name == 'repository_dispatch' }}
|
||||
continue-on-error: true
|
||||
env:
|
||||
GH_TOKEN: ${{ secrets.PF_TESTING_GH_TOKEN }}
|
||||
run: |
|
||||
set -uo pipefail
|
||||
if [ -z "${{GH_TOKEN:-}}" ]; then
|
||||
echo "PF_TESTING_GH_TOKEN is not configured; cannot dispatch the next suite" >&2
|
||||
exit 1
|
||||
fi
|
||||
DISPATCHED=0
|
||||
for attempt in 1 2 3; do
|
||||
if gh api --method POST repos/rustfs/rustfs/dispatches \
|
||||
-f event_type='rustfs-chain-pool' \
|
||||
-F 'client_payload[from_suite]=heal'; then
|
||||
echo "dispatched next suite Pool expansion (attempt ${{attempt}})"
|
||||
DISPATCHED=1
|
||||
break
|
||||
fi
|
||||
echo "dispatch attempt ${{attempt}} failed; retrying in ${{attempt}}0s" >&2
|
||||
sleep "${{attempt}}0"
|
||||
done
|
||||
if [ "${{DISPATCHED:-0}}" -ne 1 ]; then
|
||||
echo "ERROR: functional chain stalled: could not dispatch Pool expansion after 3 attempts" >&2
|
||||
TITLE="[functional][chain] stalled after heal (run ${{GITHUB_RUN_ID}})"
|
||||
BODY_FILE="$(mktemp)"
|
||||
{
|
||||
echo "The functional chain could not hand off from **heal** to **Pool expansion** after 3 attempts."
|
||||
echo ""
|
||||
echo "- Failed suite job: ${{GITHUB_SERVER_URL}}/${{GITHUB_REPOSITORY}}/actions/runs/${{GITHUB_RUN_ID}}"
|
||||
echo "- Expected next event: `rustfs-chain-pool`"
|
||||
echo "- Likely cause: PF_TESTING_GH_TOKEN lacks contents:write on rustfs/rustfs, or the GitHub API was unavailable."
|
||||
echo "- Recovery: re-dispatch manually with"
|
||||
echo " ```"
|
||||
echo " gh api --method POST repos/rustfs/rustfs/dispatches -f event_type='rustfs-chain-pool'"
|
||||
echo " ```"
|
||||
} > "${{BODY_FILE}}"
|
||||
gh issue create -R rustfs/backlog --title "${{TITLE}}" \
|
||||
--body-file "${{BODY_FILE}}" --label functional-test \
|
||||
|| gh issue create -R rustfs/backlog --title "${{TITLE}}" --body-file "${{BODY_FILE}}" \
|
||||
|| echo "could not file the stall alert issue either; check the token" >&2
|
||||
exit 1
|
||||
fi
|
||||
|
||||
- name: Notify on failure
|
||||
if: failure()
|
||||
|
||||
@@ -11,10 +11,21 @@ on:
|
||||
description: 'Direct .deb URL (nightly/R2/dev). Overrides rustfs_version.'
|
||||
required: false
|
||||
type: string
|
||||
workflow_run:
|
||||
# Strict shared-environment order: run after S3 compatibility test succeeds.
|
||||
workflows: ["RustFS S3 Compatibility Test"]
|
||||
types: [completed]
|
||||
enforce_sse_key_policy:
|
||||
description: 'Enable RUSTFS_KMS_ENFORCE_SSE_KEY_POLICY (runs KMS-401/402)'
|
||||
type: boolean
|
||||
default: false
|
||||
frame_v2:
|
||||
description: 'Enable RUSTFS_ENCRYPTION_FRAME_V2 (runs KMS-318)'
|
||||
type: boolean
|
||||
default: false
|
||||
config_secret:
|
||||
description: 'Set RUSTFS_KMS_CONFIG_SECRET (runs KMS-107 config sealing)'
|
||||
required: false
|
||||
type: string
|
||||
repository_dispatch:
|
||||
# Chain handoff: dispatched when the S3 compatibility suite finishes.
|
||||
types: [rustfs-chain-kms]
|
||||
|
||||
permissions:
|
||||
contents: read
|
||||
@@ -38,17 +49,29 @@ env:
|
||||
jobs:
|
||||
kms-test:
|
||||
runs-on: smoke-testing
|
||||
continue-on-error: true
|
||||
timeout-minutes: 420
|
||||
if: ${{ github.event_name == 'workflow_dispatch' || github.event.workflow_run.conclusion == 'success' }}
|
||||
if: ${{ github.event_name == 'workflow_dispatch' || github.event_name == 'repository_dispatch' }}
|
||||
steps:
|
||||
- name: Checkout auto-testing scripts
|
||||
uses: actions/checkout@9c091bb21b7c1c1d1991bb908d89e4e9dddfe3e0 # v7
|
||||
with:
|
||||
repository: rustfs/auto-testing
|
||||
ref: main
|
||||
path: auto-testing
|
||||
persist-credentials: false
|
||||
token: ${{ secrets.PF_TESTING_GH_TOKEN }}
|
||||
# auto-testing is private: clone it with the dedicated PF token (not
|
||||
# GITHUB_TOKEN) and retry transient GitHub/network failures.
|
||||
- name: Checkout auto-testing scripts (with retry)
|
||||
env:
|
||||
GH_TOKEN: ${{ secrets.PF_TESTING_GH_TOKEN }}
|
||||
run: |
|
||||
set -euo pipefail
|
||||
rm -rf auto-testing
|
||||
for attempt in 1 2 3 4 5; do
|
||||
if gh repo clone rustfs/auto-testing auto-testing -- --depth 1 --quiet; then
|
||||
echo "auto-testing cloned (attempt ${attempt})"
|
||||
exit 0
|
||||
fi
|
||||
rm -rf auto-testing
|
||||
echo "clone attempt ${attempt} failed; retrying in $((attempt * 15))s" >&2
|
||||
sleep $((attempt * 15))
|
||||
done
|
||||
echo "ERROR: unable to clone rustfs/auto-testing after 5 attempts" >&2
|
||||
exit 1
|
||||
|
||||
- name: Show environment
|
||||
run: |
|
||||
@@ -86,6 +109,7 @@ jobs:
|
||||
|
||||
- name: Run KMS suite
|
||||
id: test
|
||||
continue-on-error: true
|
||||
env:
|
||||
LOG_FILE: /tmp/rustfs-kms.log
|
||||
run: |
|
||||
@@ -94,6 +118,19 @@ jobs:
|
||||
PACKAGE_URL='${{ inputs.package_url }}'
|
||||
RUSTFS_VERSION='${{ inputs.rustfs_version }}'
|
||||
ARGS=(--all-topologies --backends "local,vault-kv2" -y --log-file "${LOG_FILE}")
|
||||
EXTRA_ENV=""
|
||||
if [ "${{ inputs.enforce_sse_key_policy }}" = "true" ]; then
|
||||
EXTRA_ENV+="RUSTFS_KMS_ENFORCE_SSE_KEY_POLICY=true"$'\n'
|
||||
fi
|
||||
if [ "${{ inputs.frame_v2 }}" = "true" ]; then
|
||||
EXTRA_ENV+="RUSTFS_ENCRYPTION_FRAME_V2=true"$'\n'
|
||||
fi
|
||||
if [ -n "${{ inputs.config_secret }}" ]; then
|
||||
EXTRA_ENV+="RUSTFS_KMS_CONFIG_SECRET=${{ inputs.config_secret }}"$'\n'
|
||||
fi
|
||||
if [ -n "${EXTRA_ENV}" ]; then
|
||||
ARGS+=(--extra-env "${EXTRA_ENV}")
|
||||
fi
|
||||
if [ -n "${PACKAGE_URL}" ]; then
|
||||
ARGS+=(--package-url "${PACKAGE_URL}")
|
||||
elif [ -n "${RUSTFS_VERSION}" ]; then
|
||||
@@ -119,12 +156,65 @@ jobs:
|
||||
else
|
||||
PACKAGE_SOURCE="${RUSTFS_NIGHTLY_PACKAGE_URL}"
|
||||
fi
|
||||
CASE_TABLE="/tmp/rustfs-kms-cases.md"
|
||||
python3 - "${LOG_FILE}" "${CASE_TABLE}" <<'PY'
|
||||
import re
|
||||
import sys
|
||||
|
||||
log_file, out_file = sys.argv[1], sys.argv[2]
|
||||
ansi = re.compile(r'\x1b\[[0-9;]*m')
|
||||
start_re = re.compile(r'^---\s+([A-Z]+-[0-9]+)\s+(.+?)\s+---$')
|
||||
done_re = re.compile(r'^\[(PASS|FAIL|UNSUPPORTED)\]\s+([A-Z]+-[0-9]+)\b')
|
||||
|
||||
rows = []
|
||||
index = {}
|
||||
try:
|
||||
with open(log_file, 'r', encoding='utf-8', errors='replace') as fh:
|
||||
for raw in fh:
|
||||
line = ansi.sub('', raw).strip()
|
||||
m = start_re.match(line)
|
||||
if m:
|
||||
case_id, name = m.group(1), m.group(2)
|
||||
if case_id not in index:
|
||||
index[case_id] = len(rows)
|
||||
rows.append([case_id, name, 'RUNNING'])
|
||||
continue
|
||||
m = done_re.match(line)
|
||||
if m:
|
||||
status, case_id = m.group(1), m.group(2)
|
||||
if case_id in index:
|
||||
rows[index[case_id]][2] = status
|
||||
else:
|
||||
rows.append([case_id, case_id, status])
|
||||
index[case_id] = len(rows) - 1
|
||||
except FileNotFoundError:
|
||||
rows = []
|
||||
|
||||
counts = {'PASS': 0, 'FAIL': 0, 'UNSUPPORTED': 0, 'RUNNING': 0}
|
||||
for _, _, status in rows:
|
||||
counts[status] = counts.get(status, 0) + 1
|
||||
|
||||
with open(out_file, 'w', encoding='utf-8') as out:
|
||||
out.write('## Case Summary\n\n')
|
||||
out.write(f"- Total: {len(rows)}\\n")
|
||||
out.write(f"- PASS: {counts.get('PASS', 0)}\\n")
|
||||
out.write(f"- FAIL: {counts.get('FAIL', 0)}\\n")
|
||||
out.write(f"- UNSUPPORTED: {counts.get('UNSUPPORTED', 0)}\\n")
|
||||
out.write('\\n')
|
||||
out.write('| Case | Name | Status |\\n')
|
||||
out.write('| --- | --- | --- |\\n')
|
||||
for case_id, name, status in rows:
|
||||
out.write(f'| {case_id} | {name} | {status} |\\n')
|
||||
PY
|
||||
{
|
||||
echo "# RustFS KMS test report"
|
||||
echo ""
|
||||
echo "- Run: ${{ github.server_url }}/${{ github.repository }}/actions/runs/${{ github.run_id }}"
|
||||
echo "- Trigger: ${{ github.event_name }}"
|
||||
echo "- Package: ${PACKAGE_SOURCE}"
|
||||
echo "- Test Step Outcome: ${{ steps.test.outcome }}"
|
||||
echo ""
|
||||
cat "${CASE_TABLE}" || true
|
||||
echo ""
|
||||
echo "## Log tail"
|
||||
echo '```text'
|
||||
@@ -133,6 +223,92 @@ jobs:
|
||||
} | tee "${REPORT_FILE}"
|
||||
cat "${REPORT_FILE}" >> "${GITHUB_STEP_SUMMARY}"
|
||||
|
||||
- name: Upload functional report to dashboard
|
||||
if: always()
|
||||
continue-on-error: true
|
||||
env:
|
||||
GH_TOKEN: ${{ env.PF_TESTING_GH_TOKEN }}
|
||||
REPORT_FILE: /tmp/rustfs-kms-report.md
|
||||
SUITE: kms
|
||||
run: |
|
||||
set -euo pipefail
|
||||
if [ -z "${GH_TOKEN:-}" ]; then
|
||||
echo "PF_TESTING_GH_TOKEN is not configured; skipping dashboard upload"
|
||||
exit 0
|
||||
fi
|
||||
DATE="$(date -u +%Y-%m-%d)"
|
||||
REPORT_PATH="functional-reports/${SUITE}/${DATE}.md"
|
||||
CONTENT="$(python3 -c 'import base64,sys;print(base64.b64encode(open(sys.argv[1],"rb").read()).decode())' "${REPORT_FILE}")"
|
||||
SHA="$(gh api "repos/rustfs/dashboard/contents/${REPORT_PATH}" -q '.sha' 2>/dev/null || true)"
|
||||
if [ -n "${SHA}" ]; then
|
||||
jq -n --arg msg "report(${SUITE}): ${DATE}" --arg content "${CONTENT}" --arg sha "${SHA}" \
|
||||
'{message:$msg, content:$content, sha:$sha}' \
|
||||
| gh api --method PUT "repos/rustfs/dashboard/contents/${REPORT_PATH}" --input - >/dev/null
|
||||
else
|
||||
jq -n --arg msg "report(${SUITE}): ${DATE}" --arg content "${CONTENT}" \
|
||||
'{message:$msg, content:$content}' \
|
||||
| gh api --method PUT "repos/rustfs/dashboard/contents/${REPORT_PATH}" --input - >/dev/null
|
||||
fi
|
||||
|
||||
- name: File failure issue in rustfs/backlog
|
||||
if: ${{ always() && (failure() || steps.test.outcome == 'failure' || steps.test.outcome == 'cancelled') }}
|
||||
continue-on-error: true
|
||||
env:
|
||||
GH_TOKEN: ${{ secrets.PF_TESTING_GH_TOKEN }}
|
||||
SUITE: 'kms'
|
||||
SUITE_LABEL: 'KMS'
|
||||
RUN_URL: ${{ github.server_url }}/${{ github.repository }}/actions/runs/${{ github.run_id }}
|
||||
REPORT_FILE: '/tmp/rustfs-kms-report.md'
|
||||
LOG_FILE: '/tmp/rustfs-kms.log'
|
||||
run: |
|
||||
set -euo pipefail
|
||||
if [ -z "${GH_TOKEN:-}" ]; then
|
||||
echo "PF_TESTING_GH_TOKEN is not configured; skipping backlog issue"
|
||||
exit 0
|
||||
fi
|
||||
TITLE="[functional][${SUITE}] ${SUITE_LABEL} suite failed (run ${GITHUB_RUN_ID})"
|
||||
EXISTING="$(gh issue list -R rustfs/backlog --state all \
|
||||
--search "in:title \"run ${GITHUB_RUN_ID}\"" \
|
||||
--json number --jq '.[].number' || true)"
|
||||
if [ -n "${EXISTING}" ]; then
|
||||
echo "backlog issue already exists for run ${GITHUB_RUN_ID}; skipping"
|
||||
exit 0
|
||||
fi
|
||||
redact() {
|
||||
sed -E \
|
||||
-e 's/(RUSTFS_(ACCESS_KEY|SECRET_KEY)[=: ]+)[^[:space:]]+/\1[REDACTED]/Ig' \
|
||||
-e 's/(Authorization:).*/\1 [REDACTED]/Ig' \
|
||||
-e 's/(X-Amz-Signature=)[^&[:space:]]+/\1[REDACTED]/Ig' \
|
||||
-e 's/^.*(password|secret|token)[=: ].*/[REDACTED SENSITIVE LINE]/Ig'
|
||||
}
|
||||
BODY_FILE="$(mktemp)"
|
||||
{
|
||||
echo "The **${SUITE_LABEL}** functional suite failed."
|
||||
echo ""
|
||||
echo "- Suite: \`${SUITE}\`"
|
||||
echo "- Run: ${RUN_URL}"
|
||||
echo "- Trigger: ${GITHUB_EVENT_NAME}"
|
||||
echo "- Date: $(date -u +%Y-%m-%d)"
|
||||
echo ""
|
||||
echo "## Report (errors and symptoms)"
|
||||
echo ""
|
||||
if [ -s "${REPORT_FILE}" ]; then
|
||||
redact < "${REPORT_FILE}"
|
||||
elif [ -s "${LOG_FILE:-}" ]; then
|
||||
echo "(report file missing; log tail below)"
|
||||
echo ""
|
||||
tail -n 200 "${LOG_FILE}" | redact
|
||||
else
|
||||
echo "(no report or log file was produced)"
|
||||
fi
|
||||
} | head -c 55000 > "${BODY_FILE}"
|
||||
gh label create functional-test -R rustfs/backlog --color d73a4a 2>/dev/null || true
|
||||
if ! gh issue create -R rustfs/backlog --title "${TITLE}" \
|
||||
--body-file "${BODY_FILE}" --label functional-test; then
|
||||
gh issue create -R rustfs/backlog --title "${TITLE}" --body-file "${BODY_FILE}"
|
||||
fi
|
||||
echo "filed backlog issue for suite ${SUITE}"
|
||||
|
||||
- name: Upload report and logs
|
||||
if: always()
|
||||
uses: actions/upload-artifact@b7c566a772e6b6bfb58ed0dc250532a479d7789f # v6
|
||||
@@ -162,6 +338,55 @@ jobs:
|
||||
'
|
||||
done
|
||||
|
||||
- name: "Continue functional chain (next: Tier)"
|
||||
# Only chain-triggered runs forward to the next suite; standalone
|
||||
# workflow_dispatch runs stop after their own cleanup. A failed
|
||||
# handoff must never pass silently: it retries, then files an alert
|
||||
# issue in rustfs/backlog so a stalled chain is visible.
|
||||
if: ${{ always() && github.event_name == 'repository_dispatch' }}
|
||||
continue-on-error: true
|
||||
env:
|
||||
GH_TOKEN: ${{ secrets.PF_TESTING_GH_TOKEN }}
|
||||
run: |
|
||||
set -uo pipefail
|
||||
if [ -z "${{GH_TOKEN:-}}" ]; then
|
||||
echo "PF_TESTING_GH_TOKEN is not configured; cannot dispatch the next suite" >&2
|
||||
exit 1
|
||||
fi
|
||||
DISPATCHED=0
|
||||
for attempt in 1 2 3; do
|
||||
if gh api --method POST repos/rustfs/rustfs/dispatches \
|
||||
-f event_type='rustfs-chain-tier' \
|
||||
-F 'client_payload[from_suite]=kms'; then
|
||||
echo "dispatched next suite Tier (attempt ${{attempt}})"
|
||||
DISPATCHED=1
|
||||
break
|
||||
fi
|
||||
echo "dispatch attempt ${{attempt}} failed; retrying in ${{attempt}}0s" >&2
|
||||
sleep "${{attempt}}0"
|
||||
done
|
||||
if [ "${{DISPATCHED:-0}}" -ne 1 ]; then
|
||||
echo "ERROR: functional chain stalled: could not dispatch Tier after 3 attempts" >&2
|
||||
TITLE="[functional][chain] stalled after kms (run ${{GITHUB_RUN_ID}})"
|
||||
BODY_FILE="$(mktemp)"
|
||||
{
|
||||
echo "The functional chain could not hand off from **kms** to **Tier** after 3 attempts."
|
||||
echo ""
|
||||
echo "- Failed suite job: ${{GITHUB_SERVER_URL}}/${{GITHUB_REPOSITORY}}/actions/runs/${{GITHUB_RUN_ID}}"
|
||||
echo "- Expected next event: `rustfs-chain-tier`"
|
||||
echo "- Likely cause: PF_TESTING_GH_TOKEN lacks contents:write on rustfs/rustfs, or the GitHub API was unavailable."
|
||||
echo "- Recovery: re-dispatch manually with"
|
||||
echo " ```"
|
||||
echo " gh api --method POST repos/rustfs/rustfs/dispatches -f event_type='rustfs-chain-tier'"
|
||||
echo " ```"
|
||||
} > "${{BODY_FILE}}"
|
||||
gh issue create -R rustfs/backlog --title "${{TITLE}}" \
|
||||
--body-file "${{BODY_FILE}}" --label functional-test \
|
||||
|| gh issue create -R rustfs/backlog --title "${{TITLE}}" --body-file "${{BODY_FILE}}" \
|
||||
|| echo "could not file the stall alert issue either; check the token" >&2
|
||||
exit 1
|
||||
fi
|
||||
|
||||
- name: Notify on failure
|
||||
if: failure()
|
||||
run: |
|
||||
|
||||
@@ -48,10 +48,10 @@ on:
|
||||
description: 'Reset the nodes after the test (DESTROYS test data/config)'
|
||||
type: boolean
|
||||
default: true
|
||||
workflow_run:
|
||||
# Run after the nightly build completes; the nightly deb is what the test installs.
|
||||
workflows: ["Nightly GNU Build"]
|
||||
types: [completed]
|
||||
repository_dispatch:
|
||||
# Chain entry: dispatched by rustfs-functional-chain.yml (runs on its own
|
||||
# pf-testing runner, in parallel with the shared-VM chain).
|
||||
types: [rustfs-chain-performance]
|
||||
|
||||
permissions:
|
||||
contents: read
|
||||
@@ -84,19 +84,33 @@ env:
|
||||
jobs:
|
||||
performance-test:
|
||||
runs-on: pf-testing
|
||||
# Requirement: a failing benchmark must not fail the workflow;
|
||||
# failures are filed to rustfs/backlog.
|
||||
continue-on-error: true
|
||||
timeout-minutes: 900
|
||||
# Run on manual dispatch, or when the nightly build completed successfully.
|
||||
# Skipped when nightly failed.
|
||||
if: ${{ github.event_name == 'workflow_dispatch' || github.event.workflow_run.conclusion == 'success' }}
|
||||
if: ${{ github.event_name == 'workflow_dispatch' || github.event_name == 'repository_dispatch' }}
|
||||
steps:
|
||||
- name: Checkout auto-testing scripts
|
||||
uses: actions/checkout@9c091bb21b7c1c1d1991bb908d89e4e9dddfe3e0 # v7
|
||||
with:
|
||||
repository: rustfs/auto-testing
|
||||
ref: main
|
||||
path: auto-testing
|
||||
persist-credentials: false
|
||||
token: ${{ secrets.PF_TESTING_GH_TOKEN }}
|
||||
# auto-testing is private: clone it with the dedicated PF token (not
|
||||
# GITHUB_TOKEN) and retry transient GitHub/network failures.
|
||||
- name: Checkout auto-testing scripts (with retry)
|
||||
env:
|
||||
GH_TOKEN: ${{ secrets.PF_TESTING_GH_TOKEN }}
|
||||
run: |
|
||||
set -euo pipefail
|
||||
rm -rf auto-testing
|
||||
for attempt in 1 2 3 4 5; do
|
||||
if gh repo clone rustfs/auto-testing auto-testing -- --depth 1 --quiet; then
|
||||
echo "auto-testing cloned (attempt ${attempt})"
|
||||
exit 0
|
||||
fi
|
||||
rm -rf auto-testing
|
||||
echo "clone attempt ${attempt} failed; retrying in $((attempt * 15))s" >&2
|
||||
sleep $((attempt * 15))
|
||||
done
|
||||
echo "ERROR: unable to clone rustfs/auto-testing after 5 attempts" >&2
|
||||
exit 1
|
||||
|
||||
- name: Show environment
|
||||
run: |
|
||||
@@ -212,6 +226,65 @@ jobs:
|
||||
echo "created ${REPORT_PATH} in rustfs/dashboard"
|
||||
fi
|
||||
|
||||
- name: File failure issue in rustfs/backlog
|
||||
if: ${{ always() && (failure() || steps.benchmark.outcome == 'failure' || steps.benchmark.outcome == 'cancelled') }}
|
||||
continue-on-error: true
|
||||
env:
|
||||
GH_TOKEN: ${{ secrets.PF_TESTING_GH_TOKEN }}
|
||||
SUITE: 'performance'
|
||||
SUITE_LABEL: 'Performance'
|
||||
RUN_URL: ${{ github.server_url }}/${{ github.repository }}/actions/runs/${{ github.run_id }}
|
||||
REPORT_FILE: '/tmp/rustfs-perf-report.md'
|
||||
LOG_FILE: '/tmp/rustfs-perf-test.log'
|
||||
run: |
|
||||
set -euo pipefail
|
||||
if [ -z "${GH_TOKEN:-}" ]; then
|
||||
echo "PF_TESTING_GH_TOKEN is not configured; skipping backlog issue"
|
||||
exit 0
|
||||
fi
|
||||
TITLE="[functional][${SUITE}] ${SUITE_LABEL} suite failed (run ${GITHUB_RUN_ID})"
|
||||
EXISTING="$(gh issue list -R rustfs/backlog --state all \
|
||||
--search "in:title \"run ${GITHUB_RUN_ID}\"" \
|
||||
--json number --jq '.[].number' || true)"
|
||||
if [ -n "${EXISTING}" ]; then
|
||||
echo "backlog issue already exists for run ${GITHUB_RUN_ID}; skipping"
|
||||
exit 0
|
||||
fi
|
||||
redact() {
|
||||
sed -E \
|
||||
-e 's/(RUSTFS_(ACCESS_KEY|SECRET_KEY)[=: ]+)[^[:space:]]+/\1[REDACTED]/Ig' \
|
||||
-e 's/(Authorization:).*/\1 [REDACTED]/Ig' \
|
||||
-e 's/(X-Amz-Signature=)[^&[:space:]]+/\1[REDACTED]/Ig' \
|
||||
-e 's/^.*(password|secret|token)[=: ].*/[REDACTED SENSITIVE LINE]/Ig'
|
||||
}
|
||||
BODY_FILE="$(mktemp)"
|
||||
{
|
||||
echo "The **${SUITE_LABEL}** functional suite failed."
|
||||
echo ""
|
||||
echo "- Suite: \`${SUITE}\`"
|
||||
echo "- Run: ${RUN_URL}"
|
||||
echo "- Trigger: ${GITHUB_EVENT_NAME}"
|
||||
echo "- Date: $(date -u +%Y-%m-%d)"
|
||||
echo ""
|
||||
echo "## Report (errors and symptoms)"
|
||||
echo ""
|
||||
if [ -s "${REPORT_FILE}" ]; then
|
||||
redact < "${REPORT_FILE}"
|
||||
elif [ -s "${LOG_FILE:-}" ]; then
|
||||
echo "(report file missing; log tail below)"
|
||||
echo ""
|
||||
tail -n 200 "${LOG_FILE}" | redact
|
||||
else
|
||||
echo "(no report or log file was produced)"
|
||||
fi
|
||||
} | head -c 55000 > "${BODY_FILE}"
|
||||
gh label create functional-test -R rustfs/backlog --color d73a4a 2>/dev/null || true
|
||||
if ! gh issue create -R rustfs/backlog --title "${TITLE}" \
|
||||
--body-file "${BODY_FILE}" --label functional-test; then
|
||||
gh issue create -R rustfs/backlog --title "${TITLE}" --body-file "${BODY_FILE}"
|
||||
fi
|
||||
echo "filed backlog issue for suite ${SUITE}"
|
||||
|
||||
- name: Upload test logs & results
|
||||
if: always()
|
||||
uses: actions/upload-artifact@b7c566a772e6b6bfb58ed0dc250532a479d7789f # v6
|
||||
|
||||
@@ -1,4 +1,4 @@
|
||||
name: RustFS Pool Expansion / Heal Test
|
||||
name: RustFS Pool Expansion Test
|
||||
|
||||
on:
|
||||
workflow_dispatch:
|
||||
@@ -25,18 +25,14 @@ on:
|
||||
description: 'warp write duration (e.g. 5m, 10m)'
|
||||
required: false
|
||||
default: '10m'
|
||||
warp_concurrent:
|
||||
description: 'Pool fill: concurrent warp operations'
|
||||
required: false
|
||||
default: '32'
|
||||
run_decommission:
|
||||
description: 'Run the pool decommission step (3-pool topology only)'
|
||||
type: boolean
|
||||
default: true
|
||||
stop_node_gb:
|
||||
description: 'Heal: stop the outage node when surviving nodes reach N GiB'
|
||||
required: false
|
||||
default: '15'
|
||||
warp_stop_gb:
|
||||
description: 'Heal: stop warp when surviving nodes reach N GiB'
|
||||
required: false
|
||||
default: '40'
|
||||
cleanup_before:
|
||||
description: 'Reset the nodes before the test (DESTROYS existing data/config)'
|
||||
type: boolean
|
||||
@@ -45,17 +41,16 @@ on:
|
||||
description: 'Reset the nodes after the test (DESTROYS test data/config)'
|
||||
type: boolean
|
||||
default: true
|
||||
workflow_run:
|
||||
# Strict shared-environment order: run after tier test succeeds.
|
||||
workflows: ["RustFS Tier Test"]
|
||||
types: [completed]
|
||||
repository_dispatch:
|
||||
# Chain handoff: dispatched when the heal suite finishes.
|
||||
types: [rustfs-chain-pool]
|
||||
|
||||
permissions:
|
||||
contents: read
|
||||
|
||||
# Only one test run at a time: every job mutates the same shared test
|
||||
# Only one test run at a time: the job mutates the same shared test
|
||||
# environment (vm000/vm001/vm002), so concurrent runs must not clobber each
|
||||
# other. Jobs inside a run are chained with needs to serialize them.
|
||||
# other.
|
||||
concurrency:
|
||||
group: rustfs-shared-functional-tests
|
||||
cancel-in-progress: false
|
||||
@@ -70,25 +65,55 @@ env:
|
||||
RUSTFS_API_ENDPOINT: ${{ secrets.RUSTFS_API_ENDPOINT || vars.RUSTFS_API_ENDPOINT || vars.RUSTFS_RC_ENDPOINT }}
|
||||
RUSTFS_NODES: ${{ secrets.RUSTFS_NODES || vars.RUSTFS_NODES }}
|
||||
RUSTFS_SSH_USER: ${{ secrets.RUSTFS_SSH_USER || vars.RUSTFS_SSH_USER }}
|
||||
PF_TESTING_GH_TOKEN: ${{ secrets.PF_TESTING_GH_TOKEN }}
|
||||
# Package used by the nightly run (workflow_dispatch inputs are empty for
|
||||
# workflow_run events), i.e. the latest nightly deb published by nightly-gnu.yml.
|
||||
RUSTFS_NIGHTLY_PACKAGE_URL: ${{ vars.RUSTFS_NIGHTLY_PACKAGE_URL || 'https://dl.rustfs.com/artifacts/rustfs/packages/nightly/rustfs-nightly-latest.deb' }}
|
||||
|
||||
jobs:
|
||||
# Pool expansion: dispatched by the heal suite's chain handoff. Heal
|
||||
# itself lives in rustfs-heal-test.yml and runs exactly once per chain.
|
||||
pool-expansion-test:
|
||||
name: Pool expansion / decommission test
|
||||
runs-on: smoke-testing
|
||||
# Requirement: a failing suite must not fail the workflow; failures
|
||||
# are filed to rustfs/backlog and the chain continues.
|
||||
continue-on-error: true
|
||||
timeout-minutes: 360
|
||||
if: ${{ github.event_name == 'workflow_dispatch' || github.event.workflow_run.conclusion == 'success' }}
|
||||
if: ${{ github.event_name == 'workflow_dispatch' || github.event_name == 'repository_dispatch' }}
|
||||
env:
|
||||
RUSTFS_POOL_ADMIN_ENDPOINT: ${{ secrets.RUSTFS_POOL_ADMIN_ENDPOINT || vars.RUSTFS_POOL_ADMIN_ENDPOINT || 'http://rustfs-node1:9000' }}
|
||||
RUSTFS_POOL_PROXY_ENDPOINT: http://127.0.0.1:19000
|
||||
RUSTFS_POOL_WARP_ENDPOINT: http://127.0.0.1:19000
|
||||
RUSTFS_SHARED_PROXY_ENDPOINT: ${{ secrets.RUSTFS_API_ENDPOINT || vars.RUSTFS_API_ENDPOINT || vars.RUSTFS_RC_ENDPOINT }}
|
||||
RUSTFS_POOL_NODE_ENDPOINTS: ${{ secrets.RUSTFS_POOL_NODE_ENDPOINTS || vars.RUSTFS_POOL_NODE_ENDPOINTS || 'http://rustfs-node1:9000 http://rustfs-node2:9000 http://rustfs-node3:9000' }}
|
||||
steps:
|
||||
- name: Checkout auto-testing scripts
|
||||
uses: actions/checkout@9c091bb21b7c1c1d1991bb908d89e4e9dddfe3e0 # v7
|
||||
with:
|
||||
repository: rustfs/auto-testing
|
||||
ref: main
|
||||
path: auto-testing
|
||||
persist-credentials: false
|
||||
token: ${{ secrets.PF_TESTING_GH_TOKEN }}
|
||||
# auto-testing is private: clone it with the dedicated PF token (not
|
||||
# GITHUB_TOKEN) and retry transient GitHub/network failures.
|
||||
- name: Checkout auto-testing scripts (with retry)
|
||||
env:
|
||||
GH_TOKEN: ${{ secrets.PF_TESTING_GH_TOKEN }}
|
||||
run: |
|
||||
set -euo pipefail
|
||||
rm -rf auto-testing
|
||||
for attempt in 1 2 3 4 5; do
|
||||
if gh repo clone rustfs/auto-testing auto-testing -- --depth 1 --quiet; then
|
||||
echo "auto-testing cloned (attempt ${attempt})"
|
||||
exit 0
|
||||
fi
|
||||
rm -rf auto-testing
|
||||
echo "clone attempt ${attempt} failed; retrying in $((attempt * 15))s" >&2
|
||||
sleep $((attempt * 15))
|
||||
done
|
||||
echo "ERROR: unable to clone rustfs/auto-testing after 5 attempts" >&2
|
||||
exit 1
|
||||
|
||||
- name: Initialize pool test artifacts
|
||||
run: |
|
||||
set -euo pipefail
|
||||
ARTIFACT_DIR="${RUNNER_TEMP}/rustfs-pool-${GITHUB_RUN_ID}-${GITHUB_RUN_ATTEMPT}"
|
||||
mkdir -p "${ARTIFACT_DIR}"
|
||||
echo "POOL_ARTIFACT_DIR=${ARTIFACT_DIR}" >> "${GITHUB_ENV}"
|
||||
|
||||
- name: Show environment
|
||||
run: |
|
||||
@@ -98,15 +123,32 @@ jobs:
|
||||
warp --version || true
|
||||
df -h /data | tail -1
|
||||
|
||||
- name: Reset test environment (before)
|
||||
- name: Cleanup environment (before)
|
||||
if: ${{ inputs.cleanup_before != 'false' }}
|
||||
run: |
|
||||
chmod +x auto-testing/rustfs_pool_expand.sh
|
||||
./auto-testing/rustfs_pool_expand.sh --reset -y
|
||||
set -euo pipefail
|
||||
read -r -a NODES <<< "${RUSTFS_NODES:-vm000 vm001 vm002}"
|
||||
SSH_USER="${RUSTFS_SSH_USER:-azureuser}"
|
||||
for node in "${NODES[@]}"; do
|
||||
ssh -o BatchMode=yes -o ConnectTimeout=10 -o StrictHostKeyChecking=accept-new "${SSH_USER}@${node}" '
|
||||
set -euo pipefail
|
||||
SUDO=""; [ "$(id -u)" -ne 0 ] && SUDO="sudo -n"
|
||||
${SUDO} systemctl stop rustfs 2>/dev/null || true
|
||||
if ${SUDO} dpkg -l rustfs 2>/dev/null | grep -q "^ii"; then
|
||||
${SUDO} dpkg -P rustfs
|
||||
fi
|
||||
for i in 1 2 3 4; do ${SUDO} rm -rf /data/rustfs${i}/mnmd; done
|
||||
${SUDO} rm -rf /var/log/rustfs /var/lib/rustfs/kms /var/lib/rustfs/kms-backup
|
||||
'
|
||||
done
|
||||
|
||||
- name: Install RustFS package & start first pool
|
||||
- name: Install RustFS package & start cluster
|
||||
run: |
|
||||
ARGS=(--steps "1,2,3" -y --endpoint "${{ env.RUSTFS_API_ENDPOINT }}")
|
||||
ARGS=(--steps "1,2,3" -y \
|
||||
--admin-endpoint "${RUSTFS_POOL_ADMIN_ENDPOINT}" \
|
||||
--warp-endpoint "${RUSTFS_POOL_WARP_ENDPOINT}" \
|
||||
--node-endpoints "${RUSTFS_POOL_NODE_ENDPOINTS}" \
|
||||
--log-file "${POOL_ARTIFACT_DIR}/pool-test.log")
|
||||
if [ -n "${{ inputs.package_url }}" ]; then
|
||||
ARGS+=(--package-url "${{ inputs.package_url }}")
|
||||
elif [ -n "${{ inputs.rustfs_version }}" ]; then
|
||||
@@ -118,7 +160,11 @@ jobs:
|
||||
|
||||
- name: Preflight checks
|
||||
run: |
|
||||
ARGS=(--preflight --endpoint "${{ env.RUSTFS_API_ENDPOINT }}")
|
||||
ARGS=(--preflight \
|
||||
--admin-endpoint "${RUSTFS_POOL_ADMIN_ENDPOINT}" \
|
||||
--warp-endpoint "${RUSTFS_POOL_WARP_ENDPOINT}" \
|
||||
--node-endpoints "${RUSTFS_POOL_NODE_ENDPOINTS}" \
|
||||
--log-file "${POOL_ARTIFACT_DIR}/pool-test.log")
|
||||
if [ -n "${{ inputs.package_url }}" ]; then
|
||||
ARGS+=(--package-url "${{ inputs.package_url }}")
|
||||
elif [ -n "${{ inputs.rustfs_version }}" ]; then
|
||||
@@ -128,6 +174,60 @@ jobs:
|
||||
fi
|
||||
./auto-testing/rustfs_pool_expand.sh "${ARGS[@]}"
|
||||
|
||||
- name: Reset dedicated pool proxy
|
||||
run: |
|
||||
set -euo pipefail
|
||||
RUSTFS_POOL_NGINX_CONFIG_PATH=/etc/nginx/conf.d/rustfs-pool-test.conf \
|
||||
RUSTFS_POOL_NGINX_LISTEN="${RUSTFS_POOL_PROXY_ENDPOINT#http://}" \
|
||||
RUSTFS_POOL_NGINX_ACCESS_LOG=/var/log/nginx/rustfs-pool-test-access.log \
|
||||
RUSTFS_POOL_NGINX_ERROR_LOG=/var/log/nginx/rustfs-pool-test-error.log \
|
||||
./auto-testing/rustfs_pool_nginx_stage.sh cleanup
|
||||
|
||||
- name: Capture pool test baseline
|
||||
run: |
|
||||
set -uo pipefail
|
||||
BASELINE_FILE="${POOL_ARTIFACT_DIR}/pool-baseline.log"
|
||||
read -r -a NODES <<< "${RUSTFS_NODES:-vm000 vm001 vm002}"
|
||||
read -r -a DIRECT_ENDPOINTS <<< "${RUSTFS_POOL_NODE_ENDPOINTS}"
|
||||
SSH_USER="${RUSTFS_SSH_USER:-azureuser}"
|
||||
failed=0
|
||||
: > "${BASELINE_FILE}"
|
||||
|
||||
if [ "${#DIRECT_ENDPOINTS[@]}" -lt "${#NODES[@]}" ]; then
|
||||
echo "not enough direct endpoints for the configured nodes" | tee -a "${BASELINE_FILE}" >&2
|
||||
exit 1
|
||||
fi
|
||||
|
||||
for index in "${!NODES[@]}"; do
|
||||
node="${NODES[$index]}"
|
||||
endpoint="${DIRECT_ENDPOINTS[$index]}"
|
||||
body_file="${POOL_ARTIFACT_DIR}/ready-baseline-$((index + 1)).body"
|
||||
{
|
||||
echo "--- node=${node} endpoint=${endpoint} ---"
|
||||
if ! ssh -o BatchMode=yes -o ConnectTimeout=10 -o StrictHostKeyChecking=accept-new \
|
||||
"${SSH_USER}@${node}" '
|
||||
set -euo pipefail
|
||||
SUDO=""; [ "$(id -u)" -ne 0 ] && SUDO="sudo -n"
|
||||
echo "--- rustfs version ---"
|
||||
rustfs --version
|
||||
echo "--- systemd state ---"
|
||||
${SUDO} systemctl show rustfs --no-pager \
|
||||
--property=ActiveState,SubState,Result,ExecMainPID,ExecMainStartTimestamp,NRestarts
|
||||
'; then
|
||||
echo "baseline collection failed for ${node}"
|
||||
failed=1
|
||||
fi
|
||||
curl -sS --connect-timeout 5 --max-time 15 -o "${body_file}" \
|
||||
-w "baseline_ready=${endpoint} http=%{http_code} connect=%{time_connect} ttfb=%{time_starttransfer} total=%{time_total}\n" \
|
||||
"${endpoint%/}/health/ready" || true
|
||||
echo "--- readiness body ---"
|
||||
cat "${body_file}" 2>/dev/null || true
|
||||
echo
|
||||
} >> "${BASELINE_FILE}" 2>&1
|
||||
done
|
||||
|
||||
[ "${failed}" -eq 0 ] || exit 1
|
||||
|
||||
- name: Run pool expansion & decommission test
|
||||
id: pool_test
|
||||
run: |
|
||||
@@ -140,10 +240,16 @@ jobs:
|
||||
fi
|
||||
fi
|
||||
ARGS=(--steps "$STEPS" --with-warp -y \
|
||||
--endpoint "${{ env.RUSTFS_API_ENDPOINT }}" \
|
||||
--admin-endpoint "${RUSTFS_POOL_ADMIN_ENDPOINT}" \
|
||||
--warp-endpoint "${RUSTFS_POOL_WARP_ENDPOINT}" \
|
||||
--node-endpoints "${RUSTFS_POOL_NODE_ENDPOINTS}" \
|
||||
--storage-threshold "${{ inputs.storage_threshold || '50' }}" \
|
||||
--warp-duration "${{ inputs.warp_duration || '10m' }}" \
|
||||
--log-file /tmp/rustfs-pool-test.log)
|
||||
--warp-concurrent "${{ inputs.warp_concurrent || '32' }}" \
|
||||
--log-file "${POOL_ARTIFACT_DIR}/pool-test.log")
|
||||
if [ -n "${RUSTFS_POOL_PROXY_ENDPOINT}" ]; then
|
||||
ARGS+=(--proxy-endpoint "${RUSTFS_POOL_PROXY_ENDPOINT}")
|
||||
fi
|
||||
if [ -n "${{ inputs.package_url }}" ]; then
|
||||
ARGS+=(--package-url "${{ inputs.package_url }}")
|
||||
elif [ -n "${{ inputs.rustfs_version }}" ]; then
|
||||
@@ -151,22 +257,391 @@ jobs:
|
||||
else
|
||||
ARGS+=(--package-url "${{ env.RUSTFS_NIGHTLY_PACKAGE_URL }}")
|
||||
fi
|
||||
./auto-testing/rustfs_pool_expand.sh "${ARGS[@]}"
|
||||
RUSTFS_WARP_LOG_FILE="${POOL_ARTIFACT_DIR}/warp.log" \
|
||||
RUSTFS_PROXY_STAGE_HOOK=./auto-testing/rustfs_pool_nginx_stage.sh \
|
||||
RUSTFS_POOL_NGINX_CONFIG_PATH=/etc/nginx/conf.d/rustfs-pool-test.conf \
|
||||
RUSTFS_POOL_NGINX_LISTEN="${RUSTFS_POOL_PROXY_ENDPOINT#http://}" \
|
||||
RUSTFS_POOL_NGINX_ACCESS_LOG=/var/log/nginx/rustfs-pool-test-access.log \
|
||||
RUSTFS_POOL_NGINX_ERROR_LOG=/var/log/nginx/rustfs-pool-test-error.log \
|
||||
./auto-testing/rustfs_pool_expand.sh "${ARGS[@]}"
|
||||
|
||||
- name: Collect pool test diagnostics
|
||||
if: always()
|
||||
run: |
|
||||
set -uo pipefail
|
||||
ARTIFACT_DIR="${POOL_ARTIFACT_DIR:-${RUNNER_TEMP}/rustfs-pool-${GITHUB_RUN_ID}-${GITHUB_RUN_ATTEMPT}}"
|
||||
mkdir -p "${ARTIFACT_DIR}"
|
||||
echo "POOL_ARTIFACT_DIR=${ARTIFACT_DIR}" >> "${GITHUB_ENV}"
|
||||
|
||||
redact() {
|
||||
sed -E \
|
||||
-e 's/(RUSTFS_(ACCESS_KEY|SECRET_KEY)=).*/\1[REDACTED]/Ig' \
|
||||
-e 's/(Authorization:).*/\1 [REDACTED]/Ig' \
|
||||
-e 's/(X-Amz-Signature=)[^&[:space:]]+/\1[REDACTED]/Ig' \
|
||||
-e 's/(proxy_set_header[[:space:]]+Authorization[[:space:]]+).*/\1[REDACTED];/Ig' \
|
||||
-e 's/^.*(password|secret|token).*/[REDACTED SENSITIVE LINE]/Ig'
|
||||
}
|
||||
|
||||
if [ "$(id -u)" -eq 0 ]; then
|
||||
SUDO=()
|
||||
else
|
||||
SUDO=(sudo -n)
|
||||
fi
|
||||
|
||||
{
|
||||
echo "captured_at=$(date -u +%Y-%m-%dT%H:%M:%SZ)"
|
||||
echo "run_id=${GITHUB_RUN_ID}"
|
||||
echo "run_attempt=${GITHUB_RUN_ATTEMPT}"
|
||||
if command -v nginx >/dev/null 2>&1; then
|
||||
"${SUDO[@]}" nginx -T 2>&1 || echo "nginx -T failed"
|
||||
else
|
||||
echo "nginx is not installed on the runner"
|
||||
fi
|
||||
} | redact > "${ARTIFACT_DIR}/nginx-config-redacted.txt"
|
||||
|
||||
for log_path in \
|
||||
/var/log/nginx/access.log \
|
||||
/var/log/nginx/error.log \
|
||||
/var/log/nginx/rustfs-pool-test-access.log \
|
||||
/var/log/nginx/rustfs-pool-test-error.log; do
|
||||
log_name="$(basename "${log_path}")"
|
||||
if "${SUDO[@]}" test -r "${log_path}" 2>/dev/null; then
|
||||
"${SUDO[@]}" cat "${log_path}" 2>&1 | redact \
|
||||
> "${ARTIFACT_DIR}/nginx-${log_name%.log}-redacted.log"
|
||||
else
|
||||
echo "unavailable: ${log_path}" > "${ARTIFACT_DIR}/nginx-${log_name%.log}-redacted.log"
|
||||
fi
|
||||
done
|
||||
"${SUDO[@]}" journalctl -u nginx --no-pager -n 5000 2>&1 | redact \
|
||||
> "${ARTIFACT_DIR}/nginx-journal-redacted.log" || true
|
||||
|
||||
read -r -a NODES <<< "${RUSTFS_NODES:-vm000 vm001 vm002}"
|
||||
SSH_USER="${RUSTFS_SSH_USER:-azureuser}"
|
||||
for node in "${NODES[@]}"; do
|
||||
safe_node="${node//[^A-Za-z0-9_.-]/_}"
|
||||
{
|
||||
if ! ssh -o BatchMode=yes -o ConnectTimeout=10 -o StrictHostKeyChecking=accept-new \
|
||||
"${SSH_USER}@${node}" '
|
||||
SUDO=""; [ "$(id -u)" -ne 0 ] && SUDO="sudo -n"
|
||||
echo "--- rustfs version ---"
|
||||
rustfs --version 2>&1 || true
|
||||
echo "--- systemd state ---"
|
||||
${SUDO} systemctl show rustfs --no-pager \
|
||||
--property=ActiveState,SubState,Result,ExecMainPID,ExecMainStartTimestamp,NRestarts 2>&1 || true
|
||||
echo "--- rustfs journal ---"
|
||||
${SUDO} journalctl -u rustfs --no-pager -n 10000 2>&1 || true
|
||||
echo "--- rustfs file logs ---"
|
||||
if ${SUDO} test -d /var/log/rustfs; then
|
||||
${SUDO} find /var/log/rustfs -maxdepth 2 -type f -print 2>/dev/null | while IFS= read -r file; do
|
||||
echo "--- ${file} (last 5000 lines) ---"
|
||||
${SUDO} tail -n 5000 "${file}" 2>&1 || true
|
||||
done
|
||||
else
|
||||
echo "/var/log/rustfs is unavailable"
|
||||
fi
|
||||
'; then
|
||||
echo "SSH diagnostics failed for ${node}"
|
||||
fi
|
||||
} 2>&1 | redact > "${ARTIFACT_DIR}/${safe_node}-rustfs-redacted.log"
|
||||
done
|
||||
|
||||
: > "${ARTIFACT_DIR}/endpoint-ready-probes.log"
|
||||
read -r -a DIRECT_ENDPOINTS <<< "${RUSTFS_POOL_NODE_ENDPOINTS}"
|
||||
probe_index=0
|
||||
for endpoint in "${DIRECT_ENDPOINTS[@]}"; do
|
||||
probe_index=$((probe_index + 1))
|
||||
curl -sS --connect-timeout 5 --max-time 15 -o "${ARTIFACT_DIR}/ready-direct-${probe_index}.body" \
|
||||
-w "direct[${probe_index}]=${endpoint} http=%{http_code} connect=%{time_connect} ttfb=%{time_starttransfer} total=%{time_total}\n" \
|
||||
"${endpoint%/}/health/ready" >> "${ARTIFACT_DIR}/endpoint-ready-probes.log" 2>&1 || true
|
||||
done
|
||||
if [ -n "${RUSTFS_POOL_PROXY_ENDPOINT}" ]; then
|
||||
curl -sS --connect-timeout 5 --max-time 15 -o "${ARTIFACT_DIR}/ready-proxy.body" \
|
||||
-w "proxy=${RUSTFS_POOL_PROXY_ENDPOINT} http=%{http_code} connect=%{time_connect} ttfb=%{time_starttransfer} total=%{time_total}\n" \
|
||||
"${RUSTFS_POOL_PROXY_ENDPOINT%/}/health/ready" >> "${ARTIFACT_DIR}/endpoint-ready-probes.log" 2>&1 || true
|
||||
fi
|
||||
if [ -n "${RUSTFS_SHARED_PROXY_ENDPOINT}" ]; then
|
||||
curl -sS --connect-timeout 5 --max-time 15 -o "${ARTIFACT_DIR}/ready-shared-proxy.body" \
|
||||
-w "shared_proxy=${RUSTFS_SHARED_PROXY_ENDPOINT} http=%{http_code} connect=%{time_connect} ttfb=%{time_starttransfer} total=%{time_total}\n" \
|
||||
"${RUSTFS_SHARED_PROXY_ENDPOINT%/}/health/ready" >> "${ARTIFACT_DIR}/endpoint-ready-probes.log" 2>&1 || true
|
||||
fi
|
||||
|
||||
- name: Generate report
|
||||
if: always()
|
||||
run: |
|
||||
set -euo pipefail
|
||||
LOG_FILE="${POOL_ARTIFACT_DIR}/pool-test.log"
|
||||
REPORT_FILE="${POOL_ARTIFACT_DIR}/pool-report.md"
|
||||
PACKAGE_URL='${{ inputs.package_url }}'
|
||||
RUSTFS_VERSION='${{ inputs.rustfs_version }}'
|
||||
if [ -n "${PACKAGE_URL}" ]; then
|
||||
PACKAGE_SOURCE="${PACKAGE_URL}"
|
||||
elif [ -n "${RUSTFS_VERSION}" ]; then
|
||||
PACKAGE_SOURCE="version ${RUSTFS_VERSION}"
|
||||
else
|
||||
PACKAGE_SOURCE="${RUSTFS_NIGHTLY_PACKAGE_URL}"
|
||||
fi
|
||||
{
|
||||
echo "# RustFS pool expansion test report"
|
||||
echo ""
|
||||
echo "- Run: ${{ github.server_url }}/${{ github.repository }}/actions/runs/${{ github.run_id }}"
|
||||
echo "- Trigger: ${{ github.event_name }}"
|
||||
echo "- Package: ${PACKAGE_SOURCE}"
|
||||
echo "- Warp concurrent: ${{ inputs.warp_concurrent || '32' }}"
|
||||
echo "- Test Step Outcome: ${{ steps.pool_test.outcome }}"
|
||||
echo ""
|
||||
echo "## Log tail"
|
||||
echo '```text'
|
||||
tail -n 200 "${LOG_FILE}" || true
|
||||
echo '```'
|
||||
} | tee "${REPORT_FILE}"
|
||||
cat "${REPORT_FILE}" >> "${GITHUB_STEP_SUMMARY}"
|
||||
|
||||
- name: Validate pool diagnostic completeness
|
||||
if: always()
|
||||
run: |
|
||||
set -euo pipefail
|
||||
failed=0
|
||||
require_nonempty() {
|
||||
if [ ! -s "$1" ]; then
|
||||
echo "required diagnostic is missing or empty: $1" >&2
|
||||
failed=1
|
||||
fi
|
||||
}
|
||||
require_available() {
|
||||
if [ ! -e "$1" ]; then
|
||||
echo "required diagnostic is missing: $1" >&2
|
||||
failed=1
|
||||
elif grep -Fq 'unavailable:' "$1" 2>/dev/null; then
|
||||
echo "required diagnostic could not be collected: $1" >&2
|
||||
failed=1
|
||||
fi
|
||||
}
|
||||
|
||||
require_nonempty "${POOL_ARTIFACT_DIR}/pool-test.log"
|
||||
require_nonempty "${POOL_ARTIFACT_DIR}/warp.log"
|
||||
require_nonempty "${POOL_ARTIFACT_DIR}/pool-report.md"
|
||||
require_nonempty "${POOL_ARTIFACT_DIR}/pool-baseline.log"
|
||||
require_nonempty "${POOL_ARTIFACT_DIR}/nginx-config-redacted.txt"
|
||||
require_nonempty "${POOL_ARTIFACT_DIR}/nginx-rustfs-pool-test-access-redacted.log"
|
||||
require_available "${POOL_ARTIFACT_DIR}/nginx-rustfs-pool-test-access-redacted.log"
|
||||
require_available "${POOL_ARTIFACT_DIR}/nginx-rustfs-pool-test-error-redacted.log"
|
||||
require_nonempty "${POOL_ARTIFACT_DIR}/endpoint-ready-probes.log"
|
||||
|
||||
read -r -a NODES <<< "${RUSTFS_NODES:-vm000 vm001 vm002}"
|
||||
if grep -Fq 'baseline collection failed' "${POOL_ARTIFACT_DIR}/pool-baseline.log" 2>/dev/null; then
|
||||
echo "one or more node baselines could not be collected" >&2
|
||||
failed=1
|
||||
fi
|
||||
for node in "${NODES[@]}"; do
|
||||
safe_node="${node//[^A-Za-z0-9_.-]/_}"
|
||||
node_log="${POOL_ARTIFACT_DIR}/${safe_node}-rustfs-redacted.log"
|
||||
require_nonempty "${node_log}"
|
||||
if grep -Fq "SSH diagnostics failed for ${node}" "${node_log}" 2>/dev/null; then
|
||||
echo "node diagnostics failed: ${node_log}" >&2
|
||||
failed=1
|
||||
fi
|
||||
if ! grep -Eq '^rustfs @' "${node_log}" 2>/dev/null \
|
||||
|| ! grep -Eq '^NRestarts=[0-9]+$' "${node_log}" 2>/dev/null; then
|
||||
echo "node version or restart evidence is incomplete: ${node_log}" >&2
|
||||
failed=1
|
||||
elif grep -Eq '^NRestarts=[1-9][0-9]*$' "${node_log}"; then
|
||||
echo "RustFS restarted unexpectedly during the run: ${node_log}" >&2
|
||||
failed=1
|
||||
fi
|
||||
done
|
||||
|
||||
if ! grep -Fq "upstream_status=\"\$upstream_status\"" \
|
||||
"${POOL_ARTIFACT_DIR}/nginx-config-redacted.txt"; then
|
||||
echo "Nginx config does not expose upstream status fields" >&2
|
||||
failed=1
|
||||
fi
|
||||
if ! grep -Eq '^proxy=.* http=200([[:space:]]|$)' "${POOL_ARTIFACT_DIR}/endpoint-ready-probes.log"; then
|
||||
echo "dedicated proxy readiness probe did not return HTTP 200" >&2
|
||||
failed=1
|
||||
fi
|
||||
if grep -Eq 'status=50(2|4)|upstream_status="[^"]*50(2|4)' \
|
||||
"${POOL_ARTIFACT_DIR}/nginx-rustfs-pool-test-access-redacted.log"; then
|
||||
echo "dedicated proxy access log contains a 502/504 response" >&2
|
||||
failed=1
|
||||
fi
|
||||
if grep -Eiq 'upstream prematurely closed connection|upstream timed out|(connect\(\)|recv\(\)|send\(\)) failed.*upstream|connection reset by peer.*upstream' \
|
||||
"${POOL_ARTIFACT_DIR}/nginx-rustfs-pool-test-error-redacted.log"; then
|
||||
echo "dedicated proxy error log contains an upstream timeout or connection failure" >&2
|
||||
failed=1
|
||||
fi
|
||||
[ "${failed}" -eq 0 ] || exit 1
|
||||
|
||||
- name: Upload functional report to dashboard
|
||||
if: always()
|
||||
continue-on-error: true
|
||||
env:
|
||||
GH_TOKEN: ${{ env.PF_TESTING_GH_TOKEN }}
|
||||
SUITE: pool
|
||||
run: |
|
||||
set -euo pipefail
|
||||
REPORT_FILE="${POOL_ARTIFACT_DIR}/pool-report.md"
|
||||
if [ -z "${GH_TOKEN:-}" ]; then
|
||||
echo "PF_TESTING_GH_TOKEN is not configured; skipping dashboard upload"
|
||||
exit 0
|
||||
fi
|
||||
DATE="$(date -u +%Y-%m-%d)"
|
||||
REPORT_PATH="functional-reports/${SUITE}/${DATE}.md"
|
||||
CONTENT="$(python3 -c 'import base64,sys;print(base64.b64encode(open(sys.argv[1],"rb").read()).decode())' "${REPORT_FILE}")"
|
||||
SHA="$(gh api "repos/rustfs/dashboard/contents/${REPORT_PATH}" -q '.sha' 2>/dev/null || true)"
|
||||
if [ -n "${SHA}" ]; then
|
||||
jq -n --arg msg "report(${SUITE}): ${DATE}" --arg content "${CONTENT}" --arg sha "${SHA}" \
|
||||
'{message:$msg, content:$content, sha:$sha}' \
|
||||
| gh api --method PUT "repos/rustfs/dashboard/contents/${REPORT_PATH}" --input - >/dev/null
|
||||
else
|
||||
jq -n --arg msg "report(${SUITE}): ${DATE}" --arg content "${CONTENT}" \
|
||||
'{message:$msg, content:$content}' \
|
||||
| gh api --method PUT "repos/rustfs/dashboard/contents/${REPORT_PATH}" --input - >/dev/null
|
||||
fi
|
||||
|
||||
- name: File failure issue in rustfs/backlog
|
||||
if: ${{ always() && (failure() || steps.pool_test.outcome == 'failure' || steps.pool_test.outcome == 'cancelled') }}
|
||||
continue-on-error: true
|
||||
env:
|
||||
GH_TOKEN: ${{ secrets.PF_TESTING_GH_TOKEN }}
|
||||
SUITE: 'pool'
|
||||
SUITE_LABEL: 'Pool expansion'
|
||||
RUN_URL: ${{ github.server_url }}/${{ github.repository }}/actions/runs/${{ github.run_id }}
|
||||
REPORT_FILE: '${{ env.POOL_ARTIFACT_DIR }}/pool-report.md'
|
||||
LOG_FILE: '${{ env.POOL_ARTIFACT_DIR }}/pool-test.log'
|
||||
run: |
|
||||
set -euo pipefail
|
||||
if [ -z "${GH_TOKEN:-}" ]; then
|
||||
echo "PF_TESTING_GH_TOKEN is not configured; skipping backlog issue"
|
||||
exit 0
|
||||
fi
|
||||
TITLE="[functional][${SUITE}] ${SUITE_LABEL} suite failed (run ${GITHUB_RUN_ID})"
|
||||
EXISTING="$(gh issue list -R rustfs/backlog --state all \
|
||||
--search "in:title \"run ${GITHUB_RUN_ID}\"" \
|
||||
--json number --jq '.[].number' || true)"
|
||||
if [ -n "${EXISTING}" ]; then
|
||||
echo "backlog issue already exists for run ${GITHUB_RUN_ID}; skipping"
|
||||
exit 0
|
||||
fi
|
||||
redact() {
|
||||
sed -E \
|
||||
-e 's/(RUSTFS_(ACCESS_KEY|SECRET_KEY)[=: ]+)[^[:space:]]+/\1[REDACTED]/Ig' \
|
||||
-e 's/(Authorization:).*/\1 [REDACTED]/Ig' \
|
||||
-e 's/(X-Amz-Signature=)[^&[:space:]]+/\1[REDACTED]/Ig' \
|
||||
-e 's/^.*(password|secret|token)[=: ].*/[REDACTED SENSITIVE LINE]/Ig'
|
||||
}
|
||||
BODY_FILE="$(mktemp)"
|
||||
{
|
||||
echo "The **${SUITE_LABEL}** functional suite failed."
|
||||
echo ""
|
||||
echo "- Suite: \`${SUITE}\`"
|
||||
echo "- Run: ${RUN_URL}"
|
||||
echo "- Trigger: ${GITHUB_EVENT_NAME}"
|
||||
echo "- Date: $(date -u +%Y-%m-%d)"
|
||||
echo ""
|
||||
echo "## Report (errors and symptoms)"
|
||||
echo ""
|
||||
if [ -s "${REPORT_FILE}" ]; then
|
||||
redact < "${REPORT_FILE}"
|
||||
elif [ -s "${LOG_FILE:-}" ]; then
|
||||
echo "(report file missing; log tail below)"
|
||||
echo ""
|
||||
tail -n 200 "${LOG_FILE}" | redact
|
||||
else
|
||||
echo "(no report or log file was produced)"
|
||||
fi
|
||||
} | head -c 55000 > "${BODY_FILE}"
|
||||
gh label create functional-test -R rustfs/backlog --color d73a4a 2>/dev/null || true
|
||||
if ! gh issue create -R rustfs/backlog --title "${TITLE}" \
|
||||
--body-file "${BODY_FILE}" --label functional-test; then
|
||||
gh issue create -R rustfs/backlog --title "${TITLE}" --body-file "${BODY_FILE}"
|
||||
fi
|
||||
echo "filed backlog issue for suite ${SUITE}"
|
||||
|
||||
- name: Upload test logs
|
||||
if: always()
|
||||
uses: actions/upload-artifact@b7c566a772e6b6bfb58ed0dc250532a479d7789f # v6
|
||||
with:
|
||||
name: rustfs-pool-test-${{ github.run_id }}
|
||||
path: |
|
||||
/tmp/rustfs-pool-test*.log
|
||||
/tmp/rustfs-warp.*.log
|
||||
name: rustfs-pool-test-${{ github.run_id }}-${{ github.run_attempt }}
|
||||
path: ${{ runner.temp }}/rustfs-pool-${{ github.run_id }}-${{ github.run_attempt }}
|
||||
if-no-files-found: warn
|
||||
|
||||
- name: Reset test environment (after)
|
||||
- name: Restore dedicated pool proxy
|
||||
if: always()
|
||||
run: |
|
||||
set -euo pipefail
|
||||
RUSTFS_POOL_NGINX_CONFIG_PATH=/etc/nginx/conf.d/rustfs-pool-test.conf \
|
||||
RUSTFS_POOL_NGINX_LISTEN="${RUSTFS_POOL_PROXY_ENDPOINT#http://}" \
|
||||
RUSTFS_POOL_NGINX_ACCESS_LOG=/var/log/nginx/rustfs-pool-test-access.log \
|
||||
RUSTFS_POOL_NGINX_ERROR_LOG=/var/log/nginx/rustfs-pool-test-error.log \
|
||||
./auto-testing/rustfs_pool_nginx_stage.sh cleanup
|
||||
|
||||
- name: Cleanup environment (after)
|
||||
if: ${{ always() && inputs.cleanup_after != 'false' }}
|
||||
run: |
|
||||
./auto-testing/rustfs_pool_expand.sh --reset -y
|
||||
set -euo pipefail
|
||||
read -r -a NODES <<< "${RUSTFS_NODES:-vm000 vm001 vm002}"
|
||||
SSH_USER="${RUSTFS_SSH_USER:-azureuser}"
|
||||
for node in "${NODES[@]}"; do
|
||||
ssh -o BatchMode=yes -o ConnectTimeout=10 -o StrictHostKeyChecking=accept-new "${SSH_USER}@${node}" '
|
||||
set -euo pipefail
|
||||
SUDO=""; [ "$(id -u)" -ne 0 ] && SUDO="sudo -n"
|
||||
${SUDO} systemctl stop rustfs 2>/dev/null || true
|
||||
if ${SUDO} dpkg -l rustfs 2>/dev/null | grep -q "^ii"; then
|
||||
${SUDO} dpkg -P rustfs
|
||||
fi
|
||||
for i in 1 2 3 4; do ${SUDO} rm -rf /data/rustfs${i}/mnmd; done
|
||||
${SUDO} rm -rf /var/log/rustfs /var/lib/rustfs/kms /var/lib/rustfs/kms-backup
|
||||
'
|
||||
done
|
||||
|
||||
- name: "Continue functional chain (next: Security)"
|
||||
# Only chain-triggered runs forward to the next suite; standalone
|
||||
# workflow_dispatch runs stop after their own cleanup. A failed
|
||||
# handoff must never pass silently: it retries, then files an alert
|
||||
# issue in rustfs/backlog so a stalled chain is visible.
|
||||
if: ${{ always() && github.event_name == 'repository_dispatch' }}
|
||||
continue-on-error: true
|
||||
env:
|
||||
GH_TOKEN: ${{ secrets.PF_TESTING_GH_TOKEN }}
|
||||
run: |
|
||||
set -uo pipefail
|
||||
if [ -z "${{GH_TOKEN:-}}" ]; then
|
||||
echo "PF_TESTING_GH_TOKEN is not configured; cannot dispatch the next suite" >&2
|
||||
exit 1
|
||||
fi
|
||||
DISPATCHED=0
|
||||
for attempt in 1 2 3; do
|
||||
if gh api --method POST repos/rustfs/rustfs/dispatches \
|
||||
-f event_type='rustfs-chain-security' \
|
||||
-F 'client_payload[from_suite]=pool'; then
|
||||
echo "dispatched next suite Security (attempt ${{attempt}})"
|
||||
DISPATCHED=1
|
||||
break
|
||||
fi
|
||||
echo "dispatch attempt ${{attempt}} failed; retrying in ${{attempt}}0s" >&2
|
||||
sleep "${{attempt}}0"
|
||||
done
|
||||
if [ "${{DISPATCHED:-0}}" -ne 1 ]; then
|
||||
echo "ERROR: functional chain stalled: could not dispatch Security after 3 attempts" >&2
|
||||
TITLE="[functional][chain] stalled after pool (run ${{GITHUB_RUN_ID}})"
|
||||
BODY_FILE="$(mktemp)"
|
||||
{
|
||||
echo "The functional chain could not hand off from **pool** to **Security** after 3 attempts."
|
||||
echo ""
|
||||
echo "- Failed suite job: ${{GITHUB_SERVER_URL}}/${{GITHUB_REPOSITORY}}/actions/runs/${{GITHUB_RUN_ID}}"
|
||||
echo "- Expected next event: `rustfs-chain-security`"
|
||||
echo "- Likely cause: PF_TESTING_GH_TOKEN lacks contents:write on rustfs/rustfs, or the GitHub API was unavailable."
|
||||
echo "- Recovery: re-dispatch manually with"
|
||||
echo " ```"
|
||||
echo " gh api --method POST repos/rustfs/rustfs/dispatches -f event_type='rustfs-chain-security'"
|
||||
echo " ```"
|
||||
} > "${{BODY_FILE}}"
|
||||
gh issue create -R rustfs/backlog --title "${{TITLE}}" \
|
||||
--body-file "${{BODY_FILE}}" --label functional-test \
|
||||
|| gh issue create -R rustfs/backlog --title "${{TITLE}}" --body-file "${{BODY_FILE}}" \
|
||||
|| echo "could not file the stall alert issue either; check the token" >&2
|
||||
exit 1
|
||||
fi
|
||||
|
||||
- name: Notify on failure
|
||||
if: failure()
|
||||
@@ -174,82 +649,3 @@ jobs:
|
||||
echo "RustFS pool expansion test failed"
|
||||
echo "Package source: ${{ inputs.package_url || inputs.rustfs_version || 'nightly (R2 latest)' }}"
|
||||
echo "See the uploaded log artifact for details."
|
||||
|
||||
# Heal regression runs after the pool test regardless of its outcome: a pool
|
||||
# failure must be reported (it makes the run red) but must not block heal.
|
||||
heal-test:
|
||||
name: Heal test (after pool test)
|
||||
runs-on: smoke-testing
|
||||
timeout-minutes: 480
|
||||
needs: pool-expansion-test
|
||||
if: ${{ always() && (github.event_name == 'workflow_dispatch' || github.event.workflow_run.conclusion == 'success') }}
|
||||
steps:
|
||||
- name: Checkout auto-testing scripts
|
||||
uses: actions/checkout@9c091bb21b7c1c1d1991bb908d89e4e9dddfe3e0 # v7
|
||||
with:
|
||||
repository: rustfs/auto-testing
|
||||
ref: main
|
||||
path: auto-testing
|
||||
persist-credentials: false
|
||||
token: ${{ secrets.PF_TESTING_GH_TOKEN }}
|
||||
|
||||
- name: Reset test environment (before)
|
||||
if: ${{ inputs.cleanup_before != 'false' }}
|
||||
run: |
|
||||
chmod +x auto-testing/rustfs_heal_test.sh
|
||||
./auto-testing/rustfs_heal_test.sh --reset -y
|
||||
|
||||
- name: Install RustFS package & start cluster
|
||||
run: |
|
||||
ARGS=(--steps "1,2" -y --endpoint "${{ env.RUSTFS_API_ENDPOINT }}")
|
||||
if [ -n "${{ inputs.package_url }}" ]; then
|
||||
ARGS+=(--package-url "${{ inputs.package_url }}")
|
||||
else
|
||||
ARGS+=(--package-url "${{ env.RUSTFS_NIGHTLY_PACKAGE_URL }}")
|
||||
fi
|
||||
./auto-testing/rustfs_heal_test.sh "${ARGS[@]}"
|
||||
|
||||
- name: Preflight checks
|
||||
run: |
|
||||
ARGS=(--preflight --endpoint "${{ env.RUSTFS_API_ENDPOINT }}")
|
||||
if [ -n "${{ inputs.package_url }}" ]; then
|
||||
ARGS+=(--package-url "${{ inputs.package_url }}")
|
||||
else
|
||||
ARGS+=(--package-url "${{ env.RUSTFS_NIGHTLY_PACKAGE_URL }}")
|
||||
fi
|
||||
./auto-testing/rustfs_heal_test.sh "${ARGS[@]}"
|
||||
|
||||
- name: Run heal test (write -> outage -> heal -> verify)
|
||||
run: |
|
||||
ARGS=(--steps "3,4,5,6,7" -y \
|
||||
--endpoint "${{ env.RUSTFS_API_ENDPOINT }}" \
|
||||
--stop-node-gb "${{ inputs.stop_node_gb || '15' }}" \
|
||||
--warp-stop-gb "${{ inputs.warp_stop_gb || '40' }}" \
|
||||
--log-file /tmp/rustfs-heal-test.log)
|
||||
if [ -n "${{ inputs.package_url }}" ]; then
|
||||
ARGS+=(--package-url "${{ inputs.package_url }}")
|
||||
else
|
||||
ARGS+=(--package-url "${{ env.RUSTFS_NIGHTLY_PACKAGE_URL }}")
|
||||
fi
|
||||
./auto-testing/rustfs_heal_test.sh "${ARGS[@]}"
|
||||
|
||||
- name: Upload test logs
|
||||
if: always()
|
||||
uses: actions/upload-artifact@b7c566a772e6b6bfb58ed0dc250532a479d7789f # v6
|
||||
with:
|
||||
name: rustfs-heal-test-${{ github.run_id }}
|
||||
path: |
|
||||
/tmp/rustfs-heal-test.log
|
||||
/tmp/rustfs-warp.*.log
|
||||
if-no-files-found: warn
|
||||
|
||||
- name: Reset test environment (after)
|
||||
if: ${{ always() && inputs.cleanup_after != 'false' }}
|
||||
run: |
|
||||
./auto-testing/rustfs_heal_test.sh --reset -y
|
||||
|
||||
- name: Notify on failure
|
||||
if: failure()
|
||||
run: |
|
||||
echo "RustFS heal test failed"
|
||||
echo "See the uploaded log artifact for details."
|
||||
|
||||
@@ -11,10 +11,9 @@ on:
|
||||
description: 'Direct .deb URL (nightly/R2/dev). Overrides rustfs_version.'
|
||||
required: false
|
||||
type: string
|
||||
workflow_run:
|
||||
# Run after the nightly build completes; the nightly deb is what the test installs.
|
||||
workflows: ["Nightly GNU Build"]
|
||||
types: [completed]
|
||||
repository_dispatch:
|
||||
# Chain handoff: dispatched when the upgrade suite finishes.
|
||||
types: [rustfs-chain-s3]
|
||||
|
||||
permissions:
|
||||
contents: read
|
||||
@@ -33,21 +32,34 @@ env:
|
||||
RUSTFS_NODES: ${{ secrets.RUSTFS_NODES || vars.RUSTFS_NODES }}
|
||||
RUSTFS_SSH_USER: ${{ secrets.RUSTFS_SSH_USER || vars.RUSTFS_SSH_USER }}
|
||||
RUSTFS_NIGHTLY_PACKAGE_URL: ${{ vars.RUSTFS_NIGHTLY_PACKAGE_URL || 'https://dl.rustfs.com/artifacts/rustfs/packages/nightly/rustfs-nightly-latest.deb' }}
|
||||
PF_TESTING_GH_TOKEN: ${{ secrets.PF_TESTING_GH_TOKEN }}
|
||||
|
||||
jobs:
|
||||
s3-compat-test:
|
||||
runs-on: smoke-testing
|
||||
continue-on-error: true
|
||||
timeout-minutes: 360
|
||||
if: ${{ github.event_name == 'workflow_dispatch' || github.event.workflow_run.conclusion == 'success' }}
|
||||
if: ${{ github.event_name == 'workflow_dispatch' || github.event_name == 'repository_dispatch' }}
|
||||
steps:
|
||||
- name: Checkout auto-testing scripts
|
||||
uses: actions/checkout@9c091bb21b7c1c1d1991bb908d89e4e9dddfe3e0 # v7
|
||||
with:
|
||||
repository: rustfs/auto-testing
|
||||
ref: main
|
||||
path: auto-testing
|
||||
persist-credentials: false
|
||||
token: ${{ secrets.PF_TESTING_GH_TOKEN }}
|
||||
# auto-testing is private: clone it with the dedicated PF token (not
|
||||
# GITHUB_TOKEN) and retry transient GitHub/network failures.
|
||||
- name: Checkout auto-testing scripts (with retry)
|
||||
env:
|
||||
GH_TOKEN: ${{ secrets.PF_TESTING_GH_TOKEN }}
|
||||
run: |
|
||||
set -euo pipefail
|
||||
rm -rf auto-testing
|
||||
for attempt in 1 2 3 4 5; do
|
||||
if gh repo clone rustfs/auto-testing auto-testing -- --depth 1 --quiet; then
|
||||
echo "auto-testing cloned (attempt ${attempt})"
|
||||
exit 0
|
||||
fi
|
||||
rm -rf auto-testing
|
||||
echo "clone attempt ${attempt} failed; retrying in $((attempt * 15))s" >&2
|
||||
sleep $((attempt * 15))
|
||||
done
|
||||
echo "ERROR: unable to clone rustfs/auto-testing after 5 attempts" >&2
|
||||
exit 1
|
||||
|
||||
- name: Show environment
|
||||
run: |
|
||||
@@ -76,6 +88,7 @@ jobs:
|
||||
|
||||
- name: Run S3 compatibility suite
|
||||
id: test
|
||||
continue-on-error: true
|
||||
env:
|
||||
LOG_FILE: /tmp/rustfs-s3-compat.log
|
||||
run: |
|
||||
@@ -109,12 +122,79 @@ jobs:
|
||||
else
|
||||
PACKAGE_SOURCE="${RUSTFS_NIGHTLY_PACKAGE_URL}"
|
||||
fi
|
||||
read -r -a NODES <<< "${RUSTFS_NODES:-vm000 vm001 vm002}"
|
||||
SSH_USER="${RUSTFS_SSH_USER:-azureuser}"
|
||||
RUSTFS_VERSION_INFO="N/A"
|
||||
if [ "${#NODES[@]}" -gt 0 ]; then
|
||||
DETECTED_VERSION="$(ssh -o BatchMode=yes -o ConnectTimeout=10 -o StrictHostKeyChecking=accept-new \
|
||||
"${SSH_USER}@${NODES[0]}" 'rustfs --version' 2>/dev/null | tr -d '\r' | head -n 1 || true)"
|
||||
if [ -n "${DETECTED_VERSION}" ]; then
|
||||
RUSTFS_VERSION_INFO="${DETECTED_VERSION}"
|
||||
fi
|
||||
fi
|
||||
CASE_TABLE="/tmp/rustfs-s3-compat-cases.md"
|
||||
python3 - "${LOG_FILE}" "${CASE_TABLE}" <<'PY'
|
||||
import re
|
||||
import sys
|
||||
|
||||
log_file, out_file = sys.argv[1], sys.argv[2]
|
||||
ansi = re.compile(r'\x1b\[[0-9;]*m')
|
||||
start_re = re.compile(r'^---\s+([A-Z0-9]+-[0-9]+)\s+(.+?)\s+---$')
|
||||
done_re = re.compile(r'^\[(PASS|FAIL|UNSUPPORTED)\]\s+([A-Z0-9]+-[0-9]+)\b')
|
||||
|
||||
rows = []
|
||||
index = {}
|
||||
current = None
|
||||
try:
|
||||
with open(log_file, 'r', encoding='utf-8', errors='replace') as fh:
|
||||
for raw in fh:
|
||||
line = ansi.sub('', raw).strip()
|
||||
m = start_re.match(line)
|
||||
if m:
|
||||
case_id, name = m.group(1), m.group(2)
|
||||
current = case_id
|
||||
if case_id not in index:
|
||||
index[case_id] = len(rows)
|
||||
rows.append([case_id, name, 'RUNNING'])
|
||||
continue
|
||||
m = done_re.match(line)
|
||||
if m:
|
||||
status, case_id = m.group(1), m.group(2)
|
||||
if case_id in index:
|
||||
rows[index[case_id]][2] = status
|
||||
else:
|
||||
rows.append([case_id, case_id, status])
|
||||
index[case_id] = len(rows) - 1
|
||||
current = None
|
||||
except FileNotFoundError:
|
||||
rows = []
|
||||
|
||||
counts = {'PASS': 0, 'FAIL': 0, 'UNSUPPORTED': 0, 'RUNNING': 0}
|
||||
for _, _, status in rows:
|
||||
counts[status] = counts.get(status, 0) + 1
|
||||
|
||||
with open(out_file, 'w', encoding='utf-8') as out:
|
||||
out.write('## Case Summary\n\n')
|
||||
out.write(f"- Total: {len(rows)}\\n")
|
||||
out.write(f"- PASS: {counts.get('PASS', 0)}\\n")
|
||||
out.write(f"- FAIL: {counts.get('FAIL', 0)}\\n")
|
||||
out.write(f"- UNSUPPORTED: {counts.get('UNSUPPORTED', 0)}\\n")
|
||||
out.write('\\n')
|
||||
out.write('| Case | Name | Status |\\n')
|
||||
out.write('| --- | --- | --- |\\n')
|
||||
for case_id, name, status in rows:
|
||||
out.write(f'| {case_id} | {name} | {status} |\\n')
|
||||
PY
|
||||
{
|
||||
echo "# RustFS S3 compatibility test report"
|
||||
echo ""
|
||||
echo "- Run: ${{ github.server_url }}/${{ github.repository }}/actions/runs/${{ github.run_id }}"
|
||||
echo "- Trigger: ${{ github.event_name }}"
|
||||
echo "- Package: ${PACKAGE_SOURCE}"
|
||||
echo "- RustFS Version: ${RUSTFS_VERSION_INFO}"
|
||||
echo "- Test Step Outcome: ${{ steps.test.outcome }}"
|
||||
echo ""
|
||||
cat "${CASE_TABLE}" || true
|
||||
echo ""
|
||||
echo "## Log tail"
|
||||
echo '```text'
|
||||
@@ -123,6 +203,92 @@ jobs:
|
||||
} | tee "${REPORT_FILE}"
|
||||
cat "${REPORT_FILE}" >> "${GITHUB_STEP_SUMMARY}"
|
||||
|
||||
- name: Upload functional report to dashboard
|
||||
if: always()
|
||||
continue-on-error: true
|
||||
env:
|
||||
GH_TOKEN: ${{ env.PF_TESTING_GH_TOKEN }}
|
||||
REPORT_FILE: /tmp/rustfs-s3-compat-report.md
|
||||
SUITE: s3
|
||||
run: |
|
||||
set -euo pipefail
|
||||
if [ -z "${GH_TOKEN:-}" ]; then
|
||||
echo "PF_TESTING_GH_TOKEN is not configured; skipping dashboard upload"
|
||||
exit 0
|
||||
fi
|
||||
DATE="$(date -u +%Y-%m-%d)"
|
||||
REPORT_PATH="functional-reports/${SUITE}/${DATE}.md"
|
||||
CONTENT="$(python3 -c 'import base64,sys;print(base64.b64encode(open(sys.argv[1],"rb").read()).decode())' "${REPORT_FILE}")"
|
||||
SHA="$(gh api "repos/rustfs/dashboard/contents/${REPORT_PATH}" -q '.sha' 2>/dev/null || true)"
|
||||
if [ -n "${SHA}" ]; then
|
||||
jq -n --arg msg "report(${SUITE}): ${DATE}" --arg content "${CONTENT}" --arg sha "${SHA}" \
|
||||
'{message:$msg, content:$content, sha:$sha}' \
|
||||
| gh api --method PUT "repos/rustfs/dashboard/contents/${REPORT_PATH}" --input - >/dev/null
|
||||
else
|
||||
jq -n --arg msg "report(${SUITE}): ${DATE}" --arg content "${CONTENT}" \
|
||||
'{message:$msg, content:$content}' \
|
||||
| gh api --method PUT "repos/rustfs/dashboard/contents/${REPORT_PATH}" --input - >/dev/null
|
||||
fi
|
||||
|
||||
- name: File failure issue in rustfs/backlog
|
||||
if: ${{ always() && (failure() || steps.test.outcome == 'failure' || steps.test.outcome == 'cancelled') }}
|
||||
continue-on-error: true
|
||||
env:
|
||||
GH_TOKEN: ${{ secrets.PF_TESTING_GH_TOKEN }}
|
||||
SUITE: 's3'
|
||||
SUITE_LABEL: 'S3 compatibility'
|
||||
RUN_URL: ${{ github.server_url }}/${{ github.repository }}/actions/runs/${{ github.run_id }}
|
||||
REPORT_FILE: '/tmp/rustfs-s3-compat-report.md'
|
||||
LOG_FILE: '/tmp/rustfs-s3-compat.log'
|
||||
run: |
|
||||
set -euo pipefail
|
||||
if [ -z "${GH_TOKEN:-}" ]; then
|
||||
echo "PF_TESTING_GH_TOKEN is not configured; skipping backlog issue"
|
||||
exit 0
|
||||
fi
|
||||
TITLE="[functional][${SUITE}] ${SUITE_LABEL} suite failed (run ${GITHUB_RUN_ID})"
|
||||
EXISTING="$(gh issue list -R rustfs/backlog --state all \
|
||||
--search "in:title \"run ${GITHUB_RUN_ID}\"" \
|
||||
--json number --jq '.[].number' || true)"
|
||||
if [ -n "${EXISTING}" ]; then
|
||||
echo "backlog issue already exists for run ${GITHUB_RUN_ID}; skipping"
|
||||
exit 0
|
||||
fi
|
||||
redact() {
|
||||
sed -E \
|
||||
-e 's/(RUSTFS_(ACCESS_KEY|SECRET_KEY)[=: ]+)[^[:space:]]+/\1[REDACTED]/Ig' \
|
||||
-e 's/(Authorization:).*/\1 [REDACTED]/Ig' \
|
||||
-e 's/(X-Amz-Signature=)[^&[:space:]]+/\1[REDACTED]/Ig' \
|
||||
-e 's/^.*(password|secret|token)[=: ].*/[REDACTED SENSITIVE LINE]/Ig'
|
||||
}
|
||||
BODY_FILE="$(mktemp)"
|
||||
{
|
||||
echo "The **${SUITE_LABEL}** functional suite failed."
|
||||
echo ""
|
||||
echo "- Suite: \`${SUITE}\`"
|
||||
echo "- Run: ${RUN_URL}"
|
||||
echo "- Trigger: ${GITHUB_EVENT_NAME}"
|
||||
echo "- Date: $(date -u +%Y-%m-%d)"
|
||||
echo ""
|
||||
echo "## Report (errors and symptoms)"
|
||||
echo ""
|
||||
if [ -s "${REPORT_FILE}" ]; then
|
||||
redact < "${REPORT_FILE}"
|
||||
elif [ -s "${LOG_FILE:-}" ]; then
|
||||
echo "(report file missing; log tail below)"
|
||||
echo ""
|
||||
tail -n 200 "${LOG_FILE}" | redact
|
||||
else
|
||||
echo "(no report or log file was produced)"
|
||||
fi
|
||||
} | head -c 55000 > "${BODY_FILE}"
|
||||
gh label create functional-test -R rustfs/backlog --color d73a4a 2>/dev/null || true
|
||||
if ! gh issue create -R rustfs/backlog --title "${TITLE}" \
|
||||
--body-file "${BODY_FILE}" --label functional-test; then
|
||||
gh issue create -R rustfs/backlog --title "${TITLE}" --body-file "${BODY_FILE}"
|
||||
fi
|
||||
echo "filed backlog issue for suite ${SUITE}"
|
||||
|
||||
- name: Upload report and logs
|
||||
if: always()
|
||||
uses: actions/upload-artifact@b7c566a772e6b6bfb58ed0dc250532a479d7789f # v6
|
||||
@@ -152,6 +318,55 @@ jobs:
|
||||
'
|
||||
done
|
||||
|
||||
- name: "Continue functional chain (next: KMS)"
|
||||
# Only chain-triggered runs forward to the next suite; standalone
|
||||
# workflow_dispatch runs stop after their own cleanup. A failed
|
||||
# handoff must never pass silently: it retries, then files an alert
|
||||
# issue in rustfs/backlog so a stalled chain is visible.
|
||||
if: ${{ always() && github.event_name == 'repository_dispatch' }}
|
||||
continue-on-error: true
|
||||
env:
|
||||
GH_TOKEN: ${{ secrets.PF_TESTING_GH_TOKEN }}
|
||||
run: |
|
||||
set -uo pipefail
|
||||
if [ -z "${{GH_TOKEN:-}}" ]; then
|
||||
echo "PF_TESTING_GH_TOKEN is not configured; cannot dispatch the next suite" >&2
|
||||
exit 1
|
||||
fi
|
||||
DISPATCHED=0
|
||||
for attempt in 1 2 3; do
|
||||
if gh api --method POST repos/rustfs/rustfs/dispatches \
|
||||
-f event_type='rustfs-chain-kms' \
|
||||
-F 'client_payload[from_suite]=s3'; then
|
||||
echo "dispatched next suite KMS (attempt ${{attempt}})"
|
||||
DISPATCHED=1
|
||||
break
|
||||
fi
|
||||
echo "dispatch attempt ${{attempt}} failed; retrying in ${{attempt}}0s" >&2
|
||||
sleep "${{attempt}}0"
|
||||
done
|
||||
if [ "${{DISPATCHED:-0}}" -ne 1 ]; then
|
||||
echo "ERROR: functional chain stalled: could not dispatch KMS after 3 attempts" >&2
|
||||
TITLE="[functional][chain] stalled after s3 (run ${{GITHUB_RUN_ID}})"
|
||||
BODY_FILE="$(mktemp)"
|
||||
{
|
||||
echo "The functional chain could not hand off from **s3** to **KMS** after 3 attempts."
|
||||
echo ""
|
||||
echo "- Failed suite job: ${{GITHUB_SERVER_URL}}/${{GITHUB_REPOSITORY}}/actions/runs/${{GITHUB_RUN_ID}}"
|
||||
echo "- Expected next event: `rustfs-chain-kms`"
|
||||
echo "- Likely cause: PF_TESTING_GH_TOKEN lacks contents:write on rustfs/rustfs, or the GitHub API was unavailable."
|
||||
echo "- Recovery: re-dispatch manually with"
|
||||
echo " ```"
|
||||
echo " gh api --method POST repos/rustfs/rustfs/dispatches -f event_type='rustfs-chain-kms'"
|
||||
echo " ```"
|
||||
} > "${{BODY_FILE}}"
|
||||
gh issue create -R rustfs/backlog --title "${{TITLE}}" \
|
||||
--body-file "${{BODY_FILE}}" --label functional-test \
|
||||
|| gh issue create -R rustfs/backlog --title "${{TITLE}}" --body-file "${{BODY_FILE}}" \
|
||||
|| echo "could not file the stall alert issue either; check the token" >&2
|
||||
exit 1
|
||||
fi
|
||||
|
||||
- name: Notify on failure
|
||||
if: failure()
|
||||
run: |
|
||||
|
||||
@@ -0,0 +1,300 @@
|
||||
# Copyright 2024 RustFS Team
|
||||
#
|
||||
# Licensed under the Apache License, Version 2.0 (the "License");
|
||||
# you may not use this file except in compliance with the License.
|
||||
# You may obtain a copy of the License at
|
||||
#
|
||||
# http://www.apache.org/licenses/LICENSE-2.0
|
||||
#
|
||||
# Unless required by applicable law or agreed to in writing, software
|
||||
# distributed under the License is distributed on an "AS IS" BASIS,
|
||||
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||
# See the License for the specific language governing permissions and
|
||||
# limitations under the License.
|
||||
|
||||
name: RustFS Security Test
|
||||
|
||||
on:
|
||||
workflow_dispatch:
|
||||
inputs:
|
||||
rustfs_version:
|
||||
description: 'RustFS release tag to test (e.g. 1.0.0-rc.4-preview.1)'
|
||||
required: false
|
||||
default: '1.0.0-rc.4-preview.1'
|
||||
package_url:
|
||||
description: 'Direct .deb URL (nightly/R2/dev). Overrides rustfs_version.'
|
||||
required: false
|
||||
type: string
|
||||
topology:
|
||||
description: 'Topology to run (all = SNSD, SNMD, MNMD)'
|
||||
type: choice
|
||||
options:
|
||||
- all
|
||||
- single-single
|
||||
- single-multi
|
||||
- multi-multi
|
||||
default: all
|
||||
oidc_live:
|
||||
description: 'Run the live Keycloak OIDC/SSO gate as part of the suite'
|
||||
type: boolean
|
||||
default: true
|
||||
cleanup_before:
|
||||
description: 'Reset the nodes before the test (DESTROYS existing data/config)'
|
||||
type: boolean
|
||||
default: true
|
||||
cleanup_after:
|
||||
description: 'Reset the nodes after the test (DESTROYS test data/config)'
|
||||
type: boolean
|
||||
default: true
|
||||
repository_dispatch:
|
||||
# Chain handoff: dispatched when the pool expansion suite finishes (last link).
|
||||
types: [rustfs-chain-security]
|
||||
|
||||
permissions:
|
||||
contents: read
|
||||
|
||||
# The security suite uses the same shared VMs as the other functional tests,
|
||||
# so it must serialize with them instead of running in parallel.
|
||||
concurrency:
|
||||
group: rustfs-shared-functional-tests
|
||||
cancel-in-progress: false
|
||||
|
||||
defaults:
|
||||
run:
|
||||
shell: bash
|
||||
|
||||
env:
|
||||
RUSTFS_ACCESS_KEY: ${{ secrets.RUSTFS_ACCESS_KEY }}
|
||||
RUSTFS_SECRET_KEY: ${{ secrets.RUSTFS_SECRET_KEY }}
|
||||
RUSTFS_NODES: ${{ secrets.RUSTFS_NODES || vars.RUSTFS_NODES }}
|
||||
RUSTFS_SSH_USER: ${{ secrets.RUSTFS_SSH_USER || vars.RUSTFS_SSH_USER }}
|
||||
RUSTFS_NIGHTLY_PACKAGE_URL: ${{ vars.RUSTFS_NIGHTLY_PACKAGE_URL || 'https://dl.rustfs.com/artifacts/rustfs/packages/nightly/rustfs-nightly-latest.deb' }}
|
||||
PF_TESTING_GH_TOKEN: ${{ secrets.PF_TESTING_GH_TOKEN }}
|
||||
|
||||
jobs:
|
||||
security-test:
|
||||
runs-on: smoke-testing
|
||||
continue-on-error: true
|
||||
timeout-minutes: 360
|
||||
if: ${{ github.event_name == 'workflow_dispatch' || github.event_name == 'repository_dispatch' }}
|
||||
steps:
|
||||
# auto-testing is private: clone it with the dedicated PF token (not
|
||||
# GITHUB_TOKEN) and retry transient GitHub/network failures.
|
||||
- name: Checkout auto-testing scripts (with retry)
|
||||
env:
|
||||
GH_TOKEN: ${{ secrets.PF_TESTING_GH_TOKEN }}
|
||||
run: |
|
||||
set -euo pipefail
|
||||
rm -rf auto-testing
|
||||
for attempt in 1 2 3 4 5; do
|
||||
if gh repo clone rustfs/auto-testing auto-testing -- --depth 1 --quiet; then
|
||||
echo "auto-testing cloned (attempt ${attempt})"
|
||||
exit 0
|
||||
fi
|
||||
rm -rf auto-testing
|
||||
echo "clone attempt ${attempt} failed; retrying in $((attempt * 15))s" >&2
|
||||
sleep $((attempt * 15))
|
||||
done
|
||||
echo "ERROR: unable to clone rustfs/auto-testing after 5 attempts" >&2
|
||||
exit 1
|
||||
|
||||
- name: Checkout repository (for the OIDC live gate script)
|
||||
uses: actions/checkout@9c091bb21b7c1c1d1991bb908d89e4e9dddfe3e0 # v7
|
||||
with:
|
||||
persist-credentials: false
|
||||
|
||||
- name: Show environment
|
||||
run: |
|
||||
uname -a
|
||||
jq --version
|
||||
openssl version
|
||||
aws --version || true
|
||||
docker --version || true
|
||||
df -h /data | tail -1
|
||||
|
||||
- name: Cleanup environment (before)
|
||||
if: ${{ inputs.cleanup_before != 'false' || github.event_name != 'workflow_dispatch' }}
|
||||
run: |
|
||||
set -euo pipefail
|
||||
read -r -a NODES <<< "${RUSTFS_NODES:-vm000 vm001 vm002}"
|
||||
SSH_USER="${RUSTFS_SSH_USER:-azureuser}"
|
||||
for node in "${NODES[@]}"; do
|
||||
ssh -o BatchMode=yes -o ConnectTimeout=10 -o StrictHostKeyChecking=accept-new "${SSH_USER}@${node}" '
|
||||
set -euo pipefail
|
||||
SUDO=""; [ "$(id -u)" -ne 0 ] && SUDO="sudo -n"
|
||||
${SUDO} systemctl stop rustfs 2>/dev/null || true
|
||||
if ${SUDO} dpkg -l rustfs 2>/dev/null | grep -q "^ii"; then
|
||||
${SUDO} dpkg -P rustfs
|
||||
fi
|
||||
for i in 1 2 3 4; do ${SUDO} rm -rf /data/rustfs${i}/mnmd; done
|
||||
${SUDO} rm -rf /var/log/rustfs /var/lib/rustfs/kms
|
||||
'
|
||||
done
|
||||
|
||||
- name: Run security suite
|
||||
id: test
|
||||
continue-on-error: true
|
||||
env:
|
||||
REPORT_FILE: /tmp/rustfs-security-report.md
|
||||
RUSTFS_SECURITY_OIDC_LIVE_SCRIPT: ${{ github.workspace }}/scripts/test/oidc_keycloak_live.sh
|
||||
run: |
|
||||
set -euo pipefail
|
||||
chmod +x auto-testing/rustfs-security-test.sh
|
||||
PACKAGE_URL='${{ inputs.package_url }}'
|
||||
RUSTFS_VERSION='${{ inputs.rustfs_version }}'
|
||||
TOPOLOGY='${{ inputs.topology }}'
|
||||
ARGS=(-y)
|
||||
if [ "${TOPOLOGY}" = "all" ] || [ -z "${TOPOLOGY}" ] || [ "${TOPOLOGY}" = "null" ]; then
|
||||
ARGS+=(--all-topologies)
|
||||
else
|
||||
ARGS+=(--topology "${TOPOLOGY}")
|
||||
fi
|
||||
if [ "${{ inputs.oidc_live }}" = "true" ] || [ "${{ github.event_name }}" != "workflow_dispatch" ]; then
|
||||
ARGS+=(--oidc-live)
|
||||
fi
|
||||
if [ -n "${PACKAGE_URL}" ]; then
|
||||
ARGS+=(--package-url "${PACKAGE_URL}")
|
||||
elif [ -n "${RUSTFS_VERSION}" ] && [ "${RUSTFS_VERSION}" != "null" ]; then
|
||||
ARGS+=(--version "${RUSTFS_VERSION}")
|
||||
else
|
||||
ARGS+=(--package-url "${RUSTFS_NIGHTLY_PACKAGE_URL}")
|
||||
fi
|
||||
./auto-testing/rustfs-security-test.sh "${ARGS[@]}"
|
||||
|
||||
- name: Generate report
|
||||
if: always()
|
||||
run: |
|
||||
set -euo pipefail
|
||||
if [ ! -f /tmp/rustfs-security-report.md ]; then
|
||||
{
|
||||
echo "# RustFS security test report"
|
||||
echo ""
|
||||
echo "- Run: ${{ github.server_url }}/${{ github.repository }}/actions/runs/${{ github.run_id }}"
|
||||
echo "- Trigger: ${{ github.event_name }}"
|
||||
echo "- Test Step Outcome: failure (suite did not produce a report)"
|
||||
} > /tmp/rustfs-security-report.md
|
||||
fi
|
||||
cat /tmp/rustfs-security-report.md >> "${GITHUB_STEP_SUMMARY}"
|
||||
|
||||
- name: Upload functional report to dashboard
|
||||
if: always()
|
||||
continue-on-error: true
|
||||
env:
|
||||
GH_TOKEN: ${{ env.PF_TESTING_GH_TOKEN }}
|
||||
REPORT_FILE: /tmp/rustfs-security-report.md
|
||||
SUITE: security
|
||||
run: |
|
||||
set -euo pipefail
|
||||
if [ -z "${GH_TOKEN:-}" ]; then
|
||||
echo "PF_TESTING_GH_TOKEN is not configured; skipping dashboard upload"
|
||||
exit 0
|
||||
fi
|
||||
DATE="$(date -u +%Y-%m-%d)"
|
||||
REPORT_PATH="functional-reports/${SUITE}/${DATE}.md"
|
||||
CONTENT="$(python3 -c 'import base64,sys;print(base64.b64encode(open(sys.argv[1],"rb").read()).decode())' "${REPORT_FILE}")"
|
||||
SHA="$(gh api "repos/rustfs/dashboard/contents/${REPORT_PATH}" -q '.sha' 2>/dev/null || true)"
|
||||
if [ -n "${SHA}" ]; then
|
||||
jq -n --arg msg "report(${SUITE}): ${DATE}" --arg content "${CONTENT}" --arg sha "${SHA}" \
|
||||
'{message:$msg, content:$content, sha:$sha}' \
|
||||
| gh api --method PUT "repos/rustfs/dashboard/contents/${REPORT_PATH}" --input - >/dev/null
|
||||
else
|
||||
jq -n --arg msg "report(${SUITE}): ${DATE}" --arg content "${CONTENT}" \
|
||||
'{message:$msg, content:$content}' \
|
||||
| gh api --method PUT "repos/rustfs/dashboard/contents/${REPORT_PATH}" --input - >/dev/null
|
||||
fi
|
||||
|
||||
- name: File failure issue in rustfs/backlog
|
||||
if: ${{ always() && (failure() || steps.test.outcome == 'failure' || steps.test.outcome == 'cancelled') }}
|
||||
continue-on-error: true
|
||||
env:
|
||||
GH_TOKEN: ${{ secrets.PF_TESTING_GH_TOKEN }}
|
||||
SUITE: 'security'
|
||||
SUITE_LABEL: 'Security'
|
||||
RUN_URL: ${{ github.server_url }}/${{ github.repository }}/actions/runs/${{ github.run_id }}
|
||||
REPORT_FILE: '/tmp/rustfs-security-report.md'
|
||||
LOG_FILE: ''
|
||||
run: |
|
||||
set -euo pipefail
|
||||
if [ -z "${GH_TOKEN:-}" ]; then
|
||||
echo "PF_TESTING_GH_TOKEN is not configured; skipping backlog issue"
|
||||
exit 0
|
||||
fi
|
||||
TITLE="[functional][${SUITE}] ${SUITE_LABEL} suite failed (run ${GITHUB_RUN_ID})"
|
||||
EXISTING="$(gh issue list -R rustfs/backlog --state all \
|
||||
--search "in:title \"run ${GITHUB_RUN_ID}\"" \
|
||||
--json number --jq '.[].number' || true)"
|
||||
if [ -n "${EXISTING}" ]; then
|
||||
echo "backlog issue already exists for run ${GITHUB_RUN_ID}; skipping"
|
||||
exit 0
|
||||
fi
|
||||
redact() {
|
||||
sed -E \
|
||||
-e 's/(RUSTFS_(ACCESS_KEY|SECRET_KEY)[=: ]+)[^[:space:]]+/\1[REDACTED]/Ig' \
|
||||
-e 's/(Authorization:).*/\1 [REDACTED]/Ig' \
|
||||
-e 's/(X-Amz-Signature=)[^&[:space:]]+/\1[REDACTED]/Ig' \
|
||||
-e 's/^.*(password|secret|token)[=: ].*/[REDACTED SENSITIVE LINE]/Ig'
|
||||
}
|
||||
BODY_FILE="$(mktemp)"
|
||||
{
|
||||
echo "The **${SUITE_LABEL}** functional suite failed."
|
||||
echo ""
|
||||
echo "- Suite: \`${SUITE}\`"
|
||||
echo "- Run: ${RUN_URL}"
|
||||
echo "- Trigger: ${GITHUB_EVENT_NAME}"
|
||||
echo "- Date: $(date -u +%Y-%m-%d)"
|
||||
echo ""
|
||||
echo "## Report (errors and symptoms)"
|
||||
echo ""
|
||||
if [ -s "${REPORT_FILE}" ]; then
|
||||
redact < "${REPORT_FILE}"
|
||||
elif [ -s "${LOG_FILE:-}" ]; then
|
||||
echo "(report file missing; log tail below)"
|
||||
echo ""
|
||||
tail -n 200 "${LOG_FILE}" | redact
|
||||
else
|
||||
echo "(no report or log file was produced)"
|
||||
fi
|
||||
} | head -c 55000 > "${BODY_FILE}"
|
||||
gh label create functional-test -R rustfs/backlog --color d73a4a 2>/dev/null || true
|
||||
if ! gh issue create -R rustfs/backlog --title "${TITLE}" \
|
||||
--body-file "${BODY_FILE}" --label functional-test; then
|
||||
gh issue create -R rustfs/backlog --title "${TITLE}" --body-file "${BODY_FILE}"
|
||||
fi
|
||||
echo "filed backlog issue for suite ${SUITE}"
|
||||
|
||||
- name: Upload report and logs
|
||||
if: always()
|
||||
uses: actions/upload-artifact@b7c566a772e6b6bfb58ed0dc250532a479d7789f # v6
|
||||
with:
|
||||
name: rustfs-security-test-${{ github.run_id }}
|
||||
path: |
|
||||
/tmp/rustfs-security-report.md
|
||||
/tmp/rustfs-security.*/*
|
||||
if-no-files-found: ignore
|
||||
retention-days: 3
|
||||
|
||||
- name: Cleanup environment (after)
|
||||
if: ${{ always() && (inputs.cleanup_after != 'false' || github.event_name != 'workflow_dispatch') }}
|
||||
run: |
|
||||
set -euo pipefail
|
||||
read -r -a NODES <<< "${RUSTFS_NODES:-vm000 vm001 vm002}"
|
||||
SSH_USER="${RUSTFS_SSH_USER:-azureuser}"
|
||||
for node in "${NODES[@]}"; do
|
||||
ssh -o BatchMode=yes -o ConnectTimeout=10 -o StrictHostKeyChecking=accept-new "${SSH_USER}@${node}" '
|
||||
set -euo pipefail
|
||||
SUDO=""; [ "$(id -u)" -ne 0 ] && SUDO="sudo -n"
|
||||
${SUDO} systemctl stop rustfs 2>/dev/null || true
|
||||
if ${SUDO} dpkg -l rustfs 2>/dev/null | grep -q "^ii"; then
|
||||
${SUDO} dpkg -P rustfs
|
||||
fi
|
||||
for i in 1 2 3 4; do ${SUDO} rm -rf /data/rustfs${i}/mnmd; done
|
||||
${SUDO} rm -rf /var/log/rustfs /var/lib/rustfs/kms
|
||||
'
|
||||
done
|
||||
|
||||
- name: Notify on failure
|
||||
if: failure()
|
||||
run: |
|
||||
echo "RustFS security test failed"
|
||||
echo "Package source: ${{ inputs.package_url || 'nightly (R2 latest)' }}"
|
||||
echo "See the uploaded report and logs for details."
|
||||
@@ -0,0 +1,389 @@
|
||||
name: RustFS Storage Engine Test
|
||||
|
||||
on:
|
||||
workflow_dispatch:
|
||||
inputs:
|
||||
rustfs_version:
|
||||
description: 'RustFS release tag to test (e.g. 1.0.0-rc.4-preview.1)'
|
||||
required: false
|
||||
default: '1.0.0-rc.4-preview.1'
|
||||
package_url:
|
||||
description: 'Direct .deb URL (nightly/R2/dev). Overrides rustfs_version.'
|
||||
required: false
|
||||
type: string
|
||||
topology:
|
||||
description: 'Topology to run (all = SNSD, SNMD, MNMD)'
|
||||
type: choice
|
||||
options:
|
||||
- all
|
||||
- single-single
|
||||
- single-multi
|
||||
- multi-multi
|
||||
default: all
|
||||
repository_dispatch:
|
||||
# Chain handoff: dispatched when the tier suite finishes.
|
||||
types: [rustfs-chain-storage]
|
||||
|
||||
permissions:
|
||||
contents: read
|
||||
|
||||
concurrency:
|
||||
group: rustfs-shared-functional-tests
|
||||
cancel-in-progress: false
|
||||
|
||||
defaults:
|
||||
run:
|
||||
shell: bash
|
||||
|
||||
env:
|
||||
RUSTFS_ACCESS_KEY: ${{ secrets.RUSTFS_ACCESS_KEY }}
|
||||
RUSTFS_SECRET_KEY: ${{ secrets.RUSTFS_SECRET_KEY }}
|
||||
RUSTFS_NODES: ${{ secrets.RUSTFS_NODES || vars.RUSTFS_NODES }}
|
||||
RUSTFS_SSH_USER: ${{ secrets.RUSTFS_SSH_USER || vars.RUSTFS_SSH_USER }}
|
||||
RUSTFS_NIGHTLY_PACKAGE_URL: ${{ vars.RUSTFS_NIGHTLY_PACKAGE_URL || 'https://dl.rustfs.com/artifacts/rustfs/packages/nightly/rustfs-nightly-latest.deb' }}
|
||||
PF_TESTING_GH_TOKEN: ${{ secrets.PF_TESTING_GH_TOKEN }}
|
||||
|
||||
jobs:
|
||||
storage-test:
|
||||
runs-on: smoke-testing
|
||||
continue-on-error: true
|
||||
timeout-minutes: 360
|
||||
if: ${{ github.event_name == 'workflow_dispatch' || github.event_name == 'repository_dispatch' }}
|
||||
steps:
|
||||
# auto-testing is private: clone it with the dedicated PF token (not
|
||||
# GITHUB_TOKEN) and retry transient GitHub/network failures.
|
||||
- name: Checkout auto-testing scripts (with retry)
|
||||
env:
|
||||
GH_TOKEN: ${{ secrets.PF_TESTING_GH_TOKEN }}
|
||||
run: |
|
||||
set -euo pipefail
|
||||
rm -rf auto-testing
|
||||
for attempt in 1 2 3 4 5; do
|
||||
if gh repo clone rustfs/auto-testing auto-testing -- --depth 1 --quiet; then
|
||||
echo "auto-testing cloned (attempt ${attempt})"
|
||||
exit 0
|
||||
fi
|
||||
rm -rf auto-testing
|
||||
echo "clone attempt ${attempt} failed; retrying in $((attempt * 15))s" >&2
|
||||
sleep $((attempt * 15))
|
||||
done
|
||||
echo "ERROR: unable to clone rustfs/auto-testing after 5 attempts" >&2
|
||||
exit 1
|
||||
|
||||
- name: Show environment
|
||||
run: |
|
||||
uname -a
|
||||
jq --version
|
||||
openssl version
|
||||
df -h /data | tail -1
|
||||
|
||||
- name: Cleanup environment (before)
|
||||
run: |
|
||||
set -euo pipefail
|
||||
read -r -a NODES <<< "${RUSTFS_NODES:-vm000 vm001 vm002}"
|
||||
SSH_USER="${RUSTFS_SSH_USER:-azureuser}"
|
||||
for node in "${NODES[@]}"; do
|
||||
ssh -o BatchMode=yes -o ConnectTimeout=10 -o StrictHostKeyChecking=accept-new "${SSH_USER}@${node}" '
|
||||
set -euo pipefail
|
||||
SUDO=""; [ "$(id -u)" -ne 0 ] && SUDO="sudo -n"
|
||||
${SUDO} systemctl stop rustfs 2>/dev/null || true
|
||||
if ${SUDO} dpkg -l rustfs 2>/dev/null | grep -q "^ii"; then
|
||||
${SUDO} dpkg -P rustfs
|
||||
fi
|
||||
for i in 1 2 3 4; do ${SUDO} rm -rf /data/rustfs${i}/mnmd; done
|
||||
${SUDO} rm -rf /var/log/rustfs /var/lib/rustfs/kms /var/lib/rustfs/kms-backup
|
||||
'
|
||||
done
|
||||
|
||||
- name: Run storage engine suite
|
||||
id: test
|
||||
continue-on-error: true
|
||||
env:
|
||||
LOG_FILE: /tmp/rustfs-storage.log
|
||||
run: |
|
||||
set -euo pipefail
|
||||
chmod +x auto-testing/rustfs-storage-test.sh
|
||||
PACKAGE_URL='${{ inputs.package_url }}'
|
||||
RUSTFS_VERSION='${{ inputs.rustfs_version }}'
|
||||
TOPOLOGY='${{ inputs.topology }}'
|
||||
ARGS=(-y --log-file "${LOG_FILE}")
|
||||
if [ "${TOPOLOGY}" = "all" ] || [ -z "${TOPOLOGY}" ] || [ "${TOPOLOGY}" = "null" ]; then
|
||||
ARGS+=(--all-topologies)
|
||||
else
|
||||
ARGS+=(--topology "${TOPOLOGY}")
|
||||
fi
|
||||
if [ -n "${PACKAGE_URL}" ]; then
|
||||
ARGS+=(--package-url "${PACKAGE_URL}")
|
||||
elif [ -n "${RUSTFS_VERSION}" ]; then
|
||||
ARGS+=(--version "${RUSTFS_VERSION}")
|
||||
else
|
||||
ARGS+=(--package-url "${RUSTFS_NIGHTLY_PACKAGE_URL}")
|
||||
fi
|
||||
./auto-testing/rustfs-storage-test.sh "${ARGS[@]}"
|
||||
|
||||
- name: Generate report
|
||||
if: always()
|
||||
env:
|
||||
LOG_FILE: /tmp/rustfs-storage.log
|
||||
REPORT_FILE: /tmp/rustfs-storage-report.md
|
||||
run: |
|
||||
set -euo pipefail
|
||||
PACKAGE_URL='${{ inputs.package_url }}'
|
||||
RUSTFS_VERSION='${{ inputs.rustfs_version }}'
|
||||
if [ -n "${PACKAGE_URL}" ]; then
|
||||
PACKAGE_SOURCE="${PACKAGE_URL}"
|
||||
elif [ -n "${RUSTFS_VERSION}" ]; then
|
||||
PACKAGE_SOURCE="version ${RUSTFS_VERSION}"
|
||||
else
|
||||
PACKAGE_SOURCE="${RUSTFS_NIGHTLY_PACKAGE_URL}"
|
||||
fi
|
||||
read -r -a NODES <<< "${RUSTFS_NODES:-vm000 vm001 vm002}"
|
||||
SSH_USER="${RUSTFS_SSH_USER:-azureuser}"
|
||||
RUSTFS_VERSION_INFO="N/A"
|
||||
if [ "${#NODES[@]}" -gt 0 ]; then
|
||||
DETECTED_VERSION="$(ssh -o BatchMode=yes -o ConnectTimeout=10 -o StrictHostKeyChecking=accept-new \
|
||||
"${SSH_USER}@${NODES[0]}" 'rustfs --version' 2>/dev/null | tr -d '\r' | head -n 1 || true)"
|
||||
if [ -n "${DETECTED_VERSION}" ]; then
|
||||
RUSTFS_VERSION_INFO="${DETECTED_VERSION}"
|
||||
fi
|
||||
fi
|
||||
CASE_TABLE="/tmp/rustfs-storage-cases.md"
|
||||
python3 - "${LOG_FILE}" "${CASE_TABLE}" <<'PY'
|
||||
import re
|
||||
import sys
|
||||
|
||||
log_file, out_file = sys.argv[1], sys.argv[2]
|
||||
ansi = re.compile(r'\x1b\[[0-9;]*m')
|
||||
start_re = re.compile(r'^---\s+([A-Z0-9]+-[0-9]+)\s+(.+?)\s+---$')
|
||||
done_re = re.compile(r'^\[(PASS|FAIL|UNSUPPORTED)\]\s+([A-Z0-9]+-[0-9]+)\b')
|
||||
|
||||
rows = []
|
||||
index = {}
|
||||
current = None
|
||||
try:
|
||||
with open(log_file, 'r', encoding='utf-8', errors='replace') as fh:
|
||||
for raw in fh:
|
||||
line = ansi.sub('', raw).strip()
|
||||
m = start_re.match(line)
|
||||
if m:
|
||||
case_id, name = m.group(1), m.group(2)
|
||||
current = case_id
|
||||
if case_id not in index:
|
||||
index[case_id] = len(rows)
|
||||
rows.append([case_id, name, 'RUNNING'])
|
||||
continue
|
||||
m = done_re.match(line)
|
||||
if m:
|
||||
status, case_id = m.group(1), m.group(2)
|
||||
if case_id in index:
|
||||
rows[index[case_id]][2] = status
|
||||
else:
|
||||
rows.append([case_id, case_id, status])
|
||||
index[case_id] = len(rows) - 1
|
||||
current = None
|
||||
except FileNotFoundError:
|
||||
rows = []
|
||||
|
||||
counts = {'PASS': 0, 'FAIL': 0, 'UNSUPPORTED': 0, 'RUNNING': 0}
|
||||
for _, _, status in rows:
|
||||
counts[status] = counts.get(status, 0) + 1
|
||||
|
||||
with open(out_file, 'w', encoding='utf-8') as out:
|
||||
out.write('## Case Summary\n\n')
|
||||
out.write(f"- Total: {len(rows)}\\n")
|
||||
out.write(f"- PASS: {counts.get('PASS', 0)}\\n")
|
||||
out.write(f"- FAIL: {counts.get('FAIL', 0)}\\n")
|
||||
out.write(f"- UNSUPPORTED: {counts.get('UNSUPPORTED', 0)}\\n")
|
||||
out.write('\\n')
|
||||
out.write('| Case | Name | Status |\\n')
|
||||
out.write('| --- | --- | --- |\\n')
|
||||
for case_id, name, status in rows:
|
||||
out.write(f'| {case_id} | {name} | {status} |\\n')
|
||||
PY
|
||||
{
|
||||
echo "# RustFS storage engine test report"
|
||||
echo ""
|
||||
echo "- Run: ${{ github.server_url }}/${{ github.repository }}/actions/runs/${{ github.run_id }}"
|
||||
echo "- Trigger: ${{ github.event_name }}"
|
||||
echo "- Package: ${PACKAGE_SOURCE}"
|
||||
echo "- RustFS Version: ${RUSTFS_VERSION_INFO}"
|
||||
echo "- Test Step Outcome: ${{ steps.test.outcome }}"
|
||||
echo ""
|
||||
cat "${CASE_TABLE}" || true
|
||||
echo ""
|
||||
echo "## Log tail"
|
||||
echo '```text'
|
||||
tail -n 200 "${LOG_FILE}" || true
|
||||
echo '```'
|
||||
} | tee "${REPORT_FILE}"
|
||||
cat "${REPORT_FILE}" >> "${GITHUB_STEP_SUMMARY}"
|
||||
|
||||
- name: Upload functional report to dashboard
|
||||
if: always()
|
||||
continue-on-error: true
|
||||
env:
|
||||
GH_TOKEN: ${{ env.PF_TESTING_GH_TOKEN }}
|
||||
REPORT_FILE: /tmp/rustfs-storage-report.md
|
||||
SUITE: storage
|
||||
run: |
|
||||
set -euo pipefail
|
||||
if [ -z "${GH_TOKEN:-}" ]; then
|
||||
echo "PF_TESTING_GH_TOKEN is not configured; skipping dashboard upload"
|
||||
exit 0
|
||||
fi
|
||||
DATE="$(date -u +%Y-%m-%d)"
|
||||
REPORT_PATH="functional-reports/${SUITE}/${DATE}.md"
|
||||
CONTENT="$(python3 -c 'import base64,sys;print(base64.b64encode(open(sys.argv[1],"rb").read()).decode())' "${REPORT_FILE}")"
|
||||
SHA="$(gh api "repos/rustfs/dashboard/contents/${REPORT_PATH}" -q '.sha' 2>/dev/null || true)"
|
||||
if [ -n "${SHA}" ]; then
|
||||
jq -n --arg msg "report(${SUITE}): ${DATE}" --arg content "${CONTENT}" --arg sha "${SHA}" \
|
||||
'{message:$msg, content:$content, sha:$sha}' \
|
||||
| gh api --method PUT "repos/rustfs/dashboard/contents/${REPORT_PATH}" --input - >/dev/null
|
||||
else
|
||||
jq -n --arg msg "report(${SUITE}): ${DATE}" --arg content "${CONTENT}" \
|
||||
'{message:$msg, content:$content}' \
|
||||
| gh api --method PUT "repos/rustfs/dashboard/contents/${REPORT_PATH}" --input - >/dev/null
|
||||
fi
|
||||
|
||||
- name: File failure issue in rustfs/backlog
|
||||
if: ${{ always() && (failure() || steps.test.outcome == 'failure' || steps.test.outcome == 'cancelled') }}
|
||||
continue-on-error: true
|
||||
env:
|
||||
GH_TOKEN: ${{ secrets.PF_TESTING_GH_TOKEN }}
|
||||
SUITE: 'storage'
|
||||
SUITE_LABEL: 'Storage engine'
|
||||
RUN_URL: ${{ github.server_url }}/${{ github.repository }}/actions/runs/${{ github.run_id }}
|
||||
REPORT_FILE: '/tmp/rustfs-storage-report.md'
|
||||
LOG_FILE: '/tmp/rustfs-storage.log'
|
||||
run: |
|
||||
set -euo pipefail
|
||||
if [ -z "${GH_TOKEN:-}" ]; then
|
||||
echo "PF_TESTING_GH_TOKEN is not configured; skipping backlog issue"
|
||||
exit 0
|
||||
fi
|
||||
TITLE="[functional][${SUITE}] ${SUITE_LABEL} suite failed (run ${GITHUB_RUN_ID})"
|
||||
EXISTING="$(gh issue list -R rustfs/backlog --state all \
|
||||
--search "in:title \"run ${GITHUB_RUN_ID}\"" \
|
||||
--json number --jq '.[].number' || true)"
|
||||
if [ -n "${EXISTING}" ]; then
|
||||
echo "backlog issue already exists for run ${GITHUB_RUN_ID}; skipping"
|
||||
exit 0
|
||||
fi
|
||||
redact() {
|
||||
sed -E \
|
||||
-e 's/(RUSTFS_(ACCESS_KEY|SECRET_KEY)[=: ]+)[^[:space:]]+/\1[REDACTED]/Ig' \
|
||||
-e 's/(Authorization:).*/\1 [REDACTED]/Ig' \
|
||||
-e 's/(X-Amz-Signature=)[^&[:space:]]+/\1[REDACTED]/Ig' \
|
||||
-e 's/^.*(password|secret|token)[=: ].*/[REDACTED SENSITIVE LINE]/Ig'
|
||||
}
|
||||
BODY_FILE="$(mktemp)"
|
||||
{
|
||||
echo "The **${SUITE_LABEL}** functional suite failed."
|
||||
echo ""
|
||||
echo "- Suite: \`${SUITE}\`"
|
||||
echo "- Run: ${RUN_URL}"
|
||||
echo "- Trigger: ${GITHUB_EVENT_NAME}"
|
||||
echo "- Date: $(date -u +%Y-%m-%d)"
|
||||
echo ""
|
||||
echo "## Report (errors and symptoms)"
|
||||
echo ""
|
||||
if [ -s "${REPORT_FILE}" ]; then
|
||||
redact < "${REPORT_FILE}"
|
||||
elif [ -s "${LOG_FILE:-}" ]; then
|
||||
echo "(report file missing; log tail below)"
|
||||
echo ""
|
||||
tail -n 200 "${LOG_FILE}" | redact
|
||||
else
|
||||
echo "(no report or log file was produced)"
|
||||
fi
|
||||
} | head -c 55000 > "${BODY_FILE}"
|
||||
gh label create functional-test -R rustfs/backlog --color d73a4a 2>/dev/null || true
|
||||
if ! gh issue create -R rustfs/backlog --title "${TITLE}" \
|
||||
--body-file "${BODY_FILE}" --label functional-test; then
|
||||
gh issue create -R rustfs/backlog --title "${TITLE}" --body-file "${BODY_FILE}"
|
||||
fi
|
||||
echo "filed backlog issue for suite ${SUITE}"
|
||||
|
||||
- name: Upload report and logs
|
||||
if: always()
|
||||
uses: actions/upload-artifact@b7c566a772e6b6bfb58ed0dc250532a479d7789f # v6
|
||||
with:
|
||||
name: rustfs-storage-${{ github.run_id }}
|
||||
path: |
|
||||
/tmp/rustfs-storage.log
|
||||
/tmp/rustfs-storage-report.md
|
||||
if-no-files-found: warn
|
||||
|
||||
- name: Cleanup environment (after)
|
||||
if: always()
|
||||
run: |
|
||||
set -euo pipefail
|
||||
read -r -a NODES <<< "${RUSTFS_NODES:-vm000 vm001 vm002}"
|
||||
SSH_USER="${RUSTFS_SSH_USER:-azureuser}"
|
||||
for node in "${NODES[@]}"; do
|
||||
ssh -o BatchMode=yes -o ConnectTimeout=10 -o StrictHostKeyChecking=accept-new "${SSH_USER}@${node}" '
|
||||
set -euo pipefail
|
||||
SUDO=""; [ "$(id -u)" -ne 0 ] && SUDO="sudo -n"
|
||||
${SUDO} systemctl stop rustfs 2>/dev/null || true
|
||||
if ${SUDO} dpkg -l rustfs 2>/dev/null | grep -q "^ii"; then
|
||||
${SUDO} dpkg -P rustfs
|
||||
fi
|
||||
for i in 1 2 3 4; do ${SUDO} rm -rf /data/rustfs${i}/mnmd; done
|
||||
${SUDO} rm -rf /var/log/rustfs
|
||||
'
|
||||
done
|
||||
|
||||
- name: "Continue functional chain (next: Heal)"
|
||||
# Only chain-triggered runs forward to the next suite; standalone
|
||||
# workflow_dispatch runs stop after their own cleanup. A failed
|
||||
# handoff must never pass silently: it retries, then files an alert
|
||||
# issue in rustfs/backlog so a stalled chain is visible.
|
||||
if: ${{ always() && github.event_name == 'repository_dispatch' }}
|
||||
continue-on-error: true
|
||||
env:
|
||||
GH_TOKEN: ${{ secrets.PF_TESTING_GH_TOKEN }}
|
||||
run: |
|
||||
set -uo pipefail
|
||||
if [ -z "${{GH_TOKEN:-}}" ]; then
|
||||
echo "PF_TESTING_GH_TOKEN is not configured; cannot dispatch the next suite" >&2
|
||||
exit 1
|
||||
fi
|
||||
DISPATCHED=0
|
||||
for attempt in 1 2 3; do
|
||||
if gh api --method POST repos/rustfs/rustfs/dispatches \
|
||||
-f event_type='rustfs-chain-heal' \
|
||||
-F 'client_payload[from_suite]=storage'; then
|
||||
echo "dispatched next suite Heal (attempt ${{attempt}})"
|
||||
DISPATCHED=1
|
||||
break
|
||||
fi
|
||||
echo "dispatch attempt ${{attempt}} failed; retrying in ${{attempt}}0s" >&2
|
||||
sleep "${{attempt}}0"
|
||||
done
|
||||
if [ "${{DISPATCHED:-0}}" -ne 1 ]; then
|
||||
echo "ERROR: functional chain stalled: could not dispatch Heal after 3 attempts" >&2
|
||||
TITLE="[functional][chain] stalled after storage (run ${{GITHUB_RUN_ID}})"
|
||||
BODY_FILE="$(mktemp)"
|
||||
{
|
||||
echo "The functional chain could not hand off from **storage** to **Heal** after 3 attempts."
|
||||
echo ""
|
||||
echo "- Failed suite job: ${{GITHUB_SERVER_URL}}/${{GITHUB_REPOSITORY}}/actions/runs/${{GITHUB_RUN_ID}}"
|
||||
echo "- Expected next event: `rustfs-chain-heal`"
|
||||
echo "- Likely cause: PF_TESTING_GH_TOKEN lacks contents:write on rustfs/rustfs, or the GitHub API was unavailable."
|
||||
echo "- Recovery: re-dispatch manually with"
|
||||
echo " ```"
|
||||
echo " gh api --method POST repos/rustfs/rustfs/dispatches -f event_type='rustfs-chain-heal'"
|
||||
echo " ```"
|
||||
} > "${{BODY_FILE}}"
|
||||
gh issue create -R rustfs/backlog --title "${{TITLE}}" \
|
||||
--body-file "${{BODY_FILE}}" --label functional-test \
|
||||
|| gh issue create -R rustfs/backlog --title "${{TITLE}}" --body-file "${{BODY_FILE}}" \
|
||||
|| echo "could not file the stall alert issue either; check the token" >&2
|
||||
exit 1
|
||||
fi
|
||||
|
||||
- name: Notify on failure
|
||||
if: failure()
|
||||
run: |
|
||||
echo "RustFS storage engine suite failed"
|
||||
echo "See the uploaded report and log artifacts for details."
|
||||
@@ -11,10 +11,18 @@ on:
|
||||
description: 'Direct .deb URL (nightly/R2/dev). Overrides rustfs_version.'
|
||||
required: false
|
||||
type: string
|
||||
workflow_run:
|
||||
# Strict shared-environment order: run after KMS test succeeds.
|
||||
workflows: ["RustFS KMS Test"]
|
||||
types: [completed]
|
||||
rc_sha256:
|
||||
description: 'Optional SHA-256 for the preinstalled rc binary; mismatch is an infrastructure failure.'
|
||||
required: false
|
||||
type: string
|
||||
force_case_failure:
|
||||
description: 'Diagnostic only: rewrite single-single/TIER-101 to FAIL after execution to verify artifact and final-gate behavior.'
|
||||
required: false
|
||||
default: false
|
||||
type: boolean
|
||||
repository_dispatch:
|
||||
# Chain handoff: dispatched when the KMS suite finishes.
|
||||
types: [rustfs-chain-tier]
|
||||
|
||||
permissions:
|
||||
contents: read
|
||||
@@ -33,22 +41,50 @@ env:
|
||||
RUSTFS_NODES: ${{ secrets.RUSTFS_NODES || vars.RUSTFS_NODES }}
|
||||
RUSTFS_SSH_USER: ${{ secrets.RUSTFS_SSH_USER || vars.RUSTFS_SSH_USER }}
|
||||
RUSTFS_NIGHTLY_PACKAGE_URL: ${{ vars.RUSTFS_NIGHTLY_PACKAGE_URL || 'https://dl.rustfs.com/artifacts/rustfs/packages/nightly/rustfs-nightly-latest.deb' }}
|
||||
RUSTFS_EXPECTED_RC_SHA256: ${{ inputs.rc_sha256 || vars.RUSTFS_TIER_RC_SHA256 }}
|
||||
PF_TESTING_GH_TOKEN: ${{ secrets.PF_TESTING_GH_TOKEN }}
|
||||
TIER_ARTIFACTS_DIR: /tmp/rustfs-tier-artifacts-${{ github.run_id }}-${{ github.run_attempt }}
|
||||
|
||||
jobs:
|
||||
tier-test:
|
||||
runs-on: smoke-testing
|
||||
# Requirement: a failing suite must not fail the workflow; failures
|
||||
# are filed to rustfs/backlog and the chain continues.
|
||||
continue-on-error: true
|
||||
timeout-minutes: 420
|
||||
if: ${{ github.event_name == 'workflow_dispatch' || github.event.workflow_run.conclusion == 'success' }}
|
||||
if: ${{ github.event_name == 'workflow_dispatch' || github.event_name == 'repository_dispatch' }}
|
||||
steps:
|
||||
- name: Checkout auto-testing scripts
|
||||
uses: actions/checkout@9c091bb21b7c1c1d1991bb908d89e4e9dddfe3e0 # v7
|
||||
with:
|
||||
repository: rustfs/auto-testing
|
||||
ref: main
|
||||
path: auto-testing
|
||||
persist-credentials: false
|
||||
token: ${{ secrets.PF_TESTING_GH_TOKEN }}
|
||||
- name: Initialize run evidence directory
|
||||
id: evidence
|
||||
run: |
|
||||
set -euo pipefail
|
||||
umask 077
|
||||
if ! mkdir -- "${TIER_ARTIFACTS_DIR}"; then
|
||||
echo "refusing to reuse tier evidence path: ${TIER_ARTIFACTS_DIR}" >&2
|
||||
exit 1
|
||||
fi
|
||||
test -d "${TIER_ARTIFACTS_DIR}"
|
||||
test ! -L "${TIER_ARTIFACTS_DIR}"
|
||||
|
||||
# auto-testing is private: clone it with the dedicated PF token (not
|
||||
# GITHUB_TOKEN) and retry transient GitHub/network failures.
|
||||
- name: Checkout auto-testing scripts (with retry)
|
||||
env:
|
||||
GH_TOKEN: ${{ secrets.PF_TESTING_GH_TOKEN }}
|
||||
run: |
|
||||
set -euo pipefail
|
||||
rm -rf auto-testing
|
||||
for attempt in 1 2 3 4 5; do
|
||||
if gh repo clone rustfs/auto-testing auto-testing -- --depth 1 --quiet; then
|
||||
echo "auto-testing cloned (attempt ${attempt})"
|
||||
exit 0
|
||||
fi
|
||||
rm -rf auto-testing
|
||||
echo "clone attempt ${attempt} failed; retrying in $((attempt * 15))s" >&2
|
||||
sleep $((attempt * 15))
|
||||
done
|
||||
echo "ERROR: unable to clone rustfs/auto-testing after 5 attempts" >&2
|
||||
exit 1
|
||||
|
||||
- name: Show environment
|
||||
run: |
|
||||
@@ -60,6 +96,8 @@ jobs:
|
||||
- name: Cleanup environment (before)
|
||||
run: |
|
||||
set -euo pipefail
|
||||
sudo docker rm -f rustfs-test-mqtt >/dev/null 2>&1 || true
|
||||
sudo rm -f /tmp/rustfs-mosquitto.conf
|
||||
read -r -a NODES <<< "${RUSTFS_NODES:-vm000 vm001 vm002}"
|
||||
SSH_USER="${RUSTFS_SSH_USER:-azureuser}"
|
||||
for node in "${NODES[@]}"; do
|
||||
@@ -77,26 +115,55 @@ jobs:
|
||||
|
||||
- name: Ensure MQTT broker + clients
|
||||
run: |
|
||||
set -euo pipefail
|
||||
if ! command -v mosquitto_sub >/dev/null 2>&1; then
|
||||
sudo apt-get update
|
||||
sudo apt-get install -y mosquitto mosquitto-clients
|
||||
sudo apt-get install -y mosquitto-clients
|
||||
fi
|
||||
sudo mkdir -p /etc/mosquitto/conf.d
|
||||
printf 'listener 1883 0.0.0.0\nallow_anonymous true\n' | sudo tee /etc/mosquitto/conf.d/rustfs-test.conf >/dev/null
|
||||
sudo systemctl restart mosquitto
|
||||
sleep 2
|
||||
ss -tln 2>/dev/null | grep -q ':1883' || { echo 'mosquitto not listening on 1883'; exit 1; }
|
||||
command -v docker >/dev/null 2>&1 || { echo 'docker not found on runner'; exit 1; }
|
||||
sudo docker rm -f rustfs-test-mqtt >/dev/null 2>&1 || true
|
||||
cat <<'EOF' | sudo tee /tmp/rustfs-mosquitto.conf >/dev/null
|
||||
listener 1883 0.0.0.0
|
||||
allow_anonymous true
|
||||
EOF
|
||||
sudo docker run -d --name rustfs-test-mqtt -p 1883:1883 \
|
||||
-v /tmp/rustfs-mosquitto.conf:/mosquitto/config/mosquitto.conf:ro \
|
||||
eclipse-mosquitto:2 >/dev/null
|
||||
for _ in {1..10}; do
|
||||
if ss -tln 2>/dev/null | grep -q ':1883'; then
|
||||
break
|
||||
fi
|
||||
sleep 1
|
||||
done
|
||||
ss -tln 2>/dev/null | grep -q ':1883' || {
|
||||
echo 'mosquitto container is not listening on 1883'
|
||||
sudo docker logs rustfs-test-mqtt || true
|
||||
exit 1
|
||||
}
|
||||
|
||||
- name: Run tier suite
|
||||
id: test
|
||||
continue-on-error: true
|
||||
env:
|
||||
LOG_FILE: /tmp/rustfs-tier.log
|
||||
PACKAGE_URL_INPUT: ${{ inputs.package_url }}
|
||||
RUSTFS_VERSION_INPUT: ${{ inputs.rustfs_version }}
|
||||
run: |
|
||||
set -euo pipefail
|
||||
LOG_FILE="${TIER_ARTIFACTS_DIR}/rustfs-tier.log"
|
||||
chmod +x auto-testing/rustfs-tier-test.sh
|
||||
PACKAGE_URL='${{ inputs.package_url }}'
|
||||
RUSTFS_VERSION='${{ inputs.rustfs_version }}'
|
||||
ARGS=(--all-topologies -y --log-file "${LOG_FILE}")
|
||||
RC_BIN="$(command -v rc)"
|
||||
PACKAGE_URL="${PACKAGE_URL_INPUT}"
|
||||
RUSTFS_VERSION="${RUSTFS_VERSION_INPUT}"
|
||||
ARGS=(
|
||||
--all-topologies
|
||||
-y
|
||||
--log-file "${LOG_FILE}"
|
||||
--rc-bin "${RC_BIN}"
|
||||
--artifacts-dir "${TIER_ARTIFACTS_DIR}"
|
||||
)
|
||||
if [ -n "${RUSTFS_EXPECTED_RC_SHA256}" ]; then
|
||||
ARGS+=(--expected-rc-sha256 "${RUSTFS_EXPECTED_RC_SHA256}")
|
||||
fi
|
||||
if [ -n "${PACKAGE_URL}" ]; then
|
||||
ARGS+=(--package-url "${PACKAGE_URL}")
|
||||
elif [ -n "${RUSTFS_VERSION}" ]; then
|
||||
@@ -106,15 +173,34 @@ jobs:
|
||||
fi
|
||||
./auto-testing/rustfs-tier-test.sh "${ARGS[@]}"
|
||||
|
||||
- name: Generate report
|
||||
if: always()
|
||||
env:
|
||||
LOG_FILE: /tmp/rustfs-tier.log
|
||||
REPORT_FILE: /tmp/rustfs-tier-report.md
|
||||
- name: Inject diagnostic case failure
|
||||
if: ${{ always() && steps.evidence.outcome == 'success' && inputs.force_case_failure }}
|
||||
run: |
|
||||
set -euo pipefail
|
||||
PACKAGE_URL='${{ inputs.package_url }}'
|
||||
RUSTFS_VERSION='${{ inputs.rustfs_version }}'
|
||||
RESULT_FILE="${TIER_ARTIFACTS_DIR}/cases/single-single--TIER-101.json"
|
||||
test -s "${RESULT_FILE}"
|
||||
TMP_FILE="$(mktemp "${TIER_ARTIFACTS_DIR}/cases/.forced.XXXXXX")"
|
||||
jq '.status = "FAIL" | .case_rc = 97' "${RESULT_FILE}" > "${TMP_FILE}"
|
||||
mv "${TMP_FILE}" "${RESULT_FILE}"
|
||||
|
||||
- name: Generate report
|
||||
if: ${{ always() && steps.evidence.outcome == 'success' }}
|
||||
env:
|
||||
PACKAGE_URL_INPUT: ${{ inputs.package_url }}
|
||||
RUSTFS_VERSION_INPUT: ${{ inputs.rustfs_version }}
|
||||
TEST_OUTCOME: ${{ steps.test.outcome }}
|
||||
RUN_URL: ${{ github.server_url }}/${{ github.repository }}/actions/runs/${{ github.run_id }}
|
||||
TRIGGER_NAME: ${{ github.event_name }}
|
||||
run: |
|
||||
set -euo pipefail
|
||||
test -d "${TIER_ARTIFACTS_DIR}"
|
||||
test ! -L "${TIER_ARTIFACTS_DIR}"
|
||||
LOG_FILE="${TIER_ARTIFACTS_DIR}/rustfs-tier.log"
|
||||
REPORT_FILE="${TIER_ARTIFACTS_DIR}/rustfs-tier-report.md"
|
||||
CASE_TABLE="${TIER_ARTIFACTS_DIR}/rustfs-tier-cases.md"
|
||||
GATE_RC_FILE="${TIER_ARTIFACTS_DIR}/rustfs-tier-gate.rc"
|
||||
PACKAGE_URL="${PACKAGE_URL_INPUT}"
|
||||
RUSTFS_VERSION="${RUSTFS_VERSION_INPUT}"
|
||||
if [ -n "${PACKAGE_URL}" ]; then
|
||||
PACKAGE_SOURCE="${PACKAGE_URL}"
|
||||
elif [ -n "${RUSTFS_VERSION}" ]; then
|
||||
@@ -122,12 +208,31 @@ jobs:
|
||||
else
|
||||
PACKAGE_SOURCE="${RUSTFS_NIGHTLY_PACKAGE_URL}"
|
||||
fi
|
||||
set +e
|
||||
python3 auto-testing/rustfs_tier_report.py \
|
||||
--results-dir "${TIER_ARTIFACTS_DIR}/cases" \
|
||||
--provenance "${TIER_ARTIFACTS_DIR}/provenance.json" \
|
||||
--output "${CASE_TABLE}"
|
||||
CASE_GATE_RC=$?
|
||||
set -e
|
||||
printf '%s\n' "${CASE_GATE_RC}" > "${GATE_RC_FILE}"
|
||||
if [ ! -s "${CASE_TABLE}" ]; then
|
||||
{
|
||||
echo "## Case Summary"
|
||||
echo ""
|
||||
echo "Structured report generation failed before producing output (exit ${CASE_GATE_RC})."
|
||||
} > "${CASE_TABLE}"
|
||||
fi
|
||||
{
|
||||
echo "# RustFS tier test report"
|
||||
echo ""
|
||||
echo "- Run: ${{ github.server_url }}/${{ github.repository }}/actions/runs/${{ github.run_id }}"
|
||||
echo "- Trigger: ${{ github.event_name }}"
|
||||
echo "- Run: ${RUN_URL}"
|
||||
echo "- Trigger: ${TRIGGER_NAME}"
|
||||
echo "- Package: ${PACKAGE_SOURCE}"
|
||||
echo "- Test Step Outcome: ${TEST_OUTCOME}"
|
||||
echo "- Structured Gate Exit: ${CASE_GATE_RC}"
|
||||
echo ""
|
||||
cat "${CASE_TABLE}"
|
||||
echo ""
|
||||
echo "## Log tail"
|
||||
echo '```text'
|
||||
@@ -136,20 +241,76 @@ jobs:
|
||||
} | tee "${REPORT_FILE}"
|
||||
cat "${REPORT_FILE}" >> "${GITHUB_STEP_SUMMARY}"
|
||||
|
||||
- name: Upload functional report to dashboard
|
||||
if: ${{ always() && steps.evidence.outcome == 'success' }}
|
||||
continue-on-error: true
|
||||
env:
|
||||
GH_TOKEN: ${{ env.PF_TESTING_GH_TOKEN }}
|
||||
REPORT_FILE: ${{ env.TIER_ARTIFACTS_DIR }}/rustfs-tier-report.md
|
||||
SUITE: tier
|
||||
run: |
|
||||
set -euo pipefail
|
||||
if [ -z "${GH_TOKEN:-}" ]; then
|
||||
echo "PF_TESTING_GH_TOKEN is not configured; skipping dashboard upload"
|
||||
exit 0
|
||||
fi
|
||||
DATE="$(date -u +%Y-%m-%d)"
|
||||
REPORT_PATH="functional-reports/${SUITE}/${DATE}.md"
|
||||
CONTENT="$(python3 -c 'import base64,sys;print(base64.b64encode(open(sys.argv[1],"rb").read()).decode())' "${REPORT_FILE}")"
|
||||
SHA="$(gh api "repos/rustfs/dashboard/contents/${REPORT_PATH}" -q '.sha' 2>/dev/null || true)"
|
||||
if [ -n "${SHA}" ]; then
|
||||
jq -n --arg msg "report(${SUITE}): ${DATE}" --arg content "${CONTENT}" --arg sha "${SHA}" \
|
||||
'{message:$msg, content:$content, sha:$sha}' \
|
||||
| gh api --method PUT "repos/rustfs/dashboard/contents/${REPORT_PATH}" --input - >/dev/null
|
||||
else
|
||||
jq -n --arg msg "report(${SUITE}): ${DATE}" --arg content "${CONTENT}" \
|
||||
'{message:$msg, content:$content}' \
|
||||
| gh api --method PUT "repos/rustfs/dashboard/contents/${REPORT_PATH}" --input - >/dev/null
|
||||
fi
|
||||
|
||||
- name: Verify required tier evidence
|
||||
id: evidence_verify
|
||||
if: ${{ always() && steps.evidence.outcome == 'success' }}
|
||||
run: |
|
||||
set -euo pipefail
|
||||
failed=0
|
||||
for name in \
|
||||
rustfs-tier.log \
|
||||
rustfs-tier-report.md \
|
||||
rustfs-tier-cases.md \
|
||||
rustfs-tier-gate.rc \
|
||||
provenance.json; do
|
||||
if [ ! -s "${TIER_ARTIFACTS_DIR}/${name}" ]; then
|
||||
echo "required tier evidence is missing or empty: ${name}" >&2
|
||||
failed=1
|
||||
fi
|
||||
done
|
||||
for name in cases logs; do
|
||||
if [ ! -d "${TIER_ARTIFACTS_DIR}/${name}" ]; then
|
||||
echo "required tier evidence directory is missing: ${name}" >&2
|
||||
failed=1
|
||||
fi
|
||||
done
|
||||
if ! find "${TIER_ARTIFACTS_DIR}/cases" -maxdepth 1 -type f -name '*.json' -print -quit 2>/dev/null | grep -q .; then
|
||||
echo "no atomic tier case result was produced" >&2
|
||||
failed=1
|
||||
fi
|
||||
[ "${failed}" -eq 0 ]
|
||||
|
||||
- name: Upload report and logs
|
||||
if: always()
|
||||
if: ${{ always() && steps.evidence.outcome == 'success' }}
|
||||
uses: actions/upload-artifact@b7c566a772e6b6bfb58ed0dc250532a479d7789f # v6
|
||||
with:
|
||||
name: rustfs-tier-test-${{ github.run_id }}
|
||||
path: |
|
||||
/tmp/rustfs-tier.log
|
||||
/tmp/rustfs-tier-report.md
|
||||
if-no-files-found: warn
|
||||
name: rustfs-tier-test-${{ github.run_id }}-${{ github.run_attempt }}
|
||||
path: ${{ env.TIER_ARTIFACTS_DIR }}/
|
||||
if-no-files-found: error
|
||||
|
||||
- name: Cleanup environment (after)
|
||||
if: always()
|
||||
run: |
|
||||
set -euo pipefail
|
||||
sudo docker rm -f rustfs-test-mqtt >/dev/null 2>&1 || true
|
||||
sudo rm -f /tmp/rustfs-mosquitto.conf
|
||||
read -r -a NODES <<< "${RUSTFS_NODES:-vm000 vm001 vm002}"
|
||||
SSH_USER="${RUSTFS_SSH_USER:-azureuser}"
|
||||
for node in "${NODES[@]}"; do
|
||||
@@ -165,6 +326,157 @@ jobs:
|
||||
'
|
||||
done
|
||||
|
||||
- name: Enforce tier suite result
|
||||
id: gate
|
||||
if: always()
|
||||
env:
|
||||
EVIDENCE_OUTCOME: ${{ steps.evidence.outcome }}
|
||||
TEST_OUTCOME: ${{ steps.test.outcome }}
|
||||
GATE_RC_FILE: ${{ env.TIER_ARTIFACTS_DIR }}/rustfs-tier-gate.rc
|
||||
run: |
|
||||
set -euo pipefail
|
||||
failed=0
|
||||
if [ "${EVIDENCE_OUTCOME}" != "success" ]; then
|
||||
echo "tier evidence directory initialization is ${EVIDENCE_OUTCOME}, expected success" >&2
|
||||
failed=1
|
||||
fi
|
||||
if [ "${TEST_OUTCOME}" != "success" ]; then
|
||||
echo "tier suite step outcome is ${TEST_OUTCOME}, expected success" >&2
|
||||
failed=1
|
||||
fi
|
||||
if [ "${EVIDENCE_OUTCOME}" != "success" ]; then
|
||||
echo "structured gate result is unavailable because evidence initialization failed" >&2
|
||||
elif [ ! -s "${GATE_RC_FILE}" ]; then
|
||||
echo "structured gate result is missing" >&2
|
||||
failed=1
|
||||
else
|
||||
GATE_RC="$(tr -d '[:space:]' < "${GATE_RC_FILE}")"
|
||||
if ! [[ "${GATE_RC}" =~ ^[0-9]+$ ]] || [ "${GATE_RC}" -ne 0 ]; then
|
||||
echo "structured 56-case gate failed with exit ${GATE_RC:-invalid}" >&2
|
||||
failed=1
|
||||
fi
|
||||
fi
|
||||
[ "${failed}" -eq 0 ]
|
||||
|
||||
- name: File failure issue in rustfs/backlog
|
||||
if: ${{ always() && (failure() || steps.test.outcome == 'failure' || steps.test.outcome == 'cancelled' || steps.evidence_verify.outcome == 'failure' || steps.evidence_verify.outcome == 'cancelled' || steps.gate.outcome == 'failure' || steps.gate.outcome == 'cancelled') }}
|
||||
continue-on-error: true
|
||||
env:
|
||||
GH_TOKEN: ${{ secrets.PF_TESTING_GH_TOKEN }}
|
||||
SUITE: 'tier'
|
||||
SUITE_LABEL: 'Tier'
|
||||
RUN_URL: ${{ github.server_url }}/${{ github.repository }}/actions/runs/${{ github.run_id }}
|
||||
EVIDENCE_DIR: ${{ env.TIER_ARTIFACTS_DIR }}
|
||||
EVIDENCE_OUTCOME: ${{ steps.evidence.outcome }}
|
||||
VERIFY_OUTCOME: ${{ steps.evidence_verify.outcome }}
|
||||
GATE_OUTCOME: ${{ steps.gate.outcome }}
|
||||
REPORT_FILE: ${{ env.TIER_ARTIFACTS_DIR }}/rustfs-tier-report.md
|
||||
LOG_FILE: ${{ env.TIER_ARTIFACTS_DIR }}/rustfs-tier.log
|
||||
run: |
|
||||
set -euo pipefail
|
||||
if [ -z "${GH_TOKEN:-}" ]; then
|
||||
echo "PF_TESTING_GH_TOKEN is not configured; skipping backlog issue"
|
||||
exit 0
|
||||
fi
|
||||
TITLE="[functional][${SUITE}] ${SUITE_LABEL} suite failed (run ${GITHUB_RUN_ID})"
|
||||
EXISTING="$(gh issue list -R rustfs/backlog --state all \
|
||||
--search "in:title \"run ${GITHUB_RUN_ID}\"" \
|
||||
--json number --jq '.[].number' || true)"
|
||||
if [ -n "${EXISTING}" ]; then
|
||||
echo "backlog issue already exists for run ${GITHUB_RUN_ID}; skipping"
|
||||
exit 0
|
||||
fi
|
||||
redact() {
|
||||
sed -E \
|
||||
-e 's/(RUSTFS_(ACCESS_KEY|SECRET_KEY)[=: ]+)[^[:space:]]+/\1[REDACTED]/Ig' \
|
||||
-e 's/(Authorization:).*/\1 [REDACTED]/Ig' \
|
||||
-e 's/(X-Amz-Signature=)[^&[:space:]]+/\1[REDACTED]/Ig' \
|
||||
-e 's/^.*(password|secret|token)[=: ].*/[REDACTED SENSITIVE LINE]/Ig'
|
||||
}
|
||||
BODY_FILE="$(mktemp)"
|
||||
{
|
||||
echo "The **${SUITE_LABEL}** functional suite failed."
|
||||
echo ""
|
||||
echo "- Suite: \`${SUITE}\`"
|
||||
echo "- Run: ${RUN_URL}"
|
||||
echo "- Trigger: ${GITHUB_EVENT_NAME}"
|
||||
echo "- Date: $(date -u +%Y-%m-%d)"
|
||||
echo "- Evidence initialization: ${EVIDENCE_OUTCOME}"
|
||||
echo "- Evidence verification: ${VERIFY_OUTCOME}"
|
||||
echo "- Final gate: ${GATE_OUTCOME}"
|
||||
echo ""
|
||||
echo "## Report (errors and symptoms)"
|
||||
echo ""
|
||||
if [ "${EVIDENCE_OUTCOME}" != "success" ]; then
|
||||
echo "(the run evidence directory was rejected; its contents were not read)"
|
||||
elif [ ! -d "${EVIDENCE_DIR}" ] || [ -L "${EVIDENCE_DIR}" ]; then
|
||||
echo "(the run evidence directory is missing or unsafe; its contents were not read)"
|
||||
elif [ -s "${REPORT_FILE}" ]; then
|
||||
redact < "${REPORT_FILE}"
|
||||
elif [ -s "${LOG_FILE:-}" ]; then
|
||||
echo "(report file missing; log tail below)"
|
||||
echo ""
|
||||
tail -n 200 "${LOG_FILE}" | redact
|
||||
else
|
||||
echo "(no report or log file was produced)"
|
||||
fi
|
||||
} | head -c 55000 > "${BODY_FILE}"
|
||||
gh label create functional-test -R rustfs/backlog --color d73a4a 2>/dev/null || true
|
||||
if ! gh issue create -R rustfs/backlog --title "${TITLE}" \
|
||||
--body-file "${BODY_FILE}" --label functional-test; then
|
||||
gh issue create -R rustfs/backlog --title "${TITLE}" --body-file "${BODY_FILE}"
|
||||
fi
|
||||
echo "filed backlog issue for suite ${SUITE}"
|
||||
|
||||
- name: "Continue functional chain (next: Storage engine)"
|
||||
# Only chain-triggered runs forward to the next suite; standalone
|
||||
# workflow_dispatch runs stop after their own cleanup. A failed
|
||||
# handoff must never pass silently: it retries, then files an alert
|
||||
# issue in rustfs/backlog so a stalled chain is visible.
|
||||
if: ${{ always() && github.event_name == 'repository_dispatch' }}
|
||||
continue-on-error: true
|
||||
env:
|
||||
GH_TOKEN: ${{ secrets.PF_TESTING_GH_TOKEN }}
|
||||
run: |
|
||||
set -uo pipefail
|
||||
if [ -z "${{GH_TOKEN:-}}" ]; then
|
||||
echo "PF_TESTING_GH_TOKEN is not configured; cannot dispatch the next suite" >&2
|
||||
exit 1
|
||||
fi
|
||||
DISPATCHED=0
|
||||
for attempt in 1 2 3; do
|
||||
if gh api --method POST repos/rustfs/rustfs/dispatches \
|
||||
-f event_type='rustfs-chain-storage' \
|
||||
-F 'client_payload[from_suite]=tier'; then
|
||||
echo "dispatched next suite Storage engine (attempt ${{attempt}})"
|
||||
DISPATCHED=1
|
||||
break
|
||||
fi
|
||||
echo "dispatch attempt ${{attempt}} failed; retrying in ${{attempt}}0s" >&2
|
||||
sleep "${{attempt}}0"
|
||||
done
|
||||
if [ "${{DISPATCHED:-0}}" -ne 1 ]; then
|
||||
echo "ERROR: functional chain stalled: could not dispatch Storage engine after 3 attempts" >&2
|
||||
TITLE="[functional][chain] stalled after tier (run ${{GITHUB_RUN_ID}})"
|
||||
BODY_FILE="$(mktemp)"
|
||||
{
|
||||
echo "The functional chain could not hand off from **tier** to **Storage engine** after 3 attempts."
|
||||
echo ""
|
||||
echo "- Failed suite job: ${{GITHUB_SERVER_URL}}/${{GITHUB_REPOSITORY}}/actions/runs/${{GITHUB_RUN_ID}}"
|
||||
echo "- Expected next event: `rustfs-chain-storage`"
|
||||
echo "- Likely cause: PF_TESTING_GH_TOKEN lacks contents:write on rustfs/rustfs, or the GitHub API was unavailable."
|
||||
echo "- Recovery: re-dispatch manually with"
|
||||
echo " ```"
|
||||
echo " gh api --method POST repos/rustfs/rustfs/dispatches -f event_type='rustfs-chain-storage'"
|
||||
echo " ```"
|
||||
} > "${{BODY_FILE}}"
|
||||
gh issue create -R rustfs/backlog --title "${{TITLE}}" \
|
||||
--body-file "${{BODY_FILE}}" --label functional-test \
|
||||
|| gh issue create -R rustfs/backlog --title "${{TITLE}}" --body-file "${{BODY_FILE}}" \
|
||||
|| echo "could not file the stall alert issue either; check the token" >&2
|
||||
exit 1
|
||||
fi
|
||||
|
||||
- name: Notify on failure
|
||||
if: failure()
|
||||
run: |
|
||||
|
||||
@@ -0,0 +1,444 @@
|
||||
# Copyright 2024 RustFS Team
|
||||
#
|
||||
# Licensed under the Apache License, Version 2.0 (the "License");
|
||||
# you may not use this file except in compliance with the License.
|
||||
# You may obtain a copy of the License at
|
||||
#
|
||||
# http://www.apache.org/licenses/LICENSE-2.0
|
||||
#
|
||||
# Unless required by applicable law or agreed to in writing, software
|
||||
# distributed under the License is distributed on an "AS IS" BASIS,
|
||||
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||
# See the License for the specific language governing permissions and
|
||||
# limitations under the License.
|
||||
|
||||
name: RustFS Upgrade Test
|
||||
|
||||
on:
|
||||
workflow_dispatch:
|
||||
inputs:
|
||||
from_version:
|
||||
description: 'OLD RustFS release tag (e.g. 1.0.0-rc.4-preview.1)'
|
||||
required: false
|
||||
default: '1.0.0-rc.4-preview.1'
|
||||
from_url:
|
||||
description: 'OLD .deb URL. Overrides from_version.'
|
||||
required: false
|
||||
type: string
|
||||
to_version:
|
||||
description: 'NEW RustFS release tag (leave empty for latest nightly)'
|
||||
required: false
|
||||
to_url:
|
||||
description: 'NEW .deb URL. Overrides to_version / nightly default.'
|
||||
required: false
|
||||
type: string
|
||||
topology:
|
||||
description: 'Topology to run (all = SNSD, SNMD, MNMD)'
|
||||
type: choice
|
||||
options:
|
||||
- all
|
||||
- single-single
|
||||
- single-multi
|
||||
- multi-multi
|
||||
default: all
|
||||
backends:
|
||||
description: 'KMS backends to run (local,vault-kv2)'
|
||||
required: false
|
||||
default: 'local,vault-kv2'
|
||||
cleanup_before:
|
||||
description: 'Reset the nodes before the test (DESTROYS existing data/config)'
|
||||
type: boolean
|
||||
default: true
|
||||
cleanup_after:
|
||||
description: 'Reset the nodes after the test (DESTROYS test data/config)'
|
||||
type: boolean
|
||||
default: true
|
||||
repository_dispatch:
|
||||
# Functional-chain entry: dispatched by rustfs-functional-chain.yml.
|
||||
types: [rustfs-chain-upgrade]
|
||||
|
||||
permissions:
|
||||
contents: read
|
||||
|
||||
concurrency:
|
||||
group: rustfs-shared-functional-tests
|
||||
cancel-in-progress: false
|
||||
|
||||
defaults:
|
||||
run:
|
||||
shell: bash
|
||||
|
||||
env:
|
||||
RUSTFS_ACCESS_KEY: ${{ secrets.RUSTFS_ACCESS_KEY }}
|
||||
RUSTFS_SECRET_KEY: ${{ secrets.RUSTFS_SECRET_KEY }}
|
||||
RUSTFS_NODES: ${{ secrets.RUSTFS_NODES || vars.RUSTFS_NODES }}
|
||||
RUSTFS_SSH_USER: ${{ secrets.RUSTFS_SSH_USER || vars.RUSTFS_SSH_USER }}
|
||||
RUSTFS_NIGHTLY_PACKAGE_URL: ${{ vars.RUSTFS_NIGHTLY_PACKAGE_URL || 'https://dl.rustfs.com/artifacts/rustfs/packages/nightly/rustfs-nightly-latest.deb' }}
|
||||
PF_TESTING_GH_TOKEN: ${{ secrets.PF_TESTING_GH_TOKEN }}
|
||||
|
||||
jobs:
|
||||
upgrade-test:
|
||||
runs-on: smoke-testing
|
||||
continue-on-error: true
|
||||
timeout-minutes: 420
|
||||
if: ${{ github.event_name == 'workflow_dispatch' || github.event_name == 'repository_dispatch' }}
|
||||
steps:
|
||||
# auto-testing is private: clone it with the dedicated PF token (not
|
||||
# GITHUB_TOKEN) and retry transient GitHub/network failures.
|
||||
- name: Checkout auto-testing scripts (with retry)
|
||||
env:
|
||||
GH_TOKEN: ${{ secrets.PF_TESTING_GH_TOKEN }}
|
||||
run: |
|
||||
set -euo pipefail
|
||||
rm -rf auto-testing
|
||||
for attempt in 1 2 3 4 5; do
|
||||
if gh repo clone rustfs/auto-testing auto-testing -- --depth 1 --quiet; then
|
||||
echo "auto-testing cloned (attempt ${attempt})"
|
||||
exit 0
|
||||
fi
|
||||
rm -rf auto-testing
|
||||
echo "clone attempt ${attempt} failed; retrying in $((attempt * 15))s" >&2
|
||||
sleep $((attempt * 15))
|
||||
done
|
||||
echo "ERROR: unable to clone rustfs/auto-testing after 5 attempts" >&2
|
||||
exit 1
|
||||
|
||||
- name: Show environment
|
||||
run: |
|
||||
uname -a
|
||||
jq --version
|
||||
openssl version
|
||||
aws --version || true
|
||||
docker --version || true
|
||||
df -h /data | tail -1
|
||||
|
||||
- name: Cleanup environment (before)
|
||||
if: ${{ inputs.cleanup_before != 'false' || github.event_name != 'workflow_dispatch' }}
|
||||
run: |
|
||||
set -euo pipefail
|
||||
read -r -a NODES <<< "${RUSTFS_NODES:-vm000 vm001 vm002}"
|
||||
SSH_USER="${RUSTFS_SSH_USER:-azureuser}"
|
||||
for node in "${NODES[@]}"; do
|
||||
ssh -o BatchMode=yes -o ConnectTimeout=10 -o StrictHostKeyChecking=accept-new "${SSH_USER}@${node}" '
|
||||
set -euo pipefail
|
||||
SUDO=""; [ "$(id -u)" -ne 0 ] && SUDO="sudo -n"
|
||||
${SUDO} systemctl stop rustfs 2>/dev/null || true
|
||||
if ${SUDO} dpkg -l rustfs 2>/dev/null | grep -q "^ii"; then
|
||||
${SUDO} dpkg -P rustfs
|
||||
fi
|
||||
for i in 1 2 3 4; do ${SUDO} rm -rf /data/rustfs${i}/mnmd; done
|
||||
${SUDO} rm -rf /var/log/rustfs /var/lib/rustfs/kms /var/lib/rustfs/kms-backup
|
||||
'
|
||||
done
|
||||
|
||||
- name: Ensure docker (Vault container)
|
||||
run: |
|
||||
if ! command -v docker >/dev/null 2>&1; then
|
||||
sudo apt-get update
|
||||
sudo apt-get install -y docker.io
|
||||
fi
|
||||
sudo systemctl enable --now docker
|
||||
docker info >/dev/null 2>&1 || sudo docker info >/dev/null 2>&1
|
||||
|
||||
- name: Run upgrade compatibility suite
|
||||
id: test
|
||||
continue-on-error: true
|
||||
env:
|
||||
LOG_FILE: /tmp/rustfs-upgrade.log
|
||||
run: |
|
||||
set -euo pipefail
|
||||
chmod +x auto-testing/rustfs-upgrade-test.sh
|
||||
FROM_URL='${{ inputs.from_url }}'
|
||||
FROM_VERSION='${{ inputs.from_version }}'
|
||||
TO_URL='${{ inputs.to_url }}'
|
||||
TO_VERSION='${{ inputs.to_version }}'
|
||||
TOPOLOGY='${{ inputs.topology }}'
|
||||
BACKENDS='${{ inputs.backends }}'
|
||||
ARGS=(-y --log-file "${LOG_FILE}")
|
||||
if [ "${TOPOLOGY}" = "all" ] || [ -z "${TOPOLOGY}" ] || [ "${TOPOLOGY}" = "null" ]; then
|
||||
ARGS+=(--all-topologies)
|
||||
else
|
||||
ARGS+=(--topology "${TOPOLOGY}")
|
||||
fi
|
||||
if [ -n "${BACKENDS}" ] && [ "${BACKENDS}" != "null" ]; then
|
||||
ARGS+=(--backends "${BACKENDS}")
|
||||
fi
|
||||
if [ -n "${FROM_URL}" ]; then
|
||||
ARGS+=(--from-url "${FROM_URL}")
|
||||
elif [ -n "${FROM_VERSION}" ] && [ "${FROM_VERSION}" != "null" ]; then
|
||||
ARGS+=(--from-version "${FROM_VERSION}")
|
||||
fi
|
||||
if [ -n "${TO_URL}" ]; then
|
||||
ARGS+=(--to-url "${TO_URL}")
|
||||
elif [ -n "${TO_VERSION}" ] && [ "${TO_VERSION}" != "null" ]; then
|
||||
ARGS+=(--to-version "${TO_VERSION}")
|
||||
else
|
||||
ARGS+=(--to-url "${RUSTFS_NIGHTLY_PACKAGE_URL}")
|
||||
fi
|
||||
./auto-testing/rustfs-upgrade-test.sh "${ARGS[@]}"
|
||||
|
||||
- name: Generate report
|
||||
if: always()
|
||||
env:
|
||||
LOG_FILE: /tmp/rustfs-upgrade.log
|
||||
REPORT_FILE: /tmp/rustfs-upgrade-report.md
|
||||
run: |
|
||||
set -euo pipefail
|
||||
FROM_URL='${{ inputs.from_url }}'
|
||||
FROM_VERSION='${{ inputs.from_version }}'
|
||||
TO_URL='${{ inputs.to_url }}'
|
||||
TO_VERSION='${{ inputs.to_version }}'
|
||||
if [ -n "${FROM_URL}" ]; then
|
||||
FROM_SOURCE="${FROM_URL}"
|
||||
elif [ -n "${FROM_VERSION}" ]; then
|
||||
FROM_SOURCE="version ${FROM_VERSION}"
|
||||
else
|
||||
FROM_SOURCE="release (default)"
|
||||
fi
|
||||
if [ -n "${TO_URL}" ]; then
|
||||
TO_SOURCE="${TO_URL}"
|
||||
elif [ -n "${TO_VERSION}" ]; then
|
||||
TO_SOURCE="version ${TO_VERSION}"
|
||||
else
|
||||
TO_SOURCE="${RUSTFS_NIGHTLY_PACKAGE_URL}"
|
||||
fi
|
||||
CASE_TABLE="/tmp/rustfs-upgrade-cases.md"
|
||||
python3 - "${LOG_FILE}" "${CASE_TABLE}" <<'PY'
|
||||
import re
|
||||
import sys
|
||||
|
||||
log_file, out_file = sys.argv[1], sys.argv[2]
|
||||
ansi = re.compile(r'\x1b\[[0-9;]*m')
|
||||
start_re = re.compile(r'^---\s+([A-Z]+-[0-9]+)\s+(.+?)\s+---$')
|
||||
done_re = re.compile(r'^\[(PASS|FAIL|UNSUPPORTED)\]\s+([A-Z]+-[0-9]+)\b')
|
||||
|
||||
rows = []
|
||||
index = {}
|
||||
try:
|
||||
with open(log_file, 'r', encoding='utf-8', errors='replace') as fh:
|
||||
for raw in fh:
|
||||
line = ansi.sub('', raw).strip()
|
||||
m = start_re.match(line)
|
||||
if m:
|
||||
case_id, name = m.group(1), m.group(2)
|
||||
if case_id not in index:
|
||||
index[case_id] = len(rows)
|
||||
rows.append([case_id, name, 'RUNNING'])
|
||||
continue
|
||||
m = done_re.match(line)
|
||||
if m:
|
||||
status, case_id = m.group(1), m.group(2)
|
||||
if case_id in index:
|
||||
rows[index[case_id]][2] = status
|
||||
else:
|
||||
rows.append([case_id, case_id, status])
|
||||
index[case_id] = len(rows) - 1
|
||||
except FileNotFoundError:
|
||||
rows = []
|
||||
|
||||
counts = {'PASS': 0, 'FAIL': 0, 'UNSUPPORTED': 0, 'RUNNING': 0}
|
||||
for _, _, status in rows:
|
||||
counts[status] = counts.get(status, 0) + 1
|
||||
|
||||
with open(out_file, 'w', encoding='utf-8') as out:
|
||||
out.write('## Case Summary\n\n')
|
||||
out.write(f"- Total: {len(rows)}\\n")
|
||||
out.write(f"- PASS: {counts.get('PASS', 0)}\\n")
|
||||
out.write(f"- FAIL: {counts.get('FAIL', 0)}\\n")
|
||||
out.write(f"- UNSUPPORTED: {counts.get('UNSUPPORTED', 0)}\\n")
|
||||
out.write('\\n')
|
||||
out.write('| Case | Name | Status |\\n')
|
||||
out.write('| --- | --- | --- |\\n')
|
||||
for case_id, name, status in rows:
|
||||
out.write(f'| {case_id} | {name} | {status} |\\n')
|
||||
PY
|
||||
{
|
||||
echo "# RustFS upgrade compatibility report"
|
||||
echo ""
|
||||
echo "- Run: ${{ github.server_url }}/${{ github.repository }}/actions/runs/${{ github.run_id }}"
|
||||
echo "- Trigger: ${{ github.event_name }}"
|
||||
echo "- From: ${FROM_SOURCE}"
|
||||
echo "- To: ${TO_SOURCE}"
|
||||
echo "- Test Step Outcome: ${{ steps.test.outcome }}"
|
||||
echo ""
|
||||
cat "${CASE_TABLE}" || true
|
||||
echo ""
|
||||
echo "## Log tail"
|
||||
echo '```text'
|
||||
tail -n 200 "${LOG_FILE}" || true
|
||||
echo '```'
|
||||
} | tee "${REPORT_FILE}"
|
||||
cat "${REPORT_FILE}" >> "${GITHUB_STEP_SUMMARY}"
|
||||
|
||||
- name: Upload functional report to dashboard
|
||||
if: always()
|
||||
continue-on-error: true
|
||||
env:
|
||||
GH_TOKEN: ${{ env.PF_TESTING_GH_TOKEN }}
|
||||
REPORT_FILE: /tmp/rustfs-upgrade-report.md
|
||||
SUITE: upgrade
|
||||
run: |
|
||||
set -euo pipefail
|
||||
if [ -z "${GH_TOKEN:-}" ]; then
|
||||
echo "PF_TESTING_GH_TOKEN is not configured; skipping dashboard upload"
|
||||
exit 0
|
||||
fi
|
||||
DATE="$(date -u +%Y-%m-%d)"
|
||||
REPORT_PATH="functional-reports/${SUITE}/${DATE}.md"
|
||||
CONTENT="$(python3 -c 'import base64,sys;print(base64.b64encode(open(sys.argv[1],"rb").read()).decode())' "${REPORT_FILE}")"
|
||||
SHA="$(gh api "repos/rustfs/dashboard/contents/${REPORT_PATH}" -q '.sha' 2>/dev/null || true)"
|
||||
if [ -n "${SHA}" ]; then
|
||||
jq -n --arg msg "report(${SUITE}): ${DATE}" --arg content "${CONTENT}" --arg sha "${SHA}" \
|
||||
'{message:$msg, content:$content, sha:$sha}' \
|
||||
| gh api --method PUT "repos/rustfs/dashboard/contents/${REPORT_PATH}" --input - >/dev/null
|
||||
else
|
||||
jq -n --arg msg "report(${SUITE}): ${DATE}" --arg content "${CONTENT}" \
|
||||
'{message:$msg, content:$content}' \
|
||||
| gh api --method PUT "repos/rustfs/dashboard/contents/${REPORT_PATH}" --input - >/dev/null
|
||||
fi
|
||||
|
||||
- name: File failure issue in rustfs/backlog
|
||||
if: ${{ always() && (failure() || steps.test.outcome == 'failure' || steps.test.outcome == 'cancelled') }}
|
||||
continue-on-error: true
|
||||
env:
|
||||
GH_TOKEN: ${{ secrets.PF_TESTING_GH_TOKEN }}
|
||||
SUITE: 'upgrade'
|
||||
SUITE_LABEL: 'Upgrade compatibility'
|
||||
RUN_URL: ${{ github.server_url }}/${{ github.repository }}/actions/runs/${{ github.run_id }}
|
||||
REPORT_FILE: '/tmp/rustfs-upgrade-report.md'
|
||||
LOG_FILE: '/tmp/rustfs-upgrade.log'
|
||||
run: |
|
||||
set -euo pipefail
|
||||
if [ -z "${GH_TOKEN:-}" ]; then
|
||||
echo "PF_TESTING_GH_TOKEN is not configured; skipping backlog issue"
|
||||
exit 0
|
||||
fi
|
||||
TITLE="[functional][${SUITE}] ${SUITE_LABEL} suite failed (run ${GITHUB_RUN_ID})"
|
||||
EXISTING="$(gh issue list -R rustfs/backlog --state all \
|
||||
--search "in:title \"run ${GITHUB_RUN_ID}\"" \
|
||||
--json number --jq '.[].number' || true)"
|
||||
if [ -n "${EXISTING}" ]; then
|
||||
echo "backlog issue already exists for run ${GITHUB_RUN_ID}; skipping"
|
||||
exit 0
|
||||
fi
|
||||
redact() {
|
||||
sed -E \
|
||||
-e 's/(RUSTFS_(ACCESS_KEY|SECRET_KEY)[=: ]+)[^[:space:]]+/\1[REDACTED]/Ig' \
|
||||
-e 's/(Authorization:).*/\1 [REDACTED]/Ig' \
|
||||
-e 's/(X-Amz-Signature=)[^&[:space:]]+/\1[REDACTED]/Ig' \
|
||||
-e 's/^.*(password|secret|token)[=: ].*/[REDACTED SENSITIVE LINE]/Ig'
|
||||
}
|
||||
BODY_FILE="$(mktemp)"
|
||||
{
|
||||
echo "The **${SUITE_LABEL}** functional suite failed."
|
||||
echo ""
|
||||
echo "- Suite: \`${SUITE}\`"
|
||||
echo "- Run: ${RUN_URL}"
|
||||
echo "- Trigger: ${GITHUB_EVENT_NAME}"
|
||||
echo "- Date: $(date -u +%Y-%m-%d)"
|
||||
echo ""
|
||||
echo "## Report (errors and symptoms)"
|
||||
echo ""
|
||||
if [ -s "${REPORT_FILE}" ]; then
|
||||
redact < "${REPORT_FILE}"
|
||||
elif [ -s "${LOG_FILE:-}" ]; then
|
||||
echo "(report file missing; log tail below)"
|
||||
echo ""
|
||||
tail -n 200 "${LOG_FILE}" | redact
|
||||
else
|
||||
echo "(no report or log file was produced)"
|
||||
fi
|
||||
} | head -c 55000 > "${BODY_FILE}"
|
||||
gh label create functional-test -R rustfs/backlog --color d73a4a 2>/dev/null || true
|
||||
if ! gh issue create -R rustfs/backlog --title "${TITLE}" \
|
||||
--body-file "${BODY_FILE}" --label functional-test; then
|
||||
gh issue create -R rustfs/backlog --title "${TITLE}" --body-file "${BODY_FILE}"
|
||||
fi
|
||||
echo "filed backlog issue for suite ${SUITE}"
|
||||
|
||||
- name: Upload report and logs
|
||||
if: always()
|
||||
uses: actions/upload-artifact@b7c566a772e6b6bfb58ed0dc250532a479d7789f # v6
|
||||
with:
|
||||
name: rustfs-upgrade-test-${{ github.run_id }}
|
||||
path: |
|
||||
/tmp/rustfs-upgrade-report.md
|
||||
/tmp/rustfs-upgrade.*/*
|
||||
if-no-files-found: ignore
|
||||
retention-days: 3
|
||||
|
||||
- name: Cleanup environment (after)
|
||||
if: ${{ always() && (inputs.cleanup_after != 'false' || github.event_name != 'workflow_dispatch') }}
|
||||
run: |
|
||||
set -euo pipefail
|
||||
read -r -a NODES <<< "${RUSTFS_NODES:-vm000 vm001 vm002}"
|
||||
SSH_USER="${RUSTFS_SSH_USER:-azureuser}"
|
||||
for node in "${NODES[@]}"; do
|
||||
ssh -o BatchMode=yes -o ConnectTimeout=10 -o StrictHostKeyChecking=accept-new "${SSH_USER}@${node}" '
|
||||
set -euo pipefail
|
||||
SUDO=""; [ "$(id -u)" -ne 0 ] && SUDO="sudo -n"
|
||||
${SUDO} systemctl stop rustfs 2>/dev/null || true
|
||||
if ${SUDO} dpkg -l rustfs 2>/dev/null | grep -q "^ii"; then
|
||||
${SUDO} dpkg -P rustfs
|
||||
fi
|
||||
for i in 1 2 3 4; do ${SUDO} rm -rf /data/rustfs${i}/mnmd; done
|
||||
${SUDO} rm -rf /var/log/rustfs /var/lib/rustfs/kms /var/lib/rustfs/kms-backup
|
||||
'
|
||||
done
|
||||
|
||||
- name: "Continue functional chain (next: S3 compatibility)"
|
||||
# Only chain-triggered runs forward to the next suite; standalone
|
||||
# workflow_dispatch runs stop after their own cleanup. A failed
|
||||
# handoff must never pass silently: it retries, then files an alert
|
||||
# issue in rustfs/backlog so a stalled chain is visible.
|
||||
if: ${{ always() && github.event_name == 'repository_dispatch' }}
|
||||
continue-on-error: true
|
||||
env:
|
||||
GH_TOKEN: ${{ secrets.PF_TESTING_GH_TOKEN }}
|
||||
run: |
|
||||
set -uo pipefail
|
||||
if [ -z "${{GH_TOKEN:-}}" ]; then
|
||||
echo "PF_TESTING_GH_TOKEN is not configured; cannot dispatch the next suite" >&2
|
||||
exit 1
|
||||
fi
|
||||
DISPATCHED=0
|
||||
for attempt in 1 2 3; do
|
||||
if gh api --method POST repos/rustfs/rustfs/dispatches \
|
||||
-f event_type='rustfs-chain-s3' \
|
||||
-F 'client_payload[from_suite]=upgrade'; then
|
||||
echo "dispatched next suite S3 compatibility (attempt ${{attempt}})"
|
||||
DISPATCHED=1
|
||||
break
|
||||
fi
|
||||
echo "dispatch attempt ${{attempt}} failed; retrying in ${{attempt}}0s" >&2
|
||||
sleep "${{attempt}}0"
|
||||
done
|
||||
if [ "${{DISPATCHED:-0}}" -ne 1 ]; then
|
||||
echo "ERROR: functional chain stalled: could not dispatch S3 compatibility after 3 attempts" >&2
|
||||
TITLE="[functional][chain] stalled after upgrade (run ${{GITHUB_RUN_ID}})"
|
||||
BODY_FILE="$(mktemp)"
|
||||
{
|
||||
echo "The functional chain could not hand off from **upgrade** to **S3 compatibility** after 3 attempts."
|
||||
echo ""
|
||||
echo "- Failed suite job: ${{GITHUB_SERVER_URL}}/${{GITHUB_REPOSITORY}}/actions/runs/${{GITHUB_RUN_ID}}"
|
||||
echo "- Expected next event: `rustfs-chain-s3`"
|
||||
echo "- Likely cause: PF_TESTING_GH_TOKEN lacks contents:write on rustfs/rustfs, or the GitHub API was unavailable."
|
||||
echo "- Recovery: re-dispatch manually with"
|
||||
echo " ```"
|
||||
echo " gh api --method POST repos/rustfs/rustfs/dispatches -f event_type='rustfs-chain-s3'"
|
||||
echo " ```"
|
||||
} > "${{BODY_FILE}}"
|
||||
gh issue create -R rustfs/backlog --title "${{TITLE}}" \
|
||||
--body-file "${{BODY_FILE}}" --label functional-test \
|
||||
|| gh issue create -R rustfs/backlog --title "${{TITLE}}" --body-file "${{BODY_FILE}}" \
|
||||
|| echo "could not file the stall alert issue either; check the token" >&2
|
||||
exit 1
|
||||
fi
|
||||
|
||||
- name: Notify on failure
|
||||
if: failure()
|
||||
run: |
|
||||
echo "RustFS upgrade compatibility test failed"
|
||||
echo "From: ${{ inputs.from_url || inputs.from_version || 'release (default)' }}"
|
||||
echo "To: ${{ inputs.to_url || inputs.to_version || 'nightly (R2 latest)' }}"
|
||||
echo "See the uploaded report and logs for details."
|
||||
@@ -123,12 +123,13 @@ runtime/build output:
|
||||
- Use `make pre-commit` only when its repository-wide fast checks add confidence
|
||||
beyond the focused checks.
|
||||
|
||||
### Broad or High-Risk Changes
|
||||
### Broad Cross-Module Changes
|
||||
|
||||
After the required adversarial review, run `make pre-pr` when targeted coverage
|
||||
cannot bound the impact, including dependency/toolchain/build-matrix changes,
|
||||
unbounded cross-crate APIs, or locking, durability, erasure coding, replication,
|
||||
RPC, IAM/KMS/auth, cryptography, on-disk/on-wire, and S3-visible behavior.
|
||||
Do not run `make pre-pr` by default before opening a PR. Consider it only when
|
||||
the final diff is broad, spans multiple modules, and targeted checks cannot
|
||||
bound the impact. Decide dynamically from the affected boundaries and risks;
|
||||
otherwise use the scoped formatting, linting, compilation, and test checks
|
||||
above.
|
||||
|
||||
`make pre-pr` includes `make pre-commit`; never run both for the same unchanged
|
||||
diff. Do not repeat a check already covered by a successful umbrella gate.
|
||||
|
||||
@@ -15,7 +15,7 @@ cargo check -p <crate> # fast type-check one crate
|
||||
cargo test -p <crate> # test one crate
|
||||
cargo fmt --all # format (required before PR)
|
||||
make pre-commit # fast gate: fmt + arch checks + quick-check (NO clippy/tests)
|
||||
make pre-pr # full pre-PR gate: fmt + arch checks + clippy + tests
|
||||
make pre-pr # optional full gate for broad cross-module changes
|
||||
make build-docker BUILD_OS=ubuntu22.04
|
||||
```
|
||||
|
||||
|
||||
+20
-7
@@ -62,12 +62,20 @@ make test
|
||||
# Fast pre-commit gate — see below for exactly what it runs
|
||||
make pre-commit
|
||||
|
||||
# Full pre-PR gate (pre-commit gates + clippy + tests)
|
||||
# Optional full gate for broad cross-module changes (pre-commit + clippy + tests)
|
||||
make pre-pr
|
||||
```
|
||||
|
||||
> `make test` requires [cargo-nextest](https://nexte.st) (CI runs it and only nextest honours `.config/nextest.toml` test-groups). Install it with `cargo install cargo-nextest --locked` or a prebuilt binary (see https://nexte.st/docs/installation/). To run the plain `cargo test` fallback anyway (results not authoritative — serialization semantics differ from CI), set `RUSTFS_ALLOW_CARGO_TEST_FALLBACK=1`.
|
||||
|
||||
> Some guard checks are Python (`test-wiring-check` in `make pre-commit`, plus the
|
||||
> security-coverage and scheduled-validation self-tests in `make test`) and import
|
||||
> `tomllib`, so they need **Python 3.11+**. Make resolves the interpreter through
|
||||
> `scripts/python_bin.sh`, which prefers a `python3.11`+ on `PATH` and otherwise falls
|
||||
> back to `uv run --python 3.12`. macOS ships `/usr/bin/python3` at 3.9, so install a
|
||||
> newer one (`brew install python@3.12`) or [uv](https://docs.astral.sh/uv/); pin a
|
||||
> specific interpreter with `RUSTFS_PYTHON=/path/to/python3.12`.
|
||||
|
||||
> For the full test-layer taxonomy (unit / ecstore black-box / e2e / s3s-e2e / S3 compatibility / chaos / fuzz / bench), each layer's entry command, the naming conventions the migration gate depends on, and the serial/nextest rules, see [docs/testing/README.md](docs/testing/README.md).
|
||||
|
||||
> For the event, timeout, required-status, and local reproduction matrix, see [docs/testing/ci-gates.md](docs/testing/ci-gates.md).
|
||||
@@ -88,14 +96,16 @@ make pre-pr
|
||||
8. `quick-check` — `cargo check --workspace --exclude e2e_test`
|
||||
|
||||
**`make pre-commit` does NOT run clippy and does NOT run any tests.**
|
||||
A green `make pre-commit` is not enough to open a pull request.
|
||||
It does not replace the scoped Clippy and test checks applicable to a change.
|
||||
|
||||
`make pre-pr` is the **full** gate: it runs all of the guard checks above,
|
||||
then `clippy-check` (`cargo clippy --all-targets --all-features -- -D warnings`)
|
||||
and `test` (shell script tests, workspace tests excluding `e2e_test`, and doc
|
||||
tests). Complete the applicable multi-role adversarial review described in
|
||||
`AGENTS.md` before running `make pre-pr`; then run the gate before opening or
|
||||
updating a pull request. This is what CI enforces.
|
||||
`AGENTS.md` first. Do not run `make pre-pr` locally by default before opening or
|
||||
updating a pull request. Consider it only for a broad change that spans multiple
|
||||
modules and whose impact cannot be bounded by targeted checks; decide from the
|
||||
affected boundaries and risks. CI still runs its configured repository gates.
|
||||
|
||||
### 🔒 Git Pre-commit Hooks (optional)
|
||||
|
||||
@@ -114,8 +124,9 @@ Or manually:
|
||||
chmod +x .git/hooks/pre-commit
|
||||
```
|
||||
|
||||
With or without a hook, the expectation is the same: run `make pre-commit`
|
||||
before committing and `make pre-pr` before opening a pull request.
|
||||
With or without a hook, follow the verification tiers in `AGENTS.md`. Run the
|
||||
applicable scoped checks, and reserve `make pre-pr` for broad cross-module
|
||||
changes whose impact cannot be bounded by those checks.
|
||||
|
||||
### 📝 Formatting Configuration
|
||||
|
||||
@@ -154,7 +165,9 @@ Example output when formatting fails:
|
||||
3. **Run the fast gate**: `make pre-commit` (no clippy, no tests)
|
||||
4. **Commit your changes**: `git commit -m "your message"`
|
||||
5. **Complete the applicable multi-role adversarial review** for non-exempt changes (see `AGENTS.md`)
|
||||
6. **Run the full gate before opening/updating a PR**: `make pre-pr` (clippy + tests)
|
||||
6. **Run applicable scoped checks before opening/updating a PR**; consider
|
||||
`make pre-pr` only for broad cross-module changes whose impact cannot be
|
||||
bounded by targeted checks
|
||||
7. **Push to your branch**: `git push`
|
||||
|
||||
### 🛠️ IDE Integration
|
||||
|
||||
Generated
+119
-81
@@ -627,8 +627,7 @@ dependencies = [
|
||||
[[package]]
|
||||
name = "astral-tokio-tar"
|
||||
version = "0.7.0"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "6f2e989b33246fe9240d39accf4dd9a01e0b6c1f3ce9dd095e0a47fa02505523"
|
||||
source = "git+https://github.com/cxymds/tokio-tar.git?rev=603756478b7668436e464519c77ccac22a99ba96#603756478b7668436e464519c77ccac22a99ba96"
|
||||
dependencies = [
|
||||
"futures-core",
|
||||
"libc",
|
||||
@@ -2163,6 +2162,7 @@ dependencies = [
|
||||
"compression-core",
|
||||
"flate2",
|
||||
"liblzma",
|
||||
"lz4",
|
||||
"memchr",
|
||||
"zstd",
|
||||
"zstd-safe",
|
||||
@@ -3916,7 +3916,7 @@ checksum = "d0881ea181b1df73ff77ffaaf9c7544ecc11e82fba9b5f27b262a3c73a332555"
|
||||
|
||||
[[package]]
|
||||
name = "e2e_test"
|
||||
version = "1.0.0-rc.4"
|
||||
version = "1.0.0-rc.5"
|
||||
dependencies = [
|
||||
"anyhow",
|
||||
"astral-tokio-tar",
|
||||
@@ -3926,6 +3926,7 @@ dependencies = [
|
||||
"aws-sdk-s3",
|
||||
"aws-sdk-sts",
|
||||
"aws-smithy-http-client",
|
||||
"aws-smithy-types",
|
||||
"base64-simd",
|
||||
"bytes",
|
||||
"chrono",
|
||||
@@ -3943,6 +3944,7 @@ dependencies = [
|
||||
"hyper-util",
|
||||
"local-ip-address",
|
||||
"md-5 0.11.0",
|
||||
"minlz",
|
||||
"opentelemetry-proto",
|
||||
"prost 0.14.4",
|
||||
"rand 0.10.2",
|
||||
@@ -4208,7 +4210,7 @@ source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "39cab71617ae0d63f51a36d69f866391735b51691dbda63cf6f96d042b63efeb"
|
||||
dependencies = [
|
||||
"libc",
|
||||
"windows-sys 0.52.0",
|
||||
"windows-sys 0.61.2",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
@@ -5037,9 +5039,9 @@ checksum = "2304e00983f87ffb38b55b444b5e3b60a884b5d30c0fca7d82fe33449bbe55ea"
|
||||
|
||||
[[package]]
|
||||
name = "hermit-abi"
|
||||
version = "0.5.2"
|
||||
version = "0.5.3"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "fc0fef456e4baa96da950455cd02c081ca953b141298e41db3fc7e36b1da849c"
|
||||
checksum = "e17592d60ebacc7d5e169f4663c5f84f9161cc90328abcfe8456f41e4dfcb284"
|
||||
|
||||
[[package]]
|
||||
name = "hex"
|
||||
@@ -5666,7 +5668,7 @@ checksum = "3640c1c38b8e4e43584d8df18be5fc6b0aa314ce6ebf51b53313d4306cca8e46"
|
||||
dependencies = [
|
||||
"hermit-abi",
|
||||
"libc",
|
||||
"windows-sys 0.52.0",
|
||||
"windows-sys 0.61.2",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
@@ -6648,10 +6650,11 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "mysql_async"
|
||||
version = "0.37.0"
|
||||
version = "0.37.1"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "3519e91b0d254ac1ffa495bc42053286cb2172ad7241d5b3b1b9f8a891f21ee2"
|
||||
checksum = "40d11da0e2d9fad4640c9f9198ee431c6d68444568f83ef1f10f3367270071e4"
|
||||
dependencies = [
|
||||
"arc-swap",
|
||||
"bytes",
|
||||
"crossbeam-queue",
|
||||
"crossbeam-utils",
|
||||
@@ -6984,9 +6987,9 @@ checksum = "a3c00a0c9600379bd32f8972de90676a7672cba3bf4886986bc05902afc1e093"
|
||||
|
||||
[[package]]
|
||||
name = "nvml-wrapper"
|
||||
version = "0.12.1"
|
||||
version = "0.13.0"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "f049ae562349fefb8e837eb15443da1e7c6dcbd8a11f52a228f92220c2e5c85e"
|
||||
checksum = "d164abbde0b3c03edb9edb9cb8d31a7f5b79015c692b7c771f6e0840e9106b9f"
|
||||
dependencies = [
|
||||
"bitflags 2.13.1",
|
||||
"libloading",
|
||||
@@ -6998,9 +7001,9 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "nvml-wrapper-sys"
|
||||
version = "0.9.1"
|
||||
version = "0.10.0"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "6b4d594420fcda43b1c2c4bd44d48974aa3c7a9ab2cbf10dc18e35265767bf0b"
|
||||
checksum = "5d2079f4c9b6d2170bfb71c6355734ead6c47da75c179847395c31f9f2f66ede"
|
||||
dependencies = [
|
||||
"libloading",
|
||||
]
|
||||
@@ -7011,7 +7014,7 @@ version = "5.0.0"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "51e219e79014df21a225b1860a479e2dcd7cbd9130f4defd4bd0e191ea31d67d"
|
||||
dependencies = [
|
||||
"base64 0.22.1",
|
||||
"base64 0.21.7",
|
||||
"chrono",
|
||||
"getrandom 0.2.17",
|
||||
"http 1.5.0",
|
||||
@@ -8034,9 +8037,9 @@ checksum = "439ee305def115ba05938db6eb1644ff94165c5ab5e9420d1c1bcedbba909391"
|
||||
|
||||
[[package]]
|
||||
name = "ppmd-rust"
|
||||
version = "1.4.0"
|
||||
version = "1.4.1"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "efca4c95a19a79d1c98f791f10aebd5c1363b473244630bb7dbde1dc98455a24"
|
||||
checksum = "9e9219bcb9d7aca6b2f63c83cf100cf78bcd619ac46e6ecbd0dd90869a39345d"
|
||||
|
||||
[[package]]
|
||||
name = "ppv-lite86"
|
||||
@@ -8238,7 +8241,7 @@ version = "0.13.5"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "be769465445e8c1474e9c5dac2018218498557af32d9ed057325ec9a41ae81bf"
|
||||
dependencies = [
|
||||
"heck 0.5.0",
|
||||
"heck 0.4.1",
|
||||
"itertools 0.14.0",
|
||||
"log",
|
||||
"multimap",
|
||||
@@ -8258,7 +8261,7 @@ version = "0.14.4"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "03da047801ff44bb6a4d407d4860c05fd70bb81714e6b2f3812603d5b145b042"
|
||||
dependencies = [
|
||||
"heck 0.5.0",
|
||||
"heck 0.4.1",
|
||||
"itertools 0.14.0",
|
||||
"log",
|
||||
"multimap",
|
||||
@@ -8608,7 +8611,7 @@ dependencies = [
|
||||
"once_cell",
|
||||
"socket2",
|
||||
"tracing",
|
||||
"windows-sys 0.52.0",
|
||||
"windows-sys 0.61.2",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
@@ -9384,7 +9387,7 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "rustfs"
|
||||
version = "1.0.0-rc.4"
|
||||
version = "1.0.0-rc.5"
|
||||
dependencies = [
|
||||
"aes-gcm",
|
||||
"anyhow",
|
||||
@@ -9459,6 +9462,7 @@ dependencies = [
|
||||
"rustfs-io-metrics",
|
||||
"rustfs-keystone",
|
||||
"rustfs-kms",
|
||||
"rustfs-license",
|
||||
"rustfs-lock",
|
||||
"rustfs-log-analyzer",
|
||||
"rustfs-madmin",
|
||||
@@ -9510,7 +9514,7 @@ dependencies = [
|
||||
"tokio-util",
|
||||
"tonic",
|
||||
"tower",
|
||||
"tower-http 0.7.0",
|
||||
"tower-http 0.7.1",
|
||||
"tracing",
|
||||
"tracing-opentelemetry",
|
||||
"tracing-subscriber",
|
||||
@@ -9525,7 +9529,7 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "rustfs-audit"
|
||||
version = "1.0.0-rc.4"
|
||||
version = "1.0.0-rc.5"
|
||||
dependencies = [
|
||||
"const-str",
|
||||
"futures",
|
||||
@@ -9547,7 +9551,7 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "rustfs-checksums"
|
||||
version = "1.0.0-rc.4"
|
||||
version = "1.0.0-rc.5"
|
||||
dependencies = [
|
||||
"base64-simd",
|
||||
"bytes",
|
||||
@@ -9563,7 +9567,7 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "rustfs-common"
|
||||
version = "1.0.0-rc.4"
|
||||
version = "1.0.0-rc.5"
|
||||
dependencies = [
|
||||
"hotpath",
|
||||
"metrics",
|
||||
@@ -9576,7 +9580,7 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "rustfs-concurrency"
|
||||
version = "1.0.0-rc.4"
|
||||
version = "1.0.0-rc.5"
|
||||
dependencies = [
|
||||
"hotpath",
|
||||
"insta",
|
||||
@@ -9589,7 +9593,7 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "rustfs-config"
|
||||
version = "1.0.0-rc.4"
|
||||
version = "1.0.0-rc.5"
|
||||
dependencies = [
|
||||
"const-str",
|
||||
"hotpath",
|
||||
@@ -9599,7 +9603,7 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "rustfs-credentials"
|
||||
version = "1.0.0-rc.4"
|
||||
version = "1.0.0-rc.5"
|
||||
dependencies = [
|
||||
"base64-simd",
|
||||
"hmac 0.13.0",
|
||||
@@ -9613,7 +9617,7 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "rustfs-crypto"
|
||||
version = "1.0.0-rc.4"
|
||||
version = "1.0.0-rc.5"
|
||||
dependencies = [
|
||||
"aes-gcm",
|
||||
"argon2",
|
||||
@@ -9634,7 +9638,7 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "rustfs-data-usage"
|
||||
version = "1.0.0-rc.4"
|
||||
version = "1.0.0-rc.5"
|
||||
dependencies = [
|
||||
"hotpath",
|
||||
"rmp-serde",
|
||||
@@ -9644,7 +9648,7 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "rustfs-ecstore"
|
||||
version = "1.0.0-rc.4"
|
||||
version = "1.0.0-rc.5"
|
||||
dependencies = [
|
||||
"arc-swap",
|
||||
"async-channel",
|
||||
@@ -9779,7 +9783,7 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "rustfs-extension-schema"
|
||||
version = "1.0.0-rc.4"
|
||||
version = "1.0.0-rc.5"
|
||||
dependencies = [
|
||||
"hotpath",
|
||||
"serde",
|
||||
@@ -9789,7 +9793,7 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "rustfs-filemeta"
|
||||
version = "1.0.0-rc.4"
|
||||
version = "1.0.0-rc.5"
|
||||
dependencies = [
|
||||
"arc-swap",
|
||||
"byteorder",
|
||||
@@ -9816,7 +9820,7 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "rustfs-heal"
|
||||
version = "1.0.0-rc.4"
|
||||
version = "1.0.0-rc.5"
|
||||
dependencies = [
|
||||
"async-trait",
|
||||
"base64-simd",
|
||||
@@ -9852,7 +9856,7 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "rustfs-heal-contracts"
|
||||
version = "1.0.0-rc.4"
|
||||
version = "1.0.0-rc.5"
|
||||
dependencies = [
|
||||
"serde",
|
||||
"serde_json",
|
||||
@@ -9862,7 +9866,7 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "rustfs-iam"
|
||||
version = "1.0.0-rc.4"
|
||||
version = "1.0.0-rc.5"
|
||||
dependencies = [
|
||||
"arc-swap",
|
||||
"async-trait",
|
||||
@@ -9910,7 +9914,7 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "rustfs-io-core"
|
||||
version = "1.0.0-rc.4"
|
||||
version = "1.0.0-rc.5"
|
||||
dependencies = [
|
||||
"bytes",
|
||||
"hotpath",
|
||||
@@ -9922,7 +9926,7 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "rustfs-io-metrics"
|
||||
version = "1.0.0-rc.4"
|
||||
version = "1.0.0-rc.5"
|
||||
dependencies = [
|
||||
"criterion",
|
||||
"hotpath",
|
||||
@@ -9986,7 +9990,7 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "rustfs-keystone"
|
||||
version = "1.0.0-rc.4"
|
||||
version = "1.0.0-rc.5"
|
||||
dependencies = [
|
||||
"bytes",
|
||||
"futures",
|
||||
@@ -10013,7 +10017,7 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "rustfs-kms"
|
||||
version = "1.0.0-rc.4"
|
||||
version = "1.0.0-rc.5"
|
||||
dependencies = [
|
||||
"aes-gcm",
|
||||
"anyhow",
|
||||
@@ -10061,9 +10065,16 @@ dependencies = [
|
||||
"zeroize",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "rustfs-license"
|
||||
version = "1.0.0-rc.5"
|
||||
dependencies = [
|
||||
"thiserror 2.0.20",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "rustfs-lifecycle"
|
||||
version = "1.0.0-rc.4"
|
||||
version = "1.0.0-rc.5"
|
||||
dependencies = [
|
||||
"async-trait",
|
||||
"hotpath",
|
||||
@@ -10086,7 +10097,7 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "rustfs-lock"
|
||||
version = "1.0.0-rc.4"
|
||||
version = "1.0.0-rc.5"
|
||||
dependencies = [
|
||||
"async-trait",
|
||||
"compact_str",
|
||||
@@ -10109,7 +10120,7 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "rustfs-log-analyzer"
|
||||
version = "1.0.0-rc.4"
|
||||
version = "1.0.0-rc.5"
|
||||
dependencies = [
|
||||
"chrono",
|
||||
"flate2",
|
||||
@@ -10128,7 +10139,7 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "rustfs-madmin"
|
||||
version = "1.0.0-rc.4"
|
||||
version = "1.0.0-rc.5"
|
||||
dependencies = [
|
||||
"hotpath",
|
||||
"http 1.5.0",
|
||||
@@ -10166,7 +10177,7 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "rustfs-notify"
|
||||
version = "1.0.0-rc.4"
|
||||
version = "1.0.0-rc.5"
|
||||
dependencies = [
|
||||
"arc-swap",
|
||||
"async-trait",
|
||||
@@ -10201,7 +10212,7 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "rustfs-object-capacity"
|
||||
version = "1.0.0-rc.4"
|
||||
version = "1.0.0-rc.5"
|
||||
dependencies = [
|
||||
"criterion",
|
||||
"futures",
|
||||
@@ -10220,7 +10231,7 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "rustfs-object-data-cache"
|
||||
version = "1.0.0-rc.4"
|
||||
version = "1.0.0-rc.5"
|
||||
dependencies = [
|
||||
"bytes",
|
||||
"criterion",
|
||||
@@ -10237,7 +10248,7 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "rustfs-obs"
|
||||
version = "1.0.0-rc.4"
|
||||
version = "1.0.0-rc.5"
|
||||
dependencies = [
|
||||
"chrono",
|
||||
"crossbeam-channel",
|
||||
@@ -10295,7 +10306,7 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "rustfs-policy"
|
||||
version = "1.0.0-rc.4"
|
||||
version = "1.0.0-rc.5"
|
||||
dependencies = [
|
||||
"async-trait",
|
||||
"base64-simd",
|
||||
@@ -10326,7 +10337,7 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "rustfs-protocols"
|
||||
version = "1.0.0-rc.4"
|
||||
version = "1.0.0-rc.5"
|
||||
dependencies = [
|
||||
"astral-tokio-tar",
|
||||
"async-compression",
|
||||
@@ -10388,7 +10399,7 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "rustfs-protos"
|
||||
version = "1.0.0-rc.4"
|
||||
version = "1.0.0-rc.5"
|
||||
dependencies = [
|
||||
"flatbuffers",
|
||||
"hotpath",
|
||||
@@ -10413,7 +10424,7 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "rustfs-replication"
|
||||
version = "1.0.0-rc.4"
|
||||
version = "1.0.0-rc.5"
|
||||
dependencies = [
|
||||
"byteorder",
|
||||
"bytes",
|
||||
@@ -10431,7 +10442,7 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "rustfs-rio"
|
||||
version = "1.0.0-rc.4"
|
||||
version = "1.0.0-rc.5"
|
||||
dependencies = [
|
||||
"aes-gcm",
|
||||
"arc-swap",
|
||||
@@ -10448,6 +10459,7 @@ dependencies = [
|
||||
"hyper",
|
||||
"hyper-util",
|
||||
"md-5 0.11.0",
|
||||
"minlz",
|
||||
"pin-project-lite",
|
||||
"rand 0.10.2",
|
||||
"reqwest",
|
||||
@@ -10471,7 +10483,7 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "rustfs-rio-v2"
|
||||
version = "1.0.0-rc.4"
|
||||
version = "1.0.0-rc.5"
|
||||
dependencies = [
|
||||
"aes-gcm",
|
||||
"bytes",
|
||||
@@ -10494,7 +10506,7 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "rustfs-s3-client"
|
||||
version = "1.0.0-rc.4"
|
||||
version = "1.0.0-rc.5"
|
||||
dependencies = [
|
||||
"base64-simd",
|
||||
"bytes",
|
||||
@@ -10538,7 +10550,7 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "rustfs-s3-ops"
|
||||
version = "1.0.0-rc.4"
|
||||
version = "1.0.0-rc.5"
|
||||
dependencies = [
|
||||
"hotpath",
|
||||
"rustfs-s3-types",
|
||||
@@ -10546,7 +10558,7 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "rustfs-s3-types"
|
||||
version = "1.0.0-rc.4"
|
||||
version = "1.0.0-rc.5"
|
||||
dependencies = [
|
||||
"hotpath",
|
||||
"serde",
|
||||
@@ -10555,12 +10567,16 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "rustfs-s3select-api"
|
||||
version = "1.0.0-rc.4"
|
||||
version = "1.0.0-rc.5"
|
||||
dependencies = [
|
||||
"arc-swap",
|
||||
"async-compression",
|
||||
"async-trait",
|
||||
"bytes",
|
||||
"chrono",
|
||||
"crc-fast",
|
||||
"datafusion",
|
||||
"flate2",
|
||||
"futures",
|
||||
"futures-core",
|
||||
"hotpath",
|
||||
@@ -10576,6 +10592,7 @@ dependencies = [
|
||||
"serial_test",
|
||||
"thiserror 2.0.20",
|
||||
"tokio",
|
||||
"tokio-stream",
|
||||
"tokio-util",
|
||||
"tracing",
|
||||
"transform-stream",
|
||||
@@ -10585,7 +10602,7 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "rustfs-s3select-query"
|
||||
version = "1.0.0-rc.4"
|
||||
version = "1.0.0-rc.5"
|
||||
dependencies = [
|
||||
"async-recursion",
|
||||
"async-trait",
|
||||
@@ -10598,13 +10615,15 @@ dependencies = [
|
||||
"rustfs-s3select-api",
|
||||
"rustfs-test-utils",
|
||||
"s3s",
|
||||
"serde",
|
||||
"serde_json",
|
||||
"tokio",
|
||||
"tracing",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "rustfs-scanner"
|
||||
version = "1.0.0-rc.4"
|
||||
version = "1.0.0-rc.5"
|
||||
dependencies = [
|
||||
"async-trait",
|
||||
"bytes",
|
||||
@@ -10647,7 +10666,7 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "rustfs-scanner-contracts"
|
||||
version = "1.0.0-rc.4"
|
||||
version = "1.0.0-rc.5"
|
||||
dependencies = [
|
||||
"chrono",
|
||||
"jiff",
|
||||
@@ -10662,7 +10681,7 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "rustfs-security-governance"
|
||||
version = "1.0.0-rc.4"
|
||||
version = "1.0.0-rc.5"
|
||||
dependencies = [
|
||||
"hotpath",
|
||||
"thiserror 2.0.20",
|
||||
@@ -10670,7 +10689,7 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "rustfs-signer"
|
||||
version = "1.0.0-rc.4"
|
||||
version = "1.0.0-rc.5"
|
||||
dependencies = [
|
||||
"base64-simd",
|
||||
"bytes",
|
||||
@@ -10688,7 +10707,7 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "rustfs-storage-api"
|
||||
version = "1.0.0-rc.4"
|
||||
version = "1.0.0-rc.5"
|
||||
dependencies = [
|
||||
"async-trait",
|
||||
"hotpath",
|
||||
@@ -10703,7 +10722,7 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "rustfs-targets"
|
||||
version = "1.0.0-rc.4"
|
||||
version = "1.0.0-rc.5"
|
||||
dependencies = [
|
||||
"arc-swap",
|
||||
"async-nats",
|
||||
@@ -10757,7 +10776,7 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "rustfs-test-utils"
|
||||
version = "1.0.0-rc.4"
|
||||
version = "1.0.0-rc.5"
|
||||
dependencies = [
|
||||
"hotpath",
|
||||
"rustfs-data-usage",
|
||||
@@ -10773,7 +10792,7 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "rustfs-tls-runtime"
|
||||
version = "1.0.0-rc.4"
|
||||
version = "1.0.0-rc.5"
|
||||
dependencies = [
|
||||
"arc-swap",
|
||||
"hotpath",
|
||||
@@ -10794,7 +10813,7 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "rustfs-trusted-proxies"
|
||||
version = "1.0.0-rc.4"
|
||||
version = "1.0.0-rc.5"
|
||||
dependencies = [
|
||||
"async-trait",
|
||||
"axum",
|
||||
@@ -10831,7 +10850,7 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "rustfs-utils"
|
||||
version = "1.0.0-rc.4"
|
||||
version = "1.0.0-rc.5"
|
||||
dependencies = [
|
||||
"base64-simd",
|
||||
"blake2",
|
||||
@@ -10873,10 +10892,11 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "rustfs-zip"
|
||||
version = "1.0.0-rc.4"
|
||||
version = "1.0.0-rc.5"
|
||||
dependencies = [
|
||||
"async-compression",
|
||||
"hotpath",
|
||||
"rustfs-rio",
|
||||
"thiserror 2.0.20",
|
||||
"tokio",
|
||||
]
|
||||
@@ -10934,7 +10954,7 @@ dependencies = [
|
||||
"errno",
|
||||
"libc",
|
||||
"linux-raw-sys",
|
||||
"windows-sys 0.52.0",
|
||||
"windows-sys 0.61.2",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
@@ -11007,7 +11027,7 @@ dependencies = [
|
||||
"security-framework",
|
||||
"security-framework-sys",
|
||||
"webpki-root-certs",
|
||||
"windows-sys 0.52.0",
|
||||
"windows-sys 0.61.2",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
@@ -11064,7 +11084,7 @@ checksum = "9774ba4a74de5f7b1c1451ed6cd5285a32eddb5cccb8cc655a4e50009e06477f"
|
||||
[[package]]
|
||||
name = "s3s"
|
||||
version = "0.15.0"
|
||||
source = "git+https://github.com/rustfs/s3s.git?rev=9c4690d8e73fc8d184031a19b2c4539ebc77d180#9c4690d8e73fc8d184031a19b2c4539ebc77d180"
|
||||
source = "git+https://github.com/rustfs/s3s.git?rev=28e9ebb23dd2fb7d667084f34121b4aa4807a5c6#28e9ebb23dd2fb7d667084f34121b4aa4807a5c6"
|
||||
dependencies = [
|
||||
"arc-swap",
|
||||
"arrayvec",
|
||||
@@ -11095,6 +11115,7 @@ dependencies = [
|
||||
"pin-project-lite",
|
||||
"quick-xml",
|
||||
"regex",
|
||||
"s3s-rfc2047",
|
||||
"s3s-sigv2",
|
||||
"s3s-sigv4",
|
||||
"serde",
|
||||
@@ -11118,28 +11139,45 @@ dependencies = [
|
||||
"zeroize",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "s3s-rfc2047"
|
||||
version = "0.16.0-alpha.1"
|
||||
source = "git+https://github.com/rustfs/s3s.git?rev=28e9ebb23dd2fb7d667084f34121b4aa4807a5c6#28e9ebb23dd2fb7d667084f34121b4aa4807a5c6"
|
||||
dependencies = [
|
||||
"base64-simd",
|
||||
"thiserror 2.0.20",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "s3s-sigv2"
|
||||
version = "0.16.0-alpha.1"
|
||||
source = "git+https://github.com/rustfs/s3s.git?rev=9c4690d8e73fc8d184031a19b2c4539ebc77d180#9c4690d8e73fc8d184031a19b2c4539ebc77d180"
|
||||
source = "git+https://github.com/rustfs/s3s.git?rev=28e9ebb23dd2fb7d667084f34121b4aa4807a5c6#28e9ebb23dd2fb7d667084f34121b4aa4807a5c6"
|
||||
dependencies = [
|
||||
"base64-simd",
|
||||
"hmac 0.13.0",
|
||||
"jiff",
|
||||
"sha1 0.11.0",
|
||||
"smallvec",
|
||||
"thiserror 2.0.20",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "s3s-sigv4"
|
||||
version = "0.16.0-alpha.1"
|
||||
source = "git+https://github.com/rustfs/s3s.git?rev=9c4690d8e73fc8d184031a19b2c4539ebc77d180#9c4690d8e73fc8d184031a19b2c4539ebc77d180"
|
||||
source = "git+https://github.com/rustfs/s3s.git?rev=28e9ebb23dd2fb7d667084f34121b4aa4807a5c6#28e9ebb23dd2fb7d667084f34121b4aa4807a5c6"
|
||||
dependencies = [
|
||||
"arrayvec",
|
||||
"base64-simd",
|
||||
"hex-simd",
|
||||
"hmac 0.13.0",
|
||||
"jiff",
|
||||
"nom 8.0.0",
|
||||
"serde",
|
||||
"sha2 0.11.0",
|
||||
"smallvec",
|
||||
"std-next",
|
||||
"thiserror 2.0.20",
|
||||
"zeroize",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
@@ -12031,9 +12069,9 @@ checksum = "13c2bddecc57b384dee18652358fb23172facb8a2c51ccc10d74c157bdea3292"
|
||||
|
||||
[[package]]
|
||||
name = "suppaftp"
|
||||
version = "10.0.2"
|
||||
version = "11.0.0"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "821001051ea3d12a60fb790b8c7cb9a6f5f8698dcfdca4cd533a025fefb0b5b8"
|
||||
checksum = "46c5095831abc0d7944a2d50d6ec6abcd75b9d165d9377deb3e45798cae2343a"
|
||||
dependencies = [
|
||||
"async-trait",
|
||||
"chrono",
|
||||
@@ -12224,10 +12262,10 @@ source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "32497e9a4c7b38532efcdebeef879707aa9f794296a4f0244f6f69e9bc8574bd"
|
||||
dependencies = [
|
||||
"fastrand",
|
||||
"getrandom 0.3.4",
|
||||
"getrandom 0.4.3",
|
||||
"once_cell",
|
||||
"rustix",
|
||||
"windows-sys 0.52.0",
|
||||
"windows-sys 0.61.2",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
@@ -12720,9 +12758,9 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "tower-http"
|
||||
version = "0.7.0"
|
||||
version = "0.7.1"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "b11f75e912b0c2be01b63d8cf8057b8c3f97cf34abb3d431a3a4c8675498e233"
|
||||
checksum = "08a05a66a4fdd61cbbe0a1d755ffe0ca6aba159dd4820936a0ff8a8278245b9c"
|
||||
dependencies = [
|
||||
"async-compression",
|
||||
"bitflags 2.13.1",
|
||||
@@ -13346,7 +13384,7 @@ version = "0.1.11"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "c2a7b1c03c876122aa43f3020e6c3c3ee5c05081c9a00739faf7503aeba10d22"
|
||||
dependencies = [
|
||||
"windows-sys 0.52.0",
|
||||
"windows-sys 0.61.2",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
|
||||
+60
-57
@@ -29,6 +29,7 @@ members = [
|
||||
"crates/heal-contracts", # Heal request/response channel contracts
|
||||
"crates/iam", # Identity and Access Management
|
||||
"crates/keystone", # OpenStack Keystone integration
|
||||
"crates/license", # License and entitlement provider contracts
|
||||
"crates/lifecycle", # Lifecycle rule evaluation contracts
|
||||
"crates/kms", # Key Management Service
|
||||
"crates/lock", # Distributed locking implementation
|
||||
@@ -71,8 +72,8 @@ resolver = "3"
|
||||
edition = "2024"
|
||||
license = "Apache-2.0"
|
||||
repository = "https://github.com/rustfs/rustfs"
|
||||
rust-version = "1.97.1"
|
||||
version = "1.0.0-rc.4"
|
||||
rust-version = "1.98.0"
|
||||
version = "1.0.0-rc.5"
|
||||
homepage = "https://rustfs.com"
|
||||
description = "RustFS is a high-performance distributed object storage software built using Rust, one of the most popular languages worldwide. "
|
||||
keywords = ["RustFS", "Minio", "object-storage", "filesystem", "s3"]
|
||||
@@ -89,60 +90,61 @@ redundant_clone = "warn"
|
||||
|
||||
[workspace.dependencies]
|
||||
# RustFS Internal Crates
|
||||
rustfs = { path = "./rustfs", version = "1.0.0-rc.4" }
|
||||
rustfs-heal = { path = "crates/heal", version = "1.0.0-rc.4" }
|
||||
rustfs-heal-contracts = { path = "crates/heal-contracts", version = "1.0.0-rc.4" }
|
||||
rustfs-scanner-contracts = { path = "crates/scanner-contracts", version = "1.0.0-rc.4" }
|
||||
rustfs-audit = { path = "crates/audit", version = "1.0.0-rc.4" }
|
||||
rustfs-checksums = { path = "crates/checksums", version = "1.0.0-rc.4" }
|
||||
rustfs-common = { path = "crates/common", version = "1.0.0-rc.4" }
|
||||
rustfs-data-usage = { path = "crates/data-usage", version = "1.0.0-rc.4" }
|
||||
rustfs-config = { path = "./crates/config", version = "1.0.0-rc.4" }
|
||||
rustfs-concurrency = { path = "./crates/concurrency", version = "1.0.0-rc.4" }
|
||||
rustfs-credentials = { path = "crates/credentials", version = "1.0.0-rc.4" }
|
||||
rustfs-crypto = { path = "crates/crypto", version = "1.0.0-rc.4" }
|
||||
rustfs-ecstore = { path = "crates/ecstore", version = "1.0.0-rc.4" }
|
||||
rustfs-filemeta = { path = "crates/filemeta", version = "1.0.0-rc.4" }
|
||||
rustfs-iam = { path = "crates/iam", version = "1.0.0-rc.4" }
|
||||
rustfs-keystone = { path = "crates/keystone", version = "1.0.0-rc.4" }
|
||||
rustfs-lifecycle = { path = "crates/lifecycle", version = "1.0.0-rc.4" }
|
||||
rustfs-kms = { path = "crates/kms", version = "1.0.0-rc.4" }
|
||||
rustfs-lock = { path = "crates/lock", version = "1.0.0-rc.4" }
|
||||
rustfs-madmin = { path = "crates/madmin", version = "1.0.0-rc.4" }
|
||||
rustfs-notify = { path = "crates/notify", version = "1.0.0-rc.4" }
|
||||
rustfs-io-metrics = { path = "crates/io-metrics", version = "1.0.0-rc.4" }
|
||||
rustfs-io-core = { path = "crates/io-core", version = "1.0.0-rc.4" }
|
||||
rustfs-object-capacity = { path = "crates/object-capacity", version = "1.0.0-rc.4" }
|
||||
rustfs-object-data-cache = { path = "crates/object-data-cache", version = "1.0.0-rc.4", default-features = false }
|
||||
rustfs-log-analyzer = { path = "crates/log-analyzer", version = "1.0.0-rc.4" }
|
||||
rustfs-obs = { path = "crates/obs", version = "1.0.0-rc.4" }
|
||||
rustfs-policy = { path = "crates/policy", version = "1.0.0-rc.4" }
|
||||
rustfs-protos = { path = "crates/protos", version = "1.0.0-rc.4" }
|
||||
rustfs-protocols = { path = "crates/protocols", version = "1.0.0-rc.4" }
|
||||
rustfs-replication = { path = "crates/replication", version = "1.0.0-rc.4" }
|
||||
rustfs-rio = { path = "crates/rio", version = "1.0.0-rc.4" }
|
||||
rustfs-rio-v2 = { path = "crates/rio-v2", version = "1.0.0-rc.4" }
|
||||
rustfs-s3-client = { path = "crates/s3-client", version = "1.0.0-rc.4" }
|
||||
rustfs-s3-types = { path = "crates/s3-types", version = "1.0.0-rc.4" }
|
||||
rustfs-s3-ops = { path = "crates/s3-ops", version = "1.0.0-rc.4" }
|
||||
rustfs-s3select-api = { path = "crates/s3select-api", version = "1.0.0-rc.4" }
|
||||
rustfs-s3select-query = { path = "crates/s3select-query", version = "1.0.0-rc.4" }
|
||||
rustfs-scanner = { path = "crates/scanner", version = "1.0.0-rc.4" }
|
||||
rustfs-security-governance = { path = "crates/security-governance", version = "1.0.0-rc.4" }
|
||||
rustfs-extension-schema = { path = "crates/extension-schema", version = "1.0.0-rc.4" }
|
||||
rustfs-signer = { path = "crates/signer", version = "1.0.0-rc.4" }
|
||||
rustfs-storage-api = { path = "crates/storage-api", version = "1.0.0-rc.4" }
|
||||
rustfs-trusted-proxies = { path = "crates/trusted-proxies", version = "1.0.0-rc.4" }
|
||||
rustfs-targets = { path = "crates/targets", version = "1.0.0-rc.4" }
|
||||
rustfs-test-utils = { path = "crates/test-utils", version = "1.0.0-rc.4" }
|
||||
rustfs-tls-runtime = { path = "crates/tls-runtime", version = "1.0.0-rc.4" }
|
||||
rustfs-utils = { path = "crates/utils", version = "1.0.0-rc.4" }
|
||||
rustfs-zip = { path = "./crates/zip", version = "1.0.0-rc.4" }
|
||||
rustfs = { path = "./rustfs", version = "1.0.0-rc.5" }
|
||||
rustfs-heal = { path = "crates/heal", version = "1.0.0-rc.5" }
|
||||
rustfs-heal-contracts = { path = "crates/heal-contracts", version = "1.0.0-rc.5" }
|
||||
rustfs-scanner-contracts = { path = "crates/scanner-contracts", version = "1.0.0-rc.5" }
|
||||
rustfs-audit = { path = "crates/audit", version = "1.0.0-rc.5" }
|
||||
rustfs-checksums = { path = "crates/checksums", version = "1.0.0-rc.5" }
|
||||
rustfs-common = { path = "crates/common", version = "1.0.0-rc.5" }
|
||||
rustfs-data-usage = { path = "crates/data-usage", version = "1.0.0-rc.5" }
|
||||
rustfs-config = { path = "./crates/config", version = "1.0.0-rc.5" }
|
||||
rustfs-concurrency = { path = "./crates/concurrency", version = "1.0.0-rc.5" }
|
||||
rustfs-credentials = { path = "crates/credentials", version = "1.0.0-rc.5" }
|
||||
rustfs-crypto = { path = "crates/crypto", version = "1.0.0-rc.5" }
|
||||
rustfs-ecstore = { path = "crates/ecstore", version = "1.0.0-rc.5" }
|
||||
rustfs-filemeta = { path = "crates/filemeta", version = "1.0.0-rc.5" }
|
||||
rustfs-iam = { path = "crates/iam", version = "1.0.0-rc.5" }
|
||||
rustfs-keystone = { path = "crates/keystone", version = "1.0.0-rc.5" }
|
||||
rustfs-license = { path = "crates/license", version = "1.0.0-rc.5" }
|
||||
rustfs-lifecycle = { path = "crates/lifecycle", version = "1.0.0-rc.5" }
|
||||
rustfs-kms = { path = "crates/kms", version = "1.0.0-rc.5" }
|
||||
rustfs-lock = { path = "crates/lock", version = "1.0.0-rc.5" }
|
||||
rustfs-madmin = { path = "crates/madmin", version = "1.0.0-rc.5" }
|
||||
rustfs-notify = { path = "crates/notify", version = "1.0.0-rc.5" }
|
||||
rustfs-io-metrics = { path = "crates/io-metrics", version = "1.0.0-rc.5" }
|
||||
rustfs-io-core = { path = "crates/io-core", version = "1.0.0-rc.5" }
|
||||
rustfs-object-capacity = { path = "crates/object-capacity", version = "1.0.0-rc.5" }
|
||||
rustfs-object-data-cache = { path = "crates/object-data-cache", version = "1.0.0-rc.5", default-features = false }
|
||||
rustfs-log-analyzer = { path = "crates/log-analyzer", version = "1.0.0-rc.5" }
|
||||
rustfs-obs = { path = "crates/obs", version = "1.0.0-rc.5" }
|
||||
rustfs-policy = { path = "crates/policy", version = "1.0.0-rc.5" }
|
||||
rustfs-protos = { path = "crates/protos", version = "1.0.0-rc.5" }
|
||||
rustfs-protocols = { path = "crates/protocols", version = "1.0.0-rc.5" }
|
||||
rustfs-replication = { path = "crates/replication", version = "1.0.0-rc.5" }
|
||||
rustfs-rio = { path = "crates/rio", version = "1.0.0-rc.5" }
|
||||
rustfs-rio-v2 = { path = "crates/rio-v2", version = "1.0.0-rc.5" }
|
||||
rustfs-s3-client = { path = "crates/s3-client", version = "1.0.0-rc.5" }
|
||||
rustfs-s3-types = { path = "crates/s3-types", version = "1.0.0-rc.5" }
|
||||
rustfs-s3-ops = { path = "crates/s3-ops", version = "1.0.0-rc.5" }
|
||||
rustfs-s3select-api = { path = "crates/s3select-api", version = "1.0.0-rc.5" }
|
||||
rustfs-s3select-query = { path = "crates/s3select-query", version = "1.0.0-rc.5" }
|
||||
rustfs-scanner = { path = "crates/scanner", version = "1.0.0-rc.5" }
|
||||
rustfs-security-governance = { path = "crates/security-governance", version = "1.0.0-rc.5" }
|
||||
rustfs-extension-schema = { path = "crates/extension-schema", version = "1.0.0-rc.5" }
|
||||
rustfs-signer = { path = "crates/signer", version = "1.0.0-rc.5" }
|
||||
rustfs-storage-api = { path = "crates/storage-api", version = "1.0.0-rc.5" }
|
||||
rustfs-trusted-proxies = { path = "crates/trusted-proxies", version = "1.0.0-rc.5" }
|
||||
rustfs-targets = { path = "crates/targets", version = "1.0.0-rc.5" }
|
||||
rustfs-test-utils = { path = "crates/test-utils", version = "1.0.0-rc.5" }
|
||||
rustfs-tls-runtime = { path = "crates/tls-runtime", version = "1.0.0-rc.5" }
|
||||
rustfs-utils = { path = "crates/utils", version = "1.0.0-rc.5" }
|
||||
rustfs-zip = { path = "./crates/zip", version = "1.0.0-rc.5" }
|
||||
|
||||
# Async Runtime and Networking
|
||||
async-channel = "2.5.0"
|
||||
async_zip = { default-features = false, version = "0.0.19" }
|
||||
mysql_async = { default-features = false, version = "0.37" }
|
||||
mysql_async = { default-features = false, version = "0.37.1" }
|
||||
async-compression = { version = "0.4.43" }
|
||||
async-recursion = "1.1.1"
|
||||
async-trait = "0.1.92"
|
||||
@@ -174,7 +176,7 @@ tonic = { version = "0.14.6" }
|
||||
tonic-prost = { version = "0.14.6" }
|
||||
tonic-prost-build = { version = "0.14.6" }
|
||||
tower = { version = "0.5.3" }
|
||||
tower-http = { version = "0.7.0" }
|
||||
tower-http = { version = "0.7.1" }
|
||||
|
||||
# Serialization and Data Formats
|
||||
apache-avro = { version = "0.22.0", features = ["snappy", "zstandard"] }
|
||||
@@ -232,7 +234,8 @@ tokio-postgres-rustls = "0.14.0"
|
||||
# Utilities and Tools
|
||||
anyhow = "1.0.104"
|
||||
arc-swap = "1.9.2"
|
||||
astral-tokio-tar = "0.7.0"
|
||||
# RUSTFS_COMPAT_TODO(tokio-tar-extension-limits): keep the fork pin until every parser hardening used by Snowball is released upstream. Remove after astral-sh/tokio-tar#118 is merged and a published release includes extension, physical-entry, and sparse limits, cancellation-safe sparse parsing, and error-fused entry streams.
|
||||
astral-tokio-tar = { git = "https://github.com/cxymds/tokio-tar.git", rev = "603756478b7668436e464519c77ccac22a99ba96" }
|
||||
atoi = "3.1.0"
|
||||
atomic_enum = "0.3.0"
|
||||
aws-config = { version = "1.11.0" }
|
||||
@@ -282,7 +285,7 @@ mime_guess = "2.0.5"
|
||||
moka = { version = "0.12.16" }
|
||||
netif = "0.1.6"
|
||||
num_cpus = { version = "1.17.0" }
|
||||
nvml-wrapper = "0.12.1"
|
||||
nvml-wrapper = "0.13.0"
|
||||
parking_lot = "0.12.5"
|
||||
path-absolutize = "4.0.1"
|
||||
percent-encoding = "2.3.2"
|
||||
@@ -304,7 +307,7 @@ rustify = { version = "0.7", default-features = false }
|
||||
rustix = { version = "1.1.4" }
|
||||
rust-embed = { version = "8.12.0" }
|
||||
rustc-hash = { version = "2.1.3" }
|
||||
s3s = { git = "https://github.com/rustfs/s3s.git", rev = "9c4690d8e73fc8d184031a19b2c4539ebc77d180", version = "0.15.0", features = ["minio"] }
|
||||
s3s = { git = "https://github.com/rustfs/s3s.git", rev = "28e9ebb23dd2fb7d667084f34121b4aa4807a5c6", version = "0.15.0", features = ["minio"] }
|
||||
serial_test = "4.0.1"
|
||||
shadow-rs = { default-features = false, version = "2.0.0" }
|
||||
siphasher = "1.0.3"
|
||||
@@ -354,7 +357,7 @@ pyroscope = { version = "2.1.1" }
|
||||
# FTP and SFTP
|
||||
libunftp = { version = "0.23.0" }
|
||||
unftp-core = "0.1.0"
|
||||
suppaftp = { version = "10.0.2" }
|
||||
suppaftp = { version = "11.0.0" }
|
||||
rcgen = { version = "0.14.10", default-features = false, features = ["aws_lc_rs", "crypto", "pem"] }
|
||||
russh = { version = "0.63.1" }
|
||||
russh-sftp = "2.4.0"
|
||||
|
||||
@@ -23,6 +23,12 @@ SHELL := $(shell which bash)
|
||||
.SHELLFLAGS = -eu -o pipefail -c
|
||||
|
||||
DOCKER_CLI ?= docker
|
||||
# Python interpreter for the repository's helper scripts. They import tomllib
|
||||
# (Python 3.11+), while macOS still ships /usr/bin/python3 at 3.9, so calls go
|
||||
# through a resolver that picks a new-enough interpreter (or falls back to uv).
|
||||
# Override with RUSTFS_PYTHON=/path/to/python3.12, or replace the resolver via
|
||||
# RUSTFS_PYTHON_BIN=<command>.
|
||||
RUSTFS_PYTHON_BIN ?= ./scripts/python_bin.sh
|
||||
IMAGE_NAME ?= rustfs:v1.0.0
|
||||
CONTAINER_NAME ?= rustfs-dev
|
||||
# Docker build configurations
|
||||
|
||||
@@ -115,7 +115,7 @@ chown -R 10001:10001 data logs
|
||||
docker run -d -p 9000:9000 -p 9001:9001 -v $(pwd)/data:/data -v $(pwd)/logs:/logs rustfs/rustfs:latest
|
||||
|
||||
# Using specific version
|
||||
docker run -d -p 9000:9000 -p 9001:9001 -v $(pwd)/data:/data -v $(pwd)/logs:/logs rustfs/rustfs:1.0.0-rc.4
|
||||
docker run -d -p 9000:9000 -p 9001:9001 -v $(pwd)/data:/data -v $(pwd)/logs:/logs rustfs/rustfs:1.0.0-rc.5
|
||||
```
|
||||
|
||||
If you use [podman](https://github.com/containers/podman) instead of docker, you can install the RustFS with the below command
|
||||
|
||||
+1
-1
@@ -112,7 +112,7 @@ chown -R 10001:10001 data logs
|
||||
docker run -d -p 9000:9000 -p 9001:9001 -v $(pwd)/data:/data -v $(pwd)/logs:/logs rustfs/rustfs:latest
|
||||
|
||||
# 使用指定版本运行
|
||||
docker run -d -p 9000:9000 -p 9001:9001 -v $(pwd)/data:/data -v $(pwd)/logs:/logs rustfs/rustfs:1.0.0-rc.4
|
||||
docker run -d -p 9000:9000 -p 9001:9001 -v $(pwd)/data:/data -v $(pwd)/logs:/logs rustfs/rustfs:1.0.0-rc.5
|
||||
```
|
||||
|
||||
如果您通过绑定挂载启用 TLS 证书目录,也请用同样方式准备该目录:
|
||||
|
||||
@@ -59,20 +59,20 @@ pub const ENV_CAPACITY_MAX_TIMEOUT: &str = "RUSTFS_CAPACITY_MAX_TIMEOUT";
|
||||
// ============================================================================
|
||||
|
||||
/// Scheduled update interval in seconds
|
||||
/// Default: 120 seconds (2 minutes)
|
||||
pub const DEFAULT_SCHEDULED_UPDATE_INTERVAL_SECS: u64 = 120;
|
||||
/// Default: 600 seconds (10 minutes)
|
||||
pub const DEFAULT_SCHEDULED_UPDATE_INTERVAL_SECS: u64 = 600;
|
||||
|
||||
/// Write trigger delay in seconds
|
||||
/// Default: 5 seconds
|
||||
pub const DEFAULT_WRITE_TRIGGER_DELAY_SECS: u64 = 5;
|
||||
/// Default: 30 seconds
|
||||
pub const DEFAULT_WRITE_TRIGGER_DELAY_SECS: u64 = 30;
|
||||
|
||||
/// Write frequency threshold (writes per minute)
|
||||
/// Default: 5 writes/minute
|
||||
pub const DEFAULT_WRITE_FREQUENCY_THRESHOLD: usize = 5;
|
||||
/// Default: 20 writes/minute
|
||||
pub const DEFAULT_WRITE_FREQUENCY_THRESHOLD: usize = 20;
|
||||
|
||||
/// Fast update threshold in seconds
|
||||
/// Default: 30 seconds
|
||||
pub const DEFAULT_FAST_UPDATE_THRESHOLD_SECS: u64 = 30;
|
||||
/// Default: 120 seconds
|
||||
pub const DEFAULT_FAST_UPDATE_THRESHOLD_SECS: u64 = 120;
|
||||
|
||||
/// Maximum files threshold for sampling
|
||||
/// Default: 200,000 files
|
||||
@@ -129,4 +129,16 @@ mod tests {
|
||||
assert_eq!(ENV_CAPACITY_MIN_TIMEOUT, "RUSTFS_CAPACITY_MIN_TIMEOUT");
|
||||
assert_eq!(ENV_CAPACITY_MAX_TIMEOUT, "RUSTFS_CAPACITY_MAX_TIMEOUT");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_capacity_default_values() {
|
||||
assert_eq!(DEFAULT_SCHEDULED_UPDATE_INTERVAL_SECS, 600);
|
||||
assert_eq!(DEFAULT_WRITE_TRIGGER_DELAY_SECS, 30);
|
||||
assert_eq!(DEFAULT_WRITE_FREQUENCY_THRESHOLD, 20);
|
||||
assert_eq!(DEFAULT_FAST_UPDATE_THRESHOLD_SECS, 120);
|
||||
assert_eq!(DEFAULT_MAX_FILES_THRESHOLD, 200_000);
|
||||
assert_eq!(DEFAULT_STAT_TIMEOUT_SECS, 3);
|
||||
assert_eq!(DEFAULT_SAMPLE_RATE, 200);
|
||||
assert_eq!(DEFAULT_CAPACITY_METRICS_INTERVAL_SECS, 600);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -198,11 +198,11 @@ pub const ENV_SCANNER_IDLE_MODE: &str = "RUSTFS_SCANNER_IDLE_MODE";
|
||||
/// Environment variable that controls scanner cache save timeout in seconds.
|
||||
/// The scanner enforces a minimum value of `1`.
|
||||
/// - Unit: seconds (u64).
|
||||
/// - Example: `export RUSTFS_SCANNER_CACHE_SAVE_TIMEOUT_SECS=30`
|
||||
/// - Example: `export RUSTFS_SCANNER_CACHE_SAVE_TIMEOUT_SECS=14`
|
||||
pub const ENV_SCANNER_CACHE_SAVE_TIMEOUT_SECS: &str = "RUSTFS_SCANNER_CACHE_SAVE_TIMEOUT_SECS";
|
||||
|
||||
/// Default scanner cache save timeout in seconds.
|
||||
pub const DEFAULT_SCANNER_CACHE_SAVE_TIMEOUT_SECS: u64 = 30;
|
||||
pub const DEFAULT_SCANNER_CACHE_SAVE_TIMEOUT_SECS: u64 = 14;
|
||||
|
||||
/// Environment variable that caps concurrent scanner set tasks.
|
||||
/// A value of `0` keeps the existing topology-based concurrency.
|
||||
|
||||
@@ -100,7 +100,8 @@ aws-sdk-s3 = { workspace = true, default-features = false, features = ["sigv4a",
|
||||
aws-sdk-sts = { workspace = true, default-features = false, features = ["default-https-client", "rt-tokio"] }
|
||||
aws-config = { workspace = true }
|
||||
aws-smithy-http-client = { workspace = true, default-features = false, features = ["rustls-aws-lc"] }
|
||||
async-compression = { workspace = true, features = ["tokio", "bzip2", "xz"] }
|
||||
aws-smithy-types.workspace = true
|
||||
async-compression = { workspace = true, features = ["tokio", "bzip2", "lz4", "xz"] }
|
||||
async-trait = { workspace = true }
|
||||
flate2.workspace = true
|
||||
http.workspace = true
|
||||
@@ -114,6 +115,7 @@ rustfs-signer.workspace = true
|
||||
# server's implementation: a shared helper could agree with a bug on both sides.
|
||||
data-encoding = { workspace = true }
|
||||
hmac = { workspace = true }
|
||||
minlz.workspace = true
|
||||
sha1 = { workspace = true }
|
||||
serde_urlencoded = { workspace = true }
|
||||
tracing = { workspace = true }
|
||||
|
||||
@@ -169,7 +169,7 @@ the same profile for membership and execution with one nightly worker.
|
||||
| `s3s-e2e` black-box | `e2e-tests` + `e2e-tests-rio-v2` jobs | **Active** (external conformance tool) |
|
||||
| ILM / lifecycle (ignored) | `test-ilm-integration-serial` lane, `-j1` | **Active** (backlog#1148 ilm-1) |
|
||||
| KMS suite | `e2e-full` job, merge queue + main | **Active** |
|
||||
| Direct upgrade from pinned previous release | `e2e-upgrade.yml`, storage-sensitive PRs + release tags + weekly | **Active** |
|
||||
| Direct and mixed-version rolling upgrades from pinned previous release | `e2e-upgrade.yml`, storage-sensitive PRs + release tags + weekly | **Active** |
|
||||
| Cluster faults (`e2e-nightly` profile) | consolidated nightly workflow | **Active** (backlog#1149 ci-7) |
|
||||
| Protocols (FTPS/WebDAV/SFTP) | consolidated nightly workflow, serial | **Active** (backlog#1149 ci-7) |
|
||||
| Replication (fast subset) | `e2e-smoke` profile, `e2e-tests` job, every PR | **Active** (backlog#1147 repl-1) |
|
||||
|
||||
@@ -22,7 +22,7 @@ mod tests {
|
||||
use aws_sdk_s3::config::{Credentials, Region, RequestChecksumCalculation};
|
||||
use aws_sdk_s3::error::ProvideErrorMetadata;
|
||||
use aws_sdk_s3::primitives::ByteStream;
|
||||
use aws_sdk_s3::types::{ChecksumAlgorithm, ChecksumMode, CompletedMultipartUpload, CompletedPart};
|
||||
use aws_sdk_s3::types::{ChecksumAlgorithm, ChecksumMode, CompletedMultipartUpload, CompletedPart, ServerSideEncryption};
|
||||
use aws_smithy_http_client::Builder as SmithyHttpClientBuilder;
|
||||
use md5::{Digest as Md5Digest, Md5};
|
||||
use rustfs_rio::{Checksum, ChecksumType as RioChecksumType};
|
||||
@@ -260,6 +260,117 @@ mod tests {
|
||||
info!("PASSED: HeadObject returns stored SHA256 digest");
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
async fn test_head_object_returns_sse_s3_checksum() {
|
||||
init_logging();
|
||||
|
||||
let mut env = RustFSTestEnvironment::new().await.expect("Failed to create test environment");
|
||||
env.start_rustfs_server_with_env(
|
||||
vec![],
|
||||
&[
|
||||
("RUSTFS_SSE_S3_MASTER_KEY", "MTIzNDU2Nzg5MDEyMzQ1Njc4OTAxMjM0NTY3ODkwMTI="),
|
||||
("RUSTFS_CONSOLE_ENABLE", "false"),
|
||||
],
|
||||
)
|
||||
.await
|
||||
.expect("Failed to start RustFS");
|
||||
|
||||
let client = create_s3_client(&env);
|
||||
let bucket = "test-sse-s3-checksum-head";
|
||||
create_bucket(&client, bucket).await.expect("Failed to create bucket");
|
||||
|
||||
let put = client
|
||||
.put_object()
|
||||
.bucket(bucket)
|
||||
.key("encrypted.txt")
|
||||
.body(ByteStream::from_static(b"encrypted checksum"))
|
||||
.server_side_encryption(ServerSideEncryption::Aes256)
|
||||
.checksum_algorithm(ChecksumAlgorithm::Crc32)
|
||||
.send()
|
||||
.await
|
||||
.expect("SSE-S3 PutObject with CRC32 failed");
|
||||
let expected = put.checksum_crc32().expect("PutObject must return CRC32");
|
||||
|
||||
let head = client
|
||||
.head_object()
|
||||
.bucket(bucket)
|
||||
.key("encrypted.txt")
|
||||
.checksum_mode(ChecksumMode::Enabled)
|
||||
.send()
|
||||
.await
|
||||
.expect("SSE-S3 HeadObject failed");
|
||||
|
||||
assert_eq!(head.checksum_crc32(), Some(expected));
|
||||
|
||||
client
|
||||
.copy_object()
|
||||
.bucket(bucket)
|
||||
.key("encrypted-copy.txt")
|
||||
.copy_source(format!("{bucket}/encrypted.txt"))
|
||||
.server_side_encryption(ServerSideEncryption::Aes256)
|
||||
.send()
|
||||
.await
|
||||
.expect("SSE-S3 CopyObject failed");
|
||||
let copy_head = client
|
||||
.head_object()
|
||||
.bucket(bucket)
|
||||
.key("encrypted-copy.txt")
|
||||
.checksum_mode(ChecksumMode::Enabled)
|
||||
.send()
|
||||
.await
|
||||
.expect("SSE-S3 copied HeadObject failed");
|
||||
|
||||
assert_eq!(copy_head.checksum_crc32(), Some(expected));
|
||||
|
||||
let multipart_key = "encrypted-multipart.txt";
|
||||
let create = client
|
||||
.create_multipart_upload()
|
||||
.bucket(bucket)
|
||||
.key(multipart_key)
|
||||
.server_side_encryption(ServerSideEncryption::Aes256)
|
||||
.checksum_algorithm(ChecksumAlgorithm::Crc32)
|
||||
.send()
|
||||
.await
|
||||
.expect("SSE-S3 CreateMultipartUpload with CRC32 failed");
|
||||
let upload_id = create.upload_id().expect("CreateMultipartUpload must return an upload ID");
|
||||
let part = client
|
||||
.upload_part()
|
||||
.bucket(bucket)
|
||||
.key(multipart_key)
|
||||
.upload_id(upload_id)
|
||||
.part_number(1)
|
||||
.body(ByteStream::from_static(b"encrypted multipart checksum"))
|
||||
.checksum_algorithm(ChecksumAlgorithm::Crc32)
|
||||
.send()
|
||||
.await
|
||||
.expect("SSE-S3 UploadPart with CRC32 failed");
|
||||
let completed_part = CompletedPart::builder()
|
||||
.part_number(1)
|
||||
.e_tag(part.e_tag().expect("UploadPart must return an ETag"))
|
||||
.checksum_crc32(part.checksum_crc32().expect("UploadPart must return CRC32"))
|
||||
.build();
|
||||
let complete = client
|
||||
.complete_multipart_upload()
|
||||
.bucket(bucket)
|
||||
.key(multipart_key)
|
||||
.upload_id(upload_id)
|
||||
.multipart_upload(CompletedMultipartUpload::builder().parts(completed_part).build())
|
||||
.send()
|
||||
.await
|
||||
.expect("SSE-S3 CompleteMultipartUpload with CRC32 failed");
|
||||
let expected_multipart = complete.checksum_crc32().expect("CompleteMultipartUpload must return CRC32");
|
||||
let multipart_head = client
|
||||
.head_object()
|
||||
.bucket(bucket)
|
||||
.key(multipart_key)
|
||||
.checksum_mode(ChecksumMode::Enabled)
|
||||
.send()
|
||||
.await
|
||||
.expect("SSE-S3 multipart HeadObject failed");
|
||||
|
||||
assert_eq!(multipart_head.checksum_crc32(), Some(expected_multipart));
|
||||
}
|
||||
|
||||
/// Multipart upload with checksum: CreateMultipartUpload, UploadPart(s) with checksum_sha256, CompleteMultipartUpload; then GetObject verifies content.
|
||||
/// Uses part size >= 5MB (server minimum) for two parts.
|
||||
#[tokio::test]
|
||||
|
||||
@@ -27,8 +27,10 @@
|
||||
//! Readiness is established by the harness's `start()` handshake (TCP reachability
|
||||
//! plus an S3 `ListBuckets` poll) — there are no fixed sleeps.
|
||||
//!
|
||||
//! Out of scope for this block (tracked separately): network fault injection
|
||||
//! (toxiproxy / socket proxy) and 5GiB large-object budgets.
|
||||
//! The volume-proxy smoke below also proves that the socket-level fault proxy
|
||||
//! can be installed before startup without changing the client-facing node URL.
|
||||
//! A full lock-plane partition matrix and 5GiB large-object budget remain
|
||||
//! tracked separately.
|
||||
|
||||
use crate::common::{ClusterTopology, RustFSTestClusterEnvironment};
|
||||
|
||||
@@ -76,6 +78,28 @@ async fn cluster_multidrive_single_pool_smoke() -> TestResult {
|
||||
Ok(())
|
||||
}
|
||||
|
||||
/// 4 nodes x 4 drives, single pool: exercise the maximum local erasure layout
|
||||
/// supported by the cluster harness. This remains in the nightly lane because
|
||||
/// it starts four real server processes and sixteen data directories.
|
||||
#[tokio::test]
|
||||
async fn cluster_four_node_four_drive_single_pool_smoke() -> TestResult {
|
||||
crate::common::init_logging();
|
||||
|
||||
let mut cluster = RustFSTestClusterEnvironment::with_topology(ClusterTopology::single_pool_multidrive(4, 4)).await?;
|
||||
|
||||
let volumes = cluster.rustfs_volumes_arg();
|
||||
assert_eq!(volumes.split(' ').count(), 16, "expected 16 explicit endpoints, got: {volumes}");
|
||||
assert!(!volumes.contains('{'), "single-pool layout must not use ellipses: {volumes}");
|
||||
assert!(cluster.nodes.iter().all(|node| node.data_dirs.len() == 4));
|
||||
|
||||
cluster.start().await?;
|
||||
cluster.create_test_bucket(BUCKET).await?;
|
||||
|
||||
let payload = vec![0x3Cu8; 1024 * 1024];
|
||||
put_get_roundtrip(&cluster, "multidrive-4/object", &payload).await?;
|
||||
Ok(())
|
||||
}
|
||||
|
||||
/// Two single-node pools, 2 drives each: the multi-pool layout boots and
|
||||
/// round-trips. Every pool is a distinct erasure pool (`pool_idx` 0 and 1).
|
||||
#[tokio::test]
|
||||
@@ -103,3 +127,27 @@ async fn cluster_two_pool_smoke() -> TestResult {
|
||||
put_get_roundtrip(&cluster, "twopool/object", &payload).await?;
|
||||
Ok(())
|
||||
}
|
||||
|
||||
/// A real cluster smoke for the volume FaultProxy wiring. The proxy target is
|
||||
/// not listening yet when it is created; cluster startup must still converge
|
||||
/// once the target node starts, and peer disk/RPC traffic must traverse it.
|
||||
#[tokio::test]
|
||||
async fn cluster_volume_fault_proxy_pass_smoke() -> TestResult {
|
||||
crate::common::init_logging();
|
||||
|
||||
let mut cluster = RustFSTestClusterEnvironment::with_topology(ClusterTopology::single_pool_multidrive(2, 2)).await?;
|
||||
let proxy = cluster.start_volume_proxy_for_node(0).await?;
|
||||
let proxied = proxy.local_addr().to_string();
|
||||
assert!(cluster.rustfs_volumes_arg().contains(&proxied));
|
||||
|
||||
let result: TestResult = async {
|
||||
cluster.start().await?;
|
||||
cluster.create_test_bucket(BUCKET).await?;
|
||||
let payload = vec![0x6Du8; 256 * 1024];
|
||||
put_get_roundtrip(&cluster, "volume-proxy/object", &payload).await
|
||||
}
|
||||
.await;
|
||||
|
||||
proxy.shutdown().await;
|
||||
result
|
||||
}
|
||||
|
||||
+155
-35
@@ -34,6 +34,7 @@ use serde_json;
|
||||
use std::ffi::OsStr;
|
||||
use std::fs as stdfs;
|
||||
use std::io::ErrorKind;
|
||||
use std::net::SocketAddr;
|
||||
use std::path::{Path, PathBuf};
|
||||
use std::process::{Child, Command, Stdio};
|
||||
use std::sync::Once;
|
||||
@@ -1214,6 +1215,9 @@ pub struct RustFSTestClusterEnvironment {
|
||||
pub node_extra_env: Vec<Vec<(String, String)>>,
|
||||
pub node_capture_log_paths: Vec<Option<String>>,
|
||||
pub topology: ClusterTopology,
|
||||
/// Optional socket proxies used for the corresponding node's volume
|
||||
/// endpoints. Proxies must be installed before [`Self::start`].
|
||||
volume_proxy_addresses: Vec<Option<SocketAddr>>,
|
||||
}
|
||||
|
||||
impl RustFSTestClusterEnvironment {
|
||||
@@ -1305,6 +1309,7 @@ impl RustFSTestClusterEnvironment {
|
||||
extra_env.push(("RUSTFS_UNSAFE_BYPASS_DISK_CHECK".to_string(), "true".to_string()));
|
||||
}
|
||||
|
||||
let node_count = topology.node_count;
|
||||
Ok(Self {
|
||||
nodes,
|
||||
temp_dir,
|
||||
@@ -1314,6 +1319,7 @@ impl RustFSTestClusterEnvironment {
|
||||
node_extra_env: vec![Vec::new(); topology.node_count],
|
||||
node_capture_log_paths: vec![None; topology.node_count],
|
||||
topology,
|
||||
volume_proxy_addresses: vec![None; node_count],
|
||||
})
|
||||
}
|
||||
|
||||
@@ -1381,6 +1387,34 @@ impl RustFSTestClusterEnvironment {
|
||||
self.build_volumes_arg()
|
||||
}
|
||||
|
||||
/// Start a socket proxy for one node's volume endpoints and route all
|
||||
/// subsequent `RUSTFS_VOLUMES` references for that node through it.
|
||||
///
|
||||
/// Call this before [`Self::start`], then use the returned proxy's
|
||||
/// [`crate::fault_proxy::FaultProxy::set_mode`] to inject latency,
|
||||
/// blackhole, or one-way partition faults. The node's own listen address
|
||||
/// remains direct, so S3 clients can still reach it while peer disk/RPC
|
||||
/// traffic is steered through the proxy.
|
||||
pub async fn start_volume_proxy_for_node(
|
||||
&mut self,
|
||||
node_idx: usize,
|
||||
) -> Result<crate::fault_proxy::FaultProxy, Box<dyn std::error::Error + Send + Sync>> {
|
||||
self.ensure_node_index(node_idx)?;
|
||||
if self.volume_proxy_addresses[node_idx].is_some() {
|
||||
return Err(format!("a volume proxy is already configured for node {node_idx}").into());
|
||||
}
|
||||
let target = self.nodes[node_idx].address.parse::<SocketAddr>()?;
|
||||
let proxy = crate::fault_proxy::FaultProxy::start(target).await?;
|
||||
self.volume_proxy_addresses[node_idx] = Some(proxy.local_addr());
|
||||
Ok(proxy)
|
||||
}
|
||||
|
||||
fn volume_address(&self, node_idx: usize) -> String {
|
||||
self.volume_proxy_addresses[node_idx]
|
||||
.map(|address| address.to_string())
|
||||
.unwrap_or_else(|| self.nodes[node_idx].address.clone())
|
||||
}
|
||||
|
||||
fn build_volumes_arg(&self) -> String {
|
||||
let pools = self.topology.normalized_pools();
|
||||
|
||||
@@ -1389,7 +1423,11 @@ impl RustFSTestClusterEnvironment {
|
||||
return self
|
||||
.nodes
|
||||
.iter()
|
||||
.flat_map(|n| n.data_dirs.iter().map(move |dir| format!("http://{}{}", n.address, dir)))
|
||||
.enumerate()
|
||||
.flat_map(|(node_idx, n)| {
|
||||
let address = self.volume_address(node_idx);
|
||||
n.data_dirs.iter().map(move |dir| format!("http://{}{}", address, dir))
|
||||
})
|
||||
.collect::<Vec<_>>()
|
||||
.join(" ");
|
||||
}
|
||||
@@ -1400,13 +1438,19 @@ impl RustFSTestClusterEnvironment {
|
||||
pools
|
||||
.iter()
|
||||
.map(|nodes| {
|
||||
let node = &self.nodes[nodes[0]];
|
||||
let node_idx = nodes[0];
|
||||
let node = &self.nodes[node_idx];
|
||||
let base = node
|
||||
.data_dirs
|
||||
.first()
|
||||
.and_then(|d| d.rsplit_once('/').map(|(parent, _)| parent))
|
||||
.unwrap_or(&node.data_dir);
|
||||
format!("http://{}{}/drive{{0...{}}}", node.address, base, self.topology.drives_per_node - 1)
|
||||
format!(
|
||||
"http://{}{}/drive{{0...{}}}",
|
||||
self.volume_address(node_idx),
|
||||
base,
|
||||
self.topology.drives_per_node - 1
|
||||
)
|
||||
})
|
||||
.collect::<Vec<_>>()
|
||||
.join(" ")
|
||||
@@ -1425,31 +1469,18 @@ impl RustFSTestClusterEnvironment {
|
||||
/// times out, or cluster service readiness times out.
|
||||
pub async fn start(&mut self) -> Result<(), Box<dyn std::error::Error + Send + Sync>> {
|
||||
let binary_path = rustfs_binary_path();
|
||||
self.start_with_binary(&binary_path).await
|
||||
}
|
||||
|
||||
/// Start every cluster node with a specific RustFS binary.
|
||||
///
|
||||
/// Upgrade compatibility tests use this to initialize a cluster with a
|
||||
/// pinned previous release before replacing nodes with the workspace build.
|
||||
pub async fn start_with_binary(&mut self, binary_path: &Path) -> Result<(), Box<dyn std::error::Error + Send + Sync>> {
|
||||
let volumes_arg = self.build_volumes_arg();
|
||||
|
||||
for (i, node) in self.nodes.iter_mut().enumerate() {
|
||||
info!("Starting cluster node {} on {}", i, node.address);
|
||||
|
||||
let mut command = Command::new(&binary_path);
|
||||
command
|
||||
.env("RUSTFS_VOLUMES", &volumes_arg)
|
||||
.env("RUSTFS_ADDRESS", &node.address)
|
||||
.env("RUSTFS_ACCESS_KEY", &self.access_key)
|
||||
.env("RUSTFS_SECRET_KEY", &self.secret_key)
|
||||
.env("RUSTFS_CONSOLE_ENABLE", "false")
|
||||
.env("RUST_LOG", "rustfs=info,rustfs_notify=debug");
|
||||
|
||||
for (key, value) in &self.extra_env {
|
||||
command.env(key, value);
|
||||
}
|
||||
for (key, value) in &self.node_extra_env[i] {
|
||||
command.env(key, value);
|
||||
}
|
||||
capture_command_logs(&mut command, self.node_capture_log_paths[i].as_deref())?;
|
||||
|
||||
let process = command.current_dir(&node.data_dir).spawn()?;
|
||||
|
||||
node.process = Some(process);
|
||||
for node_idx in 0..self.nodes.len() {
|
||||
self.spawn_node(node_idx, binary_path, &volumes_arg)?;
|
||||
}
|
||||
|
||||
for (i, node) in self.nodes.iter().enumerate() {
|
||||
@@ -1465,20 +1496,46 @@ impl RustFSTestClusterEnvironment {
|
||||
|
||||
/// Start one node process using the cluster's existing volume layout.
|
||||
pub async fn start_node(&mut self, node_idx: usize) -> Result<(), Box<dyn std::error::Error + Send + Sync>> {
|
||||
let binary_path = rustfs_binary_path();
|
||||
self.start_node_from_binary(node_idx, &binary_path).await
|
||||
}
|
||||
|
||||
/// Start one stopped cluster node with a specific RustFS binary while
|
||||
/// preserving the cluster's volume layout and that node's data directory.
|
||||
pub async fn start_node_from_binary(
|
||||
&mut self,
|
||||
node_idx: usize,
|
||||
binary_path: &Path,
|
||||
) -> Result<(), Box<dyn std::error::Error + Send + Sync>> {
|
||||
let volumes_arg = self.build_volumes_arg();
|
||||
self.spawn_node(node_idx, binary_path, &volumes_arg)?;
|
||||
|
||||
self.wait_for_node_ready(&self.nodes[node_idx].address, node_idx).await?;
|
||||
self.wait_for_node_service_ready(node_idx).await?;
|
||||
Ok(())
|
||||
}
|
||||
|
||||
fn spawn_node(
|
||||
&mut self,
|
||||
node_idx: usize,
|
||||
binary_path: &Path,
|
||||
volumes_arg: &str,
|
||||
) -> Result<(), Box<dyn std::error::Error + Send + Sync>> {
|
||||
self.ensure_node_index(node_idx)?;
|
||||
if self.nodes[node_idx].process.is_some() {
|
||||
return Err(format!("cluster node {node_idx} is already running").into());
|
||||
}
|
||||
if !binary_path.is_file() {
|
||||
return Err(format!("RustFS binary does not exist: {}", binary_path.display()).into());
|
||||
}
|
||||
|
||||
let binary_path = rustfs_binary_path();
|
||||
let volumes_arg = self.build_volumes_arg();
|
||||
let log_path = self.node_capture_log_paths[node_idx].clone();
|
||||
let node = &mut self.nodes[node_idx];
|
||||
info!("Starting cluster node {} on {}", node_idx, node.address);
|
||||
info!("Starting cluster node {} on {} with {}", node_idx, node.address, binary_path.display());
|
||||
|
||||
let mut command = Command::new(&binary_path);
|
||||
let mut command = Command::new(binary_path);
|
||||
command
|
||||
.env("RUSTFS_VOLUMES", &volumes_arg)
|
||||
.env("RUSTFS_VOLUMES", volumes_arg)
|
||||
.env("RUSTFS_ADDRESS", &node.address)
|
||||
.env("RUSTFS_ACCESS_KEY", &self.access_key)
|
||||
.env("RUSTFS_SECRET_KEY", &self.secret_key)
|
||||
@@ -1495,9 +1552,6 @@ impl RustFSTestClusterEnvironment {
|
||||
|
||||
let process = command.current_dir(&node.data_dir).spawn()?;
|
||||
node.process = Some(process);
|
||||
|
||||
self.wait_for_node_ready(&self.nodes[node_idx].address, node_idx).await?;
|
||||
self.wait_for_node_service_ready(node_idx).await?;
|
||||
Ok(())
|
||||
}
|
||||
|
||||
@@ -1645,6 +1699,51 @@ impl RustFSTestClusterEnvironment {
|
||||
process.wait()?;
|
||||
Ok(())
|
||||
}
|
||||
|
||||
/// Gracefully stop one cluster node and wait for its process to exit.
|
||||
///
|
||||
/// This is intentionally separate from [`Self::stop_node`]: the latter is
|
||||
/// a hard kill used by crash-recovery tests, while this path lets RustFS
|
||||
/// complete its normal shutdown hooks before a test restarts the node.
|
||||
pub async fn stop_node_gracefully(&mut self, node_idx: usize) -> Result<(), Box<dyn std::error::Error + Send + Sync>> {
|
||||
self.ensure_node_index(node_idx)?;
|
||||
|
||||
#[cfg(unix)]
|
||||
{
|
||||
let Some(process) = self.nodes[node_idx].process.as_ref() else {
|
||||
return Ok(());
|
||||
};
|
||||
let pid = process.id().to_string();
|
||||
let signal_status = Command::new("kill").args(["-TERM", &pid]).status()?;
|
||||
if !signal_status.success() {
|
||||
return Err(format!("failed to send SIGTERM to cluster node {node_idx} (pid {pid})").into());
|
||||
}
|
||||
|
||||
let mut process = self.nodes[node_idx]
|
||||
.process
|
||||
.take()
|
||||
.ok_or_else(|| format!("cluster node {node_idx} process disappeared while stopping"))?;
|
||||
let deadline = std::time::Instant::now() + Duration::from_secs(45);
|
||||
loop {
|
||||
if let Some(status) = process.try_wait()? {
|
||||
info!("Cluster node {} stopped gracefully with {}", node_idx, status);
|
||||
return Ok(());
|
||||
}
|
||||
if std::time::Instant::now() >= deadline {
|
||||
let _ = process.kill();
|
||||
let _ = process.wait();
|
||||
return Err(format!("cluster node {node_idx} did not stop gracefully within 45 seconds").into());
|
||||
}
|
||||
sleep(Duration::from_millis(100)).await;
|
||||
}
|
||||
}
|
||||
|
||||
#[cfg(not(unix))]
|
||||
{
|
||||
let _ = node_idx;
|
||||
Err("graceful cluster-node stop is only supported on Unix E2E hosts".into())
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
impl Drop for RustFSTestClusterEnvironment {
|
||||
@@ -2000,7 +2099,7 @@ mod tests {
|
||||
}
|
||||
let multidrive = topology.drives_per_node > 1;
|
||||
|
||||
let nodes = (0..topology.node_count)
|
||||
let nodes: Vec<ClusterNode> = (0..topology.node_count)
|
||||
.map(|i| {
|
||||
let address = format!("127.0.0.1:{}", 9000 + i);
|
||||
let data_dirs: Vec<String> = if multidrive {
|
||||
@@ -2021,6 +2120,7 @@ mod tests {
|
||||
})
|
||||
.collect();
|
||||
|
||||
let node_count = nodes.len();
|
||||
RustFSTestClusterEnvironment {
|
||||
nodes,
|
||||
temp_dir,
|
||||
@@ -2030,6 +2130,7 @@ mod tests {
|
||||
node_extra_env: vec![Vec::new(); topology.node_count],
|
||||
node_capture_log_paths: vec![None; topology.node_count],
|
||||
topology,
|
||||
volume_proxy_addresses: vec![None; node_count],
|
||||
}
|
||||
}
|
||||
|
||||
@@ -2114,6 +2215,25 @@ mod tests {
|
||||
assert!(ClusterTopology::single_pool_multidrive(1, 1).validate().is_ok());
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
async fn volume_proxy_rewrites_cluster_volume_endpoint() {
|
||||
let mut env = RustFSTestClusterEnvironment::new(1)
|
||||
.await
|
||||
.expect("cluster environment should allocate a node");
|
||||
let direct = env.nodes[0].address.clone();
|
||||
let proxy = env
|
||||
.start_volume_proxy_for_node(0)
|
||||
.await
|
||||
.expect("volume proxy should bind before the target server starts");
|
||||
let proxied = proxy.local_addr().to_string();
|
||||
let volumes = env.rustfs_volumes_arg();
|
||||
|
||||
assert!(volumes.contains(&proxied), "volumes must use the proxy address: {volumes}");
|
||||
assert!(!volumes.contains(&direct), "volumes must not retain the direct address: {volumes}");
|
||||
|
||||
proxy.shutdown().await;
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn cluster_node_env_supports_per_node_overrides() {
|
||||
let mut env = fake_cluster(ClusterTopology::single_pool(4));
|
||||
|
||||
@@ -0,0 +1,174 @@
|
||||
// Copyright 2024 RustFS Team
|
||||
//
|
||||
// Licensed under the Apache License, Version 2.0 (the "License");
|
||||
// you may not use this file except in compliance with the License.
|
||||
// You may obtain a copy of the License at
|
||||
//
|
||||
// http://www.apache.org/licenses/LICENSE-2.0
|
||||
//
|
||||
// Unless required by applicable law or agreed to in writing, software
|
||||
// distributed under the License is distributed on an "AS IS" BASIS,
|
||||
// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||
// See the License for the specific language governing permissions and
|
||||
// limitations under the License.
|
||||
|
||||
//! Regression: an object legally committed at degraded write quorum must stay
|
||||
//! listable while a *different* drive is offline.
|
||||
//!
|
||||
//! On a 4-drive EC 2+2 set, a PUT made while one drive is down persists
|
||||
//! `xl.meta` on 3 of 4 drives (write quorum). If a different drive later goes
|
||||
//! offline before heal converges, a strict latest-listing quorum of 3 can only
|
||||
//! ever observe 2 copies, so ListObjectsV2 silently dropped the object even
|
||||
//! though GetObject (read quorum 2) still succeeded. Exposed by the flaky
|
||||
//! "Mixed-version rolling upgrade from rc.2" CI lane (run 33478999853); the
|
||||
//! product fix relaxes the listing's required object quorum by the number of
|
||||
//! set drives the listing could not consult (see
|
||||
//! `latest_listing_required_object_quorum` in
|
||||
//! `crates/ecstore/src/store/list_objects.rs`).
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use crate::common::{RustFSTestClusterEnvironment, init_logging};
|
||||
use aws_sdk_s3::Client;
|
||||
use bytes::Bytes;
|
||||
use std::collections::HashSet;
|
||||
use std::error::Error;
|
||||
use std::time::{Duration, Instant};
|
||||
use tracing::info;
|
||||
|
||||
type TestResult = Result<(), Box<dyn Error + Send + Sync>>;
|
||||
|
||||
const BUCKET: &str = "degraded-listing-availability";
|
||||
const OBJECT_COUNT: usize = 8;
|
||||
/// Well under the observed heal-convergence gap (~50s in the CI incident),
|
||||
/// so a listing that only completes after heal restores the missing copy
|
||||
/// still fails this deadline on a regressed build.
|
||||
const LISTING_DEADLINE: Duration = Duration::from_secs(25);
|
||||
const GET_RETRY_DEADLINE: Duration = Duration::from_secs(15);
|
||||
const PUT_RETRY_DEADLINE: Duration = Duration::from_secs(15);
|
||||
|
||||
fn object_key(idx: usize) -> String {
|
||||
format!("degraded-object-{idx:02}")
|
||||
}
|
||||
|
||||
async fn list_all_keys(client: &Client) -> Result<HashSet<String>, Box<dyn Error + Send + Sync>> {
|
||||
let mut keys = HashSet::new();
|
||||
let mut continuation_token: Option<String> = None;
|
||||
loop {
|
||||
let response = client
|
||||
.list_objects_v2()
|
||||
.bucket(BUCKET)
|
||||
.set_continuation_token(continuation_token.clone())
|
||||
.send()
|
||||
.await?;
|
||||
keys.extend(
|
||||
response
|
||||
.contents()
|
||||
.iter()
|
||||
.filter_map(|object| object.key().map(str::to_owned)),
|
||||
);
|
||||
match response.next_continuation_token() {
|
||||
Some(token) => continuation_token = Some(token.to_owned()),
|
||||
None => break,
|
||||
}
|
||||
}
|
||||
Ok(keys)
|
||||
}
|
||||
|
||||
/// 4-node single-drive cluster (EC 2+2, write quorum 3):
|
||||
/// 1. Stop node 1 and PUT objects — each commits on nodes {0, 2, 3} only.
|
||||
/// 2. Stop node 3 (a holder drive), then bring node 1 back before heal can
|
||||
/// recreate the missing copies there.
|
||||
/// 3. Every object still satisfies read quorum (nodes 0 and 2), so GET
|
||||
/// must succeed AND ListObjectsV2 must report every key well before
|
||||
/// heal converges.
|
||||
#[tokio::test]
|
||||
async fn degraded_write_remains_listable_while_a_different_drive_is_offline() -> TestResult {
|
||||
init_logging();
|
||||
|
||||
let mut cluster = RustFSTestClusterEnvironment::new(4).await?;
|
||||
// Listing availability must not depend on heal convergence: disable
|
||||
// the background healers so the degraded objects keep their metadata
|
||||
// on exactly 3 of 4 drives for the whole test.
|
||||
cluster.set_env("RUSTFS_HEAL_ENABLED", "false");
|
||||
cluster.set_env("RUSTFS_SCANNER_ENABLED", "false");
|
||||
cluster.start().await?;
|
||||
cluster.create_test_bucket(BUCKET).await?;
|
||||
let client = cluster.create_s3_client(0)?;
|
||||
|
||||
info!("stopping node 1 so the uploads commit at degraded write quorum (3 of 4)");
|
||||
cluster.stop_node(1)?;
|
||||
// The first writes after a node drops can see transient 503s while the
|
||||
// survivors notice the dead peer; retry briefly (overwrites of the same
|
||||
// unversioned key are idempotent).
|
||||
for idx in 0..OBJECT_COUNT {
|
||||
let key = object_key(idx);
|
||||
let body = format!("degraded listing payload {idx}");
|
||||
let deadline = Instant::now() + PUT_RETRY_DEADLINE;
|
||||
loop {
|
||||
let request = client
|
||||
.put_object()
|
||||
.bucket(BUCKET)
|
||||
.key(&key)
|
||||
.body(Bytes::from(body.clone()).into());
|
||||
match request.send().await {
|
||||
Ok(_) => break,
|
||||
Err(error) if Instant::now() < deadline => {
|
||||
info!("retrying degraded PUT for {key}: {error}");
|
||||
tokio::time::sleep(Duration::from_millis(500)).await;
|
||||
}
|
||||
Err(error) => return Err(format!("degraded PUT for {key} failed: {error}").into()),
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
info!("stopping node 3 (holds a copy) and restoring node 1 (holds none)");
|
||||
cluster.stop_node(3)?;
|
||||
cluster.start_node(1).await?;
|
||||
|
||||
// The first requests after a node drops can see transient 503s while
|
||||
// the survivors notice the dead peer; retry briefly before asserting.
|
||||
for idx in 0..OBJECT_COUNT {
|
||||
let key = object_key(idx);
|
||||
let deadline = Instant::now() + GET_RETRY_DEADLINE;
|
||||
let body = loop {
|
||||
match client.get_object().bucket(BUCKET).key(&key).send().await {
|
||||
Ok(response) => break response.body.collect().await?.into_bytes(),
|
||||
Err(error) if Instant::now() < deadline => {
|
||||
info!("retrying degraded GET for {key}: {error}");
|
||||
tokio::time::sleep(Duration::from_millis(500)).await;
|
||||
}
|
||||
Err(error) => return Err(format!("degraded object {key} failed read quorum GET: {error}").into()),
|
||||
}
|
||||
};
|
||||
assert!(!body.is_empty(), "degraded object {key} should read back at read quorum");
|
||||
}
|
||||
|
||||
let expected: HashSet<String> = (0..OBJECT_COUNT).map(object_key).collect();
|
||||
let deadline = Instant::now() + LISTING_DEADLINE;
|
||||
let listed = loop {
|
||||
let listed = match list_all_keys(&client).await {
|
||||
Ok(keys) => keys,
|
||||
Err(error) if Instant::now() < deadline => {
|
||||
info!("retrying degraded listing: {error}");
|
||||
tokio::time::sleep(Duration::from_millis(500)).await;
|
||||
continue;
|
||||
}
|
||||
Err(error) => return Err(error),
|
||||
};
|
||||
if expected.is_subset(&listed) {
|
||||
break listed;
|
||||
}
|
||||
assert!(
|
||||
Instant::now() < deadline,
|
||||
"objects readable at read quorum stayed missing from ListObjectsV2 for {LISTING_DEADLINE:?}: \
|
||||
missing={:?} listed={listed:?}",
|
||||
expected.difference(&listed).collect::<Vec<_>>(),
|
||||
);
|
||||
tokio::time::sleep(Duration::from_millis(500)).await;
|
||||
};
|
||||
info!(listed = listed.len(), "degraded objects are listable while node 3 is offline");
|
||||
|
||||
Ok(())
|
||||
}
|
||||
}
|
||||
@@ -16,13 +16,14 @@
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use crate::chaos::signed_admin_post;
|
||||
use crate::common::{RustFSTestClusterEnvironment, RustFSTestEnvironment, init_logging};
|
||||
use crate::chaos::{VersionShardCensus, census_object_version_on_disk, signed_admin_post};
|
||||
use crate::common::{RustFSTestClusterEnvironment, RustFSTestEnvironment, admin_request, init_logging};
|
||||
use aws_sdk_s3::primitives::ByteStream;
|
||||
use http::Method;
|
||||
use std::collections::HashSet;
|
||||
use std::error::Error;
|
||||
use std::path::{Path, PathBuf};
|
||||
use tokio::time::{Duration, sleep, timeout};
|
||||
use tokio::time::{Duration, Instant, sleep, timeout};
|
||||
use tracing::info;
|
||||
|
||||
fn has_file_under(path: &Path) -> bool {
|
||||
@@ -48,6 +49,110 @@ mod tests {
|
||||
disk.join(bucket).join(key).join("xl.meta").is_file()
|
||||
}
|
||||
|
||||
// Healing may rewrite non-identity bookkeeping in xl.meta. The census
|
||||
// therefore compares the canonical selected metadata fields plus every
|
||||
// physical shard, while the payload seed makes object mix-ups observable.
|
||||
#[derive(Debug)]
|
||||
struct PhysicalObjectManifest {
|
||||
key: String,
|
||||
payload_seed: u8,
|
||||
shard_census: VersionShardCensus,
|
||||
}
|
||||
|
||||
fn deterministic_object_body(len: usize, seed: u8) -> Vec<u8> {
|
||||
let mut value = seed;
|
||||
std::iter::repeat_with(|| {
|
||||
value = value.wrapping_mul(31).wrapping_add(17);
|
||||
value
|
||||
})
|
||||
.take(len)
|
||||
.collect()
|
||||
}
|
||||
|
||||
fn matching_manifest_count(
|
||||
disk: &Path,
|
||||
bucket: &str,
|
||||
expected_manifests: &[PhysicalObjectManifest],
|
||||
) -> Result<usize, Box<dyn Error + Send + Sync>> {
|
||||
let mut matching = 0;
|
||||
for expected in expected_manifests {
|
||||
let actual = census_object_version_on_disk(disk, bucket, &expected.key, None)?;
|
||||
if actual.matches_manifest(&expected.shard_census) {
|
||||
matching += 1;
|
||||
}
|
||||
}
|
||||
Ok(matching)
|
||||
}
|
||||
|
||||
fn metadata_count(disk: &Path, bucket: &str, expected_manifests: &[PhysicalObjectManifest]) -> usize {
|
||||
expected_manifests
|
||||
.iter()
|
||||
.filter(|expected| object_metadata_exists_on_disk(disk, bucket, &expected.key))
|
||||
.count()
|
||||
}
|
||||
|
||||
fn heal_task_status_diagnostic(body: &str) -> String {
|
||||
let Ok(status) = serde_json::from_str::<serde_json::Value>(body) else {
|
||||
return body.to_string();
|
||||
};
|
||||
let items = status["items"].as_array();
|
||||
let mut unresolved_states = HashSet::new();
|
||||
for item in items.into_iter().flatten() {
|
||||
for drive in item["after"]["drives"].as_array().into_iter().flatten() {
|
||||
if let Some(state) = drive["state"].as_str()
|
||||
&& state != "ok"
|
||||
{
|
||||
unresolved_states.insert(state.to_string());
|
||||
}
|
||||
}
|
||||
}
|
||||
let mut unresolved_states = unresolved_states.into_iter().collect::<Vec<_>>();
|
||||
unresolved_states.sort();
|
||||
format!(
|
||||
"summary={:?}, detail={:?}, item_count={}, unresolved_drive_states={unresolved_states:?}",
|
||||
status["summary"].as_str(),
|
||||
status["detail"].as_str(),
|
||||
items.map_or(0, Vec::len)
|
||||
)
|
||||
}
|
||||
|
||||
fn cluster_heal_is_idle(status: &serde_json::Value) -> bool {
|
||||
let operations = &status["healOperations"];
|
||||
status["clusterStatusComplete"] == serde_json::Value::Bool(true)
|
||||
&& status["state"].as_str() == Some("idle")
|
||||
&& operations["queueLength"].as_u64() == Some(0)
|
||||
&& operations["activeTasks"].as_u64() == Some(0)
|
||||
&& operations["retryingTasks"].as_u64() == Some(0)
|
||||
}
|
||||
|
||||
fn only_admin_heal_is_active(status: &serde_json::Value) -> bool {
|
||||
let operations = &status["healOperations"];
|
||||
status["clusterStatusComplete"] == serde_json::Value::Bool(true)
|
||||
&& status["state"].as_str() == Some("active")
|
||||
&& operations["queueLength"].as_u64() == Some(0)
|
||||
&& operations["activeTasks"].as_u64() == Some(1)
|
||||
&& operations["retryingTasks"].as_u64() == Some(0)
|
||||
&& operations["activeBySource"]["admin"].as_u64() == Some(1)
|
||||
}
|
||||
|
||||
async fn replacement_recovery_status(
|
||||
cluster: &RustFSTestClusterEnvironment,
|
||||
) -> Result<serde_json::Value, Box<dyn Error + Send + Sync>> {
|
||||
let (status, body) = admin_request(
|
||||
&cluster.nodes[0].url,
|
||||
Method::GET,
|
||||
"/rustfs/admin/v4/heal/replacement-recovery",
|
||||
None,
|
||||
&cluster.access_key,
|
||||
&cluster.secret_key,
|
||||
)
|
||||
.await?;
|
||||
if !status.is_success() {
|
||||
return Err(format!("replacement recovery status failed: {status} {body}").into());
|
||||
}
|
||||
serde_json::from_str(&body).map_err(|err| format!("replacement recovery status is not JSON ({err}): {body}").into())
|
||||
}
|
||||
|
||||
async fn assert_object_body(env: &RustFSTestEnvironment, bucket: &str, key: &str, expected: &[u8]) {
|
||||
let client = env.create_s3_client();
|
||||
let response = client
|
||||
@@ -442,6 +547,380 @@ mod tests {
|
||||
.into())
|
||||
}
|
||||
|
||||
// Keep the original unformatted-disk scenario above. This case retains the
|
||||
// format identity so only the explicit admin task can rebuild missing data.
|
||||
#[tokio::test(flavor = "multi_thread")]
|
||||
async fn test_cluster_root_heal_resumes_missing_remote_shards_after_node_restart() -> Result<(), Box<dyn Error + Send + Sync>>
|
||||
{
|
||||
init_logging();
|
||||
info!(
|
||||
event = "heal_restart_started",
|
||||
component = "e2e_test",
|
||||
subsystem = "heal",
|
||||
"Starting root-heal restart test"
|
||||
);
|
||||
|
||||
let mut cluster = RustFSTestClusterEnvironment::new(4).await?;
|
||||
cluster.set_env("RUSTFS_UNSAFE_BYPASS_DISK_CHECK", "true");
|
||||
cluster.set_env("RUSTFS_HEAL_ENABLED", "true");
|
||||
cluster.set_env("RUSTFS_HEAL_AUTO_HEAL_ENABLE", "false");
|
||||
cluster.set_env("RUSTFS_HEAL_MRF_ENABLE", "false");
|
||||
cluster.set_env("RUSTFS_SCANNER_ENABLED", "false");
|
||||
cluster.set_env("RUSTFS_HEAL_MAX_CONCURRENT_HEALS", "1");
|
||||
cluster.set_env("RUSTFS_HEAL_MAX_CONCURRENT_PER_SET", "1");
|
||||
cluster.set_env("RUSTFS_HEAL_PAGE_OBJECT_CONCURRENCY", "1");
|
||||
cluster.set_env("RUSTFS_HEAL_PAGE_PARALLEL_ENABLE", "false");
|
||||
// Keep all storage nodes' Heal runtimes enabled so their disk services
|
||||
// complete normal registration after restart. Scanner, auto-heal and
|
||||
// MRF are disabled; the pre-root idle barrier below drains the direct
|
||||
// outage-object repair before the explicit admin task starts.
|
||||
let server_rust_log = std::env::var("RUSTFS_HEAL_CHAOS_SERVER_RUST_LOG")
|
||||
.unwrap_or_else(|_| "rustfs::heal::task=info,rustfs=error".to_string());
|
||||
cluster.set_env("RUST_LOG", server_rust_log);
|
||||
if let Ok(log_dir) = std::env::var("RUSTFS_HEAL_CHAOS_LOG_DIR") {
|
||||
std::fs::create_dir_all(&log_dir)?;
|
||||
for node_index in 0..cluster.nodes.len() {
|
||||
cluster.set_node_capture_log_path(node_index, format!("{log_dir}/node{node_index}.log"))?;
|
||||
}
|
||||
}
|
||||
cluster.start().await?;
|
||||
let clients = cluster.create_all_clients()?;
|
||||
|
||||
let bucket = "heal-restart-during-rebuild";
|
||||
clients[0].create_bucket().bucket(bucket).send().await?;
|
||||
|
||||
let replaced_disk = PathBuf::from(&cluster.nodes[1].data_dir);
|
||||
let replacement_format_path = replaced_disk.join(".rustfs.sys").join("format.json");
|
||||
let replacement_format = std::fs::read(&replacement_format_path).map_err(|err| {
|
||||
format!("failed to capture target format before replacement wipe at {replacement_format_path:?}: {err}")
|
||||
})?;
|
||||
let online_object_count = std::env::var("RUSTFS_HEAL_CHAOS_OBJECT_COUNT")
|
||||
.ok()
|
||||
.and_then(|value| value.parse::<usize>().ok())
|
||||
.unwrap_or(24)
|
||||
.clamp(8, 64);
|
||||
let object_size_bytes = std::env::var("RUSTFS_HEAL_CHAOS_OBJECT_SIZE_BYTES")
|
||||
.ok()
|
||||
.and_then(|value| value.parse::<usize>().ok())
|
||||
.unwrap_or(4 * 1024 * 1024)
|
||||
.clamp(1024 * 1024, 16 * 1024 * 1024);
|
||||
let mut expected_manifests = Vec::with_capacity(online_object_count);
|
||||
for index in 0..online_object_count {
|
||||
let key = format!("cluster/online/object-{index:04}.bin");
|
||||
let payload_seed = u8::try_from(index + 1).expect("clamped object count must fit in u8");
|
||||
timeout(
|
||||
Duration::from_secs(30),
|
||||
clients[0]
|
||||
.put_object()
|
||||
.bucket(bucket)
|
||||
.key(&key)
|
||||
.body(ByteStream::from(deterministic_object_body(object_size_bytes, payload_seed)))
|
||||
.send(),
|
||||
)
|
||||
.await??;
|
||||
let shard_census = census_object_version_on_disk(&replaced_disk, bucket, &key, None)?;
|
||||
assert!(
|
||||
shard_census.is_complete(),
|
||||
"node 1 should hold a complete baseline shard for {key}: {shard_census:?}"
|
||||
);
|
||||
assert!(
|
||||
!shard_census.expected_part_numbers.is_empty(),
|
||||
"chaos objects must use physical part shards rather than inline data: {shard_census:?}"
|
||||
);
|
||||
expected_manifests.push(PhysicalObjectManifest {
|
||||
key,
|
||||
payload_seed,
|
||||
shard_census,
|
||||
});
|
||||
}
|
||||
|
||||
cluster.stop_node(1)?;
|
||||
std::fs::remove_dir_all(&replaced_disk)?;
|
||||
std::fs::create_dir_all(
|
||||
replacement_format_path
|
||||
.parent()
|
||||
.ok_or("replacement format path has no parent")?,
|
||||
)?;
|
||||
std::fs::write(&replacement_format_path, replacement_format)?;
|
||||
assert!(
|
||||
replacement_format_path.is_file(),
|
||||
"replacement target must retain only its preformatted topology identity"
|
||||
);
|
||||
|
||||
let outage_key = "cluster/written-while-node-down.bin";
|
||||
let outage_payload_seed = 0xf1;
|
||||
timeout(
|
||||
Duration::from_secs(30),
|
||||
clients[2]
|
||||
.put_object()
|
||||
.bucket(bucket)
|
||||
.key(outage_key)
|
||||
.body(ByteStream::from(deterministic_object_body(object_size_bytes, outage_payload_seed)))
|
||||
.send(),
|
||||
)
|
||||
.await??;
|
||||
|
||||
let mut outage_peer_erasure_indices = HashSet::new();
|
||||
for (node_index, node) in cluster.nodes.iter().enumerate() {
|
||||
if node_index == 1 {
|
||||
continue;
|
||||
}
|
||||
let census = census_object_version_on_disk(Path::new(&node.data_dir), bucket, outage_key, None)?;
|
||||
assert!(
|
||||
census.is_complete(),
|
||||
"online node {node_index} must hold a complete outage-object shard: {census:?}"
|
||||
);
|
||||
let erasure_index = census
|
||||
.erasure_index
|
||||
.ok_or_else(|| format!("online node {node_index} outage-object shard has no erasure index: {census:?}"))?;
|
||||
assert!(
|
||||
(1..=cluster.nodes.len()).contains(&erasure_index),
|
||||
"online node {node_index} outage-object erasure index is out of range: {census:?}"
|
||||
);
|
||||
assert!(
|
||||
outage_peer_erasure_indices.insert(erasure_index),
|
||||
"outage-object erasure index {erasure_index} is duplicated across online nodes"
|
||||
);
|
||||
}
|
||||
assert_eq!(
|
||||
outage_peer_erasure_indices.len(),
|
||||
cluster.nodes.len().saturating_sub(1),
|
||||
"every online node must contribute one unique outage-object erasure index"
|
||||
);
|
||||
let expected_outage_target_erasure_index = (1..=cluster.nodes.len())
|
||||
.find(|index| !outage_peer_erasure_indices.contains(index))
|
||||
.ok_or("online outage-object shards leave no erasure index for the replacement target")?;
|
||||
|
||||
// The PUT path may have admitted a direct Internal object repair while
|
||||
// node 1 was offline. Cancel the isolated bucket path before the target
|
||||
// returns; otherwise it could rebuild the outage object and invalidate
|
||||
// the explicit-root ownership assertion below.
|
||||
let cancel_outage_heal_path = format!("/rustfs/admin/v3/heal/{bucket}?forceStop=true");
|
||||
let (cancel_status, cancel_body) = admin_request(
|
||||
&cluster.nodes[0].url,
|
||||
Method::POST,
|
||||
&cancel_outage_heal_path,
|
||||
Some(
|
||||
r#"{"recursive":true,"dryRun":false,"remove":false,"recreate":true,"scanMode":2,"updateParity":false,"nolock":false}"#
|
||||
.to_string(),
|
||||
),
|
||||
&cluster.access_key,
|
||||
&cluster.secret_key,
|
||||
)
|
||||
.await?;
|
||||
if !cancel_status.is_success() {
|
||||
return Err(format!("cancel outage heal failed: {cancel_status} {cancel_body}").into());
|
||||
}
|
||||
|
||||
cluster.start_node(1).await?;
|
||||
|
||||
let status_url = format!("{}/rustfs/admin/v3/background-heal/status", cluster.nodes[0].url);
|
||||
let recovery_deadline = Instant::now() + Duration::from_secs(60);
|
||||
loop {
|
||||
let status_body = signed_admin_post(&status_url, None, &cluster.access_key, &cluster.secret_key).await?;
|
||||
assert!(
|
||||
!status_body.contains("MissingContentLength"),
|
||||
"background heal status should not fail without an explicit Content-Length: {status_body}"
|
||||
);
|
||||
let recovered: serde_json::Value = serde_json::from_str(&status_body)
|
||||
.map_err(|err| format!("background heal status is not JSON ({err}): {status_body}"))?;
|
||||
if cluster_heal_is_idle(&recovered) {
|
||||
break;
|
||||
}
|
||||
if Instant::now() >= recovery_deadline {
|
||||
return Err(format!("cluster heal operations did not become idle before root heal: {recovered}").into());
|
||||
}
|
||||
sleep(Duration::from_millis(250)).await;
|
||||
}
|
||||
assert_eq!(
|
||||
matching_manifest_count(&replaced_disk, bucket, &expected_manifests)?,
|
||||
0,
|
||||
"non-admin Heal is disabled, so the replacement target must remain empty before the explicit root heal"
|
||||
);
|
||||
assert!(
|
||||
!census_object_version_on_disk(&replaced_disk, bucket, outage_key, None)?.has_xl_meta,
|
||||
"the object written during the outage must be absent before the explicit root heal"
|
||||
);
|
||||
let pre_heal_replacement = replacement_recovery_status(&cluster).await?;
|
||||
assert_eq!(
|
||||
pre_heal_replacement["cluster"]["records"].as_array().map(Vec::len),
|
||||
Some(0),
|
||||
"isolated target must not retain an automatic replacement generation: {pre_heal_replacement}"
|
||||
);
|
||||
|
||||
let heal_body = r#"{"recursive":true,"dryRun":false,"remove":false,"recreate":true,"scanMode":2,"updateParity":false,"nolock":false}"#;
|
||||
let heal_url = format!("{}/rustfs/admin/v3/heal/?forceStart=true", cluster.nodes[0].url);
|
||||
let heal_start_body = signed_admin_post(&heal_url, Some(heal_body), &cluster.access_key, &cluster.secret_key).await?;
|
||||
let heal_start: serde_json::Value = serde_json::from_str(&heal_start_body)
|
||||
.map_err(|err| format!("heal start response is not JSON ({err}): {heal_start_body}"))?;
|
||||
let client_token = heal_start["clientToken"]
|
||||
.as_str()
|
||||
.filter(|token| !token.is_empty())
|
||||
.ok_or_else(|| format!("heal start response has no client token: {heal_start}"))?;
|
||||
let task_status_url = format!("{}/rustfs/admin/v3/heal/?clientToken={client_token}", cluster.nodes[0].url);
|
||||
|
||||
let partial_timeout_secs = std::env::var("RUSTFS_HEAL_CHAOS_PARTIAL_TIMEOUT_SECS")
|
||||
.ok()
|
||||
.and_then(|value| value.parse::<u64>().ok())
|
||||
.unwrap_or(60);
|
||||
let partial_deadline = Instant::now() + Duration::from_secs(partial_timeout_secs);
|
||||
let pre_interrupt_status = loop {
|
||||
let status_body = signed_admin_post(&status_url, None, &cluster.access_key, &cluster.secret_key).await?;
|
||||
let active_status: serde_json::Value = serde_json::from_str(&status_body)
|
||||
.map_err(|err| format!("background heal status is not JSON ({err}): {status_body}"))?;
|
||||
if only_admin_heal_is_active(&active_status) {
|
||||
break active_status;
|
||||
}
|
||||
if Instant::now() >= partial_deadline {
|
||||
return Err(format!("root heal never became active within {partial_timeout_secs}s: {active_status}").into());
|
||||
}
|
||||
sleep(Duration::from_millis(50)).await;
|
||||
};
|
||||
let partial_count = loop {
|
||||
let matching = matching_manifest_count(&replaced_disk, bucket, &expected_manifests)?;
|
||||
if matching > 0 && matching < expected_manifests.len() {
|
||||
break matching;
|
||||
}
|
||||
if matching == expected_manifests.len() {
|
||||
return Err(format!(
|
||||
"root heal rebuilt all {} baseline objects before the target could be interrupted",
|
||||
expected_manifests.len()
|
||||
)
|
||||
.into());
|
||||
}
|
||||
if Instant::now() >= partial_deadline {
|
||||
return Err(format!(
|
||||
"root heal made no observable partial progress on the replacement target within {partial_timeout_secs}s"
|
||||
)
|
||||
.into());
|
||||
}
|
||||
sleep(Duration::from_millis(10)).await;
|
||||
};
|
||||
info!(
|
||||
event = "heal_restart_checkpoint",
|
||||
component = "e2e_test",
|
||||
subsystem = "heal",
|
||||
partial_count,
|
||||
"Verified unique admin owner before target interruption"
|
||||
);
|
||||
|
||||
cluster.stop_node(1)?;
|
||||
let stopped_count = matching_manifest_count(&replaced_disk, bucket, &expected_manifests)?;
|
||||
assert!(
|
||||
stopped_count > 0 && stopped_count < expected_manifests.len(),
|
||||
"the target must stop after a partial rebuild, observed before stop={partial_count}, after stop={stopped_count}, total={}",
|
||||
expected_manifests.len()
|
||||
);
|
||||
let unclean_shutdown_marker = replaced_disk.join(".rustfs.sys").join("unclean-shutdown");
|
||||
match std::fs::remove_file(&unclean_shutdown_marker) {
|
||||
Ok(()) => {}
|
||||
Err(error) if error.kind() == std::io::ErrorKind::NotFound => {}
|
||||
Err(error) => {
|
||||
return Err(format!("failed to isolate unclean recovery marker {unclean_shutdown_marker:?}: {error}").into());
|
||||
}
|
||||
}
|
||||
cluster.start_node(1).await?;
|
||||
|
||||
let heal_timeout_secs = std::env::var("RUSTFS_HEAL_REPLACED_DISK_TIMEOUT_SECS")
|
||||
.ok()
|
||||
.and_then(|value| value.parse::<u64>().ok())
|
||||
.unwrap_or(180);
|
||||
let heal_deadline = Instant::now() + Duration::from_secs(heal_timeout_secs);
|
||||
loop {
|
||||
if metadata_count(&replaced_disk, bucket, &expected_manifests) == expected_manifests.len()
|
||||
&& object_metadata_exists_on_disk(&replaced_disk, bucket, outage_key)
|
||||
{
|
||||
let matching = matching_manifest_count(&replaced_disk, bucket, &expected_manifests)?;
|
||||
let outage_census = census_object_version_on_disk(&replaced_disk, bucket, outage_key, None)?;
|
||||
if matching == expected_manifests.len() && outage_census.is_complete() {
|
||||
break;
|
||||
}
|
||||
}
|
||||
if Instant::now() >= heal_deadline {
|
||||
let matching = matching_manifest_count(&replaced_disk, bucket, &expected_manifests)?;
|
||||
let outage_census = census_object_version_on_disk(&replaced_disk, bucket, outage_key, None)?;
|
||||
let final_status = signed_admin_post(&status_url, None, &cluster.access_key, &cluster.secret_key)
|
||||
.await
|
||||
.unwrap_or_else(|err| format!("status request failed: {err}"));
|
||||
let task_status = match timeout(
|
||||
Duration::from_secs(5),
|
||||
signed_admin_post(&task_status_url, None, &cluster.access_key, &cluster.secret_key),
|
||||
)
|
||||
.await
|
||||
{
|
||||
Ok(Ok(body)) => heal_task_status_diagnostic(&body),
|
||||
Ok(Err(err)) => format!("task status request failed: {err}"),
|
||||
Err(_) => "task status request exceeded 5s diagnostic budget".to_string(),
|
||||
};
|
||||
let replacement_status = match timeout(Duration::from_secs(5), replacement_recovery_status(&cluster)).await {
|
||||
Ok(Ok(status)) => status.to_string(),
|
||||
Ok(Err(err)) => format!("replacement status request failed: {err}"),
|
||||
Err(_) => "replacement status request exceeded 5s diagnostic budget".to_string(),
|
||||
};
|
||||
return Err(format!(
|
||||
"root heal did not resume after target restart within {heal_timeout_secs}s: baseline={matching}/{}, outage={outage_census:?}, status={final_status}, task_status={task_status}, pre_interrupt_status={pre_interrupt_status}, pre_heal_replacement={pre_heal_replacement}, replacement_status={replacement_status}",
|
||||
expected_manifests.len()
|
||||
)
|
||||
.into());
|
||||
}
|
||||
sleep(Duration::from_millis(250)).await;
|
||||
}
|
||||
|
||||
for expected in &expected_manifests {
|
||||
let actual = census_object_version_on_disk(&replaced_disk, bucket, &expected.key, None)?;
|
||||
assert!(
|
||||
actual.matches_manifest(&expected.shard_census),
|
||||
"rebuilt target shard differs from its baseline for {}: {actual:?}",
|
||||
expected.key
|
||||
);
|
||||
}
|
||||
let outage_census = census_object_version_on_disk(&replaced_disk, bucket, outage_key, None)?;
|
||||
assert!(
|
||||
outage_census.is_complete(),
|
||||
"outage object must have a complete target shard: {outage_census:?}"
|
||||
);
|
||||
assert_eq!(
|
||||
outage_census.erasure_index,
|
||||
Some(expected_outage_target_erasure_index),
|
||||
"the outage object must be rebuilt into its own missing erasure slot"
|
||||
);
|
||||
|
||||
let target_client = cluster.create_s3_client(1)?;
|
||||
for expected in &expected_manifests {
|
||||
let response = target_client.get_object().bucket(bucket).key(&expected.key).send().await?;
|
||||
let actual = response.body.collect().await?.into_bytes();
|
||||
let expected_body = deterministic_object_body(object_size_bytes, expected.payload_seed);
|
||||
assert_eq!(actual.as_ref(), expected_body.as_slice(), "object body changed for {}", expected.key);
|
||||
}
|
||||
let response = target_client.get_object().bucket(bucket).key(outage_key).send().await?;
|
||||
let actual = response.body.collect().await?.into_bytes();
|
||||
let expected_outage_body = deterministic_object_body(object_size_bytes, outage_payload_seed);
|
||||
assert_eq!(actual.as_ref(), expected_outage_body.as_slice(), "object body changed for {outage_key}");
|
||||
|
||||
let terminal_deadline = Instant::now() + Duration::from_secs(30);
|
||||
loop {
|
||||
let status_body = signed_admin_post(&status_url, None, &cluster.access_key, &cluster.secret_key).await?;
|
||||
let status: serde_json::Value = serde_json::from_str(&status_body)
|
||||
.map_err(|err| format!("background heal status is not JSON ({err}): {status_body}"))?;
|
||||
if cluster_heal_is_idle(&status) {
|
||||
break;
|
||||
}
|
||||
if Instant::now() >= terminal_deadline {
|
||||
return Err(format!("heal data rebuilt but operations did not converge to terminal idle: {status}").into());
|
||||
}
|
||||
sleep(Duration::from_millis(250)).await;
|
||||
}
|
||||
|
||||
let task_status_body = signed_admin_post(&task_status_url, None, &cluster.access_key, &cluster.secret_key).await?;
|
||||
let task_status: serde_json::Value = serde_json::from_str(&task_status_body)
|
||||
.map_err(|err| format!("heal task status is not JSON ({err}): {task_status_body}"))?;
|
||||
if task_status["summary"].as_str() != Some("finished") {
|
||||
return Err(format!("heal data rebuilt but task did not finish successfully: {task_status}").into());
|
||||
}
|
||||
|
||||
Ok(())
|
||||
}
|
||||
|
||||
/// Issue #5850: `background-heal/status` must answer while a peer is down.
|
||||
///
|
||||
/// Exercises the production path in `read_cluster_heal_status` end to end,
|
||||
|
||||
@@ -560,18 +560,12 @@ async fn test_multipart_encryption_type(
|
||||
.set_parts(Some(completed_parts))
|
||||
.build();
|
||||
|
||||
let mut complete_request = s3_client
|
||||
let complete_request = s3_client
|
||||
.complete_multipart_upload()
|
||||
.bucket(bucket)
|
||||
.key(object_key)
|
||||
.upload_id(upload_id)
|
||||
.multipart_upload(completed_multipart_upload);
|
||||
if matches!(encryption_type, EncryptionType::SSEC) {
|
||||
complete_request = complete_request
|
||||
.sse_customer_algorithm("AES256")
|
||||
.sse_customer_key(sse_c_key.as_ref().unwrap())
|
||||
.sse_customer_key_md5(sse_c_md5.as_ref().unwrap());
|
||||
}
|
||||
let _complete_output = complete_request.send().await?;
|
||||
|
||||
// Download and verify
|
||||
|
||||
@@ -348,6 +348,11 @@ mod delete_regression_test;
|
||||
#[cfg(test)]
|
||||
mod listing_regression_test;
|
||||
|
||||
// Cluster regression: objects committed at degraded write quorum must stay
|
||||
// listable while a different drive is offline (CI run 33478999853).
|
||||
#[cfg(test)]
|
||||
mod degraded_listing_availability_test;
|
||||
|
||||
// P1 regression: bucket statistics accuracy (rustfs#5615, #5008, #5116, #5055, #3898, #1012)
|
||||
#[cfg(test)]
|
||||
mod bucket_stats_regression_test;
|
||||
|
||||
@@ -15,7 +15,7 @@
|
||||
//! Regression coverage for anonymous access on multipart control APIs.
|
||||
|
||||
use crate::common::{RustFSTestEnvironment, init_logging, local_http_client};
|
||||
use async_compression::tokio::write::{BzEncoder, XzEncoder};
|
||||
use async_compression::tokio::write::{BzEncoder, Lz4Encoder, XzEncoder};
|
||||
use aws_sdk_s3::error::{ProvideErrorMetadata, SdkError};
|
||||
use aws_sdk_s3::operation::head_object::HeadObjectOutput;
|
||||
use aws_sdk_s3::primitives::ByteStream;
|
||||
@@ -23,7 +23,10 @@ use aws_sdk_s3::types::{
|
||||
ServerSideEncryption, ServerSideEncryptionByDefault, ServerSideEncryptionConfiguration, ServerSideEncryptionRule,
|
||||
};
|
||||
use chrono::{Duration as ChronoDuration, Utc};
|
||||
use flate2::{Compression, write::GzEncoder};
|
||||
use flate2::{
|
||||
Compression,
|
||||
write::{GzEncoder, ZlibEncoder},
|
||||
};
|
||||
use http::HeaderValue;
|
||||
use http::header::{CONTENT_TYPE, HOST};
|
||||
use md5::{Digest as Md5Digest, Md5};
|
||||
@@ -187,6 +190,12 @@ fn gzip_bytes(data: &[u8]) -> Vec<u8> {
|
||||
encoder.finish().expect("gzip encoder should finish")
|
||||
}
|
||||
|
||||
fn zlib_bytes(data: &[u8]) -> Vec<u8> {
|
||||
let mut encoder = ZlibEncoder::new(Vec::new(), Compression::default());
|
||||
encoder.write_all(data).expect("zlib encoder should accept input");
|
||||
encoder.finish().expect("zlib encoder should finish")
|
||||
}
|
||||
|
||||
fn zstd_bytes(data: &[u8]) -> Vec<u8> {
|
||||
let mut encoder = zstd::Encoder::new(Vec::new(), 0).expect("zstd encoder should initialize");
|
||||
encoder.write_all(data).expect("zstd encoder should accept input");
|
||||
@@ -209,6 +218,45 @@ async fn xz_bytes(data: &[u8]) -> Vec<u8> {
|
||||
encoder.into_inner().into_inner()
|
||||
}
|
||||
|
||||
async fn lz4_bytes(data: &[u8]) -> Vec<u8> {
|
||||
let cursor = Cursor::new(Vec::new());
|
||||
let mut encoder = Lz4Encoder::new(cursor);
|
||||
encoder.write_all(data).await.expect("LZ4 encoder should accept input");
|
||||
encoder.shutdown().await.expect("LZ4 encoder should finish");
|
||||
encoder.into_inner().into_inner()
|
||||
}
|
||||
|
||||
/// Encode the S2 framed stream shape emitted by minio-go PutObjectsSnowball
|
||||
/// with `Compress: true`: 1 MiB independent blocks, better compression,
|
||||
/// masked CRC-32C, and the `S2sTwO` stream identifier.
|
||||
fn minio_go_snowball_s2_bytes(data: &[u8]) -> Vec<u8> {
|
||||
const BLOCK_SIZE: usize = 1 << 20;
|
||||
const CHECKSUM_SIZE: usize = 4;
|
||||
|
||||
let mut output = b"\xff\x06\x00\x00S2sTwO".to_vec();
|
||||
let mut encoder = minlz::Encoder::new();
|
||||
for block in data.chunks(BLOCK_SIZE) {
|
||||
let compressed = encoder.encode_better(block);
|
||||
let compressed_limit = block.len().saturating_sub(block.len() / 32).saturating_sub(5);
|
||||
let (chunk_type, payload) = if compressed.len() <= compressed_limit {
|
||||
(0x00, compressed.as_slice())
|
||||
} else {
|
||||
(0x01, block)
|
||||
};
|
||||
let chunk_len = payload.len() + CHECKSUM_SIZE;
|
||||
assert!(chunk_len < 1 << 24, "S2 fixture chunk must fit the 24-bit frame length");
|
||||
output.extend_from_slice(&[
|
||||
chunk_type,
|
||||
(chunk_len & 0xff) as u8,
|
||||
((chunk_len >> 8) & 0xff) as u8,
|
||||
((chunk_len >> 16) & 0xff) as u8,
|
||||
]);
|
||||
output.extend_from_slice(&minlz::crc::crc(block).to_le_bytes());
|
||||
output.extend_from_slice(payload);
|
||||
}
|
||||
output
|
||||
}
|
||||
|
||||
fn assert_s3_error_code<T, E>(result: Result<T, SdkError<E>>, code: &str)
|
||||
where
|
||||
T: std::fmt::Debug,
|
||||
@@ -3456,6 +3504,62 @@ async fn test_signed_put_object_extract_expands_tar_entries_with_prefix_headers(
|
||||
Ok(())
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
async fn test_signed_put_object_extract_ignore_dirs_skips_unauthorized_directory()
|
||||
-> Result<(), Box<dyn std::error::Error + Send + Sync>> {
|
||||
init_logging();
|
||||
|
||||
let mut env = RustFSTestEnvironment::new().await?;
|
||||
env.start_rustfs_server(vec![]).await?;
|
||||
|
||||
let bucket = "signed-extract-ignore-dirs-auth";
|
||||
let archive_key = "bundle.tar";
|
||||
let allowed_member = "allowed/member.txt";
|
||||
let denied_directory = "denied/";
|
||||
let username = "snowball-ignore-dirs";
|
||||
let secret_key = "snowball-ignore-dirs-secret";
|
||||
let expected_body = b"allowed-body";
|
||||
|
||||
let admin_client = env.create_s3_client();
|
||||
admin_client.create_bucket().bucket(bucket).send().await?;
|
||||
create_restricted_user(&env, username, secret_key).await?;
|
||||
|
||||
let policy = serde_json::json!({
|
||||
"Version": "2012-10-17",
|
||||
"Statement": [{
|
||||
"Effect": "Allow",
|
||||
"Principal": { "AWS": [username] },
|
||||
"Action": ["s3:PutObject"],
|
||||
"Resource": [
|
||||
format!("arn:aws:s3:::{bucket}/{archive_key}"),
|
||||
format!("arn:aws:s3:::{bucket}/{allowed_member}")
|
||||
]
|
||||
}]
|
||||
})
|
||||
.to_string();
|
||||
admin_client.put_bucket_policy().bucket(bucket).policy(policy).send().await?;
|
||||
|
||||
let restricted_client = restricted_user_client(&env, username, secret_key);
|
||||
let tar_bytes = make_tar(&[(allowed_member, expected_body)], &[denied_directory]).await;
|
||||
restricted_client
|
||||
.put_object()
|
||||
.bucket(bucket)
|
||||
.key(archive_key)
|
||||
.body(ByteStream::from(tar_bytes))
|
||||
.customize()
|
||||
.mutate_request(|req| {
|
||||
req.headers_mut().insert("x-amz-meta-snowball-auto-extract", "true");
|
||||
req.headers_mut().insert("x-amz-meta-snowball-ignore-dirs", "true");
|
||||
})
|
||||
.send()
|
||||
.await?;
|
||||
|
||||
let stored = admin_client.get_object().bucket(bucket).key(allowed_member).send().await?;
|
||||
assert_eq!(stored.body.collect().await?.into_bytes().as_ref(), expected_body);
|
||||
|
||||
Ok(())
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
async fn test_signed_put_object_extract_preserves_request_metadata_on_extracted_objects()
|
||||
-> Result<(), Box<dyn std::error::Error + Send + Sync>> {
|
||||
@@ -4185,6 +4289,60 @@ async fn test_signed_put_object_extract_returns_archive_etag() -> Result<(), Box
|
||||
Ok(())
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
async fn test_signed_put_object_extract_expands_s2_and_lz4_by_magic_with_raw_etags()
|
||||
-> Result<(), Box<dyn std::error::Error + Send + Sync>> {
|
||||
init_logging();
|
||||
|
||||
let mut env = RustFSTestEnvironment::new().await?;
|
||||
env.start_rustfs_server(vec![]).await?;
|
||||
|
||||
let bucket = "signed-extract-magic-codecs";
|
||||
let client = env.create_s3_client();
|
||||
client.create_bucket().bucket(bucket).send().await?;
|
||||
|
||||
let s2_tar = make_tar(&[("s2/object.txt", b"s2-body")], &[]).await;
|
||||
let s2_archive = minio_go_snowball_s2_bytes(&s2_tar);
|
||||
let expected_s2_etag = format!("\"{}\"", md5_hex(&s2_archive));
|
||||
let s2_response = client
|
||||
.put_object()
|
||||
.bucket(bucket)
|
||||
// minio-go intentionally uploads a compressed S2 stream with a .tar key.
|
||||
.key("snowball-upload-0123456789abcdef.tar")
|
||||
.body(ByteStream::from(s2_archive))
|
||||
.customize()
|
||||
.mutate_request(|req| {
|
||||
req.headers_mut().insert("x-amz-meta-snowball-auto-extract", "true");
|
||||
})
|
||||
.send()
|
||||
.await?;
|
||||
assert_eq!(s2_response.e_tag(), Some(expected_s2_etag.as_str()));
|
||||
|
||||
let s2_object = client.get_object().bucket(bucket).key("s2/object.txt").send().await?;
|
||||
assert_eq!(s2_object.body.collect().await?.into_bytes().as_ref(), b"s2-body");
|
||||
|
||||
let lz4_tar = make_tar(&[("lz4/object.txt", b"lz4-body")], &[]).await;
|
||||
let lz4_archive = lz4_bytes(&lz4_tar).await;
|
||||
let expected_lz4_etag = format!("\"{}\"", md5_hex(&lz4_archive));
|
||||
let lz4_response = client
|
||||
.put_object()
|
||||
.bucket(bucket)
|
||||
.key("also-looks-like-a-plain.tar")
|
||||
.body(ByteStream::from(lz4_archive))
|
||||
.customize()
|
||||
.mutate_request(|req| {
|
||||
req.headers_mut().insert("x-amz-meta-snowball-auto-extract", "true");
|
||||
})
|
||||
.send()
|
||||
.await?;
|
||||
assert_eq!(lz4_response.e_tag(), Some(expected_lz4_etag.as_str()));
|
||||
|
||||
let lz4_object = client.get_object().bucket(bucket).key("lz4/object.txt").send().await?;
|
||||
assert_eq!(lz4_object.body.collect().await?.into_bytes().as_ref(), b"lz4-body");
|
||||
|
||||
Ok(())
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
async fn test_signed_put_object_extract_preserves_entry_mtime() -> Result<(), Box<dyn std::error::Error + Send + Sync>> {
|
||||
init_logging();
|
||||
@@ -4309,9 +4467,15 @@ async fn test_signed_put_object_extract_authorizes_each_pax_privilege_and_retent
|
||||
let context_archive_resources = [
|
||||
format!("arn:aws:s3:::{bucket}/tag-context.tar"),
|
||||
format!("arn:aws:s3:::{bucket}/lock-context.tar"),
|
||||
format!("arn:aws:s3:::{bucket}/legal-hold-context.tar"),
|
||||
format!("arn:aws:s3:::{bucket}/user-agent-bypass.tar"),
|
||||
format!("arn:aws:s3:::{bucket}/sse-bypass.tar"),
|
||||
];
|
||||
let tag_entry_resource = format!("arn:aws:s3:::{bucket}/tag-context-entry.txt");
|
||||
let lock_entry_resource = format!("arn:aws:s3:::{bucket}/lock-context-entry.txt");
|
||||
let legal_hold_entry_resource = format!("arn:aws:s3:::{bucket}/legal-hold-context-entry.txt");
|
||||
let user_agent_entry_resource = format!("arn:aws:s3:::{bucket}/user-agent-bypass-entry.txt");
|
||||
let sse_entry_resource = format!("arn:aws:s3:::{bucket}/sse-bypass-entry.txt");
|
||||
let policy = serde_json::json!({
|
||||
"Version": "2012-10-17",
|
||||
"Statement": [
|
||||
@@ -4371,7 +4535,7 @@ async fn test_signed_put_object_extract_authorizes_each_pax_privilege_and_retent
|
||||
"Sid": "PaxContextArchives",
|
||||
"Effect": "Allow",
|
||||
"Principal": { "AWS": [pax_context_user] },
|
||||
"Action": ["s3:PutObject", "s3:PutObjectRetention", "s3:PutObjectTagging"],
|
||||
"Action": ["s3:PutObject", "s3:PutObjectRetention", "s3:PutObjectLegalHold", "s3:PutObjectTagging"],
|
||||
"Resource": context_archive_resources
|
||||
},
|
||||
{
|
||||
@@ -4411,6 +4575,49 @@ async fn test_signed_put_object_extract_authorizes_each_pax_privilege_and_retent
|
||||
"Principal": { "AWS": [pax_context_user] },
|
||||
"Action": ["s3:PutObjectRetention"],
|
||||
"Resource": [lock_entry_resource]
|
||||
},
|
||||
{
|
||||
"Sid": "PaxLegalHoldContextPut",
|
||||
"Effect": "Allow",
|
||||
"Principal": { "AWS": [pax_context_user] },
|
||||
"Action": ["s3:PutObject"],
|
||||
"Resource": [legal_hold_entry_resource.clone()]
|
||||
},
|
||||
{
|
||||
"Sid": "PaxLegalHoldContextAction",
|
||||
"Effect": "Allow",
|
||||
"Principal": { "AWS": [pax_context_user] },
|
||||
"Action": ["s3:PutObjectLegalHold"],
|
||||
"Resource": [legal_hold_entry_resource],
|
||||
"Condition": {
|
||||
"StringEquals": {
|
||||
"s3:object-lock-legal-hold": "OFF"
|
||||
}
|
||||
}
|
||||
},
|
||||
{
|
||||
"Sid": "MemberUserAgentCondition",
|
||||
"Effect": "Allow",
|
||||
"Principal": { "AWS": [pax_context_user] },
|
||||
"Action": ["s3:PutObject"],
|
||||
"Resource": [user_agent_entry_resource],
|
||||
"Condition": {
|
||||
"StringEquals": {
|
||||
"aws:UserAgent": "trusted"
|
||||
}
|
||||
}
|
||||
},
|
||||
{
|
||||
"Sid": "MemberSseCondition",
|
||||
"Effect": "Allow",
|
||||
"Principal": { "AWS": [pax_context_user] },
|
||||
"Action": ["s3:PutObject"],
|
||||
"Resource": [sse_entry_resource],
|
||||
"Condition": {
|
||||
"StringEquals": {
|
||||
"s3:x-amz-server-side-encryption": "AES256"
|
||||
}
|
||||
}
|
||||
}
|
||||
]
|
||||
})
|
||||
@@ -4423,9 +4630,14 @@ async fn test_signed_put_object_extract_authorizes_each_pax_privilege_and_retent
|
||||
let cases = [
|
||||
(
|
||||
"legal-hold.tar",
|
||||
put_only_client,
|
||||
put_only_client.clone(),
|
||||
HashMap::from([("minio.metadata.x-amz-object-lock-legal-hold", "ON".to_string())]),
|
||||
),
|
||||
(
|
||||
"tagging.tar",
|
||||
put_only_client,
|
||||
HashMap::from([("minio.metadata.x-amz-tagging", "classification=restricted".to_string())]),
|
||||
),
|
||||
(
|
||||
"retention-condition.tar",
|
||||
conditional_client,
|
||||
@@ -4512,6 +4724,57 @@ async fn test_signed_put_object_extract_authorizes_each_pax_privilege_and_retent
|
||||
assert_eq!(stored.body.collect().await?.into_bytes().as_ref(), b"condition-body");
|
||||
|
||||
let pax_context_client = restricted_user_client(&env, pax_context_user, pax_context_secret);
|
||||
for (archive_key, entry_key, pax_key, injected_value, outer_user_agent) in [
|
||||
(
|
||||
"user-agent-bypass.tar",
|
||||
"user-agent-bypass-entry.txt",
|
||||
"minio.metadata.user-agent",
|
||||
"trusted",
|
||||
Some("untrusted"),
|
||||
),
|
||||
(
|
||||
"sse-bypass.tar",
|
||||
"sse-bypass-entry.txt",
|
||||
"minio.metadata.x-amz-server-side-encryption",
|
||||
"AES256",
|
||||
None,
|
||||
),
|
||||
] {
|
||||
let pax = HashMap::from([(pax_key, injected_value.to_string())]);
|
||||
let archive = make_tar_with_pax_entry(entry_key, b"must-not-write", None, &pax).await;
|
||||
let err = pax_context_client
|
||||
.put_object()
|
||||
.bucket(bucket)
|
||||
.key(archive_key)
|
||||
.body(ByteStream::from(archive))
|
||||
.customize()
|
||||
.mutate_request(move |req| {
|
||||
req.headers_mut().insert("x-amz-meta-snowball-auto-extract", "true");
|
||||
if let Some(user_agent) = outer_user_agent {
|
||||
req.headers_mut().insert("user-agent", user_agent);
|
||||
}
|
||||
})
|
||||
.send()
|
||||
.await
|
||||
.expect_err("PAX metadata must not satisfy unrelated IAM request conditions");
|
||||
assert_eq!(
|
||||
err.as_service_error().and_then(|error| error.meta().code()),
|
||||
Some("AccessDenied"),
|
||||
"{archive_key}"
|
||||
);
|
||||
let err = admin_client
|
||||
.head_object()
|
||||
.bucket(bucket)
|
||||
.key(entry_key)
|
||||
.send()
|
||||
.await
|
||||
.expect_err("a denied PAX member must not be written");
|
||||
assert!(matches!(
|
||||
err.as_service_error().and_then(|error| error.meta().code()),
|
||||
Some("NoSuchKey" | "NotFound")
|
||||
));
|
||||
}
|
||||
|
||||
let tag_pax = HashMap::from([("minio.metadata.x-amz-tagging", "classification=public".to_string())]);
|
||||
let archive = make_tar_with_pax_entry("tag-context-entry.txt", b"tag-context-body", None, &tag_pax).await;
|
||||
pax_context_client
|
||||
@@ -4575,6 +4838,34 @@ async fn test_signed_put_object_extract_authorizes_each_pax_privilege_and_retent
|
||||
pax_retain_until
|
||||
);
|
||||
|
||||
let legal_hold_pax = HashMap::from([("minio.metadata.x-amz-object-lock-legal-hold", "ON".to_string())]);
|
||||
let archive = make_tar_with_pax_entry("legal-hold-context-entry.txt", b"must-not-write", None, &legal_hold_pax).await;
|
||||
let err = pax_context_client
|
||||
.put_object()
|
||||
.bucket(bucket)
|
||||
.key("legal-hold-context.tar")
|
||||
.object_lock_legal_hold_status(aws_sdk_s3::types::ObjectLockLegalHoldStatus::Off)
|
||||
.body(ByteStream::from(archive))
|
||||
.customize()
|
||||
.mutate_request(|req| {
|
||||
req.headers_mut().insert("x-amz-meta-snowball-auto-extract", "true");
|
||||
})
|
||||
.send()
|
||||
.await
|
||||
.expect_err("PAX legal hold must replace the outer value in the member IAM condition context");
|
||||
assert_eq!(err.as_service_error().and_then(|error| error.meta().code()), Some("AccessDenied"));
|
||||
let err = admin_client
|
||||
.head_object()
|
||||
.bucket(bucket)
|
||||
.key("legal-hold-context-entry.txt")
|
||||
.send()
|
||||
.await
|
||||
.expect_err("a denied PAX legal-hold member must not be written");
|
||||
assert!(matches!(
|
||||
err.as_service_error().and_then(|error| error.meta().code()),
|
||||
Some("NoSuchKey" | "NotFound")
|
||||
));
|
||||
|
||||
Ok(())
|
||||
}
|
||||
|
||||
@@ -5050,8 +5341,8 @@ async fn test_signed_put_object_extract_expands_tzst_archive() -> Result<(), Box
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
async fn test_signed_put_object_extract_rejects_missing_archive_extension() -> Result<(), Box<dyn std::error::Error + Send + Sync>>
|
||||
{
|
||||
async fn test_signed_put_object_extract_uses_magic_without_requiring_or_trusting_extension()
|
||||
-> Result<(), Box<dyn std::error::Error + Send + Sync>> {
|
||||
init_logging();
|
||||
|
||||
let mut env = RustFSTestEnvironment::new().await?;
|
||||
@@ -5064,8 +5355,7 @@ async fn test_signed_put_object_extract_rejects_missing_archive_extension() -> R
|
||||
admin_client.create_bucket().bucket(bucket).send().await?;
|
||||
|
||||
let tar_bytes = make_tar(&[("plain.txt", b"plain-body")], &[]).await;
|
||||
|
||||
let result = admin_client
|
||||
admin_client
|
||||
.put_object()
|
||||
.bucket(bucket)
|
||||
.key(archive_key)
|
||||
@@ -5075,15 +5365,80 @@ async fn test_signed_put_object_extract_rejects_missing_archive_extension() -> R
|
||||
req.headers_mut().insert("x-amz-meta-snowball-auto-extract", "true");
|
||||
})
|
||||
.send()
|
||||
.await;
|
||||
.await?;
|
||||
|
||||
assert_s3_error_code(result, "InvalidArgument");
|
||||
let plain = admin_client.get_object().bucket(bucket).key("plain.txt").send().await?;
|
||||
assert_eq!(plain.body.collect().await?.into_bytes().as_ref(), b"plain-body");
|
||||
|
||||
let raw_with_gzip_suffix = make_tar(&[("raw-with-wrong-suffix.txt", b"raw-body")], &[]).await;
|
||||
admin_client
|
||||
.put_object()
|
||||
.bucket(bucket)
|
||||
.key("raw-but-named.tar.gz")
|
||||
.body(ByteStream::from(raw_with_gzip_suffix))
|
||||
.customize()
|
||||
.mutate_request(|req| {
|
||||
req.headers_mut().insert("x-amz-meta-snowball-auto-extract", "true");
|
||||
})
|
||||
.send()
|
||||
.await?;
|
||||
|
||||
let raw = admin_client
|
||||
.get_object()
|
||||
.bucket(bucket)
|
||||
.key("raw-with-wrong-suffix.txt")
|
||||
.send()
|
||||
.await?;
|
||||
assert_eq!(raw.body.collect().await?.into_bytes().as_ref(), b"raw-body");
|
||||
|
||||
let gzip_with_tar_suffix = gzip_bytes(&make_tar(&[("gzip-with-wrong-suffix.txt", b"gzip-body")], &[]).await);
|
||||
admin_client
|
||||
.put_object()
|
||||
.bucket(bucket)
|
||||
.key("gzip-but-named.tar")
|
||||
.body(ByteStream::from(gzip_with_tar_suffix))
|
||||
.customize()
|
||||
.mutate_request(|req| {
|
||||
req.headers_mut().insert("x-amz-meta-snowball-auto-extract", "true");
|
||||
})
|
||||
.send()
|
||||
.await?;
|
||||
|
||||
let gzip = admin_client
|
||||
.get_object()
|
||||
.bucket(bucket)
|
||||
.key("gzip-with-wrong-suffix.txt")
|
||||
.send()
|
||||
.await?;
|
||||
assert_eq!(gzip.body.collect().await?.into_bytes().as_ref(), b"gzip-body");
|
||||
|
||||
let zlib_archive = zlib_bytes(&make_tar(&[("zlib-extension.txt", b"zlib-body")], &[]).await);
|
||||
admin_client
|
||||
.put_object()
|
||||
.bucket(bucket)
|
||||
.key("bundle.zlib")
|
||||
.body(ByteStream::from(zlib_archive))
|
||||
.customize()
|
||||
.mutate_request(|req| {
|
||||
req.headers_mut().insert("x-amz-meta-snowball-auto-extract", "true");
|
||||
})
|
||||
.send()
|
||||
.await?;
|
||||
|
||||
let zlib = admin_client
|
||||
.get_object()
|
||||
.bucket(bucket)
|
||||
.key("zlib-extension.txt")
|
||||
.send()
|
||||
.await?;
|
||||
assert_eq!(zlib.body.collect().await?.into_bytes().as_ref(), b"zlib-body");
|
||||
|
||||
Ok(())
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
async fn test_signed_put_object_extract_rejects_invalid_tar_gz_payload() -> Result<(), Box<dyn std::error::Error + Send + Sync>> {
|
||||
async fn test_signed_put_object_extract_rejects_invalid_archive_payload() -> Result<(), Box<dyn std::error::Error + Send + Sync>>
|
||||
{
|
||||
init_logging();
|
||||
|
||||
let mut env = RustFSTestEnvironment::new().await?;
|
||||
|
||||
@@ -36,9 +36,10 @@
|
||||
use crate::common::{RustFSTestEnvironment, init_logging, local_http_client};
|
||||
use aws_sdk_s3::error::ProvideErrorMetadata;
|
||||
use aws_sdk_s3::primitives::ByteStream;
|
||||
use rustfs_signer::constants::UNSIGNED_PAYLOAD;
|
||||
use rustfs_signer::constants::{UNSIGNED_PAYLOAD, UNSIGNED_PAYLOAD_TRAILER};
|
||||
use rustfs_signer::request_signature_v4::{SIGN_V4_ALGORITHM, get_scope, get_signature, get_signing_key};
|
||||
use std::fmt::Write as _;
|
||||
use std::io::Cursor;
|
||||
use time::macros::format_description;
|
||||
use time::{Duration, OffsetDateTime};
|
||||
use tracing::info;
|
||||
@@ -98,15 +99,37 @@ impl SigV4 {
|
||||
/// header AND folded into the canonical request — pass the hash of the
|
||||
/// body you *claim* to send, which may differ from what you actually send.
|
||||
fn sign(&self, method: &str, path: &str, canonical_query: &str, content_sha256: &str) -> SignedHeaders {
|
||||
let amz_date = amz_datetime(self.time);
|
||||
let signed_headers = "host;x-amz-content-sha256;x-amz-date";
|
||||
self.sign_with_extra_headers(method, path, canonical_query, content_sha256, &[])
|
||||
}
|
||||
|
||||
let canonical_headers = format!(
|
||||
"host:{host}\nx-amz-content-sha256:{sha}\nx-amz-date:{date}\n",
|
||||
host = self.host,
|
||||
sha = content_sha256,
|
||||
date = amz_date,
|
||||
);
|
||||
/// Sign additional request headers while preserving SigV4's lowercase,
|
||||
/// lexicographically sorted canonical-header representation.
|
||||
fn sign_with_extra_headers(
|
||||
&self,
|
||||
method: &str,
|
||||
path: &str,
|
||||
canonical_query: &str,
|
||||
content_sha256: &str,
|
||||
extra_signed_headers: &[(&str, &str)],
|
||||
) -> SignedHeaders {
|
||||
let amz_date = amz_datetime(self.time);
|
||||
let mut canonical_header_values = vec![
|
||||
("host", self.host.as_str()),
|
||||
("x-amz-content-sha256", content_sha256),
|
||||
("x-amz-date", amz_date.as_str()),
|
||||
];
|
||||
canonical_header_values.extend(extra_signed_headers.iter().copied());
|
||||
canonical_header_values.sort_unstable_by(|left, right| left.0.cmp(right.0));
|
||||
|
||||
let signed_headers = canonical_header_values
|
||||
.iter()
|
||||
.map(|(name, _)| *name)
|
||||
.collect::<Vec<_>>()
|
||||
.join(";");
|
||||
let mut canonical_headers = String::new();
|
||||
for (name, value) in canonical_header_values {
|
||||
let _ = writeln!(canonical_headers, "{name}:{value}");
|
||||
}
|
||||
let canonical_request =
|
||||
format!("{method}\n{path}\n{canonical_query}\n{canonical_headers}\n{signed_headers}\n{content_sha256}");
|
||||
|
||||
@@ -179,6 +202,34 @@ async fn setup(env: &mut RustFSTestEnvironment) -> Result<(), Box<dyn std::error
|
||||
Ok(())
|
||||
}
|
||||
|
||||
async fn build_single_member_archive(
|
||||
member_key: &str,
|
||||
member_body: &[u8],
|
||||
) -> Result<Vec<u8>, Box<dyn std::error::Error + Send + Sync>> {
|
||||
let mut builder = tokio_tar::Builder::new(Cursor::new(Vec::new()));
|
||||
let mut header = tokio_tar::Header::new_gnu();
|
||||
header.set_size(member_body.len() as u64);
|
||||
header.set_mode(0o644);
|
||||
header.set_cksum();
|
||||
builder.append_data(&mut header, member_key, Cursor::new(member_body)).await?;
|
||||
Ok(builder.into_inner().await?.into_inner())
|
||||
}
|
||||
|
||||
fn sha256_base64(data: &[u8]) -> String {
|
||||
use sha2::{Digest, Sha256};
|
||||
|
||||
base64_simd::STANDARD.encode_to_string(Sha256::digest(data))
|
||||
}
|
||||
|
||||
fn encode_unsigned_aws_chunked_with_sha256_trailer(decoded: &[u8]) -> Vec<u8> {
|
||||
let checksum = sha256_base64(decoded);
|
||||
let mut encoded = format!("{:x}\r\n", decoded.len()).into_bytes();
|
||||
encoded.extend_from_slice(decoded);
|
||||
encoded.extend_from_slice(b"\r\n0\r\n\r\n");
|
||||
encoded.extend_from_slice(format!("x-amz-checksum-sha256:{checksum}").as_bytes());
|
||||
encoded
|
||||
}
|
||||
|
||||
/// Positive control: a correctly hand-signed request must succeed. Without
|
||||
/// this, every negative assertion below could pass for the wrong reason (a
|
||||
/// broken signer that never produces a valid signature).
|
||||
@@ -249,6 +300,128 @@ async fn tampered_signature_returns_signature_does_not_match() -> Result<(), Box
|
||||
Ok(())
|
||||
}
|
||||
|
||||
/// `STREAMING-UNSIGNED-PAYLOAD-TRAILER` disables per-chunk signatures, not the
|
||||
/// seed/header SigV4 signature. A forged request must be rejected before the
|
||||
/// Snowball handler can publish any archive member.
|
||||
#[tokio::test]
|
||||
async fn snowball_streaming_unsigned_trailer_rejects_forged_signature() -> Result<(), Box<dyn std::error::Error + Send + Sync>> {
|
||||
init_logging();
|
||||
let mut env = RustFSTestEnvironment::new().await?;
|
||||
setup(&mut env).await?;
|
||||
|
||||
let archive_key = "forged-streaming-snowball.tar";
|
||||
let member_key = "must-not-be-published.txt";
|
||||
let archive = build_single_member_archive(member_key, b"forged request payload").await?;
|
||||
let decoded_content_length = archive.len().to_string();
|
||||
let encoded_body = encode_unsigned_aws_chunked_with_sha256_trailer(&archive);
|
||||
let path = format!("/{BUCKET}/{archive_key}");
|
||||
|
||||
let mut signer = SigV4::new(&env);
|
||||
signer.secret_key = "wrong-secret-for-forged-streaming-request".to_string();
|
||||
let extra_signed_headers = [
|
||||
("content-encoding", "aws-chunked"),
|
||||
("x-amz-decoded-content-length", decoded_content_length.as_str()),
|
||||
("x-amz-meta-snowball-auto-extract", "true"),
|
||||
("x-amz-trailer", "x-amz-checksum-sha256"),
|
||||
];
|
||||
let headers = signer.sign_with_extra_headers("PUT", &path, "", UNSIGNED_PAYLOAD_TRAILER, &extra_signed_headers);
|
||||
|
||||
let response = local_http_client()
|
||||
.put(format!("{}{}", env.url, path))
|
||||
.header("authorization", &headers.authorization)
|
||||
.header("content-encoding", "aws-chunked")
|
||||
.header("x-amz-content-sha256", &headers.content_sha256)
|
||||
.header("x-amz-date", &headers.amz_date)
|
||||
.header("x-amz-decoded-content-length", &decoded_content_length)
|
||||
.header("x-amz-meta-snowball-auto-extract", "true")
|
||||
.header("x-amz-trailer", "x-amz-checksum-sha256")
|
||||
.body(encoded_body)
|
||||
.send()
|
||||
.await?;
|
||||
let status = response.status();
|
||||
let body = response.text().await?;
|
||||
assert_eq!(status.as_u16(), 403, "forged streaming signature must be 403, body:\n{body}");
|
||||
assert_error_code(&body, "SignatureDoesNotMatch");
|
||||
|
||||
let absent = env
|
||||
.create_s3_client()
|
||||
.get_object()
|
||||
.bucket(BUCKET)
|
||||
.key(member_key)
|
||||
.send()
|
||||
.await
|
||||
.expect_err("a forged streaming request must not publish a Snowball member");
|
||||
assert_eq!(absent.raw_response().map(|response| response.status().as_u16()), Some(404));
|
||||
assert_eq!(absent.as_service_error().and_then(ProvideErrorMetadata::code), Some("NoSuchKey"));
|
||||
|
||||
env.stop_server();
|
||||
Ok(())
|
||||
}
|
||||
|
||||
/// Snowball must consume the complete aws-chunked body before reading the
|
||||
/// trailing checksum exported by s3s into the PutObject response.
|
||||
#[tokio::test]
|
||||
async fn snowball_streaming_unsigned_trailer_returns_sha256_checksum() -> Result<(), Box<dyn std::error::Error + Send + Sync>> {
|
||||
init_logging();
|
||||
let mut env = RustFSTestEnvironment::new().await?;
|
||||
setup(&mut env).await?;
|
||||
|
||||
let archive_key = "valid-streaming-snowball.tar";
|
||||
let member_key = "streaming-checksum-member.txt";
|
||||
let member_body = b"valid streaming Snowball payload";
|
||||
let archive = build_single_member_archive(member_key, member_body).await?;
|
||||
let expected_checksum = sha256_base64(&archive);
|
||||
let decoded_content_length = archive.len().to_string();
|
||||
let encoded_body = encode_unsigned_aws_chunked_with_sha256_trailer(&archive);
|
||||
let path = format!("/{BUCKET}/{archive_key}");
|
||||
|
||||
let signer = SigV4::new(&env);
|
||||
let extra_signed_headers = [
|
||||
("content-encoding", "aws-chunked"),
|
||||
("x-amz-decoded-content-length", decoded_content_length.as_str()),
|
||||
("x-amz-meta-snowball-auto-extract", "true"),
|
||||
("x-amz-sdk-checksum-algorithm", "SHA256"),
|
||||
("x-amz-trailer", "x-amz-checksum-sha256"),
|
||||
];
|
||||
let headers = signer.sign_with_extra_headers("PUT", &path, "", UNSIGNED_PAYLOAD_TRAILER, &extra_signed_headers);
|
||||
|
||||
let response = local_http_client()
|
||||
.put(format!("{}{}", env.url, path))
|
||||
.header("authorization", &headers.authorization)
|
||||
.header("content-encoding", "aws-chunked")
|
||||
.header("x-amz-content-sha256", &headers.content_sha256)
|
||||
.header("x-amz-date", &headers.amz_date)
|
||||
.header("x-amz-decoded-content-length", &decoded_content_length)
|
||||
.header("x-amz-meta-snowball-auto-extract", "true")
|
||||
.header("x-amz-sdk-checksum-algorithm", "SHA256")
|
||||
.header("x-amz-trailer", "x-amz-checksum-sha256")
|
||||
.body(encoded_body)
|
||||
.send()
|
||||
.await?;
|
||||
let status = response.status();
|
||||
let response_checksum = response
|
||||
.headers()
|
||||
.get("x-amz-checksum-sha256")
|
||||
.and_then(|value| value.to_str().ok())
|
||||
.map(str::to_owned);
|
||||
let response_body = response.text().await?;
|
||||
assert_eq!(status.as_u16(), 200, "valid streaming Snowball PUT failed, body:\n{response_body}");
|
||||
assert_eq!(response_checksum.as_deref(), Some(expected_checksum.as_str()));
|
||||
|
||||
let member = env
|
||||
.create_s3_client()
|
||||
.get_object()
|
||||
.bucket(BUCKET)
|
||||
.key(member_key)
|
||||
.send()
|
||||
.await?;
|
||||
let stored = member.body.collect().await?.into_bytes();
|
||||
assert_eq!(stored.as_ref(), member_body);
|
||||
|
||||
env.stop_server();
|
||||
Ok(())
|
||||
}
|
||||
|
||||
/// (b) A valid AccessKeyId paired with the wrong secret key must be rejected
|
||||
/// with SignatureDoesNotMatch / 403.
|
||||
#[tokio::test]
|
||||
|
||||
@@ -21,5 +21,6 @@ mod head_tls_bodyless_test;
|
||||
mod lifecycle;
|
||||
mod lock;
|
||||
mod node_interact_test;
|
||||
mod s3_select_compression;
|
||||
mod sql;
|
||||
mod tiering;
|
||||
|
||||
@@ -0,0 +1,351 @@
|
||||
#![cfg(test)]
|
||||
// Copyright 2024 RustFS Team
|
||||
//
|
||||
// Licensed under the Apache License, Version 2.0 (the "License");
|
||||
// you may not use this file except in compliance with the License.
|
||||
// You may obtain a copy of the License at
|
||||
//
|
||||
// http://www.apache.org/licenses/LICENSE-2.0
|
||||
//
|
||||
// Unless required by applicable law or agreed to in writing, software
|
||||
// distributed under the License is distributed on an "AS IS" BASIS,
|
||||
// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||
// See the License for the specific language governing permissions and
|
||||
// limitations under the License.
|
||||
|
||||
use crate::common::{RustFSTestEnvironment, init_logging};
|
||||
use async_compression::tokio::write::BzEncoder;
|
||||
use aws_sdk_s3::{
|
||||
Client,
|
||||
error::ProvideErrorMetadata,
|
||||
operation::select_object_content::{SelectObjectContentOutput, builders::SelectObjectContentFluentBuilder},
|
||||
types::{
|
||||
CompressionType, CsvInput, CsvOutput, ExpressionType, FileHeaderInfo, InputSerialization, JsonInput, JsonOutput,
|
||||
JsonType, OutputSerialization, SelectObjectContentEventStream,
|
||||
},
|
||||
};
|
||||
use aws_smithy_types::event_stream::RawMessage;
|
||||
use bytes::Bytes;
|
||||
use flate2::{Compression, write::GzEncoder};
|
||||
use std::{error::Error, io::Cursor, time::Duration};
|
||||
use tokio::io::AsyncWriteExt;
|
||||
|
||||
const BUCKET: &str = "s3-select-compression";
|
||||
const SELECT_RESPONSE_TIMEOUT: Duration = Duration::from_secs(30);
|
||||
|
||||
type TestResult<T> = Result<T, Box<dyn Error + Send + Sync>>;
|
||||
|
||||
async fn create_test_environment(extra_env: &[(&str, &str)]) -> TestResult<(RustFSTestEnvironment, Client)> {
|
||||
init_logging();
|
||||
let mut env = RustFSTestEnvironment::new().await?;
|
||||
env.start_rustfs_server_with_env(vec![], extra_env).await?;
|
||||
let client = env.create_s3_client();
|
||||
client.create_bucket().bucket(BUCKET).send().await?;
|
||||
Ok((env, client))
|
||||
}
|
||||
|
||||
async fn put_object(client: &Client, key: &str, body: &[u8]) -> TestResult<()> {
|
||||
client
|
||||
.put_object()
|
||||
.bucket(BUCKET)
|
||||
.key(key)
|
||||
.body(Bytes::copy_from_slice(body).into())
|
||||
.send()
|
||||
.await?;
|
||||
Ok(())
|
||||
}
|
||||
|
||||
fn gzip(input: &[u8]) -> TestResult<Vec<u8>> {
|
||||
let mut encoder = GzEncoder::new(Vec::new(), Compression::default());
|
||||
std::io::Write::write_all(&mut encoder, input)?;
|
||||
Ok(encoder.finish()?)
|
||||
}
|
||||
|
||||
async fn bzip2(input: &[u8]) -> TestResult<Vec<u8>> {
|
||||
let mut encoder = BzEncoder::new(Cursor::new(Vec::new()));
|
||||
encoder.write_all(input).await?;
|
||||
encoder.shutdown().await?;
|
||||
Ok(encoder.into_inner().into_inner())
|
||||
}
|
||||
|
||||
fn csv_select_request(
|
||||
client: &Client,
|
||||
key: &str,
|
||||
compression: CompressionType,
|
||||
expression: &str,
|
||||
) -> SelectObjectContentFluentBuilder {
|
||||
client
|
||||
.select_object_content()
|
||||
.bucket(BUCKET)
|
||||
.key(key)
|
||||
.expression(expression)
|
||||
.expression_type(ExpressionType::Sql)
|
||||
.input_serialization(
|
||||
InputSerialization::builder()
|
||||
.compression_type(compression)
|
||||
.csv(CsvInput::builder().file_header_info(FileHeaderInfo::Use).build())
|
||||
.build(),
|
||||
)
|
||||
.output_serialization(OutputSerialization::builder().csv(CsvOutput::builder().build()).build())
|
||||
}
|
||||
|
||||
fn json_select_request(
|
||||
client: &Client,
|
||||
key: &str,
|
||||
compression: CompressionType,
|
||||
json_type: JsonType,
|
||||
) -> SelectObjectContentFluentBuilder {
|
||||
client
|
||||
.select_object_content()
|
||||
.bucket(BUCKET)
|
||||
.key(key)
|
||||
.expression("SELECT name FROM S3Object")
|
||||
.expression_type(ExpressionType::Sql)
|
||||
.input_serialization(
|
||||
InputSerialization::builder()
|
||||
.compression_type(compression)
|
||||
.json(JsonInput::builder().set_type(Some(json_type)).build())
|
||||
.build(),
|
||||
)
|
||||
.output_serialization(OutputSerialization::builder().json(JsonOutput::builder().build()).build())
|
||||
}
|
||||
|
||||
async fn collect_success(
|
||||
mut response: SelectObjectContentOutput,
|
||||
compressed_bytes: usize,
|
||||
processed_bytes: usize,
|
||||
) -> TestResult<Vec<u8>> {
|
||||
tokio::time::timeout(SELECT_RESPONSE_TIMEOUT, async move {
|
||||
let mut records = Vec::new();
|
||||
let mut stats = None;
|
||||
let mut saw_end = false;
|
||||
|
||||
while let Some(event) = response.payload.recv().await? {
|
||||
assert!(!saw_end, "Select emitted an event after End");
|
||||
match event {
|
||||
SelectObjectContentEventStream::Records(event) => {
|
||||
assert!(stats.is_none(), "Select emitted Records after Stats");
|
||||
if let Some(payload) = event.payload {
|
||||
records.extend_from_slice(payload.as_ref());
|
||||
}
|
||||
}
|
||||
SelectObjectContentEventStream::Stats(event) => {
|
||||
assert!(stats.is_none(), "Select emitted more than one Stats event");
|
||||
stats = event.details;
|
||||
}
|
||||
SelectObjectContentEventStream::End(_) => {
|
||||
assert!(stats.is_some(), "Select emitted End before Stats");
|
||||
saw_end = true;
|
||||
}
|
||||
_ => assert!(stats.is_none(), "Select emitted a non-terminal event after Stats"),
|
||||
}
|
||||
}
|
||||
|
||||
let stats = stats.ok_or("Select response ended without a Stats event")?;
|
||||
assert_eq!(stats.bytes_scanned(), Some(i64::try_from(compressed_bytes)?));
|
||||
assert_eq!(stats.bytes_processed(), Some(i64::try_from(processed_bytes)?));
|
||||
assert_eq!(stats.bytes_returned(), Some(i64::try_from(records.len())?));
|
||||
assert!(saw_end, "Select response ended without an End event");
|
||||
Ok::<_, Box<dyn Error + Send + Sync>>(records)
|
||||
})
|
||||
.await
|
||||
.map_err(|_| -> Box<dyn Error + Send + Sync> { "Select response timed out".into() })?
|
||||
}
|
||||
|
||||
async fn assert_truncated_stream_failure(mut response: SelectObjectContentOutput) -> TestResult<()> {
|
||||
tokio::time::timeout(SELECT_RESPONSE_TIMEOUT, async move {
|
||||
loop {
|
||||
match response.payload.recv().await {
|
||||
Err(error) => {
|
||||
// S3 Select request-level errors use `error` frames, which this SDK version exposes as raw response errors.
|
||||
if let Some(code) = error.code() {
|
||||
assert_eq!(code, "TruncatedInput", "unexpected modeled event-stream error: {error:?}");
|
||||
} else if let aws_sdk_s3::error::SdkError::ResponseError(context) = &error
|
||||
&& let RawMessage::Decoded(message) = context.raw()
|
||||
{
|
||||
let header = |name: &str| {
|
||||
message
|
||||
.headers()
|
||||
.iter()
|
||||
.find(|header| header.name().as_str() == name)
|
||||
.and_then(|header| header.value().as_string().ok())
|
||||
.map(|value| value.as_str())
|
||||
};
|
||||
assert_eq!(header(":message-type"), Some("error"));
|
||||
assert_eq!(header(":error-code"), Some("TruncatedInput"));
|
||||
} else {
|
||||
panic!("unexpected event-stream error: {error:?}");
|
||||
}
|
||||
return Ok(());
|
||||
}
|
||||
Ok(Some(SelectObjectContentEventStream::Stats(_))) | Ok(Some(SelectObjectContentEventStream::End(_))) => {
|
||||
return Err("truncated compressed input reached a success terminal event".into());
|
||||
}
|
||||
Ok(Some(_)) => {}
|
||||
Ok(None) => return Err("truncated compressed input ended without an error event".into()),
|
||||
}
|
||||
}
|
||||
})
|
||||
.await
|
||||
.map_err(|_| -> Box<dyn Error + Send + Sync> { "truncated Select response timed out".into() })?
|
||||
}
|
||||
|
||||
#[tokio::test(flavor = "multi_thread", worker_threads = 4)]
|
||||
async fn test_select_object_content_compressed_csv_and_json() -> TestResult<()> {
|
||||
const CSV: &[u8] = b"name,age\nAlice,30\nBob,25\n";
|
||||
const JSON_LINES: &[u8] = b"{\"name\":\"Alice\"}\n{\"name\":\"Bob\"}\n";
|
||||
const JSON_DOCUMENT: &[u8] = br#"[{"name":"Alice"},{"name":"Bob"}]"#;
|
||||
|
||||
let (_env, client) = create_test_environment(&[]).await?;
|
||||
|
||||
let gzip_csv = gzip(CSV)?;
|
||||
put_object(&client, "records.csv.gz", &gzip_csv).await?;
|
||||
let gzip_csv_records = collect_success(
|
||||
csv_select_request(&client, "records.csv.gz", CompressionType::Gzip, "SELECT * FROM S3Object")
|
||||
.send()
|
||||
.await?,
|
||||
gzip_csv.len(),
|
||||
CSV.len(),
|
||||
)
|
||||
.await?;
|
||||
assert_eq!(gzip_csv_records, b"Alice,30\nBob,25\n");
|
||||
|
||||
let bzip_csv = bzip2(CSV).await?;
|
||||
put_object(&client, "records.csv.bz2", &bzip_csv).await?;
|
||||
let bzip_csv_records = collect_success(
|
||||
csv_select_request(&client, "records.csv.bz2", CompressionType::Bzip2, "SELECT * FROM S3Object")
|
||||
.send()
|
||||
.await?,
|
||||
bzip_csv.len(),
|
||||
CSV.len(),
|
||||
)
|
||||
.await?;
|
||||
assert_eq!(bzip_csv_records, gzip_csv_records);
|
||||
|
||||
let gzip_json_lines = gzip(JSON_LINES)?;
|
||||
put_object(&client, "json-lines", &gzip_json_lines).await?;
|
||||
let gzip_json_records = collect_success(
|
||||
json_select_request(&client, "json-lines", CompressionType::Gzip, JsonType::Lines)
|
||||
.send()
|
||||
.await?,
|
||||
gzip_json_lines.len(),
|
||||
JSON_LINES.len(),
|
||||
)
|
||||
.await?;
|
||||
assert_eq!(gzip_json_records, JSON_LINES);
|
||||
|
||||
let bzip_json_lines = bzip2(JSON_LINES).await?;
|
||||
put_object(&client, "records.jsonl.bz2", &bzip_json_lines).await?;
|
||||
let bzip_json_records = collect_success(
|
||||
json_select_request(&client, "records.jsonl.bz2", CompressionType::Bzip2, JsonType::Lines)
|
||||
.send()
|
||||
.await?,
|
||||
bzip_json_lines.len(),
|
||||
JSON_LINES.len(),
|
||||
)
|
||||
.await?;
|
||||
assert_eq!(bzip_json_records, gzip_json_records);
|
||||
|
||||
let gzip_json_document = gzip(JSON_DOCUMENT)?;
|
||||
put_object(&client, "document.json.gz", &gzip_json_document).await?;
|
||||
let document_records = collect_success(
|
||||
json_select_request(&client, "document.json.gz", CompressionType::Gzip, JsonType::Document)
|
||||
.send()
|
||||
.await?,
|
||||
gzip_json_document.len(),
|
||||
JSON_DOCUMENT.len(),
|
||||
)
|
||||
.await?;
|
||||
assert_eq!(document_records, JSON_LINES);
|
||||
|
||||
Ok(())
|
||||
}
|
||||
|
||||
#[tokio::test(flavor = "multi_thread", worker_threads = 4)]
|
||||
async fn test_select_object_content_invalid_compressed_stream_fails() -> TestResult<()> {
|
||||
const CSV: &[u8] = b"name\nAlice\n";
|
||||
|
||||
let (_env, client) = create_test_environment(&[]).await?;
|
||||
|
||||
put_object(&client, "invalid.csv.gz", CSV).await?;
|
||||
let invalid = csv_select_request(&client, "invalid.csv.gz", CompressionType::Gzip, "SELECT * FROM S3Object")
|
||||
.send()
|
||||
.await
|
||||
.expect_err("invalid GZIP header must fail before streaming");
|
||||
assert_eq!(
|
||||
invalid.as_service_error().and_then(ProvideErrorMetadata::code),
|
||||
Some("InvalidCompressionFormat")
|
||||
);
|
||||
|
||||
put_object(&client, "empty.csv.gz", b"").await?;
|
||||
let empty = csv_select_request(&client, "empty.csv.gz", CompressionType::Gzip, "SELECT * FROM S3Object")
|
||||
.send()
|
||||
.await
|
||||
.expect_err("empty GZIP input must fail as truncated");
|
||||
assert_eq!(empty.as_service_error().and_then(ProvideErrorMetadata::code), Some("TruncatedInput"));
|
||||
|
||||
let mut truncated = bzip2(CSV).await?;
|
||||
truncated.pop();
|
||||
put_object(&client, "truncated.csv.bz2", &truncated).await?;
|
||||
let truncated = csv_select_request(&client, "truncated.csv.bz2", CompressionType::Bzip2, "SELECT * FROM S3Object")
|
||||
.send()
|
||||
.await?;
|
||||
assert_truncated_stream_failure(truncated).await?;
|
||||
|
||||
Ok(())
|
||||
}
|
||||
|
||||
#[tokio::test(flavor = "multi_thread", worker_threads = 4)]
|
||||
async fn test_select_object_content_compressed_disconnect_releases_query() -> TestResult<()> {
|
||||
const OBJECT: &str = "disconnect.csv.gz";
|
||||
const ROWS: usize = 16 * 1024;
|
||||
const RELEASE_ATTEMPTS: usize = 20;
|
||||
const RELEASE_BACKOFF: Duration = Duration::from_millis(25);
|
||||
|
||||
let (_env, client) = create_test_environment(&[("RUSTFS_S3SELECT_MAX_CONCURRENT_QUERIES", "1")]).await?;
|
||||
let row = format!("{}\n", "x".repeat(1023));
|
||||
let mut body = Vec::with_capacity("value\n".len() + ROWS * row.len());
|
||||
body.extend_from_slice(b"value\n");
|
||||
for _ in 0..ROWS {
|
||||
body.extend_from_slice(row.as_bytes());
|
||||
}
|
||||
let compressed = gzip(&body)?;
|
||||
put_object(&client, OBJECT, &compressed).await?;
|
||||
|
||||
let first = csv_select_request(&client, OBJECT, CompressionType::Gzip, "SELECT * FROM S3Object")
|
||||
.send()
|
||||
.await?;
|
||||
let saturated = csv_select_request(&client, OBJECT, CompressionType::Gzip, "SELECT * FROM S3Object")
|
||||
.send()
|
||||
.await
|
||||
.expect_err("the unread compressed response should retain the only query permit");
|
||||
assert_eq!(saturated.as_service_error().and_then(ProvideErrorMetadata::code), Some("SlowDown"));
|
||||
|
||||
drop(first);
|
||||
let second = tokio::time::timeout(Duration::from_secs(5), async {
|
||||
for attempt in 0..RELEASE_ATTEMPTS {
|
||||
match csv_select_request(&client, OBJECT, CompressionType::Gzip, "SELECT * FROM S3Object")
|
||||
.send()
|
||||
.await
|
||||
{
|
||||
Ok(response) => return Ok::<_, Box<dyn Error + Send + Sync>>(response),
|
||||
Err(error)
|
||||
if error.as_service_error().and_then(ProvideErrorMetadata::code) == Some("SlowDown")
|
||||
&& attempt + 1 < RELEASE_ATTEMPTS =>
|
||||
{
|
||||
tokio::time::sleep(RELEASE_BACKOFF).await;
|
||||
}
|
||||
Err(error) if error.as_service_error().and_then(ProvideErrorMetadata::code) == Some("SlowDown") => {
|
||||
return Err("disconnected compressed Select retained its query permit".into());
|
||||
}
|
||||
Err(error) => return Err(format!("unexpected Select error after disconnect: {error}").into()),
|
||||
}
|
||||
}
|
||||
Err("query permit release retry loop ended unexpectedly".into())
|
||||
})
|
||||
.await
|
||||
.map_err(|_| -> Box<dyn Error + Send + Sync> { "compressed Select did not release its query permit".into() })??;
|
||||
drop(second);
|
||||
|
||||
Ok(())
|
||||
}
|
||||
@@ -17,7 +17,8 @@ use crate::common::{RustFSTestEnvironment, init_logging};
|
||||
use aws_sdk_s3::Client;
|
||||
use aws_sdk_s3::error::ProvideErrorMetadata;
|
||||
use aws_sdk_s3::types::{
|
||||
CsvInput, CsvOutput, ExpressionType, FileHeaderInfo, InputSerialization, JsonInput, JsonOutput, JsonType, OutputSerialization,
|
||||
CsvInput, CsvOutput, ExpressionType, FileHeaderInfo, InputSerialization, JsonInput, JsonOutput, JsonType,
|
||||
OutputSerialization, RequestProgress,
|
||||
};
|
||||
use bytes::Bytes;
|
||||
use std::error::Error;
|
||||
@@ -26,6 +27,9 @@ use std::time::Duration;
|
||||
const BUCKET: &str = "test-sql-bucket";
|
||||
const CSV_OBJECT: &str = "test-data.csv";
|
||||
const JSON_OBJECT: &str = "test-data.json";
|
||||
const JSON_DOCUMENT_OBJECT: &str = "nested-data.json";
|
||||
const JSON_ROOT_ARRAY_OBJECT: &str = "root-array.json";
|
||||
const JSON_ROOT_SCALAR_ARRAY_OBJECT: &str = "root-scalars.json";
|
||||
const SELECT_RESPONSE_TIMEOUT: Duration = Duration::from_secs(30);
|
||||
|
||||
type TestResult<T> = Result<T, Box<dyn Error + Send + Sync>>;
|
||||
@@ -73,6 +77,69 @@ async fn upload_test_json(client: &Client) -> TestResult<()> {
|
||||
Ok(())
|
||||
}
|
||||
|
||||
async fn upload_nested_json_document(client: &Client) -> TestResult<()> {
|
||||
let json_data = r#"{"departments":[{"employees":[{"name":"Alice","active":true},{"name":"Bob","active":false}]},{"employees":[{"name":"Charlie","active":true}]}]}"#;
|
||||
|
||||
client
|
||||
.put_object()
|
||||
.bucket(BUCKET)
|
||||
.key(JSON_DOCUMENT_OBJECT)
|
||||
.body(Bytes::from_static(json_data.as_bytes()).into())
|
||||
.send()
|
||||
.await?;
|
||||
client
|
||||
.put_object()
|
||||
.bucket(BUCKET)
|
||||
.key(JSON_ROOT_ARRAY_OBJECT)
|
||||
.body(Bytes::from_static(br#"[{"name":"Alice"},{"name":"Bob"}]"#).into())
|
||||
.send()
|
||||
.await?;
|
||||
client
|
||||
.put_object()
|
||||
.bucket(BUCKET)
|
||||
.key(JSON_ROOT_SCALAR_ARRAY_OBJECT)
|
||||
.body(Bytes::from_static(b"[1,2]").into())
|
||||
.send()
|
||||
.await?;
|
||||
Ok(())
|
||||
}
|
||||
|
||||
async fn select_json_document(client: &Client, key: &str, expression: &str) -> TestResult<String> {
|
||||
let response = client
|
||||
.select_object_content()
|
||||
.bucket(BUCKET)
|
||||
.key(key)
|
||||
.expression(expression)
|
||||
.expression_type(ExpressionType::Sql)
|
||||
.input_serialization(
|
||||
InputSerialization::builder()
|
||||
.json(JsonInput::builder().set_type(Some(JsonType::Document)).build())
|
||||
.build(),
|
||||
)
|
||||
.output_serialization(OutputSerialization::builder().json(JsonOutput::builder().build()).build())
|
||||
.send()
|
||||
.await?;
|
||||
process_select_response(response).await
|
||||
}
|
||||
|
||||
fn csv_select_request(
|
||||
client: &Client,
|
||||
key: &str,
|
||||
) -> aws_sdk_s3::operation::select_object_content::builders::SelectObjectContentFluentBuilder {
|
||||
client
|
||||
.select_object_content()
|
||||
.bucket(BUCKET)
|
||||
.key(key)
|
||||
.expression("SELECT * FROM S3Object")
|
||||
.expression_type(ExpressionType::Sql)
|
||||
.input_serialization(
|
||||
InputSerialization::builder()
|
||||
.csv(CsvInput::builder().file_header_info(FileHeaderInfo::Use).build())
|
||||
.build(),
|
||||
)
|
||||
.output_serialization(OutputSerialization::builder().csv(CsvOutput::builder().build()).build())
|
||||
}
|
||||
|
||||
async fn process_select_response(
|
||||
mut event_stream: aws_sdk_s3::operation::select_object_content::SelectObjectContentOutput,
|
||||
) -> TestResult<String> {
|
||||
@@ -104,6 +171,209 @@ async fn process_select_response(
|
||||
.map_err(|_| -> Box<dyn Error + Send + Sync> { "Select response timed out".into() })?
|
||||
}
|
||||
|
||||
async fn assert_input_byte_stats(
|
||||
client: &Client,
|
||||
object: &str,
|
||||
body: &[u8],
|
||||
expression: &str,
|
||||
input_serialization: InputSerialization,
|
||||
output_serialization: OutputSerialization,
|
||||
progress_enabled: bool,
|
||||
) -> TestResult<()> {
|
||||
client
|
||||
.put_object()
|
||||
.bucket(BUCKET)
|
||||
.key(object)
|
||||
.body(Bytes::copy_from_slice(body).into())
|
||||
.send()
|
||||
.await?;
|
||||
|
||||
let mut request = client
|
||||
.select_object_content()
|
||||
.bucket(BUCKET)
|
||||
.key(object)
|
||||
.expression(expression)
|
||||
.expression_type(ExpressionType::Sql)
|
||||
.input_serialization(input_serialization)
|
||||
.output_serialization(output_serialization);
|
||||
if progress_enabled {
|
||||
request = request.request_progress(RequestProgress::builder().enabled(true).build());
|
||||
}
|
||||
let response = request.send().await?;
|
||||
|
||||
let mut payload = response.payload;
|
||||
let mut records_len = 0_u64;
|
||||
let mut last_progress: Option<aws_sdk_s3::types::Progress> = None;
|
||||
let mut stats = None;
|
||||
let mut saw_end = false;
|
||||
tokio::time::timeout(SELECT_RESPONSE_TIMEOUT, async {
|
||||
// The AWS SDK validates both event-stream CRCs before yielding an event.
|
||||
while let Some(event) = payload.recv().await? {
|
||||
assert!(!saw_end, "Select emitted an event after End");
|
||||
match event {
|
||||
aws_sdk_s3::types::SelectObjectContentEventStream::Records(records) => {
|
||||
assert!(stats.is_none(), "Select emitted Records after Stats");
|
||||
if let Some(bytes) = records.payload {
|
||||
records_len = records_len.saturating_add(u64::try_from(bytes.as_ref().len())?);
|
||||
}
|
||||
}
|
||||
aws_sdk_s3::types::SelectObjectContentEventStream::Progress(event) => {
|
||||
assert!(stats.is_none(), "Select emitted Progress after Stats");
|
||||
let details = event.details.ok_or("Progress event did not contain details")?;
|
||||
if let Some(previous) = last_progress.as_ref() {
|
||||
assert!(details.bytes_scanned() >= previous.bytes_scanned());
|
||||
assert!(details.bytes_processed() >= previous.bytes_processed());
|
||||
assert!(details.bytes_returned() >= previous.bytes_returned());
|
||||
}
|
||||
last_progress = Some(details);
|
||||
}
|
||||
aws_sdk_s3::types::SelectObjectContentEventStream::Stats(event) => {
|
||||
assert!(stats.is_none(), "Select emitted more than one Stats event");
|
||||
stats = event.details;
|
||||
}
|
||||
aws_sdk_s3::types::SelectObjectContentEventStream::End(_) => {
|
||||
assert!(stats.is_some(), "Select emitted End before Stats");
|
||||
saw_end = true;
|
||||
}
|
||||
_ => assert!(stats.is_none(), "Select emitted a non-terminal event after Stats"),
|
||||
}
|
||||
}
|
||||
Ok::<(), Box<dyn Error + Send + Sync>>(())
|
||||
})
|
||||
.await
|
||||
.map_err(|_| -> Box<dyn Error + Send + Sync> { "Select response timed out".into() })??;
|
||||
|
||||
let stats = stats.ok_or("Select response ended without a Stats event")?;
|
||||
let input_len = i64::try_from(body.len())?;
|
||||
assert_eq!(stats.bytes_scanned(), Some(input_len));
|
||||
assert_eq!(stats.bytes_processed(), Some(input_len));
|
||||
assert_eq!(stats.bytes_returned(), Some(i64::try_from(records_len)?));
|
||||
if progress_enabled {
|
||||
if let Some(progress) = last_progress {
|
||||
assert!(stats.bytes_scanned() >= progress.bytes_scanned());
|
||||
assert!(stats.bytes_processed() >= progress.bytes_processed());
|
||||
assert!(stats.bytes_returned() >= progress.bytes_returned());
|
||||
}
|
||||
} else {
|
||||
assert!(last_progress.is_none(), "disabled request progress emitted a Progress event");
|
||||
}
|
||||
assert!(saw_end, "Select response ended without an End event");
|
||||
Ok(())
|
||||
}
|
||||
|
||||
#[tokio::test(flavor = "multi_thread", worker_threads = 4)]
|
||||
async fn test_select_object_content_http_event_order_crc_and_input_byte_stats() -> TestResult<()> {
|
||||
const CSV_BODY: &[u8] = b"name,age\nAlice,30\nBob,25\n";
|
||||
const JSON_LINES_BODY: &[u8] = b"{\"name\":\"Alice\"}\n{\"name\":\"Bob\"}\n";
|
||||
const JSON_DOCUMENT_BODY: &[u8] = b"[{\"name\":\"Alice\"},{\"name\":\"Bob\"}]";
|
||||
|
||||
let (_env, client) = create_test_environment().await?;
|
||||
setup_test_bucket(&client).await?;
|
||||
assert_input_byte_stats(
|
||||
&client,
|
||||
"input-metrics.csv",
|
||||
CSV_BODY,
|
||||
"SELECT name FROM S3Object",
|
||||
InputSerialization::builder()
|
||||
.csv(CsvInput::builder().file_header_info(FileHeaderInfo::Use).build())
|
||||
.build(),
|
||||
OutputSerialization::builder().csv(CsvOutput::builder().build()).build(),
|
||||
true,
|
||||
)
|
||||
.await?;
|
||||
assert_input_byte_stats(
|
||||
&client,
|
||||
"input-metrics.jsonl",
|
||||
JSON_LINES_BODY,
|
||||
"SELECT name FROM S3Object",
|
||||
InputSerialization::builder()
|
||||
.json(JsonInput::builder().set_type(Some(JsonType::Lines)).build())
|
||||
.build(),
|
||||
OutputSerialization::builder().json(JsonOutput::builder().build()).build(),
|
||||
true,
|
||||
)
|
||||
.await?;
|
||||
assert_input_byte_stats(
|
||||
&client,
|
||||
"input-metrics.json",
|
||||
JSON_DOCUMENT_BODY,
|
||||
"SELECT name FROM S3Object",
|
||||
InputSerialization::builder()
|
||||
.json(JsonInput::builder().set_type(Some(JsonType::Document)).build())
|
||||
.build(),
|
||||
OutputSerialization::builder().json(JsonOutput::builder().build()).build(),
|
||||
true,
|
||||
)
|
||||
.await?;
|
||||
assert_input_byte_stats(
|
||||
&client,
|
||||
"input-metrics-without-progress.csv",
|
||||
CSV_BODY,
|
||||
"SELECT name FROM S3Object",
|
||||
InputSerialization::builder()
|
||||
.csv(CsvInput::builder().file_header_info(FileHeaderInfo::Use).build())
|
||||
.build(),
|
||||
OutputSerialization::builder().csv(CsvOutput::builder().build()).build(),
|
||||
false,
|
||||
)
|
||||
.await?;
|
||||
Ok(())
|
||||
}
|
||||
|
||||
#[tokio::test(flavor = "multi_thread", worker_threads = 4)]
|
||||
async fn test_select_object_content_http_disconnect_releases_query() -> TestResult<()> {
|
||||
const OBJECT: &str = "disconnect.csv";
|
||||
const ROWS: usize = 16 * 1024;
|
||||
const RELEASE_BACKOFF: Duration = Duration::from_millis(25);
|
||||
|
||||
init_logging();
|
||||
let mut env = RustFSTestEnvironment::new().await?;
|
||||
env.start_rustfs_server_with_env(vec![], &[("RUSTFS_S3SELECT_MAX_CONCURRENT_QUERIES", "1")])
|
||||
.await?;
|
||||
let client = env.create_s3_client();
|
||||
setup_test_bucket(&client).await?;
|
||||
|
||||
let row = format!("{}\n", "x".repeat(1023));
|
||||
let mut body = Vec::with_capacity("value\n".len() + ROWS * row.len());
|
||||
body.extend_from_slice(b"value\n");
|
||||
for _ in 0..ROWS {
|
||||
body.extend_from_slice(row.as_bytes());
|
||||
}
|
||||
client
|
||||
.put_object()
|
||||
.bucket(BUCKET)
|
||||
.key(OBJECT)
|
||||
.body(Bytes::from(body).into())
|
||||
.send()
|
||||
.await?;
|
||||
|
||||
// Leaving this response body unread fills the bounded HTTP/event channels before the query can finish.
|
||||
let first = csv_select_request(&client, OBJECT).send().await?;
|
||||
let saturated = csv_select_request(&client, OBJECT)
|
||||
.send()
|
||||
.await
|
||||
.expect_err("the first HTTP stream should retain the only query permit");
|
||||
assert_eq!(saturated.as_service_error().and_then(ProvideErrorMetadata::code), Some("SlowDown"));
|
||||
|
||||
drop(first);
|
||||
let second = tokio::time::timeout(Duration::from_secs(5), async {
|
||||
loop {
|
||||
match csv_select_request(&client, OBJECT).send().await {
|
||||
Ok(response) => return Ok::<_, Box<dyn Error + Send + Sync>>(response),
|
||||
Err(error) if error.as_service_error().and_then(ProvideErrorMetadata::code) == Some("SlowDown") => {
|
||||
tokio::time::sleep(RELEASE_BACKOFF).await;
|
||||
}
|
||||
Err(error) => return Err(format!("unexpected Select error after disconnect: {error}").into()),
|
||||
}
|
||||
}
|
||||
})
|
||||
.await
|
||||
.map_err(|_| -> Box<dyn Error + Send + Sync> { "disconnected Select did not release its query permit".into() })??;
|
||||
drop(second);
|
||||
|
||||
Ok(())
|
||||
}
|
||||
|
||||
#[tokio::test(flavor = "multi_thread", worker_threads = 4)]
|
||||
async fn test_select_object_content_csv_basic() -> TestResult<()> {
|
||||
let (_env, client) = create_test_environment().await?;
|
||||
@@ -228,6 +498,107 @@ async fn test_select_object_content_json_basic() -> TestResult<()> {
|
||||
Ok(())
|
||||
}
|
||||
|
||||
#[tokio::test(flavor = "multi_thread", worker_threads = 4)]
|
||||
async fn test_select_object_content_nested_json_source_path() -> TestResult<()> {
|
||||
let (_env, client) = create_test_environment().await?;
|
||||
setup_test_bucket(&client).await?;
|
||||
upload_nested_json_document(&client).await?;
|
||||
|
||||
let result = select_json_document(
|
||||
&client,
|
||||
JSON_DOCUMENT_OBJECT,
|
||||
"SELECT e.name FROM S3Object[*].departments[*].employees[*] AS e WHERE e.active = true",
|
||||
)
|
||||
.await?;
|
||||
let names: Vec<String> = result
|
||||
.lines()
|
||||
.filter(|line| !line.trim().is_empty())
|
||||
.map(|line| -> TestResult<String> {
|
||||
let value: serde_json::Value = serde_json::from_str(line)?;
|
||||
Ok(value["name"].as_str().ok_or("missing name field")?.to_string())
|
||||
})
|
||||
.collect::<TestResult<_>>()?;
|
||||
|
||||
assert_eq!(names, vec!["Alice", "Charlie"]);
|
||||
|
||||
let terminal_scalars = select_json_document(
|
||||
&client,
|
||||
JSON_DOCUMENT_OBJECT,
|
||||
"SELECT NAME FROM S3Object[*].DEPARTMENTS[*].employees[*].NAME",
|
||||
)
|
||||
.await?;
|
||||
let scalar_names: Vec<String> = terminal_scalars
|
||||
.lines()
|
||||
.map(|line| -> TestResult<String> {
|
||||
let value: serde_json::Value = serde_json::from_str(line)?;
|
||||
Ok(value["name"].as_str().ok_or("missing scalar name field")?.to_string())
|
||||
})
|
||||
.collect::<TestResult<_>>()?;
|
||||
assert_eq!(scalar_names, vec!["Alice", "Bob", "Charlie"]);
|
||||
|
||||
let aliased_scalars = select_json_document(
|
||||
&client,
|
||||
JSON_DOCUMENT_OBJECT,
|
||||
"SELECT v FROM S3Object[*].departments[*].employees[*].name AS v",
|
||||
)
|
||||
.await?;
|
||||
let aliased_names: Vec<String> = aliased_scalars
|
||||
.lines()
|
||||
.map(|line| -> TestResult<String> {
|
||||
let value: serde_json::Value = serde_json::from_str(line)?;
|
||||
Ok(value["v"].as_str().ok_or("missing aliased scalar field")?.to_string())
|
||||
})
|
||||
.collect::<TestResult<_>>()?;
|
||||
assert_eq!(aliased_names, vec!["Alice", "Bob", "Charlie"]);
|
||||
|
||||
let root_array = select_json_document(&client, JSON_ROOT_ARRAY_OBJECT, "SELECT c.name FROM S3Object[*][*] AS c").await?;
|
||||
let root_names: Vec<String> = root_array
|
||||
.lines()
|
||||
.map(|line| -> TestResult<String> {
|
||||
let value: serde_json::Value = serde_json::from_str(line)?;
|
||||
Ok(value["name"].as_str().ok_or("missing root-array name field")?.to_string())
|
||||
})
|
||||
.collect::<TestResult<_>>()?;
|
||||
assert_eq!(root_names, vec!["Alice", "Bob"]);
|
||||
|
||||
let root_index = select_json_document(&client, JSON_ROOT_ARRAY_OBJECT, "SELECT c.name FROM S3Object[*][0] AS c").await?;
|
||||
let root_index_value: serde_json::Value = serde_json::from_str(root_index.trim())?;
|
||||
assert_eq!(root_index_value["name"], "Alice");
|
||||
|
||||
let root_scalars = select_json_document(&client, JSON_ROOT_SCALAR_ARRAY_OBJECT, "SELECT V FROM S3Object AS V").await?;
|
||||
let scalar_values: Vec<i64> = root_scalars
|
||||
.lines()
|
||||
.map(|line| -> TestResult<i64> {
|
||||
let value: serde_json::Value = serde_json::from_str(line)?;
|
||||
Ok(value["v"].as_i64().ok_or("missing root scalar value")?)
|
||||
})
|
||||
.collect::<TestResult<_>>()?;
|
||||
assert_eq!(scalar_values, vec![1, 2]);
|
||||
|
||||
let implicit_root_scalars =
|
||||
select_json_document(&client, JSON_ROOT_SCALAR_ARRAY_OBJECT, "SELECT S3Object FROM S3Object").await?;
|
||||
let implicit_scalar_values: Vec<i64> = implicit_root_scalars
|
||||
.lines()
|
||||
.map(|line| -> TestResult<i64> {
|
||||
let value: serde_json::Value = serde_json::from_str(line)?;
|
||||
Ok(value["s3object"].as_i64().ok_or("missing implicit root scalar value")?)
|
||||
})
|
||||
.collect::<TestResult<_>>()?;
|
||||
assert_eq!(implicit_scalar_values, vec![1, 2]);
|
||||
|
||||
let quoted_root_scalars =
|
||||
select_json_document(&client, JSON_ROOT_SCALAR_ARRAY_OBJECT, "SELECT \"S3Object\" FROM \"S3Object\"").await?;
|
||||
let quoted_scalar_values: Vec<i64> = quoted_root_scalars
|
||||
.lines()
|
||||
.map(|line| -> TestResult<i64> {
|
||||
let value: serde_json::Value = serde_json::from_str(line)?;
|
||||
Ok(value["S3Object"].as_i64().ok_or("missing quoted root scalar value")?)
|
||||
})
|
||||
.collect::<TestResult<_>>()?;
|
||||
assert_eq!(quoted_scalar_values, vec![1, 2]);
|
||||
Ok(())
|
||||
}
|
||||
|
||||
#[tokio::test(flavor = "multi_thread", worker_threads = 4)]
|
||||
async fn test_select_object_content_csv_limit() -> TestResult<()> {
|
||||
let (_env, client) = create_test_environment().await?;
|
||||
|
||||
@@ -40,10 +40,10 @@ mod tests {
|
||||
|
||||
const ENABLE_ENV: &str = "RUSTFS_PRIVILEGED_REPLACEMENT_E2E";
|
||||
const NAMESPACE_ENV: &str = "RUSTFS_PRIVILEGED_REPLACEMENT_E2E_IN_NAMESPACE";
|
||||
const LOG_DIR_ENV: &str = "RUSTFS_PRIVILEGED_REPLACEMENT_LOG_DIR";
|
||||
const TARGET_NODE: usize = 1;
|
||||
const TARGET_DRIVE: usize = 0;
|
||||
const MOUNT_SIZE: &str = "size=128m,mode=0700";
|
||||
const ABSENT_SCANNER_OBSERVATION_TIMEOUT_SECS: u64 = 180;
|
||||
const REPLACEMENT_RECOVERY_DIR: &str = ".rustfs.sys/buckets/ahm-replacement";
|
||||
const REPLACEMENT_INTENT_SUFFIX: &str = "_ahm_replacement_intent.json";
|
||||
const REPLACEMENT_COMPLETION_PROOF_SUFFIX: &str = "_ahm_replacement_completion_proof.json";
|
||||
@@ -142,6 +142,23 @@ mod tests {
|
||||
run_command("dmsetup", &["resume", &self.dm_name])
|
||||
}
|
||||
|
||||
fn verify_raw_io_is_unavailable(&self) -> Result<(), Box<dyn Error + Send + Sync>> {
|
||||
let mapper = format!("/dev/mapper/{}", self.dm_name);
|
||||
let output = Command::new("dd")
|
||||
.env("LC_ALL", "C")
|
||||
.arg(format!("if={mapper}"))
|
||||
.args(["of=/dev/null", "bs=4096", "count=1", "iflag=direct", "status=none"])
|
||||
.output()?;
|
||||
if output.status.success() {
|
||||
return Err(format!("dm-error target unexpectedly allowed a raw read from {mapper}").into());
|
||||
}
|
||||
let stderr = String::from_utf8_lossy(&output.stderr);
|
||||
if !stderr.contains("Input/output error") {
|
||||
return Err(format!("raw read from dm-error target failed unexpectedly: {stderr}").into());
|
||||
}
|
||||
Ok(())
|
||||
}
|
||||
|
||||
fn restore_available(&self) -> Result<(), Box<dyn Error + Send + Sync>> {
|
||||
let sectors = run_command_stdout("blockdev", &["--getsz", &self.loop_device])?;
|
||||
let linear_table = format!("0 {sectors} linear {} 0", self.loop_device);
|
||||
@@ -194,6 +211,72 @@ mod tests {
|
||||
}
|
||||
}
|
||||
|
||||
struct ZramBlockMount {
|
||||
target: PathBuf,
|
||||
device: String,
|
||||
mounted: bool,
|
||||
}
|
||||
|
||||
impl ZramBlockMount {
|
||||
fn reserve(target: &Path) -> Result<Self, Box<dyn Error + Send + Sync>> {
|
||||
if !Path::new("/dev/zram-control").exists() {
|
||||
run_command("modprobe", &["zram"])?;
|
||||
}
|
||||
let device = run_command_stdout("zramctl", &["--find", "--size", "256M"])?;
|
||||
if device.is_empty() {
|
||||
return Err("zramctl --find --size returned an empty device".into());
|
||||
}
|
||||
|
||||
Ok(Self {
|
||||
target: target.to_path_buf(),
|
||||
device,
|
||||
mounted: false,
|
||||
})
|
||||
}
|
||||
|
||||
fn mount_target(&mut self) -> Result<(), Box<dyn Error + Send + Sync>> {
|
||||
let result = (|| {
|
||||
run_command("mkfs.ext4", &["-F", &self.device])?;
|
||||
let target_arg = path_to_string(&self.target, "zram replacement mount target")?;
|
||||
run_command("mount", &[&self.device, &target_arg])
|
||||
})();
|
||||
if let Err(error) = result {
|
||||
let _ = self.cleanup();
|
||||
return Err(error);
|
||||
}
|
||||
self.mounted = true;
|
||||
Ok(())
|
||||
}
|
||||
|
||||
fn cleanup(&mut self) -> Result<(), Box<dyn Error + Send + Sync>> {
|
||||
let mut first_error: Option<Box<dyn Error + Send + Sync>> = None;
|
||||
if self.mounted {
|
||||
if let Err(error) = detach_mount(&self.target) {
|
||||
first_error.get_or_insert(error);
|
||||
} else {
|
||||
self.mounted = false;
|
||||
}
|
||||
}
|
||||
if !self.device.is_empty() {
|
||||
if let Err(error) = run_command("zramctl", &["--reset", &self.device]) {
|
||||
first_error.get_or_insert(error);
|
||||
} else {
|
||||
self.device.clear();
|
||||
}
|
||||
}
|
||||
if let Some(error) = first_error {
|
||||
return Err(error);
|
||||
}
|
||||
Ok(())
|
||||
}
|
||||
}
|
||||
|
||||
impl Drop for ZramBlockMount {
|
||||
fn drop(&mut self) {
|
||||
let _ = self.cleanup();
|
||||
}
|
||||
}
|
||||
|
||||
fn checked_command_output(program: &str, args: &[&str]) -> Result<std::process::Output, Box<dyn Error + Send + Sync>> {
|
||||
let output = Command::new(program).args(args).output()?;
|
||||
if output.status.success() {
|
||||
@@ -298,6 +381,18 @@ mod tests {
|
||||
Err(format!("{ENABLE_ENV}=1 requires root or CAP_SYS_ADMIN; unshare exited with status {status}").into())
|
||||
}
|
||||
|
||||
fn replacement_node_log_path(
|
||||
cluster_temp_dir: &str,
|
||||
parity: usize,
|
||||
node_index: usize,
|
||||
) -> Result<PathBuf, Box<dyn Error + Send + Sync>> {
|
||||
let log_dir = std::env::var_os(LOG_DIR_ENV)
|
||||
.map(PathBuf::from)
|
||||
.unwrap_or_else(|| PathBuf::from(cluster_temp_dir));
|
||||
fs::create_dir_all(&log_dir)?;
|
||||
Ok(log_dir.join(format!("replacement-ec{parity}-node{node_index}-{}.log", std::process::id())))
|
||||
}
|
||||
|
||||
fn payload(len: usize, seed: u8) -> Vec<u8> {
|
||||
let mut next = seed;
|
||||
(0..len)
|
||||
@@ -472,8 +567,20 @@ mod tests {
|
||||
if let Some(version_id) = &version.version_id {
|
||||
request = request.version_id(version_id);
|
||||
}
|
||||
let response = request.send().await?;
|
||||
let body = response.body.collect().await?.into_bytes();
|
||||
let response = request.send().await.map_err(|error| {
|
||||
format!("body GET failed for {}/{}@{:?}: {error}", version.bucket, version.key, version.version_id)
|
||||
})?;
|
||||
let body = response
|
||||
.body
|
||||
.collect()
|
||||
.await
|
||||
.map_err(|error| {
|
||||
format!(
|
||||
"body stream failed for {}/{}@{:?}: {error}",
|
||||
version.bucket, version.key, version.version_id
|
||||
)
|
||||
})?
|
||||
.into_bytes();
|
||||
assert_eq!(
|
||||
sha256_hex(&body),
|
||||
*expected_sha256,
|
||||
@@ -582,81 +689,6 @@ mod tests {
|
||||
Ok(())
|
||||
}
|
||||
|
||||
fn log_tail(log: &str) -> String {
|
||||
let mut lines = log.lines().rev().take(80).collect::<Vec<_>>();
|
||||
lines.reverse();
|
||||
lines.join("\n")
|
||||
}
|
||||
|
||||
fn log_len(path: &Path) -> Result<u64, Box<dyn Error + Send + Sync>> {
|
||||
match fs::metadata(path) {
|
||||
Ok(metadata) => Ok(metadata.len()),
|
||||
Err(error) if error.kind() == std::io::ErrorKind::NotFound => Ok(0),
|
||||
Err(error) => Err(format!("failed to stat target node log {path:?}: {error}").into()),
|
||||
}
|
||||
}
|
||||
|
||||
fn log_from_offset(path: &Path, offset: u64) -> Result<String, Box<dyn Error + Send + Sync>> {
|
||||
let log = match fs::read(path) {
|
||||
Ok(log) => log,
|
||||
Err(error) if error.kind() == std::io::ErrorKind::NotFound => Vec::new(),
|
||||
Err(error) => return Err(format!("failed to read target node log {path:?}: {error}").into()),
|
||||
};
|
||||
let start = usize::try_from(offset).unwrap_or(usize::MAX).min(log.len());
|
||||
Ok(String::from_utf8_lossy(&log[start..]).into_owned())
|
||||
}
|
||||
|
||||
fn live_disk_loss_scan_completed(log: &str, target_disk: &Path) -> bool {
|
||||
let target = target_disk.to_string_lossy();
|
||||
let mut saw_live_loss = false;
|
||||
for line in log.lines() {
|
||||
if line.contains("Heal auto-scan disk inspection failed")
|
||||
&& line.contains("check_failed")
|
||||
&& line.contains(target.as_ref())
|
||||
{
|
||||
saw_live_loss = true;
|
||||
continue;
|
||||
}
|
||||
if saw_live_loss && (line.contains("Heal auto disk scanner idle") || line.contains("Heal auto-scan cycle completed"))
|
||||
{
|
||||
return true;
|
||||
}
|
||||
}
|
||||
false
|
||||
}
|
||||
|
||||
fn live_disk_loss_scan_completed_from_path(
|
||||
log_path: &Path,
|
||||
start_offset: u64,
|
||||
target_disk: &Path,
|
||||
) -> Result<bool, Box<dyn Error + Send + Sync>> {
|
||||
Ok(live_disk_loss_scan_completed(&log_from_offset(log_path, start_offset)?, target_disk))
|
||||
}
|
||||
|
||||
async fn wait_for_live_disk_loss_observation(
|
||||
log_path: &Path,
|
||||
target_disk: &Path,
|
||||
start_offset: u64,
|
||||
timeout_secs: u64,
|
||||
) -> Result<(), Box<dyn Error + Send + Sync>> {
|
||||
let deadline = Instant::now() + Duration::from_secs(timeout_secs);
|
||||
let mut tick = interval(Duration::from_secs(1));
|
||||
loop {
|
||||
if live_disk_loss_scan_completed_from_path(log_path, start_offset, target_disk)? {
|
||||
return Ok(());
|
||||
}
|
||||
if Instant::now() >= deadline {
|
||||
let log = log_from_offset(log_path, start_offset)?;
|
||||
return Err(format!(
|
||||
"scanner did not finish a live target-loss scan for {target_disk:?} within {timeout_secs}s; log tail:\n{}",
|
||||
log_tail(&log)
|
||||
)
|
||||
.into());
|
||||
}
|
||||
tick.tick().await;
|
||||
}
|
||||
}
|
||||
|
||||
fn cluster_status_is_definitive(status: &serde_json::Value) -> Result<bool, Box<dyn Error + Send + Sync>> {
|
||||
status["cluster"]["definitive"]
|
||||
.as_bool()
|
||||
@@ -707,6 +739,13 @@ mod tests {
|
||||
.collect()
|
||||
}
|
||||
|
||||
fn is_transient_recovery_version_absence(error: &(dyn Error + 'static)) -> bool {
|
||||
matches!(
|
||||
error.downcast_ref::<rustfs_filemeta::Error>(),
|
||||
Some(rustfs_filemeta::Error::FileVersionNotFound)
|
||||
)
|
||||
}
|
||||
|
||||
fn incomplete_versions(
|
||||
target_disk: &Path,
|
||||
versions: &[BaselineVersion],
|
||||
@@ -714,7 +753,21 @@ mod tests {
|
||||
let mut missing = BTreeSet::new();
|
||||
for version in versions {
|
||||
let actual =
|
||||
census_object_version_on_disk(target_disk, &version.bucket, &version.key, version.version_id.as_deref())?;
|
||||
match census_object_version_on_disk(target_disk, &version.bucket, &version.key, version.version_id.as_deref()) {
|
||||
Ok(actual) => actual,
|
||||
// During replacement recovery, xl.meta may arrive before this
|
||||
// particular historical version. The generic census helper
|
||||
// correctly reports that as an error; this progress poll must
|
||||
// instead wait for the version to be restored.
|
||||
Err(error) if is_transient_recovery_version_absence(error.as_ref()) => {
|
||||
missing.insert(format!(
|
||||
"{}/{}@{:?}: version metadata not yet present on replacement",
|
||||
version.bucket, version.key, version.version_id
|
||||
));
|
||||
continue;
|
||||
}
|
||||
Err(error) => return Err(error),
|
||||
};
|
||||
if !actual.matches_manifest(&version.expected) {
|
||||
missing.insert(format!("{}/{}@{:?}: {actual:?}", version.bucket, version.key, version.version_id));
|
||||
}
|
||||
@@ -822,13 +875,15 @@ mod tests {
|
||||
|
||||
let mut mount_ns = MountNamespaceGuard::new()?;
|
||||
let mut cluster = RustFSTestClusterEnvironment::with_topology(ClusterTopology::single_pool_multidrive(3, 4)).await?;
|
||||
let target_log_path = PathBuf::from(&cluster.temp_dir).join(format!("replacement-node{TARGET_NODE}.log"));
|
||||
cluster.set_node_capture_log_path(TARGET_NODE, target_log_path.to_string_lossy())?;
|
||||
for node_index in 0..cluster.nodes.len() {
|
||||
let node_log_path = replacement_node_log_path(&cluster.temp_dir, parity, node_index)?;
|
||||
cluster.set_node_capture_log_path(node_index, node_log_path.to_string_lossy())?;
|
||||
}
|
||||
let target_disk = PathBuf::from(&cluster.nodes[TARGET_NODE].data_dirs[TARGET_DRIVE]);
|
||||
// Each drive below is an independent tmpfs mount, so this privileged
|
||||
// path must exercise the production distinct-device/readiness fences.
|
||||
// The blank target uses a temporary zram block device, so the
|
||||
// replacement readiness fence sees no root or sibling alias.
|
||||
cluster.extra_env.retain(|(key, _)| key != "RUSTFS_UNSAFE_BYPASS_DISK_CHECK");
|
||||
let image_root = PathBuf::from(&cluster.temp_dir).join("replacement-faultable-images");
|
||||
let image_root = PathBuf::from(&cluster.temp_dir).join("replacement-block-images");
|
||||
let mut target_mount = None;
|
||||
for (node_index, node) in cluster.nodes.iter().enumerate() {
|
||||
for (drive_index, drive) in node.data_dirs.iter().enumerate() {
|
||||
@@ -845,6 +900,7 @@ mod tests {
|
||||
}
|
||||
}
|
||||
let mut target_mount = target_mount.ok_or("target drive was not mounted with the faultable block fixture")?;
|
||||
let mut replacement_mount = ZramBlockMount::reserve(&target_disk)?;
|
||||
|
||||
cluster.set_env("RUSTFS_HEAL_ENABLED", "true");
|
||||
cluster.set_env("RUSTFS_SCANNER_ENABLED", "true");
|
||||
@@ -852,28 +908,34 @@ mod tests {
|
||||
cluster.set_env("RUSTFS_SCANNER_CYCLE", "1");
|
||||
cluster.set_env("RUSTFS_SCANNER_START_DELAY_SECS", "0");
|
||||
cluster.set_env("RUSTFS_STORAGE_CLASS_STANDARD", format!("EC:{parity}"));
|
||||
cluster.set_node_env(TARGET_NODE, "RUST_LOG", "rustfs=info,rustfs::heal::manager=debug,rustfs_notify=debug")?;
|
||||
for node_index in 0..cluster.nodes.len() {
|
||||
cluster.set_node_env(node_index, "RUST_LOG", "rustfs=info,rustfs::heal::manager=debug,rustfs_notify=debug")?;
|
||||
}
|
||||
cluster.start().await?;
|
||||
|
||||
let clients = cluster.create_all_clients()?;
|
||||
let versions = seed_baseline(&clients[0], &target_disk).await?;
|
||||
verify_bodies(&clients[0], &versions).await?;
|
||||
let versions = seed_baseline(&clients[0], &target_disk)
|
||||
.await
|
||||
.map_err(|error| format!("pre-fault baseline seeding failed: {error}"))?;
|
||||
verify_bodies(&clients[0], &versions)
|
||||
.await
|
||||
.map_err(|error| format!("pre-fault body verification failed: {error}"))?;
|
||||
|
||||
let live_loss_log_offset = log_len(&target_log_path)?;
|
||||
target_mount.make_unavailable()?;
|
||||
wait_for_live_disk_loss_observation(
|
||||
&target_log_path,
|
||||
&target_disk,
|
||||
live_loss_log_offset,
|
||||
ABSENT_SCANNER_OBSERVATION_TIMEOUT_SECS,
|
||||
)
|
||||
.await?;
|
||||
assert_no_replacement_status_records(&cluster, &target_disk).await?;
|
||||
assert_no_replacement_admission_artifacts(&cluster, &target_disk)?;
|
||||
target_mount
|
||||
.make_unavailable()
|
||||
.map_err(|error| format!("failed to install the dm-error target: {error}"))?;
|
||||
target_mount
|
||||
.verify_raw_io_is_unavailable()
|
||||
.map_err(|error| format!("dm-error target was not proven by a direct raw read: {error}"))?;
|
||||
assert_no_replacement_status_records(&cluster, &target_disk)
|
||||
.await
|
||||
.map_err(|error| format!("live-fault replacement status check failed: {error}"))?;
|
||||
assert_no_replacement_admission_artifacts(&cluster, &target_disk)
|
||||
.map_err(|error| format!("live-fault replacement artifact check failed: {error}"))?;
|
||||
|
||||
cluster.stop_node(TARGET_NODE)?;
|
||||
cluster.stop_node_gracefully(TARGET_NODE).await?;
|
||||
target_mount.cleanup()?;
|
||||
mount_ns.mount_tmpfs(&target_disk, &format!("rustfs-e2e-p{parity}-replacement"))?;
|
||||
replacement_mount.mount_target()?;
|
||||
let missing_before_restart = incomplete_versions(&target_disk, &versions)?;
|
||||
assert_eq!(
|
||||
missing_before_restart.len(),
|
||||
@@ -882,46 +944,26 @@ mod tests {
|
||||
);
|
||||
cluster.start_node(TARGET_NODE).await?;
|
||||
|
||||
wait_for_completed_replacement_with_census(&cluster, &target_disk, &versions, 420).await?;
|
||||
verify_bodies(&clients[0], &versions).await?;
|
||||
let recovery_result = async {
|
||||
wait_for_completed_replacement_with_census(&cluster, &target_disk, &versions, 420).await?;
|
||||
verify_bodies(&clients[0], &versions).await
|
||||
}
|
||||
.await;
|
||||
let stop_result = cluster.stop_node_gracefully(TARGET_NODE).await;
|
||||
let replacement_cleanup_result = replacement_mount.cleanup();
|
||||
|
||||
Ok(())
|
||||
}
|
||||
if let Err(error) = recovery_result {
|
||||
if let Err(stop_error) = stop_result {
|
||||
info!(%stop_error, "replacement target stop failed while preserving recovery failure");
|
||||
}
|
||||
if let Err(cleanup_error) = replacement_cleanup_result {
|
||||
info!(%cleanup_error, "replacement zram cleanup failed while preserving recovery failure");
|
||||
}
|
||||
return Err(error);
|
||||
}
|
||||
stop_result?;
|
||||
replacement_cleanup_result?;
|
||||
|
||||
#[test]
|
||||
fn live_loss_barrier_requires_scanner_failure_after_log_offset() -> Result<(), Box<dyn Error + Send + Sync>> {
|
||||
let target = Path::new("/mnt/target");
|
||||
assert!(live_disk_loss_scan_completed(
|
||||
"Heal auto-scan disk inspection failed endpoint=/mnt/target disk_state=check_failed\nHeal auto-scan cycle completed",
|
||||
target
|
||||
));
|
||||
assert!(live_disk_loss_scan_completed(
|
||||
"Heal auto-scan disk inspection failed endpoint=/mnt/target disk_state=check_failed\nHeal auto disk scanner idle",
|
||||
target
|
||||
));
|
||||
assert!(!live_disk_loss_scan_completed(
|
||||
"Heal auto disk scanner idle\nHeal auto-scan disk inspection failed endpoint=/mnt/target disk_state=check_failed",
|
||||
target
|
||||
));
|
||||
assert!(!live_disk_loss_scan_completed(
|
||||
"event=disk_health_check_failed endpoint=/mnt/target disk_state=check_failed\nHeal auto disk scanner idle",
|
||||
target
|
||||
));
|
||||
assert!(!live_disk_loss_scan_completed(
|
||||
"Heal auto-scan disk inspection failed endpoint=/mnt/other disk_state=check_failed\nHeal auto disk scanner idle",
|
||||
target
|
||||
));
|
||||
let path = std::env::temp_dir().join(format!("rustfs-replacement-scan-{}.log", std::process::id()));
|
||||
let stale =
|
||||
"Heal auto-scan disk inspection failed endpoint=/mnt/target disk_state=check_failed\nHeal auto disk scanner idle\n";
|
||||
fs::write(&path, stale)?;
|
||||
let offset = log_len(&path)?;
|
||||
assert!(!live_disk_loss_scan_completed_from_path(&path, offset, target)?);
|
||||
let fresh =
|
||||
"Heal auto-scan disk inspection failed endpoint=/mnt/target disk_state=check_failed\nHeal auto disk scanner idle\n";
|
||||
fs::write(&path, format!("{stale}{fresh}"))?;
|
||||
assert!(live_disk_loss_scan_completed_from_path(&path, offset, target)?);
|
||||
fs::remove_file(path)?;
|
||||
Ok(())
|
||||
}
|
||||
|
||||
@@ -954,6 +996,15 @@ mod tests {
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn recovery_census_only_treats_missing_version_as_transient() {
|
||||
let missing_version: Box<dyn Error + Send + Sync> = Box::new(rustfs_filemeta::Error::FileVersionNotFound);
|
||||
let missing_file: Box<dyn Error + Send + Sync> = Box::new(rustfs_filemeta::Error::FileNotFound);
|
||||
|
||||
assert!(is_transient_recovery_version_absence(missing_version.as_ref()));
|
||||
assert!(!is_transient_recovery_version_absence(missing_file.as_ref()));
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
async fn completion_poll_samples_census_before_status() {
|
||||
let order = std::rc::Rc::new(std::cell::RefCell::new(Vec::new()));
|
||||
|
||||
@@ -17,8 +17,56 @@ mod tests {
|
||||
use crate::common::{RustFSTestEnvironment, init_logging};
|
||||
use aws_sdk_s3::error::ProvideErrorMetadata;
|
||||
use aws_sdk_s3::primitives::ByteStream;
|
||||
use flate2::{Compression, write::GzEncoder};
|
||||
use std::error::Error;
|
||||
use std::io::Cursor;
|
||||
use std::io::{Cursor, Write};
|
||||
|
||||
fn pax_record(key: &str, value: &str) -> Vec<u8> {
|
||||
let payload = format!("{key}={value}\n");
|
||||
let mut len = payload.len() + 3;
|
||||
loop {
|
||||
let record = format!("{len} {payload}");
|
||||
if record.len() == len {
|
||||
return record.into_bytes();
|
||||
}
|
||||
len = record.len();
|
||||
}
|
||||
}
|
||||
|
||||
async fn append_pax_header(
|
||||
builder: &mut tokio_tar::Builder<Cursor<Vec<u8>>>,
|
||||
entry_type: tokio_tar::EntryType,
|
||||
records: &[(&str, &str)],
|
||||
) -> Result<(), Box<dyn Error + Send + Sync>> {
|
||||
let mut payload = Vec::new();
|
||||
for (key, value) in records {
|
||||
payload.extend(pax_record(key, value));
|
||||
}
|
||||
let mut header = tokio_tar::Header::new_ustar();
|
||||
header.set_entry_type(entry_type);
|
||||
header.set_size(u64::try_from(payload.len()).expect("PAX payload length should fit in u64"));
|
||||
header.set_mode(0o644);
|
||||
header.set_cksum();
|
||||
builder
|
||||
.append_data(&mut header, "PaxHeaders.X/snowball", Cursor::new(payload))
|
||||
.await?;
|
||||
Ok(())
|
||||
}
|
||||
|
||||
async fn append_typed_entry(
|
||||
builder: &mut tokio_tar::Builder<Cursor<Vec<u8>>>,
|
||||
path: &str,
|
||||
entry_type: tokio_tar::EntryType,
|
||||
body: &[u8],
|
||||
) -> Result<(), Box<dyn Error + Send + Sync>> {
|
||||
let mut header = tokio_tar::Header::new_gnu();
|
||||
header.set_entry_type(entry_type);
|
||||
header.set_size(u64::try_from(body.len()).expect("TAR member length should fit in u64"));
|
||||
header.set_mode(0o644);
|
||||
header.set_cksum();
|
||||
builder.append_data(&mut header, path, Cursor::new(body)).await?;
|
||||
Ok(())
|
||||
}
|
||||
|
||||
async fn build_test_archive() -> Result<Vec<u8>, Box<dyn Error + Send + Sync>> {
|
||||
let mut builder = tokio_tar::Builder::new(Cursor::new(Vec::new()));
|
||||
@@ -69,12 +117,50 @@ mod tests {
|
||||
Ok(builder.into_inner().await?.into_inner())
|
||||
}
|
||||
|
||||
fn build_archive_with_parent_dir_entry(victim_bucket: &str) -> Vec<u8> {
|
||||
let path = format!("../{victim_bucket}/evil-injected.txt");
|
||||
let data = b"injected-body";
|
||||
async fn build_archive_with_invalid_checksum() -> Result<Vec<u8>, Box<dyn Error + Send + Sync>> {
|
||||
let mut archive = build_test_archive().await?;
|
||||
archive[0] ^= 1;
|
||||
Ok(archive)
|
||||
}
|
||||
|
||||
async fn build_archive_with_negative_gnu_mtime() -> Result<Vec<u8>, Box<dyn Error + Send + Sync>> {
|
||||
let mut builder = tokio_tar::Builder::new(Cursor::new(Vec::new()));
|
||||
let mut header = tokio_tar::Header::new_gnu();
|
||||
header.set_size(b"negative-mtime-body".len() as u64);
|
||||
header.set_mode(0o644);
|
||||
header.as_old_mut().mtime.fill(0xff);
|
||||
builder
|
||||
.append_data(&mut header, "negative-mtime.txt", Cursor::new(b"negative-mtime-body".as_slice()))
|
||||
.await?;
|
||||
Ok(builder.into_inner().await?.into_inner())
|
||||
}
|
||||
|
||||
fn gzip_member(payload: &[u8]) -> Result<Vec<u8>, Box<dyn Error + Send + Sync>> {
|
||||
let mut encoder = GzEncoder::new(Vec::new(), Compression::default());
|
||||
encoder.write_all(payload)?;
|
||||
Ok(encoder.finish()?)
|
||||
}
|
||||
|
||||
async fn build_concatenated_gzip_archive() -> Result<Vec<u8>, Box<dyn Error + Send + Sync>> {
|
||||
let archive = build_test_archive().await?;
|
||||
let split_at = archive.len() / 2;
|
||||
let mut encoded = gzip_member(&archive[..split_at])?;
|
||||
encoded.extend(gzip_member(&archive[split_at..])?);
|
||||
Ok(encoded)
|
||||
}
|
||||
|
||||
async fn build_gzip_archive_with_invalid_crc() -> Result<Vec<u8>, Box<dyn Error + Send + Sync>> {
|
||||
let mut encoded = gzip_member(&build_test_archive().await?)?;
|
||||
let crc_offset = encoded.len().checked_sub(8).expect("gzip fixture must contain a trailer");
|
||||
encoded[crc_offset] ^= 1;
|
||||
Ok(encoded)
|
||||
}
|
||||
|
||||
fn append_raw_tar_entry_with_type(archive: &mut Vec<u8>, path: &[u8], data: &[u8], entry_type: u8) {
|
||||
assert!(path.len() <= 100, "raw TAR fixture path must fit in the name field");
|
||||
let mut header = [0u8; 512];
|
||||
|
||||
header[..path.len()].copy_from_slice(path.as_bytes());
|
||||
header[..path.len()].copy_from_slice(path);
|
||||
header[100..108].copy_from_slice(b"0000644\0");
|
||||
header[108..116].copy_from_slice(b"0000000\0");
|
||||
header[116..124].copy_from_slice(b"0000000\0");
|
||||
@@ -82,7 +168,7 @@ mod tests {
|
||||
header[124..136].copy_from_slice(size.as_bytes());
|
||||
header[136..148].copy_from_slice(b"00000000000\0");
|
||||
header[148..156].fill(b' ');
|
||||
header[156] = b'0';
|
||||
header[156] = entry_type;
|
||||
header[257..263].copy_from_slice(b"ustar\0");
|
||||
header[263..265].copy_from_slice(b"00");
|
||||
|
||||
@@ -90,11 +176,87 @@ mod tests {
|
||||
let checksum = format!("{:06o}\0 ", checksum);
|
||||
header[148..156].copy_from_slice(checksum.as_bytes());
|
||||
|
||||
let mut archive = Vec::new();
|
||||
archive.extend_from_slice(&header);
|
||||
archive.extend_from_slice(data);
|
||||
let padding = (512 - (data.len() % 512)) % 512;
|
||||
archive.extend(std::iter::repeat_n(0, padding));
|
||||
}
|
||||
|
||||
fn append_raw_tar_entry(archive: &mut Vec<u8>, path: &[u8], data: &[u8]) {
|
||||
append_raw_tar_entry_with_type(archive, path, data, b'0');
|
||||
}
|
||||
|
||||
fn build_archive_with_parent_dir_entry(victim_bucket: &str) -> Vec<u8> {
|
||||
let path = format!("../{victim_bucket}/evil-injected.txt");
|
||||
let mut archive = Vec::new();
|
||||
append_raw_tar_entry(&mut archive, path.as_bytes(), b"injected-body");
|
||||
archive.extend_from_slice(&[0u8; 1024]);
|
||||
archive
|
||||
}
|
||||
|
||||
async fn build_member_semantics_archive() -> Result<Vec<u8>, Box<dyn Error + Send + Sync>> {
|
||||
let mut builder = tokio_tar::Builder::new(Cursor::new(Vec::new()));
|
||||
append_pax_header(
|
||||
&mut builder,
|
||||
tokio_tar::EntryType::XGlobalHeader,
|
||||
&[
|
||||
("minio.metadata.x-amz-meta-owner", "global"),
|
||||
("minio.metadata.x-amz-meta-snowball-auto-extract", "true"),
|
||||
],
|
||||
)
|
||||
.await?;
|
||||
append_pax_header(
|
||||
&mut builder,
|
||||
tokio_tar::EntryType::XHeader,
|
||||
&[("minio.metadata.x-amz-meta-owner", "local")],
|
||||
)
|
||||
.await?;
|
||||
append_typed_entry(&mut builder, "regular.txt", tokio_tar::EntryType::Regular, b"regular-body").await?;
|
||||
for (path, entry_type) in [
|
||||
("char", tokio_tar::EntryType::Char),
|
||||
("block", tokio_tar::EntryType::Block),
|
||||
("fifo", tokio_tar::EntryType::Fifo),
|
||||
] {
|
||||
append_typed_entry(&mut builder, path, entry_type, b"").await?;
|
||||
}
|
||||
let mut directory = tokio_tar::Header::new_gnu();
|
||||
directory.set_entry_type(tokio_tar::EntryType::Directory);
|
||||
directory.set_size(0);
|
||||
directory.set_mode(0o755);
|
||||
directory.set_cksum();
|
||||
builder
|
||||
.append_data(&mut directory, "directory/", Cursor::new(Vec::new()))
|
||||
.await?;
|
||||
for (path, entry_type) in [
|
||||
("hard-link", tokio_tar::EntryType::Link),
|
||||
("symlink", tokio_tar::EntryType::Symlink),
|
||||
("continuous", tokio_tar::EntryType::Continuous),
|
||||
("unknown", tokio_tar::EntryType::Other(b'9')),
|
||||
] {
|
||||
append_typed_entry(&mut builder, path, entry_type, b"").await?;
|
||||
}
|
||||
Ok(builder.into_inner().await?.into_inner())
|
||||
}
|
||||
|
||||
async fn build_versioned_member_archive(path: &str, version_id: &str) -> Result<Vec<u8>, Box<dyn Error + Send + Sync>> {
|
||||
let mut builder = tokio_tar::Builder::new(Cursor::new(Vec::new()));
|
||||
append_pax_header(&mut builder, tokio_tar::EntryType::XHeader, &[("minio.versionId", version_id)]).await?;
|
||||
append_typed_entry(&mut builder, path, tokio_tar::EntryType::Regular, b"versioned-body").await?;
|
||||
Ok(builder.into_inner().await?.into_inner())
|
||||
}
|
||||
|
||||
fn build_archive_with_invalid_utf8_entry() -> Vec<u8> {
|
||||
let mut archive = Vec::new();
|
||||
append_raw_tar_entry(&mut archive, b"invalid-\xff.txt", b"ignored-body");
|
||||
append_raw_tar_entry(&mut archive, b"valid.txt", b"valid-body");
|
||||
archive.extend_from_slice(&[0u8; 1024]);
|
||||
archive
|
||||
}
|
||||
|
||||
fn build_archive_with_invalid_utf8_symlink() -> Vec<u8> {
|
||||
let mut archive = Vec::new();
|
||||
append_raw_tar_entry_with_type(&mut archive, b"invalid-\xff-link", b"", b'2');
|
||||
append_raw_tar_entry(&mut archive, b"valid.txt", b"valid-body");
|
||||
archive.extend_from_slice(&[0u8; 1024]);
|
||||
archive
|
||||
}
|
||||
@@ -135,6 +297,147 @@ mod tests {
|
||||
Ok(())
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
async fn snowball_auto_extract_applies_member_semantics_and_metadata_precedence() -> Result<(), Box<dyn Error + Send + Sync>>
|
||||
{
|
||||
init_logging();
|
||||
|
||||
let mut env = RustFSTestEnvironment::new().await?;
|
||||
env.start_rustfs_server(vec![]).await?;
|
||||
|
||||
let client = env.create_s3_client();
|
||||
let bucket = "snowball-member-semantics";
|
||||
client.create_bucket().bucket(bucket).send().await?;
|
||||
client
|
||||
.put_object()
|
||||
.bucket(bucket)
|
||||
.key("fixture.tar")
|
||||
.metadata("Snowball-Auto-Extract", "true")
|
||||
.metadata("Minio-Snowball-Prefix", "members")
|
||||
.metadata("owner", "outer")
|
||||
.body(ByteStream::from(build_member_semantics_archive().await?))
|
||||
.send()
|
||||
.await?;
|
||||
|
||||
let regular = client.head_object().bucket(bucket).key("members/regular.txt").send().await?;
|
||||
let regular_metadata = regular.metadata().expect("regular member should expose metadata");
|
||||
assert_eq!(regular_metadata.get("owner").map(String::as_str), Some("local"));
|
||||
assert!(!regular_metadata.contains_key("snowball-auto-extract"));
|
||||
assert!(!regular_metadata.contains_key("minio-snowball-prefix"));
|
||||
|
||||
for key in ["char", "block", "fifo"] {
|
||||
let head = client
|
||||
.head_object()
|
||||
.bucket(bucket)
|
||||
.key(format!("members/{key}"))
|
||||
.send()
|
||||
.await?;
|
||||
assert_eq!(head.content_length(), Some(0), "{key} should be materialized as an empty object");
|
||||
assert_eq!(
|
||||
head.metadata().and_then(|metadata| metadata.get("owner")).map(String::as_str),
|
||||
Some("outer"),
|
||||
"{key} should not inherit global PAX metadata"
|
||||
);
|
||||
}
|
||||
let directory = client.head_object().bucket(bucket).key("members/directory/").send().await?;
|
||||
assert_eq!(directory.content_length(), Some(0));
|
||||
|
||||
for key in ["hard-link", "symlink", "continuous", "unknown"] {
|
||||
let error = client
|
||||
.head_object()
|
||||
.bucket(bucket)
|
||||
.key(format!("members/{key}"))
|
||||
.send()
|
||||
.await
|
||||
.expect_err("unsupported TAR entry type must be skipped");
|
||||
assert_eq!(error.into_service_error().code(), Some("NotFound"), "{key}");
|
||||
}
|
||||
|
||||
env.stop_server();
|
||||
Ok(())
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
async fn snowball_auto_extract_validates_pax_version_id_against_bucket_state() -> Result<(), Box<dyn Error + Send + Sync>> {
|
||||
init_logging();
|
||||
|
||||
let mut env = RustFSTestEnvironment::new().await?;
|
||||
env.start_rustfs_server(vec![]).await?;
|
||||
|
||||
let client = env.create_s3_client();
|
||||
let bucket = "snowball-version-semantics";
|
||||
client.create_bucket().bucket(bucket).send().await?;
|
||||
|
||||
client
|
||||
.put_object()
|
||||
.bucket(bucket)
|
||||
.key("null.tar")
|
||||
.metadata("Snowball-Auto-Extract", "true")
|
||||
.body(ByteStream::from(build_versioned_member_archive("null.txt", "null").await?))
|
||||
.send()
|
||||
.await?;
|
||||
let null_member = client.get_object().bucket(bucket).key("null.txt").send().await?;
|
||||
assert_eq!(null_member.body.collect().await?.into_bytes().as_ref(), b"versioned-body");
|
||||
|
||||
for (archive_key, member_key, version_id) in [
|
||||
("uuid.tar", "uuid.txt", uuid::Uuid::new_v4().to_string()),
|
||||
("uppercase-null.tar", "uppercase-null.txt", "NULL".to_string()),
|
||||
] {
|
||||
let error = client
|
||||
.put_object()
|
||||
.bucket(bucket)
|
||||
.key(archive_key)
|
||||
.metadata("Snowball-Auto-Extract", "true")
|
||||
.body(ByteStream::from(build_versioned_member_archive(member_key, &version_id).await?))
|
||||
.send()
|
||||
.await
|
||||
.expect_err("invalid or unversioned UUID import must be rejected");
|
||||
assert_eq!(error.into_service_error().code(), Some("InvalidArgument"), "{archive_key}");
|
||||
let missing = client
|
||||
.head_object()
|
||||
.bucket(bucket)
|
||||
.key(member_key)
|
||||
.send()
|
||||
.await
|
||||
.expect_err("rejected version import must not create an object");
|
||||
assert_eq!(missing.into_service_error().code(), Some("NotFound"), "{member_key}");
|
||||
}
|
||||
|
||||
client
|
||||
.put_bucket_versioning()
|
||||
.bucket(bucket)
|
||||
.versioning_configuration(
|
||||
aws_sdk_s3::types::VersioningConfiguration::builder()
|
||||
.status(aws_sdk_s3::types::BucketVersioningStatus::Enabled)
|
||||
.build(),
|
||||
)
|
||||
.send()
|
||||
.await?;
|
||||
let imported_version_id = uuid::Uuid::new_v4().to_string();
|
||||
client
|
||||
.put_object()
|
||||
.bucket(bucket)
|
||||
.key("versioned-uuid.tar")
|
||||
.metadata("Snowball-Auto-Extract", "true")
|
||||
.body(ByteStream::from(
|
||||
build_versioned_member_archive("versioned-uuid.txt", &imported_version_id).await?,
|
||||
))
|
||||
.send()
|
||||
.await?;
|
||||
let imported = client
|
||||
.get_object()
|
||||
.bucket(bucket)
|
||||
.key("versioned-uuid.txt")
|
||||
.version_id(&imported_version_id)
|
||||
.send()
|
||||
.await?;
|
||||
assert_eq!(imported.version_id(), Some(imported_version_id.as_str()));
|
||||
assert_eq!(imported.body.collect().await?.into_bytes().as_ref(), b"versioned-body");
|
||||
|
||||
env.stop_server();
|
||||
Ok(())
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
async fn snowball_auto_extract_supports_standard_headers_with_combined_extract_options()
|
||||
-> Result<(), Box<dyn Error + Send + Sync>> {
|
||||
@@ -263,6 +566,113 @@ mod tests {
|
||||
Ok(())
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
async fn snowball_auto_extract_accepts_negative_gnu_mtime() -> Result<(), Box<dyn Error + Send + Sync>> {
|
||||
init_logging();
|
||||
|
||||
let mut env = RustFSTestEnvironment::new().await?;
|
||||
env.start_rustfs_server(vec![]).await?;
|
||||
|
||||
let client = env.create_s3_client();
|
||||
let bucket = "snowball-negative-mtime";
|
||||
client.create_bucket().bucket(bucket).send().await?;
|
||||
client
|
||||
.put_object()
|
||||
.bucket(bucket)
|
||||
.key("fixture.tar")
|
||||
.metadata("Snowball-Auto-Extract", "true")
|
||||
.body(ByteStream::from(build_archive_with_negative_gnu_mtime().await?))
|
||||
.send()
|
||||
.await?;
|
||||
|
||||
let object = client.get_object().bucket(bucket).key("negative-mtime.txt").send().await?;
|
||||
assert_eq!(object.body.collect().await?.into_bytes().as_ref(), b"negative-mtime-body");
|
||||
|
||||
env.stop_server();
|
||||
Ok(())
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
async fn snowball_auto_extract_consumes_concatenated_gzip_members() -> Result<(), Box<dyn Error + Send + Sync>> {
|
||||
init_logging();
|
||||
|
||||
let mut env = RustFSTestEnvironment::new().await?;
|
||||
env.start_rustfs_server(vec![]).await?;
|
||||
|
||||
let client = env.create_s3_client();
|
||||
let bucket = "snowball-concatenated-gzip";
|
||||
client.create_bucket().bucket(bucket).send().await?;
|
||||
client
|
||||
.put_object()
|
||||
.bucket(bucket)
|
||||
.key("fixture.tar.gz")
|
||||
.metadata("Snowball-Auto-Extract", "true")
|
||||
.body(ByteStream::from(build_concatenated_gzip_archive().await?))
|
||||
.send()
|
||||
.await?;
|
||||
|
||||
let object = client.get_object().bucket(bucket).key("root.txt").send().await?;
|
||||
assert_eq!(object.body.collect().await?.into_bytes().as_ref(), b"root payload\n");
|
||||
|
||||
env.stop_server();
|
||||
Ok(())
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
async fn snowball_auto_extract_rejects_gzip_crc_error_when_ignore_errors_enabled() -> Result<(), Box<dyn Error + Send + Sync>>
|
||||
{
|
||||
init_logging();
|
||||
|
||||
let mut env = RustFSTestEnvironment::new().await?;
|
||||
env.start_rustfs_server(vec![]).await?;
|
||||
|
||||
let client = env.create_s3_client();
|
||||
let bucket = "snowball-gzip-crc-ignore-errors";
|
||||
client.create_bucket().bucket(bucket).send().await?;
|
||||
let err = client
|
||||
.put_object()
|
||||
.bucket(bucket)
|
||||
.key("fixture.tar.gz")
|
||||
.metadata("Snowball-Auto-Extract", "true")
|
||||
.metadata("Minio-Snowball-Ignore-Errors", "true")
|
||||
.body(ByteStream::from(build_gzip_archive_with_invalid_crc().await?))
|
||||
.send()
|
||||
.await
|
||||
.expect_err("gzip integrity failures must remain fatal under ignore-errors");
|
||||
|
||||
assert_eq!(err.into_service_error().code(), Some("InvalidArgument"));
|
||||
|
||||
env.stop_server();
|
||||
Ok(())
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
async fn snowball_auto_extract_rejects_mismatched_content_md5() -> Result<(), Box<dyn Error + Send + Sync>> {
|
||||
init_logging();
|
||||
|
||||
let mut env = RustFSTestEnvironment::new().await?;
|
||||
env.start_rustfs_server(vec![]).await?;
|
||||
|
||||
let client = env.create_s3_client();
|
||||
let bucket = "snowball-content-md5";
|
||||
client.create_bucket().bucket(bucket).send().await?;
|
||||
let err = client
|
||||
.put_object()
|
||||
.bucket(bucket)
|
||||
.key("fixture.tar")
|
||||
.metadata("Snowball-Auto-Extract", "true")
|
||||
.content_md5("AAAAAAAAAAAAAAAAAAAAAA==")
|
||||
.body(ByteStream::from(build_test_archive().await?))
|
||||
.send()
|
||||
.await
|
||||
.expect_err("mismatched Content-MD5 must fail after the raw body reaches EOF");
|
||||
|
||||
assert_eq!(err.into_service_error().code(), Some("BadDigest"));
|
||||
|
||||
env.stop_server();
|
||||
Ok(())
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
async fn snowball_auto_extract_ignores_invalid_entries_when_requested() -> Result<(), Box<dyn Error + Send + Sync>> {
|
||||
init_logging();
|
||||
@@ -299,7 +709,100 @@ mod tests {
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
async fn snowball_auto_extract_rejects_parent_dir_entry_without_cross_bucket_write()
|
||||
async fn snowball_auto_extract_skips_non_utf8_symlink_without_ignore_errors() -> Result<(), Box<dyn Error + Send + Sync>> {
|
||||
init_logging();
|
||||
|
||||
let mut env = RustFSTestEnvironment::new().await?;
|
||||
env.start_rustfs_server(vec![]).await?;
|
||||
|
||||
let client = env.create_s3_client();
|
||||
let bucket = "snowball-invalid-utf8-link";
|
||||
client.create_bucket().bucket(bucket).send().await?;
|
||||
|
||||
client
|
||||
.put_object()
|
||||
.bucket(bucket)
|
||||
.key("fixture.tar")
|
||||
.metadata("Snowball-Auto-Extract", "true")
|
||||
.body(ByteStream::from(build_archive_with_invalid_utf8_symlink()))
|
||||
.send()
|
||||
.await?;
|
||||
|
||||
let valid = client.get_object().bucket(bucket).key("valid.txt").send().await?;
|
||||
assert_eq!(valid.body.collect().await?.into_bytes().as_ref(), b"valid-body");
|
||||
let listed = client.list_objects_v2().bucket(bucket).send().await?;
|
||||
let keys: Vec<_> = listed.contents().iter().filter_map(|entry| entry.key()).collect();
|
||||
assert_eq!(keys, vec!["valid.txt"]);
|
||||
|
||||
env.stop_server();
|
||||
Ok(())
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
async fn snowball_auto_extract_skips_non_utf8_member_without_lossy_key_collision() -> Result<(), Box<dyn Error + Send + Sync>>
|
||||
{
|
||||
init_logging();
|
||||
|
||||
let mut env = RustFSTestEnvironment::new().await?;
|
||||
env.start_rustfs_server(vec![]).await?;
|
||||
|
||||
let client = env.create_s3_client();
|
||||
let bucket = "snowball-invalid-utf8";
|
||||
client.create_bucket().bucket(bucket).send().await?;
|
||||
|
||||
client
|
||||
.put_object()
|
||||
.bucket(bucket)
|
||||
.key("fixture.tar")
|
||||
.metadata("Snowball-Auto-Extract", "true")
|
||||
.metadata("Minio-Snowball-Ignore-Errors", "true")
|
||||
.body(ByteStream::from(build_archive_with_invalid_utf8_entry()))
|
||||
.send()
|
||||
.await?;
|
||||
|
||||
let valid = client.get_object().bucket(bucket).key("valid.txt").send().await?;
|
||||
assert_eq!(valid.body.collect().await?.into_bytes().as_ref(), b"valid-body");
|
||||
let listed = client.list_objects_v2().bucket(bucket).send().await?;
|
||||
let keys: Vec<_> = listed.contents().iter().filter_map(|entry| entry.key()).collect();
|
||||
assert_eq!(keys, vec!["valid.txt"]);
|
||||
|
||||
env.stop_server();
|
||||
Ok(())
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
async fn snowball_auto_extract_rejects_corrupt_tar_when_ignore_errors_enabled() -> Result<(), Box<dyn Error + Send + Sync>> {
|
||||
init_logging();
|
||||
|
||||
let mut env = RustFSTestEnvironment::new().await?;
|
||||
env.start_rustfs_server(vec![]).await?;
|
||||
|
||||
let client = env.create_s3_client();
|
||||
let bucket = "snowball-corrupt-ignore-errors";
|
||||
let archive = build_archive_with_invalid_checksum().await?;
|
||||
client.create_bucket().bucket(bucket).send().await?;
|
||||
|
||||
let err = client
|
||||
.put_object()
|
||||
.bucket(bucket)
|
||||
.key("fixture.tar")
|
||||
.metadata("Snowball-Auto-Extract", "true")
|
||||
.metadata("Minio-Snowball-Ignore-Errors", "true")
|
||||
.body(ByteStream::from(archive))
|
||||
.send()
|
||||
.await
|
||||
.expect_err("corrupt TAR structure must remain fatal under ignore-errors");
|
||||
assert_eq!(err.into_service_error().code(), Some("InvalidArgument"));
|
||||
|
||||
let listed = client.list_objects_v2().bucket(bucket).send().await?;
|
||||
assert!(listed.contents().is_empty(), "corrupt archive must not produce objects");
|
||||
|
||||
env.stop_server();
|
||||
Ok(())
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
async fn snowball_auto_extract_rejects_parent_dir_entry_even_when_ignore_errors_enabled()
|
||||
-> Result<(), Box<dyn Error + Send + Sync>> {
|
||||
init_logging();
|
||||
|
||||
@@ -319,6 +822,7 @@ mod tests {
|
||||
.bucket(attacker_bucket)
|
||||
.key("fixture.tar")
|
||||
.metadata("Snowball-Auto-Extract", "true")
|
||||
.metadata("Minio-Snowball-Ignore-Errors", "true")
|
||||
.body(ByteStream::from(archive))
|
||||
.send()
|
||||
.await
|
||||
|
||||
@@ -12,14 +12,17 @@
|
||||
// See the License for the specific language governing permissions and
|
||||
// limitations under the License.
|
||||
|
||||
use crate::common::{RustFSTestEnvironment, init_logging};
|
||||
use crate::common::{RustFSTestClusterEnvironment, RustFSTestEnvironment, init_logging, rustfs_binary_path};
|
||||
use aws_sdk_s3::Client;
|
||||
use aws_sdk_s3::error::ProvideErrorMetadata;
|
||||
use aws_sdk_s3::primitives::ByteStream;
|
||||
use aws_sdk_s3::types::{
|
||||
BucketVersioningStatus, CompletedMultipartUpload, CompletedPart, ServerSideEncryption, VersioningConfiguration,
|
||||
};
|
||||
use std::path::PathBuf;
|
||||
use std::path::{Path, PathBuf};
|
||||
use std::time::Duration;
|
||||
use tokio::task::JoinSet;
|
||||
use tokio::time::{Instant, sleep};
|
||||
|
||||
type TestResult = Result<(), Box<dyn std::error::Error + Send + Sync>>;
|
||||
|
||||
@@ -28,6 +31,14 @@ const SSE_MASTER_KEY_ENV: &str = "RUSTFS_SSE_S3_MASTER_KEY";
|
||||
const SSE_MASTER_KEY: &str = "QkJCQkJCQkJCQkJCQkJCQkJCQkJCQkJCQkJCQkJCQkI=";
|
||||
const PLAIN_BUCKET: &str = "upgrade-plain-data";
|
||||
const VERSIONED_BUCKET: &str = "upgrade-versioned-data";
|
||||
const MIXED_BUCKET: &str = "upgrade-mixed-version-data";
|
||||
const MIXED_NODE_COUNT: usize = 4;
|
||||
const MULTIPART_WORKERS: usize = 16;
|
||||
const MULTIPART_UPLOADS_PER_WORKER: usize = 16;
|
||||
// Peers keep a restarted node's drive in Suspect/Returning for roughly
|
||||
// probe_interval (2s) x success_threshold (3) after it comes back; 30s
|
||||
// comfortably covers that window plus CI scheduling jitter.
|
||||
const LISTING_CONVERGENCE_TIMEOUT: Duration = Duration::from_secs(30);
|
||||
|
||||
fn source_binary() -> Result<PathBuf, Box<dyn std::error::Error + Send + Sync>> {
|
||||
let path = std::env::var_os(SOURCE_BINARY_ENV)
|
||||
@@ -103,6 +114,132 @@ async fn write_multipart(client: &Client, bucket: &str, key: &str, parts: &[Vec<
|
||||
Ok(())
|
||||
}
|
||||
|
||||
fn configure_cluster_logs(cluster: &mut RustFSTestClusterEnvironment) -> TestResult {
|
||||
let Some(log_dir) = std::env::var_os("RUSTFS_E2E_LOG_DIR") else {
|
||||
return Ok(());
|
||||
};
|
||||
std::fs::create_dir_all(&log_dir)?;
|
||||
for node_idx in 0..cluster.nodes.len() {
|
||||
let path = Path::new(&log_dir).join(format!("mixed-upgrade-node-{node_idx}.log"));
|
||||
cluster.set_node_capture_log_path(node_idx, path.to_string_lossy().into_owned())?;
|
||||
}
|
||||
Ok(())
|
||||
}
|
||||
|
||||
async fn write_multipart_load(clients: &[Client], phase: &str) -> Result<Vec<String>, Box<dyn std::error::Error + Send + Sync>> {
|
||||
let mut tasks = JoinSet::new();
|
||||
for worker in 0..MULTIPART_WORKERS {
|
||||
let client = clients[worker % clients.len()].clone();
|
||||
let phase = phase.to_string();
|
||||
tasks.spawn(async move {
|
||||
let mut keys = Vec::with_capacity(MULTIPART_UPLOADS_PER_WORKER);
|
||||
for upload in 0..MULTIPART_UPLOADS_PER_WORKER {
|
||||
let key = format!("{phase}/multipart/{worker:02}/{upload:02}");
|
||||
let part = vec![u8::try_from(worker)?; 64 * 1024];
|
||||
write_multipart(&client, MIXED_BUCKET, &key, &[part]).await?;
|
||||
keys.push(key);
|
||||
}
|
||||
Ok::<_, Box<dyn std::error::Error + Send + Sync>>(keys)
|
||||
});
|
||||
}
|
||||
|
||||
let mut keys = Vec::with_capacity(MULTIPART_WORKERS * MULTIPART_UPLOADS_PER_WORKER);
|
||||
while let Some(result) = tasks.join_next().await {
|
||||
keys.extend(result??);
|
||||
}
|
||||
Ok(keys)
|
||||
}
|
||||
|
||||
/// Assert that `client` eventually lists exactly `expected` objects under
|
||||
/// `{phase}/`, polling until [`LISTING_CONVERGENCE_TIMEOUT`].
|
||||
///
|
||||
/// A single-snapshot assertion here is racy by construction: each phase both
|
||||
/// writes and lists within seconds of a node restart. While a peer still holds
|
||||
/// the restarted node's drive in Suspect/Returning, strict-quorum listing
|
||||
/// consults only the remaining three drives and drops any object that was
|
||||
/// itself legally written at write quorum (3/4 drives) during an earlier
|
||||
/// node's identical post-restart window — its xl.meta is then visible on only
|
||||
/// two of the three consulted drives, below the required object quorum of
|
||||
/// three. GET still succeeds for such objects; only the listing under-counts
|
||||
/// until drive health converges. A genuine upgrade data-loss regression still
|
||||
/// fails after the deadline.
|
||||
async fn wait_for_phase_listing(client: &Client, phase: &str, expected: usize, context: &str) -> TestResult {
|
||||
let deadline = Instant::now() + LISTING_CONVERGENCE_TIMEOUT;
|
||||
loop {
|
||||
let listed = client
|
||||
.list_objects_v2()
|
||||
.bucket(MIXED_BUCKET)
|
||||
.prefix(format!("{phase}/"))
|
||||
.send()
|
||||
.await?;
|
||||
let count = listed.contents().len();
|
||||
if count == expected {
|
||||
return Ok(());
|
||||
}
|
||||
if Instant::now() >= deadline {
|
||||
return Err(format!(
|
||||
"{context}: listing under {phase}/ returned {count} of {expected} objects even after {}s of post-restart convergence",
|
||||
LISTING_CONVERGENCE_TIMEOUT.as_secs()
|
||||
)
|
||||
.into());
|
||||
}
|
||||
sleep(Duration::from_millis(500)).await;
|
||||
}
|
||||
}
|
||||
|
||||
async fn exercise_mixed_cluster(
|
||||
cluster: &RustFSTestClusterEnvironment,
|
||||
phase: &str,
|
||||
current_node: usize,
|
||||
previous_node: usize,
|
||||
) -> TestResult {
|
||||
let clients = cluster.create_all_clients()?;
|
||||
let current_client = &clients[current_node];
|
||||
let previous_client = &clients[previous_node];
|
||||
|
||||
let current_key = format!("{phase}/written-by-current");
|
||||
let current_body = format!("{phase}: current RustFS build").into_bytes();
|
||||
current_client
|
||||
.put_object()
|
||||
.bucket(MIXED_BUCKET)
|
||||
.key(¤t_key)
|
||||
.body(ByteStream::from(current_body.clone()))
|
||||
.send()
|
||||
.await?;
|
||||
assert_eq!(read_object(previous_client, MIXED_BUCKET, ¤t_key, None).await?.1, current_body);
|
||||
|
||||
let previous_key = format!("{phase}/written-by-previous");
|
||||
let previous_body = format!("{phase}: previous RustFS release").into_bytes();
|
||||
previous_client
|
||||
.put_object()
|
||||
.bucket(MIXED_BUCKET)
|
||||
.key(&previous_key)
|
||||
.body(ByteStream::from(previous_body.clone()))
|
||||
.send()
|
||||
.await?;
|
||||
assert_eq!(read_object(current_client, MIXED_BUCKET, &previous_key, None).await?.1, previous_body);
|
||||
|
||||
let multipart_keys = write_multipart_load(&clients, phase).await?;
|
||||
let expected_count = multipart_keys.len() + 2;
|
||||
for (label, client) in [("current", current_client), ("previous", previous_client)] {
|
||||
wait_for_phase_listing(
|
||||
client,
|
||||
phase,
|
||||
expected_count,
|
||||
&format!("the {label} RustFS version must stream the complete mixed-version listing"),
|
||||
)
|
||||
.await?;
|
||||
}
|
||||
|
||||
let last_multipart_key = format!("{phase}/multipart/{:02}/{:02}", MULTIPART_WORKERS - 1, MULTIPART_UPLOADS_PER_WORKER - 1);
|
||||
assert_eq!(
|
||||
read_object(previous_client, MIXED_BUCKET, &last_multipart_key, None).await?.1,
|
||||
vec![u8::try_from(MULTIPART_WORKERS - 1)?; 64 * 1024]
|
||||
);
|
||||
|
||||
Ok(())
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
#[ignore = "requires a pinned previous RustFS release binary"]
|
||||
async fn direct_upgrade_from_rc2_preserves_object_contracts() -> TestResult {
|
||||
@@ -252,3 +389,43 @@ async fn direct_upgrade_from_rc2_preserves_object_contracts() -> TestResult {
|
||||
|
||||
Ok(())
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
#[ignore = "requires a pinned previous RustFS release binary"]
|
||||
async fn rolling_upgrade_from_rc2_preserves_mixed_version_contracts() -> TestResult {
|
||||
init_logging();
|
||||
let previous_binary = source_binary()?;
|
||||
let current_binary = rustfs_binary_path();
|
||||
let mut cluster = RustFSTestClusterEnvironment::new(MIXED_NODE_COUNT).await?;
|
||||
cluster.set_env("RUST_LOG", "rustfs=warn,rustfs_notify=warn");
|
||||
configure_cluster_logs(&mut cluster)?;
|
||||
cluster.start_with_binary(&previous_binary).await?;
|
||||
cluster.create_test_bucket(MIXED_BUCKET).await?;
|
||||
|
||||
cluster.stop_node(0)?;
|
||||
cluster.start_node_from_binary(0, ¤t_binary).await?;
|
||||
exercise_mixed_cluster(&cluster, "one-current-node", 0, 1).await?;
|
||||
|
||||
for node_idx in [1, 2] {
|
||||
cluster.stop_node(node_idx)?;
|
||||
cluster.start_node_from_binary(node_idx, ¤t_binary).await?;
|
||||
}
|
||||
exercise_mixed_cluster(&cluster, "one-previous-node", 0, 3).await?;
|
||||
|
||||
cluster.stop_node(3)?;
|
||||
cluster.start_node_from_binary(3, ¤t_binary).await?;
|
||||
|
||||
for (node_idx, client) in cluster.create_all_clients()?.iter().enumerate() {
|
||||
for phase in ["one-current-node", "one-previous-node"] {
|
||||
wait_for_phase_listing(
|
||||
client,
|
||||
phase,
|
||||
MULTIPART_WORKERS * MULTIPART_UPLOADS_PER_WORKER + 2,
|
||||
&format!("node {node_idx}: the homogeneous current cluster must preserve every object"),
|
||||
)
|
||||
.await?;
|
||||
}
|
||||
}
|
||||
|
||||
Ok(())
|
||||
}
|
||||
|
||||
@@ -75,6 +75,10 @@ pub mod bucket {
|
||||
delete_transition_candidate_for_operator, finalize_missing_transition_transaction_for_operator,
|
||||
inspect_transition_transaction_for_operator,
|
||||
};
|
||||
#[cfg(feature = "test-util")]
|
||||
pub use crate::bucket::lifecycle::transition_transaction::{
|
||||
TransitionTransactionRecoveryStats, recover_transition_transaction_records,
|
||||
};
|
||||
}
|
||||
|
||||
pub mod evaluator {
|
||||
@@ -99,6 +103,17 @@ pub mod bucket {
|
||||
pub use crate::bucket::lifecycle::tier_delete_journal::{
|
||||
persist_tier_delete_journal_entry, record_tier_delete_journal_backend_identity,
|
||||
};
|
||||
|
||||
#[cfg(feature = "test-util")]
|
||||
pub mod test_util {
|
||||
/// Model a single-node, all-v6 fleet after its capability probe has completed.
|
||||
///
|
||||
/// Call this only once while constructing an isolated test store, before any
|
||||
/// tier-delete journal permit or background worker can be active.
|
||||
pub fn install_all_v6_fleet_capability_proof() {
|
||||
crate::services::notification_sys::install_cross_pool_fence_fleet_proof_for_test();
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
pub mod tier_last_day_stats {
|
||||
@@ -406,7 +421,7 @@ pub mod notification {
|
||||
pub use crate::services::notification_sys::{
|
||||
CrossPoolFenceFleetProofToken, NotificationPeerErr, NotificationSys, ScannerPublicationLeaseGrant,
|
||||
acquire_cross_pool_fence_fleet_proof, cross_pool_fence_fleet_proof_matches, get_global_notification_sys,
|
||||
new_global_notification_sys, start_remote_version_state_fleet_probe,
|
||||
new_global_notification_sys, scanner_peer_transport_error_message_is_retryable, start_remote_version_state_fleet_probe,
|
||||
};
|
||||
}
|
||||
|
||||
@@ -416,9 +431,10 @@ pub mod object {
|
||||
GetObjectBodyCacheHookLookup, GetObjectBodySource, GetObjectReader, NamespaceLockFence, ObjectEncryptionResolver,
|
||||
ObjectInfo, ObjectLockConfigSnapshot, ObjectMutationHook, ObjectOptions, PutObjReader, QuotaAdmission,
|
||||
RangedDecompressReader, ReadEncryptionMaterial, ReadEncryptionMode, ReadEncryptionRequest,
|
||||
SCANNER_PUBLICATION_LEASE_FENCE_METADATA_KEY, StreamConsumer, get_object_body_cache_plaintext_len,
|
||||
lookup_get_object_body_cache_hook, register_get_object_body_cache_hook, register_object_mutation_hook,
|
||||
unregister_get_object_body_cache_hook, unregister_object_mutation_hook,
|
||||
SCANNER_PUBLICATION_LEASE_FENCE_METADATA_KEY, ScannerPublicationCommitScope, ScannerPublicationCommitStartError,
|
||||
ScannerPublicationCommitState, StreamConsumer, get_object_body_cache_plaintext_len, lookup_get_object_body_cache_hook,
|
||||
register_get_object_body_cache_hook, register_object_mutation_hook, unregister_get_object_body_cache_hook,
|
||||
unregister_object_mutation_hook,
|
||||
};
|
||||
pub use crate::store::{
|
||||
PrepareSelectObjectSnapshotError, PreparedGetObjectReader, SelectObjectSnapshot, SelectObjectSnapshotReadError,
|
||||
@@ -460,8 +476,8 @@ pub mod rpc {
|
||||
tonic_boot_epoch_challenge, tonic_boot_epoch_response_headers, tonic_rpc_auth_failure_reason,
|
||||
verify_ns_scanner_capability, verify_ns_scanner_capability_with_tier_registry_generation, verify_put_file_auth_trailer,
|
||||
verify_put_file_capability, verify_rpc_signature, verify_tonic_boot_epoch_response, verify_tonic_canonical_body_digest,
|
||||
verify_tonic_mutation_body_digest, verify_tonic_rpc_response_proof, verify_tonic_rpc_signature,
|
||||
verify_tonic_rpc_signature_with_bootstrap,
|
||||
verify_tonic_mutation_body_digest, verify_tonic_mutation_body_digest_reject_unsigned, verify_tonic_rpc_response_proof,
|
||||
verify_tonic_rpc_signature, verify_tonic_rpc_signature_with_bootstrap,
|
||||
};
|
||||
}
|
||||
|
||||
@@ -488,9 +504,9 @@ pub mod storage {
|
||||
pub use crate::core::pools::HealLifecycleExpiryContext;
|
||||
pub use crate::store::HealWalkVersion;
|
||||
pub use crate::store::{
|
||||
ECStore, SCANNER_PUBLICATION_LEASE_TTL_MS, all_local_disk, all_local_disk_path, find_local_disk_by_ref, init_local_disks,
|
||||
init_local_disks_with_instance_ctx, init_lock_clients, prewarm_local_disk_id_map,
|
||||
prewarm_local_disk_id_map_with_instance_ctx,
|
||||
ECStore, SCANNER_PUBLICATION_LEASE_TTL_MS, ScannerDataMovementPauseStatus, all_local_disk, all_local_disk_path,
|
||||
find_local_disk_by_ref, init_local_disks, init_local_disks_with_instance_ctx, init_lock_clients,
|
||||
prewarm_local_disk_id_map, prewarm_local_disk_id_map_with_instance_ctx,
|
||||
};
|
||||
}
|
||||
|
||||
|
||||
@@ -22,7 +22,9 @@ use crate::bucket::target::{self, BucketTarget, BucketTargets, Credentials};
|
||||
use crate::bucket::versioning_sys::BucketVersioningSys;
|
||||
use crate::runtime::sources as runtime_sources;
|
||||
use aws_credential_types::Credentials as SdkCredentials;
|
||||
use aws_credential_types::provider::{ProvideCredentials, error::CredentialsError, future};
|
||||
use aws_sdk_s3::config::Region as SdkRegion;
|
||||
use aws_sdk_s3::config::RequestChecksumCalculation;
|
||||
use aws_sdk_s3::config::SharedHttpClient;
|
||||
use aws_sdk_s3::error::ProvideErrorMetadata;
|
||||
use aws_sdk_s3::error::SdkError;
|
||||
@@ -38,6 +40,7 @@ use aws_sdk_s3::primitives::ByteStream;
|
||||
use aws_sdk_s3::types::Tagging as SdkTagging;
|
||||
use aws_sdk_s3::types::{
|
||||
ChecksumMode, CompletedMultipartUpload, CompletedPart, ObjectLockLegalHoldStatus, ObjectLockRetentionMode,
|
||||
ServerSideEncryption,
|
||||
};
|
||||
use aws_sdk_s3::{Client as S3Client, Config as S3Config, operation::head_object::HeadObjectOutput};
|
||||
use aws_sdk_s3::{config::SharedCredentialsProvider, types::BucketVersioningStatus};
|
||||
@@ -77,7 +80,7 @@ use std::str::FromStr as _;
|
||||
use std::sync::Arc;
|
||||
use std::sync::OnceLock;
|
||||
use std::sync::Weak;
|
||||
use std::time::{Duration, Instant};
|
||||
use std::time::{Duration, Instant, SystemTime};
|
||||
use time::{OffsetDateTime, format_description::well_known::Rfc3339};
|
||||
use tokio::sync::Mutex;
|
||||
use tokio::sync::RwLock;
|
||||
@@ -89,6 +92,71 @@ use uuid::Uuid;
|
||||
|
||||
const MAX_CONCURRENT_TARGET_HEALTH_CHECKS: usize = 16;
|
||||
const REDACTED_CREDENTIAL: &str = "<redacted>";
|
||||
const EXPIRED_REMOTE_TARGET_CREDENTIALS: &str = "remote target credentials have expired";
|
||||
|
||||
#[derive(Clone)]
|
||||
struct RemoteTargetCredentialsProvider {
|
||||
credentials: SdkCredentials,
|
||||
}
|
||||
|
||||
impl RemoteTargetCredentialsProvider {
|
||||
fn resolve_at(&self, now: SystemTime) -> aws_credential_types::provider::Result {
|
||||
if self.credentials.expiry().is_some_and(|expiration| expiration <= now) {
|
||||
return Err(CredentialsError::provider_error(std::io::Error::other(EXPIRED_REMOTE_TARGET_CREDENTIALS)));
|
||||
}
|
||||
Ok(self.credentials.clone())
|
||||
}
|
||||
}
|
||||
|
||||
impl fmt::Debug for RemoteTargetCredentialsProvider {
|
||||
fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result {
|
||||
f.debug_struct("RemoteTargetCredentialsProvider")
|
||||
.field("temporary", &self.credentials.session_token().is_some())
|
||||
.field("expiration", &self.credentials.expiry())
|
||||
.finish()
|
||||
}
|
||||
}
|
||||
|
||||
impl ProvideCredentials for RemoteTargetCredentialsProvider {
|
||||
fn provide_credentials<'a>(&'a self) -> future::ProvideCredentials<'a>
|
||||
where
|
||||
Self: 'a,
|
||||
{
|
||||
future::ProvideCredentials::ready(self.resolve_at(SystemTime::now()))
|
||||
}
|
||||
|
||||
fn fallback_on_interrupt(&self) -> Option<SdkCredentials> {
|
||||
self.resolve_at(SystemTime::now()).ok()
|
||||
}
|
||||
}
|
||||
|
||||
fn remote_target_sdk_credentials(
|
||||
credentials: &Credentials,
|
||||
account_id: &str,
|
||||
now: SystemTime,
|
||||
) -> Result<SdkCredentials, &'static str> {
|
||||
let session_token = credentials.effective_session_token();
|
||||
let expiration = credentials.effective_expiration().map(SystemTime::from);
|
||||
if expiration.is_some() && session_token.is_none() {
|
||||
return Err("remote target credential expiration requires a session token");
|
||||
}
|
||||
if expiration.is_some_and(|expiration| expiration <= now) {
|
||||
return Err(EXPIRED_REMOTE_TARGET_CREDENTIALS);
|
||||
}
|
||||
|
||||
let mut builder = SdkCredentials::builder()
|
||||
.access_key_id(credentials.access_key.clone())
|
||||
.secret_access_key(credentials.secret_key.clone())
|
||||
.account_id(account_id.to_string())
|
||||
.provider_name("bucket_target_sys");
|
||||
if let Some(session_token) = session_token {
|
||||
builder = builder.session_token(session_token.to_string());
|
||||
}
|
||||
if let Some(expiration) = expiration {
|
||||
builder = builder.expiry(expiration);
|
||||
}
|
||||
Ok(builder.build())
|
||||
}
|
||||
|
||||
pub type HeadObjectSdkError = Box<SdkError<HeadObjectError>>;
|
||||
pub type GetObjectSdkError = Box<SdkError<GetObjectError>>;
|
||||
@@ -417,6 +485,19 @@ impl BucketTargetSys {
|
||||
mutex
|
||||
}
|
||||
|
||||
/// Snapshot the heartbeat-tracked health of `url`'s endpoint.
|
||||
///
|
||||
/// Returns `None` when the heartbeat has never seen the endpoint. Unlike
|
||||
/// [`Self::is_offline`] this deliberately does not call `init_hc`: a caller
|
||||
/// that only reports metrics must not create health entries as a side
|
||||
/// effect, or merely rendering a status page would mark an unknown peer
|
||||
/// online.
|
||||
pub async fn endpoint_health(&self, url: &Url) -> Option<EpHealth> {
|
||||
let key = endpoint_health_key(url);
|
||||
let health_map = self.h_mutex.read().await;
|
||||
health_map.get(&key).cloned()
|
||||
}
|
||||
|
||||
pub async fn is_offline(&self, url: &Url) -> bool {
|
||||
let key = endpoint_health_key(url);
|
||||
{
|
||||
@@ -845,13 +926,26 @@ impl BucketTargetSys {
|
||||
Ok(BucketTargets { targets: new_targets })
|
||||
}
|
||||
|
||||
async fn mark_refresh_attempt(&self, arn: &str) {
|
||||
// Rate-limit a failed config fetch as well as a failed client build.
|
||||
// A successful rebuild replaces this timestamp during publication.
|
||||
self.arn_remotes_map
|
||||
.write()
|
||||
.await
|
||||
.entry(arn.to_string())
|
||||
.or_default()
|
||||
.last_refresh = OffsetDateTime::now_utc();
|
||||
}
|
||||
|
||||
pub async fn mark_refresh_in_progress(&self, bucket: &str, arn: &str) {
|
||||
let mut arn_errs = self.arn_errs_map.write().await;
|
||||
arn_errs.entry(arn.to_string()).or_insert_with(|| ArnErrs {
|
||||
bucket: bucket.to_string(),
|
||||
update_in_progress: true,
|
||||
let err = arn_errs.entry(arn.to_string()).or_insert_with(|| ArnErrs {
|
||||
count: 1,
|
||||
bucket: bucket.to_string(),
|
||||
..Default::default()
|
||||
});
|
||||
err.update_in_progress = true;
|
||||
err.bucket = bucket.to_string();
|
||||
}
|
||||
|
||||
pub async fn mark_refresh_done(&self, bucket: &str, arn: &str) {
|
||||
@@ -863,15 +957,21 @@ impl BucketTargetSys {
|
||||
}
|
||||
|
||||
pub async fn is_reloading_target(&self, _bucket: &str, arn: &str) -> bool {
|
||||
let arn_errs = self.arn_errs_map.read().await;
|
||||
arn_errs.get(arn).map(|err| err.update_in_progress).unwrap_or(false)
|
||||
self.arn_errs_map
|
||||
.read()
|
||||
.await
|
||||
.get(arn)
|
||||
.is_some_and(|err| err.update_in_progress)
|
||||
}
|
||||
|
||||
pub async fn inc_arn_errs(&self, _bucket: &str, arn: &str) {
|
||||
pub async fn inc_arn_errs(&self, bucket: &str, arn: &str) {
|
||||
let mut arn_errs = self.arn_errs_map.write().await;
|
||||
if let Some(err) = arn_errs.get_mut(arn) {
|
||||
err.count += 1;
|
||||
}
|
||||
let err = arn_errs.entry(arn.to_string()).or_insert_with(|| ArnErrs {
|
||||
bucket: bucket.to_string(),
|
||||
..Default::default()
|
||||
});
|
||||
err.count += 1;
|
||||
err.bucket = bucket.to_string();
|
||||
}
|
||||
|
||||
pub async fn get_remote_target_client(&self, bucket: &str, arn: &str) -> Option<Arc<TargetClient>> {
|
||||
@@ -884,15 +984,15 @@ impl BucketTargetSys {
|
||||
.unwrap_or((None, None))
|
||||
};
|
||||
|
||||
if let Some(cli) = cli {
|
||||
let credentials_expired = cli
|
||||
.as_ref()
|
||||
.is_some_and(|client| client.credentials_expired_at(jiff::Timestamp::now()));
|
||||
if let Some(cli) = cli
|
||||
&& !credentials_expired
|
||||
{
|
||||
return Some(cli);
|
||||
}
|
||||
|
||||
// TODO(backlog): spawn an async task to proactively reload the replication target
|
||||
if self.is_reloading_target(bucket, arn).await {
|
||||
return None;
|
||||
}
|
||||
|
||||
if let Some(last_refresh) = last_refresh {
|
||||
let now = OffsetDateTime::now_utc();
|
||||
if now - last_refresh < Duration::from_secs(60 * 5) {
|
||||
@@ -900,16 +1000,24 @@ impl BucketTargetSys {
|
||||
}
|
||||
}
|
||||
|
||||
// The existing per-bucket publication lock is also the reload claim:
|
||||
// try-locking keeps the request path non-blocking, is cancellation-safe,
|
||||
// and prevents a stale reload from publishing after a credential update.
|
||||
let update_mutex = self.target_update_mutex(bucket).await;
|
||||
let Ok(update_guard) = update_mutex.try_lock() else {
|
||||
return None;
|
||||
};
|
||||
self.mark_refresh_attempt(arn).await;
|
||||
|
||||
match get_bucket_targets_config(bucket).await {
|
||||
Ok(bucket_targets) => {
|
||||
self.mark_refresh_in_progress(bucket, arn).await;
|
||||
self.update_all_targets(bucket, Some(&bucket_targets)).await;
|
||||
self.mark_refresh_done(bucket, arn).await;
|
||||
self.update_all_targets_locked(bucket, Some(&bucket_targets)).await;
|
||||
}
|
||||
Err(e) => {
|
||||
error!("get bucket targets config error:{}", e);
|
||||
}
|
||||
};
|
||||
drop(update_guard);
|
||||
|
||||
let cli = self
|
||||
.arn_remotes_map
|
||||
@@ -917,8 +1025,10 @@ impl BucketTargetSys {
|
||||
.await
|
||||
.get(arn)
|
||||
.and_then(|target| target.client.clone());
|
||||
if cli.is_some() {
|
||||
return cli;
|
||||
if let Some(cli) = cli
|
||||
&& !cli.credentials_expired_at(jiff::Timestamp::now())
|
||||
{
|
||||
return Some(cli);
|
||||
}
|
||||
|
||||
self.inc_arn_errs(bucket, arn).await;
|
||||
@@ -948,12 +1058,13 @@ impl BucketTargetSys {
|
||||
});
|
||||
};
|
||||
|
||||
let creds = SdkCredentials::builder()
|
||||
.access_key_id(credentials.access_key.clone())
|
||||
.secret_access_key(credentials.secret_key.clone())
|
||||
.account_id(target.reset_id.clone())
|
||||
.provider_name("bucket_target_sys")
|
||||
.build();
|
||||
let creds = remote_target_sdk_credentials(credentials, &target.reset_id, SystemTime::now()).map_err(|error| {
|
||||
BucketTargetError::RemoteTargetConnectionErr {
|
||||
bucket: target.target_bucket.clone(),
|
||||
access_key: credentials.access_key.clone(),
|
||||
error: error.to_string(),
|
||||
}
|
||||
})?;
|
||||
|
||||
let endpoint = if target.secure {
|
||||
format!("https://{}", target.endpoint)
|
||||
@@ -973,9 +1084,10 @@ impl BucketTargetSys {
|
||||
|
||||
let mut config_builder = S3Config::builder()
|
||||
.endpoint_url(endpoint.clone())
|
||||
.credentials_provider(SharedCredentialsProvider::new(creds))
|
||||
.credentials_provider(SharedCredentialsProvider::new(RemoteTargetCredentialsProvider { credentials: creds }))
|
||||
.region(SdkRegion::new(target.region.clone()))
|
||||
.behavior_version(aws_sdk_s3::config::BehaviorVersion::latest());
|
||||
.behavior_version(aws_sdk_s3::config::BehaviorVersion::latest())
|
||||
.request_checksum_calculation(replication_request_checksum_calculation());
|
||||
|
||||
if should_force_path_style(target) {
|
||||
config_builder = config_builder.force_path_style(true);
|
||||
@@ -1047,6 +1159,13 @@ impl BucketTargetSys {
|
||||
let update_mutex = self.target_update_mutex(bucket).await;
|
||||
let _update_guard = update_mutex.lock().await;
|
||||
|
||||
self.update_all_targets_locked(bucket, targets).await;
|
||||
}
|
||||
|
||||
/// Builds and publishes one bucket snapshot while its update mutex is held.
|
||||
/// Keeping persisted-config reads under the same mutex prevents a stale
|
||||
/// reload from overwriting a concurrent credential rotation.
|
||||
async fn update_all_targets_locked(&self, bucket: &str, targets: Option<&BucketTargets>) {
|
||||
let mut clients = Vec::new();
|
||||
if let Some(new_targets) = targets {
|
||||
for target in &new_targets.targets {
|
||||
@@ -1078,6 +1197,17 @@ impl BucketTargetSys {
|
||||
&& !new_targets.is_empty()
|
||||
{
|
||||
for (target, client) in clients {
|
||||
// Keep a timestamped placeholder for configured targets whose
|
||||
// client cannot be built. Replication records these attempts as
|
||||
// failed, while the placeholder prevents every object from
|
||||
// triggering another metadata reload/client build for five minutes.
|
||||
arn_remotes_map.insert(
|
||||
target.arn.clone(),
|
||||
ArnTarget {
|
||||
client: None,
|
||||
last_refresh: OffsetDateTime::now_utc(),
|
||||
},
|
||||
);
|
||||
match client {
|
||||
Ok(client) => {
|
||||
arn_remotes_map.insert(
|
||||
@@ -1090,11 +1220,6 @@ impl BucketTargetSys {
|
||||
health_map.insert(client.arn.clone(), target_health(&client));
|
||||
self.update_bandwidth_limit(bucket, &target.arn, target.bandwidth_limit);
|
||||
}
|
||||
// The target stays in `targets_map`, so it keeps showing up in
|
||||
// `bucket remote ls` while no client exists to replicate through it —
|
||||
// replication then drops every object for this ARN. Without this the
|
||||
// rejection (loopback endpoint, bad CA, unparseable URL) left no trace
|
||||
// anywhere.
|
||||
Err(err) => warn!(
|
||||
bucket = %bucket,
|
||||
arn = %target.arn,
|
||||
@@ -1258,6 +1383,25 @@ fn loopback_replication_targets_allowed() -> bool {
|
||||
.unwrap_or(false)
|
||||
}
|
||||
|
||||
const REPLICATION_STREAMING_CHECKSUMS_ENV: &str = "RUSTFS_REPLICATION_STREAMING_CHECKSUMS";
|
||||
|
||||
/// Streaming trailer checksums make the SDK frame request bodies as
|
||||
/// `aws-chunked`; a target that does not decode that framing stores the frames
|
||||
/// verbatim, silently corrupting every replica while the transfer itself
|
||||
/// succeeds (#6853). Plain signed payloads are the compatible default; the env
|
||||
/// knob restores trailer checksums for fleets whose targets are all known to
|
||||
/// decode them.
|
||||
fn replication_request_checksum_calculation() -> RequestChecksumCalculation {
|
||||
if std::env::var(REPLICATION_STREAMING_CHECKSUMS_ENV)
|
||||
.map(|v| v.eq_ignore_ascii_case("true") || v == "1")
|
||||
.unwrap_or(false)
|
||||
{
|
||||
RequestChecksumCalculation::WhenSupported
|
||||
} else {
|
||||
RequestChecksumCalculation::WhenRequired
|
||||
}
|
||||
}
|
||||
|
||||
fn validate_replication_target_endpoint(url: &Url) -> Result<(), OutboundUrlError> {
|
||||
validate_replication_target_endpoint_inner(url, loopback_replication_targets_allowed())
|
||||
}
|
||||
@@ -1637,6 +1781,17 @@ impl Default for AdvancedPutOptions {
|
||||
}
|
||||
}
|
||||
|
||||
/// The subset of the target's PutObject response replication audits.
|
||||
#[derive(Debug, Clone)]
|
||||
pub struct RemotePutObjectResponse {
|
||||
/// Version id the target assigned (`x-amz-version-id`).
|
||||
pub version_id: Option<String>,
|
||||
/// ETag of what the target stored; `None` when the target withheld it or
|
||||
/// when its encryption mode (SSE-KMS / SSE-C) makes it incomparable to
|
||||
/// the source ETag. `None` is therefore "not decidable", never evidence.
|
||||
pub etag: Option<String>,
|
||||
}
|
||||
|
||||
#[derive(Clone)]
|
||||
pub struct PutObjectOptions {
|
||||
pub user_metadata: HashMap<String, String>,
|
||||
@@ -1962,6 +2117,13 @@ pub struct TargetClient {
|
||||
}
|
||||
|
||||
impl TargetClient {
|
||||
fn credentials_expired_at(&self, now: jiff::Timestamp) -> bool {
|
||||
self.credentials
|
||||
.as_ref()
|
||||
.and_then(Credentials::effective_expiration)
|
||||
.is_some_and(|expiration| expiration <= now)
|
||||
}
|
||||
|
||||
pub fn to_url(&self) -> Url {
|
||||
Url::parse(&self.endpoint).unwrap()
|
||||
}
|
||||
@@ -2175,7 +2337,9 @@ impl TargetClient {
|
||||
|
||||
/// On success returns the version id the target assigned (from
|
||||
/// `x-amz-version-id`), letting callers audit the version-identity
|
||||
/// contract — a target that adopts the source version echoes it back.
|
||||
/// contract — a target that adopts the source version echoes it back —
|
||||
/// together with the ETag of what the target actually stored, so callers
|
||||
/// can detect a target that persisted transformed bytes (#6853).
|
||||
pub async fn put_object(
|
||||
&self,
|
||||
bucket: &str,
|
||||
@@ -2183,7 +2347,7 @@ impl TargetClient {
|
||||
size: i64,
|
||||
body: ByteStream,
|
||||
opts: &PutObjectOptions,
|
||||
) -> Result<Option<String>, S3ClientError> {
|
||||
) -> Result<RemotePutObjectResponse, S3ClientError> {
|
||||
let mut headers = opts.header();
|
||||
|
||||
let builder = self.client.put_object();
|
||||
@@ -2218,7 +2382,25 @@ impl TargetClient {
|
||||
.send()
|
||||
.await
|
||||
{
|
||||
Ok(output) => Ok(output.version_id().map(ToOwned::to_owned)),
|
||||
Ok(output) => {
|
||||
// Under SSE-KMS/DSSE or SSE-C the target's ETag is not the MD5
|
||||
// of the stored plaintext, so it cannot be compared against the
|
||||
// source ETag; withhold it rather than let a caller conclude
|
||||
// corruption from an opaque value.
|
||||
let etag_comparable = output.sse_customer_algorithm().is_none()
|
||||
&& !matches!(
|
||||
output.server_side_encryption(),
|
||||
Some(ServerSideEncryption::AwsKms) | Some(ServerSideEncryption::AwsKmsDsse)
|
||||
);
|
||||
Ok(RemotePutObjectResponse {
|
||||
version_id: output.version_id().map(ToOwned::to_owned),
|
||||
etag: if etag_comparable {
|
||||
output.e_tag().map(ToOwned::to_owned)
|
||||
} else {
|
||||
None
|
||||
},
|
||||
})
|
||||
}
|
||||
Err(e) => match e {
|
||||
SdkError::ServiceError(service_err) => {
|
||||
let err = service_err.into_err();
|
||||
@@ -2557,6 +2739,165 @@ mod tests {
|
||||
}
|
||||
}
|
||||
|
||||
type RecordedHeaders = Arc<std::sync::Mutex<Vec<Vec<(String, String)>>>>;
|
||||
|
||||
/// Records full request headers and answers with canned response headers,
|
||||
/// for asserting wire framing and response parsing.
|
||||
#[derive(Clone, Debug)]
|
||||
struct RecordingHeaderConnector {
|
||||
request_headers: RecordedHeaders,
|
||||
response_headers: Vec<(String, String)>,
|
||||
}
|
||||
|
||||
impl SmithyHttpConnector for RecordingHeaderConnector {
|
||||
fn call(&self, request: HttpRequest) -> HttpConnectorFuture {
|
||||
self.request_headers
|
||||
.lock()
|
||||
.expect("recorded header lock should not be poisoned")
|
||||
.push(
|
||||
request
|
||||
.headers()
|
||||
.iter()
|
||||
.map(|(k, v)| (k.to_string(), v.to_string()))
|
||||
.collect(),
|
||||
);
|
||||
let mut response = HttpResponse::new(
|
||||
aws_smithy_runtime_api::http::StatusCode::try_from(200_u16).expect("200 should be a valid response status"),
|
||||
SdkBody::empty(),
|
||||
);
|
||||
for (name, value) in &self.response_headers {
|
||||
response.headers_mut().insert(name.clone(), value.clone());
|
||||
}
|
||||
HttpConnectorFuture::ready(Ok(response))
|
||||
}
|
||||
}
|
||||
|
||||
fn header_recording_target_client(response_headers: Vec<(String, String)>) -> (TargetClient, RecordedHeaders) {
|
||||
let request_headers: RecordedHeaders = Arc::new(std::sync::Mutex::new(Vec::new()));
|
||||
let connector = SharedHttpConnector::new(RecordingHeaderConnector {
|
||||
request_headers: Arc::clone(&request_headers),
|
||||
response_headers,
|
||||
});
|
||||
let http_client = http_client_fn(move |_settings, _components| connector.clone());
|
||||
let client = s3_client_for_test(443, Some(http_client));
|
||||
(
|
||||
TargetClient {
|
||||
endpoint: "https://localhost:443".to_string(),
|
||||
credentials: None,
|
||||
bucket: "target-bucket".to_string(),
|
||||
storage_class: String::new(),
|
||||
disable_proxy: false,
|
||||
arn: "arn:rustfs:replication:us-east-1:target:bucket".to_string(),
|
||||
reset_id: String::new(),
|
||||
secure: true,
|
||||
health_check_duration: Duration::from_secs(5),
|
||||
replicate_sync: false,
|
||||
client: Arc::new(client),
|
||||
},
|
||||
request_headers,
|
||||
)
|
||||
}
|
||||
|
||||
fn streaming_test_body(payload: &'static [u8]) -> ByteStream {
|
||||
let stream = tokio_util::io::ReaderStream::new(std::io::Cursor::new(payload));
|
||||
let body = http_body_util::StreamBody::new(futures::StreamExt::map(stream, |r| r.map(http_body::Frame::data)));
|
||||
ByteStream::new(SdkBody::from_body_1_x(body))
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn replication_checksums_default_to_plain_payloads() {
|
||||
assert!(matches!(
|
||||
replication_request_checksum_calculation(),
|
||||
RequestChecksumCalculation::WhenRequired
|
||||
));
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
async fn replication_put_object_sends_plain_signed_payloads_by_default() {
|
||||
let (client, recorded) = header_recording_target_client(Vec::new());
|
||||
client
|
||||
.put_object("target-bucket", "object", 4, streaming_test_body(b"data"), &PutObjectOptions::default())
|
||||
.await
|
||||
.expect("recorded put_object should succeed");
|
||||
|
||||
let recorded = recorded.lock().expect("recorded header lock should not be poisoned");
|
||||
let headers = &recorded[0];
|
||||
let header = |name: &str| {
|
||||
headers
|
||||
.iter()
|
||||
.find(|(k, _)| k.eq_ignore_ascii_case(name))
|
||||
.map(|(_, v)| v.as_str())
|
||||
};
|
||||
// The #6853 regression shape: trailer checksums force aws-chunked
|
||||
// framing, which a non-decoding target stores verbatim as the object.
|
||||
assert_eq!(header("x-amz-trailer"), None, "streaming uploads must not carry a trailer checksum");
|
||||
assert!(
|
||||
header("content-encoding").is_none_or(|v| !v.contains("aws-chunked")),
|
||||
"streaming uploads must not be aws-chunked framed"
|
||||
);
|
||||
assert_eq!(header("x-amz-decoded-content-length"), None);
|
||||
assert_eq!(header("content-length"), Some("4"));
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
async fn put_object_returns_the_etag_the_target_stored() {
|
||||
let (client, _) =
|
||||
header_recording_target_client(vec![("etag".to_string(), "\"9a0364b9e99bb480dd25e1f0284c8555\"".to_string())]);
|
||||
let response = client
|
||||
.put_object(
|
||||
"target-bucket",
|
||||
"object",
|
||||
4,
|
||||
ByteStream::from_static(b"data"),
|
||||
&PutObjectOptions::default(),
|
||||
)
|
||||
.await
|
||||
.expect("recorded put_object should succeed");
|
||||
assert_eq!(response.etag.as_deref(), Some("\"9a0364b9e99bb480dd25e1f0284c8555\""));
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
async fn put_object_withholds_the_etag_under_target_side_kms() {
|
||||
let (client, _) = header_recording_target_client(vec![
|
||||
("etag".to_string(), "\"9a0364b9e99bb480dd25e1f0284c8555\"".to_string()),
|
||||
("x-amz-server-side-encryption".to_string(), "aws:kms".to_string()),
|
||||
]);
|
||||
let response = client
|
||||
.put_object(
|
||||
"target-bucket",
|
||||
"object",
|
||||
4,
|
||||
ByteStream::from_static(b"data"),
|
||||
&PutObjectOptions::default(),
|
||||
)
|
||||
.await
|
||||
.expect("recorded put_object should succeed");
|
||||
assert!(
|
||||
response.etag.is_none(),
|
||||
"a KMS-encrypted replica's etag is not the content MD5 and must be withheld"
|
||||
);
|
||||
}
|
||||
|
||||
#[derive(Clone, Debug)]
|
||||
struct RecordingAuthConnector {
|
||||
signed_requests: Arc<std::sync::Mutex<Vec<(bool, bool)>>>,
|
||||
}
|
||||
|
||||
impl SmithyHttpConnector for RecordingAuthConnector {
|
||||
fn call(&self, request: HttpRequest) -> HttpConnectorFuture {
|
||||
let has_expected_token = request.headers().get("x-amz-security-token") == Some("temporary-session-token");
|
||||
let has_authorization = request.headers().contains_key("authorization");
|
||||
self.signed_requests
|
||||
.lock()
|
||||
.expect("recorded auth request lock should not be poisoned")
|
||||
.push((has_expected_token, has_authorization));
|
||||
HttpConnectorFuture::ready(Ok(HttpResponse::new(
|
||||
aws_smithy_runtime_api::http::StatusCode::try_from(200_u16).expect("200 should be a valid response status"),
|
||||
SdkBody::empty(),
|
||||
)))
|
||||
}
|
||||
}
|
||||
|
||||
fn recording_target_client() -> (TargetClient, Arc<std::sync::Mutex<Vec<String>>>) {
|
||||
let request_uris = Arc::new(std::sync::Mutex::new(Vec::new()));
|
||||
let connector = SharedHttpConnector::new(RecordingHttpConnector {
|
||||
@@ -2582,6 +2923,150 @@ mod tests {
|
||||
)
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn remote_target_sdk_credentials_preserve_temporary_credential_fields() {
|
||||
let now = SystemTime::UNIX_EPOCH + Duration::from_secs(1_000);
|
||||
let expiration = SystemTime::UNIX_EPOCH + Duration::from_secs(2_000);
|
||||
let credentials = Credentials {
|
||||
access_key: "access".to_string(),
|
||||
secret_key: "secret".to_string(),
|
||||
session_token: Some("temporary-session-token".to_string()),
|
||||
expiration: Some(jiff::Timestamp::try_from(expiration).expect("test expiration should convert")),
|
||||
};
|
||||
|
||||
let sdk_credentials =
|
||||
remote_target_sdk_credentials(&credentials, "account", now).expect("unexpired temporary credentials should build");
|
||||
|
||||
assert_eq!(sdk_credentials.session_token(), Some("temporary-session-token"));
|
||||
assert_eq!(sdk_credentials.expiry(), Some(expiration));
|
||||
assert_eq!(sdk_credentials.account_id().map(|id| id.as_str()), Some("account"));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn remote_target_sdk_credentials_normalize_go_zero_expiration() {
|
||||
let credentials = Credentials {
|
||||
access_key: "access".to_string(),
|
||||
secret_key: "secret".to_string(),
|
||||
session_token: None,
|
||||
expiration: Some("0001-01-01T00:00:00Z".parse().expect("Go zero time should parse")),
|
||||
};
|
||||
|
||||
let sdk_credentials = remote_target_sdk_credentials(&credentials, "", SystemTime::now())
|
||||
.expect("Go zero expiration should remain compatible with static credentials");
|
||||
|
||||
assert!(sdk_credentials.session_token().is_none());
|
||||
assert!(sdk_credentials.expiry().is_none());
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn remote_target_sdk_credentials_reject_invalid_expiration_boundaries() {
|
||||
let expiration = SystemTime::UNIX_EPOCH + Duration::from_secs(2_000);
|
||||
let mut credentials = Credentials {
|
||||
access_key: "access".to_string(),
|
||||
secret_key: "secret".to_string(),
|
||||
session_token: None,
|
||||
expiration: Some(jiff::Timestamp::try_from(expiration).expect("test expiration should convert")),
|
||||
};
|
||||
|
||||
assert_eq!(
|
||||
remote_target_sdk_credentials(&credentials, "", SystemTime::UNIX_EPOCH + Duration::from_secs(1_000))
|
||||
.expect_err("expiration without a session token must fail"),
|
||||
"remote target credential expiration requires a session token"
|
||||
);
|
||||
|
||||
credentials.session_token = Some("temporary-session-token".to_string());
|
||||
assert_eq!(
|
||||
remote_target_sdk_credentials(&credentials, "", expiration)
|
||||
.expect_err("credentials expire at the exact expiration boundary"),
|
||||
EXPIRED_REMOTE_TARGET_CREDENTIALS
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn remote_target_credentials_provider_fails_closed_after_expiration() {
|
||||
let expiration = SystemTime::UNIX_EPOCH + Duration::from_secs(2_000);
|
||||
let provider = RemoteTargetCredentialsProvider {
|
||||
credentials: SdkCredentials::new(
|
||||
"access",
|
||||
"secret",
|
||||
Some("temporary-session-token".to_string()),
|
||||
Some(expiration),
|
||||
"test",
|
||||
),
|
||||
};
|
||||
|
||||
assert!(provider.resolve_at(expiration - Duration::from_nanos(1)).is_ok());
|
||||
let err = provider
|
||||
.resolve_at(expiration)
|
||||
.expect_err("expired credentials must not be returned");
|
||||
assert_eq!(err.source().map(ToString::to_string).as_deref(), Some(EXPIRED_REMOTE_TARGET_CREDENTIALS));
|
||||
assert!(!format!("{provider:?}").contains("temporary-session-token"));
|
||||
assert!(!format!("{provider:?}").contains("secret"));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn target_client_detects_expiration_for_cache_refresh() {
|
||||
let expiration: jiff::Timestamp = "2099-01-01T00:00:00Z".parse().expect("expiration should parse");
|
||||
let (mut client, _) = recording_target_client();
|
||||
client.credentials = Some(Credentials {
|
||||
access_key: "access".to_string(),
|
||||
secret_key: "secret".to_string(),
|
||||
session_token: Some("temporary-session-token".to_string()),
|
||||
expiration: Some(expiration),
|
||||
});
|
||||
|
||||
assert!(!client.credentials_expired_at("2098-12-31T23:59:59Z".parse().expect("pre-expiration timestamp should parse")));
|
||||
assert!(client.credentials_expired_at(expiration));
|
||||
|
||||
client.credentials.as_mut().expect("credentials should exist").expiration =
|
||||
Some("0001-01-01T00:00:00Z".parse().expect("Go zero time should parse"));
|
||||
assert!(!client.credentials_expired_at(jiff::Timestamp::now()));
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
async fn temporary_credentials_add_security_token_to_sigv4_requests() {
|
||||
let signed_requests = Arc::new(std::sync::Mutex::new(Vec::new()));
|
||||
let connector = SharedHttpConnector::new(RecordingAuthConnector {
|
||||
signed_requests: Arc::clone(&signed_requests),
|
||||
});
|
||||
let http_client = http_client_fn(move |_settings, _components| connector.clone());
|
||||
let credentials = Credentials {
|
||||
access_key: "access".to_string(),
|
||||
secret_key: "secret".to_string(),
|
||||
session_token: Some("temporary-session-token".to_string()),
|
||||
expiration: Some("2099-01-01T00:00:00Z".parse().expect("future expiration should parse")),
|
||||
};
|
||||
let sdk_credentials = remote_target_sdk_credentials(&credentials, "", SystemTime::now())
|
||||
.expect("unexpired temporary credentials should build");
|
||||
let client = S3Client::from_conf(
|
||||
S3Config::builder()
|
||||
.endpoint_url("https://target.example")
|
||||
.credentials_provider(SharedCredentialsProvider::new(RemoteTargetCredentialsProvider {
|
||||
credentials: sdk_credentials,
|
||||
}))
|
||||
.region(SdkRegion::new("us-east-1"))
|
||||
.http_client(http_client)
|
||||
.behavior_version(aws_sdk_s3::config::BehaviorVersion::latest())
|
||||
.build(),
|
||||
);
|
||||
|
||||
client
|
||||
.head_bucket()
|
||||
.bucket("target-bucket")
|
||||
.send()
|
||||
.await
|
||||
.expect("recording connector should accept the signed request");
|
||||
|
||||
assert_eq!(
|
||||
signed_requests
|
||||
.lock()
|
||||
.expect("recorded auth request lock should not be poisoned")
|
||||
.as_slice(),
|
||||
&[(true, true)],
|
||||
"SigV4 request must include both authorization and the session-token header"
|
||||
);
|
||||
}
|
||||
|
||||
fn spawn_https_server(cert: &rcgen::CertifiedKey<rcgen::KeyPair>, requests: usize) -> (u16, std::thread::JoinHandle<()>) {
|
||||
use std::io::{Read, Write};
|
||||
|
||||
@@ -2689,7 +3174,10 @@ mod tests {
|
||||
.credentials_provider(SharedCredentialsProvider::new(credentials))
|
||||
.region(SdkRegion::new("us-east-1"))
|
||||
.force_path_style(true)
|
||||
.behavior_version(aws_sdk_s3::config::BehaviorVersion::latest());
|
||||
.behavior_version(aws_sdk_s3::config::BehaviorVersion::latest())
|
||||
// Mirror the production remote-target builder so recorded requests
|
||||
// exercise the same checksum/framing behavior (#6853).
|
||||
.request_checksum_calculation(replication_request_checksum_calculation());
|
||||
if let Some(http_client) = http_client {
|
||||
config = config.http_client(http_client);
|
||||
}
|
||||
@@ -3513,6 +4001,29 @@ mod tests {
|
||||
assert!(mutexes.contains_key("second"));
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
async fn target_refresh_attempt_updates_retry_timestamp_and_error_count() {
|
||||
let sys = BucketTargetSys::default();
|
||||
|
||||
sys.mark_refresh_attempt("arn:reload").await;
|
||||
let last_refresh = sys.arn_remotes_map.read().await["arn:reload"].last_refresh;
|
||||
assert!(OffsetDateTime::now_utc() - last_refresh < Duration::from_secs(5));
|
||||
|
||||
sys.inc_arn_errs("bucket", "arn:reload").await;
|
||||
sys.inc_arn_errs("bucket", "arn:reload").await;
|
||||
let errors = sys.arn_errs_map.read().await;
|
||||
assert_eq!(errors["arn:reload"].count, 2);
|
||||
assert_eq!(errors["arn:reload"].bucket, "bucket");
|
||||
drop(errors);
|
||||
|
||||
sys.mark_refresh_in_progress("bucket", "arn:reload").await;
|
||||
assert!(sys.is_reloading_target("bucket", "arn:reload").await);
|
||||
sys.mark_refresh_done("bucket", "arn:reload").await;
|
||||
assert!(!sys.is_reloading_target("bucket", "arn:reload").await);
|
||||
sys.mark_refresh_in_progress("bucket", "arn:reload").await;
|
||||
assert!(sys.is_reloading_target("bucket", "arn:reload").await);
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
async fn update_all_targets_publishes_disable_proxy_on_target_client() {
|
||||
// The read-proxy selector (replication_proxy::get_proxy_targets) skips
|
||||
@@ -3551,6 +4062,88 @@ mod tests {
|
||||
assert!(opted_out.disable_proxy, "disable_proxy must reach the published TargetClient");
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
async fn update_all_targets_keeps_failed_client_placeholder() {
|
||||
let sys = BucketTargetSys::default();
|
||||
let target = BucketTarget {
|
||||
arn: "arn:expired".to_string(),
|
||||
endpoint: "192.168.1.10:9000".to_string(),
|
||||
target_bucket: "target-bucket".to_string(),
|
||||
region: "us-east-1".to_string(),
|
||||
credentials: Some(Credentials {
|
||||
access_key: "access".to_string(),
|
||||
secret_key: "secret".to_string(),
|
||||
session_token: Some("temporary-session-token".to_string()),
|
||||
expiration: Some("2000-01-01T00:00:00Z".parse().expect("expired timestamp should parse")),
|
||||
}),
|
||||
..Default::default()
|
||||
};
|
||||
let targets = BucketTargets { targets: vec![target] };
|
||||
|
||||
sys.update_all_targets("bucket", Some(&targets)).await;
|
||||
|
||||
let remotes = sys.arn_remotes_map.read().await;
|
||||
let placeholder = remotes
|
||||
.get("arn:expired")
|
||||
.expect("configured target should retain a cache entry");
|
||||
assert!(placeholder.client.is_none());
|
||||
assert!(OffsetDateTime::now_utc() - placeholder.last_refresh < Duration::from_secs(5));
|
||||
drop(remotes);
|
||||
assert!(sys.get_remote_target_client("bucket", "arn:expired").await.is_none());
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
async fn credential_rotation_atomically_replaces_published_client() {
|
||||
let sys = BucketTargetSys::default();
|
||||
let target = |session_token: &str| BucketTarget {
|
||||
arn: "arn:rotating".to_string(),
|
||||
endpoint: "192.168.1.10:9000".to_string(),
|
||||
target_bucket: "target-bucket".to_string(),
|
||||
region: "us-east-1".to_string(),
|
||||
credentials: Some(Credentials {
|
||||
access_key: "access".to_string(),
|
||||
secret_key: "secret".to_string(),
|
||||
session_token: Some(session_token.to_string()),
|
||||
expiration: None,
|
||||
}),
|
||||
..Default::default()
|
||||
};
|
||||
|
||||
sys.update_all_targets(
|
||||
"bucket",
|
||||
Some(&BucketTargets {
|
||||
targets: vec![target("old-session-token")],
|
||||
}),
|
||||
)
|
||||
.await;
|
||||
let old_client = sys
|
||||
.get_remote_target_client("bucket", "arn:rotating")
|
||||
.await
|
||||
.expect("initial client should be published");
|
||||
|
||||
sys.update_all_targets(
|
||||
"bucket",
|
||||
Some(&BucketTargets {
|
||||
targets: vec![target("new-session-token")],
|
||||
}),
|
||||
)
|
||||
.await;
|
||||
let new_client = sys
|
||||
.get_remote_target_client("bucket", "arn:rotating")
|
||||
.await
|
||||
.expect("rotated client should be published");
|
||||
|
||||
assert!(!Arc::ptr_eq(&old_client, &new_client));
|
||||
assert_eq!(
|
||||
old_client.credentials.as_ref().and_then(Credentials::effective_session_token),
|
||||
Some("old-session-token")
|
||||
);
|
||||
assert_eq!(
|
||||
new_client.credentials.as_ref().and_then(Credentials::effective_session_token),
|
||||
Some("new-session-token")
|
||||
);
|
||||
}
|
||||
|
||||
#[tokio::test(flavor = "multi_thread", worker_threads = 2)]
|
||||
async fn target_updates_serialize_client_build_through_publication_per_bucket() {
|
||||
let sys = Arc::new(BucketTargetSys::default());
|
||||
|
||||
@@ -41,7 +41,7 @@ use crate::bucket::lifecycle::tier_free_version_recovery::{
|
||||
DEFAULT_FREE_VERSION_RECOVERY_LIMIT, FreeVersionRecoveryStats, recover_tier_free_versions_with_cancel,
|
||||
};
|
||||
use crate::bucket::lifecycle::tier_last_day_stats::{DailyAllTierStats, LastDayTierStats};
|
||||
use crate::bucket::lifecycle::tier_sweeper::{Jentry, delete_object_from_remote_tier_idempotent_with_manager_and_identity};
|
||||
use crate::bucket::lifecycle::tier_sweeper::{Jentry, delete_object_from_remote_tier_with_lease_idempotent};
|
||||
use crate::bucket::lifecycle::transition_transaction::run_transition_transaction_recovery_loop;
|
||||
use crate::bucket::object_lock::ObjectLockApi;
|
||||
use crate::bucket::versioning::VersioningApi as _;
|
||||
@@ -50,7 +50,10 @@ use crate::disk::error::DiskError;
|
||||
use crate::disk::{DeleteOptions, Disk, DiskAPI, RUSTFS_META_BUCKET, RUSTFS_META_MULTIPART_BUCKET, STORAGE_FORMAT_FILE};
|
||||
use crate::error::Error;
|
||||
use crate::error::StorageError;
|
||||
use crate::error::{is_err_object_not_found, is_err_read_quorum, is_err_version_not_found, is_network_or_host_down};
|
||||
use crate::error::{
|
||||
is_err_object_not_found, is_err_read_quorum, is_err_strict_volume_not_found, is_err_version_not_found,
|
||||
is_network_or_host_down,
|
||||
};
|
||||
use crate::object_api::{GetObjectReader, ObjectInfo, ObjectOptions};
|
||||
use crate::object_api::{ObjectEncryptionResolver, ReadPlan};
|
||||
use crate::services::tier::{
|
||||
@@ -586,25 +589,217 @@ impl ExpiryOp for FreeVersionTask {
|
||||
}
|
||||
}
|
||||
|
||||
async fn delete_free_version_remote_object(
|
||||
async fn acquire_free_version_tier_lease(
|
||||
oi: &ObjectInfo,
|
||||
tier_config_mgr: &Arc<RwLock<TierConfigMgr>>,
|
||||
) -> Result<(), std::io::Error> {
|
||||
) -> Result<(TierOperationLease, bool), std::io::Error> {
|
||||
let version_id_exact = validate_transition_remote_version(oi)?;
|
||||
let identity = tier_destination_id_from_metadata(&oi.user_defined)?
|
||||
.ok_or_else(|| std::io::Error::other("tier free-version has no durable backend identity"))?;
|
||||
delete_object_from_remote_tier_idempotent_with_manager_and_identity(
|
||||
let lease =
|
||||
TierConfigMgr::acquire_operation_lease_for_backend_identity(tier_config_mgr, &oi.transitioned_object.tier, identity)
|
||||
.await
|
||||
.map_err(std::io::Error::other)?;
|
||||
Ok((lease, version_id_exact))
|
||||
}
|
||||
|
||||
async fn delete_free_version_remote_object_with_lease(
|
||||
oi: &ObjectInfo,
|
||||
lease: &TierOperationLease,
|
||||
version_id_exact: bool,
|
||||
) -> Result<(), std::io::Error> {
|
||||
delete_object_from_remote_tier_with_lease_idempotent(
|
||||
&oi.transitioned_object.name,
|
||||
&oi.transitioned_object.version_id,
|
||||
&oi.transitioned_object.tier,
|
||||
identity,
|
||||
tier_config_mgr,
|
||||
lease,
|
||||
version_id_exact,
|
||||
)
|
||||
.await?;
|
||||
Ok(())
|
||||
}
|
||||
|
||||
fn free_version_physical_topology_generation(api: &ECStore) -> String {
|
||||
let mut hasher = Sha256::new();
|
||||
for pool in &api.pools {
|
||||
hasher.update(pool.pool_idx.to_be_bytes());
|
||||
hasher.update(pool.disk_set.len().to_be_bytes());
|
||||
for set in &pool.disk_set {
|
||||
hasher.update(set.set_index.to_be_bytes());
|
||||
}
|
||||
}
|
||||
rustfs_utils::crypto::hex(hasher.finalize().as_slice())
|
||||
}
|
||||
|
||||
fn free_version_remote_tuple_matches(candidate: &ObjectInfo, expected: &ObjectInfo) -> std::io::Result<bool> {
|
||||
if candidate.transitioned_object.tier != expected.transitioned_object.tier
|
||||
|| candidate.transitioned_object.name != expected.transitioned_object.name
|
||||
{
|
||||
return Ok(false);
|
||||
}
|
||||
let candidate_identity = tier_destination_id_from_metadata(&candidate.user_defined)?
|
||||
.ok_or_else(|| std::io::Error::other("tier free-version is missing its backend identity"))?;
|
||||
let expected_identity = tier_destination_id_from_metadata(&expected.user_defined)?
|
||||
.ok_or_else(|| std::io::Error::other("tier free-version task is missing its backend identity"))?;
|
||||
if candidate_identity != expected_identity {
|
||||
return Ok(false);
|
||||
}
|
||||
if candidate.transition_version_state == rustfs_filemeta::TransitionVersionState::Unknown
|
||||
|| expected.transition_version_state == rustfs_filemeta::TransitionVersionState::Unknown
|
||||
{
|
||||
return Err(std::io::Error::new(
|
||||
std::io::ErrorKind::WouldBlock,
|
||||
"tier free-version remote version state is unknown",
|
||||
));
|
||||
}
|
||||
Ok(candidate.transition_version_state == expected.transition_version_state
|
||||
&& candidate.transitioned_object.version_id == expected.transitioned_object.version_id)
|
||||
}
|
||||
|
||||
async fn scan_exact_free_version_targets(
|
||||
api: &ECStore,
|
||||
oi: &ObjectInfo,
|
||||
local_object: &str,
|
||||
) -> std::io::Result<Vec<(Arc<SetDisks>, FileInfo)>> {
|
||||
let mut targets = Vec::new();
|
||||
for pool in &api.pools {
|
||||
for set in &pool.disk_set {
|
||||
let versions = match set.load_file_info_versions_exact(&oi.bucket, &oi.name).await {
|
||||
Ok(Some(versions)) => versions,
|
||||
Ok(None) => continue,
|
||||
Err(err) if is_err_strict_volume_not_found(&err) => continue,
|
||||
Err(err) => return Err(std::io::Error::other(err)),
|
||||
};
|
||||
for version in versions.versions.iter().chain(versions.free_versions.iter()) {
|
||||
let candidate = ObjectInfo::from_file_info(version, &oi.bucket, &oi.name, true);
|
||||
if free_version_remote_tuple_matches(&candidate, oi)? {
|
||||
if candidate.transitioned_object.free_version {
|
||||
// Data movement can leave the same remote tuple in
|
||||
// several physical pools. Ordinary deletion assigns a
|
||||
// fresh local free-version UUID to each copy, but all
|
||||
// of those markers own the same idempotent remote
|
||||
// DELETE. Consume them together while holding every
|
||||
// physical object lock; treating their local UUIDs as
|
||||
// conflicting would strand cleanup forever.
|
||||
let mut actual = version.clone();
|
||||
actual.name = local_object.to_string();
|
||||
targets.push((Arc::clone(set), actual));
|
||||
} else {
|
||||
return Err(std::io::Error::new(
|
||||
std::io::ErrorKind::WouldBlock,
|
||||
"a live transitioned source still references the free-version remote tuple",
|
||||
));
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
Ok(targets)
|
||||
}
|
||||
|
||||
fn free_version_cleanup_fences_current(
|
||||
topology_generation: &str,
|
||||
api: &ECStore,
|
||||
bucket_guard: &rustfs_lock::NamespaceLockGuard,
|
||||
object_guards: &[crate::store::ObjectLockDiagGuard],
|
||||
lease: &TierOperationLease,
|
||||
cancel: &CancellationToken,
|
||||
deadline: tokio::time::Instant,
|
||||
) -> bool {
|
||||
!cancel.is_cancelled()
|
||||
&& tokio::time::Instant::now() < deadline
|
||||
&& !bucket_guard.is_lock_lost()
|
||||
&& object_guards.iter().all(|guard| !guard.is_lock_lost())
|
||||
&& lease.is_current_generation()
|
||||
&& free_version_physical_topology_generation(api) == topology_generation
|
||||
}
|
||||
|
||||
async fn cleanup_free_version_exact(api: Arc<ECStore>, oi: &ObjectInfo, cancel: &CancellationToken) -> std::io::Result<bool> {
|
||||
const FREE_VERSION_REMOTE_DEADLINE: StdDuration = StdDuration::from_secs(30);
|
||||
|
||||
let topology_generation = free_version_physical_topology_generation(&api);
|
||||
let bucket_guard = api
|
||||
.acquire_bucket_lifecycle_read_lock(&oi.bucket)
|
||||
.await
|
||||
.map_err(std::io::Error::other)?;
|
||||
let (lease, version_id_exact) = acquire_free_version_tier_lease(oi, &api.tier_config_mgr()).await?;
|
||||
let local_object = encode_dir_object(&oi.name);
|
||||
let object_guards = api
|
||||
.acquire_all_physical_object_write_locks("tier_free_version_cleanup", &oi.bucket, &local_object)
|
||||
.await
|
||||
.map_err(std::io::Error::other)?;
|
||||
let targets = scan_exact_free_version_targets(&api, oi, &local_object).await?;
|
||||
if targets.is_empty() {
|
||||
return Ok(false);
|
||||
}
|
||||
|
||||
let deadline = tokio::time::Instant::now() + FREE_VERSION_REMOTE_DEADLINE;
|
||||
if !free_version_cleanup_fences_current(&topology_generation, &api, &bucket_guard, &object_guards, &lease, cancel, deadline) {
|
||||
return Err(std::io::Error::new(
|
||||
std::io::ErrorKind::WouldBlock,
|
||||
"tier free-version cleanup fence is invalid before remote delete",
|
||||
));
|
||||
}
|
||||
tokio::select! {
|
||||
_ = cancel.cancelled() => {
|
||||
return Err(std::io::Error::new(std::io::ErrorKind::Interrupted, "tier free-version cleanup was cancelled"));
|
||||
}
|
||||
result = tokio::time::timeout_at(
|
||||
deadline,
|
||||
delete_free_version_remote_object_with_lease(oi, &lease, version_id_exact),
|
||||
) => {
|
||||
result
|
||||
.map_err(|_| std::io::Error::new(std::io::ErrorKind::TimedOut, "tier free-version remote delete timed out"))??;
|
||||
}
|
||||
}
|
||||
if !free_version_cleanup_fences_current(&topology_generation, &api, &bucket_guard, &object_guards, &lease, cancel, deadline) {
|
||||
// Remote DELETE is idempotent, but a changed fence makes the local
|
||||
// outcome ambiguous. Keep every marker for a fully fenced retry.
|
||||
return Err(std::io::Error::new(
|
||||
std::io::ErrorKind::WouldBlock,
|
||||
"tier free-version cleanup fence changed after remote delete",
|
||||
));
|
||||
}
|
||||
|
||||
let mut first_error = None;
|
||||
for (set, actual) in &targets {
|
||||
let mut delete_request = FileInfo {
|
||||
name: local_object.clone(),
|
||||
version_id: actual.version_id,
|
||||
..Default::default()
|
||||
};
|
||||
delete_request.set_tier_free_version();
|
||||
if let Err(err) = set
|
||||
.delete_object_version(&oi.bucket, &local_object, &delete_request, false)
|
||||
.await
|
||||
&& first_error.is_none()
|
||||
{
|
||||
first_error = Some(std::io::Error::other(err));
|
||||
}
|
||||
}
|
||||
let remaining = scan_exact_free_version_targets(&api, oi, &local_object).await?;
|
||||
if !remaining.is_empty() {
|
||||
return Err(first_error.unwrap_or_else(|| {
|
||||
std::io::Error::new(
|
||||
std::io::ErrorKind::WouldBlock,
|
||||
"tier free-version cleanup remained on at least one physical set",
|
||||
)
|
||||
}));
|
||||
}
|
||||
if let Some(err) = first_error {
|
||||
return Err(err);
|
||||
}
|
||||
Ok(true)
|
||||
}
|
||||
|
||||
#[cfg(all(test, feature = "test-util"))]
|
||||
async fn delete_free_version_remote_object(
|
||||
oi: &ObjectInfo,
|
||||
tier_config_mgr: &Arc<RwLock<TierConfigMgr>>,
|
||||
) -> Result<(), std::io::Error> {
|
||||
let (lease, version_id_exact) = acquire_free_version_tier_lease(oi, tier_config_mgr).await?;
|
||||
delete_free_version_remote_object_with_lease(oi, &lease, version_id_exact).await
|
||||
}
|
||||
|
||||
#[allow(
|
||||
dead_code,
|
||||
reason = "MinIO-parity tier/lifecycle entry point that this port never wired (backlog#1823)"
|
||||
@@ -618,8 +813,11 @@ where
|
||||
F: FnOnce() -> Fut,
|
||||
Fut: std::future::Future<Output = T>,
|
||||
{
|
||||
delete_free_version_remote_object(oi, tier_config_mgr).await?;
|
||||
Ok(delete_local().await)
|
||||
let (lease, version_id_exact) = acquire_free_version_tier_lease(oi, tier_config_mgr).await?;
|
||||
delete_free_version_remote_object_with_lease(oi, &lease, version_id_exact).await?;
|
||||
let result = delete_local().await;
|
||||
drop(lease);
|
||||
Ok(result)
|
||||
}
|
||||
|
||||
struct NewerNoncurrentTask {
|
||||
@@ -690,6 +888,10 @@ impl ExpiryState {
|
||||
usize::try_from(self.stats.pending_tasks().max(0)).unwrap_or(usize::MAX)
|
||||
}
|
||||
|
||||
pub fn active_tasks(&self) -> usize {
|
||||
usize::try_from(self.stats.active_tasks().max(0)).unwrap_or(usize::MAX)
|
||||
}
|
||||
|
||||
fn send_expiry_task(&self, wrkr: Sender<Option<ExpiryOpType>>, task: ExpiryOpType) -> bool {
|
||||
let queued = wrkr.try_send(Some(task)).is_ok();
|
||||
if queued {
|
||||
@@ -826,7 +1028,7 @@ impl ExpiryState {
|
||||
}
|
||||
|
||||
pub async fn resize_workers(n: usize, api: Arc<ECStore>) {
|
||||
let expiry_state = runtime_sources::expiry_state_handle();
|
||||
let expiry_state = api.ctx.expiry_state();
|
||||
if n == expiry_state.read().await.tasks_tx.len() || n < 1 {
|
||||
return;
|
||||
}
|
||||
@@ -867,7 +1069,7 @@ impl ExpiryState {
|
||||
stats: Arc<ExpiryStats>,
|
||||
recovery_notify: Arc<Notify>,
|
||||
) {
|
||||
let cancel_token = runtime_sources::background_services_cancel_token().unwrap_or_else(|| {
|
||||
let cancel_token = api.ctx.background_cancel_token().unwrap_or_else(|| {
|
||||
static FALLBACK: std::sync::OnceLock<tokio_util::sync::CancellationToken> = std::sync::OnceLock::new();
|
||||
FALLBACK.get_or_init(tokio_util::sync::CancellationToken::new).clone()
|
||||
});
|
||||
@@ -968,119 +1170,33 @@ impl ExpiryState {
|
||||
else if v.as_any().is::<FreeVersionTask>() {
|
||||
let v = v.as_any().downcast_ref::<FreeVersionTask>().expect("FreeVersionTask downcast failed");
|
||||
let oi = v.0.clone();
|
||||
if let Err(err) = delete_free_version_remote_object(&oi, &api.tier_config_mgr()).await {
|
||||
recovery_notify.notify_one();
|
||||
debug!(
|
||||
bucket = %oi.bucket,
|
||||
object = %oi.name,
|
||||
remote_object = %oi.transitioned_object.name,
|
||||
remote_version_id = %oi.transitioned_object.version_id,
|
||||
tier = %oi.transitioned_object.tier,
|
||||
error = ?err,
|
||||
event = EVENT_LIFECYCLE_WORKER_STATE,
|
||||
component = LOG_COMPONENT_ECSTORE,
|
||||
subsystem = LOG_SUBSYSTEM_LIFECYCLE,
|
||||
reason = "remote_tier_delete_failed",
|
||||
"Lifecycle worker skipped remote tier delete"
|
||||
);
|
||||
continue;
|
||||
}
|
||||
|
||||
let local_object = encode_dir_object(&oi.name);
|
||||
let mut fi = FileInfo {
|
||||
name: local_object.clone(),
|
||||
version_id: oi.version_id,
|
||||
..Default::default()
|
||||
};
|
||||
// This removes an existing internal cleanup marker. Keeping
|
||||
// `deleted` false makes duplicate tasks return not-found
|
||||
// instead of creating an ordinary delete marker.
|
||||
fi.set_tier_free_version();
|
||||
|
||||
let mut deleted_locally = false;
|
||||
for pool in &api.pools {
|
||||
let set = pool.get_disks_by_key(&local_object);
|
||||
let ns_lock = match set.new_ns_lock(&oi.bucket, &local_object).await {
|
||||
Ok(lock) => lock,
|
||||
Err(err) => {
|
||||
recovery_notify.notify_one();
|
||||
debug!(
|
||||
event = EVENT_LIFECYCLE_WORKER_STATE,
|
||||
component = LOG_COMPONENT_ECSTORE,
|
||||
subsystem = LOG_SUBSYSTEM_LIFECYCLE,
|
||||
bucket = %oi.bucket,
|
||||
object = %oi.name,
|
||||
pool_index = pool.pool_idx,
|
||||
set_index = set.set_index,
|
||||
error = ?err,
|
||||
reason = "local_free_version_lock_failed",
|
||||
"Lifecycle worker failed to create local free-version cleanup lock"
|
||||
);
|
||||
continue;
|
||||
}
|
||||
};
|
||||
let _object_lock_guard =
|
||||
match ns_lock.get_write_lock_quiet(get_lock_acquire_timeout()).await {
|
||||
Ok(guard) => guard,
|
||||
Err(err) => {
|
||||
recovery_notify.notify_one();
|
||||
debug!(
|
||||
event = EVENT_LIFECYCLE_WORKER_STATE,
|
||||
component = LOG_COMPONENT_ECSTORE,
|
||||
subsystem = LOG_SUBSYSTEM_LIFECYCLE,
|
||||
bucket = %oi.bucket,
|
||||
object = %oi.name,
|
||||
pool_index = pool.pool_idx,
|
||||
set_index = set.set_index,
|
||||
error = ?err,
|
||||
reason = "local_free_version_lock_failed",
|
||||
"Lifecycle worker failed to acquire local free-version cleanup lock"
|
||||
);
|
||||
continue;
|
||||
}
|
||||
};
|
||||
match set
|
||||
.delete_object_version(&oi.bucket, &local_object, &fi, false)
|
||||
.await
|
||||
{
|
||||
Ok(()) => {
|
||||
deleted_locally = true;
|
||||
break;
|
||||
}
|
||||
Err(err) if is_err_version_not_found(&err) || is_err_object_not_found(&err) => continue,
|
||||
Err(err) => {
|
||||
recovery_notify.notify_one();
|
||||
debug!(
|
||||
event = EVENT_LIFECYCLE_WORKER_STATE,
|
||||
component = LOG_COMPONENT_ECSTORE,
|
||||
subsystem = LOG_SUBSYSTEM_LIFECYCLE,
|
||||
bucket = %oi.bucket,
|
||||
object = %oi.name,
|
||||
remote_object = %oi.transitioned_object.name,
|
||||
remote_version_id = %oi.transitioned_object.version_id,
|
||||
tier = %oi.transitioned_object.tier,
|
||||
error = ?err,
|
||||
reason = "local_free_version_delete_failed",
|
||||
"Lifecycle worker failed local free-version cleanup"
|
||||
);
|
||||
break;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
if !deleted_locally {
|
||||
debug!(
|
||||
match cleanup_free_version_exact(api.clone(), &oi, &cancel_token).await {
|
||||
Ok(true) => {}
|
||||
Ok(false) => debug!(
|
||||
event = EVENT_LIFECYCLE_WORKER_STATE,
|
||||
component = LOG_COMPONENT_ECSTORE,
|
||||
subsystem = LOG_SUBSYSTEM_LIFECYCLE,
|
||||
bucket = %oi.bucket,
|
||||
object = %oi.name,
|
||||
remote_object = %oi.transitioned_object.name,
|
||||
remote_version_id = %oi.transitioned_object.version_id,
|
||||
tier = %oi.transitioned_object.tier,
|
||||
reason = "local_free_version_missing",
|
||||
"Lifecycle worker could not find transitioned free version locally"
|
||||
);
|
||||
"Lifecycle worker found that the exact free-version was already absent"
|
||||
),
|
||||
Err(err) => {
|
||||
recovery_notify.notify_one();
|
||||
debug!(
|
||||
event = EVENT_LIFECYCLE_WORKER_STATE,
|
||||
component = LOG_COMPONENT_ECSTORE,
|
||||
subsystem = LOG_SUBSYSTEM_LIFECYCLE,
|
||||
bucket = %oi.bucket,
|
||||
object = %oi.name,
|
||||
remote_object = %oi.transitioned_object.name,
|
||||
remote_version_id = %oi.transitioned_object.version_id,
|
||||
tier = %oi.transitioned_object.tier,
|
||||
error = ?err,
|
||||
reason = "free_version_exact_cleanup_deferred",
|
||||
"Lifecycle worker retained the exact free-version for a fenced retry"
|
||||
);
|
||||
}
|
||||
}
|
||||
}
|
||||
else {
|
||||
@@ -1152,8 +1268,8 @@ fn set_recovered_free_version_enqueue_observer(
|
||||
RecoveredFreeVersionEnqueueObserverGuard
|
||||
}
|
||||
|
||||
pub async fn enqueue_recovered_free_version(oi: ObjectInfo) -> bool {
|
||||
let expiry_state = runtime_sources::expiry_state_handle();
|
||||
pub async fn enqueue_recovered_free_version(api: &ECStore, oi: ObjectInfo) -> bool {
|
||||
let expiry_state = api.ctx.expiry_state();
|
||||
let queued = enqueue_recovered_free_version_with_state(&expiry_state, oi).await;
|
||||
|
||||
#[cfg(test)]
|
||||
@@ -2580,8 +2696,8 @@ fn spawn_tier_free_version_recovery_once(api: Arc<ECStore>, started: &OnceLock<(
|
||||
}
|
||||
|
||||
Some(tokio::spawn(async move {
|
||||
let cancel_token = runtime_sources::background_services_cancel_token().unwrap_or_default();
|
||||
let expiry_state = runtime_sources::expiry_state_handle();
|
||||
let cancel_token = api.ctx.background_cancel_token().unwrap_or_default();
|
||||
let expiry_state = api.ctx.expiry_state();
|
||||
run_tier_free_version_recovery_loop(
|
||||
cancel_token,
|
||||
expiry_state,
|
||||
@@ -6229,9 +6345,18 @@ mod tests {
|
||||
rustfs_utils::crypto::hex(old_identity),
|
||||
);
|
||||
oi.user_defined = Arc::new(metadata.clone());
|
||||
let lease_observed_during_local_delete = Arc::new(std::sync::atomic::AtomicBool::new(false));
|
||||
delete_free_version_remote_object_then(&oi, &manager, {
|
||||
let local_delete_calls = Arc::clone(&local_delete_calls);
|
||||
let lease_observed_during_local_delete = Arc::clone(&lease_observed_during_local_delete);
|
||||
let manager = manager.clone();
|
||||
move || async move {
|
||||
assert_eq!(
|
||||
crate::services::tier::tier::TierConfigMgr::active_operation_lease_count(&manager, "WARM").await,
|
||||
1,
|
||||
"the identity-bound tier lease must span the exact local marker delete"
|
||||
);
|
||||
lease_observed_during_local_delete.store(true, Ordering::Relaxed);
|
||||
local_delete_calls.fetch_add(1, Ordering::Relaxed);
|
||||
}
|
||||
})
|
||||
@@ -6239,6 +6364,12 @@ mod tests {
|
||||
.expect("matching destination identity should allow idempotent remote cleanup");
|
||||
assert_eq!(old_backend.remove_count().await, 1);
|
||||
assert_eq!(local_delete_calls.load(Ordering::Relaxed), 1);
|
||||
assert!(lease_observed_during_local_delete.load(Ordering::Relaxed));
|
||||
assert_eq!(
|
||||
crate::services::tier::tier::TierConfigMgr::active_operation_lease_count(&manager, "WARM").await,
|
||||
0,
|
||||
"the tier lease should be released after the local marker delete completes"
|
||||
);
|
||||
|
||||
let mut single_prefix_metadata = HashMap::new();
|
||||
single_prefix_metadata.insert(
|
||||
@@ -6472,6 +6603,7 @@ mod tests {
|
||||
let state = ExpiryState::new();
|
||||
let mut state = state.write().await;
|
||||
let je = Jentry {
|
||||
persisted_version: 0,
|
||||
obj_name: "remote/object".to_string(),
|
||||
version_id: "remote-version".to_string(),
|
||||
tier_name: "WARM".to_string(),
|
||||
@@ -6480,6 +6612,7 @@ mod tests {
|
||||
version_state: rustfs_filemeta::TransitionVersionState::Exact,
|
||||
state: crate::bucket::lifecycle::tier_sweeper::TierDeleteJournalState::Committed,
|
||||
source: None,
|
||||
dispatch: None,
|
||||
};
|
||||
|
||||
let err = state
|
||||
@@ -6620,6 +6753,7 @@ mod tests {
|
||||
let state = ExpiryState::new_with_unconsumed_worker_channel(1);
|
||||
let mut state = state.write().await;
|
||||
let je = Jentry {
|
||||
persisted_version: 0,
|
||||
obj_name: "remote/object".to_string(),
|
||||
version_id: "remote-version".to_string(),
|
||||
tier_name: "WARM".to_string(),
|
||||
@@ -6628,6 +6762,7 @@ mod tests {
|
||||
version_state: rustfs_filemeta::TransitionVersionState::Exact,
|
||||
state: crate::bucket::lifecycle::tier_sweeper::TierDeleteJournalState::Committed,
|
||||
source: None,
|
||||
dispatch: None,
|
||||
};
|
||||
|
||||
state
|
||||
@@ -6759,7 +6894,7 @@ mod tests {
|
||||
};
|
||||
|
||||
assert!(
|
||||
super::enqueue_recovered_free_version(oi).await,
|
||||
super::enqueue_recovered_free_version(&ecstore, oi).await,
|
||||
"the resized production worker queue should accept the task"
|
||||
);
|
||||
stop_tx.send(None).await.expect("worker stop signal should be delivered");
|
||||
@@ -6875,12 +7010,12 @@ mod tests {
|
||||
.await
|
||||
.expect("free-version task should reach the worker");
|
||||
tokio::time::timeout(StdDuration::from_secs(30), async {
|
||||
while remote_backend.remove_count().await == 0 {
|
||||
while stats.active_tasks() == 0 {
|
||||
tokio::task::yield_now().await;
|
||||
}
|
||||
})
|
||||
.await
|
||||
.expect("worker should complete remote cleanup before taking the local lock");
|
||||
.expect("worker should mark the cleanup task active before the lock assertion");
|
||||
let completed_while_locked = tokio::time::timeout(StdDuration::from_millis(100), async {
|
||||
while stats.active_tasks() != 0 {
|
||||
tokio::task::yield_now().await;
|
||||
@@ -6889,7 +7024,12 @@ mod tests {
|
||||
.await;
|
||||
assert!(
|
||||
completed_while_locked.is_err(),
|
||||
"local cleanup must wait while a competing object writer owns the namespace lock"
|
||||
"the cleanup task must wait while a competing object writer owns the namespace lock"
|
||||
);
|
||||
assert_eq!(
|
||||
remote_backend.remove_count().await,
|
||||
0,
|
||||
"the remote tuple must not be deleted before the all-physical namespace fence is acquired"
|
||||
);
|
||||
for disk_path in &disk_paths {
|
||||
assert!(
|
||||
@@ -6900,6 +7040,13 @@ mod tests {
|
||||
}
|
||||
|
||||
drop(object_lock_guard);
|
||||
tokio::time::timeout(StdDuration::from_secs(30), async {
|
||||
while remote_backend.remove_count().await == 0 {
|
||||
tokio::task::yield_now().await;
|
||||
}
|
||||
})
|
||||
.await
|
||||
.expect("worker should delete the remote tuple after acquiring the released namespace fence");
|
||||
tx.send(None).await.expect("worker stop signal should be delivered");
|
||||
worker.await.expect("free-version worker should stop cleanly");
|
||||
|
||||
@@ -6995,6 +7142,7 @@ mod tests {
|
||||
.next()
|
||||
.expect("seeded free version should be recoverable");
|
||||
let stale_version_id = oi.version_id.expect("free version should have a concrete UUID");
|
||||
let ordinary_marker_mod_time = OffsetDateTime::now_utc();
|
||||
|
||||
for disk_path in &disk_paths {
|
||||
let metadata_path = disk_path.join(&bucket).join(object).join(STORAGE_FORMAT_FILE);
|
||||
@@ -7017,7 +7165,7 @@ mod tests {
|
||||
name: object.to_string(),
|
||||
version_id: Some(stale_version_id),
|
||||
deleted: true,
|
||||
mod_time: Some(OffsetDateTime::now_utc()),
|
||||
mod_time: Some(ordinary_marker_mod_time),
|
||||
..Default::default()
|
||||
})
|
||||
.expect("same-ID ordinary marker should replace the stale free version");
|
||||
@@ -7031,6 +7179,13 @@ mod tests {
|
||||
.expect("same-ID ordinary marker metadata should be written");
|
||||
}
|
||||
|
||||
assert!(
|
||||
!super::cleanup_free_version_exact(Arc::clone(&ecstore), &oi, &CancellationToken::new())
|
||||
.await
|
||||
.expect("a stale task whose local UUID now names an ordinary marker should be an idempotent no-op"),
|
||||
"the stale free-version task must not report local cleanup"
|
||||
);
|
||||
|
||||
let state = ExpiryState::new();
|
||||
let (stats, recovery_notify) = {
|
||||
let state = state.read().await;
|
||||
@@ -11522,7 +11677,7 @@ mod tests {
|
||||
|
||||
#[cfg(feature = "test-util")]
|
||||
#[tokio::test]
|
||||
async fn journal_replay_rejects_unknown_version_state_before_backend_io() {
|
||||
async fn journal_replay_quarantines_legacy_unknown_version_state_before_backend_io() {
|
||||
let (_disk_paths, ecstore) = setup_test_env().await;
|
||||
let (backend, _) = register_recovery_mock_tier(&ecstore).await;
|
||||
let identity = TierConfigMgr::acquire_operation_lease(&ecstore.tier_config_mgr(), "WARM")
|
||||
@@ -11530,6 +11685,7 @@ mod tests {
|
||||
.expect("mock tier lease should be available")
|
||||
.backend_identity();
|
||||
let je = Jentry {
|
||||
persisted_version: 0,
|
||||
obj_name: "remote/object".to_string(),
|
||||
version_id: "legacy-version".to_string(),
|
||||
tier_name: "WARM".to_string(),
|
||||
@@ -11538,25 +11694,28 @@ mod tests {
|
||||
version_state: rustfs_filemeta::TransitionVersionState::Unknown,
|
||||
state: crate::bucket::lifecycle::tier_sweeper::TierDeleteJournalState::Committed,
|
||||
source: None,
|
||||
dispatch: None,
|
||||
};
|
||||
|
||||
crate::bucket::lifecycle::tier_delete_journal::persist_tier_delete_journal_entry(ecstore.clone(), &je)
|
||||
.await
|
||||
.expect("legacy unknown journal should remain byte-compatible and persistable");
|
||||
let err = crate::bucket::lifecycle::tier_delete_journal::process_tier_delete_journal_entry(ecstore, &je)
|
||||
.await
|
||||
.expect_err("unknown journal state must fail before backend IO");
|
||||
.expect_err("legacy unknown journal must be quarantined before backend IO");
|
||||
|
||||
assert_eq!(err.kind(), std::io::ErrorKind::InvalidData);
|
||||
assert_eq!(err.kind(), std::io::ErrorKind::WouldBlock);
|
||||
assert_eq!(backend.remove_count().await, 0);
|
||||
}
|
||||
|
||||
#[cfg(feature = "test-util")]
|
||||
#[tokio::test]
|
||||
async fn journal_replay_deletes_confirmed_exact_provider_token() {
|
||||
async fn rejected_upload_cleanup_retries_confirmed_exact_provider_token_without_legacy_journal() {
|
||||
let (_disk_paths, ecstore) = setup_test_env().await;
|
||||
let (backend, _) = register_recovery_mock_tier(&ecstore).await;
|
||||
let lease = TierConfigMgr::acquire_operation_lease(&ecstore.tier_config_mgr(), "WARM")
|
||||
.await
|
||||
.expect("mock tier lease should be available");
|
||||
let identity = lease.backend_identity();
|
||||
backend
|
||||
.set_put_remote_version(Some("provider-version-token".to_string()))
|
||||
.await;
|
||||
@@ -11570,34 +11729,30 @@ mod tests {
|
||||
.expect("confirmed remote candidate should be seeded");
|
||||
backend.set_remove_failure(true);
|
||||
backend.set_reject_non_empty_remote_versions(true);
|
||||
let je = Jentry {
|
||||
obj_name: "remote/object".to_string(),
|
||||
version_id: "provider-version-token".to_string(),
|
||||
tier_name: "WARM".to_string(),
|
||||
backend_identity: Some(identity),
|
||||
version_id_exact: true,
|
||||
version_state: rustfs_filemeta::TransitionVersionState::Exact,
|
||||
state: crate::bucket::lifecycle::tier_sweeper::TierDeleteJournalState::Committed,
|
||||
source: None,
|
||||
};
|
||||
|
||||
crate::set_disk::cleanup_rejected_transition_upload_durably(
|
||||
let err = crate::set_disk::cleanup_rejected_transition_upload_durably(
|
||||
&lease,
|
||||
&je.obj_name,
|
||||
&je.version_id,
|
||||
"remote/object",
|
||||
"provider-version-token",
|
||||
true,
|
||||
Some(ecstore.clone()),
|
||||
)
|
||||
.await
|
||||
.expect("failed immediate cleanup should remain durable in the journal");
|
||||
assert!(backend.contains(&je.obj_name).await);
|
||||
.expect_err("a failed immediate cleanup must remain owned by the caller's transition transaction");
|
||||
assert_eq!(err.kind(), std::io::ErrorKind::Other);
|
||||
assert!(backend.contains("remote/object").await);
|
||||
|
||||
backend.set_remove_failure(false);
|
||||
crate::bucket::lifecycle::tier_delete_journal::process_tier_delete_journal_entry(ecstore, &je)
|
||||
.await
|
||||
.expect("identity-bound exact journal must retry confirmed candidate cleanup");
|
||||
crate::set_disk::cleanup_rejected_transition_upload_durably(
|
||||
&lease,
|
||||
"remote/object",
|
||||
"provider-version-token",
|
||||
true,
|
||||
Some(ecstore),
|
||||
)
|
||||
.await
|
||||
.expect("the transaction retry must delete the same confirmed candidate");
|
||||
|
||||
assert!(!backend.contains(&je.obj_name).await);
|
||||
assert!(!backend.contains("remote/object").await);
|
||||
assert_eq!(backend.exact_remove_count(), 2);
|
||||
assert_eq!(
|
||||
backend.remove_versions().await,
|
||||
@@ -11760,11 +11915,14 @@ mod tests {
|
||||
};
|
||||
let mut recovery_rx = recovery_rx.lock().await;
|
||||
assert!(
|
||||
super::enqueue_recovered_free_version(ObjectInfo {
|
||||
bucket: "prefill".to_string(),
|
||||
name: "prefill".to_string(),
|
||||
..Default::default()
|
||||
})
|
||||
super::enqueue_recovered_free_version(
|
||||
&ecstore,
|
||||
ObjectInfo {
|
||||
bucket: "prefill".to_string(),
|
||||
name: "prefill".to_string(),
|
||||
..Default::default()
|
||||
},
|
||||
)
|
||||
.await,
|
||||
"the production recovery queue should accept its first task"
|
||||
);
|
||||
@@ -12195,7 +12353,7 @@ mod tests {
|
||||
#[tokio::test]
|
||||
#[serial]
|
||||
async fn tier_free_version_recovery_continues_after_deleted_marker_bucket() {
|
||||
let (_paths, ecstore) = setup_test_env().await;
|
||||
let (disk_paths, ecstore) = setup_test_env().await;
|
||||
let suffix = Uuid::new_v4().simple();
|
||||
let earlier_bucket = format!("zzzz-recovery-{suffix}-a");
|
||||
let deleted_marker = format!("zzzz-recovery-{suffix}-m");
|
||||
@@ -12203,11 +12361,7 @@ mod tests {
|
||||
let later_object = "a-before-stale-marker";
|
||||
create_test_bucket(&ecstore, &earlier_bucket).await;
|
||||
create_test_bucket(&ecstore, &later_bucket).await;
|
||||
let mut reader = PutObjReader::from_vec(b"cursor reset probe".to_vec());
|
||||
ecstore
|
||||
.put_object(&later_bucket, later_object, &mut reader, &ObjectOptions::default())
|
||||
.await
|
||||
.expect("successor bucket object should be created");
|
||||
seed_recoverable_free_version(&disk_paths, &later_bucket, later_object, None, None).await;
|
||||
|
||||
let page = list_tier_free_versions(
|
||||
Arc::clone(&ecstore),
|
||||
@@ -12220,14 +12374,10 @@ mod tests {
|
||||
.expect("recovery should resume at the first bucket after a deleted marker bucket");
|
||||
|
||||
assert_eq!(page.buckets_scanned, 1, "the later bucket must not be skipped");
|
||||
assert_eq!(
|
||||
page.scanned_entries, 1,
|
||||
"the deleted bucket's object marker must not skip objects in the successor bucket"
|
||||
);
|
||||
ecstore
|
||||
.delete_object(&later_bucket, later_object, ObjectOptions::default())
|
||||
.await
|
||||
.expect("successor bucket object should be removed");
|
||||
assert_eq!(page.items.len(), 1, "the successor bucket's recoverable object must be returned");
|
||||
assert_eq!(page.items[0].bucket, later_bucket);
|
||||
assert_eq!(page.items[0].name, later_object);
|
||||
remove_seeded_free_version(&disk_paths, &later_bucket, later_object).await;
|
||||
for bucket in [&earlier_bucket, &later_bucket] {
|
||||
ecstore
|
||||
.delete_bucket(bucket, &DeleteBucketOptions::default())
|
||||
|
||||
@@ -33,6 +33,7 @@ const MANUAL_TRANSITION_CURSOR_MARKER_PROOF_MAX_SIZE: usize = 1024;
|
||||
#[derive(Debug, Clone, Copy, PartialEq, Eq)]
|
||||
pub(crate) enum DurableIlmRecordKind {
|
||||
TierDeleteJournal,
|
||||
TierDeleteDispatchManifest,
|
||||
TransitionTransaction,
|
||||
ManualTransitionJob,
|
||||
ManualTransitionScope,
|
||||
@@ -54,6 +55,18 @@ pub(crate) const TIER_DELETE_JOURNAL_NAMESPACE: DurableIlmNamespace = DurableIlm
|
||||
max_record_size: 64 * 1024,
|
||||
kind: DurableIlmRecordKind::TierDeleteJournal,
|
||||
};
|
||||
pub(crate) const TIER_DELETE_JOURNAL_V6_NAMESPACE: DurableIlmNamespace = DurableIlmNamespace {
|
||||
name: "tier-delete-journal-v6",
|
||||
prefix: "ilm/tier-delete-journal-v6/",
|
||||
max_record_size: 64 * 1024,
|
||||
kind: DurableIlmRecordKind::TierDeleteJournal,
|
||||
};
|
||||
pub(crate) const TIER_DELETE_DISPATCH_MANIFEST_NAMESPACE: DurableIlmNamespace = DurableIlmNamespace {
|
||||
name: "tier-delete-dispatch-manifest",
|
||||
prefix: tier_delete_journal::TIER_DELETE_DISPATCH_MANIFEST_PREFIX,
|
||||
max_record_size: tier_delete_journal::MAX_TIER_DELETE_DISPATCH_MANIFEST_SIZE,
|
||||
kind: DurableIlmRecordKind::TierDeleteDispatchManifest,
|
||||
};
|
||||
pub(crate) const TRANSITION_TRANSACTION_NAMESPACE: DurableIlmNamespace = DurableIlmNamespace {
|
||||
name: "transition-transaction",
|
||||
prefix: "ilm/transition-transactions/records",
|
||||
@@ -85,8 +98,10 @@ pub(crate) const MANUAL_TRANSITION_WORKER_RESULT_NAMESPACE: DurableIlmNamespace
|
||||
kind: DurableIlmRecordKind::ManualTransitionWorkerResult,
|
||||
};
|
||||
|
||||
pub(crate) const DURABLE_ILM_NAMESPACES: [DurableIlmNamespace; 6] = [
|
||||
pub(crate) const DURABLE_ILM_NAMESPACES: [DurableIlmNamespace; 8] = [
|
||||
TIER_DELETE_JOURNAL_NAMESPACE,
|
||||
TIER_DELETE_JOURNAL_V6_NAMESPACE,
|
||||
TIER_DELETE_DISPATCH_MANIFEST_NAMESPACE,
|
||||
TRANSITION_TRANSACTION_NAMESPACE,
|
||||
MANUAL_TRANSITION_JOB_NAMESPACE,
|
||||
MANUAL_TRANSITION_SCOPE_NAMESPACE,
|
||||
@@ -157,6 +172,15 @@ pub(crate) enum DurableIlmRecordCheckpoint {
|
||||
content_sha256: String,
|
||||
identity_sha256: String,
|
||||
committed: bool,
|
||||
#[serde(default, skip_serializing_if = "Option::is_none")]
|
||||
dispatch_identity_sha256: Option<String>,
|
||||
#[serde(default, skip_serializing_if = "Option::is_none")]
|
||||
state: Option<super::tier_sweeper::TierDeleteJournalState>,
|
||||
},
|
||||
TierDeleteDispatchManifest {
|
||||
content_sha256: String,
|
||||
identity_sha256: String,
|
||||
state: tier_delete_journal::TierDeleteDispatchManifestState,
|
||||
},
|
||||
TransitionTransaction {
|
||||
content_sha256: String,
|
||||
@@ -195,6 +219,7 @@ impl DurableIlmRecordCheckpoint {
|
||||
pub(crate) fn content_sha256(&self) -> &str {
|
||||
match self {
|
||||
Self::TierDeleteJournal { content_sha256, .. }
|
||||
| Self::TierDeleteDispatchManifest { content_sha256, .. }
|
||||
| Self::TransitionTransaction { content_sha256, .. }
|
||||
| Self::ManualTransitionJob { content_sha256, .. }
|
||||
| Self::ManualTransitionScope { content_sha256, .. }
|
||||
@@ -228,6 +253,19 @@ impl DurableIlmRecordCheckpoint {
|
||||
}
|
||||
|
||||
pub(crate) fn validate_successor(&self, next: &Self) -> Result<()> {
|
||||
for checkpoint in [self, next] {
|
||||
if let Self::TierDeleteJournal {
|
||||
committed,
|
||||
dispatch_identity_sha256,
|
||||
state,
|
||||
..
|
||||
} = checkpoint
|
||||
&& (state.is_some() != dispatch_identity_sha256.is_some()
|
||||
|| state.is_some_and(|state| *committed != (state == super::tier_sweeper::TierDeleteJournalState::Committed)))
|
||||
{
|
||||
return Err(Error::other("durable ILM tier delete journal checkpoint is invalid"));
|
||||
}
|
||||
}
|
||||
if self == next {
|
||||
if let Self::ManualTransitionJob {
|
||||
progress,
|
||||
@@ -244,18 +282,64 @@ impl DurableIlmRecordCheckpoint {
|
||||
let valid = match (self, next) {
|
||||
(
|
||||
Self::TierDeleteJournal {
|
||||
content_sha256: previous_content,
|
||||
identity_sha256: previous_identity,
|
||||
committed: previous_committed,
|
||||
dispatch_identity_sha256: previous_dispatch_identity,
|
||||
state: previous_state,
|
||||
..
|
||||
},
|
||||
Self::TierDeleteJournal {
|
||||
content_sha256: next_content,
|
||||
identity_sha256: next_identity,
|
||||
committed: next_committed,
|
||||
dispatch_identity_sha256: next_dispatch_identity,
|
||||
state: next_state,
|
||||
..
|
||||
},
|
||||
) => {
|
||||
use super::tier_sweeper::TierDeleteJournalState::{Committed, Dispatched, Prepared};
|
||||
|
||||
let dispatch_identity_is_monotonic = match (previous_dispatch_identity, next_dispatch_identity) {
|
||||
(Some(previous), Some(next)) => previous == next,
|
||||
(None, None) => true,
|
||||
// Old receipts did not record the v6 dispatch binding. A
|
||||
// byte-identical observation may adopt the stronger proof,
|
||||
// but an in-flight mutation must fail closed instead of
|
||||
// guessing which operation owned the journal.
|
||||
(None, Some(_)) => previous_content == next_content,
|
||||
(Some(_), None) => false,
|
||||
};
|
||||
let state_is_monotonic = match (previous_state, next_state) {
|
||||
(Some(previous), Some(next)) => {
|
||||
previous == next || matches!((previous, next), (Prepared, Dispatched) | (Dispatched, Committed))
|
||||
}
|
||||
(None, None) => previous_committed == next_committed || (!previous_committed && *next_committed),
|
||||
(None, Some(_)) => previous_content == next_content,
|
||||
(Some(_), None) => false,
|
||||
};
|
||||
previous_identity == next_identity && dispatch_identity_is_monotonic && state_is_monotonic
|
||||
}
|
||||
(
|
||||
Self::TierDeleteDispatchManifest {
|
||||
identity_sha256: previous_identity,
|
||||
state: previous_state,
|
||||
..
|
||||
},
|
||||
Self::TierDeleteDispatchManifest {
|
||||
identity_sha256: next_identity,
|
||||
state: next_state,
|
||||
..
|
||||
},
|
||||
) => {
|
||||
use tier_delete_journal::TierDeleteDispatchManifestState::{
|
||||
Aborted, Aborting, Completed, DispatchAuthorized, Preparing,
|
||||
};
|
||||
previous_identity == next_identity
|
||||
&& (previous_committed == next_committed || (!previous_committed && *next_committed))
|
||||
&& matches!(
|
||||
(previous_state, next_state),
|
||||
(Preparing, DispatchAuthorized | Aborting) | (Aborting, Aborted) | (DispatchAuthorized, Completed)
|
||||
)
|
||||
}
|
||||
(
|
||||
Self::TransitionTransaction {
|
||||
@@ -351,6 +435,49 @@ impl DurableIlmRecordCheckpoint {
|
||||
Err(Error::other("durable ILM record generation is not a monotonic successor"))
|
||||
}
|
||||
}
|
||||
|
||||
/// Whether `self` is an older generation of the same immutable record
|
||||
/// that can reach `terminal` through one or more valid state transitions.
|
||||
/// This is deliberately broader than `validate_successor`, which remains
|
||||
/// adjacent-only for receipt advancement. Terminal cleanup uses this only
|
||||
/// after the exact terminal ETag and terminal receipt were committed, to
|
||||
/// purge older object versions exposed by that deletion.
|
||||
pub(crate) fn is_predecessor_of_terminal(&self, terminal: &Self) -> bool {
|
||||
if self == terminal || self.validate_successor(terminal).is_ok() {
|
||||
return true;
|
||||
}
|
||||
match (self, terminal) {
|
||||
(
|
||||
Self::TierDeleteJournal {
|
||||
identity_sha256: previous_identity,
|
||||
dispatch_identity_sha256: previous_dispatch,
|
||||
state: Some(super::tier_sweeper::TierDeleteJournalState::Prepared),
|
||||
..
|
||||
},
|
||||
Self::TierDeleteJournal {
|
||||
identity_sha256: terminal_identity,
|
||||
dispatch_identity_sha256: terminal_dispatch,
|
||||
state: Some(super::tier_sweeper::TierDeleteJournalState::Committed),
|
||||
..
|
||||
},
|
||||
) => previous_identity == terminal_identity && previous_dispatch == terminal_dispatch,
|
||||
(
|
||||
Self::TierDeleteDispatchManifest {
|
||||
identity_sha256: previous_identity,
|
||||
state: tier_delete_journal::TierDeleteDispatchManifestState::Preparing,
|
||||
..
|
||||
},
|
||||
Self::TierDeleteDispatchManifest {
|
||||
identity_sha256: terminal_identity,
|
||||
state:
|
||||
tier_delete_journal::TierDeleteDispatchManifestState::Aborted
|
||||
| tier_delete_journal::TierDeleteDispatchManifestState::Completed,
|
||||
..
|
||||
},
|
||||
) => previous_identity == terminal_identity,
|
||||
_ => false,
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
fn transition_state_distance(
|
||||
@@ -750,10 +877,19 @@ pub(crate) fn validate_durable_ilm_record(path: &str, data: &[u8]) -> Result<Val
|
||||
if tier_delete_journal::tier_delete_journal_object_name(&entry) != path {
|
||||
return Err(Error::other("tier delete journal content does not match its path"));
|
||||
}
|
||||
let operation_id = path
|
||||
let legacy_operation_id = path
|
||||
.strip_prefix(namespace.prefix)
|
||||
.and_then(|suffix| suffix.strip_suffix(".json"))
|
||||
.ok_or_else(|| Error::other("tier delete journal path is invalid"))?;
|
||||
// Legacy v1-v5 paths already expose a 64-hex operation id and
|
||||
// must remain receipt-compatible. V6 uses an operation-scoped
|
||||
// nested path, so derive a fixed, path-unique receipt id instead
|
||||
// of embedding slashes in the receipt locator.
|
||||
let operation_id = if entry.persisted_version == 6 {
|
||||
hex_sha256(path.as_bytes(), ToOwned::to_owned)
|
||||
} else {
|
||||
legacy_operation_id.to_string()
|
||||
};
|
||||
let identity_sha256 = checkpoint_hash(&(
|
||||
&entry.obj_name,
|
||||
&entry.version_id,
|
||||
@@ -763,13 +899,29 @@ pub(crate) fn validate_durable_ilm_record(path: &str, data: &[u8]) -> Result<Val
|
||||
entry.version_state,
|
||||
&entry.source,
|
||||
))?;
|
||||
let dispatch_identity_sha256 = entry.dispatch.as_ref().map(checkpoint_hash).transpose()?;
|
||||
(
|
||||
"operation_id",
|
||||
operation_id.to_string(),
|
||||
operation_id,
|
||||
DurableIlmRecordCheckpoint::TierDeleteJournal {
|
||||
content_sha256,
|
||||
identity_sha256,
|
||||
committed: entry.state == super::tier_sweeper::TierDeleteJournalState::Committed,
|
||||
dispatch_identity_sha256,
|
||||
state: (entry.persisted_version == 6).then_some(entry.state),
|
||||
},
|
||||
)
|
||||
}
|
||||
DurableIlmRecordKind::TierDeleteDispatchManifest => {
|
||||
let (operation_id, identity_sha256, state) =
|
||||
tier_delete_journal::validate_tier_delete_dispatch_manifest_record(path, data)?;
|
||||
(
|
||||
"operation_id",
|
||||
hex_sha256(operation_id.as_bytes(), ToOwned::to_owned),
|
||||
DurableIlmRecordCheckpoint::TierDeleteDispatchManifest {
|
||||
content_sha256,
|
||||
identity_sha256,
|
||||
state,
|
||||
},
|
||||
)
|
||||
}
|
||||
@@ -956,6 +1108,87 @@ mod tests {
|
||||
}
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn tier_delete_dispatch_manifest_namespace_validates_monotonic_branches() {
|
||||
use tier_delete_journal::TierDeleteDispatchManifestState::{Aborted, Aborting, Completed, DispatchAuthorized, Preparing};
|
||||
|
||||
let operation_id = Uuid::new_v4();
|
||||
let checkpoint = |state| {
|
||||
let (path, data) = tier_delete_journal::test_tier_delete_dispatch_manifest_record(operation_id, state);
|
||||
let namespace = classify_durable_ilm_record(&path)
|
||||
.expect("dispatch manifest namespace should classify")
|
||||
.expect("dispatch manifest should be durable");
|
||||
assert_eq!(namespace, &TIER_DELETE_DISPATCH_MANIFEST_NAMESPACE);
|
||||
validate_durable_ilm_record(&path, &data)
|
||||
.expect("dispatch manifest should validate")
|
||||
.checkpoint
|
||||
};
|
||||
|
||||
let preparing = checkpoint(Preparing);
|
||||
let authorized = checkpoint(DispatchAuthorized);
|
||||
let completed = checkpoint(Completed);
|
||||
let aborting = checkpoint(Aborting);
|
||||
let aborted = checkpoint(Aborted);
|
||||
|
||||
preparing
|
||||
.validate_successor(&authorized)
|
||||
.expect("Preparing may become DispatchAuthorized");
|
||||
authorized
|
||||
.validate_successor(&completed)
|
||||
.expect("DispatchAuthorized may become Completed");
|
||||
preparing.validate_successor(&aborting).expect("Preparing may enter rollback");
|
||||
aborting.validate_successor(&aborted).expect("Aborting may become Aborted");
|
||||
assert!(authorized.validate_successor(&aborting).is_err());
|
||||
assert!(completed.validate_successor(&authorized).is_err());
|
||||
assert!(aborted.validate_successor(&preparing).is_err());
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn tier_delete_journal_checkpoint_binds_dispatch_and_full_state_monotonically() {
|
||||
use crate::bucket::lifecycle::tier_sweeper::TierDeleteJournalState::{Committed, Dispatched, Prepared};
|
||||
|
||||
let checkpoint = |content: &str, dispatch: Option<&str>, state| DurableIlmRecordCheckpoint::TierDeleteJournal {
|
||||
content_sha256: content.repeat(64),
|
||||
identity_sha256: "i".repeat(64),
|
||||
committed: state == Some(Committed),
|
||||
dispatch_identity_sha256: dispatch.map(|value| value.repeat(64)),
|
||||
state,
|
||||
};
|
||||
let prepared = checkpoint("a", Some("d"), Some(Prepared));
|
||||
let dispatched = checkpoint("b", Some("d"), Some(Dispatched));
|
||||
let committed = checkpoint("c", Some("d"), Some(Committed));
|
||||
prepared
|
||||
.validate_successor(&dispatched)
|
||||
.expect("Prepared may advance to Dispatched");
|
||||
dispatched
|
||||
.validate_successor(&committed)
|
||||
.expect("Dispatched may advance to Committed");
|
||||
assert!(prepared.validate_successor(&committed).is_err());
|
||||
assert!(dispatched.validate_successor(&prepared).is_err());
|
||||
|
||||
let rebound = checkpoint("b", Some("e"), Some(Dispatched));
|
||||
assert!(dispatched.validate_successor(&rebound).is_err());
|
||||
|
||||
let legacy: DurableIlmRecordCheckpoint = serde_json::from_value(serde_json::json!({
|
||||
"kind": "tier_delete_journal",
|
||||
"content_sha256": "a".repeat(64),
|
||||
"identity_sha256": "i".repeat(64),
|
||||
"committed": false
|
||||
}))
|
||||
.expect("legacy tier-delete checkpoint should remain decodable");
|
||||
legacy
|
||||
.validate_successor(&prepared)
|
||||
.expect("byte-identical legacy receipt may adopt the stronger v6 proof");
|
||||
let changed_legacy = DurableIlmRecordCheckpoint::TierDeleteJournal {
|
||||
content_sha256: "z".repeat(64),
|
||||
identity_sha256: "i".repeat(64),
|
||||
committed: false,
|
||||
dispatch_identity_sha256: None,
|
||||
state: None,
|
||||
};
|
||||
assert!(changed_legacy.validate_successor(&prepared).is_err());
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn manual_transition_job_checkpoint_compacts_legacy_progress_compatibly() {
|
||||
let options = super::super::bucket_lifecycle_ops::ManualTransitionRunOptions::default();
|
||||
|
||||
@@ -34,6 +34,6 @@ pub mod tier_sweeper;
|
||||
pub mod transition_transaction;
|
||||
|
||||
pub(crate) use durable_namespace::{
|
||||
DurableIlmRecordCheckpoint, ILM_META_PREFIX, ValidatedDurableIlmRecord, classify_durable_ilm_record,
|
||||
validate_durable_ilm_record,
|
||||
DurableIlmRecordCheckpoint, ILM_META_PREFIX, TIER_DELETE_DISPATCH_MANIFEST_NAMESPACE, ValidatedDurableIlmRecord,
|
||||
classify_durable_ilm_record, validate_durable_ilm_record,
|
||||
};
|
||||
|
||||
File diff suppressed because it is too large
Load Diff
@@ -172,7 +172,8 @@ pub(super) async fn recover_tier_free_versions_with_cancel(
|
||||
return Err(std::io::Error::other("free-version recovery limit must be greater than zero").into());
|
||||
}
|
||||
|
||||
let page = list_tier_free_versions(api, limit, bucket_marker.clone(), object_marker.clone(), cancel_token.clone()).await?;
|
||||
let page =
|
||||
list_tier_free_versions(api.clone(), limit, bucket_marker.clone(), object_marker.clone(), cancel_token.clone()).await?;
|
||||
let mut stats = FreeVersionRecoveryStats {
|
||||
scanned: 0,
|
||||
enqueued: 0,
|
||||
@@ -190,7 +191,7 @@ pub(super) async fn recover_tier_free_versions_with_cancel(
|
||||
return Err(tier_free_version_recovery_cancelled());
|
||||
}
|
||||
retry_cursor.visit(&oi);
|
||||
if !record_recovered_free_version_enqueue(&mut stats, enqueue_recovered_free_version(oi).await) {
|
||||
if !record_recovered_free_version_enqueue(&mut stats, enqueue_recovered_free_version(&api, oi).await) {
|
||||
let (bucket_marker, object_marker) = retry_cursor.retry_markers();
|
||||
stats.truncated = true;
|
||||
stats.next_bucket_marker = bucket_marker;
|
||||
|
||||
@@ -255,6 +255,7 @@ impl ObjSweeper {
|
||||
}
|
||||
if del_tier {
|
||||
return Some(Jentry {
|
||||
persisted_version: 0,
|
||||
obj_name: self.remote_object.clone(),
|
||||
version_id: self.transition_version_id.clone(),
|
||||
tier_name: self.transition_tier.clone(),
|
||||
@@ -266,6 +267,7 @@ impl ObjSweeper {
|
||||
version_state: self.transition_version_state,
|
||||
state: TierDeleteJournalState::Committed,
|
||||
source: None,
|
||||
dispatch: None,
|
||||
});
|
||||
}
|
||||
None
|
||||
@@ -298,9 +300,19 @@ impl ObjSweeper {
|
||||
#[derive(Debug, Clone, Copy, Serialize, Deserialize, PartialEq, Eq)]
|
||||
pub(crate) enum TierDeleteJournalState {
|
||||
Prepared,
|
||||
Dispatched,
|
||||
Committed,
|
||||
}
|
||||
|
||||
#[derive(Debug, Clone, Serialize, Deserialize, PartialEq, Eq)]
|
||||
#[serde(deny_unknown_fields)]
|
||||
pub(crate) struct TierDeleteDispatchBinding {
|
||||
pub(crate) operation_id: Uuid,
|
||||
pub(crate) manifest_object: String,
|
||||
pub(crate) journal_set_sha256: String,
|
||||
pub(crate) topology_generation: String,
|
||||
}
|
||||
|
||||
#[derive(Debug, Clone, Serialize, Deserialize, PartialEq, Eq)]
|
||||
#[serde(deny_unknown_fields)]
|
||||
pub(crate) struct TierDeleteSourceIdentity {
|
||||
@@ -342,6 +354,10 @@ impl TierDeleteSourceIdentity {
|
||||
#[derive(Debug, Clone)]
|
||||
#[allow(unused_assignments)]
|
||||
pub struct Jentry {
|
||||
/// On-disk format version when decoded. Newly constructed entries use 0;
|
||||
/// the encoder chooses their format from the durable ownership fields.
|
||||
/// Recovery uses this value to quarantine v1-v5 without rewriting them.
|
||||
pub(crate) persisted_version: u8,
|
||||
pub(crate) obj_name: String,
|
||||
pub(crate) version_id: String,
|
||||
pub(crate) tier_name: String,
|
||||
@@ -350,6 +366,23 @@ pub struct Jentry {
|
||||
pub(crate) version_state: rustfs_filemeta::TransitionVersionState,
|
||||
pub(crate) state: TierDeleteJournalState,
|
||||
pub(crate) source: Option<TierDeleteSourceIdentity>,
|
||||
pub(crate) dispatch: Option<TierDeleteDispatchBinding>,
|
||||
}
|
||||
|
||||
impl Jentry {
|
||||
/// Whether this prepared transaction is eligible to become the sole
|
||||
/// cleanup owner for its transitioned source. The caller may use this to
|
||||
/// decide whether to persist it, but must not set `skip_free_version`
|
||||
/// until persistence succeeds.
|
||||
pub(crate) fn can_replace_tier_free_version(&self) -> bool {
|
||||
self.state == TierDeleteJournalState::Prepared
|
||||
&& self.backend_identity.is_some()
|
||||
&& self.version_state != rustfs_filemeta::TransitionVersionState::Unknown
|
||||
&& self
|
||||
.source
|
||||
.as_ref()
|
||||
.is_some_and(TierDeleteSourceIdentity::has_stable_identity)
|
||||
}
|
||||
}
|
||||
|
||||
impl ExpiryOp for Jentry {
|
||||
@@ -617,6 +650,7 @@ pub fn transitioned_force_delete_journal_entry(
|
||||
}
|
||||
|
||||
Some(Jentry {
|
||||
persisted_version: 0,
|
||||
obj_name: transitioned.name.clone(),
|
||||
version_id: transitioned.version_id.clone(),
|
||||
tier_name: transitioned.tier.clone(),
|
||||
@@ -628,6 +662,7 @@ pub fn transitioned_force_delete_journal_entry(
|
||||
version_state: transition_version_state,
|
||||
state: TierDeleteJournalState::Committed,
|
||||
source: None,
|
||||
dispatch: None,
|
||||
})
|
||||
}
|
||||
|
||||
@@ -673,17 +708,73 @@ mod test {
|
||||
use rustfs_s3_client::signer_error::invalid_utf8_header_error;
|
||||
|
||||
use super::{
|
||||
CONFIRMED_TRANSITION_EMPTY_GUARD_DISPATCHES, ERR_REMOTE_DELETE_BREAKER_OPEN, ERR_REMOTE_DELETE_LIMITER_CLOSED,
|
||||
RemoteDeleteBreaker, RemoteTierDeleteOutcome, delete_confirmed_transition_candidate_exact_with_manager_and_identity,
|
||||
delete_object_from_remote_tier_idempotent, delete_object_from_remote_tier_idempotent_with_manager_and_identity,
|
||||
is_remote_tier_not_found_error, is_signer_header_error, lifecycle, set_remote_tier_delete_test_hook,
|
||||
should_record_remote_delete_failure, transitioned_delete_journal_entry, transitioned_force_delete_journal_entry,
|
||||
CONFIRMED_TRANSITION_EMPTY_GUARD_DISPATCHES, ERR_REMOTE_DELETE_BREAKER_OPEN, ERR_REMOTE_DELETE_LIMITER_CLOSED, Jentry,
|
||||
RemoteDeleteBreaker, RemoteTierDeleteOutcome, TierDeleteJournalState, TierDeleteSourceIdentity,
|
||||
delete_confirmed_transition_candidate_exact_with_manager_and_identity, delete_object_from_remote_tier_idempotent,
|
||||
delete_object_from_remote_tier_idempotent_with_manager_and_identity, is_remote_tier_not_found_error,
|
||||
is_signer_header_error, lifecycle, set_remote_tier_delete_test_hook, should_record_remote_delete_failure,
|
||||
transitioned_delete_journal_entry, transitioned_force_delete_journal_entry,
|
||||
};
|
||||
use crate::storage_api_contracts::lifecycle::TransitionedObject;
|
||||
use rustfs_filemeta::TransitionVersionState;
|
||||
use std::io::{Error, ErrorKind};
|
||||
use std::time::{Duration, Instant};
|
||||
|
||||
fn stable_prepared_journal() -> Jentry {
|
||||
Jentry {
|
||||
persisted_version: 0,
|
||||
obj_name: "remote/object".to_string(),
|
||||
version_id: "remote-version".to_string(),
|
||||
tier_name: "WARM".to_string(),
|
||||
backend_identity: Some([7; 32]),
|
||||
version_id_exact: true,
|
||||
version_state: TransitionVersionState::Exact,
|
||||
state: TierDeleteJournalState::Prepared,
|
||||
source: Some(TierDeleteSourceIdentity {
|
||||
bucket: "bucket".to_string(),
|
||||
object: "object".to_string(),
|
||||
version_id: Some(uuid::Uuid::new_v4().to_string()),
|
||||
versioned: true,
|
||||
version_suspended: false,
|
||||
data_dir: None,
|
||||
etag: None,
|
||||
mod_time: None,
|
||||
}),
|
||||
dispatch: None,
|
||||
}
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn only_stable_prepared_journal_can_replace_tier_free_version() {
|
||||
let stable = stable_prepared_journal();
|
||||
assert!(stable.can_replace_tier_free_version());
|
||||
|
||||
let mut committed = stable.clone();
|
||||
committed.state = TierDeleteJournalState::Committed;
|
||||
assert!(!committed.can_replace_tier_free_version());
|
||||
|
||||
let mut unbound = stable.clone();
|
||||
unbound.backend_identity = None;
|
||||
assert!(!unbound.can_replace_tier_free_version());
|
||||
|
||||
let mut unknown = stable.clone();
|
||||
unknown.version_state = TransitionVersionState::Unknown;
|
||||
assert!(!unknown.can_replace_tier_free_version());
|
||||
|
||||
let mut unstable = stable;
|
||||
unstable.source = Some(TierDeleteSourceIdentity {
|
||||
bucket: "bucket".to_string(),
|
||||
object: "object".to_string(),
|
||||
version_id: None,
|
||||
versioned: false,
|
||||
version_suspended: false,
|
||||
data_dir: None,
|
||||
etag: Some("etag-only".to_string()),
|
||||
mod_time: None,
|
||||
});
|
||||
assert!(!unstable.can_replace_tier_free_version());
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn signer_header_error_detection_matches_utf8_failures() {
|
||||
let err = Error::new(
|
||||
|
||||
@@ -412,8 +412,14 @@ pub(crate) fn require_bucket_metadata_sys_in(
|
||||
}
|
||||
|
||||
pub(crate) async fn object_store_in(ctx: &crate::runtime::instance::InstanceContext) -> Result<Arc<ECStore>> {
|
||||
let sys = bucket_metadata_sys_of(ctx)?;
|
||||
Ok(sys.read().await.api.clone())
|
||||
object_store_if_initialized_in(ctx)
|
||||
.await
|
||||
.ok_or_else(|| Error::other("bucket metadata sys not initialized for this instance"))
|
||||
}
|
||||
|
||||
pub(crate) async fn object_store_if_initialized_in(ctx: &crate::runtime::instance::InstanceContext) -> Option<Arc<ECStore>> {
|
||||
let sys = ctx.bucket_metadata_sys().or_else(get_global_bucket_metadata_sys)?;
|
||||
Some(sys.read().await.api.clone())
|
||||
}
|
||||
|
||||
pub(crate) async fn get_in(ctx: &crate::runtime::instance::InstanceContext, bucket: &str) -> Result<Arc<BucketMetadata>> {
|
||||
@@ -1130,6 +1136,16 @@ pub(crate) async fn has_authoritative_never_versioned_state(bucket: &str) -> Res
|
||||
bucket_meta_sys.has_authoritative_never_versioned_state(bucket).await
|
||||
}
|
||||
|
||||
pub(crate) async fn has_authoritative_never_versioned_state_in(
|
||||
ctx: &crate::runtime::instance::InstanceContext,
|
||||
bucket: &str,
|
||||
) -> Result<bool> {
|
||||
let bucket_meta_sys_lock = bucket_metadata_sys_of(ctx)?;
|
||||
let bucket_meta_sys = bucket_meta_sys_lock.read().await.clone();
|
||||
|
||||
bucket_meta_sys.has_authoritative_never_versioned_state(bucket).await
|
||||
}
|
||||
|
||||
pub async fn get_website_config(bucket: &str) -> Result<(WebsiteConfiguration, OffsetDateTime)> {
|
||||
let bucket_meta_sys_lock = get_bucket_metadata_sys()?;
|
||||
let bucket_meta_sys = bucket_meta_sys_lock.read().await;
|
||||
@@ -2512,11 +2528,169 @@ pub(crate) mod test_support {
|
||||
mod tests {
|
||||
use super::test_support::isolated_store_over_temp_disks;
|
||||
use super::*;
|
||||
use crate::bucket::metadata::{
|
||||
BUCKET_ACCELERATE_CONFIG, BUCKET_CORS_CONFIG, BUCKET_LIFECYCLE_CONFIG, BUCKET_LOGGING_CONFIG, BUCKET_NOTIFICATION_CONFIG,
|
||||
BUCKET_POLICY_CONFIG, BUCKET_PUBLIC_ACCESS_BLOCK_CONFIG, BUCKET_REPLICATION_CONFIG, BUCKET_REQUEST_PAYMENT_CONFIG,
|
||||
BUCKET_SSECONFIG, BUCKET_TAGGING_CONFIG, BUCKET_VERSIONING_CONFIG, BUCKET_WEBSITE_CONFIG, OBJECT_LOCK_CONFIG,
|
||||
};
|
||||
use crate::bucket::target::{BucketTarget, BucketTargetType, Credentials};
|
||||
use crate::config::com::read_config;
|
||||
use crate::storage_api_contracts::bucket::{BucketOperations as _, DeleteBucketOptions, MakeBucketOptions};
|
||||
use byteorder::{ByteOrder as _, LittleEndian};
|
||||
use serial_test::serial;
|
||||
use tokio::time::timeout;
|
||||
|
||||
const NEW_WRITER_REPLICATION_XML: &[u8] = br#"<ReplicationConfiguration xmlns="http://s3.amazonaws.com/doc/2006-03-01/"><Role>arn:aws:iam::111122223333:role/replication-role</Role><Rule><ID>rollback</ID><Priority>1</Priority><Filter><Prefix>documents/</Prefix></Filter><Status>Enabled</Status><Destination><Bucket>arn:aws:s3:::replica-bucket</Bucket></Destination><DeleteMarkerReplication><Status>Disabled</Status></DeleteMarkerReplication></Rule></ReplicationConfiguration>"#;
|
||||
|
||||
const NEW_WRITER_CONFIGS: [(&str, &[u8]); 14] = [
|
||||
(BUCKET_POLICY_CONFIG, br#"{"Version":"2012-10-17","Statement":[]}"#),
|
||||
(BUCKET_NOTIFICATION_CONFIG, br#"<NotificationConfiguration/>"#),
|
||||
(
|
||||
BUCKET_LIFECYCLE_CONFIG,
|
||||
br#"<LifecycleConfiguration><Rule><ID>expire</ID><Status>Enabled</Status><Filter><Prefix>logs/</Prefix></Filter><Expiration><Days>30</Days></Expiration></Rule></LifecycleConfiguration>"#,
|
||||
),
|
||||
(
|
||||
OBJECT_LOCK_CONFIG,
|
||||
br#"<ObjectLockConfiguration><ObjectLockEnabled>Enabled</ObjectLockEnabled><Rule><DefaultRetention><Mode>GOVERNANCE</Mode><Days>7</Days></DefaultRetention></Rule></ObjectLockConfiguration>"#,
|
||||
),
|
||||
(
|
||||
BUCKET_VERSIONING_CONFIG,
|
||||
br#"<VersioningConfiguration><Status>Enabled</Status></VersioningConfiguration>"#,
|
||||
),
|
||||
(
|
||||
BUCKET_SSECONFIG,
|
||||
br#"<ServerSideEncryptionConfiguration><Rule><ApplyServerSideEncryptionByDefault><SSEAlgorithm>AES256</SSEAlgorithm></ApplyServerSideEncryptionByDefault></Rule></ServerSideEncryptionConfiguration>"#,
|
||||
),
|
||||
(
|
||||
BUCKET_TAGGING_CONFIG,
|
||||
r#"<Tagging><TagSet><Tag><Key>environment</Key><Value>测试-🦀</Value></Tag></TagSet></Tagging>"#.as_bytes(),
|
||||
),
|
||||
(BUCKET_REPLICATION_CONFIG, NEW_WRITER_REPLICATION_XML),
|
||||
(
|
||||
BUCKET_CORS_CONFIG,
|
||||
br#"<CORSConfiguration><CORSRule><AllowedMethod>GET</AllowedMethod><AllowedOrigin>https://example.test</AllowedOrigin></CORSRule></CORSConfiguration>"#,
|
||||
),
|
||||
(BUCKET_LOGGING_CONFIG, br#"<BucketLoggingStatus/>"#),
|
||||
(
|
||||
BUCKET_WEBSITE_CONFIG,
|
||||
br#"<WebsiteConfiguration><IndexDocument><Suffix>index.html</Suffix></IndexDocument></WebsiteConfiguration>"#,
|
||||
),
|
||||
(
|
||||
BUCKET_ACCELERATE_CONFIG,
|
||||
br#"<AccelerateConfiguration><Status>Enabled</Status></AccelerateConfiguration>"#,
|
||||
),
|
||||
(
|
||||
BUCKET_REQUEST_PAYMENT_CONFIG,
|
||||
br#"<RequestPaymentConfiguration><Payer>Requester</Payer></RequestPaymentConfiguration>"#,
|
||||
),
|
||||
(
|
||||
BUCKET_PUBLIC_ACCESS_BLOCK_CONFIG,
|
||||
br#"<PublicAccessBlockConfiguration><BlockPublicAcls>true</BlockPublicAcls><IgnorePublicAcls>true</IgnorePublicAcls><BlockPublicPolicy>true</BlockPublicPolicy><RestrictPublicBuckets>false</RestrictPublicBuckets></PublicAccessBlockConfiguration>"#,
|
||||
),
|
||||
];
|
||||
|
||||
#[tokio::test]
|
||||
async fn g_d3_003_new_writer_replication_loads_without_fail_closed_state() {
|
||||
let (dirs, store) = isolated_store_over_temp_disks().await;
|
||||
let bucket = "rollback-new-replication";
|
||||
for dir in &dirs {
|
||||
std::fs::create_dir_all(dir.path().join(bucket)).expect("rollback fixture bucket should be created");
|
||||
}
|
||||
|
||||
let writer = BucketMetadataSys::new(store.clone());
|
||||
let mut metadata = BucketMetadata::new(bucket);
|
||||
metadata
|
||||
.update_config(BUCKET_REPLICATION_CONFIG, NEW_WRITER_REPLICATION_XML.to_vec())
|
||||
.expect("new-writer replication XML should be accepted before persistence");
|
||||
writer
|
||||
.persist_new_and_set(metadata)
|
||||
.await
|
||||
.expect("new-writer replication metadata should persist");
|
||||
|
||||
let old_reader = BucketMetadataSys::new(store);
|
||||
let (loaded, _) = old_reader
|
||||
.get_replication_config(bucket)
|
||||
.await
|
||||
.expect("old metadata_sys must not classify new-writer replication XML as invalid");
|
||||
assert_eq!(loaded.role, "arn:aws:iam::111122223333:role/replication-role");
|
||||
assert_eq!(loaded.rules.len(), 1);
|
||||
assert_eq!(loaded.rules[0].id.as_deref(), Some("rollback"));
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
async fn g_d3_004_new_writer_metadata_blob_keeps_legacy_header_and_configs() {
|
||||
let (dirs, store) = isolated_store_over_temp_disks().await;
|
||||
let bucket = "rollback-new-metadata";
|
||||
for dir in &dirs {
|
||||
std::fs::create_dir_all(dir.path().join(bucket)).expect("rollback fixture bucket should be created");
|
||||
}
|
||||
|
||||
let writer = BucketMetadataSys::new(store.clone());
|
||||
let mut metadata = BucketMetadata::new(bucket);
|
||||
for (config_file, bytes) in NEW_WRITER_CONFIGS {
|
||||
metadata
|
||||
.update_config(config_file, bytes.to_vec())
|
||||
.unwrap_or_else(|err| panic!("new-writer {config_file} fixture must be valid: {err}"));
|
||||
}
|
||||
writer
|
||||
.persist_new_and_set(metadata)
|
||||
.await
|
||||
.expect("new-writer metadata should persist");
|
||||
|
||||
let path = BucketMetadata::new(bucket).save_file_path();
|
||||
let blob = read_config(store.clone(), &path)
|
||||
.await
|
||||
.expect("persisted .metadata.bin should be readable");
|
||||
assert_eq!(
|
||||
LittleEndian::read_u16(&blob[0..2]),
|
||||
1,
|
||||
"bucket metadata format must stay rollback-readable"
|
||||
);
|
||||
assert_eq!(
|
||||
LittleEndian::read_u16(&blob[2..4]),
|
||||
1,
|
||||
"bucket metadata version must stay rollback-readable"
|
||||
);
|
||||
|
||||
let loaded = load_bucket_metadata(store, bucket)
|
||||
.await
|
||||
.expect("old read_bucket_metadata path must load the new-writer blob");
|
||||
let loaded_configs: [(&str, &[u8]); 14] = [
|
||||
(BUCKET_POLICY_CONFIG, &loaded.policy_config_json),
|
||||
(BUCKET_NOTIFICATION_CONFIG, &loaded.notification_config_xml),
|
||||
(BUCKET_LIFECYCLE_CONFIG, &loaded.lifecycle_config_xml),
|
||||
(OBJECT_LOCK_CONFIG, &loaded.object_lock_config_xml),
|
||||
(BUCKET_VERSIONING_CONFIG, &loaded.versioning_config_xml),
|
||||
(BUCKET_SSECONFIG, &loaded.encryption_config_xml),
|
||||
(BUCKET_TAGGING_CONFIG, &loaded.tagging_config_xml),
|
||||
(BUCKET_REPLICATION_CONFIG, &loaded.replication_config_xml),
|
||||
(BUCKET_CORS_CONFIG, &loaded.cors_config_xml),
|
||||
(BUCKET_LOGGING_CONFIG, &loaded.logging_config_xml),
|
||||
(BUCKET_WEBSITE_CONFIG, &loaded.website_config_xml),
|
||||
(BUCKET_ACCELERATE_CONFIG, &loaded.accelerate_config_xml),
|
||||
(BUCKET_REQUEST_PAYMENT_CONFIG, &loaded.request_payment_config_xml),
|
||||
(BUCKET_PUBLIC_ACCESS_BLOCK_CONFIG, &loaded.public_access_block_config_xml),
|
||||
];
|
||||
for ((expected_name, expected), (loaded_name, actual)) in NEW_WRITER_CONFIGS.into_iter().zip(loaded_configs) {
|
||||
assert_eq!(loaded_name, expected_name);
|
||||
assert_eq!(actual, expected, "old read_bucket_metadata changed {expected_name} bytes");
|
||||
}
|
||||
assert!(loaded.policy_config.is_some());
|
||||
assert!(loaded.notification_config.is_some());
|
||||
assert!(loaded.lifecycle_config.is_some());
|
||||
assert!(loaded.object_lock_config.is_some());
|
||||
assert!(loaded.versioning_config.is_some());
|
||||
assert!(loaded.sse_config.is_some());
|
||||
assert!(loaded.tagging_config.is_some());
|
||||
assert!(loaded.replication_config.is_some());
|
||||
assert!(loaded.cors_config.is_some());
|
||||
assert!(loaded.logging_config.is_some());
|
||||
assert!(loaded.website_config.is_some());
|
||||
assert!(loaded.accelerate_config.is_some());
|
||||
assert!(loaded.request_payment_config.is_some());
|
||||
assert!(loaded.public_access_block_config.is_some());
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
async fn malformed_delete_configs_are_not_treated_as_absent() {
|
||||
let (_dirs, ecstore) = isolated_store_over_temp_disks().await;
|
||||
|
||||
@@ -177,6 +177,28 @@ pub fn replication_write_may_pass_worm_gate(
|
||||
Ok(!(retention_locked && opts.replication_retention_timestamp.is_none()))
|
||||
}
|
||||
|
||||
/// Whether an authorized replication delete (`ObjectOptions::replication_request`)
|
||||
/// addressed to an explicit version may bypass GOVERNANCE retention on the
|
||||
/// local replica, exactly as an `x-amz-bypass-governance-retention` caller
|
||||
/// with the bypass permission would.
|
||||
///
|
||||
/// The source is authoritative for a replicated version purge (issue #6850):
|
||||
/// the same WORM deletion gate already ran there, and GOVERNANCE retention
|
||||
/// with an authorized bypass is the only lock state it can purge through.
|
||||
/// Requiring the bypass header again here makes the purge permanently
|
||||
/// undeliverable — replication senders never carry it — and the sites diverge
|
||||
/// forever. COMPLIANCE retention and legal hold stay blocking: the source
|
||||
/// gate can never purge through them, so a replication purge that meets one
|
||||
/// here is divergence or forgery and fails closed.
|
||||
///
|
||||
/// The trust judgment is the same one the write-path exemption uses:
|
||||
/// `replication_request` is only set once the receiving handler has
|
||||
/// authorized the caller for the replication action
|
||||
/// (`ReplicateDeleteAction`), never straight from request headers.
|
||||
pub fn replication_delete_may_bypass_governance(opts: &ObjectOptions) -> bool {
|
||||
opts.replication_request && opts.version_id.is_some()
|
||||
}
|
||||
|
||||
/// Check if an object is locked based on its metadata.
|
||||
/// This is a common function used by both lifecycle evaluation and deletion checks.
|
||||
///
|
||||
@@ -680,6 +702,32 @@ mod tests {
|
||||
assert!(err.to_string().contains("modification time"));
|
||||
}
|
||||
|
||||
/// The replicated-purge GOVERNANCE bypass (#6850) applies only to an
|
||||
/// authorized replication delete addressed to an explicit version: a
|
||||
/// local delete never gets it, and a replicated delete without a version
|
||||
/// id creates a delete marker rather than purging anything.
|
||||
#[test]
|
||||
fn replication_delete_bypasses_governance_only_for_authorized_version_purges() {
|
||||
let version_purge = ObjectOptions {
|
||||
replication_request: true,
|
||||
version_id: Some("6b6ffbc0-b0d3-4a86-8f6c-fe19163b8dcd".to_string()),
|
||||
..Default::default()
|
||||
};
|
||||
assert!(replication_delete_may_bypass_governance(&version_purge));
|
||||
|
||||
let local_version_delete = ObjectOptions {
|
||||
replication_request: false,
|
||||
..version_purge.clone()
|
||||
};
|
||||
assert!(!replication_delete_may_bypass_governance(&local_version_delete));
|
||||
|
||||
let replicated_marker_creation = ObjectOptions {
|
||||
version_id: None,
|
||||
..version_purge
|
||||
};
|
||||
assert!(!replication_delete_may_bypass_governance(&replicated_marker_creation));
|
||||
}
|
||||
|
||||
/// A local PutObjectRetention / PutObjectLegalHold "clear" persists the
|
||||
/// lock keys as empty strings (the MinIO on-disk shape, see
|
||||
/// `parse_object_lock_retention`); that is "no lock", not corruption, and
|
||||
|
||||
@@ -436,16 +436,21 @@ pub(crate) async fn check_replicate_delete_strict(
|
||||
}
|
||||
|
||||
for target in decision.targets_map.values_mut() {
|
||||
if let Some(client) = ReplicationTargetStore::remote_target_client(bucket, &target.arn).await {
|
||||
target.synchronous = client.replicate_sync;
|
||||
} else {
|
||||
target.replicate = false;
|
||||
target.synchronous = false;
|
||||
}
|
||||
let replicate_sync = ReplicationTargetStore::remote_target_client(bucket, &target.arn)
|
||||
.await
|
||||
.map(|client| client.replicate_sync);
|
||||
apply_target_delivery_mode(target, replicate_sync);
|
||||
}
|
||||
Ok(decision)
|
||||
}
|
||||
|
||||
fn apply_target_delivery_mode(target: &mut ReplicateTargetDecision, replicate_sync: Option<bool>) {
|
||||
// A missing runtime client is a delivery failure, not a rule mismatch.
|
||||
// Preserve admission and fall back to the asynchronous worker, which can
|
||||
// persist FAILED state for the heal/retry path.
|
||||
target.synchronous = replicate_sync.unwrap_or(false);
|
||||
}
|
||||
|
||||
pub(crate) fn check_replicate_delete_with_snapshot(
|
||||
dobj: &ObjectToDelete,
|
||||
oi: &ObjectInfo,
|
||||
@@ -629,6 +634,23 @@ mod tests {
|
||||
}));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn missing_target_client_preserves_delete_admission_as_async() {
|
||||
let mut target = ReplicateTargetDecision::new("arn:target".to_string(), true, true);
|
||||
|
||||
apply_target_delivery_mode(&mut target, None);
|
||||
|
||||
assert!(target.replicate, "a runtime client miss must not erase the replication rule decision");
|
||||
assert!(
|
||||
!target.synchronous,
|
||||
"unavailable synchronous targets must fall back to the async retry path"
|
||||
);
|
||||
|
||||
apply_target_delivery_mode(&mut target, Some(true));
|
||||
assert!(target.replicate);
|
||||
assert!(target.synchronous);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn must_replicate_options_preserve_request_flag() {
|
||||
let user_defined = HashMap::new();
|
||||
|
||||
@@ -20,8 +20,9 @@ pub use rustfs_replication::{
|
||||
pub(crate) use rustfs_replication::{
|
||||
ReplicationDeleteSource, ReplicationMultipartPartInput, ReplicationResyncTargetObject, delete_marker_purge_mrf_entry,
|
||||
delete_marker_purge_version_id, delete_replication_creates_marker, delete_replication_missing_source_decision,
|
||||
delete_replication_object_opts, heal_uses_delete_replication_path, is_retryable_delete_replication_head_error,
|
||||
is_version_delete_replication, replicate_delete_outcome, replication_etags_match, replication_multipart_complete_actual_size,
|
||||
replication_multipart_part_plan, resync_existing_delete_replication_info, resync_target_for_object,
|
||||
should_retry_delete_marker_purge, target_delete_version_id,
|
||||
delete_replication_object_opts, heal_uses_delete_replication_path, is_object_lock_denied_delete,
|
||||
is_retryable_delete_replication_head_error, is_version_delete_replication, replicate_delete_outcome, replication_etags_match,
|
||||
replication_multipart_complete_actual_size, replication_multipart_part_plan, replication_single_put_size_error,
|
||||
resync_existing_delete_replication_info, resync_target_for_object, should_retry_delete_marker_purge,
|
||||
single_part_replica_etag_mismatch, target_delete_version_id,
|
||||
};
|
||||
|
||||
@@ -3177,6 +3177,19 @@ pub(crate) async fn queue_replication_heal_internal(
|
||||
}
|
||||
}
|
||||
ReplicationHealQueueAction::QueueDelete(dv) => {
|
||||
// A purge the peer denied under object lock cannot succeed until
|
||||
// the lock lapses (#6850); requeuing it every heal cycle only
|
||||
// burns bandwidth and failure counters. The backoff expires on
|
||||
// its own, so the purge is probed again — and converges — once
|
||||
// the retention window has a chance of being over.
|
||||
if super::replication_object_decision_boundary::is_version_delete_replication(&dv.delete_object)
|
||||
&& super::replication_resyncer::object_lock_denied_purge_backoff_active(&dv)
|
||||
{
|
||||
return ReplicationHealQueueResult {
|
||||
object_info: roi,
|
||||
admission: ReplicationQueueAdmission::Skipped,
|
||||
};
|
||||
}
|
||||
let admission = if let Some(pool) = runtime_sources::replication_pool() {
|
||||
pool.queue_replica_delete_task(dv).await
|
||||
} else {
|
||||
|
||||
File diff suppressed because it is too large
Load Diff
@@ -36,8 +36,8 @@ use time::OffsetDateTime;
|
||||
use time::format_description::well_known::Rfc3339;
|
||||
|
||||
pub(crate) use crate::bucket::bucket_target_sys::{
|
||||
AdvancedPutOptions, HeadObjectSdkError, PutObjectOptions, PutObjectPartOptions, RemoveObjectOptions, S3ClientError,
|
||||
TargetClient, resolve_read_api_version_id,
|
||||
AdvancedPutOptions, HeadObjectSdkError, PutObjectOptions, PutObjectPartOptions, RemotePutObjectResponse, RemoveObjectOptions,
|
||||
S3ClientError, TargetClient, resolve_read_api_version_id,
|
||||
};
|
||||
#[cfg(test)]
|
||||
pub(crate) use crate::bucket::target::BucketTarget;
|
||||
|
||||
@@ -25,6 +25,8 @@ use time::OffsetDateTime;
|
||||
use url::Url;
|
||||
|
||||
const REDACTED_CREDENTIAL: &str = "<redacted>";
|
||||
const GO_YEAR_ONE_START_UNIX_SECONDS: i64 = -62_135_596_800;
|
||||
const GO_YEAR_TWO_START_UNIX_SECONDS: i64 = -62_104_060_800;
|
||||
|
||||
#[derive(Deserialize, Serialize, Default, Clone)]
|
||||
pub struct Credentials {
|
||||
@@ -41,6 +43,26 @@ pub struct Credentials {
|
||||
}
|
||||
|
||||
impl Credentials {
|
||||
/// Returns the session token used for request signing.
|
||||
///
|
||||
/// MinIO-compatible payloads may carry an empty token. Treat whitespace-only
|
||||
/// values as absent without rewriting a real token, whose bytes are opaque.
|
||||
pub fn effective_session_token(&self) -> Option<&str> {
|
||||
self.session_token.as_deref().filter(|token| !token.trim().is_empty())
|
||||
}
|
||||
|
||||
/// Returns the credential expiry after normalizing Go's zero `time.Time`.
|
||||
///
|
||||
/// Go JSON encoders emit year 1 for an unset `time.Time`; persisted MinIO
|
||||
/// target metadata can therefore contain that sentinel even for static
|
||||
/// credentials.
|
||||
pub fn effective_expiration(&self) -> Option<Timestamp> {
|
||||
self.expiration.filter(|expiration| {
|
||||
let unix_seconds = expiration.as_second();
|
||||
!(GO_YEAR_ONE_START_UNIX_SECONDS..GO_YEAR_TWO_START_UNIX_SECONDS).contains(&unix_seconds)
|
||||
})
|
||||
}
|
||||
|
||||
pub fn redacted(&self) -> Self {
|
||||
Self {
|
||||
access_key: self.access_key.clone(),
|
||||
@@ -355,6 +377,24 @@ mod tests {
|
||||
use std::time::Duration;
|
||||
use time::OffsetDateTime;
|
||||
|
||||
#[test]
|
||||
fn credential_effective_values_normalize_only_compatibility_sentinels() {
|
||||
let mut credentials = Credentials {
|
||||
access_key: "access".to_string(),
|
||||
secret_key: "secret".to_string(),
|
||||
session_token: Some(" ".to_string()),
|
||||
expiration: Some("0001-01-01T08:00:00+08:00".parse().expect("Go zero time should parse")),
|
||||
};
|
||||
|
||||
assert!(credentials.effective_session_token().is_none());
|
||||
assert!(credentials.effective_expiration().is_none());
|
||||
|
||||
credentials.session_token = Some(" opaque token ".to_string());
|
||||
credentials.expiration = Some("2099-01-01T00:00:00Z".parse().expect("future timestamp should parse"));
|
||||
assert_eq!(credentials.effective_session_token(), Some(" opaque token "));
|
||||
assert_eq!(credentials.effective_expiration(), credentials.expiration);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_bucket_target_json_deserialize() {
|
||||
let json = r#"
|
||||
|
||||
@@ -73,6 +73,7 @@ pub fn check_valid_bucket_name_strict(bucket_name: &str) -> Result<()> {
|
||||
check_bucket_name_common(bucket_name, true)
|
||||
}
|
||||
|
||||
// RUSTFS_COMPAT_TODO(s3gate-metadata-xml): the s3s codec reads persisted XML during migration. Remove after every supported writer uses the gateway codec and every retained metadata object and backup archive is verified or rewritten.
|
||||
pub fn deserialize<T>(input: &[u8]) -> xml::DeResult<T>
|
||||
where
|
||||
T: for<'xml> xml::Deserialize<'xml>,
|
||||
|
||||
@@ -1307,6 +1307,32 @@ pub fn verify_tonic_mutation_body_digest<T>(request: &tonic::Request<T>, canonic
|
||||
verify_tonic_mutation_body_digest_with_strictness(request, canonical_body, internode_rpc_body_digest_strict())
|
||||
}
|
||||
|
||||
/// Verify a non-disk mutation without accepting a newly-generated unsigned v2 body.
|
||||
///
|
||||
/// The disk mutation lane has a rolling-upgrade exception for `UNSIGNED-PAYLOAD`
|
||||
/// while peer replay-cache capability is being discovered. Historical v2 peers
|
||||
/// used the fixed `unsigned` nonce before body-digest rollout; preserve that
|
||||
/// exact marker for mixed-version compatibility, but reject unsigned v2
|
||||
/// requests that omit it or present a different nonce.
|
||||
pub fn verify_tonic_mutation_body_digest_reject_unsigned<T>(
|
||||
request: &tonic::Request<T>,
|
||||
canonical_body: &[u8],
|
||||
) -> std::io::Result<()> {
|
||||
let version = request
|
||||
.metadata()
|
||||
.get(RPC_AUTH_VERSION_HEADER)
|
||||
.and_then(|value| value.to_str().ok());
|
||||
let digest = request
|
||||
.metadata()
|
||||
.get(RPC_CONTENT_SHA256_HEADER)
|
||||
.and_then(|value| value.to_str().ok());
|
||||
let nonce = request.metadata().get(RPC_NONCE_HEADER).and_then(|value| value.to_str().ok());
|
||||
if version == Some(RPC_AUTH_VERSION_V2) && digest == Some(UNSIGNED_PAYLOAD) && nonce != Some("unsigned") {
|
||||
return Err(std::io::Error::other("RPC mutation requires a body-bound v2 signature"));
|
||||
}
|
||||
verify_tonic_mutation_body_digest(request, canonical_body)
|
||||
}
|
||||
|
||||
/// [`verify_tonic_mutation_body_digest`] with the strict gate injected as a parameter, so both
|
||||
/// rollout postures are unit-testable without racing on process-global environment variables.
|
||||
fn verify_tonic_mutation_body_digest_with_strictness<T>(
|
||||
|
||||
@@ -36,6 +36,7 @@ use rustfs_rio::{ChunkReaderBox, HttpChunkReader, HttpReader, HttpWriter};
|
||||
use sha2::{Digest, Sha256};
|
||||
use std::collections::HashMap;
|
||||
use std::future::Future;
|
||||
use std::io;
|
||||
use std::pin::Pin;
|
||||
use std::sync::{Arc, LazyLock, OnceLock};
|
||||
use std::task::{Context, Poll};
|
||||
@@ -105,9 +106,13 @@ struct PutFileCapabilityCacheState {
|
||||
cached: Option<PutFileCapabilityState>,
|
||||
generation: u64,
|
||||
in_flight: Option<PutFileCapabilityFlight>,
|
||||
rejected_server_epoch: Option<Uuid>,
|
||||
}
|
||||
|
||||
type PutFileCapabilityCacheEntry = Arc<tokio::sync::RwLock<PutFileCapabilityCacheState>>;
|
||||
// The registry lock is released before taking an entry lock. Entry guards cover
|
||||
// only cache transitions, never a probe or await; poll-based writers must be
|
||||
// able to reject an epoch atomically with those transitions.
|
||||
type PutFileCapabilityCacheEntry = Arc<parking_lot::RwLock<PutFileCapabilityCacheState>>;
|
||||
|
||||
static PUT_FILE_CAPABILITY_CACHE: LazyLock<parking_lot::RwLock<HashMap<String, PutFileCapabilityCacheEntry>>> =
|
||||
LazyLock::new(|| parking_lot::RwLock::new(HashMap::new()));
|
||||
@@ -119,7 +124,7 @@ fn put_file_capability_cache_entry(endpoint: &str) -> PutFileCapabilityCacheEntr
|
||||
PUT_FILE_CAPABILITY_CACHE
|
||||
.write()
|
||||
.entry(endpoint.to_owned())
|
||||
.or_insert_with(|| Arc::new(tokio::sync::RwLock::new(PutFileCapabilityCacheState::default())))
|
||||
.or_insert_with(|| Arc::new(parking_lot::RwLock::new(PutFileCapabilityCacheState::default())))
|
||||
.clone()
|
||||
}
|
||||
|
||||
@@ -134,6 +139,23 @@ fn fresh_put_file_capability(state: Option<PutFileCapabilityState>, now: Instant
|
||||
}
|
||||
}
|
||||
|
||||
fn reject_put_file_server_epoch(endpoint: &str, server_epoch: Uuid) {
|
||||
let entry = PUT_FILE_CAPABILITY_CACHE.read().get(endpoint).cloned();
|
||||
if let Some(entry) = entry {
|
||||
let mut state = entry.write();
|
||||
if matches!(state.cached, Some(PutFileCapabilityState::V1 { server_epoch: cached, .. }) if cached == server_epoch) {
|
||||
state.rejected_server_epoch = Some(server_epoch);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
fn usable_put_file_capability(state: &PutFileCapabilityCacheState, now: Instant) -> Option<Option<Uuid>> {
|
||||
match fresh_put_file_capability(state.cached, now)? {
|
||||
Some(server_epoch) if state.rejected_server_epoch == Some(server_epoch) => None,
|
||||
capability => Some(capability),
|
||||
}
|
||||
}
|
||||
|
||||
fn put_file_capability_status_is_legacy(status: u16) -> bool {
|
||||
status == 404
|
||||
}
|
||||
@@ -322,13 +344,14 @@ impl InternodeDataTransport for TcpHttpInternodeDataTransport {
|
||||
|
||||
async fn open_write(&self, request: WriteStreamRequest) -> Result<FileWriter> {
|
||||
let server_epoch = self.put_file_auth_capability(&request.endpoint).await?;
|
||||
let nonce = server_epoch.map(|_| Uuid::new_v4());
|
||||
let url = build_put_file_stream_url(&request, nonce.zip(server_epoch));
|
||||
let auth_scope = server_epoch.map(|server_epoch| (Uuid::new_v4(), server_epoch));
|
||||
let url = build_put_file_stream_url(&request, auth_scope);
|
||||
let endpoint = request.endpoint;
|
||||
let mut headers = json_headers();
|
||||
build_auth_headers(&url, &Method::PUT, &mut headers)?;
|
||||
let writer = HttpWriter::new(url.clone(), Method::PUT, headers).await?;
|
||||
match nonce {
|
||||
Some(nonce) => Ok(Box::new(PutFileAuthWriter::new(writer, url, nonce))),
|
||||
match auth_scope {
|
||||
Some((nonce, server_epoch)) => Ok(Box::new(PutFileAuthWriter::new(writer, url, nonce, endpoint, server_epoch))),
|
||||
None => Ok(Box::new(writer)),
|
||||
}
|
||||
}
|
||||
@@ -498,15 +521,15 @@ where
|
||||
{
|
||||
let entry = put_file_capability_cache_entry(endpoint);
|
||||
{
|
||||
let state = entry.read().await;
|
||||
if let Some(cached) = fresh_put_file_capability(state.cached, Instant::now()) {
|
||||
let state = entry.read();
|
||||
if let Some(cached) = usable_put_file_capability(&state, Instant::now()) {
|
||||
return Ok(cached);
|
||||
}
|
||||
}
|
||||
|
||||
let flight = {
|
||||
let mut state = entry.write().await;
|
||||
if let Some(cached) = fresh_put_file_capability(state.cached, Instant::now()) {
|
||||
let mut state = entry.write();
|
||||
if let Some(cached) = usable_put_file_capability(&state, Instant::now()) {
|
||||
return Ok(cached);
|
||||
}
|
||||
if let Some(flight) = state.in_flight.clone() {
|
||||
@@ -532,7 +555,7 @@ where
|
||||
.await;
|
||||
|
||||
{
|
||||
let mut state = entry.write().await;
|
||||
let mut state = entry.write();
|
||||
let is_current_flight = state
|
||||
.in_flight
|
||||
.as_ref()
|
||||
@@ -540,6 +563,9 @@ where
|
||||
if is_current_flight {
|
||||
match outcome {
|
||||
Ok(Some(server_epoch)) => {
|
||||
if state.rejected_server_epoch != Some(*server_epoch) {
|
||||
state.rejected_server_epoch = None;
|
||||
}
|
||||
state.cached = Some(PutFileCapabilityState::V1 {
|
||||
server_epoch: *server_epoch,
|
||||
revalidate_after: Instant::now() + PUT_FILE_V1_CAPABILITY_TTL,
|
||||
@@ -630,17 +656,23 @@ struct PutFileAuthWriter<W> {
|
||||
inner: W,
|
||||
url: String,
|
||||
nonce: Uuid,
|
||||
endpoint: String,
|
||||
server_epoch: Uuid,
|
||||
server_epoch_rejected: bool,
|
||||
hasher: Sha256,
|
||||
trailer: Option<Vec<u8>>,
|
||||
trailer_offset: usize,
|
||||
}
|
||||
|
||||
impl<W> PutFileAuthWriter<W> {
|
||||
fn new(inner: W, url: String, nonce: Uuid) -> Self {
|
||||
fn new(inner: W, url: String, nonce: Uuid, endpoint: String, server_epoch: Uuid) -> Self {
|
||||
Self {
|
||||
inner,
|
||||
url,
|
||||
nonce,
|
||||
endpoint,
|
||||
server_epoch,
|
||||
server_epoch_rejected: false,
|
||||
hasher: Sha256::new(),
|
||||
trailer: None,
|
||||
trailer_offset: 0,
|
||||
@@ -656,6 +688,14 @@ impl<W> PutFileAuthWriter<W> {
|
||||
Ok(())
|
||||
}
|
||||
|
||||
fn reject_server_epoch_on_conflict(&mut self, error: &io::Error) {
|
||||
if self.server_epoch_rejected || !io_error_has_put_file_epoch_conflict(error) {
|
||||
return;
|
||||
}
|
||||
reject_put_file_server_epoch(&self.endpoint, self.server_epoch);
|
||||
self.server_epoch_rejected = true;
|
||||
}
|
||||
|
||||
fn poll_write_trailer(&mut self, cx: &mut Context<'_>) -> Poll<std::io::Result<()>>
|
||||
where
|
||||
W: AsyncWrite + Unpin,
|
||||
@@ -673,7 +713,10 @@ impl<W> PutFileAuthWriter<W> {
|
||||
)));
|
||||
}
|
||||
Poll::Ready(Ok(written)) => written,
|
||||
Poll::Ready(Err(err)) => return Poll::Ready(Err(err)),
|
||||
Poll::Ready(Err(err)) => {
|
||||
self.reject_server_epoch_on_conflict(&err);
|
||||
return Poll::Ready(Err(err));
|
||||
}
|
||||
Poll::Pending => return Poll::Pending,
|
||||
};
|
||||
self.trailer_offset += written;
|
||||
@@ -682,6 +725,15 @@ impl<W> PutFileAuthWriter<W> {
|
||||
}
|
||||
}
|
||||
|
||||
fn io_error_has_put_file_epoch_conflict(error: &io::Error) -> bool {
|
||||
error
|
||||
.get_ref()
|
||||
.and_then(|source| source.downcast_ref::<rustfs_rio::InternodeHttpError>())
|
||||
.is_some_and(
|
||||
|error| matches!(error.kind(), rustfs_rio::InternodeHttpErrorKind::HttpStatus(status) if status.as_u16() == 409),
|
||||
)
|
||||
}
|
||||
|
||||
impl<W> AsyncWrite for PutFileAuthWriter<W>
|
||||
where
|
||||
W: AsyncWrite + Unpin,
|
||||
@@ -698,12 +750,22 @@ where
|
||||
self.hasher.update(&buf[..written]);
|
||||
Poll::Ready(Ok(written))
|
||||
}
|
||||
other => other,
|
||||
Poll::Ready(Err(err)) => {
|
||||
self.reject_server_epoch_on_conflict(&err);
|
||||
Poll::Ready(Err(err))
|
||||
}
|
||||
Poll::Pending => Poll::Pending,
|
||||
}
|
||||
}
|
||||
|
||||
fn poll_flush(mut self: Pin<&mut Self>, cx: &mut Context<'_>) -> Poll<std::io::Result<()>> {
|
||||
Pin::new(&mut self.inner).poll_flush(cx)
|
||||
match Pin::new(&mut self.inner).poll_flush(cx) {
|
||||
Poll::Ready(Err(err)) => {
|
||||
self.reject_server_epoch_on_conflict(&err);
|
||||
Poll::Ready(Err(err))
|
||||
}
|
||||
other => other,
|
||||
}
|
||||
}
|
||||
|
||||
fn poll_shutdown(mut self: Pin<&mut Self>, cx: &mut Context<'_>) -> Poll<std::io::Result<()>> {
|
||||
@@ -712,7 +774,13 @@ where
|
||||
Poll::Ready(Err(err)) => return Poll::Ready(Err(err)),
|
||||
Poll::Pending => return Poll::Pending,
|
||||
}
|
||||
Pin::new(&mut self.inner).poll_shutdown(cx)
|
||||
match Pin::new(&mut self.inner).poll_shutdown(cx) {
|
||||
Poll::Ready(Err(err)) => {
|
||||
self.reject_server_epoch_on_conflict(&err);
|
||||
Poll::Ready(Err(err))
|
||||
}
|
||||
other => other,
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
@@ -840,7 +908,6 @@ mod tests {
|
||||
loop {
|
||||
let strong_count = entry
|
||||
.read()
|
||||
.await
|
||||
.in_flight
|
||||
.as_ref()
|
||||
.map(|flight| Arc::strong_count(&flight.outcome))
|
||||
@@ -858,6 +925,50 @@ mod tests {
|
||||
#[derive(Debug)]
|
||||
struct LegacyTestTransport;
|
||||
|
||||
#[derive(Clone, Copy, Debug)]
|
||||
enum PutFileFailurePhase {
|
||||
Write,
|
||||
Flush,
|
||||
Shutdown,
|
||||
}
|
||||
|
||||
struct PutFileFailureWriter {
|
||||
phase: PutFileFailurePhase,
|
||||
status: reqwest::StatusCode,
|
||||
}
|
||||
|
||||
impl PutFileFailureWriter {
|
||||
fn error(&self) -> io::Error {
|
||||
rustfs_rio::new_test_internode_http_io_error(rustfs_rio::InternodeHttpErrorKind::HttpStatus(self.status))
|
||||
}
|
||||
}
|
||||
|
||||
impl tokio::io::AsyncWrite for PutFileFailureWriter {
|
||||
fn poll_write(self: Pin<&mut Self>, _cx: &mut Context<'_>, buf: &[u8]) -> Poll<std::io::Result<usize>> {
|
||||
Poll::Ready(if matches!(self.phase, PutFileFailurePhase::Write) {
|
||||
Err(self.error())
|
||||
} else {
|
||||
Ok(buf.len())
|
||||
})
|
||||
}
|
||||
|
||||
fn poll_flush(self: Pin<&mut Self>, _cx: &mut Context<'_>) -> Poll<std::io::Result<()>> {
|
||||
Poll::Ready(if matches!(self.phase, PutFileFailurePhase::Flush) {
|
||||
Err(self.error())
|
||||
} else {
|
||||
Ok(())
|
||||
})
|
||||
}
|
||||
|
||||
fn poll_shutdown(self: Pin<&mut Self>, _cx: &mut Context<'_>) -> Poll<std::io::Result<()>> {
|
||||
Poll::Ready(if matches!(self.phase, PutFileFailurePhase::Shutdown) {
|
||||
Err(self.error())
|
||||
} else {
|
||||
Ok(())
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
#[async_trait::async_trait]
|
||||
impl InternodeDataTransport for LegacyTestTransport {
|
||||
async fn open_read(&self, _request: ReadStreamRequest) -> Result<FileReader> {
|
||||
@@ -1048,7 +1159,7 @@ mod tests {
|
||||
let v1_endpoint = format!("http://v1-{}.invalid", Uuid::new_v4());
|
||||
let v1_entry = put_file_capability_cache_entry(&v1_endpoint);
|
||||
let server_epoch = Uuid::new_v4();
|
||||
v1_entry.write().await.cached = Some(PutFileCapabilityState::V1 {
|
||||
v1_entry.write().cached = Some(PutFileCapabilityState::V1 {
|
||||
server_epoch,
|
||||
revalidate_after: Instant::now() + PUT_FILE_V1_CAPABILITY_TTL,
|
||||
});
|
||||
@@ -1067,7 +1178,7 @@ mod tests {
|
||||
Some(server_epoch)
|
||||
);
|
||||
assert!(!cache_probe_called.load(Ordering::SeqCst));
|
||||
v1_entry.write().await.cached = Some(PutFileCapabilityState::V1 {
|
||||
v1_entry.write().cached = Some(PutFileCapabilityState::V1 {
|
||||
server_epoch,
|
||||
revalidate_after: Instant::now(),
|
||||
});
|
||||
@@ -1086,8 +1197,7 @@ mod tests {
|
||||
|
||||
let legacy_endpoint = format!("http://legacy-{}.invalid", Uuid::new_v4());
|
||||
let legacy_entry = put_file_capability_cache_entry(&legacy_endpoint);
|
||||
legacy_entry.write().await.cached =
|
||||
Some(PutFileCapabilityState::LegacyUntil(Instant::now() + PUT_FILE_LEGACY_CAPABILITY_TTL));
|
||||
legacy_entry.write().cached = Some(PutFileCapabilityState::LegacyUntil(Instant::now() + PUT_FILE_LEGACY_CAPABILITY_TTL));
|
||||
assert!(
|
||||
transport
|
||||
.put_file_auth_capability(&legacy_endpoint)
|
||||
@@ -1098,7 +1208,7 @@ mod tests {
|
||||
|
||||
let expired_endpoint = format!("http://expired-legacy-{}.invalid", Uuid::new_v4());
|
||||
let expired_entry = put_file_capability_cache_entry(&expired_endpoint);
|
||||
expired_entry.write().await.cached = Some(PutFileCapabilityState::LegacyUntil(Instant::now()));
|
||||
expired_entry.write().cached = Some(PutFileCapabilityState::LegacyUntil(Instant::now()));
|
||||
let reprobed = std::sync::atomic::AtomicBool::new(false);
|
||||
assert_eq!(
|
||||
resolve_put_file_auth_capability(&expired_endpoint, || async {
|
||||
@@ -1349,7 +1459,7 @@ mod tests {
|
||||
};
|
||||
probe_started.notified().await;
|
||||
{
|
||||
let mut state = entry.write().await;
|
||||
let mut state = entry.write();
|
||||
state.generation = state.generation.checked_add(1).expect("test generation should advance");
|
||||
state.cached = Some(PutFileCapabilityState::V1 {
|
||||
server_epoch: newer_epoch,
|
||||
@@ -1362,10 +1472,7 @@ mod tests {
|
||||
task.await.expect("stale task should finish").expect("stale probe result"),
|
||||
Some(stale_epoch)
|
||||
);
|
||||
assert_eq!(
|
||||
fresh_put_file_capability(entry.read().await.cached, Instant::now()),
|
||||
Some(Some(newer_epoch))
|
||||
);
|
||||
assert_eq!(fresh_put_file_capability(entry.read().cached, Instant::now()), Some(Some(newer_epoch)));
|
||||
}
|
||||
|
||||
#[test]
|
||||
@@ -1398,6 +1505,8 @@ mod tests {
|
||||
|
||||
let _ = rustfs_credentials::set_global_rpc_secret("put-file-auth-writer-test-secret".to_string());
|
||||
let nonce = Uuid::parse_str("11111111-2222-4333-8444-555555555555").expect("nonce");
|
||||
let server_epoch = Uuid::parse_str("aaaaaaaa-bbbb-4ccc-8ddd-eeeeeeeeeeee").expect("server epoch");
|
||||
let endpoint = "http://node1:9000".to_string();
|
||||
let url = concat!(
|
||||
"http://node1:9000/rustfs/rpc/put_file_stream?disk=disk-a&volume=bucket&path=object%2Fpart.1",
|
||||
"&append=false&size=11&put_file_auth=digest-trailer-v1&put_file_nonce=11111111-2222-4333-8444-555555555555"
|
||||
@@ -1406,7 +1515,7 @@ mod tests {
|
||||
let mut sink = Vec::new();
|
||||
|
||||
{
|
||||
let mut writer = PutFileAuthWriter::new(&mut sink, url.clone(), nonce);
|
||||
let mut writer = PutFileAuthWriter::new(&mut sink, url.clone(), nonce, endpoint, server_epoch);
|
||||
writer.write_all(b"hello world").await.expect("body write should succeed");
|
||||
writer.shutdown().await.expect("shutdown should append auth trailer");
|
||||
let err = writer
|
||||
@@ -1424,6 +1533,143 @@ mod tests {
|
||||
assert_eq!(verified, expected_digest);
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
async fn put_file_auth_writer_reprobes_after_server_epoch_conflict() {
|
||||
use tokio::io::AsyncWriteExt;
|
||||
|
||||
let _ = rustfs_credentials::set_global_rpc_secret("put-file-epoch-conflict-test-secret".to_string());
|
||||
for status in [reqwest::StatusCode::CONFLICT, reqwest::StatusCode::BAD_REQUEST] {
|
||||
for (phase, trailer_write) in [
|
||||
(PutFileFailurePhase::Write, false),
|
||||
(PutFileFailurePhase::Write, true),
|
||||
(PutFileFailurePhase::Flush, false),
|
||||
(PutFileFailurePhase::Shutdown, false),
|
||||
] {
|
||||
let endpoint = format!("http://epoch-conflict-{}.invalid", Uuid::new_v4());
|
||||
let stale_epoch = Uuid::new_v4();
|
||||
let replacement_epoch = Uuid::new_v4();
|
||||
resolve_put_file_auth_capability(&endpoint, || async { Ok(Some(stale_epoch)) })
|
||||
.await
|
||||
.expect("initial capability should resolve");
|
||||
let mut writer = PutFileAuthWriter::new(
|
||||
PutFileFailureWriter { phase, status },
|
||||
format!("{endpoint}{PUT_FILE_AUTH_STREAM_PATH}"),
|
||||
Uuid::new_v4(),
|
||||
endpoint.clone(),
|
||||
stale_epoch,
|
||||
);
|
||||
let error = match (phase, trailer_write) {
|
||||
(PutFileFailurePhase::Write, false) => writer.write_all(b"body").await,
|
||||
(PutFileFailurePhase::Flush, _) => writer.flush().await,
|
||||
_ => writer.shutdown().await,
|
||||
}
|
||||
.expect_err("injected writer error must reach the caller");
|
||||
let conflict = status == reqwest::StatusCode::CONFLICT;
|
||||
assert_eq!(io_error_has_put_file_epoch_conflict(&error), conflict);
|
||||
|
||||
let probe_called = AtomicBool::new(false);
|
||||
let resolved = resolve_put_file_auth_capability(&endpoint, || async {
|
||||
probe_called.store(true, Ordering::SeqCst);
|
||||
Ok(Some(replacement_epoch))
|
||||
})
|
||||
.await
|
||||
.expect("capability should remain usable or be reprobed");
|
||||
assert_eq!(probe_called.load(Ordering::SeqCst), conflict, "phase={phase:?}, trailer={trailer_write}");
|
||||
assert_eq!(resolved, Some(if conflict { replacement_epoch } else { stale_epoch }));
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
async fn late_put_file_epoch_rejection_preserves_current_rejection() {
|
||||
let endpoint = format!("http://late-epoch-conflict-{}.invalid", Uuid::new_v4());
|
||||
let old_epoch = Uuid::new_v4();
|
||||
let current_epoch = Uuid::new_v4();
|
||||
let replacement_epoch = Uuid::new_v4();
|
||||
assert_eq!(
|
||||
resolve_put_file_auth_capability(&endpoint, || async { Ok(Some(old_epoch)) })
|
||||
.await
|
||||
.expect("initial epoch should be cached"),
|
||||
Some(old_epoch)
|
||||
);
|
||||
reject_put_file_server_epoch(&endpoint, old_epoch);
|
||||
assert_eq!(
|
||||
resolve_put_file_auth_capability(&endpoint, || async { Ok(Some(current_epoch)) })
|
||||
.await
|
||||
.expect("first restart should install a new epoch"),
|
||||
Some(current_epoch)
|
||||
);
|
||||
|
||||
reject_put_file_server_epoch(&endpoint, current_epoch);
|
||||
// A writer opened before the first restart can report its 409 after
|
||||
// a newer writer has already rejected the second server incarnation.
|
||||
reject_put_file_server_epoch(&endpoint, old_epoch);
|
||||
let probe_called = AtomicBool::new(false);
|
||||
let resolved = resolve_put_file_auth_capability(&endpoint, || async {
|
||||
probe_called.store(true, Ordering::SeqCst);
|
||||
Ok(Some(replacement_epoch))
|
||||
})
|
||||
.await
|
||||
.expect("late old-epoch rejection must preserve the current rejection");
|
||||
|
||||
assert!(probe_called.load(Ordering::SeqCst), "known-rejected current epoch must be reprobed");
|
||||
assert_eq!(resolved, Some(replacement_epoch));
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
async fn put_file_epoch_rejection_is_endpoint_and_epoch_scoped() {
|
||||
let endpoint = format!("http://scoped-epoch-{}.invalid", Uuid::new_v4());
|
||||
let other_endpoint = format!("http://other-epoch-{}.invalid", Uuid::new_v4());
|
||||
let current_epoch = Uuid::new_v4();
|
||||
for endpoint in [&endpoint, &other_endpoint] {
|
||||
resolve_put_file_auth_capability(endpoint, || async { Ok(Some(current_epoch)) })
|
||||
.await
|
||||
.expect("initial epoch should resolve");
|
||||
}
|
||||
reject_put_file_server_epoch(&endpoint, Uuid::new_v4());
|
||||
assert_eq!(
|
||||
resolve_put_file_auth_capability(&endpoint, || async { panic!("old writer must not invalidate a new epoch") })
|
||||
.await
|
||||
.expect("new epoch must remain cached"),
|
||||
Some(current_epoch)
|
||||
);
|
||||
reject_put_file_server_epoch(&endpoint, current_epoch);
|
||||
assert_eq!(
|
||||
resolve_put_file_auth_capability(&other_endpoint, || async { panic!("another endpoint must stay cached") })
|
||||
.await
|
||||
.expect("other endpoint must remain cached"),
|
||||
Some(current_epoch)
|
||||
);
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
async fn put_file_rejected_epoch_survives_failed_stale_and_downgrade_probes() {
|
||||
let endpoint = format!("http://rejected-probe-{}.invalid", Uuid::new_v4());
|
||||
let rejected_epoch = Uuid::new_v4();
|
||||
let replacement_epoch = Uuid::new_v4();
|
||||
resolve_put_file_auth_capability(&endpoint, || async { Ok(Some(rejected_epoch)) })
|
||||
.await
|
||||
.expect("initial epoch should resolve");
|
||||
reject_put_file_server_epoch(&endpoint, rejected_epoch);
|
||||
let failure = resolve_put_file_auth_capability(&endpoint, || async { Err(Error::other("injected probe failure")) })
|
||||
.await
|
||||
.expect_err("probe failure must be returned");
|
||||
assert!(failure.to_string().contains("injected probe failure"));
|
||||
let downgrade = resolve_put_file_auth_capability(&endpoint, || async { Ok(None) })
|
||||
.await
|
||||
.expect_err("rejection must not unpin authenticated v1");
|
||||
assert!(downgrade.to_string().contains("downgrade rejected"));
|
||||
resolve_put_file_auth_capability(&endpoint, || async { Ok(Some(rejected_epoch)) })
|
||||
.await
|
||||
.expect("a probe racing a restart can still return the old epoch");
|
||||
assert_eq!(
|
||||
resolve_put_file_auth_capability(&endpoint, || async { Ok(Some(replacement_epoch)) })
|
||||
.await
|
||||
.expect("same-epoch probe must not clear known rejection"),
|
||||
Some(replacement_epoch)
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn walk_dir_url_encodes_disk_ref() {
|
||||
let url = build_walk_dir_url(&WalkDirStreamRequest {
|
||||
|
||||
@@ -39,8 +39,8 @@ pub use http_auth::{
|
||||
sign_tonic_rpc_response_proof, tonic_boot_epoch_challenge, tonic_boot_epoch_response_headers, tonic_rpc_auth_failure_reason,
|
||||
verify_ns_scanner_capability, verify_ns_scanner_capability_with_tier_registry_generation, verify_put_file_auth_trailer,
|
||||
verify_put_file_capability, verify_rpc_signature, verify_tonic_boot_epoch_response, verify_tonic_canonical_body_digest,
|
||||
verify_tonic_mutation_body_digest, verify_tonic_rpc_response_proof, verify_tonic_rpc_signature,
|
||||
verify_tonic_rpc_signature_with_bootstrap,
|
||||
verify_tonic_mutation_body_digest, verify_tonic_mutation_body_digest_reject_unsigned, verify_tonic_rpc_response_proof,
|
||||
verify_tonic_rpc_signature, verify_tonic_rpc_signature_with_bootstrap,
|
||||
};
|
||||
#[cfg(test)]
|
||||
pub(crate) use internode_data_transport::TcpHttpInternodeDataTransport;
|
||||
|
||||
@@ -49,8 +49,8 @@ use rustfs_protos::proto_gen::node_service::{
|
||||
ScannerActivityRequest, ScannerActivityResponse, ScannerPublicationLeaseReleaseRequest, ScannerPublicationLeaseRequest,
|
||||
ScannerPublicationLeaseResponse, ServerInfoRequest, SignalServiceRequest, SignalServiceResponse, StartDecommissionRequest,
|
||||
StartProfilingRequest, StopRebalanceRequest, TierMutationAbortRequest, TierMutationCommitRequest,
|
||||
TierMutationControlResponse, TierMutationPeerState, TierMutationPrepareRequest, node_service_client::NodeServiceClient,
|
||||
tier_mutation_control_service_client::TierMutationControlServiceClient,
|
||||
TierMutationControlResponse, TierMutationFailureClass, TierMutationPeerState, TierMutationPrepareRequest,
|
||||
node_service_client::NodeServiceClient, tier_mutation_control_service_client::TierMutationControlServiceClient,
|
||||
};
|
||||
pub use rustfs_protos::{PEER_RESTDRY_RUN, PEER_RESTSIGNAL, PEER_RESTSUB_SYS};
|
||||
use rustfs_protos::{TierMutationRpcPhase, evict_failed_connection};
|
||||
@@ -122,6 +122,14 @@ fn control_plane_failure(op: &str, bucket: Option<&str>, error_code: Option<i32>
|
||||
if error_code == Some(rustfs_protos::proto_gen::node_service::ControlPlaneErrorCode::ControlPlaneErrorNotInitialized as i32) {
|
||||
return Error::RemoteNotInitialized;
|
||||
}
|
||||
if error_code == Some(rustfs_protos::proto_gen::node_service::ControlPlaneErrorCode::ControlPlaneErrorInvalidArgument as i32)
|
||||
{
|
||||
return Error::InvalidArgument(
|
||||
"control-plane".to_string(),
|
||||
op.to_string(),
|
||||
error_info.unwrap_or_else(|| format!("{op}: peer rejected invalid argument without details")),
|
||||
);
|
||||
}
|
||||
match error_info {
|
||||
Some(msg) => Error::other(msg),
|
||||
None => peer_failure_without_details(op, bucket),
|
||||
@@ -454,6 +462,31 @@ pub struct PeerTierMutationOutcome {
|
||||
pub applied: bool,
|
||||
}
|
||||
|
||||
#[derive(Debug, thiserror::Error)]
|
||||
#[error("{message}")]
|
||||
struct TierMutationDefinitelyRejected {
|
||||
message: String,
|
||||
}
|
||||
|
||||
fn tier_mutation_definitely_rejected_error(message: String) -> Error {
|
||||
Error::other(TierMutationDefinitelyRejected { message })
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
pub(crate) fn test_tier_mutation_definitely_rejected_error(message: &str) -> Error {
|
||||
tier_mutation_definitely_rejected_error(message.to_string())
|
||||
}
|
||||
|
||||
pub(crate) fn tier_mutation_error_is_definitely_rejected(error: &Error) -> bool {
|
||||
matches!(
|
||||
error,
|
||||
Error::Io(io_error)
|
||||
if io_error
|
||||
.get_ref()
|
||||
.is_some_and(|source| source.downcast_ref::<TierMutationDefinitelyRejected>().is_some())
|
||||
)
|
||||
}
|
||||
|
||||
fn validate_tier_mutation_response_proof(
|
||||
version: u32,
|
||||
phase: TierMutationRpcPhase,
|
||||
@@ -461,6 +494,16 @@ fn validate_tier_mutation_response_proof(
|
||||
canonical_payload: &[u8],
|
||||
response: &TierMutationControlResponse,
|
||||
) -> Result<()> {
|
||||
if response.response_proof.len() > rustfs_protos::TIER_MUTATION_RPC_MAX_RESPONSE_PROOF_SIZE {
|
||||
return Err(Error::other("peer tier mutation response proof exceeds size limit"));
|
||||
}
|
||||
if response
|
||||
.error_info
|
||||
.as_ref()
|
||||
.is_some_and(|error| error.len() > rustfs_protos::TIER_MUTATION_RPC_MAX_ERROR_INFO_SIZE)
|
||||
{
|
||||
return Err(Error::other("peer tier mutation error response exceeds size limit"));
|
||||
}
|
||||
let canonical_response =
|
||||
rustfs_protos::canonical_tier_mutation_rpc_response_body(rustfs_protos::TierMutationRpcResponseProofInput {
|
||||
version,
|
||||
@@ -471,6 +514,7 @@ fn validate_tier_mutation_response_proof(
|
||||
state: response.state,
|
||||
applied: response.applied,
|
||||
error_info: response.error_info.as_deref(),
|
||||
failure_class: response.failure_class,
|
||||
})
|
||||
.map_err(|_| Error::other("tier mutation response length cannot be represented"))?;
|
||||
verify_tonic_rpc_response_proof(&canonical_response, &response.response_proof)
|
||||
@@ -492,9 +536,9 @@ fn validate_tier_mutation_payload_len(phase: TierMutationRpcPhase, payload_len:
|
||||
TierMutationRpcPhase::Commit => rustfs_protos::TIER_MUTATION_RPC_MAX_COMMIT_PAYLOAD_SIZE,
|
||||
TierMutationRpcPhase::Abort => {
|
||||
if payload_len == 0 {
|
||||
return Ok(());
|
||||
return Err(Error::other("tier mutation abort payload is empty"));
|
||||
}
|
||||
return Err(Error::other("tier mutation abort payload must be empty"));
|
||||
rustfs_protos::TIER_MUTATION_RPC_MAX_ABORT_PAYLOAD_SIZE
|
||||
}
|
||||
_ => return Err(Error::other("tier mutation rpc phase is unsupported")),
|
||||
};
|
||||
@@ -513,8 +557,29 @@ fn tier_mutation_phase_label(phase: TierMutationRpcPhase) -> &'static str {
|
||||
}
|
||||
}
|
||||
|
||||
fn tier_mutation_control_status_error(phase: TierMutationRpcPhase, status: tonic::Status) -> Error {
|
||||
Error::other(format!("peer tier mutation {} RPC failed: {status}", tier_mutation_phase_label(phase)))
|
||||
fn tier_mutation_control_status_error(phase: TierMutationRpcPhase, requested_version: u32, status: tonic::Status) -> Error {
|
||||
let message = format!("peer tier mutation {} RPC failed: {status}", tier_mutation_phase_label(phase));
|
||||
let legacy_rejection = format!("unsupported tier mutation peer protocol version: {requested_version}");
|
||||
// RUSTFS_COMPAT_TODO(backlog-2097-tier-mutation-v4-error-text): retain this exact v3-server rejection classifier for mixed-version peers. Remove after every supported peer returns the signed v4 failure class.
|
||||
if requested_version == rustfs_protos::TIER_MUTATION_RPC_PROTOCOL_VERSION
|
||||
&& status.code() == tonic::Code::FailedPrecondition
|
||||
&& status.message().as_bytes() == legacy_rejection.as_bytes()
|
||||
{
|
||||
return tier_mutation_definitely_rejected_error(message);
|
||||
}
|
||||
Error::other(message)
|
||||
}
|
||||
|
||||
fn tier_mutation_failed_response_error(version: u32, failure_class: i32, error_info: Option<String>) -> Error {
|
||||
let message = error_info.unwrap_or_else(|| "peer tier mutation failed without an error".to_string());
|
||||
if version == rustfs_protos::TIER_MUTATION_RPC_PROTOCOL_VERSION
|
||||
&& TierMutationFailureClass::try_from(failure_class).ok() == Some(TierMutationFailureClass::PreDispatchRejected)
|
||||
{
|
||||
return tier_mutation_definitely_rejected_error(message);
|
||||
}
|
||||
// Missing/zero, unknown, and explicit Ambiguous are deliberately the same
|
||||
// fail-closed result: the coordinator must include this peer in Abort.
|
||||
Error::other(message)
|
||||
}
|
||||
|
||||
impl PeerRestClient {
|
||||
@@ -1307,8 +1372,12 @@ impl PeerRestClient {
|
||||
.await
|
||||
}
|
||||
|
||||
pub async fn abort_tier_mutation(&self, mutation_id: Uuid) -> Result<PeerTierMutationOutcome> {
|
||||
self.tier_mutation_control(TierMutationRpcPhase::Abort, mutation_id, Bytes::new())
|
||||
pub async fn abort_tier_mutation(
|
||||
&self,
|
||||
mutation_id: Uuid,
|
||||
canonical_prepare_payload: Bytes,
|
||||
) -> Result<PeerTierMutationOutcome> {
|
||||
self.tier_mutation_control(TierMutationRpcPhase::Abort, mutation_id, canonical_prepare_payload)
|
||||
.await
|
||||
}
|
||||
|
||||
@@ -1342,7 +1411,7 @@ impl PeerRestClient {
|
||||
client
|
||||
.prepare_tier_mutation(request)
|
||||
.await
|
||||
.map_err(|status| tier_mutation_control_status_error(phase, status))?
|
||||
.map_err(|status| tier_mutation_control_status_error(phase, version, status))?
|
||||
.into_inner()
|
||||
}
|
||||
TierMutationRpcPhase::Commit => {
|
||||
@@ -1355,7 +1424,7 @@ impl PeerRestClient {
|
||||
client
|
||||
.commit_tier_mutation(request)
|
||||
.await
|
||||
.map_err(|status| tier_mutation_control_status_error(phase, status))?
|
||||
.map_err(|status| tier_mutation_control_status_error(phase, version, status))?
|
||||
.into_inner()
|
||||
}
|
||||
TierMutationRpcPhase::Abort => {
|
||||
@@ -1368,18 +1437,19 @@ impl PeerRestClient {
|
||||
client
|
||||
.abort_tier_mutation(request)
|
||||
.await
|
||||
.map_err(|status| tier_mutation_control_status_error(phase, status))?
|
||||
.map_err(|status| tier_mutation_control_status_error(phase, version, status))?
|
||||
.into_inner()
|
||||
}
|
||||
_ => return Err(Error::other("tier mutation rpc phase is unsupported")),
|
||||
};
|
||||
validate_tier_mutation_response_proof(version, phase, mutation_id, &canonical_payload, &response)?;
|
||||
if !response.success {
|
||||
return Err(Error::other(
|
||||
response
|
||||
.error_info
|
||||
.unwrap_or_else(|| "peer tier mutation failed without an error".to_string()),
|
||||
));
|
||||
return Err(tier_mutation_failed_response_error(version, response.failure_class, response.error_info));
|
||||
}
|
||||
if version == rustfs_protos::TIER_MUTATION_RPC_PROTOCOL_VERSION
|
||||
&& response.failure_class != TierMutationFailureClass::Unspecified as i32
|
||||
{
|
||||
return Err(Error::other("successful peer tier mutation response carried a failure class"));
|
||||
}
|
||||
let state = decode_tier_mutation_peer_state(response.state)?;
|
||||
Ok(PeerTierMutationOutcome {
|
||||
@@ -2335,6 +2405,29 @@ mod tests {
|
||||
use rustfs_protos::proto_gen::node_service::ControlPlaneErrorCode;
|
||||
assert_eq!(ControlPlaneErrorCode::ControlPlaneErrorUnspecified as i32, 0);
|
||||
assert_eq!(ControlPlaneErrorCode::ControlPlaneErrorNotInitialized as i32, 1);
|
||||
assert_eq!(ControlPlaneErrorCode::ControlPlaneErrorInvalidArgument as i32, 2);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn control_plane_failure_preserves_typed_invalid_argument_reason() {
|
||||
use rustfs_protos::proto_gen::node_service::ControlPlaneErrorCode;
|
||||
|
||||
let reason = "durable unresolved-entry recovery requires pool metadata V2 or V3";
|
||||
let err = control_plane_failure(
|
||||
"start_decommission",
|
||||
None,
|
||||
Some(ControlPlaneErrorCode::ControlPlaneErrorInvalidArgument as i32),
|
||||
Some(reason.to_string()),
|
||||
);
|
||||
|
||||
assert!(
|
||||
matches!(
|
||||
err,
|
||||
Error::InvalidArgument(ref scope, ref operation, ref actual_reason)
|
||||
if scope == "control-plane" && operation == "start_decommission" && actual_reason == reason
|
||||
),
|
||||
"forwarded validation failures must remain typed and actionable"
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
@@ -3301,6 +3394,7 @@ mod tests {
|
||||
state: i32,
|
||||
applied: bool,
|
||||
error_info: Option<&'a str>,
|
||||
failure_class: i32,
|
||||
}
|
||||
|
||||
fn signed_tier_mutation_response(input: TierMutationResponseFixture<'_>) -> TierMutationControlResponse {
|
||||
@@ -3314,6 +3408,7 @@ mod tests {
|
||||
state: input.state,
|
||||
applied: input.applied,
|
||||
error_info: input.error_info,
|
||||
failure_class: input.failure_class,
|
||||
})
|
||||
.expect("small tier mutation response should encode");
|
||||
let response_proof =
|
||||
@@ -3324,6 +3419,7 @@ mod tests {
|
||||
applied: input.applied,
|
||||
error_info: input.error_info.map(str::to_string),
|
||||
response_proof: response_proof.into(),
|
||||
failure_class: input.failure_class,
|
||||
}
|
||||
}
|
||||
|
||||
@@ -3341,6 +3437,7 @@ mod tests {
|
||||
state: TierMutationPeerState::Prepared as i32,
|
||||
applied: true,
|
||||
error_info: None,
|
||||
failure_class: TierMutationFailureClass::Unspecified as i32,
|
||||
});
|
||||
validate_tier_mutation_response_proof(
|
||||
rustfs_protos::TIER_MUTATION_RPC_PROTOCOL_VERSION,
|
||||
@@ -3365,6 +3462,10 @@ mod tests {
|
||||
applied: false,
|
||||
..response.clone()
|
||||
},
|
||||
TierMutationControlResponse {
|
||||
failure_class: TierMutationFailureClass::Ambiguous as i32,
|
||||
..response.clone()
|
||||
},
|
||||
] {
|
||||
let err = validate_tier_mutation_response_proof(
|
||||
rustfs_protos::TIER_MUTATION_RPC_PROTOCOL_VERSION,
|
||||
@@ -3388,6 +3489,44 @@ mod tests {
|
||||
assert!(err.to_string().contains("invalid tier mutation response proof"));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn tier_mutation_response_rejects_oversized_proof_and_error_before_verification() {
|
||||
let mutation_id = Uuid::new_v4();
|
||||
let payload = b"tier-mutation-prepare";
|
||||
let oversized_proof = TierMutationControlResponse {
|
||||
success: false,
|
||||
state: TierMutationPeerState::Unspecified as i32,
|
||||
applied: false,
|
||||
error_info: None,
|
||||
response_proof: vec![0; rustfs_protos::TIER_MUTATION_RPC_MAX_RESPONSE_PROOF_SIZE + 1].into(),
|
||||
failure_class: TierMutationFailureClass::Ambiguous as i32,
|
||||
};
|
||||
let err = validate_tier_mutation_response_proof(
|
||||
rustfs_protos::TIER_MUTATION_RPC_PROTOCOL_VERSION,
|
||||
TierMutationRpcPhase::Prepare,
|
||||
mutation_id,
|
||||
payload,
|
||||
&oversized_proof,
|
||||
)
|
||||
.expect_err("oversized proof must fail before cryptographic verification");
|
||||
assert!(err.to_string().contains("response proof exceeds size limit"));
|
||||
|
||||
let oversized_error = TierMutationControlResponse {
|
||||
response_proof: Bytes::new(),
|
||||
error_info: Some("e".repeat(rustfs_protos::TIER_MUTATION_RPC_MAX_ERROR_INFO_SIZE + 1)),
|
||||
..oversized_proof
|
||||
};
|
||||
let err = validate_tier_mutation_response_proof(
|
||||
rustfs_protos::TIER_MUTATION_RPC_PROTOCOL_VERSION,
|
||||
TierMutationRpcPhase::Prepare,
|
||||
mutation_id,
|
||||
payload,
|
||||
&oversized_error,
|
||||
)
|
||||
.expect_err("oversized error detail must fail before proof construction");
|
||||
assert!(err.to_string().contains("error response exceeds size limit"));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn tier_mutation_peer_state_decode_fails_closed() {
|
||||
assert_eq!(
|
||||
@@ -3432,8 +3571,17 @@ mod tests {
|
||||
)
|
||||
.is_err()
|
||||
);
|
||||
validate_tier_mutation_payload_len(TierMutationRpcPhase::Abort, 0).expect("empty abort payload should fit");
|
||||
assert!(validate_tier_mutation_payload_len(TierMutationRpcPhase::Abort, 1).is_err());
|
||||
assert!(validate_tier_mutation_payload_len(TierMutationRpcPhase::Abort, 0).is_err());
|
||||
validate_tier_mutation_payload_len(TierMutationRpcPhase::Abort, 1).expect("non-empty abort payload should fit");
|
||||
validate_tier_mutation_payload_len(TierMutationRpcPhase::Abort, rustfs_protos::TIER_MUTATION_RPC_MAX_ABORT_PAYLOAD_SIZE)
|
||||
.expect("max abort payload should fit");
|
||||
assert!(
|
||||
validate_tier_mutation_payload_len(
|
||||
TierMutationRpcPhase::Abort,
|
||||
rustfs_protos::TIER_MUTATION_RPC_MAX_ABORT_PAYLOAD_SIZE + 1,
|
||||
)
|
||||
.is_err()
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
@@ -3448,7 +3596,7 @@ mod tests {
|
||||
tonic::Status::deadline_exceeded("peer tier mutation control timed out"),
|
||||
tonic::Status::unavailable("peer tier mutation control unavailable"),
|
||||
] {
|
||||
let err = tier_mutation_control_status_error(phase, status);
|
||||
let err = tier_mutation_control_status_error(phase, rustfs_protos::TIER_MUTATION_RPC_PROTOCOL_VERSION, status);
|
||||
let rendered = err.to_string();
|
||||
assert!(rendered.contains(&format!("peer tier mutation {label} RPC failed")), "{rendered}");
|
||||
assert!(
|
||||
@@ -3462,6 +3610,61 @@ mod tests {
|
||||
}
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn tier_mutation_v4_to_v3_rejection_classification_requires_exact_status_and_message() {
|
||||
let version = rustfs_protos::TIER_MUTATION_RPC_PROTOCOL_VERSION;
|
||||
let exact = format!("unsupported tier mutation peer protocol version: {version}");
|
||||
let rejected = tier_mutation_control_status_error(
|
||||
TierMutationRpcPhase::Prepare,
|
||||
version,
|
||||
tonic::Status::failed_precondition(exact.clone()),
|
||||
);
|
||||
assert!(tier_mutation_error_is_definitely_rejected(&rejected));
|
||||
|
||||
for status in [
|
||||
tonic::Status::failed_precondition(format!("{exact}.")),
|
||||
tonic::Status::failed_precondition(format!("unsupported tier mutation peer protocol version: {}", version - 1)),
|
||||
tonic::Status::invalid_argument(exact.clone()),
|
||||
tonic::Status::unimplemented(exact),
|
||||
] {
|
||||
let ambiguous = tier_mutation_control_status_error(TierMutationRpcPhase::Prepare, version, status);
|
||||
assert!(
|
||||
!tier_mutation_error_is_definitely_rejected(&ambiguous),
|
||||
"near-text, wrong-code, and Unimplemented failures must remain ambiguous"
|
||||
);
|
||||
}
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn tier_mutation_v4_failure_class_is_typed_and_fails_closed() {
|
||||
let version = rustfs_protos::TIER_MUTATION_RPC_PROTOCOL_VERSION;
|
||||
let rejected = tier_mutation_failed_response_error(
|
||||
version,
|
||||
TierMutationFailureClass::PreDispatchRejected as i32,
|
||||
Some("rejected".to_string()),
|
||||
);
|
||||
assert!(tier_mutation_error_is_definitely_rejected(&rejected));
|
||||
|
||||
for failure_class in [
|
||||
TierMutationFailureClass::Unspecified as i32,
|
||||
TierMutationFailureClass::Ambiguous as i32,
|
||||
99,
|
||||
] {
|
||||
let ambiguous = tier_mutation_failed_response_error(version, failure_class, None);
|
||||
assert!(
|
||||
!tier_mutation_error_is_definitely_rejected(&ambiguous),
|
||||
"missing, unknown, and explicit ambiguous classes must trigger Abort fanout"
|
||||
);
|
||||
}
|
||||
|
||||
let v3_ignores_v4_class = tier_mutation_failed_response_error(
|
||||
rustfs_protos::TIER_MUTATION_RPC_PREVIOUS_PROTOCOL_VERSION,
|
||||
TierMutationFailureClass::PreDispatchRejected as i32,
|
||||
Some("legacy failure".to_string()),
|
||||
);
|
||||
assert!(!tier_mutation_error_is_definitely_rejected(&v3_ignores_v4_class));
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
async fn peer_rest_client_rejects_oversized_tier_prepare_before_dialing() {
|
||||
let client = test_peer_client();
|
||||
|
||||
+6170
-449
File diff suppressed because it is too large
Load Diff
File diff suppressed because it is too large
Load Diff
@@ -37,8 +37,9 @@ use crate::{
|
||||
runtime::instance::{InstanceContext, bootstrap_ctx},
|
||||
runtime::sources as runtime_sources,
|
||||
set_disk::{PreparedGetObjectMetadata, SetDisks},
|
||||
store::init_format::{
|
||||
check_format_erasure_values, load_format_erasure_all, save_format_file, select_format_erasure_in_quorum,
|
||||
store::{
|
||||
RemoteTuplePublicationFence,
|
||||
init_format::{check_format_erasure_values, load_format_erasure_all, save_format_file, select_format_erasure_in_quorum},
|
||||
},
|
||||
};
|
||||
use futures::{
|
||||
@@ -625,6 +626,19 @@ impl Sets {
|
||||
.put_object_with_old_current_size(bucket, object, data, opts)
|
||||
.await
|
||||
}
|
||||
|
||||
pub(crate) async fn put_object_with_old_current_size_for_data_movement(
|
||||
&self,
|
||||
bucket: &str,
|
||||
object: &str,
|
||||
data: &mut PutObjReader,
|
||||
opts: &ObjectOptions,
|
||||
publication_fence: RemoteTuplePublicationFence,
|
||||
) -> Result<(ObjectInfo, Option<crate::disk::OldCurrentSize>)> {
|
||||
self.get_disks_by_key(object)
|
||||
.put_object_with_old_current_size_for_data_movement(bucket, object, data, opts, publication_fence)
|
||||
.await
|
||||
}
|
||||
}
|
||||
|
||||
#[async_trait::async_trait]
|
||||
@@ -1351,11 +1365,20 @@ pub(crate) async fn make_local_two_set_sets_with_ctx(ctx: Arc<InstanceContext>)
|
||||
pub(crate) async fn make_local_two_set_sets_for_pool_with_ctx(
|
||||
ctx: Arc<InstanceContext>,
|
||||
pool_idx: usize,
|
||||
) -> (Vec<tempfile::TempDir>, Arc<Sets>) {
|
||||
make_local_two_set_sets_for_pool_with_drive_count_and_ctx(ctx, pool_idx, 2).await
|
||||
}
|
||||
|
||||
#[cfg(any(test, feature = "test-util"))]
|
||||
pub(crate) async fn make_local_two_set_sets_for_pool_with_drive_count_and_ctx(
|
||||
ctx: Arc<InstanceContext>,
|
||||
pool_idx: usize,
|
||||
set_drive_count: usize,
|
||||
) -> (Vec<tempfile::TempDir>, Arc<Sets>) {
|
||||
use crate::layout::endpoint::Endpoint;
|
||||
use rustfs_lock::client::local::LocalClient;
|
||||
|
||||
let format = FormatV3::new(2, 2);
|
||||
let format = FormatV3::new(2, set_drive_count);
|
||||
let mut temp_dirs = Vec::new();
|
||||
let mut all_endpoints = Vec::new();
|
||||
let mut disk_sets = Vec::new();
|
||||
@@ -1363,7 +1386,7 @@ pub(crate) async fn make_local_two_set_sets_for_pool_with_ctx(
|
||||
for set_index in 0..2 {
|
||||
let mut endpoints = Vec::new();
|
||||
let mut disks = Vec::new();
|
||||
for disk_index in 0..2 {
|
||||
for disk_index in 0..set_drive_count {
|
||||
let temp_dir = tempfile::tempdir().expect("tempdir should be created");
|
||||
let mut endpoint = Endpoint::try_from(temp_dir.path().to_str().expect("tempdir path should be utf8"))
|
||||
.expect("endpoint should parse");
|
||||
@@ -1389,7 +1412,7 @@ pub(crate) async fn make_local_two_set_sets_for_pool_with_ctx(
|
||||
endpoints.push(endpoint);
|
||||
disks.push(Some(disk));
|
||||
}
|
||||
let lockers = (0..2)
|
||||
let lockers = (0..set_drive_count)
|
||||
.map(|_| {
|
||||
Arc::new(LocalClient::with_manager(Arc::new(rustfs_lock::GlobalLockManager::Enabled(Arc::new(
|
||||
rustfs_lock::FastObjectLockManager::new(),
|
||||
@@ -1400,7 +1423,7 @@ pub(crate) async fn make_local_two_set_sets_for_pool_with_ctx(
|
||||
SetDisks::new_with_instance_ctx(
|
||||
"test-owner".to_string(),
|
||||
Arc::new(RwLock::new(disks)),
|
||||
2,
|
||||
set_drive_count,
|
||||
1,
|
||||
set_index,
|
||||
pool_idx,
|
||||
@@ -1420,7 +1443,7 @@ pub(crate) async fn make_local_two_set_sets_for_pool_with_ctx(
|
||||
endpoints: PoolEndpoints {
|
||||
legacy: false,
|
||||
set_count: 2,
|
||||
drives_per_set: 2,
|
||||
drives_per_set: set_drive_count,
|
||||
endpoints: Endpoints::from(all_endpoints),
|
||||
cmd_line: String::new(),
|
||||
platform: String::new(),
|
||||
@@ -1428,7 +1451,7 @@ pub(crate) async fn make_local_two_set_sets_for_pool_with_ctx(
|
||||
format,
|
||||
parity_count: 1,
|
||||
set_count: 2,
|
||||
set_drive_count: 2,
|
||||
set_drive_count,
|
||||
default_parity_count: 1,
|
||||
distribution_algo: DistributionAlgoVersion::V1,
|
||||
exit_signal: None,
|
||||
|
||||
@@ -16,17 +16,18 @@
|
||||
|
||||
pub(crate) mod backpressure;
|
||||
|
||||
use crate::core::pools::{DecommissionCapacityOwner, decommission_capacity_mutation_id};
|
||||
use crate::error::{
|
||||
Error, Result, is_err_data_movement_overwrite, is_err_invalid_upload_id, is_err_object_not_found, is_err_version_not_found,
|
||||
};
|
||||
use crate::object_api::{GetObjectReader, ObjectInfo, ObjectOptions, PutObjReader};
|
||||
use crate::set_disk::{SetDisks, get_lock_acquire_timeout};
|
||||
use crate::storage_api_contracts::{
|
||||
multipart::{CompletePart, MultipartOperations as _},
|
||||
multipart::CompletePart,
|
||||
namespace::NamespaceLocking as _,
|
||||
object::{HTTPPreconditions, ObjectOperations as _},
|
||||
};
|
||||
use crate::store::{ECStore, ObjectLockDiagGuard, SourceCleanupMutationFence};
|
||||
use crate::store::{DecommissionFixedReadAnchor, ECStore, SourceCleanupMutationFence};
|
||||
use bytes::Bytes;
|
||||
use rustfs_filemeta::{FileInfo, FileInfoVersions, ObjectPartInfo};
|
||||
use rustfs_rio::{EtagResolvable, HashReader, HashReaderDetector, Index, TryGetIndex};
|
||||
@@ -160,6 +161,99 @@ pub fn mark_multipart_upload_completed(flag: &Arc<AtomicBool>) {
|
||||
flag.store(false, Ordering::Relaxed);
|
||||
}
|
||||
|
||||
#[cfg(all(test, feature = "test-util"))]
|
||||
struct DataMovementMultipartAbortBarrierState {
|
||||
bucket: String,
|
||||
object: String,
|
||||
arrived: tokio::sync::Notify,
|
||||
release: tokio::sync::Notify,
|
||||
}
|
||||
|
||||
#[cfg(all(test, feature = "test-util"))]
|
||||
pub(crate) struct DataMovementMultipartAbortBarrier {
|
||||
state: Arc<DataMovementMultipartAbortBarrierState>,
|
||||
}
|
||||
|
||||
#[cfg(all(test, feature = "test-util"))]
|
||||
static DATA_MOVEMENT_MULTIPART_ABORT_BARRIER: std::sync::OnceLock<
|
||||
std::sync::Mutex<Option<Arc<DataMovementMultipartAbortBarrierState>>>,
|
||||
> = std::sync::OnceLock::new();
|
||||
|
||||
#[cfg(all(test, feature = "test-util"))]
|
||||
impl DataMovementMultipartAbortBarrier {
|
||||
pub(crate) fn install(bucket: &str, object: &str) -> Self {
|
||||
let state = Arc::new(DataMovementMultipartAbortBarrierState {
|
||||
bucket: bucket.to_string(),
|
||||
object: object.to_string(),
|
||||
arrived: tokio::sync::Notify::new(),
|
||||
release: tokio::sync::Notify::new(),
|
||||
});
|
||||
let mut slot = DATA_MOVEMENT_MULTIPART_ABORT_BARRIER
|
||||
.get_or_init(|| std::sync::Mutex::new(None))
|
||||
.lock()
|
||||
.expect("data movement multipart abort barrier mutex should not poison");
|
||||
assert!(slot.is_none(), "data movement multipart abort barrier must be unique");
|
||||
*slot = Some(Arc::clone(&state));
|
||||
Self { state }
|
||||
}
|
||||
|
||||
pub(crate) async fn wait_until_paused(&self) {
|
||||
tokio::time::timeout(StdDuration::from_secs(30), self.state.arrived.notified())
|
||||
.await
|
||||
.expect("data movement multipart failure should reach abort cleanup");
|
||||
}
|
||||
}
|
||||
|
||||
#[cfg(all(test, feature = "test-util"))]
|
||||
impl Drop for DataMovementMultipartAbortBarrier {
|
||||
fn drop(&mut self) {
|
||||
self.state.release.notify_one();
|
||||
let mut slot = DATA_MOVEMENT_MULTIPART_ABORT_BARRIER
|
||||
.get_or_init(|| std::sync::Mutex::new(None))
|
||||
.lock()
|
||||
.expect("data movement multipart abort barrier mutex should not poison");
|
||||
if slot.as_ref().is_some_and(|state| Arc::ptr_eq(state, &self.state)) {
|
||||
*slot = None;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
#[cfg(all(test, feature = "test-util"))]
|
||||
async fn pause_data_movement_multipart_before_abort(bucket: &str, object: &str) {
|
||||
let barrier = DATA_MOVEMENT_MULTIPART_ABORT_BARRIER
|
||||
.get_or_init(|| std::sync::Mutex::new(None))
|
||||
.lock()
|
||||
.expect("data movement multipart abort barrier mutex should not poison")
|
||||
.as_ref()
|
||||
.filter(|barrier| barrier.bucket == bucket && barrier.object == object)
|
||||
.cloned();
|
||||
if let Some(barrier) = barrier {
|
||||
barrier.arrived.notify_one();
|
||||
barrier.release.notified().await;
|
||||
}
|
||||
}
|
||||
|
||||
fn data_movement_abort_opts(
|
||||
src_pool_idx: usize,
|
||||
expected_bucket_incarnation_id: Option<uuid::Uuid>,
|
||||
lock_lost_signal: Option<&Arc<rustfs_lock::distributed_lock::LockLostSignal>>,
|
||||
capacity_owner: Option<DecommissionCapacityOwner>,
|
||||
) -> ObjectOptions {
|
||||
let mut opts = ObjectOptions {
|
||||
data_movement: true,
|
||||
src_pool_idx,
|
||||
expected_bucket_incarnation_id,
|
||||
..Default::default()
|
||||
};
|
||||
if let Some(capacity_owner) = capacity_owner {
|
||||
capacity_owner.apply_to(&mut opts);
|
||||
}
|
||||
if let Some(signal) = lock_lost_signal {
|
||||
opts.add_namespace_lock_lost_signal(Arc::clone(signal));
|
||||
}
|
||||
opts
|
||||
}
|
||||
|
||||
fn insert_data_movement_checksum(user_defined: &mut HashMap<String, String>, object_info: &ObjectInfo) {
|
||||
rustfs_utils::http::remove_header_map(user_defined, rustfs_utils::http::SUFFIX_REPLICATION_SSEC_CRC);
|
||||
if let Some(checksum) = object_info.checksum.as_ref().filter(|checksum| !checksum.is_empty()) {
|
||||
@@ -192,7 +286,7 @@ fn data_movement_new_multipart_opts(object_info: &ObjectInfo, src_pool_idx: usiz
|
||||
preserve_etag: object_info.etag.clone(),
|
||||
src_pool_idx,
|
||||
data_movement: true,
|
||||
..Default::default()
|
||||
..ObjectOptions::with_capacity_expected_data_bytes(usize::try_from(object_info.size).ok())
|
||||
}
|
||||
}
|
||||
|
||||
@@ -363,7 +457,7 @@ fn data_movement_complete_multipart_opts(
|
||||
preserve_etag: object_info.etag.clone(),
|
||||
user_defined,
|
||||
src_pool_idx,
|
||||
..Default::default()
|
||||
..ObjectOptions::with_capacity_expected_data_bytes(usize::try_from(object_info.size).ok())
|
||||
})
|
||||
}
|
||||
|
||||
@@ -533,6 +627,7 @@ fn schedule_data_movement_multipart_abort_cleanup(
|
||||
bucket: String,
|
||||
object: String,
|
||||
upload_id: String,
|
||||
opts: ObjectOptions,
|
||||
op_label: &str,
|
||||
) {
|
||||
let op_label = op_label.to_string();
|
||||
@@ -540,23 +635,32 @@ fn schedule_data_movement_multipart_abort_cleanup(
|
||||
for attempt in 1..=DATA_MOVEMENT_MULTIPART_ABORT_RETRY_ATTEMPTS {
|
||||
tokio::time::sleep(StdDuration::from_secs(DATA_MOVEMENT_MULTIPART_ABORT_RETRY_DELAY_SECS)).await;
|
||||
|
||||
let Some(pool) = store.pools.get(target_pool_idx).cloned() else {
|
||||
if store.pools.get(target_pool_idx).is_none() {
|
||||
error!(
|
||||
"{op_label}: background abort_multipart_upload cleanup skipped for {bucket}/{object} upload {upload_id}: target pool {target_pool_idx} is out of range"
|
||||
);
|
||||
return;
|
||||
}
|
||||
|
||||
let mut cleanup_opts = opts.clone();
|
||||
let _multipart_mutation_fence = match DecommissionCapacityOwner::from_options(&cleanup_opts) {
|
||||
Some(owner) => match store.acquire_decommission_multipart_mutation_fence(owner).await {
|
||||
Ok(fence) => {
|
||||
fence.add_namespace_lock_fence(&mut cleanup_opts);
|
||||
Some(fence)
|
||||
}
|
||||
Err(err) => {
|
||||
error!(
|
||||
"{op_label}: background abort_multipart_upload cleanup could not fence {bucket}/{object} upload {upload_id} on attempt {attempt}: {err:?}"
|
||||
);
|
||||
continue;
|
||||
}
|
||||
},
|
||||
None => None,
|
||||
};
|
||||
|
||||
match pool
|
||||
.abort_multipart_upload(
|
||||
&bucket,
|
||||
&object,
|
||||
&upload_id,
|
||||
&ObjectOptions {
|
||||
data_movement: true,
|
||||
..Default::default()
|
||||
},
|
||||
)
|
||||
match store
|
||||
.abort_multipart_upload_for_data_movement(target_pool_idx, &bucket, &object, &upload_id, &cleanup_opts)
|
||||
.await
|
||||
{
|
||||
Ok(()) => {
|
||||
@@ -1334,27 +1438,43 @@ fn resolve_data_movement_overwrite_resume_result_for(
|
||||
Ok(matches!(err, Error::PreconditionFailed) && is_superseding_unversioned_data_movement_object(source, &target))
|
||||
}
|
||||
|
||||
#[derive(Clone, Copy)]
|
||||
struct DataMovementOverwriteCapacity {
|
||||
owner: Option<DecommissionCapacityOwner>,
|
||||
expected_data_bytes: Option<usize>,
|
||||
}
|
||||
|
||||
async fn should_treat_data_movement_overwrite_as_complete(
|
||||
store: &ECStore,
|
||||
src_pool_idx: usize,
|
||||
target_pool_idx: usize,
|
||||
pool_indices: (usize, usize),
|
||||
bucket: &str,
|
||||
object_info: &ObjectInfo,
|
||||
err: &Error,
|
||||
compare_part_checksums: bool,
|
||||
capacity: DataMovementOverwriteCapacity,
|
||||
) -> Result<bool> {
|
||||
if !should_check_data_movement_overwrite_resume(err) {
|
||||
return Ok(false);
|
||||
}
|
||||
let (src_pool_idx, target_pool_idx) = pool_indices;
|
||||
|
||||
resolve_data_movement_overwrite_resume_result_for(
|
||||
let equivalent = resolve_data_movement_overwrite_resume_result_for(
|
||||
err,
|
||||
find_data_movement_target_info(store, target_pool_idx, bucket, object_info).await,
|
||||
object_info,
|
||||
src_pool_idx,
|
||||
target_pool_idx,
|
||||
compare_part_checksums,
|
||||
)
|
||||
)?;
|
||||
if equivalent && let Some(owner) = capacity.owner {
|
||||
let expected_data_bytes = capacity
|
||||
.expected_data_bytes
|
||||
.ok_or_else(|| Error::other("equivalent data-movement target cannot reconcile unknown committed data size"))?;
|
||||
store
|
||||
.reconcile_decommission_capacity_after_equivalent_target(owner, target_pool_idx, expected_data_bytes)
|
||||
.await?;
|
||||
}
|
||||
Ok(equivalent)
|
||||
}
|
||||
|
||||
fn data_movement_part_stage_error(
|
||||
@@ -1395,9 +1515,10 @@ pub(crate) async fn migrate_decommission_object(
|
||||
rd: GetObjectReader,
|
||||
source_bucket_incarnation_id: Option<uuid::Uuid>,
|
||||
op_label: &str,
|
||||
capacity_owner: Option<DecommissionCapacityOwner>,
|
||||
) -> Result<()> {
|
||||
let source = rd.object_info.clone();
|
||||
let _mutation_fence = store
|
||||
let mutation_fence = store
|
||||
.acquire_decommission_object_mutation_fence(&bucket, &source.name)
|
||||
.await?;
|
||||
let current = find_data_movement_target_info(store.as_ref(), pool_idx, &bucket, &source)
|
||||
@@ -1415,7 +1536,8 @@ pub(crate) async fn migrate_decommission_object(
|
||||
source_bucket_incarnation_id,
|
||||
op_label,
|
||||
None,
|
||||
Some(&_mutation_fence),
|
||||
capacity_owner,
|
||||
Some(mutation_fence),
|
||||
)
|
||||
.await
|
||||
}
|
||||
@@ -1451,6 +1573,7 @@ pub(crate) async fn migrate_object_with_lock_lost_signal(
|
||||
op_label,
|
||||
lock_lost_signal,
|
||||
None,
|
||||
None,
|
||||
)
|
||||
.await
|
||||
}
|
||||
@@ -1464,24 +1587,117 @@ async fn migrate_object_inner(
|
||||
source_bucket_incarnation_id: Option<uuid::Uuid>,
|
||||
op_label: &str,
|
||||
lock_lost_signal: Option<Arc<rustfs_lock::distributed_lock::LockLostSignal>>,
|
||||
mutation_fence: Option<&ObjectLockDiagGuard>,
|
||||
capacity_owner: Option<DecommissionCapacityOwner>,
|
||||
mutation_fence: Option<DecommissionFixedReadAnchor>,
|
||||
) -> Result<()> {
|
||||
let mut mutation_fence = mutation_fence;
|
||||
let object_info = rd.object_info.clone();
|
||||
let capacity_owner = capacity_owner.map(|owner| {
|
||||
let version_id = object_info.version_id.map(|version_id| version_id.to_string());
|
||||
let mutation_id = owner.mutation_id.unwrap_or_else(|| {
|
||||
decommission_capacity_mutation_id(
|
||||
owner,
|
||||
&bucket,
|
||||
&object_info.name,
|
||||
version_id.as_deref(),
|
||||
object_info.delete_marker,
|
||||
object_info.mod_time,
|
||||
)
|
||||
});
|
||||
owner.with_mutation_id(mutation_id)
|
||||
});
|
||||
// Capture the exact source/tier identity before any client-paced read, but
|
||||
// defer both the tier lease and source/target write locks to the final
|
||||
// publication. Decommission already owns main's fixed-domain mutation
|
||||
// fence, so reacquiring that domain as a write lock would self-deadlock.
|
||||
let remote_tuple_publication_fence = store
|
||||
.acquire_remote_tuple_publication_fence(&bucket, pool_idx, &object_info, false)
|
||||
.await?;
|
||||
let has_part_checksums = object_info
|
||||
.parts
|
||||
.iter()
|
||||
.any(|part| part.checksums.as_ref().is_some_and(|checksums| !checksums.is_empty()));
|
||||
|
||||
let preserve_part_checksums = data_movement_part_checksum_writer_enabled();
|
||||
let capacity_expected_data_bytes = usize::try_from(object_info.size).ok();
|
||||
|
||||
if should_use_multipart_data_movement(&object_info, has_part_checksums) {
|
||||
// The decommission object fence already covers the source/target
|
||||
// namespace for this migration. Acquiring the synthetic multipart
|
||||
// fence while holding that read lock deadlocks local lock domains;
|
||||
// retain the extra fence only for callers without the outer fence.
|
||||
let multipart_mutation_fence = match (capacity_owner, mutation_fence.is_some()) {
|
||||
(Some(owner), false) => Some(store.acquire_decommission_multipart_mutation_fence(owner).await?),
|
||||
_ => None,
|
||||
};
|
||||
let mut new_multipart_opts = data_movement_new_multipart_opts(&object_info, pool_idx);
|
||||
if let Some(capacity_owner) = capacity_owner {
|
||||
capacity_owner.apply_to(&mut new_multipart_opts);
|
||||
}
|
||||
new_multipart_opts.expected_bucket_incarnation_id = source_bucket_incarnation_id;
|
||||
if let Some(signal) = lock_lost_signal.as_ref() {
|
||||
new_multipart_opts.add_namespace_lock_lost_signal(Arc::clone(signal));
|
||||
}
|
||||
if let Some(fence) = multipart_mutation_fence.as_ref() {
|
||||
fence.add_namespace_lock_fence(&mut new_multipart_opts);
|
||||
}
|
||||
if let Some(owner) = capacity_owner {
|
||||
let existing_target_pool_idx = store
|
||||
.select_data_movement_pool_idx(&bucket, &object_info.name, -1, &new_multipart_opts, false)
|
||||
.await?;
|
||||
if existing_target_pool_idx != pool_idx
|
||||
&& let Some(target) =
|
||||
find_data_movement_target_info(store.as_ref(), existing_target_pool_idx, &bucket, &object_info).await?
|
||||
&& is_equivalent_data_movement_object_identity(&object_info, &target, true, preserve_part_checksums)
|
||||
{
|
||||
let expected_data_bytes = capacity_expected_data_bytes
|
||||
.ok_or_else(|| Error::other("equivalent multipart target cannot reconcile unknown committed data size"))?;
|
||||
store
|
||||
.reconcile_decommission_capacity_after_equivalent_target(owner, existing_target_pool_idx, expected_data_bytes)
|
||||
.await?;
|
||||
info!(
|
||||
"{op_label}: multipart upload restart reconciled equivalent target for {}/{}",
|
||||
bucket.as_str(),
|
||||
object_info.name.as_str()
|
||||
);
|
||||
return Ok(());
|
||||
}
|
||||
let mut cleanup_opts =
|
||||
data_movement_abort_opts(pool_idx, source_bucket_incarnation_id, lock_lost_signal.as_ref(), capacity_owner);
|
||||
if let Some(anchor) = mutation_fence.as_ref() {
|
||||
anchor.guard().add_namespace_lock_fence(&mut cleanup_opts);
|
||||
}
|
||||
if let Some(fence) = multipart_mutation_fence.as_ref() {
|
||||
fence.add_namespace_lock_fence(&mut cleanup_opts);
|
||||
}
|
||||
for target_pool_idx in store.decommission_capacity_cleanup_target_indices(owner).await? {
|
||||
store
|
||||
.reconcile_multipart_uploads_for_data_movement(
|
||||
target_pool_idx,
|
||||
&bucket,
|
||||
&object_info.name,
|
||||
&data_movement_upload_identity(&object_info),
|
||||
&cleanup_opts,
|
||||
)
|
||||
.await
|
||||
.map_err(|err| {
|
||||
data_movement_stage_error(
|
||||
op_label,
|
||||
"reconcile_multipart_upload",
|
||||
bucket.as_str(),
|
||||
object_info.name.as_str(),
|
||||
err,
|
||||
)
|
||||
})?;
|
||||
}
|
||||
}
|
||||
let (res, target_pool_idx, expected_bucket_incarnation_id) = match store
|
||||
.handle_new_multipart_upload_with_pool_idx(&bucket, &object_info.name, &new_multipart_opts, mutation_fence)
|
||||
.handle_new_multipart_upload_with_pool_idx(
|
||||
&bucket,
|
||||
&object_info.name,
|
||||
&new_multipart_opts,
|
||||
mutation_fence.as_ref().map(DecommissionFixedReadAnchor::guard),
|
||||
)
|
||||
.await
|
||||
{
|
||||
Ok(res) => res,
|
||||
@@ -1532,9 +1748,15 @@ async fn migrate_object_inner(
|
||||
expected_bucket_incarnation_id,
|
||||
..Default::default()
|
||||
};
|
||||
if let Some(capacity_owner) = capacity_owner {
|
||||
capacity_owner.apply_to(&mut part_opts);
|
||||
}
|
||||
if let Some(signal) = lock_lost_signal.as_ref() {
|
||||
part_opts.add_namespace_lock_lost_signal(Arc::clone(signal));
|
||||
}
|
||||
if let Some(fence) = multipart_mutation_fence.as_ref() {
|
||||
fence.add_namespace_lock_fence(&mut part_opts);
|
||||
}
|
||||
let pi = match store
|
||||
.put_object_part_for_data_movement(
|
||||
target_pool_idx,
|
||||
@@ -1578,30 +1800,44 @@ async fn migrate_object_inner(
|
||||
err,
|
||||
)
|
||||
})?;
|
||||
if let Some(capacity_owner) = capacity_owner {
|
||||
capacity_owner.apply_to(&mut complete_multipart_opts);
|
||||
}
|
||||
complete_multipart_opts.expected_bucket_incarnation_id = expected_bucket_incarnation_id;
|
||||
if let Some(signal) = lock_lost_signal.as_ref() {
|
||||
complete_multipart_opts.add_namespace_lock_lost_signal(Arc::clone(signal));
|
||||
}
|
||||
if let Some(fence) = multipart_mutation_fence.as_ref() {
|
||||
fence.add_namespace_lock_fence(&mut complete_multipart_opts);
|
||||
}
|
||||
let remote_tuple_publication_fence = match mutation_fence.take() {
|
||||
Some(anchor) => remote_tuple_publication_fence.under_fixed_read_anchor(anchor)?,
|
||||
None => remote_tuple_publication_fence,
|
||||
};
|
||||
if let Err(err) = store
|
||||
.clone()
|
||||
.complete_multipart_upload_for_data_movement(
|
||||
(target_pool_idx, mutation_fence),
|
||||
.complete_multipart_upload_for_data_movement_with_publication_fence(
|
||||
target_pool_idx,
|
||||
&bucket,
|
||||
&object_info.name,
|
||||
&res.upload_id,
|
||||
parts,
|
||||
&complete_multipart_opts,
|
||||
remote_tuple_publication_fence,
|
||||
)
|
||||
.await
|
||||
{
|
||||
if should_treat_data_movement_overwrite_as_complete(
|
||||
store.as_ref(),
|
||||
pool_idx,
|
||||
target_pool_idx,
|
||||
(pool_idx, target_pool_idx),
|
||||
bucket.as_str(),
|
||||
&object_info,
|
||||
&err,
|
||||
preserve_part_checksums,
|
||||
DataMovementOverwriteCapacity {
|
||||
owner: capacity_owner,
|
||||
expected_data_bytes: capacity_expected_data_bytes,
|
||||
},
|
||||
)
|
||||
.await?
|
||||
{
|
||||
@@ -1629,31 +1865,37 @@ async fn migrate_object_inner(
|
||||
.await;
|
||||
|
||||
if multipart_result.is_ok() && should_abort_multipart_upload(&abort_multipart_flag) {
|
||||
let mut abort_opts =
|
||||
data_movement_abort_opts(pool_idx, expected_bucket_incarnation_id, lock_lost_signal.as_ref(), capacity_owner);
|
||||
if let Some(anchor) = mutation_fence.as_ref() {
|
||||
anchor.guard().add_namespace_lock_fence(&mut abort_opts);
|
||||
}
|
||||
if let Some(fence) = multipart_mutation_fence.as_ref() {
|
||||
fence.add_namespace_lock_fence(&mut abort_opts);
|
||||
}
|
||||
let abort_result = store
|
||||
.abort_multipart_upload_for_data_movement(target_pool_idx, &bucket, &object_info.name, &res.upload_id, &{
|
||||
let mut opts = ObjectOptions {
|
||||
data_movement: true,
|
||||
src_pool_idx: pool_idx,
|
||||
expected_bucket_incarnation_id,
|
||||
..Default::default()
|
||||
};
|
||||
if let Some(signal) = lock_lost_signal.as_ref() {
|
||||
opts.add_namespace_lock_lost_signal(Arc::clone(signal));
|
||||
}
|
||||
opts
|
||||
})
|
||||
.abort_multipart_upload_for_data_movement(
|
||||
target_pool_idx,
|
||||
&bucket,
|
||||
&object_info.name,
|
||||
&res.upload_id,
|
||||
&abort_opts,
|
||||
)
|
||||
.await;
|
||||
match abort_result {
|
||||
Ok(()) => return Ok(()),
|
||||
Err(abort_err) if is_err_invalid_upload_id(&abort_err) => {
|
||||
if should_treat_data_movement_overwrite_as_complete(
|
||||
store.as_ref(),
|
||||
pool_idx,
|
||||
target_pool_idx,
|
||||
(pool_idx, target_pool_idx),
|
||||
bucket.as_str(),
|
||||
&object_info,
|
||||
&abort_err,
|
||||
preserve_part_checksums,
|
||||
DataMovementOverwriteCapacity {
|
||||
owner: capacity_owner,
|
||||
expected_data_bytes: capacity_expected_data_bytes,
|
||||
},
|
||||
)
|
||||
.await?
|
||||
{
|
||||
@@ -1683,6 +1925,7 @@ async fn migrate_object_inner(
|
||||
bucket.clone(),
|
||||
object_info.name.clone(),
|
||||
res.upload_id.clone(),
|
||||
abort_opts,
|
||||
op_label,
|
||||
);
|
||||
return Err(data_movement_stage_error(
|
||||
@@ -1698,19 +1941,24 @@ async fn migrate_object_inner(
|
||||
|
||||
if let Err(primary_err) = multipart_result {
|
||||
if should_abort_multipart_upload(&abort_multipart_flag) {
|
||||
#[cfg(all(test, feature = "test-util"))]
|
||||
pause_data_movement_multipart_before_abort(&bucket, &object_info.name).await;
|
||||
let mut abort_opts =
|
||||
data_movement_abort_opts(pool_idx, expected_bucket_incarnation_id, lock_lost_signal.as_ref(), capacity_owner);
|
||||
if let Some(anchor) = mutation_fence.as_ref() {
|
||||
anchor.guard().add_namespace_lock_fence(&mut abort_opts);
|
||||
}
|
||||
if let Some(fence) = multipart_mutation_fence.as_ref() {
|
||||
fence.add_namespace_lock_fence(&mut abort_opts);
|
||||
}
|
||||
return match store
|
||||
.abort_multipart_upload_for_data_movement(target_pool_idx, &bucket, &object_info.name, &res.upload_id, &{
|
||||
let mut opts = ObjectOptions {
|
||||
data_movement: true,
|
||||
src_pool_idx: pool_idx,
|
||||
expected_bucket_incarnation_id,
|
||||
..Default::default()
|
||||
};
|
||||
if let Some(signal) = lock_lost_signal.as_ref() {
|
||||
opts.add_namespace_lock_lost_signal(Arc::clone(signal));
|
||||
}
|
||||
opts
|
||||
})
|
||||
.abort_multipart_upload_for_data_movement(
|
||||
target_pool_idx,
|
||||
&bucket,
|
||||
&object_info.name,
|
||||
&res.upload_id,
|
||||
&abort_opts,
|
||||
)
|
||||
.await
|
||||
{
|
||||
Ok(()) => Err(primary_err),
|
||||
@@ -1722,6 +1970,7 @@ async fn migrate_object_inner(
|
||||
bucket.clone(),
|
||||
object_info.name.clone(),
|
||||
res.upload_id.clone(),
|
||||
abort_opts,
|
||||
op_label,
|
||||
);
|
||||
Err(resolve_data_movement_abort_result(
|
||||
@@ -1744,23 +1993,39 @@ async fn migrate_object_inner(
|
||||
let mut data = data_movement_put_object_reader(bucket.as_str(), &object_info, rd, op_label)?;
|
||||
|
||||
let mut put_opts = data_movement_put_object_opts(&object_info, pool_idx);
|
||||
if let Some(capacity_owner) = capacity_owner {
|
||||
capacity_owner.apply_to(&mut put_opts);
|
||||
}
|
||||
put_opts.expected_bucket_incarnation_id = source_bucket_incarnation_id;
|
||||
if let Some(signal) = lock_lost_signal {
|
||||
put_opts.add_namespace_lock_lost_signal(signal);
|
||||
}
|
||||
let remote_tuple_publication_fence = match mutation_fence.take() {
|
||||
Some(anchor) => remote_tuple_publication_fence.under_fixed_read_anchor(anchor)?,
|
||||
None => remote_tuple_publication_fence,
|
||||
};
|
||||
let (target_pool_idx, put_result) = store
|
||||
.put_object_for_data_movement(&bucket, &object_info.name, &mut data, &put_opts, mutation_fence)
|
||||
.put_object_for_data_movement_with_publication_fence(
|
||||
&bucket,
|
||||
&object_info.name,
|
||||
&mut data,
|
||||
&put_opts,
|
||||
remote_tuple_publication_fence,
|
||||
)
|
||||
.await
|
||||
.map_err(|err| data_movement_stage_error(op_label, "prepare_put_object", &bucket, &object_info.name, err))?;
|
||||
if let Err(err) = put_result {
|
||||
if should_treat_data_movement_overwrite_as_complete(
|
||||
store.as_ref(),
|
||||
pool_idx,
|
||||
target_pool_idx,
|
||||
(pool_idx, target_pool_idx),
|
||||
bucket.as_str(),
|
||||
&object_info,
|
||||
&err,
|
||||
preserve_part_checksums,
|
||||
DataMovementOverwriteCapacity {
|
||||
owner: capacity_owner,
|
||||
expected_data_bytes: capacity_expected_data_bytes,
|
||||
},
|
||||
)
|
||||
.await?
|
||||
{
|
||||
|
||||
@@ -109,7 +109,10 @@ static USAGE_MEMORY_GENERATION: AtomicU64 = AtomicU64::new(0);
|
||||
/// strictly tighter than beta.11 (usage treated as 0) and strictly more
|
||||
/// available than a blanket 503. The fallback applies to any window without
|
||||
/// authoritative usage, not only pre-v2 upgrades; the values always come from
|
||||
/// the last persisted scanner output. Loads go through the TTL-bounded
|
||||
/// the last persisted scanner output — pre-discard sizes of the
|
||||
/// authoritative snapshot first, backfilled per bucket from the observed
|
||||
/// (nonconverged) snapshot for buckets no authoritative cycle has covered
|
||||
/// yet (issue #6852). Loads go through the TTL-bounded
|
||||
/// snapshot cache, so the quota path adds at most one backend read per
|
||||
/// [`DATA_USAGE_CACHE_TTL_SECS`] window. Returns `None` for buckets absent
|
||||
/// from every persisted snapshot — those still fail closed.
|
||||
@@ -168,7 +171,7 @@ fn fresh_cached_data_usage_snapshot(
|
||||
|
||||
fn cache_data_usage_snapshot_result(
|
||||
cache: &mut Option<CachedDataUsageSnapshot>,
|
||||
result: Result<(DataUsageInfo, HashMap<String, u64>), Error>,
|
||||
result: Result<LoadedUsageBaseline, Error>,
|
||||
loaded_at: tokio::time::Instant,
|
||||
refresh_generation: u64,
|
||||
current_generation: u64,
|
||||
@@ -178,7 +181,19 @@ fn cache_data_usage_snapshot_result(
|
||||
}
|
||||
|
||||
Some(match result {
|
||||
Ok((info, degraded_baseline)) => {
|
||||
Ok(LoadedUsageBaseline {
|
||||
info,
|
||||
mut degraded_baseline,
|
||||
observed_unavailable,
|
||||
}) => {
|
||||
// A flaky observed read must not shrink quota coverage for a TTL
|
||||
// window: carry the previous refresh's baseline entries forward,
|
||||
// letting the fresh (authoritative) values win where they exist.
|
||||
if observed_unavailable && let Some(previous) = cache.as_ref() {
|
||||
for (bucket, size) in &previous.degraded_baseline {
|
||||
degraded_baseline.entry(bucket.clone()).or_insert(*size);
|
||||
}
|
||||
}
|
||||
*cache = Some(CachedDataUsageSnapshot {
|
||||
info: Some(info.clone()),
|
||||
loaded_at,
|
||||
@@ -1113,24 +1128,78 @@ async fn load_data_usage_snapshot(store: Arc<ECStore>) -> Result<(DataUsageInfo,
|
||||
/// Load data usage info from backend storage
|
||||
#[instrument(skip(store))]
|
||||
pub async fn load_data_usage_from_backend(store: Arc<ECStore>) -> Result<DataUsageInfo, Error> {
|
||||
Ok(load_data_usage_from_backend_with_baseline(store).await?.0)
|
||||
Ok(load_data_usage_from_backend_with_baseline(store).await?.info)
|
||||
}
|
||||
|
||||
/// One refresh of the persisted usage snapshot plus the quota-admission
|
||||
/// baseline derived from it.
|
||||
struct LoadedUsageBaseline {
|
||||
info: DataUsageInfo,
|
||||
degraded_baseline: HashMap<String, u64>,
|
||||
/// True when the observed snapshot could not be read (a transport error,
|
||||
/// not absence): the cached loader then carries the previous refresh's
|
||||
/// baseline entries forward instead of shrinking quota coverage for a
|
||||
/// whole TTL window over one flaky read.
|
||||
observed_unavailable: bool,
|
||||
}
|
||||
|
||||
/// Like [`load_data_usage_from_backend`], but also returns the pre-discard
|
||||
/// per-bucket sizes so the cached loader can retain them as the degraded
|
||||
/// quota-admission baseline (issue #5716).
|
||||
async fn load_data_usage_from_backend_with_baseline(store: Arc<ECStore>) -> Result<(DataUsageInfo, HashMap<String, u64>), Error> {
|
||||
let (data_usage_info, source) = load_data_usage_snapshot(store).await?;
|
||||
Ok(normalize_loaded_data_usage(data_usage_info, source.is_authoritative()).await)
|
||||
async fn load_data_usage_from_backend_with_baseline(store: Arc<ECStore>) -> Result<LoadedUsageBaseline, Error> {
|
||||
let (loaded_snapshot, source) = load_data_usage_snapshot(store.clone()).await?;
|
||||
// The observed-newness gate below compares against the snapshot as
|
||||
// persisted, before normalization demotes or discards anything.
|
||||
let authoritative_as_persisted = loaded_snapshot.clone();
|
||||
let (info, mut degraded_baseline) = normalize_loaded_data_usage(loaded_snapshot, source.is_authoritative()).await;
|
||||
|
||||
// A bucket without a converged scanner cycle behind it — a freshly joined
|
||||
// replica whose every cycle is superseded by the sustained replication
|
||||
// write stream, or a bucket created after the last converged cycle on a
|
||||
// busy site (#6852) — has no authoritative size, and quota admission
|
||||
// fails its writes closed indefinitely. The observed (nonconverged)
|
||||
// snapshot those superseded cycles still publish is the only grounded
|
||||
// usage in that window, so it backfills buckets the loaded baseline does
|
||||
// not cover; a value already in the baseline always wins. The newness
|
||||
// gate ties the observation to this exact authoritative snapshot, so a
|
||||
// stale observed object left behind by an earlier incarnation (e.g. a
|
||||
// deleted and recreated bucket) cannot inject ghost usage. Loads sit
|
||||
// behind the same TTL cache as the snapshot itself, so this adds at most
|
||||
// one backend read per TTL window.
|
||||
let mut observed_unavailable = false;
|
||||
match load_observed_data_usage_snapshot(store).await {
|
||||
Ok(Some(observed)) if observed_data_usage_is_newer(&observed, &authoritative_as_persisted) => {
|
||||
backfill_degraded_baseline_from_observed(&mut degraded_baseline, &observed);
|
||||
}
|
||||
Ok(_) => {}
|
||||
Err(_) => observed_unavailable = true,
|
||||
}
|
||||
|
||||
Ok(LoadedUsageBaseline {
|
||||
info,
|
||||
degraded_baseline,
|
||||
observed_unavailable,
|
||||
})
|
||||
}
|
||||
|
||||
async fn load_observed_data_usage_snapshot(store: Arc<ECStore>) -> Option<DataUsageInfo> {
|
||||
/// Fill quota-baseline gaps from an observed (nonconverged) snapshot without
|
||||
/// overriding any bucket the authoritative baseline already covers.
|
||||
fn backfill_degraded_baseline_from_observed(degraded_baseline: &mut HashMap<String, u64>, observed: &DataUsageInfo) {
|
||||
for (bucket, usage) in &observed.buckets_usage {
|
||||
degraded_baseline.entry(bucket.clone()).or_insert(usage.size);
|
||||
}
|
||||
}
|
||||
|
||||
/// `Ok(None)` means the observed snapshot is absent or invalid (a settled
|
||||
/// answer); `Err` means it could not be read at all, so the caller may keep
|
||||
/// using what it learned from a previous read.
|
||||
async fn load_observed_data_usage_snapshot(store: Arc<ECStore>) -> Result<Option<DataUsageInfo>, Error> {
|
||||
let data = match read_config_preserve_empty(store, &DATA_USAGE_OBSERVED_OBJ_NAME_PATH).await {
|
||||
Ok(data) => data,
|
||||
Err(Error::ConfigNotFound) => return None,
|
||||
Err(Error::ConfigNotFound) => return Ok(None),
|
||||
Err(err) => {
|
||||
record_usage_snapshot_failure("read_observed", DATA_USAGE_OBSERVED_OBJ_NAME_PATH.as_str(), &err);
|
||||
return None;
|
||||
return Err(err);
|
||||
}
|
||||
};
|
||||
|
||||
@@ -1139,7 +1208,7 @@ async fn load_observed_data_usage_snapshot(store: Arc<ECStore>) -> Option<DataUs
|
||||
if info.usage_snapshot_converged == Some(false)
|
||||
&& (info.is_complete_bucket_usage_snapshot() || info.is_valid_partial_snapshot()) =>
|
||||
{
|
||||
Some(info)
|
||||
Ok(Some(info))
|
||||
}
|
||||
Ok(_) => {
|
||||
error!(
|
||||
@@ -1150,11 +1219,11 @@ async fn load_observed_data_usage_snapshot(store: Arc<ECStore>) -> Option<DataUs
|
||||
object = %DATA_USAGE_OBSERVED_OBJ_NAME_PATH.as_str(),
|
||||
"observed data usage snapshot was not a structurally complete nonconverged view"
|
||||
);
|
||||
None
|
||||
Ok(None)
|
||||
}
|
||||
Err(err) => {
|
||||
record_usage_snapshot_decode_failure("parse_observed", DATA_USAGE_OBSERVED_OBJ_NAME_PATH.as_str(), &err);
|
||||
None
|
||||
Ok(None)
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -1212,7 +1281,9 @@ fn merge_partial_observation_for_admin(mut authoritative: DataUsageInfo, observe
|
||||
|
||||
async fn load_admin_data_usage_from_backend(store: Arc<ECStore>) -> Result<DataUsageInfo, Error> {
|
||||
let (authoritative, source) = load_data_usage_snapshot(store.clone()).await?;
|
||||
let observed = load_observed_data_usage_snapshot(store).await;
|
||||
// For the one-shot admin view a failed observed read degrades to "no
|
||||
// observation", same as before the read was fallible.
|
||||
let observed = load_observed_data_usage_snapshot(store).await.ok().flatten();
|
||||
let (selected, selected_is_current_format) =
|
||||
select_admin_data_usage_snapshot(authoritative, source.is_authoritative(), observed);
|
||||
Ok(normalize_loaded_data_usage(selected, selected_is_current_format).await.0)
|
||||
@@ -1375,7 +1446,11 @@ pub async fn load_admin_data_usage_from_backend_cached(store: Arc<ECStore>) -> R
|
||||
let refresh_generation = admin_data_usage_snapshot_generation();
|
||||
let result = load_admin_data_usage_from_backend(store.clone())
|
||||
.await
|
||||
.map(|info| (info, HashMap::new()));
|
||||
.map(|info| LoadedUsageBaseline {
|
||||
info,
|
||||
degraded_baseline: HashMap::new(),
|
||||
observed_unavailable: false,
|
||||
});
|
||||
let loaded_at = tokio::time::Instant::now();
|
||||
let mut cache = admin_data_usage_snapshot_cache().write().await;
|
||||
if let Some(result) = cache_data_usage_snapshot_result(
|
||||
@@ -2526,6 +2601,37 @@ mod tests {
|
||||
use std::sync::Arc;
|
||||
use tokio::{io::AsyncReadExt, sync::Mutex};
|
||||
|
||||
#[test]
|
||||
fn observed_snapshot_only_backfills_baseline_gaps() {
|
||||
let mut baseline = HashMap::from([("covered".to_string(), 111_u64)]);
|
||||
let observed = DataUsageInfo {
|
||||
buckets_usage: HashMap::from([
|
||||
(
|
||||
"covered".to_string(),
|
||||
BucketUsageInfo {
|
||||
size: 999,
|
||||
..Default::default()
|
||||
},
|
||||
),
|
||||
(
|
||||
"replica-only".to_string(),
|
||||
BucketUsageInfo {
|
||||
size: 42,
|
||||
..Default::default()
|
||||
},
|
||||
),
|
||||
]),
|
||||
..Default::default()
|
||||
};
|
||||
|
||||
backfill_degraded_baseline_from_observed(&mut baseline, &observed);
|
||||
|
||||
// The authoritative value must win; only the uncovered bucket (#6852:
|
||||
// a replica that never landed a converged cycle) is filled in.
|
||||
assert_eq!(baseline.get("covered"), Some(&111));
|
||||
assert_eq!(baseline.get("replica-only"), Some(&42));
|
||||
}
|
||||
|
||||
#[derive(Debug, Default)]
|
||||
struct UsageCasState {
|
||||
object: Option<(Vec<u8>, u64)>,
|
||||
@@ -2858,6 +2964,7 @@ mod tests {
|
||||
decommission_cancelers: RwLock::new(Vec::new()),
|
||||
start_gate: TokioMutex::new(()),
|
||||
pool_meta_save_gate: TokioMutex::default(),
|
||||
decommission_capacity_entry_gate: TokioMutex::default(),
|
||||
ctx,
|
||||
bucket_fence_registry: Arc::default(),
|
||||
})
|
||||
@@ -3479,7 +3586,11 @@ mod tests {
|
||||
|
||||
let first = cache_data_usage_snapshot_result(
|
||||
&mut cache,
|
||||
Ok((expected, HashMap::new())),
|
||||
Ok(LoadedUsageBaseline {
|
||||
info: expected,
|
||||
degraded_baseline: HashMap::new(),
|
||||
observed_unavailable: false,
|
||||
}),
|
||||
loaded_at,
|
||||
refresh_generation,
|
||||
data_usage_snapshot_generation(),
|
||||
@@ -3494,6 +3605,38 @@ mod tests {
|
||||
assert_snapshot_bucket(&cached, "bucket");
|
||||
}
|
||||
|
||||
#[test]
|
||||
#[serial]
|
||||
fn unavailable_observed_read_keeps_previous_baseline_coverage() {
|
||||
let loaded_at = tokio::time::Instant::now();
|
||||
let refresh_generation = data_usage_snapshot_generation();
|
||||
let mut cache = Some(CachedDataUsageSnapshot {
|
||||
info: Some(data_usage_info_for_test("bucket", 1, 42, SystemTime::UNIX_EPOCH)),
|
||||
loaded_at,
|
||||
degraded_baseline: HashMap::from([("observed-only".to_string(), 7_u64), ("covered".to_string(), 1)]),
|
||||
});
|
||||
|
||||
cache_data_usage_snapshot_result(
|
||||
&mut cache,
|
||||
Ok(LoadedUsageBaseline {
|
||||
info: data_usage_info_for_test("bucket", 1, 42, SystemTime::UNIX_EPOCH),
|
||||
degraded_baseline: HashMap::from([("covered".to_string(), 2_u64)]),
|
||||
observed_unavailable: true,
|
||||
}),
|
||||
loaded_at,
|
||||
refresh_generation,
|
||||
data_usage_snapshot_generation(),
|
||||
)
|
||||
.expect("an uninterrupted refresh should populate the cache")
|
||||
.expect("successful load must be returned");
|
||||
|
||||
let baseline = &cache.as_ref().expect("cache must be populated").degraded_baseline;
|
||||
// The bucket only the (now unreadable) observed snapshot covered must
|
||||
// survive the refresh; the freshly loaded value wins where it exists.
|
||||
assert_eq!(baseline.get("observed-only"), Some(&7));
|
||||
assert_eq!(baseline.get("covered"), Some(&2));
|
||||
}
|
||||
|
||||
#[test]
|
||||
#[serial]
|
||||
fn cache_invalidation_during_refresh_prevents_stale_snapshot_resurrection() {
|
||||
@@ -3508,7 +3651,11 @@ mod tests {
|
||||
|
||||
let stale_result = cache_data_usage_snapshot_result(
|
||||
&mut cache,
|
||||
Ok((data_usage_info_for_test("stale", 1, 42, SystemTime::UNIX_EPOCH), HashMap::new())),
|
||||
Ok(LoadedUsageBaseline {
|
||||
info: data_usage_info_for_test("stale", 1, 42, SystemTime::UNIX_EPOCH),
|
||||
degraded_baseline: HashMap::new(),
|
||||
observed_unavailable: false,
|
||||
}),
|
||||
loaded_at,
|
||||
refresh_generation,
|
||||
data_usage_snapshot_generation(),
|
||||
|
||||
@@ -250,6 +250,40 @@ pub(crate) trait DiskStoreRenameDataExt {
|
||||
dst_volume: &str,
|
||||
dst_path: &str,
|
||||
) -> Result<RenameDataResp>;
|
||||
|
||||
async fn rename_data_borrowed_with_guard(
|
||||
&self,
|
||||
src_volume: &str,
|
||||
src_path: &str,
|
||||
fi: &FileInfo,
|
||||
dst_volume: &str,
|
||||
dst_path: &str,
|
||||
external_guard: Option<Arc<dyn Send + Sync>>,
|
||||
) -> Result<RenameDataResp> {
|
||||
let _ = external_guard;
|
||||
self.rename_data_borrowed(src_volume, src_path, fi, dst_volume, dst_path)
|
||||
.await
|
||||
}
|
||||
}
|
||||
|
||||
/// Run a mutation in an owned task when a caller supplied publication guard.
|
||||
/// RPC cancellation drops only the waiter; the mutation owner keeps the guard
|
||||
/// until its operation has returned, including any detached blocking syscall.
|
||||
async fn run_owned_mutation<T, F, Fut>(external_guard: Option<Arc<dyn Send + Sync>>, operation: F) -> Result<T>
|
||||
where
|
||||
T: Send + 'static,
|
||||
F: FnOnce() -> Fut + Send + 'static,
|
||||
Fut: std::future::Future<Output = Result<T>> + Send + 'static,
|
||||
{
|
||||
if external_guard.is_none() {
|
||||
return operation().await;
|
||||
}
|
||||
tokio::spawn(async move {
|
||||
let _external_guard = external_guard;
|
||||
operation().await
|
||||
})
|
||||
.await
|
||||
.map_err(|_| Error::other("owned mutation task failed"))?
|
||||
}
|
||||
|
||||
impl DiskStoreRenameDataExt for LocalDiskWrapper {
|
||||
@@ -273,6 +307,49 @@ impl DiskStoreRenameDataExt for LocalDiskWrapper {
|
||||
)
|
||||
.await
|
||||
}
|
||||
|
||||
async fn rename_data_borrowed_with_guard(
|
||||
&self,
|
||||
src_volume: &str,
|
||||
src_path: &str,
|
||||
fi: &FileInfo,
|
||||
dst_volume: &str,
|
||||
dst_path: &str,
|
||||
external_guard: Option<Arc<dyn Send + Sync>>,
|
||||
) -> Result<RenameDataResp> {
|
||||
let operation = self.clone();
|
||||
let src_volume = src_volume.to_owned();
|
||||
let src_path = src_path.to_owned();
|
||||
let fi = fi.clone();
|
||||
let dst_volume = dst_volume.to_owned();
|
||||
let dst_path = dst_path.to_owned();
|
||||
let timeout_duration = if external_guard.is_some() {
|
||||
// A fenced mutation owns the publication guard until the storage
|
||||
// operation returns. Timing out this waiter would cancel the
|
||||
// LocalDisk future while a spawn_blocking namespace syscall could
|
||||
// still be committing, reopening the movement window. The caller
|
||||
// may drop its waiter; the owned task drains the mutation.
|
||||
Duration::ZERO
|
||||
} else {
|
||||
get_max_timeout_duration()
|
||||
};
|
||||
run_owned_mutation(external_guard, move || async move {
|
||||
operation
|
||||
.track_disk_health_mutation(
|
||||
"rename_data",
|
||||
DiskMetricMutation::Write,
|
||||
|| async {
|
||||
operation
|
||||
.disk
|
||||
.rename_data_borrowed(&src_volume, &src_path, &fi, &dst_volume, &dst_path)
|
||||
.await
|
||||
},
|
||||
timeout_duration,
|
||||
)
|
||||
.await
|
||||
})
|
||||
.await
|
||||
}
|
||||
}
|
||||
|
||||
pub fn get_drive_walkdir_timeout() -> Duration {
|
||||
@@ -678,17 +755,20 @@ impl DiskOperationMetrics {
|
||||
let elapsed_nanos = u64::try_from(elapsed.as_nanos()).unwrap_or(u64::MAX);
|
||||
let slot = &self.last_minute[(now_sec % 60) as usize];
|
||||
loop {
|
||||
let version = slot.version.load(Ordering::Acquire);
|
||||
// The successful CAS below is AcqRel, so it is the publication
|
||||
// fence for the writer that owns this slot. The initial parity
|
||||
// check does not need to acquire the slot payload.
|
||||
let version = slot.version.load(Ordering::Relaxed);
|
||||
if !version.is_multiple_of(2) {
|
||||
std::hint::spin_loop();
|
||||
continue;
|
||||
}
|
||||
if slot
|
||||
.version
|
||||
.compare_exchange(version, version.wrapping_add(1), Ordering::AcqRel, Ordering::Acquire)
|
||||
.compare_exchange(version, version.wrapping_add(1), Ordering::AcqRel, Ordering::Relaxed)
|
||||
.is_ok()
|
||||
{
|
||||
if slot.unix_sec.load(Ordering::Acquire) != now_sec {
|
||||
if slot.unix_sec.load(Ordering::Relaxed) != now_sec {
|
||||
slot.count.store(0, Ordering::Relaxed);
|
||||
slot.acc_time.store(0, Ordering::Relaxed);
|
||||
slot.unix_sec.store(now_sec, Ordering::Release);
|
||||
@@ -704,14 +784,10 @@ impl DiskOperationMetrics {
|
||||
fn last_minute_snapshot(&self, now_sec: u64) -> TimedAction {
|
||||
let mut snapshot = TimedAction::default();
|
||||
for slot in &self.last_minute {
|
||||
let version = slot.version.load(Ordering::Acquire);
|
||||
if !version.is_multiple_of(2) {
|
||||
let Some((slot_sec, count, acc_time)) = slot.snapshot() else {
|
||||
continue;
|
||||
}
|
||||
let slot_sec = slot.unix_sec.load(Ordering::Acquire);
|
||||
let count = slot.count.load(Ordering::Acquire);
|
||||
let acc_time = slot.acc_time.load(Ordering::Acquire);
|
||||
if slot.version.load(Ordering::Acquire) == version && slot_sec <= now_sec && now_sec.saturating_sub(slot_sec) < 60 {
|
||||
};
|
||||
if slot_sec <= now_sec && now_sec.saturating_sub(slot_sec) < 60 {
|
||||
snapshot.count = snapshot.count.saturating_add(count);
|
||||
snapshot.acc_time = snapshot.acc_time.saturating_add(acc_time);
|
||||
}
|
||||
@@ -720,6 +796,23 @@ impl DiskOperationMetrics {
|
||||
}
|
||||
}
|
||||
|
||||
impl TimedActionSlot {
|
||||
fn snapshot(&self) -> Option<(u64, u64, u64)> {
|
||||
let version = self.version.load(Ordering::Acquire);
|
||||
if !version.is_multiple_of(2) {
|
||||
return None;
|
||||
}
|
||||
|
||||
// The first Acquire load publishes the payload written before the
|
||||
// matching Release store. Relaxed payload loads are sufficient while
|
||||
// the final Acquire version load validates that no writer intervened.
|
||||
let slot_sec = self.unix_sec.load(Ordering::Relaxed);
|
||||
let count = self.count.load(Ordering::Relaxed);
|
||||
let acc_time = self.acc_time.load(Ordering::Relaxed);
|
||||
(self.version.load(Ordering::Acquire) == version).then_some((slot_sec, count, acc_time))
|
||||
}
|
||||
}
|
||||
|
||||
pub(crate) struct DiskHealthWaitingGuard<'a> {
|
||||
health: &'a DiskHealthTracker,
|
||||
}
|
||||
@@ -1097,6 +1190,37 @@ impl LocalDiskWrapper {
|
||||
)
|
||||
}
|
||||
|
||||
/// Run a delete under an owned coordinator task when a publication guard
|
||||
/// is present. This keeps the guard alive if the RPC waiter is cancelled
|
||||
/// while the local namespace mutation is still in progress.
|
||||
pub(crate) async fn delete_with_publication_guard(
|
||||
&self,
|
||||
volume: &str,
|
||||
path: &str,
|
||||
options: DeleteOptions,
|
||||
external_guard: Option<Arc<dyn Send + Sync>>,
|
||||
) -> Result<()> {
|
||||
let operation = self.clone();
|
||||
let volume = volume.to_owned();
|
||||
let path = path.to_owned();
|
||||
let timeout_duration = if external_guard.is_some() {
|
||||
Duration::ZERO
|
||||
} else {
|
||||
get_max_timeout_duration()
|
||||
};
|
||||
run_owned_mutation(external_guard, move || async move {
|
||||
operation
|
||||
.track_disk_health_mutation(
|
||||
"delete",
|
||||
DiskMetricMutation::Delete,
|
||||
|| async { operation.disk.delete(&volume, &path, options).await },
|
||||
timeout_duration,
|
||||
)
|
||||
.await
|
||||
})
|
||||
.await
|
||||
}
|
||||
|
||||
pub(crate) fn new_with_reconnect_state(
|
||||
disk: Arc<LocalDisk>,
|
||||
health_check: bool,
|
||||
@@ -2247,6 +2371,44 @@ mod tests {
|
||||
};
|
||||
use tokio::io::AsyncWrite;
|
||||
|
||||
struct DropProbe(Arc<std::sync::atomic::AtomicUsize>);
|
||||
|
||||
impl Drop for DropProbe {
|
||||
fn drop(&mut self) {
|
||||
self.0.fetch_add(1, std::sync::atomic::Ordering::SeqCst);
|
||||
}
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
async fn owned_mutation_keeps_publication_guard_after_waiter_cancellation() {
|
||||
let drops = Arc::new(std::sync::atomic::AtomicUsize::new(0));
|
||||
let guard: Arc<dyn Send + Sync> = Arc::new(DropProbe(Arc::clone(&drops)));
|
||||
let (started_tx, started_rx) = tokio::sync::oneshot::channel();
|
||||
let (release_tx, release_rx) = tokio::sync::oneshot::channel();
|
||||
let (finished_tx, finished_rx) = tokio::sync::oneshot::channel();
|
||||
|
||||
let waiter = tokio::spawn(run_owned_mutation(Some(guard), move || async move {
|
||||
started_tx.send(()).expect("mutation should signal start");
|
||||
release_rx.await.expect("mutation should be released");
|
||||
finished_tx.send(()).expect("mutation should signal completion");
|
||||
Ok::<_, Error>(())
|
||||
}));
|
||||
|
||||
started_rx.await.expect("mutation owner should start");
|
||||
waiter.abort();
|
||||
assert_eq!(drops.load(std::sync::atomic::Ordering::SeqCst), 0);
|
||||
|
||||
release_tx.send(()).expect("mutation owner should still be alive");
|
||||
finished_rx.await.expect("mutation owner should finish");
|
||||
tokio::time::timeout(Duration::from_secs(1), async {
|
||||
while drops.load(std::sync::atomic::Ordering::SeqCst) == 0 {
|
||||
tokio::task::yield_now().await;
|
||||
}
|
||||
})
|
||||
.await
|
||||
.expect("publication guard should be released after mutation completion");
|
||||
}
|
||||
|
||||
struct PendingWriter;
|
||||
|
||||
#[test]
|
||||
@@ -2273,6 +2435,25 @@ mod tests {
|
||||
assert_eq!(window.acc_time, 18_000);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn timed_action_slot_snapshot_skips_writer_owned_slot() {
|
||||
let slot = TimedActionSlot::default();
|
||||
slot.unix_sec.store(70, Ordering::Relaxed);
|
||||
slot.count.store(2, Ordering::Relaxed);
|
||||
slot.acc_time.store(18_000, Ordering::Relaxed);
|
||||
slot.version.store(2, Ordering::Release);
|
||||
assert_eq!(slot.snapshot(), Some((70, 2, 18_000)));
|
||||
|
||||
assert_eq!(slot.version.compare_exchange(2, 3, Ordering::AcqRel, Ordering::Relaxed), Ok(2));
|
||||
slot.unix_sec.store(71, Ordering::Relaxed);
|
||||
slot.count.store(1, Ordering::Relaxed);
|
||||
slot.acc_time.store(11_000, Ordering::Relaxed);
|
||||
assert_eq!(slot.snapshot(), None);
|
||||
|
||||
slot.version.store(4, Ordering::Release);
|
||||
assert_eq!(slot.snapshot(), Some((71, 1, 11_000)));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn disk_health_metrics_snapshot_exports_waiting_errors_and_operation_windows() {
|
||||
let metrics = DiskHealthMetricEpoch::default();
|
||||
|
||||
@@ -12,6 +12,7 @@
|
||||
// See the License for the specific language governing permissions and
|
||||
// limitations under the License.
|
||||
|
||||
use rustfs_io_metrics::internode_metrics::INTERNODE_OPERATION_PUT_FILE_STREAM;
|
||||
use rustfs_rio::{InternodeHttpError, InternodeHttpErrorKind};
|
||||
use std::error::Error as StdError;
|
||||
use std::hash::{Hash, Hasher};
|
||||
@@ -229,6 +230,19 @@ fn classify_internode_missing_error(error: &InternodeHttpError) -> Option<DiskEr
|
||||
None
|
||||
}
|
||||
|
||||
fn internode_write_error_is_retryable(error: &InternodeHttpError) -> bool {
|
||||
error.kind().is_retryable()
|
||||
|| (matches!(error.kind(), InternodeHttpErrorKind::HttpStatus(status) if status.as_u16() == 409)
|
||||
&& error.context().operation() == Some(INTERNODE_OPERATION_PUT_FILE_STREAM))
|
||||
}
|
||||
|
||||
fn io_error_contains_retryable_internode_write(error: &io::Error) -> bool {
|
||||
error
|
||||
.get_ref()
|
||||
.and_then(|source| source.downcast_ref::<InternodeHttpError>())
|
||||
.is_some_and(internode_write_error_is_retryable)
|
||||
}
|
||||
|
||||
/// Wrap a terminal shard-read failure without changing its typed
|
||||
/// classification. Timeout-like disk errors retain `TimedOut`; other errors
|
||||
/// retain their inner I/O kind or use `Other` when no more specific kind exists.
|
||||
@@ -336,10 +350,7 @@ impl DiskError {
|
||||
|
||||
pub fn is_retryable_internode_write_failure(&self) -> bool {
|
||||
match self {
|
||||
DiskError::Io(io_error) => io_error
|
||||
.get_ref()
|
||||
.and_then(|source| source.downcast_ref::<InternodeHttpError>())
|
||||
.is_some_and(|err| err.kind().is_retryable()),
|
||||
DiskError::Io(io_error) => io_error_contains_retryable_internode_write(io_error),
|
||||
_ => false,
|
||||
}
|
||||
}
|
||||
@@ -1240,6 +1251,68 @@ mod tests {
|
||||
assert!(!DiskError::FileNotFound.is_internode_http_status(429));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_put_file_server_epoch_conflict_is_retryable_write_failure() {
|
||||
let conflict = DiskError::from(rustfs_rio::new_test_internode_http_io_error(
|
||||
rustfs_rio::InternodeHttpErrorKind::HttpStatus(http::StatusCode::CONFLICT),
|
||||
));
|
||||
let bad_request = DiskError::from(rustfs_rio::new_test_internode_http_io_error(
|
||||
rustfs_rio::InternodeHttpErrorKind::HttpStatus(http::StatusCode::BAD_REQUEST),
|
||||
));
|
||||
|
||||
assert!(conflict.is_retryable_internode_write_failure());
|
||||
assert!(!bad_request.is_retryable_internode_write_failure());
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
async fn read_stream_conflict_is_not_a_retryable_put_file_failure() {
|
||||
use tokio::io::{AsyncReadExt, AsyncWriteExt};
|
||||
|
||||
tokio::time::timeout(std::time::Duration::from_secs(5), async {
|
||||
let listener = tokio::net::TcpListener::bind("127.0.0.1:0")
|
||||
.await
|
||||
.expect("bind isolated HTTP fixture");
|
||||
let address = listener.local_addr().expect("fixture address");
|
||||
let server = tokio::spawn(async move {
|
||||
let (mut stream, _) = listener.accept().await.expect("accept read request");
|
||||
let mut request = [0_u8; 4096];
|
||||
let mut read = 0;
|
||||
loop {
|
||||
let count = stream.read(&mut request[read..]).await.expect("read HTTP request");
|
||||
assert!(count > 0, "request ended before its complete headers");
|
||||
read += count;
|
||||
if request[..read].windows(4).any(|bytes| bytes == b"\r\n\r\n") {
|
||||
break;
|
||||
}
|
||||
assert!(read < request.len(), "fixture request headers exceed their budget");
|
||||
}
|
||||
stream
|
||||
.write_all(b"HTTP/1.1 409 Conflict\r\nContent-Length: 0\r\nConnection: close\r\n\r\n")
|
||||
.await
|
||||
.expect("send typed conflict response");
|
||||
});
|
||||
let error = match rustfs_rio::HttpReader::new(
|
||||
format!("http://{address}/rustfs/rpc/read_file_stream"),
|
||||
http::Method::GET,
|
||||
http::HeaderMap::new(),
|
||||
None,
|
||||
)
|
||||
.await
|
||||
{
|
||||
Ok(_) => panic!("HTTP 409 must fail the read"),
|
||||
Err(error) => DiskError::from(error),
|
||||
};
|
||||
server.await.expect("fixture task should complete");
|
||||
assert!(error.is_internode_http_status(409));
|
||||
assert!(
|
||||
!error.is_retryable_internode_write_failure(),
|
||||
"read-operation 409 must not trigger put-file retry"
|
||||
);
|
||||
})
|
||||
.await
|
||||
.expect("isolated read-conflict test must finish within its budget");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_internode_missing_errors_preserve_disk_error_types() {
|
||||
let file_missing = DiskError::from(rustfs_rio::new_test_remote_file_not_found_http_io_error());
|
||||
|
||||
@@ -3148,9 +3148,10 @@ impl LocalIoBackend for StdBackend {
|
||||
direct_read_copy_fault_delta: MmapPageFaultDelta,
|
||||
blocking_task_duration: StdDuration,
|
||||
used_direct_io: bool,
|
||||
/// The descriptor opened by THIS call (None on a cache hit), handed
|
||||
/// back so the async caller can index it in the fd cache.
|
||||
opened_fd: Option<Arc<std::fs::File>>,
|
||||
/// The descriptor and size snapshot opened by THIS call (None on a
|
||||
/// cache hit), handed back so the async caller can index it in the
|
||||
/// fd cache.
|
||||
opened_fd: Option<Arc<FdCacheEntry>>,
|
||||
}
|
||||
|
||||
enum MmapCopyReadError {
|
||||
@@ -3197,12 +3198,12 @@ impl LocalIoBackend for StdBackend {
|
||||
(cache, key, gen_at_open)
|
||||
});
|
||||
#[cfg(target_os = "linux")]
|
||||
let cached_fd: Option<Arc<std::fs::File>> = match &fd_lookup {
|
||||
let cached_fd: Option<Arc<FdCacheEntry>> = match &fd_lookup {
|
||||
Some((cache, key, _)) => cache.get(key).await,
|
||||
None => None,
|
||||
};
|
||||
#[cfg(not(target_os = "linux"))]
|
||||
let cached_fd: Option<Arc<std::fs::File>> = None;
|
||||
let cached_fd: Option<Arc<FdCacheEntry>> = None;
|
||||
|
||||
let blocking_wait_start = metrics_enabled.then(std::time::Instant::now);
|
||||
let read_result = tokio::task::spawn_blocking(move || {
|
||||
@@ -3224,8 +3225,15 @@ impl LocalIoBackend for StdBackend {
|
||||
// the read below is positioned (mmap offset argument / `read_exact_at`)
|
||||
// and never depends on the descriptor's current offset. `cached_fd` being
|
||||
// None also marks this call as a miss for the cache-insert side-channel.
|
||||
let (file, access_check_duration) = if let Some(cached) = cached_fd.as_ref() {
|
||||
(cached.as_ref().try_clone().map_err(DiskError::from)?, StdDuration::ZERO)
|
||||
// The cached length is the metadata snapshot captured at open time;
|
||||
// all in-place/replacement writers invalidate this entry before
|
||||
// publishing a mutation, so cache hits avoid a redundant fstat.
|
||||
let (file, cached_len, access_check_duration) = if let Some(cached) = cached_fd.as_ref() {
|
||||
(
|
||||
cached.file.as_ref().try_clone().map_err(DiskError::from)?,
|
||||
Some(cached.len),
|
||||
StdDuration::ZERO,
|
||||
)
|
||||
} else {
|
||||
// Measure the volume access probe only — the part-path resolution
|
||||
// above is accounted in `path_resolve_duration` (rustfs/backlog#1801).
|
||||
@@ -3236,20 +3244,27 @@ impl LocalIoBackend for StdBackend {
|
||||
.map_err(|e| DiskError::from(to_access_error(e, DiskError::VolumeAccessDenied)))?;
|
||||
}
|
||||
let access_check_duration = access_check_start.map_or(StdDuration::ZERO, |started_at| started_at.elapsed());
|
||||
(std::fs::File::open(&file_path).map_err(DiskError::from)?, access_check_duration)
|
||||
(std::fs::File::open(&file_path).map_err(DiskError::from)?, None, access_check_duration)
|
||||
};
|
||||
let file_open_duration = file_open_start.map_or(StdDuration::ZERO, |started_at| started_at.elapsed());
|
||||
|
||||
let metadata_lookup_start = metrics_enabled.then(StdInstant::now);
|
||||
// On a cache hit this fstats the cached descriptor — the inode it was
|
||||
// opened against, which invalidation keeps current for live entries. EC
|
||||
// shards are fixed-length, so a still-cached pre-heal length is benign.
|
||||
let meta = file.metadata().map_err(DiskError::from)?;
|
||||
let metadata_lookup_duration = metadata_lookup_start.map_or(StdDuration::ZERO, |started_at| started_at.elapsed());
|
||||
let (metadata_len, metadata_lookup_duration) = if let Some(len) = cached_len {
|
||||
// Reuse the open-time metadata snapshot on a cache hit. The
|
||||
// generation fence and mutation invalidation keep this value
|
||||
// tied to the inode held by `file`.
|
||||
(len, StdDuration::ZERO)
|
||||
} else {
|
||||
let metadata_lookup_start = metrics_enabled.then(StdInstant::now);
|
||||
let meta = file.metadata().map_err(DiskError::from)?;
|
||||
let duration = metadata_lookup_start.map_or(StdDuration::ZERO, |started_at| started_at.elapsed());
|
||||
(meta.len(), duration)
|
||||
};
|
||||
|
||||
let metadata_validate_start = metrics_enabled.then(StdInstant::now);
|
||||
if meta.len() < end_offset_u64 {
|
||||
return Err(MmapCopyReadError::OutOfBounds { actual_size: meta.len() });
|
||||
if metadata_len < end_offset_u64 {
|
||||
return Err(MmapCopyReadError::OutOfBounds {
|
||||
actual_size: metadata_len,
|
||||
});
|
||||
}
|
||||
let metadata_validate_duration =
|
||||
metadata_validate_start.map_or(StdDuration::ZERO, |started_at| started_at.elapsed());
|
||||
@@ -3395,9 +3410,14 @@ impl LocalIoBackend for StdBackend {
|
||||
// Arc; `cached_fd.is_none()` is true exactly when this call did the open.
|
||||
// Non-Linux has no fd cache, so skip the Arc allocation there.
|
||||
#[cfg(target_os = "linux")]
|
||||
let opened_fd: Option<Arc<std::fs::File>> = cached_fd.is_none().then(|| Arc::new(file));
|
||||
let opened_fd: Option<Arc<FdCacheEntry>> = cached_fd.is_none().then(|| {
|
||||
Arc::new(FdCacheEntry {
|
||||
file: Arc::new(file),
|
||||
len: metadata_len,
|
||||
})
|
||||
});
|
||||
#[cfg(not(target_os = "linux"))]
|
||||
let opened_fd: Option<Arc<std::fs::File>> = None;
|
||||
let opened_fd: Option<Arc<FdCacheEntry>> = None;
|
||||
|
||||
Ok::<MmapCopyReadResult, MmapCopyReadError>(MmapCopyReadResult {
|
||||
bytes,
|
||||
@@ -3520,7 +3540,7 @@ impl LocalIoBackend for StdBackend {
|
||||
}
|
||||
}
|
||||
}
|
||||
// Index the freshly opened descriptor for future cache hits
|
||||
// Index the freshly opened descriptor and metadata snapshot for future cache hits
|
||||
// (rustfs/backlog#1801). `insert_if_fresh` refuses to cache if an
|
||||
// invalidation (heal/delete/rename) bumped the generation between the
|
||||
// open snapshot and now, so a stale pre-mutation inode is never served
|
||||
@@ -3872,6 +3892,18 @@ struct FdKey {
|
||||
direct: bool,
|
||||
}
|
||||
|
||||
/// Descriptor and immutable size snapshot retained for one cached shard inode.
|
||||
///
|
||||
/// The generation fence and explicit mutation invalidation keep the snapshot
|
||||
/// tied to the inode held by `file`, allowing cache hits to avoid a repeated
|
||||
/// metadata syscall without weakening replacement/heal semantics.
|
||||
struct FdCacheEntry {
|
||||
/// An independently cloneable descriptor for the immutable shard inode.
|
||||
file: Arc<std::fs::File>,
|
||||
/// File length captured together with the descriptor.
|
||||
len: u64,
|
||||
}
|
||||
|
||||
/// Per-disk cache of open descriptors for io_uring reads (backlog#1145).
|
||||
///
|
||||
/// Why this exists: `pread_uring` opened the file on the blocking pool for every
|
||||
@@ -3901,7 +3933,7 @@ struct FdKey {
|
||||
/// the descriptor once no in-flight read still holds it.
|
||||
#[cfg(target_os = "linux")]
|
||||
struct FdCache {
|
||||
cache: moka::future::Cache<FdKey, Arc<std::fs::File>>,
|
||||
cache: moka::future::Cache<FdKey, Arc<FdCacheEntry>>,
|
||||
/// Bumped by every invalidation. A miss-path open snapshots this before it
|
||||
/// opens and refuses to insert if it moved, so an fd opened before a
|
||||
/// heal/delete commit can never be resurrected into the cache after the
|
||||
@@ -3931,7 +3963,7 @@ impl FdCache {
|
||||
}
|
||||
}
|
||||
|
||||
async fn get(&self, key: &FdKey) -> Option<Arc<std::fs::File>> {
|
||||
async fn get(&self, key: &FdKey) -> Option<Arc<FdCacheEntry>> {
|
||||
self.cache.get(key).await
|
||||
}
|
||||
|
||||
@@ -3946,11 +3978,11 @@ impl FdCache {
|
||||
/// open bumped the generation, so a stale pre-heal/pre-delete inode is never
|
||||
/// cached. The post-insert re-check closes the tiny window where an
|
||||
/// invalidate races the insert itself, by removing the entry we just added.
|
||||
async fn insert_if_fresh(&self, key: FdKey, file: Arc<std::fs::File>, gen_at_open: u64) {
|
||||
async fn insert_if_fresh(&self, key: FdKey, entry: Arc<FdCacheEntry>, gen_at_open: u64) {
|
||||
if self.generation.load(Ordering::Acquire) != gen_at_open {
|
||||
return;
|
||||
}
|
||||
self.cache.insert(key.clone(), file).await;
|
||||
self.cache.insert(key.clone(), entry).await;
|
||||
if self.generation.load(Ordering::Acquire) != gen_at_open {
|
||||
self.cache.invalidate(&key).await;
|
||||
}
|
||||
@@ -3986,7 +4018,7 @@ impl FdCache {
|
||||
self.generation.fetch_add(1, Ordering::AcqRel);
|
||||
let volume = volume.to_owned();
|
||||
let prefix = prefix.trim_end_matches('/').to_owned();
|
||||
let matches = move |k: &FdKey, _: &Arc<std::fs::File>| {
|
||||
let matches = move |k: &FdKey, _: &Arc<FdCacheEntry>| {
|
||||
k.volume == volume && (k.path == prefix || k.path.strip_prefix(&prefix).is_some_and(|r| r.starts_with('/')))
|
||||
};
|
||||
if self.cache.invalidate_entries_if(matches).is_err() {
|
||||
@@ -4002,7 +4034,7 @@ impl FdCache {
|
||||
fn invalidate_volume(&self, volume: &str) {
|
||||
self.generation.fetch_add(1, Ordering::AcqRel);
|
||||
let volume = volume.to_owned();
|
||||
let matches = move |k: &FdKey, _: &Arc<std::fs::File>| k.volume == volume;
|
||||
let matches = move |k: &FdKey, _: &Arc<FdCacheEntry>| k.volume == volume;
|
||||
if self.cache.invalidate_entries_if(matches).is_err() {
|
||||
self.cache.invalidate_all();
|
||||
}
|
||||
@@ -4020,7 +4052,8 @@ impl FdCache {
|
||||
/// tests that drive the cache directly.
|
||||
#[cfg(test)]
|
||||
async fn insert(&self, key: FdKey, file: Arc<std::fs::File>) {
|
||||
self.cache.insert(key, file).await;
|
||||
let len = file.metadata().map(|metadata| metadata.len()).unwrap_or_default();
|
||||
self.cache.insert(key, Arc::new(FdCacheEntry { file, len })).await;
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
@@ -4399,7 +4432,12 @@ impl UringBackend {
|
||||
};
|
||||
|
||||
let file = match cached {
|
||||
Some(file) => file,
|
||||
Some(entry) => {
|
||||
if entry.len < u64::try_from(end_offset).map_err(|_| DiskError::FileCorrupt)? {
|
||||
return Err(DiskError::FileCorrupt);
|
||||
}
|
||||
Arc::clone(&entry.file)
|
||||
}
|
||||
None => {
|
||||
// Snapshot the cache generation BEFORE opening (rustfs/backlog#1176):
|
||||
// if a heal/delete invalidation runs while this open is in flight,
|
||||
@@ -4409,7 +4447,7 @@ impl UringBackend {
|
||||
let root = self.root.clone();
|
||||
let volume_owned = volume.to_owned();
|
||||
let path_owned = path.to_owned();
|
||||
let file = tokio::task::spawn_blocking(move || -> Result<std::fs::File> {
|
||||
let (file, len) = tokio::task::spawn_blocking(move || -> Result<(std::fs::File, u64)> {
|
||||
let file_path = resolve_uring_object_path(&root, &volume_owned, &path_owned)?;
|
||||
let file = std::fs::File::open(&file_path).map_err(DiskError::from)?;
|
||||
let meta = file.metadata().map_err(DiskError::from)?;
|
||||
@@ -4417,30 +4455,22 @@ impl UringBackend {
|
||||
if meta.len() < end_offset_u64 {
|
||||
return Err(DiskError::FileCorrupt);
|
||||
}
|
||||
Ok(file)
|
||||
Ok((file, meta.len()))
|
||||
})
|
||||
.await
|
||||
.map_err(|e| DiskError::other(format!("uring pread join error: {e}")))??;
|
||||
let file = Arc::new(file);
|
||||
let file = Arc::new(FdCacheEntry {
|
||||
file: Arc::new(file),
|
||||
len,
|
||||
});
|
||||
if let (Some((cache, key)), Some(gen_at_open)) = (cache_entry, gen_at_open) {
|
||||
cache.insert_if_fresh(key, Arc::clone(&file), gen_at_open).await;
|
||||
}
|
||||
file
|
||||
file.file.clone()
|
||||
}
|
||||
};
|
||||
|
||||
if length == 0 {
|
||||
// Parity with StdBackend and the miss path (rustfs/backlog#1173): a
|
||||
// zero-length read still rejects an offset past EOF. The miss path
|
||||
// validated `meta.len() < end_offset` (end_offset == offset here), but
|
||||
// a cache hit skipped it — so fstat the descriptor and match. This is
|
||||
// a rare path (callers do not issue zero-length reads), so the one
|
||||
// extra fstat is negligible.
|
||||
match file.metadata() {
|
||||
Ok(meta) if offset_u64 > meta.len() => return Err(DiskError::FileCorrupt),
|
||||
Ok(_) => {}
|
||||
Err(e) => return Err(DiskError::from(e)),
|
||||
}
|
||||
return Ok(Bytes::new());
|
||||
}
|
||||
|
||||
@@ -7062,6 +7092,9 @@ impl LocalDisk {
|
||||
.await?
|
||||
{
|
||||
meta.name.push_str(SLASH_SEPARATOR);
|
||||
// Conservative listings verify physical prefixes. Never-versioned
|
||||
// buckets use the bounded fast path and reclaim residue after an
|
||||
// exact recursive listing proves that prefix empty.
|
||||
if opts.recursive
|
||||
|| opts.incl_deleted
|
||||
|| opts.skip_hidden_prefix_check
|
||||
@@ -17589,7 +17622,7 @@ mod test {
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
async fn test_scan_dir_nonrecursive_visible_prefix_probe_cost() {
|
||||
async fn test_scan_dir_nonrecursive_fast_path_preserves_probe_bound() {
|
||||
use rustfs_filemeta::MetacacheReader;
|
||||
use tempfile::tempdir;
|
||||
|
||||
@@ -17621,6 +17654,10 @@ mod test {
|
||||
expected_names.push(format!("{prefix}/"));
|
||||
}
|
||||
|
||||
fs::create_dir_all(bucket_dir.join("stale/nested/residue"))
|
||||
.await
|
||||
.expect("stale backing directory should be created");
|
||||
|
||||
async fn scan_prefixes(disk: &LocalDisk, bucket: &str, skip_hidden_prefix_check: bool) -> (Vec<String>, usize) {
|
||||
let probe_count = Arc::new(AtomicUsize::new(0));
|
||||
let (reader, mut writer) = tokio::io::duplex(64 * 1024);
|
||||
@@ -17663,8 +17700,11 @@ mod test {
|
||||
let (fast_path_names, fast_path_probes) = scan_prefixes(&disk, bucket, true).await;
|
||||
|
||||
assert_eq!(conservative_names, expected_names);
|
||||
assert_eq!(fast_path_names, expected_names);
|
||||
assert_eq!(conservative_probes, PREFIX_COUNT * 3);
|
||||
let mut expected_fast_path_names = expected_names.clone();
|
||||
expected_fast_path_names.push("stale/".to_owned());
|
||||
assert_eq!(fast_path_names, expected_fast_path_names);
|
||||
let expected_probes = PREFIX_COUNT * 3 + 3;
|
||||
assert_eq!(conservative_probes, expected_probes);
|
||||
assert_eq!(fast_path_probes, 0);
|
||||
}
|
||||
|
||||
@@ -21407,11 +21447,10 @@ mod test {
|
||||
|
||||
/// Zero-length read bounds parity on the cache-HIT path (backlog#1173/#1180).
|
||||
/// A `length == 0` read past EOF must be rejected identically whether the
|
||||
/// descriptor is freshly opened (miss path) or served from the cache: the
|
||||
/// cache-hit branch fstats the descriptor to reproduce the miss path's
|
||||
/// `offset > len` check instead of returning empty unconditionally. Seeds
|
||||
/// the cache with a normal read so the zero-length reads are hits, then pins
|
||||
/// that UringBackend and StdBackend agree on every case.
|
||||
/// descriptor is freshly opened (miss path) or served from the cache. Seeds
|
||||
/// the cache with a normal read so the zero-length reads reuse the same
|
||||
/// open-time size snapshot, then pins that UringBackend and StdBackend agree
|
||||
/// on every case.
|
||||
#[cfg(target_os = "linux")]
|
||||
#[tokio::test(flavor = "multi_thread")]
|
||||
async fn uring_zero_length_read_bounds_match_std_on_cache_hit() {
|
||||
|
||||
@@ -677,15 +677,20 @@ impl DiskAPI for Disk {
|
||||
}
|
||||
|
||||
impl Disk {
|
||||
pub(crate) async fn delete_with_scanner_publication_lease(
|
||||
pub async fn delete_with_scanner_publication_lease_and_guard(
|
||||
&self,
|
||||
volume: &str,
|
||||
path: &str,
|
||||
opts: DeleteOptions,
|
||||
scanner_publication_lease_token: Option<Uuid>,
|
||||
external_guard: Option<Arc<dyn Send + Sync>>,
|
||||
) -> Result<()> {
|
||||
match self {
|
||||
Disk::Local(local_disk) => local_disk.delete(volume, path, opts).await,
|
||||
Disk::Local(local_disk) => {
|
||||
local_disk
|
||||
.delete_with_publication_guard(volume, path, opts, external_guard)
|
||||
.await
|
||||
}
|
||||
Disk::Remote(remote_disk) => {
|
||||
remote_disk
|
||||
.delete_with_scanner_publication_lease(volume, path, opts, scanner_publication_lease_token)
|
||||
@@ -714,11 +719,34 @@ impl Disk {
|
||||
dst_volume: &str,
|
||||
dst_path: &str,
|
||||
scanner_publication_lease_token: Option<Uuid>,
|
||||
) -> Result<RenameDataResp> {
|
||||
self.rename_data_borrowed_with_fence_and_guard(
|
||||
src_volume,
|
||||
src_path,
|
||||
fi,
|
||||
dst_volume,
|
||||
dst_path,
|
||||
scanner_publication_lease_token,
|
||||
None,
|
||||
)
|
||||
.await
|
||||
}
|
||||
|
||||
#[allow(clippy::too_many_arguments)]
|
||||
pub async fn rename_data_borrowed_with_fence_and_guard(
|
||||
&self,
|
||||
src_volume: &str,
|
||||
src_path: &str,
|
||||
fi: &FileInfo,
|
||||
dst_volume: &str,
|
||||
dst_path: &str,
|
||||
scanner_publication_lease_token: Option<Uuid>,
|
||||
external_guard: Option<Arc<dyn Send + Sync>>,
|
||||
) -> Result<RenameDataResp> {
|
||||
match self {
|
||||
Disk::Local(local_disk) => {
|
||||
local_disk
|
||||
.rename_data_borrowed(src_volume, src_path, fi, dst_volume, dst_path)
|
||||
.rename_data_borrowed_with_guard(src_volume, src_path, fi, dst_volume, dst_path, external_guard)
|
||||
.await
|
||||
}
|
||||
Disk::Remote(remote_disk) => {
|
||||
|
||||
+248
-33
@@ -298,6 +298,39 @@ pub(crate) mod windows_rename_test_hooks {
|
||||
}
|
||||
}
|
||||
|
||||
/// Test-only hooks into the destination-parent walk of rename preparation.
|
||||
///
|
||||
/// The prune race lives between two syscalls inside
|
||||
/// [`mkdir_all_below_existing_base_std`], so only an injection at that exact
|
||||
/// point reproduces it deterministically. Hooks are keyed by the absolute path
|
||||
/// of the component just opened and queued per path: a retrying preparation
|
||||
/// visits the same component again, so a test models a pruner that keeps
|
||||
/// walking upward by queueing one hook per visit.
|
||||
#[cfg(all(test, unix))]
|
||||
pub(crate) mod prepare_rename_test_hooks {
|
||||
use super::*;
|
||||
|
||||
type Hook = Box<dyn FnOnce() + Send>;
|
||||
|
||||
static AFTER_COMPONENT_OPENED: LazyLock<Mutex<HashMap<PathBuf, VecDeque<Hook>>>> =
|
||||
LazyLock::new(|| Mutex::new(HashMap::new()));
|
||||
|
||||
pub(crate) fn queue_after_component_opened(path: &Path, hook: impl FnOnce() + Send + 'static) {
|
||||
AFTER_COMPONENT_OPENED
|
||||
.lock()
|
||||
.entry(path.to_path_buf())
|
||||
.or_default()
|
||||
.push_back(Box::new(hook));
|
||||
}
|
||||
|
||||
pub(crate) fn run_after_component_opened(path: &Path) {
|
||||
let hook = AFTER_COMPONENT_OPENED.lock().get_mut(path).and_then(VecDeque::pop_front);
|
||||
if let Some(hook) = hook {
|
||||
hook();
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// Fsync a directory so recently created or renamed entries survive power loss.
|
||||
/// No-op on non-Unix platforms where directories cannot be opened for syncing.
|
||||
pub fn fsync_dir_std(dir: impl AsRef<Path>) -> io::Result<()> {
|
||||
@@ -1905,8 +1938,8 @@ pub(crate) async fn rename_all_with_prepared_source(
|
||||
let base_dir = base_dir.clone();
|
||||
move || {
|
||||
validate_prepared_rename_source(&prepared_source, &src_file_path)?;
|
||||
let (preparation, attempt) = prepare_rename_with_retry(&src_file_path, &dst_file_path, &base_dir, &publication_root)?;
|
||||
rename_prepared(&src_file_path, &dst_file_path, &preparation, attempt)
|
||||
let preparation = prepare_rename_with_retry(&src_file_path, &dst_file_path, &base_dir, &publication_root)?;
|
||||
rename_prepared(&src_file_path, &dst_file_path, &preparation)
|
||||
}
|
||||
};
|
||||
let result = run_blocking_namespace_operation(lease, operation).await;
|
||||
@@ -2008,8 +2041,8 @@ async fn reliable_rename_inner_with_lease(
|
||||
let dst_file_path = dst_file_path.clone();
|
||||
let base_dir = base_dir.clone();
|
||||
move || {
|
||||
let (preparation, attempt) = prepare_rename_with_retry(&src_file_path, &dst_file_path, &base_dir, &publication_root)?;
|
||||
rename_prepared(&src_file_path, &dst_file_path, &preparation, attempt)
|
||||
let preparation = prepare_rename_with_retry(&src_file_path, &dst_file_path, &base_dir, &publication_root)?;
|
||||
rename_prepared(&src_file_path, &dst_file_path, &preparation)
|
||||
}
|
||||
};
|
||||
let result = run_blocking_namespace_operation(lease, operation).await;
|
||||
@@ -2233,12 +2266,13 @@ fn prepare_rename_with_retry(
|
||||
dst_file_path: &Path,
|
||||
base_dir: &Path,
|
||||
publication_root: &PublicationRoot,
|
||||
) -> io::Result<(RenamePreparation, usize)> {
|
||||
) -> io::Result<RenamePreparation> {
|
||||
let prune_budget = prepare_prune_budget(dst_file_path, base_dir);
|
||||
let mut attempt = 0;
|
||||
loop {
|
||||
match prepare_rename(src_file_path, dst_file_path, base_dir, publication_root) {
|
||||
Ok(preparation) => return Ok((preparation, attempt)),
|
||||
Err(err) if should_retry_rename(&err, attempt) => {
|
||||
Ok(preparation) => return Ok(preparation),
|
||||
Err(err) if should_retry_prepare(&err, attempt, prune_budget) => {
|
||||
attempt += 1;
|
||||
}
|
||||
Err(err) => return Err(err),
|
||||
@@ -2252,23 +2286,26 @@ fn prepare_rename_with_retry(
|
||||
dst_file_path: &Path,
|
||||
base_dir: &Path,
|
||||
publication_root: &PublicationRoot,
|
||||
) -> io::Result<(RenamePreparation, usize)> {
|
||||
) -> io::Result<RenamePreparation> {
|
||||
let source_parent = src_file_path
|
||||
.parent()
|
||||
.ok_or_else(|| io::Error::new(io::ErrorKind::InvalidInput, "rename source must have a parent directory"))?;
|
||||
let destination_parent = dst_file_path.parent();
|
||||
let mut attempt = 0;
|
||||
let prepare_destination_parent = |attempt: &mut usize| -> io::Result<Option<ExistingBaseDirectoryGuard>> {
|
||||
let prune_budget = prepare_prune_budget(dst_file_path, base_dir);
|
||||
// The destination walk and the source open below keep separate counters:
|
||||
// exhausting one must not deny the other its own retry.
|
||||
let prepare_destination_parent = || -> io::Result<Option<ExistingBaseDirectoryGuard>> {
|
||||
let mut attempt = 0;
|
||||
loop {
|
||||
let result = destination_parent
|
||||
.map(|parent| mkdir_all_below_existing_base_std(parent, base_dir, publication_root))
|
||||
.transpose();
|
||||
match result {
|
||||
Ok(parent_guard) => break Ok(parent_guard),
|
||||
Err(err) if should_retry_rename(&err, *attempt) => {
|
||||
Err(err) if should_retry_prepare(&err, attempt, prune_budget) => {
|
||||
#[cfg(test)]
|
||||
windows_rename_test_hooks::run_before_rename_retry(dst_file_path);
|
||||
*attempt += 1;
|
||||
attempt += 1;
|
||||
}
|
||||
Err(err) => break Err(err),
|
||||
}
|
||||
@@ -2281,7 +2318,7 @@ fn prepare_rename_with_retry(
|
||||
None => false,
|
||||
};
|
||||
let (source_parent_guard, parent_guard, source_identity_anchor, expected_source_identity) = if same_parent {
|
||||
let parent_guard = prepare_destination_parent(&mut attempt)?;
|
||||
let parent_guard = prepare_destination_parent()?;
|
||||
let source_parent_guard = parent_guard
|
||||
.as_ref()
|
||||
.ok_or_else(|| io::Error::new(io::ErrorKind::InvalidInput, "rename destination must have a parent directory"))?
|
||||
@@ -2293,16 +2330,17 @@ fn prepare_rename_with_retry(
|
||||
let source_parent_guard = lock_windows_directory_tree(source_parent, destination_parent, publication_root)?;
|
||||
let (source_identity_anchor, expected_source_identity) =
|
||||
open_windows_rename_source_identity(src_file_path, &source_parent_guard)?;
|
||||
let parent_guard = prepare_destination_parent(&mut attempt)?;
|
||||
let parent_guard = prepare_destination_parent()?;
|
||||
(source_parent_guard, parent_guard, source_identity_anchor, expected_source_identity)
|
||||
};
|
||||
let mut source_attempt = 0;
|
||||
let source = loop {
|
||||
match open_windows_rename_source(src_file_path, &source_parent_guard) {
|
||||
Ok(source) => break source,
|
||||
Err(err) if should_retry_rename(&err, attempt) => {
|
||||
Err(err) if should_retry_rename(&err, source_attempt) => {
|
||||
#[cfg(test)]
|
||||
windows_rename_test_hooks::run_before_rename_retry(dst_file_path);
|
||||
attempt += 1;
|
||||
source_attempt += 1;
|
||||
}
|
||||
Err(err) => return Err(err),
|
||||
}
|
||||
@@ -2315,14 +2353,11 @@ fn prepare_rename_with_retry(
|
||||
}
|
||||
drop(source_identity_anchor);
|
||||
|
||||
Ok((
|
||||
RenamePreparation {
|
||||
parent_guard,
|
||||
_source_parent_guard: source_parent_guard,
|
||||
source,
|
||||
},
|
||||
attempt,
|
||||
))
|
||||
Ok(RenamePreparation {
|
||||
parent_guard,
|
||||
_source_parent_guard: source_parent_guard,
|
||||
source,
|
||||
})
|
||||
}
|
||||
|
||||
#[cfg(not(windows))]
|
||||
@@ -2339,24 +2374,22 @@ fn prepare_rename(
|
||||
Ok(RenamePreparation { parent_guard })
|
||||
}
|
||||
|
||||
fn rename_prepared(
|
||||
_src_file_path: &Path,
|
||||
dst_file_path: &Path,
|
||||
preparation: &RenamePreparation,
|
||||
attempt: usize,
|
||||
) -> io::Result<()> {
|
||||
/// Publish a prepared rename. The retry budget starts fresh here: preparation
|
||||
/// keeps its own counter, so a chain rebuilt after a concurrent prune must not
|
||||
/// cost the rename its one retry.
|
||||
fn rename_prepared(_src_file_path: &Path, dst_file_path: &Path, preparation: &RenamePreparation) -> io::Result<()> {
|
||||
#[cfg(windows)]
|
||||
{
|
||||
let parent_guard = preparation
|
||||
.parent_guard
|
||||
.as_ref()
|
||||
.ok_or_else(|| io::Error::new(io::ErrorKind::InvalidInput, "rename destination must have a parent directory"))?;
|
||||
rename_windows_prepared(dst_file_path, parent_guard, &preparation.source, attempt)
|
||||
rename_windows_prepared(dst_file_path, parent_guard, &preparation.source, 0)
|
||||
}
|
||||
|
||||
#[cfg(not(windows))]
|
||||
{
|
||||
let mut attempt = attempt;
|
||||
let mut attempt = 0;
|
||||
loop {
|
||||
let rename_result = rename_into_existing_parent(_src_file_path, dst_file_path, preparation.parent_guard.as_ref());
|
||||
match rename_result {
|
||||
@@ -3756,6 +3789,8 @@ pub(crate) fn mkdir_all_below_existing_base_std(
|
||||
let mode = Mode::RWXU | Mode::RWXG | Mode::RWXO;
|
||||
let mut parents = vec![open(base_dir, flags, Mode::empty()).map_err(io::Error::from)?];
|
||||
|
||||
#[cfg(test)]
|
||||
let mut walked_path = base_dir.to_path_buf();
|
||||
for component in relative.components() {
|
||||
let Component::Normal(component) = component else {
|
||||
continue;
|
||||
@@ -3769,6 +3804,11 @@ pub(crate) fn mkdir_all_below_existing_base_std(
|
||||
Err(err) => return Err(err.into()),
|
||||
}
|
||||
parents.push(openat(parent, component, flags, Mode::empty()).map_err(io::Error::from)?);
|
||||
#[cfg(test)]
|
||||
{
|
||||
walked_path.push(component);
|
||||
prepare_rename_test_hooks::run_after_component_opened(&walked_path);
|
||||
}
|
||||
}
|
||||
|
||||
Ok(parents)
|
||||
@@ -3866,11 +3906,51 @@ fn warn_reliable_rename_failure(src_file_path: &Path, dst_file_path: &Path, base
|
||||
/// cleanup renames (e.g. `move_to_trash` on an already-removed tmp path) a
|
||||
/// pointless second syscall. This predicate is shared by the `rename_data`
|
||||
/// commit path via `rename_all`, so any relaxation here must keep genuine
|
||||
/// transient errors retryable.
|
||||
/// transient errors retryable. The *preparation* phase deliberately uses
|
||||
/// [`should_retry_prepare`] instead — see there for why `NotFound` is
|
||||
/// recoverable while the destination parent chain is still being built.
|
||||
fn should_retry_rename(err: &io::Error, attempt: usize) -> bool {
|
||||
attempt == 0 && err.kind() != io::ErrorKind::NotFound
|
||||
}
|
||||
|
||||
/// How many times rename preparation may retry a `NotFound`.
|
||||
///
|
||||
/// A pruning walk (`LocalDisk::delete_file`) removes empty ancestors
|
||||
/// monotonically upward and stops at the volume root, so it can invalidate
|
||||
/// each component *below* the base at most once. One attempt per such
|
||||
/// component therefore outlasts a pruning walk, and concurrent walks only
|
||||
/// steal an attempt by making that same upward progress. A destination whose
|
||||
/// parent *is* the base gets a budget of zero, keeping `NotFound` immediately
|
||||
/// terminal for speculative cleanup renames.
|
||||
fn prepare_prune_budget(dst_file_path: &Path, base_dir: &Path) -> usize {
|
||||
dst_file_path
|
||||
.parent()
|
||||
.and_then(|parent| parent.strip_prefix(base_dir).ok())
|
||||
.map(|relative| relative.components().count())
|
||||
.unwrap_or(0)
|
||||
}
|
||||
|
||||
/// Whether a failed rename *preparation* attempt (building the destination
|
||||
/// parent chain) should be retried.
|
||||
///
|
||||
/// Unlike [`should_retry_rename`], `NotFound` is recoverable here: a concurrent
|
||||
/// delete prunes now-empty parent directories, so it can unlink an intermediate
|
||||
/// destination component between this walk opening a directory and creating the
|
||||
/// next child inside it, which a handle-relative `mkdirat`/`openat` reports as
|
||||
/// `NotFound`. Each retry rebuilds the whole chain from the base directory,
|
||||
/// which no walk below it can remove; `prune_budget` bounds how far a pruner
|
||||
/// can push the walk back. A genuinely missing base directory fails identically
|
||||
/// on every attempt — the base is only ever opened, never created — so the
|
||||
/// missing-base contract holds at the cost of a few extra syscalls on an
|
||||
/// already-failing path.
|
||||
fn should_retry_prepare(err: &io::Error, attempt: usize, prune_budget: usize) -> bool {
|
||||
if err.kind() == io::ErrorKind::NotFound {
|
||||
attempt < prune_budget
|
||||
} else {
|
||||
attempt == 0
|
||||
}
|
||||
}
|
||||
|
||||
pub async fn reliable_mkdir_all(path: impl AsRef<Path>, base_dir: impl AsRef<Path>) -> io::Result<()> {
|
||||
let mut i = 0;
|
||||
|
||||
@@ -4401,6 +4481,141 @@ mod tests {
|
||||
assert!(!should_retry_rename(&denied, 1));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn prepare_retry_budget_covers_every_prunable_component() {
|
||||
// A pruner can invalidate each component below the base once, so the
|
||||
// budget must match the chain depth, not a fixed count.
|
||||
let not_found = io::Error::new(io::ErrorKind::NotFound, "pruned");
|
||||
assert!(should_retry_prepare(¬_found, 0, 2));
|
||||
assert!(should_retry_prepare(¬_found, 1, 2));
|
||||
assert!(!should_retry_prepare(¬_found, 2, 2));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn prepare_retry_keeps_other_errors_at_a_single_retry() {
|
||||
// Only a prune produces a recoverable NotFound; everything else keeps
|
||||
// the historical single retry so persistent failures stay cheap.
|
||||
let denied = io::Error::new(io::ErrorKind::PermissionDenied, "denied");
|
||||
assert!(should_retry_prepare(&denied, 0, 3));
|
||||
assert!(!should_retry_prepare(&denied, 1, 3));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn prepare_retry_budget_is_zero_when_the_parent_is_the_base() {
|
||||
// Speculative cleanup renames (move_to_trash) land directly in their
|
||||
// base, so a NotFound there is a missing base: terminal, not a prune.
|
||||
let base = Path::new("/vol");
|
||||
assert_eq!(prepare_prune_budget(Path::new("/vol/entry"), base), 0);
|
||||
assert_eq!(prepare_prune_budget(Path::new("/vol/data-movement/sha/id/xl.meta"), base), 3);
|
||||
let not_found = io::Error::new(io::ErrorKind::NotFound, "missing base");
|
||||
assert!(!should_retry_prepare(¬_found, 0, 0));
|
||||
}
|
||||
|
||||
#[cfg(unix)]
|
||||
#[tokio::test]
|
||||
async fn rename_all_survives_concurrent_empty_parent_prune() {
|
||||
// A multipart staging cleanup prunes the momentarily empty shared
|
||||
// `data-movement/` prefix while a concurrent upload publishes its
|
||||
// xl.meta below that same prefix. The writer's walk holds an fd to the
|
||||
// pruned component, so its next handle-relative mkdirat fails
|
||||
// NotFound; preparation must rebuild the chain and still publish.
|
||||
let temp_dir = tempdir().expect("create temp dir");
|
||||
let base = temp_dir.path().join("multipart-volume");
|
||||
let shared = base.join("data-movement");
|
||||
std::fs::create_dir_all(&shared).expect("create shared prefix");
|
||||
let src = temp_dir.path().join("staged.meta");
|
||||
std::fs::write(&src, b"payload").expect("write staged meta");
|
||||
let dst = shared.join("sha").join("upload-id").join("xl.meta");
|
||||
|
||||
let pruned = shared.clone();
|
||||
prepare_rename_test_hooks::queue_after_component_opened(&shared, move || {
|
||||
// The cleanup chain's empty-parent prune lands after the writer
|
||||
// opened the shared component but before it creates its child.
|
||||
std::fs::remove_dir(&pruned).expect("prune the empty shared prefix");
|
||||
});
|
||||
|
||||
rename_all(&src, &dst, &base)
|
||||
.await
|
||||
.expect("a concurrently pruned intermediate directory must not fail the publish");
|
||||
|
||||
assert_eq!(std::fs::read(&dst).expect("read published meta"), b"payload");
|
||||
assert!(!src.exists(), "publish must consume the staged source");
|
||||
}
|
||||
|
||||
#[cfg(unix)]
|
||||
#[tokio::test]
|
||||
async fn rename_all_survives_a_prune_walking_up_every_shared_component() {
|
||||
// The multipart data-movement chain has TWO shared components below the
|
||||
// volume (`data-movement/` and the per-object `<sha>/`), so one cleanup
|
||||
// walk pruning upward can invalidate the writer twice: once at <sha>,
|
||||
// then again at data-movement while the writer rebuilds. A budget that
|
||||
// covers only a single component would still break write quorum here.
|
||||
let temp_dir = tempdir().expect("create temp dir");
|
||||
let base = temp_dir.path().join("multipart-volume");
|
||||
let movement = base.join("data-movement");
|
||||
let sha = movement.join("sha");
|
||||
std::fs::create_dir_all(&sha).expect("create shared chain");
|
||||
let src = temp_dir.path().join("staged.meta");
|
||||
std::fs::write(&src, b"payload").expect("write staged meta");
|
||||
let dst = sha.join("upload-id").join("xl.meta");
|
||||
|
||||
// First visit of `data-movement` is the writer's initial walk, which the
|
||||
// pruner has not reached yet; it prunes on the writer's rebuild.
|
||||
prepare_rename_test_hooks::queue_after_component_opened(&movement, || {});
|
||||
let pruned_sha = sha.clone();
|
||||
prepare_rename_test_hooks::queue_after_component_opened(&sha, move || {
|
||||
std::fs::remove_dir(&pruned_sha).expect("prune the empty per-object prefix");
|
||||
});
|
||||
let pruned_movement = movement.clone();
|
||||
prepare_rename_test_hooks::queue_after_component_opened(&movement, move || {
|
||||
std::fs::remove_dir(&pruned_movement).expect("prune the empty data-movement prefix");
|
||||
});
|
||||
|
||||
rename_all(&src, &dst, &base)
|
||||
.await
|
||||
.expect("a prune walking up the whole shared chain must not fail the publish");
|
||||
|
||||
assert_eq!(std::fs::read(&dst).expect("read published meta"), b"payload");
|
||||
assert!(!src.exists(), "publish must consume the staged source");
|
||||
}
|
||||
|
||||
#[cfg(unix)]
|
||||
#[tokio::test]
|
||||
async fn rename_all_rejects_a_symlink_swapped_in_between_prepare_attempts() {
|
||||
// The retry must not become a traversal window: replacing the pruned
|
||||
// component with a symlink out of the volume before the rebuilt walk
|
||||
// reopens it must fail closed, exactly as a symlink staged before the
|
||||
// first attempt does.
|
||||
use std::os::unix::fs::symlink;
|
||||
|
||||
let temp_dir = tempdir().expect("create temp dir");
|
||||
let base = temp_dir.path().join("multipart-volume");
|
||||
let shared = base.join("data-movement");
|
||||
let outside = temp_dir.path().join("outside");
|
||||
std::fs::create_dir_all(&shared).expect("create shared prefix");
|
||||
std::fs::create_dir_all(&outside).expect("create outside target");
|
||||
let src = temp_dir.path().join("staged.meta");
|
||||
std::fs::write(&src, b"payload").expect("write staged meta");
|
||||
let dst = shared.join("sha").join("upload-id").join("xl.meta");
|
||||
|
||||
let swapped = shared.clone();
|
||||
let outside_target = outside.clone();
|
||||
prepare_rename_test_hooks::queue_after_component_opened(&shared, move || {
|
||||
std::fs::remove_dir(&swapped).expect("prune the shared prefix");
|
||||
symlink(&outside_target, &swapped).expect("replace the pruned component with a symlink");
|
||||
});
|
||||
|
||||
rename_all(&src, &dst, &base)
|
||||
.await
|
||||
.expect_err("a symlink swapped in between attempts must not be followed");
|
||||
|
||||
assert!(src.exists(), "rejected publish must preserve the staged source");
|
||||
assert!(
|
||||
!outside.join("sha").exists(),
|
||||
"the rebuilt walk must not create or publish through the replacement symlink"
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn is_dir_not_empty_error_recognizes_directory_not_empty_kind() {
|
||||
let err = io::Error::from(io::ErrorKind::DirectoryNotEmpty);
|
||||
|
||||
@@ -16,10 +16,143 @@ use rustfs_filemeta::{MetacacheReader, MetacacheWriter};
|
||||
use std::io::Cursor;
|
||||
use std::path::PathBuf;
|
||||
use std::sync::Arc;
|
||||
use std::sync::atomic::{AtomicBool, Ordering};
|
||||
use tokio::fs;
|
||||
use tokio::io::AsyncReadExt;
|
||||
use tokio::sync::RwLock;
|
||||
|
||||
/// Test-only lock client whose refresh path can be rejected independently of
|
||||
/// every other lock operation. The observed event is awaitable so lock-loss
|
||||
/// tests do not depend on sleeps or scheduler timing.
|
||||
#[derive(Debug)]
|
||||
pub(crate) struct RefreshLossLockClient {
|
||||
inner: rustfs_lock::LocalClient,
|
||||
reject_refresh: AtomicBool,
|
||||
rejected_refresh: AtomicBool,
|
||||
rejected_refresh_notify: tokio::sync::Notify,
|
||||
}
|
||||
|
||||
impl RefreshLossLockClient {
|
||||
pub(crate) fn with_manager(manager: Arc<rustfs_lock::GlobalLockManager>) -> Self {
|
||||
Self {
|
||||
inner: rustfs_lock::LocalClient::with_manager(manager),
|
||||
reject_refresh: AtomicBool::new(false),
|
||||
rejected_refresh: AtomicBool::new(false),
|
||||
rejected_refresh_notify: tokio::sync::Notify::new(),
|
||||
}
|
||||
}
|
||||
|
||||
pub(crate) fn reject_refreshes(&self) {
|
||||
self.reject_refresh.store(true, Ordering::Release);
|
||||
}
|
||||
|
||||
pub(crate) fn refreshes_rejected(&self) -> bool {
|
||||
self.rejected_refresh.load(Ordering::Acquire)
|
||||
}
|
||||
|
||||
pub(crate) async fn wait_for_rejected_refresh(
|
||||
&self,
|
||||
timeout: std::time::Duration,
|
||||
) -> std::result::Result<(), tokio::time::error::Elapsed> {
|
||||
tokio::time::timeout(timeout, async {
|
||||
loop {
|
||||
let notified = self.rejected_refresh_notify.notified();
|
||||
if self.refreshes_rejected() {
|
||||
return;
|
||||
}
|
||||
notified.await;
|
||||
}
|
||||
})
|
||||
.await
|
||||
}
|
||||
}
|
||||
|
||||
#[async_trait::async_trait]
|
||||
impl rustfs_lock::LockClient for RefreshLossLockClient {
|
||||
async fn acquire_lock(&self, request: &rustfs_lock::LockRequest) -> rustfs_lock::Result<rustfs_lock::LockResponse> {
|
||||
rustfs_lock::LockClient::acquire_lock(&self.inner, request).await
|
||||
}
|
||||
|
||||
async fn release(&self, lock_id: &rustfs_lock::LockId) -> rustfs_lock::Result<bool> {
|
||||
rustfs_lock::LockClient::release(&self.inner, lock_id).await
|
||||
}
|
||||
|
||||
async fn refresh(&self, lock_id: &rustfs_lock::LockId) -> rustfs_lock::Result<bool> {
|
||||
if self.reject_refresh.load(Ordering::Acquire) {
|
||||
self.rejected_refresh.store(true, Ordering::Release);
|
||||
self.rejected_refresh_notify.notify_waiters();
|
||||
return Ok(false);
|
||||
}
|
||||
rustfs_lock::LockClient::refresh(&self.inner, lock_id).await
|
||||
}
|
||||
|
||||
async fn force_release(&self, lock_id: &rustfs_lock::LockId) -> rustfs_lock::Result<bool> {
|
||||
rustfs_lock::LockClient::force_release(&self.inner, lock_id).await
|
||||
}
|
||||
|
||||
async fn check_status(&self, lock_id: &rustfs_lock::LockId) -> rustfs_lock::Result<Option<rustfs_lock::LockInfo>> {
|
||||
rustfs_lock::LockClient::check_status(&self.inner, lock_id).await
|
||||
}
|
||||
|
||||
async fn list_lock_leases(&self) -> Vec<rustfs_lock::LockLeaseInfo> {
|
||||
rustfs_lock::LockClient::list_lock_leases(&self.inner).await
|
||||
}
|
||||
|
||||
async fn get_stats(&self) -> rustfs_lock::Result<rustfs_lock::LockStats> {
|
||||
rustfs_lock::LockClient::get_stats(&self.inner).await
|
||||
}
|
||||
|
||||
async fn close(&self) -> rustfs_lock::Result<()> {
|
||||
rustfs_lock::LockClient::close(&self.inner).await
|
||||
}
|
||||
|
||||
async fn is_online(&self) -> bool {
|
||||
rustfs_lock::LockClient::is_online(&self.inner).await
|
||||
}
|
||||
|
||||
async fn is_local(&self) -> bool {
|
||||
rustfs_lock::LockClient::is_local(&self.inner).await
|
||||
}
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
async fn refresh_loss_lock_client_keeps_rejection_observable_for_late_waiters() {
|
||||
let manager = Arc::new(rustfs_lock::GlobalLockManager::Enabled(Arc::new(
|
||||
rustfs_lock::FastObjectLockManager::new(),
|
||||
)));
|
||||
let client = RefreshLossLockClient::with_manager(manager);
|
||||
let resource = rustfs_lock::ObjectKey::new("bucket", "object");
|
||||
let response = rustfs_lock::LockClient::acquire_lock(
|
||||
&client,
|
||||
&rustfs_lock::LockRequest::new(resource, rustfs_lock::LockType::Shared, "refresh-loss-harness"),
|
||||
)
|
||||
.await
|
||||
.expect("acquire should reach the inner local client");
|
||||
let lock_id = response.lock_info.expect("the inner local client should acquire the lock").id;
|
||||
assert_eq!(
|
||||
rustfs_lock::LockClient::list_lock_leases(&client).await.len(),
|
||||
1,
|
||||
"lease diagnostics must remain transparent through the refresh wrapper"
|
||||
);
|
||||
|
||||
client.reject_refreshes();
|
||||
assert!(
|
||||
!rustfs_lock::LockClient::refresh(&client, &lock_id)
|
||||
.await
|
||||
.expect("refresh should return a response")
|
||||
);
|
||||
client
|
||||
.wait_for_rejected_refresh(std::time::Duration::from_millis(50))
|
||||
.await
|
||||
.expect("a waiter registered after rejection must still observe the event");
|
||||
assert!(client.refreshes_rejected());
|
||||
assert!(
|
||||
rustfs_lock::LockClient::release(&client, &lock_id)
|
||||
.await
|
||||
.expect("release should reach the inner local client")
|
||||
);
|
||||
}
|
||||
|
||||
/// Returns the backing [`tempfile::TempDir`]s alongside the set so callers keep
|
||||
/// them alive for the test's duration and the directories are removed on drop.
|
||||
pub(crate) async fn make_local_set_disks(drive_count: usize, parity_count: usize) -> (Vec<tempfile::TempDir>, Arc<SetDisks>) {
|
||||
|
||||
@@ -46,6 +46,7 @@ type ShardReadFuture<'a> = Pin<Box<dyn Future<Output = (usize, ShardReadCost, Re
|
||||
type OwnedShardReadFuture<'a, R> =
|
||||
Pin<Box<dyn Future<Output = (usize, ShardReadCost, Result<Vec<u8>, Error>, Option<BitrotReader<R>>, bool)> + Send + 'a>>;
|
||||
pub(crate) type DeferredReaderReopener<R> = Arc<dyn Fn(usize) -> Option<BitrotReader<R>> + Send + Sync>;
|
||||
pub(crate) type DecodeOutcome = (usize, Option<std::io::Error>, bool);
|
||||
|
||||
type ShardIndexes = SmallVec<[usize; INLINE_SHARD_SLOTS]>;
|
||||
type ActiveReaders = SmallVec<[bool; INLINE_SHARD_SLOTS]>;
|
||||
@@ -574,6 +575,7 @@ pub(crate) struct ParallelReader<R> {
|
||||
read_timeout: Duration,
|
||||
verify_reconstruction: bool,
|
||||
locality_preference_enabled: bool,
|
||||
demand_bound_lockstep: bool,
|
||||
// Request-scoped shard buffers keyed by shard index. Keeping ownership in
|
||||
// `ParallelReader` avoids dropping unused parity/backup slot buffers between stripes.
|
||||
buffers: ShardBufferPool,
|
||||
@@ -585,10 +587,8 @@ pub(crate) struct ParallelReader<R> {
|
||||
// it to the current stripe when it is engaged mid-object (backlog#923).
|
||||
engaged: SmallVec<[bool; INLINE_SHARD_SLOTS]>,
|
||||
deferred_handles: Vec<Option<DeferredReaderStripeHandle>>,
|
||||
// Copy-source hedges use a fresh deferred reader so cancelling a hedge
|
||||
// never consumes the unopened reader reserved for a later stripe. The
|
||||
// vector is empty for callers that do not provide a reopen factory (tests
|
||||
// and the ordinary GET path retain the handle-based behavior).
|
||||
// Demand-bound hedges use a fresh deferred reader so cancelling a hedge
|
||||
// never consumes the unopened reader reserved for a later stripe.
|
||||
deferred_reopeners: Vec<Option<DeferredReaderReopener<R>>>,
|
||||
stripe_index: usize,
|
||||
}
|
||||
@@ -777,9 +777,9 @@ where
|
||||
// reads all live readers on every stripe — the pre-backlog#923
|
||||
// behavior. With the gate on, only data slots start engaged; parity is
|
||||
// engaged on demand, stripe-aligned through its deferred handle.
|
||||
let data_shards_only = get_lockstep_data_shards_only_enabled();
|
||||
let demand_bound_lockstep = get_lockstep_data_shards_only_enabled();
|
||||
let engaged: SmallVec<_> = (0..readers.len())
|
||||
.map(|index| !data_shards_only || index < e.data_shards)
|
||||
.map(|index| !demand_bound_lockstep || index < e.data_shards)
|
||||
.collect();
|
||||
ParallelReader {
|
||||
readers,
|
||||
@@ -793,6 +793,7 @@ where
|
||||
read_timeout,
|
||||
verify_reconstruction,
|
||||
locality_preference_enabled: get_shard_locality_preference_enabled(),
|
||||
demand_bound_lockstep,
|
||||
buffers: ShardBufferPool::new(e.data_shards + e.parity_shards),
|
||||
stripe_state: None,
|
||||
engaged,
|
||||
@@ -1275,7 +1276,7 @@ where
|
||||
/// realigned (no pending deferred handle) is likewise retired instead of
|
||||
/// being read out of position.
|
||||
async fn read_lockstep(&mut self, state: &mut StripeReadState) {
|
||||
if matches!(decode_read_policy(), DecodeReadPolicy::DemandBound) {
|
||||
if self.demand_bound_lockstep {
|
||||
self.read_lockstep_demand_bound(state).await;
|
||||
return;
|
||||
}
|
||||
@@ -1531,17 +1532,18 @@ where
|
||||
}
|
||||
}
|
||||
|
||||
/// Demand-bound lockstep stripe read used by server-side copy sources.
|
||||
/// Demand-bound data-shards-only lockstep stripe read.
|
||||
///
|
||||
/// The ordinary lockstep path can cancel every in-flight reader once it
|
||||
/// has a quorum because all of its parity readers are already engaged.
|
||||
/// Copy sources keep parity unopened until a data reader is missing. A
|
||||
/// hedge therefore has to race the deferred parity reads against the
|
||||
/// original data reads and may retire the latter only after the parity has
|
||||
/// produced an actual decode-plus-verification quorum. The futures own
|
||||
/// their readers so disjoint data/parity slots can be admitted while the
|
||||
/// other group is still pending; dropping an abandoned future retires its
|
||||
/// stream without leaving a borrowed slot behind.
|
||||
/// Copy sources and the data-shards-only rollout gate keep parity unopened
|
||||
/// until a data reader is missing. A hedge therefore has to race the
|
||||
/// deferred parity reads against the original data reads and may retire the
|
||||
/// latter only after parity has produced an actual decode-plus-verification
|
||||
/// quorum. The futures own their readers so disjoint data/parity slots can
|
||||
/// be admitted while the other group is still pending; dropping an
|
||||
/// abandoned future retires its stream without leaving a borrowed slot
|
||||
/// behind.
|
||||
async fn read_lockstep_demand_bound(&mut self, state: &mut StripeReadState) {
|
||||
let num_readers = self.readers.len();
|
||||
state.reset(num_readers, self.data_shards);
|
||||
@@ -1576,14 +1578,14 @@ where
|
||||
let mut completed = 0usize;
|
||||
let mut failed = 0usize;
|
||||
let mut first_shard_recorded = false;
|
||||
let mut active = vec![false; num_readers];
|
||||
let mut temporary_parity = vec![false; num_readers];
|
||||
let mut active: ActiveReaders = smallvec![false; num_readers];
|
||||
let mut temporary_parity: ActiveReaders = smallvec![false; num_readers];
|
||||
// A deferred parity slot is attempted at most once per stripe. A
|
||||
// failed disposable hedge keeps its unopened reserve for the next
|
||||
// stripe, but must not be relaunched in a tight same-stripe retry
|
||||
// loop (which would defeat the bounded fan-out and amplify a remote
|
||||
// outage).
|
||||
let mut attempted_parity = vec![false; num_readers];
|
||||
let mut attempted_parity: ActiveReaders = smallvec![false; num_readers];
|
||||
// Once a data reader has returned an error (or was already missing at
|
||||
// setup), the loss is permanent for lockstep alignment. Use the
|
||||
// deferred handle and keep parity engaged across subsequent stripes;
|
||||
@@ -2189,8 +2191,10 @@ impl Erasure {
|
||||
W: AsyncWrite + Send + Sync + Unpin,
|
||||
R: crate::erasure::coding::ShardSource,
|
||||
{
|
||||
self.decode_inner(writer, readers, offset, length, total_length, None, Vec::new(), Vec::new())
|
||||
.await
|
||||
let (written, error, _) = self
|
||||
.decode_inner(writer, readers, offset, length, total_length, None, Vec::new(), Vec::new())
|
||||
.await;
|
||||
(written, error)
|
||||
}
|
||||
|
||||
#[allow(dead_code, reason = "read-cost decode path asserted by this file's tests (backlog#1823)")]
|
||||
@@ -2207,8 +2211,10 @@ impl Erasure {
|
||||
W: AsyncWrite + Send + Sync + Unpin,
|
||||
R: crate::erasure::coding::ShardSource,
|
||||
{
|
||||
self.decode_inner(writer, readers, offset, length, total_length, Some(read_costs), Vec::new(), Vec::new())
|
||||
.await
|
||||
let (written, error, _) = self
|
||||
.decode_inner(writer, readers, offset, length, total_length, Some(read_costs), Vec::new(), Vec::new())
|
||||
.await;
|
||||
(written, error)
|
||||
}
|
||||
|
||||
/// GET decode entry point that also carries the deferred-parity stripe
|
||||
@@ -2261,6 +2267,37 @@ impl Erasure {
|
||||
deferred_handles: Vec<Option<DeferredReaderStripeHandle>>,
|
||||
deferred_reopeners: Vec<Option<DeferredReaderReopener<R>>>,
|
||||
) -> (usize, Option<std::io::Error>)
|
||||
where
|
||||
W: AsyncWrite + Send + Sync + Unpin,
|
||||
R: crate::erasure::coding::ShardSource,
|
||||
{
|
||||
let (written, error, _) = self
|
||||
.decode_inner(
|
||||
writer,
|
||||
readers,
|
||||
offset,
|
||||
length,
|
||||
total_length,
|
||||
read_costs,
|
||||
deferred_handles,
|
||||
deferred_reopeners,
|
||||
)
|
||||
.await;
|
||||
(written, error)
|
||||
}
|
||||
|
||||
#[allow(clippy::too_many_arguments)]
|
||||
pub(crate) async fn decode_with_stripe_handles_and_reopeners_with_diagnostics<W, R>(
|
||||
&self,
|
||||
writer: &mut W,
|
||||
readers: Vec<Option<BitrotReader<R>>>,
|
||||
offset: usize,
|
||||
length: usize,
|
||||
total_length: usize,
|
||||
read_costs: Option<Vec<ShardReadCost>>,
|
||||
deferred_handles: Vec<Option<DeferredReaderStripeHandle>>,
|
||||
deferred_reopeners: Vec<Option<DeferredReaderReopener<R>>>,
|
||||
) -> DecodeOutcome
|
||||
where
|
||||
W: AsyncWrite + Send + Sync + Unpin,
|
||||
R: crate::erasure::coding::ShardSource,
|
||||
@@ -2298,6 +2335,7 @@ impl Erasure {
|
||||
written: &mut usize,
|
||||
ret_err: &mut Option<std::io::Error>,
|
||||
stage_metrics_enabled: bool,
|
||||
require_surplus_source: bool,
|
||||
) -> StripeFlow
|
||||
where
|
||||
W: AsyncWrite + Send + Sync + Unpin,
|
||||
@@ -2335,7 +2373,12 @@ impl Erasure {
|
||||
// missing data shard and an extra source shard was available, verify
|
||||
// the reconstructed data against that source before streaming bytes.
|
||||
let reconstruct_stage_start = get_stage_timer_if_enabled(stage_metrics_enabled);
|
||||
if let Err(e) = self.decode_data_with_reconstruction_verification(shards) {
|
||||
let decode_result = if require_surplus_source {
|
||||
self.decode_data_with_reconstruction_verification_for_lockstep(shards)
|
||||
} else {
|
||||
self.decode_data_with_reconstruction_verification(shards)
|
||||
};
|
||||
if let Err(e) = decode_result {
|
||||
record_get_stage_duration_if_enabled(GET_OBJECT_PATH_LEGACY_DUPLEX, GET_STAGE_RECONSTRUCT, reconstruct_stage_start);
|
||||
let reason = GetObjectFailureReason::DecodeError;
|
||||
error!(
|
||||
@@ -2404,36 +2447,48 @@ impl Erasure {
|
||||
read_costs: Option<Vec<ShardReadCost>>,
|
||||
deferred_handles: Vec<Option<DeferredReaderStripeHandle>>,
|
||||
deferred_reopeners: Vec<Option<DeferredReaderReopener<R>>>,
|
||||
) -> (usize, Option<std::io::Error>)
|
||||
) -> DecodeOutcome
|
||||
where
|
||||
W: AsyncWrite + Send + Sync + Unpin,
|
||||
R: crate::erasure::coding::ShardSource,
|
||||
{
|
||||
if readers.len() != self.data_shards + self.parity_shards {
|
||||
record_get_object_pipeline_failure(GET_STAGE_RANGE, GetObjectFailureReason::RangeOrLengthInvalid);
|
||||
return (0, Some(io::Error::new(ErrorKind::InvalidInput, "Invalid number of readers")));
|
||||
return (0, Some(io::Error::new(ErrorKind::InvalidInput, "Invalid number of readers")), false);
|
||||
}
|
||||
|
||||
// block_size/data_shards come from on-disk metadata; a corrupt FileInfo with a
|
||||
// zero here must surface as an error, not a divide-by-zero panic on every GET.
|
||||
if self.block_size == 0 || self.data_shards == 0 {
|
||||
record_get_object_pipeline_failure(GET_STAGE_RANGE, GetObjectFailureReason::RangeOrLengthInvalid);
|
||||
return (0, Some(io::Error::new(ErrorKind::InvalidInput, "Invalid erasure coding parameters")));
|
||||
return (
|
||||
0,
|
||||
Some(io::Error::new(ErrorKind::InvalidInput, "Invalid erasure coding parameters")),
|
||||
false,
|
||||
);
|
||||
}
|
||||
|
||||
let Some(end_offset) = offset.checked_add(length) else {
|
||||
record_get_object_pipeline_failure(GET_STAGE_RANGE, GetObjectFailureReason::RangeOrLengthInvalid);
|
||||
return (0, Some(io::Error::new(ErrorKind::InvalidInput, "offset + length exceeds total length")));
|
||||
return (
|
||||
0,
|
||||
Some(io::Error::new(ErrorKind::InvalidInput, "offset + length exceeds total length")),
|
||||
false,
|
||||
);
|
||||
};
|
||||
if end_offset > total_length {
|
||||
record_get_object_pipeline_failure(GET_STAGE_RANGE, GetObjectFailureReason::RangeOrLengthInvalid);
|
||||
return (0, Some(io::Error::new(ErrorKind::InvalidInput, "offset + length exceeds total length")));
|
||||
return (
|
||||
0,
|
||||
Some(io::Error::new(ErrorKind::InvalidInput, "offset + length exceeds total length")),
|
||||
false,
|
||||
);
|
||||
}
|
||||
|
||||
let mut ret_err = None;
|
||||
|
||||
if length == 0 {
|
||||
return (0, ret_err);
|
||||
return (0, ret_err, false);
|
||||
}
|
||||
|
||||
let mut written = 0;
|
||||
@@ -2473,6 +2528,7 @@ impl Erasure {
|
||||
}
|
||||
};
|
||||
|
||||
let mut exact_quorum = false;
|
||||
if legacy_stripe_prefetch_enabled() {
|
||||
// Depth-1 stripe prefetch (backlog#930 HP-9 step 2): while the current
|
||||
// stripe is reconstructed and emitted, the next stripe's shard reads
|
||||
@@ -2515,6 +2571,7 @@ impl Erasure {
|
||||
let Some((mut shards, errs)) = current.take() else {
|
||||
break;
|
||||
};
|
||||
exact_quorum |= shards.iter().filter(|shard| shard.is_some()).count() == self.data_shards;
|
||||
|
||||
if idx + 1 < blocks.len() {
|
||||
// Overlap: read stripe idx+1 while reconstructing/emitting idx.
|
||||
@@ -2546,6 +2603,7 @@ impl Erasure {
|
||||
// `shards` are borrowed again below. In the `Stop` case that
|
||||
// drop is what cancels the still-in-flight prefetch read.
|
||||
let (flow, next): (Option<StripeFlow>, Option<StripeReadOutput>) = {
|
||||
let require_surplus_source = reader.demand_bound_lockstep;
|
||||
let read_fut = read_stripe_timed(&mut reader, stage_metrics_enabled);
|
||||
let emit_fut = self.emit_decoded_stripe(
|
||||
writer,
|
||||
@@ -2556,6 +2614,7 @@ impl Erasure {
|
||||
&mut written,
|
||||
&mut ret_err,
|
||||
stage_metrics_enabled,
|
||||
require_surplus_source,
|
||||
);
|
||||
tokio::pin!(read_fut);
|
||||
tokio::pin!(emit_fut);
|
||||
@@ -2603,6 +2662,7 @@ impl Erasure {
|
||||
&mut written,
|
||||
&mut ret_err,
|
||||
stage_metrics_enabled,
|
||||
reader.demand_bound_lockstep,
|
||||
)
|
||||
.await
|
||||
{
|
||||
@@ -2626,6 +2686,7 @@ impl Erasure {
|
||||
let stage_metrics_enabled = rustfs_io_metrics::get_stage_metrics_enabled();
|
||||
let stripe_read_stage_start = get_stage_timer_if_enabled(stage_metrics_enabled);
|
||||
let (mut shards, errs) = reader.read().await;
|
||||
exact_quorum |= shards.iter().filter(|shard| shard.is_some()).count() == self.data_shards;
|
||||
record_get_stage_duration_if_enabled(
|
||||
GET_OBJECT_PATH_LEGACY_DUPLEX,
|
||||
GET_STAGE_STRIPE_READ,
|
||||
@@ -2642,6 +2703,7 @@ impl Erasure {
|
||||
&mut written,
|
||||
&mut ret_err,
|
||||
stage_metrics_enabled,
|
||||
reader.demand_bound_lockstep,
|
||||
)
|
||||
.await
|
||||
{
|
||||
@@ -2654,14 +2716,14 @@ impl Erasure {
|
||||
}
|
||||
|
||||
if ret_err.is_some() {
|
||||
return (written, ret_err);
|
||||
return (written, ret_err, exact_quorum);
|
||||
}
|
||||
|
||||
if written < length {
|
||||
ret_err = Some(Error::LessData.into());
|
||||
}
|
||||
|
||||
(written, ret_err)
|
||||
(written, ret_err, exact_quorum)
|
||||
}
|
||||
}
|
||||
|
||||
@@ -2866,6 +2928,7 @@ mod tests {
|
||||
cursor: Cursor<Vec<u8>>,
|
||||
stall: Duration,
|
||||
sleep: Option<Pin<Box<Sleep>>>,
|
||||
stall_polls: Arc<AtomicUsize>,
|
||||
},
|
||||
}
|
||||
|
||||
@@ -2904,7 +2967,12 @@ mod tests {
|
||||
TestShardReader::TerminalFileNotFound => {
|
||||
Poll::Ready(Err(crate::disk::error::terminal_read_error_to_io(Error::FileNotFound)))
|
||||
}
|
||||
TestShardReader::PrefixThenSlow { cursor, stall, sleep } => {
|
||||
TestShardReader::PrefixThenSlow {
|
||||
cursor,
|
||||
stall,
|
||||
sleep,
|
||||
stall_polls,
|
||||
} => {
|
||||
let before = buf.filled().len();
|
||||
match Pin::new(cursor).poll_read(cx, buf) {
|
||||
// Cursor still has bytes for the current stripe: serve them.
|
||||
@@ -2914,6 +2982,7 @@ mod tests {
|
||||
// the task cleanly (no busy `wake_by_ref` spin), letting the
|
||||
// `#[tokio::test(start_paused = true)]` clock auto-advance.
|
||||
Poll::Ready(Ok(())) => {
|
||||
stall_polls.fetch_add(1, Ordering::SeqCst);
|
||||
let stall = *stall;
|
||||
let sleeper = sleep.get_or_insert_with(|| Box::pin(tokio::time::sleep(stall)));
|
||||
let _ = sleeper.as_mut().poll(cx);
|
||||
@@ -2942,6 +3011,29 @@ mod tests {
|
||||
}
|
||||
}
|
||||
|
||||
struct YieldOnceThenFailWriter {
|
||||
yielded: bool,
|
||||
}
|
||||
|
||||
impl AsyncWrite for YieldOnceThenFailWriter {
|
||||
fn poll_write(mut self: Pin<&mut Self>, cx: &mut Context<'_>, _buf: &[u8]) -> Poll<io::Result<usize>> {
|
||||
if !self.yielded {
|
||||
self.yielded = true;
|
||||
cx.waker().wake_by_ref();
|
||||
return Poll::Pending;
|
||||
}
|
||||
Poll::Ready(Err(io::Error::new(ErrorKind::BrokenPipe, "injected emit failure after prefetch poll")))
|
||||
}
|
||||
|
||||
fn poll_flush(self: Pin<&mut Self>, _cx: &mut Context<'_>) -> Poll<io::Result<()>> {
|
||||
Poll::Ready(Ok(()))
|
||||
}
|
||||
|
||||
fn poll_shutdown(self: Pin<&mut Self>, _cx: &mut Context<'_>) -> Poll<io::Result<()>> {
|
||||
Poll::Ready(Ok(()))
|
||||
}
|
||||
}
|
||||
|
||||
struct DownstreamClosedWriter;
|
||||
|
||||
impl AsyncWrite for DownstreamClosedWriter {
|
||||
@@ -3878,6 +3970,7 @@ mod tests {
|
||||
(rustfs_config::ENV_OBJECT_DISK_READ_TIMEOUT, Some(READ_TIMEOUT_SECS)),
|
||||
];
|
||||
temp_env::async_with_vars(vars, async {
|
||||
let stall_polls = Arc::new(AtomicUsize::new(0));
|
||||
let readers: Vec<Option<BitrotReader<TestShardReader>>> = shard_bufs
|
||||
.iter()
|
||||
.map(|buf| {
|
||||
@@ -3887,12 +3980,13 @@ mod tests {
|
||||
cursor: Cursor::new(prefix),
|
||||
stall: STALL,
|
||||
sleep: None,
|
||||
stall_polls: Arc::clone(&stall_polls),
|
||||
};
|
||||
Some(BitrotReader::new(reader, shard_size, hash_algo.clone(), false))
|
||||
})
|
||||
.collect();
|
||||
|
||||
let mut writer = FailingEmitWriter;
|
||||
let mut writer = YieldOnceThenFailWriter { yielded: false };
|
||||
let start = TokioInstant::now();
|
||||
let (written, err) = erasure.decode(&mut writer, readers, 0, total_len, total_len).await;
|
||||
let elapsed = start.elapsed();
|
||||
@@ -3900,6 +3994,10 @@ mod tests {
|
||||
// Emit failed on stripe 0, so the GET fails with no bytes emitted.
|
||||
assert!(err.is_some(), "emit failure must surface as an error");
|
||||
assert_eq!(written, 0, "the failing writer accepts no bytes");
|
||||
assert!(
|
||||
stall_polls.load(Ordering::SeqCst) > 0,
|
||||
"the speculative next-stripe read must be in flight before emit fails"
|
||||
);
|
||||
// The decisive assertion: the prefetch read was cancelled rather than
|
||||
// awaited. Without cancel-safety this would take READ_TIMEOUT_SECS.
|
||||
assert!(
|
||||
@@ -4911,6 +5009,24 @@ mod tests {
|
||||
/// read timeout even though both parity readers were available to engage.
|
||||
#[tokio::test]
|
||||
async fn test_demand_bound_lockstep_hedges_to_deferred_parity_quorum() {
|
||||
with_decode_read_policy(DecodeReadPolicy::DemandBound, assert_deferred_parity_hedges_slow_data()).await;
|
||||
}
|
||||
|
||||
/// The ordinary GET rollout gate must use the same bounded parity race as
|
||||
/// CopySource. Leaving it on the legacy lockstep loop deadlocks the hedge:
|
||||
/// that loop waits for a parity success before cancelling the slow data
|
||||
/// read, but does not admit deferred parity until after the data read ends.
|
||||
#[tokio::test]
|
||||
#[serial_test::serial]
|
||||
async fn test_data_shards_only_gate_hedges_to_deferred_parity_quorum() {
|
||||
temp_env::async_with_vars(
|
||||
[(ENV_RUSTFS_GET_LOCKSTEP_DATA_SHARDS_ONLY_ENABLE, Some("true"))],
|
||||
assert_deferred_parity_hedges_slow_data(),
|
||||
)
|
||||
.await;
|
||||
}
|
||||
|
||||
async fn assert_deferred_parity_hedges_slow_data() {
|
||||
const NUM_SHARDS: usize = 1;
|
||||
const BLOCK_SIZE: usize = 64;
|
||||
const DATA_SHARDS: usize = 2;
|
||||
@@ -4951,33 +5067,27 @@ mod tests {
|
||||
];
|
||||
|
||||
let erasure = Erasure::new(DATA_SHARDS, PARITY_SHARDS, BLOCK_SIZE);
|
||||
let (bufs, errs, engaged, readers_remaining) = with_decode_read_policy(DecodeReadPolicy::DemandBound, async {
|
||||
let mut parallel_reader = ParallelReader::new_with_metrics_path_read_costs_timeout_and_reconstruction_verification(
|
||||
readers,
|
||||
erasure,
|
||||
0,
|
||||
NUM_SHARDS * BLOCK_SIZE,
|
||||
None,
|
||||
vec![ShardReadCost::Unknown; DATA_SHARDS + PARITY_SHARDS],
|
||||
Duration::from_secs(60),
|
||||
true,
|
||||
);
|
||||
let (bufs, errs) = tokio::time::timeout(Duration::from_secs(2), parallel_reader.read())
|
||||
.await
|
||||
.expect("deferred parity must cover a hedged data shard without waiting for read_timeout");
|
||||
(
|
||||
bufs,
|
||||
errs,
|
||||
parallel_reader.engaged.clone(),
|
||||
parallel_reader.readers.iter().map(Option::is_some).collect::<Vec<_>>(),
|
||||
)
|
||||
})
|
||||
.await;
|
||||
let mut parallel_reader = ParallelReader::new_with_metrics_path_read_costs_timeout_and_reconstruction_verification(
|
||||
readers,
|
||||
erasure,
|
||||
0,
|
||||
NUM_SHARDS * BLOCK_SIZE,
|
||||
None,
|
||||
vec![ShardReadCost::Unknown; DATA_SHARDS + PARITY_SHARDS],
|
||||
Duration::from_secs(60),
|
||||
true,
|
||||
);
|
||||
let (bufs, errs) = tokio::time::timeout(Duration::from_secs(2), parallel_reader.read())
|
||||
.await
|
||||
.expect("deferred parity must cover a hedged data shard without waiting for read_timeout");
|
||||
|
||||
assert!(matches!(&errs[0], Some(DiskError::Io(err)) if err.kind() == ErrorKind::TimedOut));
|
||||
assert_eq!(bufs.iter().filter(|buf| buf.is_some()).count(), DATA_SHARDS + 1);
|
||||
assert_eq!(engaged.as_slice(), &[true, true, true, true]);
|
||||
assert_eq!(readers_remaining, vec![false, true, true, true]);
|
||||
assert_eq!(parallel_reader.engaged.as_slice(), &[true, true, true, true]);
|
||||
assert_eq!(
|
||||
parallel_reader.readers.iter().map(Option::is_some).collect::<Vec<_>>(),
|
||||
vec![false, true, true, true]
|
||||
);
|
||||
}
|
||||
|
||||
/// A fast data failure must admit deferred parity immediately. There is
|
||||
@@ -5046,6 +5156,24 @@ mod tests {
|
||||
|
||||
#[tokio::test]
|
||||
async fn test_demand_bound_canceled_hedge_preserves_deferred_parity_for_next_stripe() {
|
||||
with_decode_read_policy(
|
||||
DecodeReadPolicy::DemandBound,
|
||||
assert_canceled_hedge_preserves_deferred_parity_for_next_stripe(),
|
||||
)
|
||||
.await;
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
#[serial_test::serial]
|
||||
async fn test_data_shards_only_gate_canceled_hedge_preserves_deferred_parity_for_next_stripe() {
|
||||
temp_env::async_with_vars(
|
||||
[(ENV_RUSTFS_GET_LOCKSTEP_DATA_SHARDS_ONLY_ENABLE, Some("true"))],
|
||||
assert_canceled_hedge_preserves_deferred_parity_for_next_stripe(),
|
||||
)
|
||||
.await;
|
||||
}
|
||||
|
||||
async fn assert_canceled_hedge_preserves_deferred_parity_for_next_stripe() {
|
||||
const BLOCK_SIZE: usize = 64;
|
||||
const DATA_SHARDS: usize = 2;
|
||||
const PARITY_SHARDS: usize = 2;
|
||||
@@ -5094,7 +5222,7 @@ mod tests {
|
||||
Some(BitrotReader::new(TestShardReader::Pending, SHARD_SIZE, hash_algo, false)),
|
||||
];
|
||||
|
||||
let (first_parity_reserved, second_result) = with_decode_read_policy(DecodeReadPolicy::DemandBound, async {
|
||||
let (first_parity_reserved, second_result) = {
|
||||
let erasure = Erasure::new(DATA_SHARDS, PARITY_SHARDS, BLOCK_SIZE);
|
||||
let mut parallel_reader = ParallelReader::new_with_metrics_path_read_timeout_and_reconstruction_verification(
|
||||
readers,
|
||||
@@ -5155,8 +5283,7 @@ mod tests {
|
||||
parallel_reader.readers[2].is_some() && parallel_reader.readers[3].is_some(),
|
||||
(third_buffers, third_errors),
|
||||
)
|
||||
})
|
||||
.await;
|
||||
};
|
||||
|
||||
assert!(first_parity_reserved);
|
||||
assert_eq!(parity_calls.load(Ordering::SeqCst), PARITY_SHARDS * 2);
|
||||
@@ -5240,6 +5367,58 @@ mod tests {
|
||||
assert!(error.is_none(), "a failed disposable hedge must not fail a recovered stripe: {error:?}");
|
||||
}
|
||||
|
||||
/// Rollout guard for backlog#1308: when a data shard and the first parity
|
||||
/// hedge both fail, the gate-on path must not settle at decode quorum and
|
||||
/// emit an unverified body. The second parity can restore decode quorum but
|
||||
/// cannot provide the extra source required for reconstruction verification,
|
||||
/// so the stripe must fail before exposing bytes.
|
||||
#[tokio::test]
|
||||
#[serial_test::serial]
|
||||
async fn test_data_shards_only_gate_data_and_parity_failure_fails_before_output() {
|
||||
const BLOCK_SIZE: usize = 64;
|
||||
const DATA_SHARDS: usize = 2;
|
||||
const PARITY_SHARDS: usize = 2;
|
||||
|
||||
temp_env::async_with_vars([(ENV_RUSTFS_GET_LOCKSTEP_DATA_SHARDS_ONLY_ENABLE, Some("true"))], async {
|
||||
let erasure = Erasure::new(DATA_SHARDS, PARITY_SHARDS, BLOCK_SIZE);
|
||||
let payload = (0..BLOCK_SIZE).map(|value| value as u8).collect::<Vec<_>>();
|
||||
let shards = erasure.encode_data(&payload).expect("test payload should encode");
|
||||
let shard_size = erasure.shard_size();
|
||||
|
||||
let readers = vec![
|
||||
Some(BitrotReader::new(TestShardReader::TimedOut, shard_size, HashAlgorithm::None, false)),
|
||||
Some(BitrotReader::new(
|
||||
TestShardReader::Ready(Cursor::new(shards[1].to_vec())),
|
||||
shard_size,
|
||||
HashAlgorithm::None,
|
||||
false,
|
||||
)),
|
||||
Some(BitrotReader::new(
|
||||
TestShardReader::TerminalFileNotFound,
|
||||
shard_size,
|
||||
HashAlgorithm::None,
|
||||
false,
|
||||
)),
|
||||
Some(BitrotReader::new(
|
||||
TestShardReader::Ready(Cursor::new(shards[3].to_vec())),
|
||||
shard_size,
|
||||
HashAlgorithm::None,
|
||||
false,
|
||||
)),
|
||||
];
|
||||
|
||||
let mut output = Vec::new();
|
||||
let (written, error) = erasure.decode(&mut output, readers, 0, payload.len(), payload.len()).await;
|
||||
|
||||
assert_eq!(written, 0, "an unverified stripe must not report body bytes");
|
||||
assert!(output.is_empty(), "an unverified stripe must not expose a clean short body");
|
||||
let error = error.expect("data plus parity loss must fail closed");
|
||||
assert_eq!(error.kind(), ErrorKind::InvalidData);
|
||||
assert!(error.to_string().contains("insufficient source shards"));
|
||||
})
|
||||
.await;
|
||||
}
|
||||
|
||||
/// Lockstep verification-quorum regression (backlog#1156). When a data shard is
|
||||
/// missing, the hedge must settle only at `data_shards + 1` (decode quorum plus
|
||||
/// a reconstruction-verification source), never at exactly `data_shards` — that
|
||||
|
||||
@@ -321,6 +321,13 @@ impl<'a> MultiWriter<'a> {
|
||||
}
|
||||
}
|
||||
|
||||
pub(super) fn take_retryable_internode_write_failure(&mut self) -> Option<Error> {
|
||||
self.errs
|
||||
.iter_mut()
|
||||
.find(|error| error.as_ref().is_some_and(Error::is_retryable_internode_write_failure))
|
||||
.and_then(Option::take)
|
||||
}
|
||||
|
||||
/// Effective budget for one shard operation: the smaller of the per-shard
|
||||
/// stall timeout and the time remaining until the object's absolute cap.
|
||||
/// Returns `None` when neither deadline is configured (wait indefinitely).
|
||||
|
||||
@@ -933,8 +933,29 @@ impl Erasure {
|
||||
}
|
||||
|
||||
pub(crate) fn decode_data_with_reconstruction_verification(&self, shards: &mut [Option<Vec<u8>>]) -> io::Result<()> {
|
||||
self.decode_data_with_reconstruction_verification_policy(shards, false)
|
||||
}
|
||||
|
||||
pub(crate) fn decode_data_with_reconstruction_verification_for_lockstep(
|
||||
&self,
|
||||
shards: &mut [Option<Vec<u8>>],
|
||||
) -> io::Result<()> {
|
||||
self.decode_data_with_reconstruction_verification_policy(shards, true)
|
||||
}
|
||||
|
||||
fn decode_data_with_reconstruction_verification_policy(
|
||||
&self,
|
||||
shards: &mut [Option<Vec<u8>>],
|
||||
require_surplus_source: bool,
|
||||
) -> io::Result<()> {
|
||||
let missing_data_source = shards.iter().take(self.data_shards).any(|shard| shard.is_none());
|
||||
let available_shards = shards.iter().filter(|shard| shard.is_some()).count();
|
||||
if require_surplus_source && missing_data_source && available_shards == self.data_shards {
|
||||
return Err(io::Error::new(
|
||||
io::ErrorKind::InvalidData,
|
||||
"insufficient source shards to verify reconstructed data",
|
||||
));
|
||||
}
|
||||
let source_parity = if missing_data_source && available_shards > self.data_shards {
|
||||
shards
|
||||
.iter()
|
||||
@@ -1868,6 +1889,31 @@ mod tests {
|
||||
assert_eq!(err.kind(), io::ErrorKind::InvalidData);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn decode_data_with_verification_scopes_exact_quorum_to_lockstep() {
|
||||
for uses_legacy in [false, true] {
|
||||
let erasure = Erasure::new_with_options(3, 2, 128, uses_legacy);
|
||||
let data = b"verified reads must not accept reconstruction without a surplus source";
|
||||
let encoded = erasure.encode_data(data).expect("encode should succeed");
|
||||
let mut exact_quorum = optional_shards(&encoded);
|
||||
exact_quorum[0] = None;
|
||||
exact_quorum[erasure.total_shard_count() - 1] = None;
|
||||
|
||||
let mut default_shards = exact_quorum.clone();
|
||||
erasure
|
||||
.decode_data_with_reconstruction_verification(&mut default_shards)
|
||||
.expect("default decode must preserve exact-quorum reconstruction");
|
||||
assert_eq!(default_shards[0].as_deref(), Some(encoded[0].as_ref()));
|
||||
|
||||
let err = erasure
|
||||
.decode_data_with_reconstruction_verification_for_lockstep(&mut exact_quorum)
|
||||
.expect_err("data-shards-only lockstep must reject an exact decode quorum");
|
||||
|
||||
assert_eq!(err.kind(), io::ErrorKind::InvalidData);
|
||||
assert!(err.to_string().contains("insufficient source shards"));
|
||||
}
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn verify_data_and_parity_rejects_missing_and_mismatched_shards() {
|
||||
let erasure = Erasure::new(4, 2, 128);
|
||||
|
||||
@@ -108,6 +108,13 @@ where
|
||||
(shards, errs)
|
||||
}
|
||||
|
||||
fn heal_writer_failure(writers: &mut MultiWriter<'_>, error: io::Error) -> Error {
|
||||
writers
|
||||
.take_retryable_internode_write_failure()
|
||||
.map(|error| Error::RemoteClientUnavailable(error.to_string()))
|
||||
.unwrap_or_else(|| error.into())
|
||||
}
|
||||
|
||||
impl super::Erasure {
|
||||
pub async fn heal<R>(
|
||||
&self,
|
||||
@@ -202,10 +209,14 @@ impl super::Erasure {
|
||||
.map(|s| Bytes::from(s.unwrap_or_default()))
|
||||
.collect::<Vec<_>>();
|
||||
|
||||
writers.write(shards).await?;
|
||||
if let Err(error) = writers.write(shards).await {
|
||||
return Err(heal_writer_failure(&mut writers, error));
|
||||
}
|
||||
}
|
||||
|
||||
writers.shutdown().await?;
|
||||
if let Err(error) = writers.shutdown().await {
|
||||
return Err(heal_writer_failure(&mut writers, error));
|
||||
}
|
||||
Ok(())
|
||||
}
|
||||
}
|
||||
@@ -246,6 +257,35 @@ mod tests {
|
||||
}
|
||||
}
|
||||
|
||||
struct InternodeFailureWriter {
|
||||
fail_on_write: bool,
|
||||
status: http::StatusCode,
|
||||
}
|
||||
|
||||
impl InternodeFailureWriter {
|
||||
fn error(&self) -> io::Error {
|
||||
rustfs_rio::new_test_internode_http_io_error(rustfs_rio::InternodeHttpErrorKind::HttpStatus(self.status))
|
||||
}
|
||||
}
|
||||
|
||||
impl AsyncWrite for InternodeFailureWriter {
|
||||
fn poll_write(self: Pin<&mut Self>, _cx: &mut Context<'_>, buf: &[u8]) -> Poll<io::Result<usize>> {
|
||||
Poll::Ready(if self.fail_on_write {
|
||||
Err(self.error())
|
||||
} else {
|
||||
Ok(buf.len())
|
||||
})
|
||||
}
|
||||
|
||||
fn poll_flush(self: Pin<&mut Self>, _cx: &mut Context<'_>) -> Poll<io::Result<()>> {
|
||||
Poll::Ready(Ok(()))
|
||||
}
|
||||
|
||||
fn poll_shutdown(self: Pin<&mut Self>, _cx: &mut Context<'_>) -> Poll<io::Result<()>> {
|
||||
Poll::Ready(Err(self.error()))
|
||||
}
|
||||
}
|
||||
|
||||
struct PendingReader;
|
||||
|
||||
impl AsyncRead for PendingReader {
|
||||
@@ -331,6 +371,94 @@ mod tests {
|
||||
assert!(writers.iter().all(Option::is_some));
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
async fn heal_maps_put_file_epoch_conflict_to_retryable_remote_unavailable() {
|
||||
for status in [http::StatusCode::CONFLICT, http::StatusCode::BAD_REQUEST] {
|
||||
for (fail_on_write, data) in [
|
||||
(false, b"".as_slice()),
|
||||
(false, b"payload".as_slice()),
|
||||
(true, b"payload".as_slice()),
|
||||
] {
|
||||
let erasure = Erasure::new(2, 1, 64);
|
||||
let encoded = erasure.encode_data(data).expect("source shards should encode");
|
||||
let readers = encoded
|
||||
.iter()
|
||||
.enumerate()
|
||||
.map(|(index, shard)| {
|
||||
(index < erasure.data_shards).then(|| {
|
||||
BitrotReader::new(Cursor::new(shard.to_vec()), erasure.shard_size(), HashAlgorithm::None, false)
|
||||
})
|
||||
})
|
||||
.collect::<Vec<_>>();
|
||||
let mut writers = (0..erasure.total_shard_count())
|
||||
.map(|index| {
|
||||
(index == erasure.data_shards).then(|| {
|
||||
BitrotWriterWrapper::new(
|
||||
CustomWriter::new_tokio_writer(InternodeFailureWriter { fail_on_write, status }),
|
||||
erasure.shard_size(),
|
||||
HashAlgorithm::None,
|
||||
)
|
||||
})
|
||||
})
|
||||
.collect::<Vec<_>>();
|
||||
let error = erasure
|
||||
.heal(&mut writers, readers, data.len(), &[])
|
||||
.await
|
||||
.expect_err("failed sole target must not satisfy heal write quorum");
|
||||
assert_eq!(
|
||||
matches!(error, Error::RemoteClientUnavailable(_)),
|
||||
status == http::StatusCode::CONFLICT,
|
||||
"status={status}, fail_on_write={fail_on_write}, len={}, error={error:?}",
|
||||
data.len()
|
||||
);
|
||||
assert!(writers.iter().all(Option::is_none), "failed target must not be committed");
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
async fn heal_epoch_conflict_does_not_abort_healthy_target() {
|
||||
for fail_on_write in [false, true] {
|
||||
let erasure = Erasure::new(2, 2, 64);
|
||||
let data = b"healthy target must retain exact reconstructed bytes";
|
||||
let encoded = erasure.encode_data(data).expect("source shards should encode");
|
||||
let readers = encoded
|
||||
.iter()
|
||||
.enumerate()
|
||||
.map(|(index, shard)| {
|
||||
(index < erasure.data_shards)
|
||||
.then(|| BitrotReader::new(Cursor::new(shard.to_vec()), erasure.shard_size(), HashAlgorithm::None, false))
|
||||
})
|
||||
.collect::<Vec<_>>();
|
||||
let mut writers = vec![
|
||||
None,
|
||||
None,
|
||||
Some(BitrotWriterWrapper::new(
|
||||
CustomWriter::new_tokio_writer(InternodeFailureWriter {
|
||||
fail_on_write,
|
||||
status: http::StatusCode::CONFLICT,
|
||||
}),
|
||||
erasure.shard_size(),
|
||||
HashAlgorithm::None,
|
||||
)),
|
||||
Some(inline_writer(erasure.shard_size())),
|
||||
];
|
||||
erasure
|
||||
.heal(&mut writers, readers, data.len(), &[])
|
||||
.await
|
||||
.expect("one healthy target must still satisfy the existing heal quorum");
|
||||
assert!(writers[2].is_none(), "conflicting target must be dropped");
|
||||
assert_eq!(
|
||||
writers[3]
|
||||
.take()
|
||||
.expect("healthy target remains")
|
||||
.into_inline_data()
|
||||
.expect("inline target data"),
|
||||
encoded[3].to_vec()
|
||||
);
|
||||
}
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
async fn heal_reconstructs_missing_parity_shard() {
|
||||
let erasure = Erasure::new(2, 2, 64);
|
||||
|
||||
@@ -278,3 +278,17 @@ fn reduce_errs_buckets_identical_other_messages_together() {
|
||||
assert_eq!(count, 3);
|
||||
assert_eq!(err, Some(DiskError::other("can not get client")));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn stable_io_context_buckets_by_cause_and_preserves_diagnostic_source() {
|
||||
let first = StorageError::other_with_context("tier mutation intent changed", "mutation-a");
|
||||
let second = StorageError::other_with_context("tier mutation intent changed", "mutation-b");
|
||||
|
||||
assert_eq!(first, second, "diagnostic identity must not split quorum buckets");
|
||||
let StorageError::Io(io_error) = first else {
|
||||
panic!("stable context must remain an io error");
|
||||
};
|
||||
assert_eq!(io_error.to_string(), "tier mutation intent changed");
|
||||
let context = io_error.get_ref().expect("stable context must remain downcastable");
|
||||
assert_eq!(context.source().expect("diagnostic source must be retained").to_string(), "mutation-a");
|
||||
}
|
||||
|
||||
@@ -23,6 +23,36 @@ use s3s::S3ErrorCode;
|
||||
pub type Error = StorageError;
|
||||
pub type Result<T> = core::result::Result<T, Error>;
|
||||
|
||||
/// Keeps high-cardinality diagnostic detail in the error source while making
|
||||
/// the rendered `io::Error` stable for quorum aggregation.
|
||||
#[derive(Debug)]
|
||||
struct StableIoContextError {
|
||||
message: &'static str,
|
||||
source: Box<dyn std::error::Error + Send + Sync>,
|
||||
}
|
||||
|
||||
impl std::fmt::Display for StableIoContextError {
|
||||
fn fmt(&self, formatter: &mut std::fmt::Formatter<'_>) -> std::fmt::Result {
|
||||
formatter.write_str(self.message)
|
||||
}
|
||||
}
|
||||
|
||||
impl std::error::Error for StableIoContextError {
|
||||
fn source(&self) -> Option<&(dyn std::error::Error + 'static)> {
|
||||
Some(self.source.as_ref())
|
||||
}
|
||||
}
|
||||
|
||||
pub(crate) fn stable_io_error<E>(message: &'static str, source: E) -> std::io::Error
|
||||
where
|
||||
E: Into<Box<dyn std::error::Error + Send + Sync>>,
|
||||
{
|
||||
std::io::Error::other(StableIoContextError {
|
||||
message,
|
||||
source: source.into(),
|
||||
})
|
||||
}
|
||||
|
||||
/// Storage layer error type covering disk, volume, bucket, object, multipart,
|
||||
/// erasure-coding, and operational error conditions.
|
||||
///
|
||||
@@ -183,8 +213,18 @@ pub enum StorageError {
|
||||
DecommissionNotStarted,
|
||||
#[error("Decommission already running")]
|
||||
DecommissionAlreadyRunning,
|
||||
#[error("Decommission capacity error: {0}")]
|
||||
DecommissionCapacity(String),
|
||||
#[error("decommission_capacity_blocked: Storage reached its minimum free drive threshold.: {message}")]
|
||||
DecommissionCapacityBlocked { message: String },
|
||||
#[error("Rebalance already running")]
|
||||
RebalanceAlreadyRunning,
|
||||
#[error("{operation}: stale pool metadata update rejected for pool {pool_index}; {reason}")]
|
||||
StalePoolMetadataUpdate {
|
||||
operation: String,
|
||||
pool_index: usize,
|
||||
reason: &'static str,
|
||||
},
|
||||
#[error("Operation canceled")]
|
||||
OperationCanceled,
|
||||
#[error("No heal required")]
|
||||
@@ -254,6 +294,13 @@ impl StorageError {
|
||||
StorageError::Io(std::io::Error::other(error))
|
||||
}
|
||||
|
||||
pub(crate) fn other_with_context<E>(message: &'static str, source: E) -> Self
|
||||
where
|
||||
E: Into<Box<dyn std::error::Error + Send + Sync>>,
|
||||
{
|
||||
StorageError::Io(stable_io_error(message, source))
|
||||
}
|
||||
|
||||
pub fn is_not_found(&self) -> bool {
|
||||
matches!(
|
||||
self,
|
||||
@@ -563,7 +610,20 @@ impl Clone for StorageError {
|
||||
StorageError::EntityTooLarge(a, b) => StorageError::EntityTooLarge(*a, *b),
|
||||
StorageError::DoneForNow => StorageError::DoneForNow,
|
||||
StorageError::DecommissionAlreadyRunning => StorageError::DecommissionAlreadyRunning,
|
||||
StorageError::DecommissionCapacity(message) => StorageError::DecommissionCapacity(message.clone()),
|
||||
StorageError::DecommissionCapacityBlocked { message } => StorageError::DecommissionCapacityBlocked {
|
||||
message: message.clone(),
|
||||
},
|
||||
StorageError::RebalanceAlreadyRunning => StorageError::RebalanceAlreadyRunning,
|
||||
StorageError::StalePoolMetadataUpdate {
|
||||
operation,
|
||||
pool_index,
|
||||
reason,
|
||||
} => StorageError::StalePoolMetadataUpdate {
|
||||
operation: operation.clone(),
|
||||
pool_index: *pool_index,
|
||||
reason,
|
||||
},
|
||||
StorageError::OperationCanceled => StorageError::OperationCanceled,
|
||||
StorageError::ErasureReadQuorum => StorageError::ErasureReadQuorum,
|
||||
StorageError::ErasureWriteQuorum => StorageError::ErasureWriteQuorum,
|
||||
@@ -666,7 +726,10 @@ impl StorageError {
|
||||
StorageError::InvalidPart(_, _, _) => StorageErrorCode::InvalidPart,
|
||||
StorageError::DoneForNow => StorageErrorCode::DoneForNow,
|
||||
StorageError::DecommissionAlreadyRunning => StorageErrorCode::DecommissionAlreadyRunning,
|
||||
StorageError::DecommissionCapacity(_) => StorageErrorCode::InvalidArgument,
|
||||
StorageError::DecommissionCapacityBlocked { .. } => StorageErrorCode::StorageFull,
|
||||
StorageError::RebalanceAlreadyRunning => StorageErrorCode::RebalanceAlreadyRunning,
|
||||
StorageError::StalePoolMetadataUpdate { .. } => StorageErrorCode::InvalidArgument,
|
||||
StorageError::OperationCanceled => StorageErrorCode::OperationCanceled,
|
||||
StorageError::ErasureReadQuorum => StorageErrorCode::ErasureReadQuorum,
|
||||
StorageError::ErasureWriteQuorum => StorageErrorCode::ErasureWriteQuorum,
|
||||
@@ -948,10 +1011,6 @@ pub fn is_err_data_movement_overwrite(err: &Error) -> bool {
|
||||
matches!(err, &StorageError::DataMovementOverwriteErr(_, _, _))
|
||||
}
|
||||
|
||||
pub fn is_err_decommission_running(err: &Error) -> bool {
|
||||
matches!(err, &StorageError::DecommissionAlreadyRunning)
|
||||
}
|
||||
|
||||
#[allow(dead_code, reason = "predicate asserted by this file's tests (backlog#1823)")]
|
||||
pub fn is_err_rebalance_running(err: &Error) -> bool {
|
||||
matches!(err, &StorageError::RebalanceAlreadyRunning)
|
||||
@@ -1347,9 +1406,6 @@ mod tests {
|
||||
|
||||
#[test]
|
||||
fn test_error_running_state_helpers() {
|
||||
assert!(is_err_decommission_running(&StorageError::DecommissionAlreadyRunning));
|
||||
assert!(!is_err_decommission_running(&StorageError::RebalanceAlreadyRunning));
|
||||
|
||||
assert!(is_err_rebalance_running(&StorageError::RebalanceAlreadyRunning));
|
||||
assert!(!is_err_rebalance_running(&StorageError::DecommissionAlreadyRunning));
|
||||
assert!(is_err_operation_canceled(&StorageError::OperationCanceled));
|
||||
|
||||
@@ -37,7 +37,9 @@ use rustfs_filemeta::{FileInfo, MetaCacheEntriesSorted, ObjectPartInfo, RestoreS
|
||||
use rustfs_rio::Checksum;
|
||||
use rustfs_utils::CompressionAlgorithm;
|
||||
use rustfs_utils::http::headers::AMZ_OBJECT_TAGGING;
|
||||
use rustfs_utils::http::{AMZ_BUCKET_REPLICATION_STATUS, AMZ_RESTORE, AMZ_STORAGE_CLASS};
|
||||
use rustfs_utils::http::{
|
||||
AMZ_BUCKET_REPLICATION_STATUS, AMZ_RESTORE, AMZ_STORAGE_CLASS, SUFFIX_PLAINTEXT_CHECKSUM, get_consistent_str,
|
||||
};
|
||||
use rustfs_utils::path::decode_dir_object;
|
||||
use std::collections::HashMap;
|
||||
use std::fmt::Debug;
|
||||
|
||||
@@ -19,6 +19,9 @@ use crate::storage_api_contracts::{
|
||||
HTTPPreconditions, ObjectLockRetentionOptions, ObjectPreconditionError, ObjectPreconditionPart, ObjectPreconditionState,
|
||||
},
|
||||
};
|
||||
use std::sync::atomic::{AtomicBool, AtomicU8, Ordering};
|
||||
use tokio::sync::{Mutex, Notify, OwnedRwLockReadGuard};
|
||||
use tokio_util::sync::CancellationToken;
|
||||
|
||||
#[derive(Clone)]
|
||||
pub struct NamespaceLockFence {
|
||||
@@ -298,36 +301,23 @@ pub enum LifecycleDeleteAllPhase {
|
||||
#[doc(hidden)]
|
||||
#[derive(Default)]
|
||||
pub struct LifecycleDeleteAllJournalState {
|
||||
prepared: HashMap<String, crate::bucket::lifecycle::tier_sweeper::Jentry>,
|
||||
mutation_started: bool,
|
||||
}
|
||||
|
||||
impl Debug for LifecycleDeleteAllJournalState {
|
||||
fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result {
|
||||
f.debug_struct("LifecycleDeleteAllJournalState")
|
||||
.field("prepared_count", &self.prepared.len())
|
||||
.field("mutation_started", &self.mutation_started)
|
||||
.finish()
|
||||
}
|
||||
}
|
||||
|
||||
impl LifecycleDeleteAllJournalState {
|
||||
pub(crate) fn contains(&self, name: &str) -> bool {
|
||||
self.prepared.contains_key(name)
|
||||
}
|
||||
|
||||
pub(crate) fn insert(&mut self, name: String, entry: crate::bucket::lifecycle::tier_sweeper::Jentry) {
|
||||
self.prepared.insert(name, entry);
|
||||
}
|
||||
|
||||
pub(crate) fn prepared_entries(&self) -> Vec<crate::bucket::lifecycle::tier_sweeper::Jentry> {
|
||||
self.prepared.values().cloned().collect()
|
||||
}
|
||||
|
||||
pub(crate) fn mark_mutation_started(&mut self) {
|
||||
self.mutation_started = true;
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
pub(crate) fn mutation_started(&self) -> bool {
|
||||
self.mutation_started
|
||||
}
|
||||
@@ -347,6 +337,347 @@ impl QuotaAdmission {
|
||||
}
|
||||
}
|
||||
|
||||
const SCANNER_PUBLICATION_SCOPE_ADMITTED: u8 = 0;
|
||||
const SCANNER_PUBLICATION_SCOPE_IN_FLIGHT: u8 = 1;
|
||||
const SCANNER_PUBLICATION_SCOPE_COMMITTED: u8 = 2;
|
||||
const SCANNER_PUBLICATION_SCOPE_ABORTED_BEFORE_COMMIT: u8 = 3;
|
||||
const SCANNER_PUBLICATION_SCOPE_INDETERMINATE: u8 = 4;
|
||||
|
||||
/// The terminal result of a storage-owned scanner publication mutation.
|
||||
///
|
||||
/// This state is deliberately not serialized. It is the ownership hand-off
|
||||
/// between the scanner coordinator and the storage mutation task, so a
|
||||
/// detached rename/cleanup task can retain the movement permit until it has
|
||||
/// reported a definitive result.
|
||||
#[derive(Clone, Copy, Debug, PartialEq, Eq)]
|
||||
pub enum ScannerPublicationCommitState {
|
||||
Admitted,
|
||||
InFlight,
|
||||
Committed,
|
||||
AbortedBeforeCommit,
|
||||
Indeterminate,
|
||||
}
|
||||
|
||||
impl ScannerPublicationCommitState {
|
||||
fn as_u8(self) -> u8 {
|
||||
match self {
|
||||
Self::Admitted => SCANNER_PUBLICATION_SCOPE_ADMITTED,
|
||||
Self::InFlight => SCANNER_PUBLICATION_SCOPE_IN_FLIGHT,
|
||||
Self::Committed => SCANNER_PUBLICATION_SCOPE_COMMITTED,
|
||||
Self::AbortedBeforeCommit => SCANNER_PUBLICATION_SCOPE_ABORTED_BEFORE_COMMIT,
|
||||
Self::Indeterminate => SCANNER_PUBLICATION_SCOPE_INDETERMINATE,
|
||||
}
|
||||
}
|
||||
|
||||
fn from_u8(value: u8) -> Self {
|
||||
match value {
|
||||
SCANNER_PUBLICATION_SCOPE_IN_FLIGHT => Self::InFlight,
|
||||
SCANNER_PUBLICATION_SCOPE_COMMITTED => Self::Committed,
|
||||
SCANNER_PUBLICATION_SCOPE_ABORTED_BEFORE_COMMIT => Self::AbortedBeforeCommit,
|
||||
SCANNER_PUBLICATION_SCOPE_INDETERMINATE => Self::Indeterminate,
|
||||
_ => Self::Admitted,
|
||||
}
|
||||
}
|
||||
|
||||
/// A caller may release its remote lease only after one of these states.
|
||||
/// `Indeterminate` is intentionally excluded: the mutation may have
|
||||
/// committed after cancellation or a transport failure.
|
||||
pub fn permits_lease_release(self) -> bool {
|
||||
matches!(self, Self::Committed | Self::AbortedBeforeCommit)
|
||||
}
|
||||
}
|
||||
|
||||
/// Why a storage-owned publication scope could not start its mutation.
|
||||
#[derive(Clone, Copy, Debug, PartialEq, Eq)]
|
||||
pub enum ScannerPublicationCommitStartError {
|
||||
Cancelled,
|
||||
DeadlineExceeded,
|
||||
AlreadyStarted,
|
||||
Terminal,
|
||||
}
|
||||
|
||||
struct ScannerPublicationCommitScopeInner {
|
||||
expected_movement_epoch: u64,
|
||||
safe_deadline: tokio::time::Instant,
|
||||
remote_lease_tokens: Arc<[Uuid]>,
|
||||
cancellation: CancellationToken,
|
||||
state: AtomicU8,
|
||||
completed: Notify,
|
||||
/// Set once a storage mutation task has taken ownership of the scope.
|
||||
/// The caller-side RAII guard must not classify cancellation as
|
||||
/// indeterminate while that owner can still report a definitive result.
|
||||
owner_attached: AtomicBool,
|
||||
/// The permit is storage-owned rather than borrowed from the scanner
|
||||
/// future. A detached mutation task keeps the scope alive and therefore
|
||||
/// keeps this guard alive until it reports a terminal state.
|
||||
movement_permit: Mutex<Option<OwnedRwLockReadGuard<()>>>,
|
||||
lease_release_safe: Arc<AtomicBool>,
|
||||
}
|
||||
|
||||
/// Storage-owned ownership scope for one fenced scanner metadata mutation.
|
||||
///
|
||||
/// The scope is an in-memory capability. It is intentionally carried through
|
||||
/// [`ObjectOptions`] as a hidden field and never participates in serde, object
|
||||
/// metadata, RPC wire structures, or on-disk formats.
|
||||
#[derive(Clone)]
|
||||
pub struct ScannerPublicationCommitScope {
|
||||
inner: Arc<ScannerPublicationCommitScopeInner>,
|
||||
}
|
||||
|
||||
/// RAII fallback for storage paths that return before their commit closure
|
||||
/// takes ownership. An in-flight scope is never guessed to be aborted: it is
|
||||
/// marked indeterminate so remote lease release remains blocked.
|
||||
pub(crate) struct ScannerPublicationCommitScopeGuard {
|
||||
scope: Option<ScannerPublicationCommitScope>,
|
||||
}
|
||||
|
||||
impl ScannerPublicationCommitScopeGuard {
|
||||
pub(crate) fn new(scope: ScannerPublicationCommitScope) -> Self {
|
||||
Self { scope: Some(scope) }
|
||||
}
|
||||
|
||||
pub(crate) fn disarm(&mut self) {
|
||||
self.scope = None;
|
||||
}
|
||||
}
|
||||
|
||||
impl Drop for ScannerPublicationCommitScopeGuard {
|
||||
fn drop(&mut self) {
|
||||
let Some(scope) = self.scope.as_ref() else {
|
||||
return;
|
||||
};
|
||||
if scope.owner_attached() {
|
||||
return;
|
||||
}
|
||||
match scope.state() {
|
||||
ScannerPublicationCommitState::Admitted => {
|
||||
let _ = scope.mark_aborted_before_commit();
|
||||
}
|
||||
ScannerPublicationCommitState::InFlight => {
|
||||
let _ = scope.mark_indeterminate();
|
||||
}
|
||||
ScannerPublicationCommitState::Committed
|
||||
| ScannerPublicationCommitState::AbortedBeforeCommit
|
||||
| ScannerPublicationCommitState::Indeterminate => {}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
impl Debug for ScannerPublicationCommitScope {
|
||||
fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result {
|
||||
f.debug_struct("ScannerPublicationCommitScope")
|
||||
.field("expected_movement_epoch", &self.expected_movement_epoch())
|
||||
.field("safe_deadline", &self.safe_deadline())
|
||||
.field("remote_lease_token_count", &self.remote_lease_tokens().len())
|
||||
.field("state", &self.state())
|
||||
.finish()
|
||||
}
|
||||
}
|
||||
|
||||
impl ScannerPublicationCommitScope {
|
||||
/// Construct a scope after the storage layer has acquired its movement
|
||||
/// read permit. Callers must keep the scope attached to the actual
|
||||
/// mutation owner until [`Self::wait_for_completion`] has resolved.
|
||||
pub(crate) fn new_storage_owned(
|
||||
expected_movement_epoch: u64,
|
||||
safe_deadline: tokio::time::Instant,
|
||||
remote_lease_tokens: Vec<Uuid>,
|
||||
movement_permit: OwnedRwLockReadGuard<()>,
|
||||
) -> Self {
|
||||
Self::new_storage_owned_with_release_flag(
|
||||
expected_movement_epoch,
|
||||
safe_deadline,
|
||||
remote_lease_tokens,
|
||||
movement_permit,
|
||||
Arc::new(AtomicBool::new(true)),
|
||||
)
|
||||
}
|
||||
|
||||
pub(crate) fn new_storage_owned_with_release_flag(
|
||||
expected_movement_epoch: u64,
|
||||
safe_deadline: tokio::time::Instant,
|
||||
remote_lease_tokens: Vec<Uuid>,
|
||||
movement_permit: OwnedRwLockReadGuard<()>,
|
||||
lease_release_safe: Arc<AtomicBool>,
|
||||
) -> Self {
|
||||
lease_release_safe.store(false, Ordering::Release);
|
||||
Self {
|
||||
inner: Arc::new(ScannerPublicationCommitScopeInner {
|
||||
expected_movement_epoch,
|
||||
safe_deadline,
|
||||
remote_lease_tokens: remote_lease_tokens.into(),
|
||||
cancellation: CancellationToken::new(),
|
||||
state: AtomicU8::new(SCANNER_PUBLICATION_SCOPE_ADMITTED),
|
||||
completed: Notify::new(),
|
||||
owner_attached: AtomicBool::new(false),
|
||||
movement_permit: Mutex::new(Some(movement_permit)),
|
||||
lease_release_safe,
|
||||
}),
|
||||
}
|
||||
}
|
||||
|
||||
pub fn expected_movement_epoch(&self) -> u64 {
|
||||
self.inner.expected_movement_epoch
|
||||
}
|
||||
|
||||
pub fn safe_deadline(&self) -> tokio::time::Instant {
|
||||
self.inner.safe_deadline
|
||||
}
|
||||
|
||||
pub fn is_expired(&self) -> bool {
|
||||
tokio::time::Instant::now() >= self.safe_deadline()
|
||||
}
|
||||
|
||||
pub fn remote_lease_tokens(&self) -> &[Uuid] {
|
||||
&self.inner.remote_lease_tokens
|
||||
}
|
||||
|
||||
pub fn cancellation_token(&self) -> CancellationToken {
|
||||
self.inner.cancellation.clone()
|
||||
}
|
||||
|
||||
pub fn is_cancelled(&self) -> bool {
|
||||
self.inner.cancellation.is_cancelled()
|
||||
}
|
||||
|
||||
/// Whether a mutation that has already begun may still enter its durable
|
||||
/// commit boundary. The storage owner must check this immediately before
|
||||
/// starting each irreversible fan-out/rename operation.
|
||||
pub fn can_commit(&self) -> bool {
|
||||
self.state() == ScannerPublicationCommitState::InFlight && !self.is_cancelled() && !self.is_expired()
|
||||
}
|
||||
|
||||
/// Transfer terminal-state responsibility from the caller to a detached
|
||||
/// storage mutation owner. Once set, dropping a scanner waiter leaves the
|
||||
/// scope in-flight until that owner reports committed or indeterminate.
|
||||
pub fn attach_mutation_owner(&self) {
|
||||
self.inner.owner_attached.store(true, Ordering::Release);
|
||||
}
|
||||
|
||||
fn owner_attached(&self) -> bool {
|
||||
self.inner.owner_attached.load(Ordering::Acquire)
|
||||
}
|
||||
|
||||
pub fn state(&self) -> ScannerPublicationCommitState {
|
||||
ScannerPublicationCommitState::from_u8(self.inner.state.load(Ordering::Acquire))
|
||||
}
|
||||
|
||||
/// Request cancellation without claiming that a mutation has stopped.
|
||||
/// The owner must still report `AbortedBeforeCommit` or `Indeterminate`.
|
||||
pub fn cancel(&self) {
|
||||
self.inner.cancellation.cancel();
|
||||
}
|
||||
|
||||
pub fn try_begin(&self) -> std::result::Result<(), ScannerPublicationCommitStartError> {
|
||||
if self.inner.cancellation.is_cancelled() {
|
||||
return Err(ScannerPublicationCommitStartError::Cancelled);
|
||||
}
|
||||
if self.is_expired() {
|
||||
return Err(ScannerPublicationCommitStartError::DeadlineExceeded);
|
||||
}
|
||||
self.inner
|
||||
.state
|
||||
.compare_exchange(
|
||||
SCANNER_PUBLICATION_SCOPE_ADMITTED,
|
||||
SCANNER_PUBLICATION_SCOPE_IN_FLIGHT,
|
||||
Ordering::AcqRel,
|
||||
Ordering::Acquire,
|
||||
)
|
||||
.map(|_| ())
|
||||
.map_err(|state| {
|
||||
if ScannerPublicationCommitState::from_u8(state).permits_lease_release() {
|
||||
ScannerPublicationCommitStartError::Terminal
|
||||
} else {
|
||||
ScannerPublicationCommitStartError::AlreadyStarted
|
||||
}
|
||||
})
|
||||
}
|
||||
|
||||
pub fn mark_committed(&self) -> bool {
|
||||
self.mark_terminal(ScannerPublicationCommitState::Committed)
|
||||
}
|
||||
|
||||
pub fn mark_aborted_before_commit(&self) -> bool {
|
||||
if self
|
||||
.inner
|
||||
.state
|
||||
.compare_exchange(
|
||||
SCANNER_PUBLICATION_SCOPE_ADMITTED,
|
||||
SCANNER_PUBLICATION_SCOPE_ABORTED_BEFORE_COMMIT,
|
||||
Ordering::AcqRel,
|
||||
Ordering::Acquire,
|
||||
)
|
||||
.is_ok()
|
||||
{
|
||||
self.inner.lease_release_safe.store(true, Ordering::Release);
|
||||
self.inner.completed.notify_waiters();
|
||||
return true;
|
||||
}
|
||||
false
|
||||
}
|
||||
|
||||
pub fn mark_indeterminate(&self) -> bool {
|
||||
self.mark_terminal(ScannerPublicationCommitState::Indeterminate)
|
||||
}
|
||||
|
||||
fn mark_terminal(&self, terminal: ScannerPublicationCommitState) -> bool {
|
||||
self.inner
|
||||
.state
|
||||
.compare_exchange(SCANNER_PUBLICATION_SCOPE_IN_FLIGHT, terminal.as_u8(), Ordering::AcqRel, Ordering::Acquire)
|
||||
.is_ok()
|
||||
.then(|| {
|
||||
if terminal.permits_lease_release() {
|
||||
self.inner.lease_release_safe.store(true, Ordering::Release);
|
||||
}
|
||||
self.inner.completed.notify_waiters()
|
||||
})
|
||||
.is_some()
|
||||
}
|
||||
|
||||
/// Wait until the mutation owner has reported a definitive terminal
|
||||
/// state. The permit remains owned by this scope until all scope clones are
|
||||
/// dropped or [`Self::release_movement_permit`] is called safely.
|
||||
pub async fn wait_for_completion(&self) -> ScannerPublicationCommitState {
|
||||
loop {
|
||||
let notified = self.inner.completed.notified();
|
||||
tokio::pin!(notified);
|
||||
notified.as_mut().enable();
|
||||
let state = self.state();
|
||||
if state != ScannerPublicationCommitState::Admitted && state != ScannerPublicationCommitState::InFlight {
|
||||
return state;
|
||||
}
|
||||
notified.await;
|
||||
}
|
||||
}
|
||||
|
||||
/// Release the storage-owned movement permit only after a known-safe
|
||||
/// terminal result. Returns `false` for in-flight or indeterminate work.
|
||||
pub async fn release_movement_permit(&self) -> bool {
|
||||
if !self.state().permits_lease_release() {
|
||||
return false;
|
||||
}
|
||||
self.inner.movement_permit.lock().await.take().is_some()
|
||||
}
|
||||
}
|
||||
|
||||
impl Drop for ScannerPublicationCommitScopeInner {
|
||||
fn drop(&mut self) {
|
||||
if !ScannerPublicationCommitState::from_u8(self.state.load(Ordering::Acquire)).permits_lease_release() {
|
||||
self.lease_release_safe.store(false, Ordering::Release);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
#[derive(Clone, Default)]
|
||||
#[doc(hidden)]
|
||||
pub struct DecommissionCapacityOptions {
|
||||
pub(crate) expected_data_bytes: Option<usize>,
|
||||
pub(crate) operation_id: Option<Uuid>,
|
||||
pub(crate) generation: Option<u64>,
|
||||
pub(crate) owner_nonce: Option<Uuid>,
|
||||
pub(crate) mutation_id: Option<Uuid>,
|
||||
}
|
||||
|
||||
#[derive(Default, Clone)]
|
||||
pub struct ObjectOptions {
|
||||
// Use the maximum parity (N/2), used when saving server configuration files
|
||||
@@ -362,6 +693,12 @@ pub struct ObjectOptions {
|
||||
pub lifecycle_delete_all: Option<LifecycleDeleteAllRequest>,
|
||||
#[doc(hidden)]
|
||||
pub lifecycle_delete_all_journal: Option<Arc<parking_lot::Mutex<LifecycleDeleteAllJournalState>>>,
|
||||
/// Whole-operation authorization created only by consuming a validated
|
||||
/// v6 dispatch-manifest permit. Clones share the authorization, not the
|
||||
/// one-shot permit itself.
|
||||
#[doc(hidden)]
|
||||
pub tier_delete_dispatch_authorization:
|
||||
Option<crate::bucket::lifecycle::tier_delete_journal::TierDeleteDispatchAuthorization>,
|
||||
/// RustFS-only compare-and-set condition checked under the object write lock.
|
||||
pub expected_current_version_id: Option<String>,
|
||||
/// Persisted bucket incarnation observed before authorization.
|
||||
@@ -384,8 +721,19 @@ pub struct ObjectOptions {
|
||||
#[doc(hidden)]
|
||||
pub put_object_cancellation: Option<tokio_util::sync::CancellationToken>,
|
||||
|
||||
/// Storage-owned scanner publication capability. This field is an
|
||||
/// in-memory hand-off only; it is never copied into object metadata.
|
||||
#[doc(hidden)]
|
||||
pub scanner_publication_commit_scope: Option<ScannerPublicationCommitScope>,
|
||||
|
||||
pub data_movement: bool,
|
||||
pub raw_data_movement_read: bool,
|
||||
/// Durable reservation identity carried only by decommission writes. Other
|
||||
/// data-movement users, including rebalance, leave it unset. Keep this
|
||||
/// context boxed because `ObjectOptions` is passed by value through deep
|
||||
/// storage futures.
|
||||
#[doc(hidden)]
|
||||
pub decommission_capacity: Option<Box<DecommissionCapacityOptions>>,
|
||||
/// Materialize the data-movement per-part checksum sidecar for APIs that
|
||||
/// return part checksums. Ordinary object reads leave it encoded.
|
||||
pub include_part_checksums: bool,
|
||||
@@ -449,6 +797,36 @@ pub struct ObjectOptions {
|
||||
/// Storage-owned journal writer used by the atomic delete path. This is
|
||||
/// populated only by the `ECStore` wrapper that holds the namespace locks.
|
||||
pub tier_delete_journal_api: Option<Arc<crate::store::ECStore>>,
|
||||
/// Internal staged-mutation admission supplied by `ECStore`; each local
|
||||
/// publish is fenced namespace-first and then by decommission capacity.
|
||||
#[doc(hidden)]
|
||||
pub decommission_capacity_admission: Option<Arc<crate::store::ECStore>>,
|
||||
}
|
||||
|
||||
impl ObjectOptions {
|
||||
pub(crate) fn with_capacity_expected_data_bytes(expected_data_bytes: Option<usize>) -> Self {
|
||||
Self {
|
||||
decommission_capacity: expected_data_bytes.map(|expected_data_bytes| {
|
||||
Box::new(DecommissionCapacityOptions {
|
||||
expected_data_bytes: Some(expected_data_bytes),
|
||||
..Default::default()
|
||||
})
|
||||
}),
|
||||
..Default::default()
|
||||
}
|
||||
}
|
||||
|
||||
pub(crate) fn capacity_expected_data_bytes(&self) -> Option<usize> {
|
||||
self.decommission_capacity
|
||||
.as_deref()
|
||||
.and_then(|capacity| capacity.expected_data_bytes)
|
||||
}
|
||||
|
||||
pub(crate) fn has_decommission_capacity_reservation(&self) -> bool {
|
||||
self.decommission_capacity
|
||||
.as_deref()
|
||||
.is_some_and(|capacity| capacity.operation_id.is_some())
|
||||
}
|
||||
}
|
||||
|
||||
impl std::fmt::Debug for ObjectOptions {
|
||||
@@ -462,6 +840,7 @@ impl std::fmt::Debug for ObjectOptions {
|
||||
.field("version_id", &self.version_id.is_some())
|
||||
.field("lifecycle_delete_all", &self.lifecycle_delete_all.is_some())
|
||||
.field("lifecycle_delete_all_journal", &self.lifecycle_delete_all_journal.is_some())
|
||||
.field("tier_delete_dispatch_authorization", &self.tier_delete_dispatch_authorization.is_some())
|
||||
.field("expected_current_version_id", &self.expected_current_version_id.is_some())
|
||||
.field("expected_bucket_incarnation_id", &self.expected_bucket_incarnation_id)
|
||||
.field("no_lock", &self.no_lock)
|
||||
@@ -473,6 +852,7 @@ impl std::fmt::Debug for ObjectOptions {
|
||||
.field("skip_rebalancing", &self.skip_rebalancing)
|
||||
.field("skip_free_version", &self.skip_free_version)
|
||||
.field("put_object_cancellation", &self.put_object_cancellation.is_some())
|
||||
.field("scanner_publication_commit_scope", &self.scanner_publication_commit_scope)
|
||||
.field("data_movement", &self.data_movement)
|
||||
.field("raw_data_movement_read", &self.raw_data_movement_read)
|
||||
.field("include_part_checksums", &self.include_part_checksums)
|
||||
@@ -586,7 +966,7 @@ impl ObjectOptions {
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
pub(crate) fn add_namespace_lock_fence_for_test(&mut self, fence: &NamespaceLockFence) {
|
||||
pub(crate) fn add_namespace_lock_fence(&mut self, fence: &NamespaceLockFence) {
|
||||
self.namespace_lock_fence
|
||||
.get_or_insert_with(NamespaceLockFence::new)
|
||||
.extend(fence);
|
||||
@@ -1391,9 +1771,10 @@ impl ObjectInfo {
|
||||
}
|
||||
|
||||
if let Some(data) = &self.checksum {
|
||||
if self.is_encrypted() {
|
||||
if self.is_encrypted() && get_consistent_str(&self.user_defined, SUFFIX_PLAINTEXT_CHECKSUM) != Some("true") {
|
||||
// Object-level encrypted checksum bytes require SSE decrypt material,
|
||||
// so do not expose them as plaintext checksum headers here. The
|
||||
// unless RustFS marked the stored bytes as plaintext. Do not expose
|
||||
// unmarked bytes as checksum headers here. The
|
||||
// `false` multipart flag feeds the response-path COMPOSITE
|
||||
// fallback; callers that need accurate multipart routing must
|
||||
// consult `is_multipart()` instead of this value.
|
||||
@@ -2099,6 +2480,31 @@ mod tests {
|
||||
assert!(checksums.is_empty());
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn decrypt_checksums_reads_marked_rustfs_encrypted_object_checksum() {
|
||||
let checksum = rustfs_rio::Checksum::new_from_data(rustfs_rio::ChecksumType::CRC32, b"encrypted-object")
|
||||
.expect("test checksum should be valid");
|
||||
let checksum_key = checksum.checksum_type.to_string();
|
||||
let expected_checksum = checksum.encoded.clone();
|
||||
let mut user_defined =
|
||||
HashMap::from([(rustfs_utils::http::headers::AMZ_SERVER_SIDE_ENCRYPTION.to_string(), "AES256".to_string())]);
|
||||
rustfs_utils::http::insert_str(&mut user_defined, SUFFIX_PLAINTEXT_CHECKSUM, "true".to_string());
|
||||
assert_eq!(user_defined.get("x-rustfs-internal-plaintext-checksum").map(String::as_str), Some("true"));
|
||||
assert_eq!(user_defined.get("x-minio-internal-plaintext-checksum").map(String::as_str), Some("true"));
|
||||
let info = ObjectInfo {
|
||||
checksum: Some(checksum.to_bytes(&[])),
|
||||
user_defined: Arc::new(user_defined),
|
||||
..Default::default()
|
||||
};
|
||||
|
||||
let (checksums, is_multipart) = info
|
||||
.decrypt_checksums(0, &HeaderMap::new())
|
||||
.expect("marked RustFS checksum should decode");
|
||||
|
||||
assert!(!is_multipart);
|
||||
assert_eq!(checksums.get(&checksum_key), Some(&expected_checksum));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn decrypt_checksums_keeps_encrypted_multipart_flag_false_for_response_paths() {
|
||||
let checksum = rustfs_rio::Checksum::new_from_data(rustfs_rio::ChecksumType::CRC32, b"encrypted-object")
|
||||
|
||||
@@ -368,6 +368,19 @@ impl InstanceContext {
|
||||
Arc::clone(&self.data_movement_generation_notify)
|
||||
}
|
||||
|
||||
pub(crate) fn observe_durable_data_movement_generation(&self, generation: u64) {
|
||||
if generation == 0 || self.data_movement_generation_exhausted.load(Ordering::Acquire) {
|
||||
return;
|
||||
}
|
||||
let previous = self.data_movement_generation.fetch_max(generation, Ordering::AcqRel);
|
||||
if generation == u64::MAX {
|
||||
self.data_movement_generation_exhausted.store(true, Ordering::Release);
|
||||
}
|
||||
if generation > previous {
|
||||
self.data_movement_generation_notify.notify_waiters();
|
||||
}
|
||||
}
|
||||
|
||||
pub(crate) fn scanner_publication_state_allowed(&self) -> bool {
|
||||
!self.data_movement_operation_epoch_exhausted()
|
||||
&& !self.data_movement_generation_exhausted()
|
||||
@@ -386,6 +399,20 @@ impl InstanceContext {
|
||||
}
|
||||
|
||||
pub(crate) fn advance_data_movement_operation_epoch(&self) -> u64 {
|
||||
let (previous, result) = self.advance_data_movement_operation_epoch_only();
|
||||
if result != previous {
|
||||
let _ = self.advance_data_movement_generation();
|
||||
}
|
||||
result
|
||||
}
|
||||
|
||||
pub(crate) fn advance_data_movement_operation_epoch_to_durable_generation(&self, generation: u64) -> u64 {
|
||||
let (_, result) = self.advance_data_movement_operation_epoch_only();
|
||||
self.observe_durable_data_movement_generation(generation);
|
||||
result
|
||||
}
|
||||
|
||||
fn advance_data_movement_operation_epoch_only(&self) -> (u64, u64) {
|
||||
self.scanner_publication_state
|
||||
.store(SCANNER_PUBLICATION_STATE_UNKNOWN, Ordering::Release);
|
||||
let previous = self.data_movement_operation_epoch.load(Ordering::Acquire);
|
||||
@@ -396,10 +423,7 @@ impl InstanceContext {
|
||||
if result == u64::MAX {
|
||||
self.data_movement_operation_epoch_exhausted.store(true, Ordering::Release);
|
||||
}
|
||||
if result != previous {
|
||||
let _ = self.advance_data_movement_generation();
|
||||
}
|
||||
result
|
||||
(previous, result)
|
||||
}
|
||||
|
||||
/// Advance the movement generation after a durable movement transition.
|
||||
|
||||
@@ -29,10 +29,14 @@ use rustfs_madmin::metrics::RealtimeMetrics;
|
||||
use rustfs_madmin::net::NetInfo;
|
||||
use rustfs_madmin::{ItemState, ServerProperties, StorageInfo};
|
||||
use rustfs_utils::XHost;
|
||||
use sha2::{Digest, Sha256};
|
||||
use std::collections::{BTreeMap, HashMap, hash_map::DefaultHasher};
|
||||
use std::future::Future;
|
||||
use std::hash::{Hash, Hasher};
|
||||
use std::sync::{Arc, Mutex, OnceLock};
|
||||
use std::sync::{
|
||||
Arc, Mutex, OnceLock,
|
||||
atomic::{AtomicBool, AtomicUsize, Ordering},
|
||||
};
|
||||
use std::time::{Duration, Instant, SystemTime};
|
||||
use tokio::time::{sleep, timeout};
|
||||
use tokio_util::sync::CancellationToken;
|
||||
@@ -52,6 +56,20 @@ const REMOTE_VERSION_STATE_PROBE_INTERVAL: Duration = Duration::from_secs(10);
|
||||
const REMOTE_VERSION_STATE_PROBE_TIMEOUT: Duration = Duration::from_secs(5);
|
||||
const REMOTE_VERSION_STATE_PROOF_TTL: Duration = Duration::from_secs(30);
|
||||
const CROSS_POOL_FENCE_SUPPORTED_VERSION: u32 = 2;
|
||||
const TIER_DELETE_JOURNAL_POLICY_SUPPORTED_VERSION: u32 = 3;
|
||||
type CrossPoolFencePolicyResult = Result<BTreeMap<String, Uuid>>;
|
||||
|
||||
fn cross_pool_fence_policy_results(
|
||||
peer_epochs: BTreeMap<String, Uuid>,
|
||||
minimum_version: u32,
|
||||
) -> (CrossPoolFencePolicyResult, CrossPoolFencePolicyResult) {
|
||||
let journal_result = if minimum_version >= TIER_DELETE_JOURNAL_POLICY_SUPPORTED_VERSION {
|
||||
Ok(peer_epochs.clone())
|
||||
} else {
|
||||
Err(Error::other("tier delete journal v6 policy capability version is unsupported"))
|
||||
};
|
||||
(Ok(peer_epochs), journal_result)
|
||||
}
|
||||
|
||||
#[derive(Clone, Debug)]
|
||||
pub struct ScannerPublicationLeaseGrant {
|
||||
@@ -107,15 +125,91 @@ struct FleetCapabilityProof {
|
||||
topology_fingerprint: String,
|
||||
peer_epochs: Arc<BTreeMap<String, Uuid>>,
|
||||
expires_at: Instant,
|
||||
generation: Arc<FleetCapabilityProofGeneration>,
|
||||
}
|
||||
|
||||
impl FleetCapabilityProof {
|
||||
fn new(topology_fingerprint: String, peer_epochs: Arc<BTreeMap<String, Uuid>>, expires_at: Instant) -> Self {
|
||||
Self {
|
||||
topology_fingerprint,
|
||||
peer_epochs,
|
||||
expires_at,
|
||||
generation: FleetCapabilityProofGeneration::fresh(),
|
||||
}
|
||||
}
|
||||
|
||||
fn token(&self) -> FleetCapabilityProofToken {
|
||||
FleetCapabilityProofToken {
|
||||
topology_fingerprint: self.topology_fingerprint.clone(),
|
||||
peer_epochs: self.peer_epochs.clone(),
|
||||
}
|
||||
}
|
||||
|
||||
#[cfg(any(test, feature = "test-util"))]
|
||||
fn with_fresh_generation(&self) -> Self {
|
||||
Self::new(self.topology_fingerprint.clone(), Arc::clone(&self.peer_epochs), self.expires_at)
|
||||
}
|
||||
}
|
||||
|
||||
/// Admission generation for effects that must not straddle a fleet-proof
|
||||
/// replacement. Revocation is deliberately non-blocking: it closes admission
|
||||
/// immediately, while the proof slot withholds the successor generation until
|
||||
/// every admitted operation has drained.
|
||||
#[derive(Default)]
|
||||
struct FleetCapabilityProofGeneration {
|
||||
accepting: AtomicBool,
|
||||
active: AtomicUsize,
|
||||
}
|
||||
|
||||
impl FleetCapabilityProofGeneration {
|
||||
fn fresh() -> Arc<Self> {
|
||||
Arc::new(Self {
|
||||
accepting: AtomicBool::new(true),
|
||||
active: AtomicUsize::new(0),
|
||||
})
|
||||
}
|
||||
|
||||
fn try_acquire(self: &Arc<Self>) -> Option<FleetCapabilityProofPermit> {
|
||||
if !self.accepting.load(Ordering::Acquire) {
|
||||
return None;
|
||||
}
|
||||
self.active.fetch_add(1, Ordering::AcqRel);
|
||||
if self.accepting.load(Ordering::Acquire) {
|
||||
Some(FleetCapabilityProofPermit {
|
||||
generation: Arc::clone(self),
|
||||
})
|
||||
} else {
|
||||
self.release();
|
||||
None
|
||||
}
|
||||
}
|
||||
|
||||
fn revoke(&self) {
|
||||
self.accepting.store(false, Ordering::Release);
|
||||
}
|
||||
|
||||
fn is_accepting(&self) -> bool {
|
||||
self.accepting.load(Ordering::Acquire)
|
||||
}
|
||||
|
||||
fn is_drained(&self) -> bool {
|
||||
self.active.load(Ordering::Acquire) == 0
|
||||
}
|
||||
|
||||
fn release(&self) {
|
||||
let previous = self.active.fetch_sub(1, Ordering::AcqRel);
|
||||
debug_assert!(previous > 0, "fleet capability permit count underflow");
|
||||
}
|
||||
}
|
||||
|
||||
struct FleetCapabilityProofPermit {
|
||||
generation: Arc<FleetCapabilityProofGeneration>,
|
||||
}
|
||||
|
||||
impl Drop for FleetCapabilityProofPermit {
|
||||
fn drop(&mut self) {
|
||||
self.generation.release();
|
||||
}
|
||||
}
|
||||
|
||||
#[derive(Clone, PartialEq, Eq)]
|
||||
@@ -127,6 +221,7 @@ struct FleetCapabilityProofToken {
|
||||
#[derive(Default)]
|
||||
struct FleetCapabilityProofState {
|
||||
proof: Option<FleetCapabilityProof>,
|
||||
draining_generation: Option<Arc<FleetCapabilityProofGeneration>>,
|
||||
topology_conflict: bool,
|
||||
}
|
||||
|
||||
@@ -136,8 +231,17 @@ pub(crate) struct RemoteVersionStateFleetProofToken(FleetCapabilityProofToken);
|
||||
#[derive(Clone, PartialEq, Eq)]
|
||||
pub struct CrossPoolFenceFleetProofToken(FleetCapabilityProofToken);
|
||||
|
||||
/// A point-in-time proof that every current storage member implements the v6
|
||||
/// dispatch-manifest policy. It intentionally has no `Clone` implementation:
|
||||
/// one acquisition authorizes one manifest construction attempt.
|
||||
pub(crate) struct TierDeleteJournalFleetProofToken {
|
||||
token: FleetCapabilityProofToken,
|
||||
_permit: FleetCapabilityProofPermit,
|
||||
}
|
||||
|
||||
static REMOTE_VERSION_STATE_FLEET_PROOF: OnceLock<std::sync::RwLock<FleetCapabilityProofState>> = OnceLock::new();
|
||||
static CROSS_POOL_FENCE_FLEET_PROOF: OnceLock<std::sync::RwLock<FleetCapabilityProofState>> = OnceLock::new();
|
||||
static TIER_DELETE_JOURNAL_FLEET_PROOF: OnceLock<std::sync::RwLock<FleetCapabilityProofState>> = OnceLock::new();
|
||||
static REMOTE_VERSION_STATE_PROBE_TOPOLOGY: OnceLock<String> = OnceLock::new();
|
||||
|
||||
fn cross_pool_fence_fleet_proof_slot() -> &'static std::sync::RwLock<FleetCapabilityProofState> {
|
||||
@@ -148,8 +252,35 @@ fn remote_version_state_fleet_proof_slot() -> &'static std::sync::RwLock<FleetCa
|
||||
REMOTE_VERSION_STATE_FLEET_PROOF.get_or_init(|| std::sync::RwLock::new(FleetCapabilityProofState::default()))
|
||||
}
|
||||
|
||||
fn replace_fleet_capability_proof(slot: &std::sync::RwLock<FleetCapabilityProofState>, proof: Option<FleetCapabilityProof>) {
|
||||
slot.write().unwrap_or_else(std::sync::PoisonError::into_inner).proof = proof;
|
||||
fn tier_delete_journal_fleet_proof_slot() -> &'static std::sync::RwLock<FleetCapabilityProofState> {
|
||||
TIER_DELETE_JOURNAL_FLEET_PROOF.get_or_init(|| std::sync::RwLock::new(FleetCapabilityProofState::default()))
|
||||
}
|
||||
|
||||
fn revoke_fleet_capability_proof_state(state: &mut FleetCapabilityProofState) {
|
||||
if let Some(proof) = state.proof.take() {
|
||||
proof.generation.revoke();
|
||||
if !proof.generation.is_drained() {
|
||||
state.draining_generation = Some(proof.generation);
|
||||
}
|
||||
}
|
||||
if state
|
||||
.draining_generation
|
||||
.as_ref()
|
||||
.is_some_and(|generation| generation.is_drained())
|
||||
{
|
||||
state.draining_generation = None;
|
||||
}
|
||||
}
|
||||
|
||||
fn revoke_fleet_capability_proof(slot: &std::sync::RwLock<FleetCapabilityProofState>) {
|
||||
let mut state = slot.write().unwrap_or_else(std::sync::PoisonError::into_inner);
|
||||
revoke_fleet_capability_proof_state(&mut state);
|
||||
}
|
||||
|
||||
fn mark_fleet_capability_topology_conflict(slot: &std::sync::RwLock<FleetCapabilityProofState>) {
|
||||
let mut state = slot.write().unwrap_or_else(std::sync::PoisonError::into_inner);
|
||||
state.topology_conflict = true;
|
||||
revoke_fleet_capability_proof_state(&mut state);
|
||||
}
|
||||
|
||||
fn publish_fleet_capability_probe_result(
|
||||
@@ -161,21 +292,42 @@ fn publish_fleet_capability_probe_result(
|
||||
match result {
|
||||
Ok(peer_epochs) => {
|
||||
let mut state = slot.write().unwrap_or_else(std::sync::PoisonError::into_inner);
|
||||
let peer_epochs = state
|
||||
if let Some(current) = state
|
||||
.proof
|
||||
.as_ref()
|
||||
.as_mut()
|
||||
.filter(|proof| proof.topology_fingerprint == topology_fingerprint && proof.peer_epochs.as_ref() == &peer_epochs)
|
||||
.map(|proof| Arc::clone(&proof.peer_epochs))
|
||||
.unwrap_or_else(|| Arc::new(peer_epochs));
|
||||
state.proof = Some(FleetCapabilityProof {
|
||||
topology_fingerprint: topology_fingerprint.to_string(),
|
||||
peer_epochs,
|
||||
expires_at: observed_at + REMOTE_VERSION_STATE_PROOF_TTL,
|
||||
});
|
||||
{
|
||||
current.expires_at = observed_at + REMOTE_VERSION_STATE_PROOF_TTL;
|
||||
return None;
|
||||
}
|
||||
|
||||
if let Some(previous) = state.proof.take() {
|
||||
previous.generation.revoke();
|
||||
if !previous.generation.is_drained() {
|
||||
state.draining_generation = Some(previous.generation);
|
||||
}
|
||||
}
|
||||
if state
|
||||
.draining_generation
|
||||
.as_ref()
|
||||
.is_some_and(|generation| generation.is_drained())
|
||||
{
|
||||
state.draining_generation = None;
|
||||
}
|
||||
if state.draining_generation.is_some() {
|
||||
return Some(Error::other(
|
||||
"fleet capability proof successor waits for the previous generation to drain",
|
||||
));
|
||||
}
|
||||
state.proof = Some(FleetCapabilityProof::new(
|
||||
topology_fingerprint.to_string(),
|
||||
Arc::new(peer_epochs),
|
||||
observed_at + REMOTE_VERSION_STATE_PROOF_TTL,
|
||||
));
|
||||
None
|
||||
}
|
||||
Err(err) => {
|
||||
replace_fleet_capability_proof(slot, None);
|
||||
revoke_fleet_capability_proof(slot);
|
||||
Some(err)
|
||||
}
|
||||
}
|
||||
@@ -216,7 +368,72 @@ pub fn cross_pool_fence_fleet_proof_matches(proof: &CrossPoolFenceFleetProofToke
|
||||
fleet_capability_proof_matches(cross_pool_fence_fleet_proof_slot(), &proof.0)
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
pub(crate) fn acquire_tier_delete_journal_fleet_proof() -> Option<TierDeleteJournalFleetProofToken> {
|
||||
let expected_topology = REMOTE_VERSION_STATE_PROBE_TOPOLOGY.get()?;
|
||||
let state = tier_delete_journal_fleet_proof_slot()
|
||||
.read()
|
||||
.unwrap_or_else(std::sync::PoisonError::into_inner);
|
||||
acquire_tier_delete_journal_fleet_proof_from(&state, expected_topology, Instant::now())
|
||||
}
|
||||
|
||||
fn acquire_tier_delete_journal_fleet_proof_from(
|
||||
state: &FleetCapabilityProofState,
|
||||
expected_topology: &str,
|
||||
now: Instant,
|
||||
) -> Option<TierDeleteJournalFleetProofToken> {
|
||||
let token = acquire_fleet_capability_proof_from(state, expected_topology, now)?;
|
||||
let permit = state.proof.as_ref()?.generation.try_acquire()?;
|
||||
Some(TierDeleteJournalFleetProofToken { token, _permit: permit })
|
||||
}
|
||||
|
||||
pub(crate) fn tier_delete_journal_fleet_proof_matches(proof: &TierDeleteJournalFleetProofToken) -> bool {
|
||||
let Some(expected_topology) = REMOTE_VERSION_STATE_PROBE_TOPOLOGY.get() else {
|
||||
return false;
|
||||
};
|
||||
let state = tier_delete_journal_fleet_proof_slot()
|
||||
.read()
|
||||
.unwrap_or_else(std::sync::PoisonError::into_inner);
|
||||
tier_delete_journal_fleet_proof_matches_at(&state, proof, expected_topology, Instant::now())
|
||||
}
|
||||
|
||||
fn tier_delete_journal_fleet_proof_matches_at(
|
||||
state: &FleetCapabilityProofState,
|
||||
proof: &TierDeleteJournalFleetProofToken,
|
||||
expected_topology: &str,
|
||||
now: Instant,
|
||||
) -> bool {
|
||||
proof._permit.generation.is_accepting()
|
||||
&& fleet_capability_proof_matches_at(state, &proof.token, expected_topology, now)
|
||||
&& state
|
||||
.proof
|
||||
.as_ref()
|
||||
.is_some_and(|current| Arc::ptr_eq(¤t.generation, &proof._permit.generation))
|
||||
}
|
||||
|
||||
pub(crate) fn tier_delete_journal_topology_generation(proof: &TierDeleteJournalFleetProofToken) -> String {
|
||||
stable_tier_delete_journal_topology_generation(&proof.token.topology_fingerprint)
|
||||
}
|
||||
|
||||
#[cfg(all(test, feature = "test-util"))]
|
||||
pub(crate) fn tier_delete_journal_fleet_proof_has_inflight_for_test() -> bool {
|
||||
let state = tier_delete_journal_fleet_proof_slot()
|
||||
.read()
|
||||
.unwrap_or_else(std::sync::PoisonError::into_inner);
|
||||
state.proof.as_ref().is_some_and(|proof| !proof.generation.is_drained())
|
||||
|| state
|
||||
.draining_generation
|
||||
.as_ref()
|
||||
.is_some_and(|generation| !generation.is_drained())
|
||||
}
|
||||
|
||||
fn stable_tier_delete_journal_topology_generation(topology_fingerprint: &str) -> String {
|
||||
let mut hasher = Sha256::new();
|
||||
hasher.update(b"rustfs-tier-delete-journal-topology-v1\0");
|
||||
hasher.update(topology_fingerprint.as_bytes());
|
||||
rustfs_utils::crypto::hex(hasher.finalize().as_slice())
|
||||
}
|
||||
|
||||
#[cfg(any(test, feature = "test-util"))]
|
||||
pub(crate) fn install_cross_pool_fence_fleet_proof_for_test() {
|
||||
let topology = REMOTE_VERSION_STATE_PROBE_TOPOLOGY
|
||||
.get()
|
||||
@@ -226,18 +443,39 @@ pub(crate) fn install_cross_pool_fence_fleet_proof_for_test() {
|
||||
let mut state = cross_pool_fence_fleet_proof_slot()
|
||||
.write()
|
||||
.unwrap_or_else(std::sync::PoisonError::into_inner);
|
||||
let now = Instant::now();
|
||||
let proof = if !state.topology_conflict && fleet_capability_proof_valid_at(state.proof.as_ref(), &topology, now) {
|
||||
state.proof.clone()
|
||||
} else {
|
||||
Some(FleetCapabilityProof::new(
|
||||
topology,
|
||||
Arc::new(BTreeMap::new()),
|
||||
now + Duration::from_secs(60 * 60),
|
||||
))
|
||||
};
|
||||
state.topology_conflict = false;
|
||||
state.proof = Some(FleetCapabilityProof {
|
||||
topology_fingerprint: topology,
|
||||
peer_epochs: Arc::new(BTreeMap::new()),
|
||||
expires_at: Instant::now() + Duration::from_secs(60 * 60),
|
||||
});
|
||||
state.proof = proof.clone();
|
||||
drop(state);
|
||||
let mut journal_state = tier_delete_journal_fleet_proof_slot()
|
||||
.write()
|
||||
.unwrap_or_else(std::sync::PoisonError::into_inner);
|
||||
debug_assert!(
|
||||
journal_state
|
||||
.proof
|
||||
.as_ref()
|
||||
.is_none_or(|current| current.generation.is_drained())
|
||||
);
|
||||
journal_state.topology_conflict = false;
|
||||
journal_state.draining_generation = None;
|
||||
journal_state.proof = proof.as_ref().map(FleetCapabilityProof::with_fresh_generation);
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
pub(crate) struct CrossPoolFenceFleetProofGuard {
|
||||
previous_proof: Option<FleetCapabilityProof>,
|
||||
previous_topology_conflict: bool,
|
||||
previous_journal_proof: Option<FleetCapabilityProof>,
|
||||
previous_journal_topology_conflict: bool,
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
@@ -246,8 +484,24 @@ impl Drop for CrossPoolFenceFleetProofGuard {
|
||||
let mut state = cross_pool_fence_fleet_proof_slot()
|
||||
.write()
|
||||
.unwrap_or_else(std::sync::PoisonError::into_inner);
|
||||
state.proof = self.previous_proof.take();
|
||||
state.proof = self
|
||||
.previous_proof
|
||||
.take()
|
||||
.as_ref()
|
||||
.map(FleetCapabilityProof::with_fresh_generation);
|
||||
state.draining_generation = None;
|
||||
state.topology_conflict = self.previous_topology_conflict;
|
||||
drop(state);
|
||||
let mut journal_state = tier_delete_journal_fleet_proof_slot()
|
||||
.write()
|
||||
.unwrap_or_else(std::sync::PoisonError::into_inner);
|
||||
journal_state.proof = self
|
||||
.previous_journal_proof
|
||||
.take()
|
||||
.as_ref()
|
||||
.map(FleetCapabilityProof::with_fresh_generation);
|
||||
journal_state.draining_generation = None;
|
||||
journal_state.topology_conflict = self.previous_journal_topology_conflict;
|
||||
}
|
||||
}
|
||||
|
||||
@@ -258,12 +512,29 @@ pub(crate) fn without_cross_pool_fence_fleet_proof_for_test() -> CrossPoolFenceF
|
||||
let mut state = cross_pool_fence_fleet_proof_slot()
|
||||
.write()
|
||||
.unwrap_or_else(std::sync::PoisonError::into_inner);
|
||||
let mut journal_state = tier_delete_journal_fleet_proof_slot()
|
||||
.write()
|
||||
.unwrap_or_else(std::sync::PoisonError::into_inner);
|
||||
let guard = CrossPoolFenceFleetProofGuard {
|
||||
previous_proof: state.proof.clone(),
|
||||
previous_topology_conflict: state.topology_conflict,
|
||||
previous_journal_proof: journal_state.proof.clone(),
|
||||
previous_journal_topology_conflict: journal_state.topology_conflict,
|
||||
};
|
||||
state.proof = None;
|
||||
if let Some(proof) = state.proof.take() {
|
||||
proof.generation.revoke();
|
||||
if !proof.generation.is_drained() {
|
||||
state.draining_generation = Some(proof.generation);
|
||||
}
|
||||
}
|
||||
state.topology_conflict = true;
|
||||
if let Some(proof) = journal_state.proof.take() {
|
||||
proof.generation.revoke();
|
||||
if !proof.generation.is_drained() {
|
||||
journal_state.draining_generation = Some(proof.generation);
|
||||
}
|
||||
}
|
||||
journal_state.topology_conflict = true;
|
||||
guard
|
||||
}
|
||||
|
||||
@@ -275,11 +546,33 @@ pub fn rotate_cross_pool_fence_fleet_proof_for_test() -> bool {
|
||||
let Some(current) = state.proof.as_ref() else {
|
||||
return false;
|
||||
};
|
||||
state.proof = Some(FleetCapabilityProof {
|
||||
topology_fingerprint: current.topology_fingerprint.clone(),
|
||||
peer_epochs: Arc::new(current.peer_epochs.as_ref().clone()),
|
||||
expires_at: current.expires_at,
|
||||
});
|
||||
let proof = FleetCapabilityProof::new(
|
||||
current.topology_fingerprint.clone(),
|
||||
Arc::new(current.peer_epochs.as_ref().clone()),
|
||||
current.expires_at,
|
||||
);
|
||||
state.proof = Some(proof.clone());
|
||||
drop(state);
|
||||
let mut journal_state = tier_delete_journal_fleet_proof_slot()
|
||||
.write()
|
||||
.unwrap_or_else(std::sync::PoisonError::into_inner);
|
||||
journal_state.topology_conflict = false;
|
||||
if let Some(previous) = journal_state.proof.take() {
|
||||
previous.generation.revoke();
|
||||
if !previous.generation.is_drained() {
|
||||
journal_state.draining_generation = Some(previous.generation);
|
||||
}
|
||||
}
|
||||
if journal_state
|
||||
.draining_generation
|
||||
.as_ref()
|
||||
.is_some_and(|generation| generation.is_drained())
|
||||
{
|
||||
journal_state.draining_generation = None;
|
||||
}
|
||||
if journal_state.draining_generation.is_none() {
|
||||
journal_state.proof = Some(proof.with_fresh_generation());
|
||||
}
|
||||
true
|
||||
}
|
||||
|
||||
@@ -291,15 +584,22 @@ fn fleet_capability_proof_matches(
|
||||
return false;
|
||||
};
|
||||
let state = slot.read().unwrap_or_else(std::sync::PoisonError::into_inner);
|
||||
if state.topology_conflict {
|
||||
return false;
|
||||
}
|
||||
state.proof.as_ref().is_some_and(|current| {
|
||||
current.topology_fingerprint == *expected_topology
|
||||
&& current.topology_fingerprint == proof.topology_fingerprint
|
||||
&& Arc::ptr_eq(¤t.peer_epochs, &proof.peer_epochs)
|
||||
&& Instant::now() < current.expires_at
|
||||
})
|
||||
fleet_capability_proof_matches_at(&state, proof, expected_topology, Instant::now())
|
||||
}
|
||||
|
||||
fn fleet_capability_proof_matches_at(
|
||||
state: &FleetCapabilityProofState,
|
||||
proof: &FleetCapabilityProofToken,
|
||||
expected_topology: &str,
|
||||
now: Instant,
|
||||
) -> bool {
|
||||
!state.topology_conflict
|
||||
&& state.proof.as_ref().is_some_and(|current| {
|
||||
current.topology_fingerprint == expected_topology
|
||||
&& current.topology_fingerprint == proof.topology_fingerprint
|
||||
&& Arc::ptr_eq(¤t.peer_epochs, &proof.peer_epochs)
|
||||
&& now < current.expires_at
|
||||
})
|
||||
}
|
||||
|
||||
fn fleet_capability_proof_valid_at(proof: Option<&FleetCapabilityProof>, expected_topology: &str, now: Instant) -> bool {
|
||||
@@ -312,7 +612,7 @@ pub(crate) struct RemoteVersionStateFleetProofGuard;
|
||||
#[cfg(test)]
|
||||
impl Drop for RemoteVersionStateFleetProofGuard {
|
||||
fn drop(&mut self) {
|
||||
replace_fleet_capability_proof(remote_version_state_fleet_proof_slot(), None);
|
||||
revoke_fleet_capability_proof(remote_version_state_fleet_proof_slot());
|
||||
}
|
||||
}
|
||||
|
||||
@@ -348,10 +648,12 @@ fn insert_remote_version_state_peer(peer_epochs: &mut BTreeMap<String, Uuid>, pe
|
||||
pub fn start_remote_version_state_fleet_probe(topology_fingerprint: String) {
|
||||
if REMOTE_VERSION_STATE_PROBE_TOPOLOGY.set(topology_fingerprint.clone()).is_err() {
|
||||
if REMOTE_VERSION_STATE_PROBE_TOPOLOGY.get() != Some(&topology_fingerprint) {
|
||||
for slot in [remote_version_state_fleet_proof_slot(), cross_pool_fence_fleet_proof_slot()] {
|
||||
let mut state = slot.write().unwrap_or_else(std::sync::PoisonError::into_inner);
|
||||
state.topology_conflict = true;
|
||||
state.proof = None;
|
||||
for slot in [
|
||||
remote_version_state_fleet_proof_slot(),
|
||||
cross_pool_fence_fleet_proof_slot(),
|
||||
tier_delete_journal_fleet_proof_slot(),
|
||||
] {
|
||||
mark_fleet_capability_topology_conflict(slot);
|
||||
}
|
||||
}
|
||||
return;
|
||||
@@ -373,7 +675,7 @@ pub fn start_remote_version_state_fleet_probe(topology_fingerprint: String) {
|
||||
}
|
||||
None => Err(Error::other("remote version state fleet capability notification system is unavailable")),
|
||||
};
|
||||
let fence_result = match get_global_notification_sys() {
|
||||
let fence_probe = match get_global_notification_sys() {
|
||||
Some(notification_sys) => timeout(
|
||||
REMOTE_VERSION_STATE_PROBE_TIMEOUT,
|
||||
notification_sys.probe_cross_pool_fence_fleet(&topology_fingerprint),
|
||||
@@ -382,13 +684,21 @@ pub fn start_remote_version_state_fleet_probe(topology_fingerprint: String) {
|
||||
.unwrap_or_else(|_| Err(Error::other("cross-pool fence fleet capability probe timed out"))),
|
||||
None => Err(Error::other("cross-pool fence fleet capability notification system is unavailable")),
|
||||
};
|
||||
let (fence_result, journal_result) = match fence_probe {
|
||||
Ok((peer_epochs, minimum_version)) => cross_pool_fence_policy_results(peer_epochs, minimum_version),
|
||||
Err(err) => {
|
||||
let message = err.to_string();
|
||||
(Err(Error::other(message.clone())), Err(Error::other(message)))
|
||||
}
|
||||
};
|
||||
let topology_conflict = remote_version_state_fleet_proof_slot()
|
||||
.read()
|
||||
.unwrap_or_else(std::sync::PoisonError::into_inner)
|
||||
.topology_conflict;
|
||||
if topology_conflict {
|
||||
replace_fleet_capability_proof(remote_version_state_fleet_proof_slot(), None);
|
||||
replace_fleet_capability_proof(cross_pool_fence_fleet_proof_slot(), None);
|
||||
revoke_fleet_capability_proof(remote_version_state_fleet_proof_slot());
|
||||
revoke_fleet_capability_proof(cross_pool_fence_fleet_proof_slot());
|
||||
revoke_fleet_capability_proof(tier_delete_journal_fleet_proof_slot());
|
||||
} else if let Some(err) = publish_fleet_capability_probe_result(
|
||||
remote_version_state_fleet_proof_slot(),
|
||||
&topology_fingerprint,
|
||||
@@ -409,7 +719,25 @@ pub fn start_remote_version_state_fleet_probe(topology_fingerprint: String) {
|
||||
event = EVENT_NOTIFICATION_CAPABILITY_PROBE,
|
||||
component = LOG_COMPONENT_ECSTORE,
|
||||
subsystem = LOG_SUBSYSTEM_NOTIFICATION,
|
||||
capability = "cross_pool_fence_v2",
|
||||
capability = "cross_pool_fence",
|
||||
state = "failed_closed",
|
||||
error = %err,
|
||||
"notification capability probe"
|
||||
);
|
||||
}
|
||||
if !topology_conflict
|
||||
&& let Some(err) = publish_fleet_capability_probe_result(
|
||||
tier_delete_journal_fleet_proof_slot(),
|
||||
&topology_fingerprint,
|
||||
journal_result,
|
||||
Instant::now(),
|
||||
)
|
||||
{
|
||||
debug!(
|
||||
event = EVENT_NOTIFICATION_CAPABILITY_PROBE,
|
||||
component = LOG_COMPONENT_ECSTORE,
|
||||
subsystem = LOG_SUBSYSTEM_NOTIFICATION,
|
||||
capability = "tier_delete_journal_v6_policy",
|
||||
state = "failed_closed",
|
||||
error = %err,
|
||||
"notification capability probe"
|
||||
@@ -483,7 +811,7 @@ impl NotificationSys {
|
||||
Ok(peer_epochs)
|
||||
}
|
||||
|
||||
async fn probe_cross_pool_fence_fleet(&self, topology_fingerprint: &str) -> Result<BTreeMap<String, Uuid>> {
|
||||
async fn probe_cross_pool_fence_fleet(&self, topology_fingerprint: &str) -> Result<(BTreeMap<String, Uuid>, u32)> {
|
||||
if self.peer_clients.len() != self.peer_topology_hosts.len() {
|
||||
return Err(Error::other("cross-pool fence capability fleet membership is incomplete"));
|
||||
}
|
||||
@@ -494,14 +822,21 @@ impl NotificationSys {
|
||||
client.probe_cross_pool_fence(topology_fingerprint.to_string()).await
|
||||
});
|
||||
let mut peer_epochs = BTreeMap::new();
|
||||
let mut minimum_version = u32::MAX;
|
||||
for result in join_all(probes).await {
|
||||
let (peer, version, epoch) = result?;
|
||||
if version < CROSS_POOL_FENCE_SUPPORTED_VERSION {
|
||||
return Err(Error::other("cross-pool fence capability version is unsupported"));
|
||||
}
|
||||
minimum_version = minimum_version.min(version);
|
||||
insert_remote_version_state_peer(&mut peer_epochs, peer, epoch)?;
|
||||
}
|
||||
Ok(peer_epochs)
|
||||
// A single-node deployment has no remote member to lower the local
|
||||
// policy version advertised by this binary.
|
||||
if minimum_version == u32::MAX {
|
||||
minimum_version = TIER_DELETE_JOURNAL_POLICY_SUPPORTED_VERSION;
|
||||
}
|
||||
Ok((peer_epochs, minimum_version))
|
||||
}
|
||||
}
|
||||
|
||||
@@ -1827,12 +2162,13 @@ impl NotificationSys {
|
||||
join_all(futures).await
|
||||
}
|
||||
|
||||
pub async fn abort_tier_mutation(&self, mutation_id: Uuid) -> Vec<NotificationPeerErr> {
|
||||
pub async fn abort_tier_mutation(&self, mutation_id: Uuid, canonical_prepare_payload: Bytes) -> Vec<NotificationPeerErr> {
|
||||
let mut futures = Vec::with_capacity(self.peer_clients.len());
|
||||
for client in self.peer_clients.iter().cloned() {
|
||||
let payload = canonical_prepare_payload.clone();
|
||||
futures.push(async move {
|
||||
if let Some(client) = client {
|
||||
notification_peer_result(client.host.to_string(), client.abort_tier_mutation(mutation_id).await)
|
||||
notification_peer_result(client.host.to_string(), client.abort_tier_mutation(mutation_id, payload).await)
|
||||
} else {
|
||||
unreachable_notification_peer_err()
|
||||
}
|
||||
@@ -1962,6 +2298,12 @@ where
|
||||
.map_err(|_| Error::other(format!("scanner activity peer {host} timed out after {timeout_duration:?}")))?
|
||||
}
|
||||
|
||||
/// Classify transport-only activity failures without treating an answered
|
||||
/// peer's application error as an outage.
|
||||
pub fn scanner_peer_transport_error_message_is_retryable(error: &str) -> bool {
|
||||
crate::cluster::rpc::client::message_has_network_needle(error)
|
||||
}
|
||||
|
||||
fn scanner_activity_should_retry(first_error: Option<&Error>, timed_out: bool) -> bool {
|
||||
timed_out || first_error.is_some_and(PeerRestClient::is_network_like_error)
|
||||
}
|
||||
@@ -2461,16 +2803,24 @@ fn aggregate_scanner_dirty_usage_acknowledgement_results(
|
||||
mod tests {
|
||||
use super::*;
|
||||
|
||||
#[test]
|
||||
fn cross_pool_v2_remains_generic_but_cannot_authorize_v6_journal() {
|
||||
let peers = BTreeMap::from([("node-b:9000".to_string(), Uuid::new_v4())]);
|
||||
let (generic_v2, journal_v2) = cross_pool_fence_policy_results(peers.clone(), 2);
|
||||
assert!(generic_v2.is_ok(), "v2 remains valid for existing cross-pool fencing");
|
||||
assert!(journal_v2.is_err(), "a mixed v2/v3 fleet must fail closed for journal-v6 deletion");
|
||||
|
||||
let (generic_v3, journal_v3) = cross_pool_fence_policy_results(peers, 3);
|
||||
assert!(generic_v3.is_ok());
|
||||
assert!(journal_v3.is_ok(), "an all-v3 fleet may authorize journal-v6 deletion");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn remote_version_state_fleet_proof_rejects_stale_or_mismatched_membership() {
|
||||
let now = Instant::now();
|
||||
let mut peer_epochs = BTreeMap::new();
|
||||
peer_epochs.insert("peer-a".to_string(), Uuid::new_v4());
|
||||
let proof = FleetCapabilityProof {
|
||||
topology_fingerprint: "topology-a".to_string(),
|
||||
peer_epochs: Arc::new(peer_epochs),
|
||||
expires_at: now + Duration::from_secs(1),
|
||||
};
|
||||
let proof = FleetCapabilityProof::new("topology-a".to_string(), Arc::new(peer_epochs), now + Duration::from_secs(1));
|
||||
|
||||
assert!(fleet_capability_proof_valid_at(Some(&proof), "topology-a", now));
|
||||
assert!(!fleet_capability_proof_valid_at(Some(&proof), "topology-b", now));
|
||||
@@ -2489,11 +2839,7 @@ mod tests {
|
||||
#[test]
|
||||
fn remote_version_state_fleet_proof_accepts_single_node_membership() {
|
||||
let now = Instant::now();
|
||||
let proof = FleetCapabilityProof {
|
||||
topology_fingerprint: "topology-a".to_string(),
|
||||
peer_epochs: Arc::new(BTreeMap::new()),
|
||||
expires_at: now + Duration::from_secs(1),
|
||||
};
|
||||
let proof = FleetCapabilityProof::new("topology-a".to_string(), Arc::new(BTreeMap::new()), now + Duration::from_secs(1));
|
||||
|
||||
assert!(fleet_capability_proof_valid_at(Some(&proof), "topology-a", now));
|
||||
}
|
||||
@@ -2501,21 +2847,97 @@ mod tests {
|
||||
#[test]
|
||||
fn remote_version_state_fleet_proof_token_changes_with_process_epoch() {
|
||||
let now = Instant::now();
|
||||
let proof = FleetCapabilityProof {
|
||||
topology_fingerprint: "topology-a".to_string(),
|
||||
peer_epochs: Arc::new(BTreeMap::from([("peer-a".to_string(), Uuid::new_v4())])),
|
||||
expires_at: now + Duration::from_secs(1),
|
||||
};
|
||||
let proof = FleetCapabilityProof::new(
|
||||
"topology-a".to_string(),
|
||||
Arc::new(BTreeMap::from([("peer-a".to_string(), Uuid::new_v4())])),
|
||||
now + Duration::from_secs(1),
|
||||
);
|
||||
let captured = proof.token();
|
||||
let restarted = FleetCapabilityProof {
|
||||
topology_fingerprint: proof.topology_fingerprint.clone(),
|
||||
peer_epochs: Arc::new(BTreeMap::from([("peer-a".to_string(), Uuid::new_v4())])),
|
||||
expires_at: proof.expires_at,
|
||||
};
|
||||
let restarted = FleetCapabilityProof::new(
|
||||
proof.topology_fingerprint.clone(),
|
||||
Arc::new(BTreeMap::from([("peer-a".to_string(), Uuid::new_v4())])),
|
||||
proof.expires_at,
|
||||
);
|
||||
|
||||
assert!(captured != restarted.token());
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn tier_delete_journal_generation_is_stable_across_members_and_process_restarts() {
|
||||
let topology = "topology-a";
|
||||
let now = Instant::now();
|
||||
let node_a_view = FleetCapabilityProof::new(
|
||||
topology.to_string(),
|
||||
Arc::new(BTreeMap::from([("node-b".to_string(), Uuid::new_v4())])),
|
||||
now + Duration::from_secs(1),
|
||||
);
|
||||
let node_b_view = FleetCapabilityProof::new(
|
||||
topology.to_string(),
|
||||
Arc::new(BTreeMap::from([("node-a".to_string(), Uuid::new_v4())])),
|
||||
now + Duration::from_secs(1),
|
||||
);
|
||||
let restarted_node_a_view = FleetCapabilityProof::new(
|
||||
topology.to_string(),
|
||||
Arc::new(BTreeMap::from([("node-b".to_string(), Uuid::new_v4())])),
|
||||
now + Duration::from_secs(1),
|
||||
);
|
||||
|
||||
let generations = [&node_a_view, &node_b_view, &restarted_node_a_view]
|
||||
.map(|proof| stable_tier_delete_journal_topology_generation(&proof.token().topology_fingerprint));
|
||||
assert_eq!(generations[0], generations[1]);
|
||||
assert_eq!(generations[0], generations[2]);
|
||||
assert_ne!(
|
||||
generations[0],
|
||||
stable_tier_delete_journal_topology_generation("topology-b"),
|
||||
"a real topology change must produce a different durable generation"
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn tier_delete_journal_restart_revokes_old_token_but_fresh_token_recovers_same_generation() {
|
||||
let slot = std::sync::RwLock::new(FleetCapabilityProofState::default());
|
||||
let now = Instant::now();
|
||||
let original_peers = BTreeMap::from([("node-b".to_string(), Uuid::new_v4())]);
|
||||
assert!(publish_fleet_capability_probe_result(&slot, "topology-a", Ok(original_peers), now).is_none());
|
||||
let original = slot
|
||||
.read()
|
||||
.expect("proof slot should not poison")
|
||||
.proof
|
||||
.as_ref()
|
||||
.expect("successful probe should publish proof")
|
||||
.token();
|
||||
let original_generation = stable_tier_delete_journal_topology_generation(&original.topology_fingerprint);
|
||||
|
||||
let restarted_peers = BTreeMap::from([("node-b".to_string(), Uuid::new_v4())]);
|
||||
assert!(
|
||||
publish_fleet_capability_probe_result(&slot, "topology-a", Ok(restarted_peers), now + Duration::from_millis(1))
|
||||
.is_none()
|
||||
);
|
||||
let state = slot.read().expect("proof slot should not poison");
|
||||
let fresh = state
|
||||
.proof
|
||||
.as_ref()
|
||||
.expect("restart probe should publish a fresh proof")
|
||||
.token();
|
||||
|
||||
assert!(!fleet_capability_proof_matches_at(
|
||||
&state,
|
||||
&original,
|
||||
"topology-a",
|
||||
now + Duration::from_millis(2)
|
||||
));
|
||||
assert!(fleet_capability_proof_matches_at(
|
||||
&state,
|
||||
&fresh,
|
||||
"topology-a",
|
||||
now + Duration::from_millis(2)
|
||||
));
|
||||
assert_eq!(
|
||||
original_generation,
|
||||
stable_tier_delete_journal_topology_generation(&fresh.topology_fingerprint)
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn remote_version_state_fleet_proof_renewal_preserves_only_same_epoch_token() {
|
||||
let slot = std::sync::RwLock::new(FleetCapabilityProofState::default());
|
||||
@@ -2555,15 +2977,106 @@ mod tests {
|
||||
assert!(!Arc::ptr_eq(&original.peer_epochs, &replaced.peer_epochs));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn tier_delete_journal_successor_waits_for_inflight_generation_to_drain() {
|
||||
let slot = std::sync::RwLock::new(FleetCapabilityProofState::default());
|
||||
let now = Instant::now();
|
||||
let original_peers = BTreeMap::from([("peer-a".to_string(), Uuid::new_v4())]);
|
||||
assert!(publish_fleet_capability_probe_result(&slot, "topology-a", Ok(original_peers), now).is_none());
|
||||
|
||||
let admitted = {
|
||||
let state = slot.read().expect("proof slot should not poison");
|
||||
acquire_tier_delete_journal_fleet_proof_from(&state, "topology-a", now)
|
||||
.expect("a fresh proof should admit one journal operation")
|
||||
};
|
||||
{
|
||||
let state = slot.read().expect("proof slot should not poison");
|
||||
assert!(
|
||||
tier_delete_journal_fleet_proof_matches_at(&state, &admitted, "topology-a", now),
|
||||
"a freshly admitted journal proof must remain current"
|
||||
);
|
||||
assert!(
|
||||
acquire_tier_delete_journal_fleet_proof_from(&state, "topology-a", now + REMOTE_VERSION_STATE_PROOF_TTL,)
|
||||
.is_none(),
|
||||
"TTL expiry must stop new admission"
|
||||
);
|
||||
assert!(
|
||||
!tier_delete_journal_fleet_proof_matches_at(
|
||||
&state,
|
||||
&admitted,
|
||||
"topology-a",
|
||||
now + REMOTE_VERSION_STATE_PROOF_TTL,
|
||||
),
|
||||
"TTL expiry must also stop an admitted proof at its next durable fence"
|
||||
);
|
||||
assert!(!admitted._permit.generation.is_drained());
|
||||
}
|
||||
|
||||
let restarted_peers = BTreeMap::from([("peer-a".to_string(), Uuid::new_v4())]);
|
||||
let blocked = publish_fleet_capability_probe_result(
|
||||
&slot,
|
||||
"topology-a",
|
||||
Ok(restarted_peers.clone()),
|
||||
now + Duration::from_millis(1),
|
||||
)
|
||||
.expect("a successor proof must wait for the admitted generation");
|
||||
assert!(blocked.to_string().contains("previous generation to drain"));
|
||||
{
|
||||
let state = slot.read().expect("proof slot should not poison");
|
||||
assert!(state.proof.is_none(), "new operations must remain closed while the predecessor drains");
|
||||
assert!(state.draining_generation.is_some());
|
||||
assert!(
|
||||
!tier_delete_journal_fleet_proof_matches_at(&state, &admitted, "topology-a", now + Duration::from_millis(1),),
|
||||
"a restarted peer must revoke an admitted proof before its next durable fence"
|
||||
);
|
||||
}
|
||||
|
||||
drop(admitted);
|
||||
assert!(
|
||||
publish_fleet_capability_probe_result(&slot, "topology-a", Ok(restarted_peers), now + Duration::from_millis(2),)
|
||||
.is_none(),
|
||||
"the successor may publish after the in-flight operation releases its permit"
|
||||
);
|
||||
let state = slot.read().expect("proof slot should not poison");
|
||||
assert!(state.proof.is_some());
|
||||
assert!(state.draining_generation.is_none());
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn tier_delete_journal_topology_conflict_revokes_admitted_generation() {
|
||||
let slot = std::sync::RwLock::new(FleetCapabilityProofState::default());
|
||||
let now = Instant::now();
|
||||
let peers = BTreeMap::from([("peer-a".to_string(), Uuid::new_v4())]);
|
||||
assert!(publish_fleet_capability_probe_result(&slot, "topology-a", Ok(peers), now).is_none());
|
||||
let admitted = {
|
||||
let state = slot.read().expect("proof slot should not poison");
|
||||
acquire_tier_delete_journal_fleet_proof_from(&state, "topology-a", now)
|
||||
.expect("a fresh proof should admit one journal operation")
|
||||
};
|
||||
|
||||
mark_fleet_capability_topology_conflict(&slot);
|
||||
|
||||
let state = slot.read().expect("proof slot should not poison");
|
||||
assert!(state.topology_conflict);
|
||||
assert!(state.proof.is_none());
|
||||
assert!(state.draining_generation.is_some());
|
||||
assert!(!admitted._permit.generation.is_accepting());
|
||||
assert!(
|
||||
!tier_delete_journal_fleet_proof_matches_at(&state, &admitted, "topology-a", now),
|
||||
"topology conflict must revoke an already admitted journal proof"
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn remote_version_state_fleet_proof_conflict_revokes_atomic_snapshot() {
|
||||
let now = Instant::now();
|
||||
let mut state = FleetCapabilityProofState {
|
||||
proof: Some(FleetCapabilityProof {
|
||||
topology_fingerprint: "topology-a".to_string(),
|
||||
peer_epochs: Arc::new(BTreeMap::new()),
|
||||
expires_at: now + Duration::from_secs(1),
|
||||
}),
|
||||
proof: Some(FleetCapabilityProof::new(
|
||||
"topology-a".to_string(),
|
||||
Arc::new(BTreeMap::new()),
|
||||
now + Duration::from_secs(1),
|
||||
)),
|
||||
draining_generation: None,
|
||||
topology_conflict: false,
|
||||
};
|
||||
assert!(acquire_fleet_capability_proof_from(&state, "topology-a", now).is_some());
|
||||
@@ -3273,7 +3786,7 @@ mod tests {
|
||||
assert_eq!(commit.len(), 1);
|
||||
assert!(commit[0].err.is_some());
|
||||
|
||||
let abort = sys.abort_tier_mutation(mutation_id).await;
|
||||
let abort = sys.abort_tier_mutation(mutation_id, Bytes::from_static(b"prepare")).await;
|
||||
assert_eq!(abort.len(), 1);
|
||||
assert!(abort[0].err.is_some());
|
||||
}
|
||||
|
||||
@@ -845,7 +845,7 @@ impl ECStore {
|
||||
|
||||
let mut pool_stats = Vec::with_capacity(self.pools.len());
|
||||
|
||||
let now = OffsetDateTime::now_utc();
|
||||
let now = self.next_scanner_data_movement_update(OffsetDateTime::now_utc()).await;
|
||||
|
||||
for disk_stat in disk_stats.iter() {
|
||||
let mut pool_stat = RebalanceStats {
|
||||
@@ -868,8 +868,10 @@ impl ECStore {
|
||||
pool_stats.push(pool_stat);
|
||||
}
|
||||
|
||||
let has_participating_pool = pool_stats.iter().any(|pool_stat| pool_stat.participating);
|
||||
let meta = RebalanceMeta {
|
||||
id: Uuid::new_v4().to_string(),
|
||||
stopped_at: (!has_participating_pool).then_some(now),
|
||||
percent_free_goal,
|
||||
pool_stats,
|
||||
..Default::default()
|
||||
@@ -963,6 +965,18 @@ impl ECStore {
|
||||
)));
|
||||
}
|
||||
if meta.stopped_at.is_some() {
|
||||
if !is_rebalance_conflicting_with_decommission(meta) {
|
||||
debug!(
|
||||
event = EVENT_REBALANCE_STATE,
|
||||
component = LOG_COMPONENT_ECSTORE,
|
||||
subsystem = LOG_SUBSYSTEM_REBALANCE,
|
||||
state = "start_skipped",
|
||||
reason = "not_started_terminal",
|
||||
rebalance_id = %expected_id,
|
||||
"Skipped rebalance start because metadata is already terminal"
|
||||
);
|
||||
return Ok(());
|
||||
}
|
||||
return Err(Error::other(format!("rebalance {expected_id} was stopped before start")));
|
||||
}
|
||||
}
|
||||
@@ -1214,11 +1228,11 @@ impl ECStore {
|
||||
};
|
||||
let movement_gate = self.ctx.data_movement_operation_gate();
|
||||
let _movement_guard = movement_gate.write().await;
|
||||
let stopped_at = self.next_scanner_data_movement_update(OffsetDateTime::now_utc()).await;
|
||||
let (previous_meta, meta_to_save) = {
|
||||
let mut rebalance_meta = self.rebalance_meta.write().await;
|
||||
let previous_meta = rebalance_meta.clone();
|
||||
let meta_to_save =
|
||||
stop_rebalance_meta_snapshot_for_id(rebalance_meta.as_mut(), OffsetDateTime::now_utc(), expected_id)?;
|
||||
let meta_to_save = stop_rebalance_meta_snapshot_for_id(rebalance_meta.as_mut(), stopped_at, expected_id)?;
|
||||
(previous_meta, meta_to_save)
|
||||
};
|
||||
|
||||
@@ -1250,14 +1264,10 @@ impl ECStore {
|
||||
.await?;
|
||||
let movement_gate = self.ctx.data_movement_operation_gate();
|
||||
let _movement_guard = movement_gate.write().await;
|
||||
let failed_at = self.next_scanner_data_movement_update(OffsetDateTime::now_utc()).await;
|
||||
let meta_to_save = {
|
||||
let mut rebalance_meta = self.rebalance_meta.write().await;
|
||||
rollback_rebalance_start_meta_snapshot_for_id(
|
||||
rebalance_meta.as_mut(),
|
||||
OffsetDateTime::now_utc(),
|
||||
expected_id,
|
||||
start_error,
|
||||
)
|
||||
rollback_rebalance_start_meta_snapshot_for_id(rebalance_meta.as_mut(), failed_at, expected_id, start_error)
|
||||
};
|
||||
|
||||
if let Some(meta_to_save) = meta_to_save {
|
||||
@@ -1319,14 +1329,19 @@ mod tests {
|
||||
use super::*;
|
||||
use crate::config::com::delete_config;
|
||||
use crate::core::pools::{
|
||||
POOL_META_NAME, PoolActivationDurableSaveBarrier, PoolActivationStartKind, PoolActivationStartProbe, PoolMetaWriteState,
|
||||
persist_pool_meta_identity_for_startup,
|
||||
DecommissionErasureLayout, DecommissionPoolCapacityInfo, POOL_META_NAME, PoolActivationDurableSaveBarrier,
|
||||
PoolActivationStartKind, PoolActivationStartProbe, PoolMetaWriteState, persist_pool_meta_identity_for_startup,
|
||||
set_decommission_capacity_info_overrides_for_test,
|
||||
};
|
||||
use crate::object_api::NamespaceLockFence;
|
||||
use crate::set_disk::{PutObjectCommitBarrier, PutObjectCommitPause, hermetic_set_disks_isolated};
|
||||
|
||||
async fn persist_initialized_identity_then_remove_pool_meta(store: &Arc<ECStore>) {
|
||||
let mut write_state = PoolMetaWriteState::for_startup(store.id, false);
|
||||
let deployment_id = store
|
||||
.ctx
|
||||
.deployment_id()
|
||||
.expect("test store should have a deployment identity");
|
||||
let mut write_state = PoolMetaWriteState::for_startup(deployment_id, false);
|
||||
persist_pool_meta_identity_for_startup(store.pools.clone(), &mut write_state, true)
|
||||
.await
|
||||
.expect("initialized pool metadata identity should persist");
|
||||
@@ -1402,6 +1417,62 @@ mod tests {
|
||||
assert!(cancel.is_cancelled());
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
#[serial_test::serial]
|
||||
async fn equal_free_ratio_admin_no_participant_rebalance_succeeds_and_persists_terminal_generation_after_restart() {
|
||||
let (_temp_dirs, store, restarted) =
|
||||
crate::services::rebalance::test_two_pool_stores_with_isolated_node_contexts(None).await;
|
||||
let movement_floor = OffsetDateTime::from_unix_timestamp(4_100_000_000).expect("future test timestamp should be valid");
|
||||
*store.rebalance_meta.write().await = Some(RebalanceMeta {
|
||||
id: "previous-terminal-rebalance".to_string(),
|
||||
stopped_at: Some(movement_floor),
|
||||
..Default::default()
|
||||
});
|
||||
set_rebalance_disk_stats_override_for_test(
|
||||
store.id,
|
||||
vec![
|
||||
DiskStat {
|
||||
total_space: 100,
|
||||
available_space: 50,
|
||||
},
|
||||
DiskStat {
|
||||
total_space: 100,
|
||||
available_space: 50,
|
||||
},
|
||||
],
|
||||
);
|
||||
|
||||
let rebalance_id = store
|
||||
.init_and_start_rebalance(vec!["equal-ratio-no-op".to_string()])
|
||||
.await
|
||||
.expect("equal free ratio admin rebalance should succeed as a terminal no-op");
|
||||
let stopped_at = {
|
||||
let local = store.rebalance_meta.read().await;
|
||||
let local = local.as_ref().expect("no-op rebalance metadata should remain available");
|
||||
assert_eq!(local.id, rebalance_id);
|
||||
assert!(local.pool_stats.iter().all(|pool_stat| !pool_stat.participating));
|
||||
let stopped_at = local.stopped_at.expect("no-op rebalance must persist a terminal timestamp");
|
||||
assert_eq!(stopped_at, movement_floor + time::Duration::nanoseconds(1));
|
||||
stopped_at
|
||||
};
|
||||
|
||||
let stopped_generation =
|
||||
u64::try_from(stopped_at.unix_timestamp_nanos()).expect("terminal timestamp should map to scanner generation");
|
||||
let live_status = store.scanner_data_movement_pause_status().await;
|
||||
assert!(!live_status.paused);
|
||||
assert_eq!(live_status.movement_generation, stopped_generation);
|
||||
|
||||
restarted
|
||||
.load_rebalance_meta()
|
||||
.await
|
||||
.expect("restarted store should load the persisted no-op rebalance metadata");
|
||||
let status = restarted.scanner_data_movement_pause_status().await;
|
||||
|
||||
assert!(!status.paused);
|
||||
assert_eq!(status.movement_generation, stopped_generation);
|
||||
assert_eq!(restarted.scanner_data_movement_generation(), stopped_generation);
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
#[serial_test::serial]
|
||||
async fn rebalance_activation_rejects_initialized_cluster_with_all_pool_meta_missing() {
|
||||
@@ -1667,6 +1738,8 @@ mod tests {
|
||||
async fn assert_real_activation_start_race(paused_kind: PoolActivationStartKind) {
|
||||
let (_temp_dirs, rebalance_store, decommission_store) =
|
||||
crate::services::rebalance::test_two_pool_stores_with_isolated_node_contexts(None).await;
|
||||
crate::services::rebalance::promote_test_pool_meta_to_v2(&rebalance_store).await;
|
||||
crate::services::rebalance::promote_test_pool_meta_to_v2(&decommission_store).await;
|
||||
let disk_stats = vec![
|
||||
DiskStat {
|
||||
total_space: 100,
|
||||
@@ -1679,26 +1752,15 @@ mod tests {
|
||||
];
|
||||
set_rebalance_disk_stats_override_for_test(rebalance_store.id, disk_stats.clone());
|
||||
set_rebalance_disk_stats_override_for_test(decommission_store.id, disk_stats);
|
||||
crate::core::pools::set_decommission_space_info_override_for_test(
|
||||
let layout = DecommissionErasureLayout { data: 1, parity: 0 };
|
||||
let capacity_snapshot = vec![
|
||||
DecommissionPoolCapacityInfo::for_test(0, layout, 0, 100, 100),
|
||||
DecommissionPoolCapacityInfo::for_test(1, layout, 200, 200, 0),
|
||||
];
|
||||
// Decommission start samples capacity before and inside its durable activation fence.
|
||||
set_decommission_capacity_info_overrides_for_test(
|
||||
decommission_store.id,
|
||||
vec![
|
||||
(
|
||||
0,
|
||||
crate::core::pools::PoolSpaceInfo {
|
||||
free: 0,
|
||||
total: 100,
|
||||
used: 100,
|
||||
},
|
||||
),
|
||||
(
|
||||
1,
|
||||
crate::core::pools::PoolSpaceInfo {
|
||||
free: 200,
|
||||
total: 200,
|
||||
used: 0,
|
||||
},
|
||||
),
|
||||
],
|
||||
vec![capacity_snapshot.clone(), capacity_snapshot],
|
||||
);
|
||||
let (first_object, competing_object, competing_kind) = match paused_kind {
|
||||
PoolActivationStartKind::Rebalance => {
|
||||
|
||||
@@ -356,7 +356,7 @@ impl ECStore {
|
||||
};
|
||||
run_guard.ensure_held("rebalance version migration")?;
|
||||
let result = migrate_entry_version(
|
||||
&RebalanceMigrationBackend::new(set.as_ref(), self.as_ref(), lock_lost_signal.clone()),
|
||||
&RebalanceMigrationBackend::new(set.as_ref(), self.clone(), lock_lost_signal.clone()),
|
||||
bucket.clone(),
|
||||
pool_index,
|
||||
version,
|
||||
@@ -1478,6 +1478,7 @@ mod tests {
|
||||
let (_temp_dirs, store, _unused_store) =
|
||||
crate::services::rebalance::test_two_pool_stores(Some(active_rebalance_meta(REBALANCE_ID))).await;
|
||||
prepare_rebalance_test_volumes(store.as_ref()).await;
|
||||
crate::services::tier::test_util::register_mock_tier(&store.tier_config_mgr(), "WARM").await;
|
||||
let source_set = store.pools[0].get_disks_by_key(object);
|
||||
let target_set = store.pools[1].get_disks_by_key(object);
|
||||
let version_id = uuid::Uuid::new_v4();
|
||||
@@ -1520,14 +1521,17 @@ mod tests {
|
||||
let entry = metacache_entry_from_source(source_set.as_ref(), bucket, object).await;
|
||||
let run_signal_fence = RebalanceRunSignalTestFence::install(REBALANCE_ID);
|
||||
let barrier = TieredMetadataCommitBarrier::install(bucket, object);
|
||||
let task = spawn_real_rebalance_entry(
|
||||
let mut task = spawn_real_rebalance_entry(
|
||||
Arc::clone(&store),
|
||||
Arc::clone(&source_set),
|
||||
entry,
|
||||
REBALANCE_ID,
|
||||
Arc::new(RebalanceBucketConfigs::default()),
|
||||
);
|
||||
barrier.wait_until_paused().await;
|
||||
tokio::select! {
|
||||
_ = barrier.wait_until_paused() => {}
|
||||
result = &mut task => panic!("rebalance exited before the tiered commit barrier: {result:?}"),
|
||||
}
|
||||
run_signal_fence.mark_lost();
|
||||
barrier.release();
|
||||
drop(barrier);
|
||||
|
||||
@@ -101,14 +101,14 @@ pub(crate) trait MigrationBackend: Send + Sync {
|
||||
|
||||
pub(crate) struct RebalanceMigrationBackend<'a> {
|
||||
source: &'a SetDisks,
|
||||
store: &'a ECStore,
|
||||
store: std::sync::Arc<ECStore>,
|
||||
lock_lost_signal: Option<std::sync::Arc<rustfs_lock::distributed_lock::LockLostSignal>>,
|
||||
}
|
||||
|
||||
impl<'a> RebalanceMigrationBackend<'a> {
|
||||
pub(crate) fn new(
|
||||
source: &'a SetDisks,
|
||||
store: &'a ECStore,
|
||||
store: std::sync::Arc<ECStore>,
|
||||
lock_lost_signal: Option<std::sync::Arc<rustfs_lock::distributed_lock::LockLostSignal>>,
|
||||
) -> Self {
|
||||
Self {
|
||||
|
||||
@@ -83,6 +83,7 @@ pub async fn test_store_with_persisted_rebalance_meta(
|
||||
decommission_cancelers: tokio::sync::RwLock::new(vec![None]),
|
||||
start_gate: tokio::sync::Mutex::new(()),
|
||||
pool_meta_save_gate: tokio::sync::Mutex::default(),
|
||||
decommission_capacity_entry_gate: tokio::sync::Mutex::default(),
|
||||
ctx,
|
||||
bucket_fence_registry: std::sync::Arc::default(),
|
||||
});
|
||||
@@ -97,7 +98,7 @@ pub(crate) async fn test_two_pool_stores(
|
||||
std::sync::Arc<crate::store::ECStore>,
|
||||
std::sync::Arc<crate::store::ECStore>,
|
||||
) {
|
||||
test_two_pool_stores_with_contexts(rebalance_meta, false).await
|
||||
test_pool_stores_with_contexts(rebalance_meta, false, 2, 2).await
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
@@ -108,30 +109,73 @@ pub(crate) async fn test_two_pool_stores_with_isolated_node_contexts(
|
||||
std::sync::Arc<crate::store::ECStore>,
|
||||
std::sync::Arc<crate::store::ECStore>,
|
||||
) {
|
||||
test_two_pool_stores_with_contexts(rebalance_meta, true).await
|
||||
test_pool_stores_with_contexts(rebalance_meta, true, 2, 2).await
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
async fn test_two_pool_stores_with_contexts(
|
||||
pub(crate) async fn promote_test_pool_meta_to_v2(store: &std::sync::Arc<crate::store::ECStore>) {
|
||||
let mut pool_meta = store.pool_meta.read().await.clone();
|
||||
pool_meta.version = crate::core::pools::POOL_META_VERSION;
|
||||
pool_meta
|
||||
.save(store.pools.clone())
|
||||
.await
|
||||
.expect("test pool metadata should be promoted to V2");
|
||||
*store.pool_meta.write().await = pool_meta;
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
pub(crate) async fn test_three_pool_stores_with_isolated_node_contexts(
|
||||
rebalance_meta: Option<RebalanceMeta>,
|
||||
) -> (
|
||||
Vec<tempfile::TempDir>,
|
||||
std::sync::Arc<crate::store::ECStore>,
|
||||
std::sync::Arc<crate::store::ECStore>,
|
||||
) {
|
||||
test_pool_stores_with_contexts(rebalance_meta, true, 3, 2).await
|
||||
}
|
||||
|
||||
#[cfg(all(test, feature = "test-util"))]
|
||||
pub(crate) async fn test_three_pool_stores_with_three_disk_sets_with_isolated_node_contexts(
|
||||
rebalance_meta: Option<RebalanceMeta>,
|
||||
) -> (
|
||||
Vec<tempfile::TempDir>,
|
||||
std::sync::Arc<crate::store::ECStore>,
|
||||
std::sync::Arc<crate::store::ECStore>,
|
||||
) {
|
||||
test_pool_stores_with_contexts(rebalance_meta, true, 3, 3).await
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
async fn test_pool_stores_with_contexts(
|
||||
rebalance_meta: Option<RebalanceMeta>,
|
||||
isolate_node_contexts: bool,
|
||||
pool_count: usize,
|
||||
set_drive_count: usize,
|
||||
) -> (
|
||||
Vec<tempfile::TempDir>,
|
||||
std::sync::Arc<crate::store::ECStore>,
|
||||
std::sync::Arc<crate::store::ECStore>,
|
||||
) {
|
||||
crate::services::notification_sys::install_cross_pool_fence_fleet_proof_for_test();
|
||||
use crate::core::pools::PoolMeta;
|
||||
use crate::core::pools::{POOL_META_VERSION, PoolMeta, PoolMetaWriteState, persist_pool_meta_identity_for_startup};
|
||||
use crate::layout::endpoints::{EndpointServerPools, SetupType};
|
||||
|
||||
let ctx = std::sync::Arc::new(crate::runtime::instance::InstanceContext::new());
|
||||
ctx.update_erasure_type(SetupType::DistErasure).await;
|
||||
let (mut temp_dirs, first_pool) =
|
||||
crate::core::sets::make_local_two_set_sets_for_pool_with_ctx(std::sync::Arc::clone(&ctx), 0).await;
|
||||
let (second_temp_dirs, second_pool) =
|
||||
crate::core::sets::make_local_two_set_sets_for_pool_with_ctx(std::sync::Arc::clone(&ctx), 1).await;
|
||||
temp_dirs.extend(second_temp_dirs);
|
||||
let pools = vec![first_pool, second_pool];
|
||||
let deployment_id = uuid::Uuid::new_v4();
|
||||
ctx.set_deployment_id(deployment_id);
|
||||
let mut temp_dirs = Vec::new();
|
||||
let mut pools = Vec::with_capacity(pool_count);
|
||||
for pool_index in 0..pool_count {
|
||||
let (pool_temp_dirs, pool) = crate::core::sets::make_local_two_set_sets_for_pool_with_drive_count_and_ctx(
|
||||
std::sync::Arc::clone(&ctx),
|
||||
pool_index,
|
||||
set_drive_count,
|
||||
)
|
||||
.await;
|
||||
temp_dirs.extend(pool_temp_dirs);
|
||||
pools.push(pool);
|
||||
}
|
||||
{
|
||||
let local_disk_map = ctx.local_disk_map();
|
||||
let mut local_disk_map = local_disk_map.write().await;
|
||||
@@ -143,11 +187,24 @@ async fn test_two_pool_stores_with_contexts(
|
||||
}
|
||||
}
|
||||
}
|
||||
let pool_meta = PoolMeta::new(&pools, &PoolMeta::default());
|
||||
let mut pool_meta = PoolMeta::new(&pools, &PoolMeta::default());
|
||||
pool_meta.version = POOL_META_VERSION;
|
||||
pool_meta
|
||||
.save_for_startup(pools.clone())
|
||||
.await
|
||||
.expect("baseline pool metadata should be persisted");
|
||||
let mut pool_meta_write_state = PoolMetaWriteState::for_startup(deployment_id, true);
|
||||
persist_pool_meta_identity_for_startup(pools.clone(), &mut pool_meta_write_state, false)
|
||||
.await
|
||||
.expect("pending pool metadata identity should be persisted");
|
||||
let replica_state = pool_meta
|
||||
.load_no_lock_from_replicas_observing(pools.clone(), &mut pool_meta_write_state)
|
||||
.await
|
||||
.expect("baseline pool metadata should remain readable");
|
||||
pool_meta_write_state.observe_replicas(replica_state);
|
||||
persist_pool_meta_identity_for_startup(pools.clone(), &mut pool_meta_write_state, true)
|
||||
.await
|
||||
.expect("initialized pool metadata identity should be persisted");
|
||||
if let Some(meta) = rebalance_meta.as_ref() {
|
||||
meta.save(pools[0].clone())
|
||||
.await
|
||||
@@ -158,6 +215,7 @@ async fn test_two_pool_stores_with_contexts(
|
||||
let other_ctx = if isolate_node_contexts {
|
||||
let other_ctx = std::sync::Arc::new(crate::runtime::instance::InstanceContext::new());
|
||||
other_ctx.update_erasure_type(SetupType::DistErasure).await;
|
||||
other_ctx.set_deployment_id(deployment_id);
|
||||
*other_ctx.local_disk_map().write().await = ctx.local_disk_map().read().await.clone();
|
||||
other_ctx.set_endpoints(endpoint_pools.clone());
|
||||
other_ctx
|
||||
@@ -172,9 +230,10 @@ async fn test_two_pool_stores_with_contexts(
|
||||
peer_sys: crate::cluster::rpc::S3PeerSys::new_with_instance_ctx(&endpoint_pools, std::sync::Arc::clone(&store_ctx)),
|
||||
pool_meta: tokio::sync::RwLock::new(pool_meta.clone()),
|
||||
rebalance_meta: tokio::sync::RwLock::new(rebalance_meta.clone()),
|
||||
decommission_cancelers: tokio::sync::RwLock::new(vec![None, None]),
|
||||
decommission_cancelers: tokio::sync::RwLock::new(vec![None; pool_count]),
|
||||
start_gate: tokio::sync::Mutex::new(()),
|
||||
pool_meta_save_gate: tokio::sync::Mutex::default(),
|
||||
pool_meta_save_gate: tokio::sync::Mutex::new(pool_meta_write_state.independent_clone_for_test()),
|
||||
decommission_capacity_entry_gate: tokio::sync::Mutex::default(),
|
||||
ctx: store_ctx,
|
||||
bucket_fence_registry: std::sync::Arc::default(),
|
||||
})
|
||||
@@ -184,6 +243,8 @@ async fn test_two_pool_stores_with_contexts(
|
||||
if isolate_node_contexts {
|
||||
crate::bucket::metadata_sys::init_bucket_metadata_sys(std::sync::Arc::clone(&store), Vec::new()).await;
|
||||
crate::bucket::metadata_sys::init_bucket_metadata_sys(std::sync::Arc::clone(&other_store), Vec::new()).await;
|
||||
} else {
|
||||
crate::bucket::metadata_sys::init_bucket_metadata_sys(std::sync::Arc::clone(&store), Vec::new()).await;
|
||||
}
|
||||
(temp_dirs, store, other_store)
|
||||
}
|
||||
|
||||
@@ -3009,6 +3009,7 @@ fn test_store_with_rebalance_meta(meta: RebalanceMeta) -> Arc<crate::store::ECSt
|
||||
decommission_cancelers: tokio::sync::RwLock::new(Vec::new()),
|
||||
start_gate: tokio::sync::Mutex::new(()),
|
||||
pool_meta_save_gate: tokio::sync::Mutex::default(),
|
||||
decommission_capacity_entry_gate: tokio::sync::Mutex::default(),
|
||||
ctx: crate::runtime::instance::bootstrap_ctx(),
|
||||
bucket_fence_registry: std::sync::Arc::default(),
|
||||
})
|
||||
|
||||
@@ -161,6 +161,7 @@ impl ECStore {
|
||||
|
||||
let cancel_tx = CancellationToken::new();
|
||||
let rx = cancel_tx.clone();
|
||||
let activation_at = self.next_scanner_data_movement_update(OffsetDateTime::now_utc()).await;
|
||||
let activation_outcome;
|
||||
let candidate;
|
||||
let expected_cancel;
|
||||
@@ -185,12 +186,8 @@ impl ECStore {
|
||||
return Ok(false);
|
||||
}
|
||||
expected_cancel = meta.cancel.clone();
|
||||
(candidate, activation_outcome, must_persist) = stage_local_rebalance_worker_activation(
|
||||
meta,
|
||||
expected_id.as_ref(),
|
||||
cancel_tx.clone(),
|
||||
OffsetDateTime::now_utc(),
|
||||
)?;
|
||||
(candidate, activation_outcome, must_persist) =
|
||||
stage_local_rebalance_worker_activation(meta, expected_id.as_ref(), cancel_tx.clone(), activation_at)?;
|
||||
if let Err(err) = activation_fence.ensure_held() {
|
||||
cancel_tx.cancel();
|
||||
return Err(err);
|
||||
@@ -384,11 +381,11 @@ impl ECStore {
|
||||
tokio::select! {
|
||||
result = done_rx.recv() => {
|
||||
quit = true;
|
||||
let now = OffsetDateTime::now_utc();
|
||||
let terminal_event = classify_rebalance_terminal_event(result, now);
|
||||
msg = terminal_event.message().to_string();
|
||||
let movement_gate = store.ctx.data_movement_operation_gate();
|
||||
let movement_guard = movement_gate.write().await;
|
||||
let terminal_at = store.next_scanner_data_movement_update(OffsetDateTime::now_utc()).await;
|
||||
let terminal_event = classify_rebalance_terminal_event(result, terminal_at);
|
||||
msg = terminal_event.message().to_string();
|
||||
let previous_meta = store.rebalance_meta.read().await.clone();
|
||||
let terminal_state_present = {
|
||||
let mut rebalance_meta = store.rebalance_meta.write().await;
|
||||
@@ -405,7 +402,7 @@ impl ECStore {
|
||||
{
|
||||
pool_stat.info.stopping = false;
|
||||
pool_stat.info.status = RebalStatus::Failed;
|
||||
pool_stat.info.end_time = Some(now);
|
||||
pool_stat.info.end_time = Some(terminal_at);
|
||||
pool_stat.info.last_error = Some(
|
||||
pool_stat
|
||||
.cleanup_warnings
|
||||
@@ -433,7 +430,7 @@ impl ECStore {
|
||||
&mut pool_stat.info.end_time,
|
||||
&mut pool_stat.info.last_error,
|
||||
terminal_event,
|
||||
now,
|
||||
terminal_at,
|
||||
);
|
||||
}
|
||||
true
|
||||
@@ -835,6 +832,10 @@ impl ECStore {
|
||||
opt: RebalSaveOpt,
|
||||
expected_id: Option<&str>,
|
||||
) -> Result<()> {
|
||||
let now = match opt {
|
||||
RebalSaveOpt::Stats => OffsetDateTime::now_utc(),
|
||||
RebalSaveOpt::StoppedAt => self.next_scanner_data_movement_update(OffsetDateTime::now_utc()).await,
|
||||
};
|
||||
let meta_to_save = {
|
||||
let mut rebalance_meta = self.rebalance_meta.write().await;
|
||||
if let Some(expected_id) = expected_id {
|
||||
@@ -844,7 +845,6 @@ impl ECStore {
|
||||
return Ok(());
|
||||
};
|
||||
|
||||
let now = OffsetDateTime::now_utc();
|
||||
apply_rebalance_save_option(meta, pool_idx, opt, now);
|
||||
meta.clone()
|
||||
};
|
||||
|
||||
@@ -12,7 +12,7 @@
|
||||
// See the License for the specific language governing permissions and
|
||||
// limitations under the License.
|
||||
|
||||
#[cfg(feature = "test-util")]
|
||||
#[cfg(any(test, feature = "test-util"))]
|
||||
pub mod test_util;
|
||||
pub mod tier;
|
||||
pub mod tier_admin;
|
||||
|
||||
File diff suppressed because it is too large
Load Diff
@@ -12,7 +12,7 @@
|
||||
// See the License for the specific language governing permissions and
|
||||
// limitations under the License.
|
||||
|
||||
use std::sync::Arc;
|
||||
use std::sync::{Arc, LazyLock};
|
||||
|
||||
use rustfs_utils::crypto::{hex_sha256, is_sha256_checksum};
|
||||
use serde::{Deserialize, Serialize};
|
||||
@@ -32,9 +32,34 @@ pub(crate) const TIER_MUTATION_INTENT_SCHEMA: &str = "rustfs-tier-mutation-inten
|
||||
pub(crate) const MAX_TIER_MUTATION_INTENT_SIZE: usize = rustfs_protos::TIER_MUTATION_RPC_MAX_PREPARE_PAYLOAD_SIZE;
|
||||
pub(crate) const TIER_MUTATION_INTENT_RECORD_PREFIX: &str = "tier/mutation-intents/records";
|
||||
pub(crate) const TIER_COORDINATOR_MUTATION_INTENT_RECORD_PREFIX: &str = "tier/mutation-intents/coordinators";
|
||||
pub(crate) const TIER_MUTATION_MUTEX_SHARDS: usize = 64;
|
||||
const TIER_MUTATION_INTENT_ADVANCE_CAS_ATTEMPTS: usize = 3;
|
||||
pub(crate) type TierMutationDigest = [u8; 32];
|
||||
|
||||
static TIER_MUTATION_MUTEXES: LazyLock<[tokio::sync::Mutex<()>; TIER_MUTATION_MUTEX_SHARDS]> =
|
||||
LazyLock::new(|| std::array::from_fn(|_| tokio::sync::Mutex::new(())));
|
||||
|
||||
/// Serializes every local phase and recovery action for one mutation id while
|
||||
/// retaining bounded parallelism for unrelated mutations.
|
||||
pub(crate) async fn acquire_tier_mutation_mutex(mutation_id: Uuid) -> tokio::sync::MutexGuard<'static, ()> {
|
||||
TIER_MUTATION_MUTEXES[tier_mutation_mutex_shard_index(mutation_id)]
|
||||
.lock()
|
||||
.await
|
||||
}
|
||||
|
||||
fn tier_mutation_mutex_shard_index(mutation_id: Uuid) -> usize {
|
||||
let raw = mutation_id.as_u128();
|
||||
let mut mixed = (raw as u64) ^ ((raw >> 64) as u64);
|
||||
// MurmurHash3's 64-bit finalizer gives stable diffusion without allocating
|
||||
// or relying on RandomState, whose seed differs between processes.
|
||||
mixed ^= mixed >> 33;
|
||||
mixed = mixed.wrapping_mul(0xff51_afd7_ed55_8ccd);
|
||||
mixed ^= mixed >> 33;
|
||||
mixed = mixed.wrapping_mul(0xc4ce_b9fe_1a85_ec53);
|
||||
mixed ^= mixed >> 33;
|
||||
(mixed as usize) & (TIER_MUTATION_MUTEX_SHARDS - 1)
|
||||
}
|
||||
|
||||
pub(crate) type Result<T> = std::result::Result<T, TierMutationIntentError>;
|
||||
|
||||
#[derive(Debug, thiserror::Error)]
|
||||
@@ -265,6 +290,30 @@ impl TierMutationIntent {
|
||||
&& self.expires_at_unix_nanos == other.expires_at_unix_nanos
|
||||
}
|
||||
|
||||
/// Reconstruct the exact Prepared record that originally produced this
|
||||
/// intent. Abort RPCs are identity-bound to that payload; serializing an
|
||||
/// Aborted terminal record would both violate the wire contract and use a
|
||||
/// different revision if a missing peer has to persist a tombstone.
|
||||
pub(crate) fn original_prepared(&self) -> Result<Self> {
|
||||
if self.state == TierMutationIntentState::Prepared {
|
||||
self.validate()?;
|
||||
return Ok(self.clone());
|
||||
}
|
||||
let mut prepared = self.clone();
|
||||
prepared.revision =
|
||||
prepared
|
||||
.revision
|
||||
.checked_sub(1)
|
||||
.filter(|revision| *revision != 0)
|
||||
.ok_or(TierMutationIntentError::Corrupt(
|
||||
"terminal intent cannot reconstruct its prepared revision",
|
||||
))?;
|
||||
prepared.state = TierMutationIntentState::Prepared;
|
||||
prepared.committed_config_etag = None;
|
||||
prepared.validate()?;
|
||||
Ok(prepared)
|
||||
}
|
||||
|
||||
pub(crate) fn encode(&self) -> Result<Vec<u8>> {
|
||||
self.validate()?;
|
||||
let intent_bytes = serde_json::to_vec(self)?;
|
||||
@@ -439,6 +488,16 @@ where
|
||||
load_tier_mutation_intent_record_with_etag_at_prefix(api, TIER_MUTATION_INTENT_RECORD_PREFIX, mutation_id).await
|
||||
}
|
||||
|
||||
pub(crate) async fn load_tier_coordinator_mutation_intent_record_with_etag<S>(
|
||||
api: Arc<S>,
|
||||
mutation_id: Uuid,
|
||||
) -> EcstoreResult<(TierMutationIntent, String)>
|
||||
where
|
||||
S: EcstoreObjectIO,
|
||||
{
|
||||
load_tier_mutation_intent_record_with_etag_at_prefix(api, TIER_COORDINATOR_MUTATION_INTENT_RECORD_PREFIX, mutation_id).await
|
||||
}
|
||||
|
||||
async fn load_tier_mutation_intent_record_with_etag_at_prefix<S>(
|
||||
api: Arc<S>,
|
||||
prefix: &str,
|
||||
@@ -507,6 +566,7 @@ where
|
||||
.await
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
pub(crate) async fn delete_tier_mutation_intent_record<S>(api: Arc<S>, mutation_id: Uuid) -> EcstoreResult<()>
|
||||
where
|
||||
S: EcstoreObjectOperations,
|
||||
@@ -514,13 +574,36 @@ where
|
||||
delete_tier_mutation_intent_record_with_prefix(api, TIER_MUTATION_INTENT_RECORD_PREFIX, mutation_id).await
|
||||
}
|
||||
|
||||
pub(crate) async fn delete_tier_coordinator_mutation_intent_record<S>(api: Arc<S>, mutation_id: Uuid) -> EcstoreResult<()>
|
||||
pub(crate) async fn delete_tier_mutation_intent_record_if_current<S>(
|
||||
api: Arc<S>,
|
||||
mutation_id: Uuid,
|
||||
current_etag: &str,
|
||||
) -> EcstoreResult<()>
|
||||
where
|
||||
S: EcstoreObjectOperations,
|
||||
{
|
||||
delete_tier_mutation_intent_record_with_prefix(api, TIER_COORDINATOR_MUTATION_INTENT_RECORD_PREFIX, mutation_id).await
|
||||
delete_tier_mutation_intent_record_if_current_with_prefix(api, TIER_MUTATION_INTENT_RECORD_PREFIX, mutation_id, current_etag)
|
||||
.await
|
||||
}
|
||||
|
||||
pub(crate) async fn delete_tier_coordinator_mutation_intent_record_if_current<S>(
|
||||
api: Arc<S>,
|
||||
mutation_id: Uuid,
|
||||
current_etag: &str,
|
||||
) -> EcstoreResult<()>
|
||||
where
|
||||
S: EcstoreObjectOperations,
|
||||
{
|
||||
delete_tier_mutation_intent_record_if_current_with_prefix(
|
||||
api,
|
||||
TIER_COORDINATOR_MUTATION_INTENT_RECORD_PREFIX,
|
||||
mutation_id,
|
||||
current_etag,
|
||||
)
|
||||
.await
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
async fn delete_tier_mutation_intent_record_with_prefix<S>(api: Arc<S>, prefix: &str, mutation_id: Uuid) -> EcstoreResult<()>
|
||||
where
|
||||
S: EcstoreObjectOperations,
|
||||
@@ -533,6 +616,40 @@ where
|
||||
}
|
||||
}
|
||||
|
||||
async fn delete_tier_mutation_intent_record_if_current_with_prefix<S>(
|
||||
api: Arc<S>,
|
||||
prefix: &str,
|
||||
mutation_id: Uuid,
|
||||
current_etag: &str,
|
||||
) -> EcstoreResult<()>
|
||||
where
|
||||
S: EcstoreObjectOperations,
|
||||
{
|
||||
if current_etag.trim().is_empty() {
|
||||
return Err(Error::other("tier mutation intent current ETag is empty"));
|
||||
}
|
||||
let object =
|
||||
tier_mutation_intent_record_object_name_with_prefix(prefix, mutation_id).map_err(tier_mutation_intent_store_error)?;
|
||||
match api
|
||||
.delete_object(
|
||||
RUSTFS_META_BUCKET,
|
||||
&object,
|
||||
ObjectOptions {
|
||||
http_preconditions: Some(HTTPPreconditions {
|
||||
if_match: Some(current_etag.to_string()),
|
||||
..Default::default()
|
||||
}),
|
||||
..Default::default()
|
||||
},
|
||||
)
|
||||
.await
|
||||
{
|
||||
Ok(_) => Ok(()),
|
||||
Err(err) if err == Error::FileNotFound || matches!(err, Error::ObjectNotFound(_, _)) => Err(Error::ConfigNotFound),
|
||||
Err(err) => Err(err),
|
||||
}
|
||||
}
|
||||
|
||||
pub(crate) async fn advance_tier_mutation_intent_record_idempotent<S>(
|
||||
api: Arc<S>,
|
||||
mutation_id: Uuid,
|
||||
@@ -713,6 +830,7 @@ fn digest_is_empty(digest: &TierMutationDigest) -> bool {
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use super::*;
|
||||
use std::time::Duration;
|
||||
|
||||
const OLD_IDENTITY: TierDestinationId = [1; 32];
|
||||
const NEW_IDENTITY: TierDestinationId = [2; 32];
|
||||
@@ -736,6 +854,61 @@ mod tests {
|
||||
}
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn mutation_mutex_uses_exactly_64_stable_shards() {
|
||||
assert_eq!(TIER_MUTATION_MUTEX_SHARDS, 64);
|
||||
assert_eq!(TIER_MUTATION_MUTEXES.len(), TIER_MUTATION_MUTEX_SHARDS);
|
||||
|
||||
let mutation_id = Uuid::parse_str("36e2220e-9ad2-495b-b3bc-c4d2caf70a31").expect("fixture uuid should parse");
|
||||
let shard = tier_mutation_mutex_shard_index(mutation_id);
|
||||
assert!(shard < TIER_MUTATION_MUTEX_SHARDS);
|
||||
assert_eq!(shard, tier_mutation_mutex_shard_index(mutation_id));
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
async fn mutation_mutex_serializes_the_same_id() {
|
||||
let mutation_id = Uuid::new_v4();
|
||||
let first = acquire_tier_mutation_mutex(mutation_id).await;
|
||||
let (started_tx, started_rx) = tokio::sync::oneshot::channel();
|
||||
let (acquired_tx, mut acquired_rx) = tokio::sync::oneshot::channel();
|
||||
|
||||
let waiter = tokio::spawn(async move {
|
||||
started_tx.send(()).expect("test receiver should remain alive");
|
||||
let _second = acquire_tier_mutation_mutex(mutation_id).await;
|
||||
acquired_tx.send(()).expect("test receiver should remain alive");
|
||||
});
|
||||
started_rx.await.expect("waiter should start");
|
||||
assert!(
|
||||
tokio::time::timeout(Duration::from_millis(25), &mut acquired_rx)
|
||||
.await
|
||||
.is_err(),
|
||||
"the same mutation id must not enter concurrently"
|
||||
);
|
||||
|
||||
drop(first);
|
||||
tokio::time::timeout(Duration::from_secs(1), &mut acquired_rx)
|
||||
.await
|
||||
.expect("waiter should acquire after release")
|
||||
.expect("waiter should report acquisition");
|
||||
waiter.await.expect("waiter task should finish");
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
async fn mutation_mutex_allows_different_shards_to_progress() {
|
||||
let first_id = Uuid::new_v4();
|
||||
let first_shard = tier_mutation_mutex_shard_index(first_id);
|
||||
let second_id = (0..1024)
|
||||
.map(|_| Uuid::new_v4())
|
||||
.find(|candidate| tier_mutation_mutex_shard_index(*candidate) != first_shard)
|
||||
.expect("a distinct shard should be easy to find");
|
||||
let first = acquire_tier_mutation_mutex(first_id).await;
|
||||
|
||||
let _second = tokio::time::timeout(Duration::from_secs(1), acquire_tier_mutation_mutex(second_id))
|
||||
.await
|
||||
.expect("a different shard must not wait for the first mutation");
|
||||
drop(first);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn intent_round_trip_preserves_committed_state() {
|
||||
let mut intent = prepared_intent();
|
||||
@@ -873,6 +1046,37 @@ mod tests {
|
||||
));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn terminal_intent_reconstructs_original_prepared_abort_payload() {
|
||||
for terminal in [TierMutationIntentState::Aborted, TierMutationIntentState::Committed] {
|
||||
let original = prepared_intent();
|
||||
let mut intent = original.clone();
|
||||
let committed_etag = (terminal == TierMutationIntentState::Committed).then(|| "new-etag".to_string());
|
||||
intent
|
||||
.advance(terminal, committed_etag)
|
||||
.expect("terminal transition should succeed");
|
||||
|
||||
let reconstructed = intent
|
||||
.original_prepared()
|
||||
.expect("terminal record should recover prepared payload");
|
||||
assert_eq!(reconstructed, original);
|
||||
assert!(intent.same_identity_as(&reconstructed));
|
||||
}
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn terminal_intent_with_initial_revision_fails_prepared_reconstruction() {
|
||||
let mut corrupt = prepared_intent();
|
||||
corrupt.state = TierMutationIntentState::Aborted;
|
||||
|
||||
assert!(matches!(
|
||||
corrupt.original_prepared(),
|
||||
Err(TierMutationIntentError::Corrupt(
|
||||
"terminal intent cannot reconstruct its prepared revision"
|
||||
))
|
||||
));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn intent_validation_rejects_placeholder_identity() {
|
||||
let mut intent = prepared_intent();
|
||||
|
||||
Some files were not shown because too many files have changed in this diff Show More
Reference in New Issue
Block a user