From d60dfbb826792021f6fc9e7dd09e26a7994aba4b Mon Sep 17 00:00:00 2001 From: Hauser Date: Wed, 30 Sep 2026 14:02:32 +0800 Subject: [PATCH] fix(heal): finalize durable MRF legacy responsibility lifecycle (#8254) ## Related Issues Resolves the corrected lifecycle fixture findings in PR #8254. ## Summary of Changes Preserve separate incarnation-bound durable MRF responsibilities and their lifecycle audit records, and align replay fixtures with their checkpoint identities. ## Verification Two independent source reviews and changed-delta reviews are complete. The final review approved 607a0b9bb0b2169c50dda4d00aa055f821c385ea with no remaining supported findings. Local runtime claims were not independently reproduced for this pull request. ## Impact Keeps storage generation fences and retained responsibility semantics. Test-capacity reservations preserve existing deadlines and assertions. ## Additional Notes Squash merge of the currently approved fix under the authorized CI-bypass exception. Main CI and release acceptance remain required. --- .config/nextest.toml | 18 + crates/common/src/mrf_channel.rs | 39 +- crates/ecstore/src/set_disk/mod.rs | 27 +- crates/ecstore/src/set_disk/ops/object.rs | 176 ++- crates/heal-contracts/src/heal_channel.rs | 6 + crates/heal/src/heal/channel.rs | 38 + crates/heal/src/heal/manager.rs | 2 + crates/heal/src/heal/mrf_queue.rs | 825 ++++++++++- .../heal/src/heal/mrf_queue/partial_write.rs | 1285 ++++++++++++++++- crates/heal/src/heal/mrf_queue/snapshot.rs | 174 +++ crates/heal/src/heal/task.rs | 8 + crates/heal/src/heal/task/heal_object.rs | 12 +- crates/heal/src/heal/task/tests.rs | 20 + crates/heal/tests/mrf_partial_write_test.rs | 352 ++++- crates/protos/src/heal_control.rs | 1 + docs/architecture/compat-cleanup-register.md | 1 + docs/operations/scanner-runtime-controls.md | 30 +- rustfs/src/admin/handlers/heal.rs | 368 ++++- rustfs/src/admin/route_policy.rs | 12 + rustfs/src/admin/route_registration_test.rs | 6 + .../src/app/object/on_demand_migration_put.rs | 60 +- 21 files changed, 3226 insertions(+), 234 deletions(-) diff --git a/.config/nextest.toml b/.config/nextest.toml index b75177486..37c8f3cfb 100644 --- a/.config/nextest.toml +++ b/.config/nextest.toml @@ -242,6 +242,13 @@ test-group = 'e2e-cluster-nightly' filter = 'package(e2e_test) & (test(/^kms::kms_vault_test::/) | test(/^kms::kms_rekey_sweep_test::/) | test(/^kms::configured_roundtrip_test::test_configured_vault_kms_admin_and_versioned_cleanup$/))' test-group = 'e2e-vault' +# These server-heavy smoke cases perform 1k+ sequential S3 writes or run the +# scanner and a remote tier server together. Reserve the full test budget so +# unrelated E2E processes cannot starve their storage and polling deadlines. +[[profile.default.overrides]] +filter = 'package(e2e_test) & (test(/^list_objects_v2_pagination_test::tests::(test_list_objects_v2_delimiter_small_page_traverses_all|test_list_objects_v2_max_keys_above_limit_returns_token|test_list_objects_v2_maxkeys_above_limit_with_delimiter)$/) | test(=reliant::tiering::test_hermetic_transition_restore_failure_expiry_and_retry))' +threads-required = "num-test-threads" + # This four-disk, 65-member rollback probe already drives up to 32 concurrent # durable deletions. Reserve this nextest run's capacity for its progress oracle. [[profile.default.overrides]] @@ -525,6 +532,11 @@ path = "junit.xml" [[profile.e2e-smoke.overrides]] filter = 'package(e2e_test) & test(/^list_objects_v2_pagination_test::tests::(test_list_objects_v2_delimiter_small_page_traverses_all|test_list_objects_v2_max_keys_above_limit_returns_token|test_list_objects_v2_maxkeys_above_limit_with_delimiter)$/)' slow-timeout = { period = "60s", terminate-after = 2, grace-period = "10s" } +threads-required = "num-test-threads" + +[[profile.e2e-smoke.overrides]] +filter = 'package(e2e_test) & test(=reliant::tiering::test_hermetic_transition_restore_failure_expiry_and_retry)' +threads-required = "num-test-threads" # --------------------------------------------------------------------------- # e2e-repl-nightly profile — scheduled full replication e2e lane (repl-1) @@ -763,3 +775,9 @@ test-group = 'e2e-cluster-nightly' [[profile.e2e-full.overrides]] filter = 'package(e2e_test) & (test(/^kms::kms_vault_test::/) | test(/^kms::kms_rekey_sweep_test::/) | test(/^kms::configured_roundtrip_test::test_configured_vault_kms_admin_and_versioned_cleanup$/))' test-group = 'e2e-vault' + +# Keep the storage-heavy smoke cases isolated in the full suite too. They +# start multiple server fixtures and issue many sequential S3 operations. +[[profile.e2e-full.overrides]] +filter = 'package(e2e_test) & (test(/^list_objects_v2_pagination_test::tests::(test_list_objects_v2_delimiter_small_page_traverses_all|test_list_objects_v2_max_keys_above_limit_returns_token|test_list_objects_v2_maxkeys_above_limit_with_delimiter)$/) | test(=reliant::tiering::test_hermetic_transition_restore_failure_expiry_and_retry))' +threads-required = "num-test-threads" diff --git a/crates/common/src/mrf_channel.rs b/crates/common/src/mrf_channel.rs index ddacce484..22b48c95a 100644 --- a/crates/common/src/mrf_channel.rs +++ b/crates/common/src/mrf_channel.rs @@ -333,6 +333,9 @@ pub enum MrfDurableAdmissionError { /// cancel repair. pub struct MrfDurableSubmission { pub intent: MrfIntent, + /// Bucket incarnation observed by the committed-write caller, when + /// available. Old journal records and legacy callers remain unbound. + pub source_bucket_incarnation_id: Option, pub response: oneshot::Sender>, } @@ -353,6 +356,19 @@ pub async fn persist_partial_write_intent( version_id: Option, scope: MrfScope, ) -> Result<(), MrfDurableAdmissionError> { + persist_partial_write_intent_with_incarnation(bucket, object, version_id, scope, None).await +} + +pub async fn persist_partial_write_intent_with_incarnation( + bucket: &str, + object: &str, + version_id: Option, + scope: MrfScope, + source_bucket_incarnation_id: Option, +) -> Result<(), MrfDurableAdmissionError> { + if source_bucket_incarnation_id.is_some_and(|incarnation| incarnation.is_nil()) { + return Err(MrfDurableAdmissionError::InvalidIdentity); + } let intent = MrfIntent { bucket: Arc::from(bucket), object: Arc::from(object), @@ -364,7 +380,7 @@ pub async fn persist_partial_write_intent( enqueued_at_ms: unix_now_ms(), attempts: 0, }; - persist_durable_intent(intent).await + persist_durable_intent(intent, source_bucket_incarnation_id).await } pub async fn persist_delete_marker_purge_intent( @@ -377,6 +393,7 @@ pub async fn persist_delete_marker_purge_intent( if version_id.is_nil() { return Err(MrfDurableAdmissionError::InvalidIdentity); } + let source_bucket_incarnation_id = delete_marker_purge.bucket_incarnation_id; let intent = MrfIntent { bucket: Arc::from(bucket), object: Arc::from(object), @@ -388,17 +405,23 @@ pub async fn persist_delete_marker_purge_intent( enqueued_at_ms: unix_now_ms(), attempts: 0, }; - persist_durable_intent_unconditionally(intent).await + persist_durable_intent_unconditionally(intent, Some(source_bucket_incarnation_id)).await } -async fn persist_durable_intent(intent: MrfIntent) -> Result<(), MrfDurableAdmissionError> { +async fn persist_durable_intent( + intent: MrfIntent, + source_bucket_incarnation_id: Option, +) -> Result<(), MrfDurableAdmissionError> { if !mrf_delivery_enabled() { return Err(MrfDurableAdmissionError::Disabled); } - persist_durable_intent_unconditionally(intent).await + persist_durable_intent_unconditionally(intent, source_bucket_incarnation_id).await } -async fn persist_durable_intent_unconditionally(mut intent: MrfIntent) -> Result<(), MrfDurableAdmissionError> { +async fn persist_durable_intent_unconditionally( + mut intent: MrfIntent, + source_bucket_incarnation_id: Option, +) -> Result<(), MrfDurableAdmissionError> { if intent.bucket.is_empty() || intent.object.is_empty() || intent.bucket.len() > MRF_MAX_IDENTITY_COMPONENT @@ -413,7 +436,11 @@ async fn persist_durable_intent_unconditionally(mut intent: MrfIntent) -> Result } let (response, receipt) = oneshot::channel(); sender - .try_send(MrfDurableSubmission { intent, response }) + .try_send(MrfDurableSubmission { + intent, + source_bucket_incarnation_id, + response, + }) .map_err(|err| match err { mpsc::error::TrySendError::Full(_) => MrfDurableAdmissionError::Full, mpsc::error::TrySendError::Closed(_) => MrfDurableAdmissionError::Unavailable, diff --git a/crates/ecstore/src/set_disk/mod.rs b/crates/ecstore/src/set_disk/mod.rs index fb18239a8..ac2892bff 100644 --- a/crates/ecstore/src/set_disk/mod.rs +++ b/crates/ecstore/src/set_disk/mod.rs @@ -4339,9 +4339,15 @@ impl SetDisks { } } - pub(in crate::set_disk) async fn persist_partial_write(&self, bucket: &str, object: &str, version_id: Option<&str>) -> bool { + pub(in crate::set_disk) async fn persist_partial_write( + &self, + bucket: &str, + object: &str, + version_id: Option<&str>, + source_bucket_incarnation_id: Option, + ) -> bool { use rustfs_common::mrf_channel::{ - MrfDurableAdmissionError, MrfScope, mrf_delivery_enabled, persist_partial_write_intent, + MrfDurableAdmissionError, MrfScope, mrf_delivery_enabled, persist_partial_write_intent_with_incarnation, }; if !mrf_delivery_enabled() { @@ -4360,7 +4366,9 @@ impl SetDisks { Ok::<_, MrfDurableAdmissionError>((version, scope)) })(); let result = match identity { - Ok((version, scope)) => persist_partial_write_intent(bucket, object, version, scope).await, + Ok((version, scope)) => { + persist_partial_write_intent_with_incarnation(bucket, object, version, scope, source_bucket_incarnation_id).await + } Err(err) => Err(err), }; match result { @@ -4396,15 +4404,24 @@ impl SetDisks { pub(in crate::set_disk) async fn submit_rename_tail_heal( &self, - request: rustfs_heal_contracts::heal_channel::HealChannelRequest, + mut request: rustfs_heal_contracts::heal_channel::HealChannelRequest, ) { if let Some(object) = request.object_prefix.as_deref() && self - .persist_partial_write(&request.bucket, object, request.object_version_id.as_deref()) + .persist_partial_write( + &request.bucket, + object, + request.object_version_id.as_deref(), + request.expected_bucket_incarnation_id, + ) .await { return; } + if request.expected_bucket_incarnation_id.is_none() { + return; + } + request.source = rustfs_heal_contracts::heal_channel::HealRequestSource::Mrf; #[cfg(test)] { let capture = self diff --git a/crates/ecstore/src/set_disk/ops/object.rs b/crates/ecstore/src/set_disk/ops/object.rs index e4cc784f9..720b1e2ab 100644 --- a/crates/ecstore/src/set_disk/ops/object.rs +++ b/crates/ecstore/src/set_disk/ops/object.rs @@ -3500,6 +3500,10 @@ impl SetDisks { mut publication_fence: Option, ) -> Result<(ObjectInfo, Option)> { let protect_write = opts.shard_integrity_write_enabled(); + let source_bucket_incarnation_id = match opts.expected_bucket_incarnation_id { + Some(incarnation_id) => Some(incarnation_id), + None => self.bucket_incarnation_id_from_disk(bucket).await.ok(), + }; if publication_fence.is_none() && opts.data_movement && rustfs_utils::http::metadata_compat::contains_key_str( @@ -4031,7 +4035,7 @@ impl SetDisks { { #[cfg(any(test, feature = "test-util"))] pause_put_object_commit(bucket, object, PutObjectCommitPause::BeforeNamespace).await; - if let Some(expected_incarnation_id) = opts.expected_bucket_incarnation_id + if let Some(expected_incarnation_id) = source_bucket_incarnation_id && opts.bucket_lifecycle_lock_fence.is_none() { bucket_lifecycle_guard = Some( @@ -4642,6 +4646,7 @@ impl SetDisks { request.object_version_id = committed_version_id .or_else(|| commit_version_suspended.then(Uuid::nil)) .map(|version_id| version_id.to_string()); + request.expected_bucket_incarnation_id = source_bucket_incarnation_id; let object_lock_guard = _object_lock_guard.take(); let publication_guard = _publication_guard.take(); let bucket_lifecycle_guard = _bucket_lifecycle_guard.take(); @@ -4790,6 +4795,7 @@ impl SetDisks { request.object_version_id = committed_version_id .or_else(|| commit_version_suspended.then(Uuid::nil)) .map(|version_id| version_id.to_string()); + request.expected_bucket_incarnation_id = source_bucket_incarnation_id; commit_set.submit_rename_tail_heal(request).await; } @@ -7492,8 +7498,15 @@ impl SetDisks { ensure_delete_commit_locks_held(None, bucket, &encoded_object, opts)?; authorization.authorized_journal_name(&candidate)?; begin_scanner_publication_delete_mutation(opts.scanner_publication_commit_scope.as_ref())?; - self.delete_object_version(bucket, &encoded_object, &delete_request, false) - .await?; + self.delete_object_version_with_purge( + bucket, + &encoded_object, + &delete_request, + false, + None, + opts.expected_bucket_incarnation_id, + ) + .await?; if let Some((_, deleted_object)) = replication_delete { ReplicationLifecycleBridge::schedule_delete(bucket.to_string(), deleted_object).await; } @@ -7573,6 +7586,7 @@ impl SetDisks { fi: &FileInfo, force_del_marker: bool, delete_marker_purge: Option, + source_bucket_incarnation_id: Option, ) -> Result<()> { let transported = delete_file_info_with_replication_transport_metadata(fi); let fi = &transported; @@ -7727,13 +7741,55 @@ impl SetDisks { { let version_id = fi.version_id.map(|version| version.to_string()); let _ = self - .add_partial(bucket, object, version_id.as_deref().unwrap_or_default()) + .add_partial_with_source_incarnation( + bucket, + object, + version_id.as_deref().unwrap_or_default(), + source_bucket_incarnation_id, + ) .await; } quorum_result } } +impl SetDisks { + async fn add_partial_with_source_incarnation( + &self, + bucket: &str, + object: &str, + version_id: &str, + source_bucket_incarnation_id: Option, + ) -> Result<()> { + if self + .persist_partial_write(bucket, object, Some(version_id), source_bucket_incarnation_id) + .await + { + return Ok(()); + } + let Some(source_bucket_incarnation_id) = source_bucket_incarnation_id else { + // The fallback is best-effort. Without the source generation it + // cannot safely target the object name after bucket recreation. + return Ok(()); + }; + let mut request = rustfs_heal_contracts::heal_channel::create_heal_request_with_options( + bucket.to_string(), + Some(object.to_string()), + false, + Some(HealChannelPriority::Normal), + Some(self.pool_index), + Some(self.set_index), + ); + request.object_version_id = (!version_id.is_empty()).then(|| version_id.to_string()); + request.source = rustfs_heal_contracts::heal_channel::HealRequestSource::Mrf; + request.expected_bucket_incarnation_id = Some(source_bucket_incarnation_id); + if let Err(error) = rustfs_heal_contracts::heal_channel::send_heal_request(request).await { + warn!(bucket, object, version_id, error = %error, "Failed to enqueue heal request for partial object"); + } + Ok(()) + } +} + #[async_trait::async_trait] impl crate::storage_api_contracts::object::ObjectOperations for SetDisks { type Error = Error; @@ -8021,7 +8077,8 @@ impl crate::storage_api_contracts::object::ObjectOperations for SetDisks { } #[tracing::instrument(skip(self))] async fn delete_object_version(&self, bucket: &str, object: &str, fi: &FileInfo, force_del_marker: bool) -> Result<()> { - self.delete_object_version_with_purge(bucket, object, fi, force_del_marker, None) + let source_bucket_incarnation_id = self.bucket_incarnation_id_from_disk(bucket).await.ok(); + self.delete_object_version_with_purge(bucket, object, fi, force_del_marker, None, source_bucket_incarnation_id) .await } @@ -8843,7 +8900,15 @@ impl crate::storage_api_contracts::object::ObjectOperations for SetDisks { delete_request.set_skip_tier_free_version(); } begin_scanner_publication_delete_mutation(scanner_publication_commit_scope.as_ref())?; - self.delete_object_version(bucket, object, &delete_request, false).await?; + self.delete_object_version_with_purge( + bucket, + object, + &delete_request, + false, + None, + opts.expected_bucket_incarnation_id, + ) + .await?; if let Some((_, deleted_object)) = replication_delete { ReplicationLifecycleBridge::schedule_delete(bucket.to_string(), deleted_object).await; } @@ -8858,7 +8923,15 @@ impl crate::storage_api_contracts::object::ObjectOperations for SetDisks { }; delete_request.set_tier_free_version_id(&Uuid::new_v4().to_string()); begin_scanner_publication_delete_mutation(scanner_publication_commit_scope.as_ref())?; - self.delete_object_version(bucket, object, &delete_request, false).await?; + self.delete_object_version_with_purge( + bucket, + object, + &delete_request, + false, + None, + opts.expected_bucket_incarnation_id, + ) + .await?; } for version in &versions.free_versions { ensure_delete_commit_locks_held(_lock_guard.as_ref(), bucket, object, &opts)?; @@ -8870,7 +8943,15 @@ impl crate::storage_api_contracts::object::ObjectOperations for SetDisks { }; delete_request.set_tier_free_version(); begin_scanner_publication_delete_mutation(scanner_publication_commit_scope.as_ref())?; - self.delete_object_version(bucket, object, &delete_request, false).await?; + self.delete_object_version_with_purge( + bucket, + object, + &delete_request, + false, + None, + opts.expected_bucket_incarnation_id, + ) + .await?; } } } @@ -8978,7 +9059,7 @@ impl crate::storage_api_contracts::object::ObjectOperations for SetDisks { }; ensure_delete_commit_locks_held(_lock_guard.as_ref(), bucket, object, &opts)?; begin_scanner_publication_delete_mutation(scanner_publication_commit_scope.as_ref())?; - self.delete_object_version(bucket, object, &dfi, false) + self.delete_object_version_with_purge(bucket, object, &dfi, false, None, opts.expected_bucket_incarnation_id) .await .map_err(|e| to_object_err(e, vec![bucket, object]))?; self.invalidate_get_object_metadata_cache(bucket, object).await; @@ -9072,6 +9153,7 @@ impl crate::storage_api_contracts::object::ObjectOperations for SetDisks { &fi, should_force_delete_marker_for_missing_version(&opts), delete_marker_purge.clone(), + opts.expected_bucket_incarnation_id, ) .await .map_err(|e| to_object_err(e, vec![bucket, object]))?; @@ -9124,9 +9206,16 @@ impl crate::storage_api_contracts::object::ObjectOperations for SetDisks { if opts.skip_free_version { dfi.set_skip_tier_free_version(); } - self.delete_object_version_with_purge(bucket, object, &dfi, opts.delete_marker, delete_marker_purge) - .await - .map_err(|e| to_object_err(e, vec![bucket, object]))?; + self.delete_object_version_with_purge( + bucket, + object, + &dfi, + opts.delete_marker, + delete_marker_purge, + opts.expected_bucket_incarnation_id, + ) + .await + .map_err(|e| to_object_err(e, vec![bucket, object]))?; #[cfg(test)] pause_delete_object_commit_after_publish(bucket, object).await; @@ -9188,28 +9277,9 @@ impl crate::storage_api_contracts::object::ObjectOperations for SetDisks { #[tracing::instrument(skip(self))] async fn add_partial(&self, bucket: &str, object: &str, version_id: &str) -> Result<()> { - if self.persist_partial_write(bucket, object, Some(version_id)).await { - return Ok(()); - } - let mut request = rustfs_heal_contracts::heal_channel::create_heal_request_with_options( - bucket.to_string(), - Some(object.to_string()), - false, - Some(HealChannelPriority::Normal), - Some(self.pool_index), - Some(self.set_index), - ); - request.object_version_id = (!version_id.is_empty()).then(|| version_id.to_string()); - if let Err(e) = rustfs_heal_contracts::heal_channel::send_heal_request(request).await { - warn!( - bucket, - object, - version_id, - error = %e, - "Failed to enqueue heal request for partial object" - ); - } - Ok(()) + let source_bucket_incarnation_id = self.bucket_incarnation_id_from_disk(bucket).await.ok(); + self.add_partial_with_source_incarnation(bucket, object, version_id, source_bucket_incarnation_id) + .await } #[tracing::instrument(skip(self))] @@ -9768,7 +9838,10 @@ impl crate::storage_api_contracts::object::ObjectOperations for SetDisks { #[cfg(all(test, feature = "test-util"))] pause_transition_transaction_at(bucket, object, TransitionTransactionKillPoint::CommitFenceBeforeLocalCommit).await; upload_cleanup.disarm(); - if let Err(err) = self.delete_object_version(bucket, object, &fi, false).await { + if let Err(err) = self + .delete_object_version_with_purge(bucket, object, &fi, false, None, opts.expected_bucket_incarnation_id) + .await + { warn!( bucket = bucket, object = object, @@ -9839,7 +9912,12 @@ impl crate::storage_api_contracts::object::ObjectOperations for SetDisks { continue; } let _ = self - .add_partial(bucket, object, opts.version_id.as_deref().unwrap_or_default()) + .add_partial_with_source_incarnation( + bucket, + object, + opts.version_id.as_deref().unwrap_or_default(), + opts.expected_bucket_incarnation_id, + ) .await; break; } @@ -19558,15 +19636,11 @@ mod put_object_tmp_cleanup_tests { expected.is_subset(&after_tail), "the detached tail must re-mark capacity after the first scope was drained" ); - let request = tokio::time::timeout(Duration::from_secs(30), heal_requests.recv()) - .await - .expect("a failed tail should submit heal") - .expect("the per-set heal capture should stay connected"); - assert_eq!(request.bucket, bucket); - assert_eq!(request.object_prefix.as_deref(), Some(object)); - assert_eq!(request.object_version_id.as_deref(), Some(expected_version_id.as_str())); - assert_eq!(request.pool_index, Some(set_disks.pool_index)); - assert_eq!(request.set_index, Some(set_disks.set_index)); + assert!( + heal_requests.try_recv().is_err(), + "a legacy fixture without a bucket-generation record must not enqueue an unfenced fallback heal" + ); + assert_eq!(expected_version_id, Uuid::nil().to_string()); }) .await; } @@ -19687,7 +19761,7 @@ mod put_object_tmp_cleanup_tests { #[tokio::test] #[serial_test::serial(capacity_dirty_scope)] - async fn tail_drained_put_preserves_quorum_success_and_heals_failed_tail() { + async fn tail_drained_put_preserves_quorum_success_without_unfenced_fallback_heal() { let (_dirs, disks, set) = hermetic_set_disks(4).await; let bucket = "put-full-tail-heal"; let object = "full-tail-heal-object"; @@ -19723,12 +19797,10 @@ mod put_object_tmp_cleanup_tests { .expect("PUT task should join") .expect("a minority tail error must not negate committed quorum"); assert_eq!(tasks.running(), 0); - let heal = tokio::time::timeout(Duration::from_secs(30), heals.recv()) - .await - .expect("failed tail must schedule heal") - .expect("heal capture must remain connected"); - assert_eq!(heal.bucket, bucket); - assert_eq!(heal.object_prefix.as_deref(), Some(object)); + assert!( + heals.try_recv().is_err(), + "a legacy fixture without a bucket-generation record must not enqueue an unfenced fallback heal" + ); let info = set .get_object_info(bucket, object, &ObjectOptions::default()) .await diff --git a/crates/heal-contracts/src/heal_channel.rs b/crates/heal-contracts/src/heal_channel.rs index 7732f755f..c843e5fd4 100644 --- a/crates/heal-contracts/src/heal_channel.rs +++ b/crates/heal-contracts/src/heal_channel.rs @@ -361,6 +361,9 @@ pub struct HealChannelRequest { pub object_prefix: Option, /// Object version ID (optional) pub object_version_id: Option, + /// Bucket incarnation observed by the object write that produced a local + /// durable partial-write responsibility. + pub expected_bucket_incarnation_id: Option, /// Force start heal pub force_start: bool, /// Priority @@ -598,6 +601,7 @@ pub fn create_heal_request( bucket, object_prefix, object_version_id: None, + expected_bucket_incarnation_id: None, force_start, priority: priority.unwrap_or_default(), pool_index: None, @@ -630,6 +634,7 @@ pub fn create_heal_request_with_options( bucket, object_prefix, object_version_id: None, + expected_bucket_incarnation_id: None, force_start, priority: priority.unwrap_or_default(), pool_index, @@ -661,6 +666,7 @@ fn create_auto_heal_disk_request(set_disk_id: String, priority: Option>, key: fn request_matches_task(request: &HealRequest, task: &HealTask) -> bool { request.heal_type == task.heal_type && request.bucket_incarnation_id == task.bucket_incarnation_id + && request.expected_mrf_bucket_incarnation_id == task.expected_mrf_bucket_incarnation_id && request.options == task.options && request.priority == task.priority && request.source == task.source @@ -573,6 +574,7 @@ fn request_matches_task(request: &HealRequest, task: &HealTask) -> bool { fn request_matches_request(request: &HealRequest, existing: &HealRequest) -> bool { request.heal_type == existing.heal_type && request.bucket_incarnation_id == existing.bucket_incarnation_id + && request.expected_mrf_bucket_incarnation_id == existing.expected_mrf_bucket_incarnation_id && request.options == existing.options && request.priority == existing.priority && request.source == existing.source diff --git a/crates/heal/src/heal/mrf_queue.rs b/crates/heal/src/heal/mrf_queue.rs index 94a98154c..305b9612a 100644 --- a/crates/heal/src/heal/mrf_queue.rs +++ b/crates/heal/src/heal/mrf_queue.rs @@ -21,8 +21,9 @@ //! Both paths share the existing committed snapshot format and legacy mirrors. //! Replay retains partial-write intents for live retries when a member is //! still offline at startup. Healthy legacy objects without identity proof are -//! held in memory after one check; their unchanged journal records are retried -//! on process restart. A lost proof causes another repair, not deletion. +//! stored in a checksummed lifecycle sidecar bound to the committed checkpoint. +//! An explicit operator risk acceptance remains visible and does not create a +//! repair proof; a missing or mismatched sidecar reactivates the durable intent. use super::{DiskStore, HealDiskExt as _, local_disk_map_read}; use crate::heal::manager::{HealManager, MrfRepairNoticeTarget}; @@ -32,7 +33,9 @@ use rustfs_common::mrf_channel::{ MrfDurableRepairAnchor, MrfDurableSubmission, MrfIngressResult, MrfIntent, MrfKind, }; use rustfs_heal_contracts::heal_channel::{HealAdmissionDropReason, HealAdmissionResult}; -use std::collections::{HashSet, VecDeque}; +use serde::{Deserialize, Serialize}; +use sha2::Digest; +use std::collections::{HashMap, HashSet, VecDeque}; use std::sync::Arc; use std::time::Duration; use tokio::sync::mpsc; @@ -41,7 +44,149 @@ use uuid::Uuid; use crate::heal::task::{HealOptions, HealPriority, HealRequest, HealType}; mod partial_write; -use partial_write::PartialWrites; +pub use partial_write::{MrfLegacyResponsibility, MrfOperatorAcceptance}; +use partial_write::{PartialWrites, ResponsibilityCheckpoint}; + +const MRF_LIFECYCLE_CONTROL_CAPACITY: usize = 128; +const MRF_LIFECYCLE_LIST_MAX_LIMIT: usize = 256; + +static GLOBAL_MRF_LIFECYCLE_SENDER: std::sync::OnceLock> = std::sync::OnceLock::new(); + +#[derive(Debug, thiserror::Error)] +pub enum MrfLifecycleControlError { + #[error("MRF lifecycle controller is unavailable")] + Unavailable, + #[error("the requested responsibility changed or does not exist")] + StaleGeneration, + #[error("the responsibility is not held as unverified legacy")] + NotHeld, + #[error("the bucket incarnation changed")] + IncarnationChanged, + #[error("operator action is invalid: {0}")] + InvalidAction(String), + #[error("the lifecycle checkpoint could not be persisted")] + Persistence, +} + +#[derive(Debug)] +pub struct MrfLegacyRiskAcceptanceRequest { + pub responsibility_id: Uuid, + pub expected_bucket_incarnation_id: Uuid, + pub acknowledge_unknown_source_incarnation: bool, + pub acknowledge_incarnation_mismatch: bool, + pub actor: String, + pub reason: String, + pub reference: String, + pub request_id: Uuid, +} + +#[derive(Clone, Debug, Serialize)] +#[serde(rename_all = "camelCase")] +pub struct MrfLegacyResponsibilitySnapshot { + pub contract_version: u32, + pub node_local: bool, + pub checkpoint_owner: Option, + pub checkpoint_sequence: Option, + pub held_count: usize, + pub legacy_generation_unknown_count: usize, + pub bucket_incarnation_changed_count: usize, + pub operator_accepted_count: usize, + pub responsibilities: Vec, + #[serde(skip_serializing_if = "Option::is_none")] + pub next_cursor: Option, +} + +pub async fn list_legacy_responsibilities( + after: Option, + limit: usize, +) -> Result { + if limit == 0 || limit > MRF_LIFECYCLE_LIST_MAX_LIMIT || after.is_some_and(|cursor| cursor.is_nil()) { + return Err(MrfLifecycleControlError::InvalidAction( + "cursor must be non-nil and limit must be between 1 and 256".to_string(), + )); + } + let sender = GLOBAL_MRF_LIFECYCLE_SENDER + .get() + .ok_or(MrfLifecycleControlError::Unavailable)?; + let (response, receiver) = tokio::sync::oneshot::channel(); + sender + .send(MrfLifecycleCommand::List { after, limit, response }) + .await + .map_err(|_| MrfLifecycleControlError::Unavailable)?; + receiver.await.map_err(|_| MrfLifecycleControlError::Unavailable) +} + +pub async fn recheck_legacy_responsibility( + responsibility_id: Uuid, + expected_bucket_incarnation_id: Uuid, +) -> Result<(), MrfLifecycleControlError> { + let sender = GLOBAL_MRF_LIFECYCLE_SENDER + .get() + .ok_or(MrfLifecycleControlError::Unavailable)?; + let (response, receiver) = tokio::sync::oneshot::channel(); + sender + .send(MrfLifecycleCommand::Recheck { + responsibility_id, + expected_bucket_incarnation_id, + response, + }) + .await + .map_err(|_| MrfLifecycleControlError::Unavailable)?; + receiver.await.map_err(|_| MrfLifecycleControlError::Unavailable)? +} + +pub async fn refresh_legacy_responsibility( + responsibility_id: Uuid, + expected_bucket_incarnation_id: Uuid, +) -> Result<(), MrfLifecycleControlError> { + let sender = GLOBAL_MRF_LIFECYCLE_SENDER + .get() + .ok_or(MrfLifecycleControlError::Unavailable)?; + let (response, receiver) = tokio::sync::oneshot::channel(); + sender + .send(MrfLifecycleCommand::Refresh { + responsibility_id, + expected_bucket_incarnation_id, + response, + }) + .await + .map_err(|_| MrfLifecycleControlError::Unavailable)?; + receiver.await.map_err(|_| MrfLifecycleControlError::Unavailable)? +} + +pub async fn accept_unverified_legacy_risk(request: MrfLegacyRiskAcceptanceRequest) -> Result<(), MrfLifecycleControlError> { + let sender = GLOBAL_MRF_LIFECYCLE_SENDER + .get() + .ok_or(MrfLifecycleControlError::Unavailable)?; + let (response, receiver) = tokio::sync::oneshot::channel(); + sender + .send(MrfLifecycleCommand::AcceptUnverifiedRisk { request, response }) + .await + .map_err(|_| MrfLifecycleControlError::Unavailable)?; + receiver.await.map_err(|_| MrfLifecycleControlError::Unavailable)? +} + +enum MrfLifecycleCommand { + List { + after: Option, + limit: usize, + response: tokio::sync::oneshot::Sender, + }, + Recheck { + responsibility_id: Uuid, + expected_bucket_incarnation_id: Uuid, + response: tokio::sync::oneshot::Sender>, + }, + Refresh { + responsibility_id: Uuid, + expected_bucket_incarnation_id: Uuid, + response: tokio::sync::oneshot::Sender>, + }, + AcceptUnverifiedRisk { + request: MrfLegacyRiskAcceptanceRequest, + response: tokio::sync::oneshot::Sender>, + }, +} /// Committed checkpoint publication, inspection and owner-scoped cleanup. pub mod snapshot; @@ -55,6 +200,10 @@ pub(crate) const MRF_JOURNAL_PATH: &str = "buckets/.heal/mrf/journal.bin"; /// the two files. This prevents a partial two-file flush from fabricating a /// mixed epoch. pub(crate) const MRF_SCOPED_JOURNAL_PATH: &str = "buckets/.heal/mrf/journal-scoped.bin"; +// RUSTFS_COMPAT_TODO(backlog-2682): pre-lifecycle MRF readers ignore source-incarnation records and may replay a compatibility intent against a recreated bucket. Keep rollback unsupported while durable responsibilities remain. Remove after supported rollback readers enforce the source fence and old-format responsibilities have been retired. +const MRF_LIFECYCLE_PATHS: [&str; 2] = [".heal-mrf-lifecycle.0.bin", ".heal-mrf-lifecycle.1.bin"]; +const MRF_LIFECYCLE_MAGIC: &[u8; 8] = b"RFMFLC01"; +const MRF_LIFECYCLE_FORMAT_VERSION: u8 = 1; /// Record format tag. const MRF_JOURNAL_FORMAT: u8 = 1; @@ -72,6 +221,14 @@ fn metric_f64(value: usize) -> f64 { f64::from(u32::try_from(value).unwrap_or(u32::MAX)) } +fn init_mrf_lifecycle_channel() -> Result, &'static str> { + let (sender, receiver) = mpsc::channel(MRF_LIFECYCLE_CONTROL_CAPACITY); + GLOBAL_MRF_LIFECYCLE_SENDER + .set(sender) + .map_err(|_| "MRF lifecycle control sender already initialized")?; + Ok(receiver) +} + #[derive(Debug, Clone)] pub(crate) struct MrfConsumerConfig { /// In-memory queue capacity in intents. @@ -447,6 +604,108 @@ pub(crate) fn decode_journal(data: &[u8]) -> (Vec, usize) { (intents, truncated) } +#[derive(Serialize, Deserialize)] +#[serde(deny_unknown_fields)] +struct MrfLifecyclePayload { + format_version: u8, + checkpoint_owner: Uuid, + checkpoint_sequence: u64, + records: Vec, +} + +fn encode_mrf_lifecycle_checkpoint( + owner: Uuid, + sequence: u64, + records: Vec, + byte_limit: usize, +) -> Option> { + if owner.is_nil() + || sequence == 0 + || sequence == u64::MAX + || records.iter().any(|record| !record.is_valid()) + || records + .iter() + .map(|record| record.responsibility_id) + .collect::>() + .len() + != records.len() + || records + .iter() + .map(|record| (record.intent_digest, record.source_bucket_incarnation_id)) + .collect::>() + .len() + != records.len() + { + return None; + } + let payload = MrfLifecyclePayload { + format_version: MRF_LIFECYCLE_FORMAT_VERSION, + checkpoint_owner: owner, + checkpoint_sequence: sequence, + records, + }; + let payload = serde_json::to_vec(&payload).ok()?; + if payload.len().saturating_add(8 + 1 + 8 + 32) > byte_limit { + return None; + } + let digest: [u8; 32] = sha2::Sha256::digest(&payload).into(); + let mut encoded = Vec::with_capacity(8 + 1 + 8 + 32 + payload.len()); + encoded.extend_from_slice(MRF_LIFECYCLE_MAGIC); + encoded.push(MRF_LIFECYCLE_FORMAT_VERSION); + encoded.extend_from_slice(&u64::try_from(payload.len()).ok()?.to_le_bytes()); + encoded.extend_from_slice(&digest); + encoded.extend_from_slice(&payload); + Some(encoded) +} + +fn decode_mrf_lifecycle_checkpoint(data: &[u8], byte_limit: usize) -> Option { + const HEADER_LEN: usize = 8 + 1 + 8 + 32; + if data.len() < HEADER_LEN || data.len() > byte_limit || &data[..8] != MRF_LIFECYCLE_MAGIC { + return None; + } + if data[8] != MRF_LIFECYCLE_FORMAT_VERSION { + return None; + } + let payload_len = usize::try_from(u64::from_le_bytes(data[9..17].try_into().ok()?)).ok()?; + if payload_len != data.len().saturating_sub(HEADER_LEN) { + return None; + } + let payload = &data[HEADER_LEN..]; + let digest: [u8; 32] = sha2::Sha256::digest(payload).into(); + if digest != data[17..49] { + return None; + } + let decoded: MrfLifecyclePayload = serde_json::from_slice(payload).ok()?; + if decoded.format_version != MRF_LIFECYCLE_FORMAT_VERSION + || decoded.checkpoint_owner.is_nil() + || decoded.checkpoint_sequence == 0 + || decoded.checkpoint_sequence == u64::MAX + || decoded.records.iter().any(|record| !record.is_valid()) + || decoded + .records + .iter() + .map(|record| record.responsibility_id) + .collect::>() + .len() + != decoded.records.len() + || decoded + .records + .iter() + .map(|record| (record.intent_digest, record.source_bucket_incarnation_id)) + .collect::>() + .len() + != decoded.records.len() + { + return None; + } + Some(decoded) +} + +fn intent_digest(intent: &MrfIntent) -> Option<[u8; 32]> { + let mut encoded = Vec::new(); + encode_intent(intent, &mut encoded).then(|| sha2::Sha256::digest(encoded).into()) +} + // --------------------------------------------------------------------------- // Journal disk IO (all local disks, first successful read wins) // --------------------------------------------------------------------------- @@ -466,6 +725,26 @@ async fn read_journal(path: &str) -> Option> { None } +async fn read_mrf_lifecycle_checkpoint( + owner: Uuid, + sequence: u64, + checkpoint_max_bytes: usize, + byte_limit: usize, +) -> Option> { + let bytes = snapshot::read_committed_companion( + &journal_disks().await, + owner, + sequence, + &MRF_LIFECYCLE_PATHS, + checkpoint_max_bytes, + byte_limit, + ) + .await + .ok()??; + let decoded = decode_mrf_lifecycle_checkpoint(&bytes, byte_limit)?; + (decoded.checkpoint_owner == owner && decoded.checkpoint_sequence == sequence).then_some(decoded.records) +} + /// Write the snapshot to every local disk; returns true when at least one /// disk accepted it, so a total write failure keeps the runtime dirty and /// the next tick retries the persist. @@ -513,7 +792,11 @@ async fn delete_journal(path: &str) -> bool { async fn delete_journals() -> bool { let authoritative_deleted = delete_journal(MRF_SCOPED_JOURNAL_PATH).await; let legacy_deleted = delete_journal(MRF_JOURNAL_PATH).await; - authoritative_deleted && legacy_deleted + let mut lifecycle_deleted = true; + for path in MRF_LIFECYCLE_PATHS { + lifecycle_deleted &= delete_journal(path).await; + } + authoritative_deleted && legacy_deleted && lifecycle_deleted } fn warn_mrf_journal_write(err: &super::DiskError) { @@ -531,7 +814,7 @@ fn warn_mrf_journal_write(err: &super::DiskError) { /// Translate an intent into the prioritized heal request the issue specifies: /// decode failures go Urgent ECDecode, metadata corruption goes High /// Metadata, partial writes go Normal-priority object heal with Deep verification. -pub(crate) fn build_heal_request(intent: &MrfIntent) -> HealRequest { +pub(crate) fn build_heal_request(intent: &MrfIntent, durable_anchor: Option<&MrfDurableRepairAnchor>) -> HealRequest { let bucket = intent.bucket.to_string(); let object = intent.object.to_string(); let version_id = intent @@ -583,6 +866,7 @@ pub(crate) fn build_heal_request(intent: &MrfIntent) -> HealRequest { } let mut request = HealRequest::new(heal_type, options, priority); request.source = rustfs_heal_contracts::heal_channel::HealRequestSource::Mrf; + request.expected_mrf_bucket_incarnation_id = durable_anchor.map(|anchor| anchor.bucket_incarnation_id); request } @@ -593,7 +877,7 @@ async fn submit_mrf_heal_request( ) -> crate::Result { let receipt = manager .submit_mrf_heal_request_with_receipt_and_identity( - build_heal_request(intent), + build_heal_request(intent, durable_anchor.as_ref()), MrfRepairNoticeTarget { bucket: intent.bucket.clone(), object: intent.object.clone(), @@ -642,16 +926,31 @@ struct MrfRuntime { } impl MrfRuntime { - fn adopt_replayed_partial_writes(&mut self, intents: Vec) { + fn adopt_replayed_partial_writes(&mut self, intents: Vec<(MrfIntent, Option)>) { // The decoded checkpoint bounds replay; live admission subsequently // shares these limits with the ordinary pending queue. self.queue.capacity = self.queue.capacity.saturating_add(intents.len()); - self.queue.byte_budget = self - .queue - .byte_budget - .saturating_add(intents.iter().map(PartialWrites::cost).sum::()); - for intent in intents { - if self.admit_partial_write(intent).is_err() { + self.queue.byte_budget = self.queue.byte_budget.saturating_add( + intents + .iter() + .map(|(intent, record)| match record { + Some(record) => { + PartialWrites::cost_with_state_and_audit(intent, &record.state, record.last_operator_acceptance.as_ref()) + } + None => PartialWrites::cost(intent), + }) + .sum::(), + ); + for (intent, record) in intents { + let restored_intent = intent.clone(); + let source_bucket_incarnation_id = record.as_ref().and_then(|record| record.source_bucket_incarnation_id); + if self.admit_partial_write(intent, source_bucket_incarnation_id).is_err() { + self.retain_replay_journal = true; + continue; + } + if let Some(record) = record + && !self.partial_writes.restore_state(&restored_intent, &record) + { self.retain_replay_journal = true; } } @@ -663,9 +962,14 @@ impl MrfRuntime { .retain(|anchor| !anchor.kind.is_durable() || !adopted_leases.contains(&anchor.lease)); } - fn admit_partial_write(&mut self, intent: MrfIntent) -> Result<(), MrfDurableAdmissionError> { - self.partial_writes.admit( + fn admit_partial_write( + &mut self, + intent: MrfIntent, + source_bucket_incarnation_id: Option, + ) -> Result<(), MrfDurableAdmissionError> { + self.partial_writes.admit_with_source_incarnation( intent, + source_bucket_incarnation_id, self.queue.capacity.saturating_sub(self.queue.depth()), self.queue.byte_budget.saturating_sub(self.queue.bytes()), )?; @@ -673,6 +977,160 @@ impl MrfRuntime { Ok(()) } + fn list_unverified_legacy(&self, after: Option, limit: usize) -> MrfLegacyResponsibilitySnapshot { + let (responsibilities, next_cursor) = self.partial_writes.list_unverified_legacy(after, limit); + MrfLegacyResponsibilitySnapshot { + contract_version: 1, + node_local: true, + checkpoint_owner: self.runtime_checkpoint.map(|(owner, _)| owner), + checkpoint_sequence: self.runtime_checkpoint.map(|(_, sequence)| sequence), + held_count: self.partial_writes.unverified_legacy_count(), + legacy_generation_unknown_count: self.partial_writes.legacy_generation_unknown_count(), + bucket_incarnation_changed_count: self.partial_writes.bucket_incarnation_changed_count(), + operator_accepted_count: self.partial_writes.operator_accepted_unverified_count(), + responsibilities, + next_cursor, + } + } + + async fn verify_legacy_generation_incarnation( + &self, + manager: &HealManager, + responsibility_id: Uuid, + expected_bucket_incarnation_id: Uuid, + ) -> Result { + let (intent, state, source_bucket_incarnation_id) = self + .partial_writes + .intent_for_responsibility(responsibility_id) + .ok_or(MrfLifecycleControlError::StaleGeneration)?; + if state.bucket_incarnation_id() != Some(expected_bucket_incarnation_id) { + return Err(MrfLifecycleControlError::IncarnationChanged); + } + if source_bucket_incarnation_id.is_some_and(|source| source != expected_bucket_incarnation_id) + && !matches!(state, partial_write::ResponsibilityState::BucketIncarnationChanged { .. }) + { + return Err(MrfLifecycleControlError::IncarnationChanged); + } + let anchor = manager + .durable_mrf_repair_anchor(&intent) + .await + .ok_or(MrfLifecycleControlError::IncarnationChanged)?; + if anchor.bucket_incarnation_id != expected_bucket_incarnation_id { + return Err(MrfLifecycleControlError::IncarnationChanged); + } + Ok(intent) + } + + async fn recheck_legacy_generation( + &mut self, + manager: &HealManager, + responsibility_id: Uuid, + expected_bucket_incarnation_id: Uuid, + ) -> Result<(), MrfLifecycleControlError> { + if self.partial_writes.is_active_responsibility(responsibility_id) { + return Ok(()); + } + self.verify_legacy_generation_incarnation(manager, responsibility_id, expected_bucket_incarnation_id) + .await?; + let previous_acceptance = self.partial_writes.last_operator_acceptance(responsibility_id); + let (_, previous) = self + .partial_writes + .recheck(responsibility_id, expected_bucket_incarnation_id) + .map_err(map_mrf_lifecycle_state_error)?; + self.dirty = true; + if !self.flush().await { + self.partial_writes + .restore_operator_state(responsibility_id, previous, previous_acceptance); + self.dirty = true; + return Err(MrfLifecycleControlError::Persistence); + } + self.dispatch(manager).await; + self.publish_metrics(); + Ok(()) + } + + async fn refresh_legacy_generation( + &mut self, + manager: &HealManager, + responsibility_id: Uuid, + expected_bucket_incarnation_id: Uuid, + ) -> Result<(), MrfLifecycleControlError> { + let (intent, state, source_incarnation) = self + .partial_writes + .intent_for_responsibility(responsibility_id) + .ok_or(MrfLifecycleControlError::StaleGeneration)?; + if state.bucket_incarnation_id() != Some(expected_bucket_incarnation_id) { + return Err(MrfLifecycleControlError::StaleGeneration); + } + let anchor = manager + .durable_mrf_repair_anchor(&intent) + .await + .ok_or(MrfLifecycleControlError::IncarnationChanged)?; + if anchor.bucket_incarnation_id == expected_bucket_incarnation_id { + return Ok(()); + } + let previous_acceptance = self.partial_writes.last_operator_acceptance(responsibility_id); + let (previous, _) = self + .partial_writes + .replace_observed_incarnation( + responsibility_id, + expected_bucket_incarnation_id, + anchor.bucket_incarnation_id, + source_incarnation, + ) + .map_err(map_mrf_lifecycle_state_error)?; + self.dirty = true; + if !self.flush().await { + self.partial_writes + .restore_operator_state(responsibility_id, previous, previous_acceptance); + self.dirty = true; + return Err(MrfLifecycleControlError::Persistence); + } + self.publish_metrics(); + Ok(()) + } + + async fn accept_legacy_risk( + &mut self, + manager: &HealManager, + request: MrfLegacyRiskAcceptanceRequest, + ) -> Result<(), MrfLifecycleControlError> { + let responsibility_id = request.responsibility_id; + if self + .partial_writes + .intent_for_responsibility(request.responsibility_id) + .is_some_and(|(_, state, _)| { + matches!( + state, + partial_write::ResponsibilityState::OperatorAcceptedUnverified { + request_id: stored_request_id, + .. + } if stored_request_id == request.request_id + ) + }) + { + self.partial_writes + .record_operator_acceptance(self.queue.byte_budget.saturating_sub(self.queue.bytes()), request) + .map_err(map_mrf_lifecycle_state_error)?; + return Ok(()); + } + self.verify_legacy_generation_incarnation(manager, request.responsibility_id, request.expected_bucket_incarnation_id) + .await?; + let (previous, previous_acceptance) = self + .partial_writes + .record_operator_acceptance(self.queue.byte_budget.saturating_sub(self.queue.bytes()), request) + .map_err(map_mrf_lifecycle_state_error)?; + self.dirty = true; + if !self.flush().await { + self.partial_writes + .restore_operator_state(responsibility_id, previous, previous_acceptance); + self.dirty = true; + return Err(MrfLifecycleControlError::Persistence); + } + self.publish_metrics(); + Ok(()) + } + fn snapshot(&self) -> (Vec, Vec) { let mut authoritative = Vec::new(); let mut legacy = Vec::new(); @@ -691,15 +1149,37 @@ impl MrfRuntime { async fn flush(&mut self) -> bool { let (authoritative, legacy) = self.snapshot(); + let lifecycle_byte_limit = self.config.journal_max_bytes.saturating_mul(4); + let lifecycle_checkpoint = if self.partial_writes.depth() == 0 { + None + } else { + let records = self.partial_writes.checkpoint_records(intent_digest); + (records.len() == self.partial_writes.depth()) + .then(|| { + encode_mrf_lifecycle_checkpoint( + self.checkpoint_owner, + self.next_checkpoint_sequence, + records, + lifecycle_byte_limit, + ) + }) + .flatten() + }; + let lifecycle_ready = self.partial_writes.depth() == 0 || lifecycle_checkpoint.is_some(); let (committed_persisted, committed_on_disk) = if authoritative.is_empty() { (true, false) + } else if !lifecycle_ready { + (false, false) } else { - match snapshot::publish_committed_snapshot( + match snapshot::publish_committed_snapshot_with_companion( &journal_disks().await, self.checkpoint_owner, self.next_checkpoint_sequence, &authoritative, self.config.journal_max_bytes, + lifecycle_checkpoint + .as_deref() + .map(|bytes| (&MRF_LIFECYCLE_PATHS, bytes, lifecycle_byte_limit)), ) .await { @@ -729,10 +1209,19 @@ impl MrfRuntime { // old reader from observing a newer epoch that a new reader cannot // see when the canonical write is unavailable. let legacy_persisted = authoritative_persisted && write_journal(MRF_JOURNAL_PATH, &legacy).await; + let lifecycle_persisted = if self.partial_writes.depth() == 0 { + let mut all_deleted = true; + for path in MRF_LIFECYCLE_PATHS { + all_deleted &= delete_journal(path).await; + } + all_deleted + } else { + lifecycle_checkpoint.is_some() && committed_persisted + }; // Keep dirty until the committed checkpoint, authoritative snapshot, // and compatibility mirror have all been accepted; otherwise a // one-sided failure would never retry the missing recovery anchor. - let persisted = committed_persisted && authoritative_persisted && legacy_persisted; + let persisted = committed_persisted && authoritative_persisted && legacy_persisted && lifecycle_persisted; self.new_since_flush = 0; // Keep the dirty flag when every disk write failed: a clean backlog // would otherwise never rewrite, losing the periodic persist retry a @@ -747,8 +1236,19 @@ impl MrfRuntime { /// Drain pending intents into the heal manager until it is full, the /// queue empties, or attempts are exhausted. + async fn dispatch_partial_writes(&mut self, manager: &HealManager) { + if self + .partial_writes + .dispatch_collecting_lifecycle_change(manager, &self.config) + .await + { + self.dirty = true; + self.flush().await; + } + } + async fn dispatch(&mut self, manager: &HealManager) { - self.partial_writes.dispatch(manager, &self.config).await; + self.dispatch_partial_writes(manager).await; if let Some(until) = self.backoff_until { if tokio::time::Instant::now() < until { return; @@ -757,12 +1257,12 @@ impl MrfRuntime { } while let Some(mut intent) = self.queue.pop_front() { if intent.kind.is_durable() { - if self.admit_partial_write(intent.clone()).is_err() { + if self.admit_partial_write(intent.clone(), None).is_err() { self.queue.push_back(intent); break; } if self.flush().await { - self.partial_writes.dispatch(manager, &self.config).await; + self.dispatch_partial_writes(manager).await; } continue; } @@ -870,9 +1370,9 @@ impl MrfRuntime { self.dirty |= self.partial_writes.retain_unproven(&remaining); for bucket in buckets { for event in rustfs_common::mrf_channel::take_mrf_unverified_legacy_events_for(bucket.as_ref()) { - // Parking is an in-memory dispatch decision. The unchanged - // durable journal remains the recovery/retry authority. - self.partial_writes.park_unverified_legacy(&event.anchor); + if self.partial_writes.park_unverified_legacy(&event.anchor) { + self.dirty = true; + } } } self.publish_metrics(); @@ -883,6 +1383,12 @@ impl MrfRuntime { gauge!("rustfs_heal_mrf_queue_bytes").set(metric_f64(self.queue.bytes() + self.partial_writes.bytes())); let held = self.partial_writes.unverified_legacy_count(); gauge!("rustfs_heal_mrf_unverified_legacy").set(metric_f64(held)); + gauge!("rustfs_heal_mrf_legacy_generation_unknown") + .set(metric_f64(self.partial_writes.legacy_generation_unknown_count())); + gauge!("rustfs_heal_mrf_bucket_incarnation_changed") + .set(metric_f64(self.partial_writes.bucket_incarnation_changed_count())); + gauge!("rustfs_heal_mrf_operator_accepted_unverified") + .set(metric_f64(self.partial_writes.operator_accepted_unverified_count())); let now_ms = std::time::SystemTime::now() .duration_since(std::time::UNIX_EPOCH) .map(|duration| u64::try_from(duration.as_millis()).unwrap_or(u64::MAX)) @@ -890,10 +1396,29 @@ impl MrfRuntime { let oldest_age_seconds = self .partial_writes .oldest_unverified_legacy_enqueued_at_ms() - .map(|enqueued_at_ms| now_ms.saturating_sub(enqueued_at_ms) / 1_000) + .map(|held_since_ms| now_ms.saturating_sub(held_since_ms) / 1_000) .unwrap_or_default(); gauge!("rustfs_heal_mrf_unverified_legacy_oldest_age_seconds") .set(metric_f64(usize::try_from(oldest_age_seconds).unwrap_or(usize::MAX))); + let oldest_operator_accepted_age_seconds = self + .partial_writes + .oldest_operator_accepted_at_ms() + .map(|accepted_at_ms| now_ms.saturating_sub(accepted_at_ms) / 1_000) + .unwrap_or_default(); + gauge!("rustfs_heal_mrf_operator_accepted_unverified_oldest_age_seconds") + .set(metric_f64(usize::try_from(oldest_operator_accepted_age_seconds).unwrap_or(usize::MAX))); + } +} + +fn map_mrf_lifecycle_state_error(error: &'static str) -> MrfLifecycleControlError { + if error.contains("not found") || error.contains("generation changed") { + MrfLifecycleControlError::StaleGeneration + } else if error.contains("bucket incarnation changed") { + MrfLifecycleControlError::IncarnationChanged + } else if error.contains("not held") { + MrfLifecycleControlError::NotHeld + } else { + MrfLifecycleControlError::InvalidAction(error.to_string()) } } @@ -911,21 +1436,24 @@ pub fn spawn_mrf_consumer(manager: Arc) { "Best-effort MRF ingress disabled by configuration; durable responsibilities remain active" ); } - let (receiver, durable_receiver) = match rustfs_common::mrf_channel::init_mrf_channel().and_then(|receiver| { - rustfs_common::mrf_channel::init_durable_mrf_channel().map(|durable_receiver| (receiver, durable_receiver)) - }) { - Ok(receivers) => receivers, - Err(err) => { - tracing::warn!( - target: "rustfs::heal::mrf", - error = err, - "MRF channel initialization failed; intents will be dropped at producers" - ); - return; - } - }; + let (receiver, durable_receiver, lifecycle_receiver) = + match rustfs_common::mrf_channel::init_mrf_channel().and_then(|receiver| { + rustfs_common::mrf_channel::init_durable_mrf_channel().and_then(|durable_receiver| { + init_mrf_lifecycle_channel().map(|lifecycle_receiver| (receiver, durable_receiver, lifecycle_receiver)) + }) + }) { + Ok(receivers) => receivers, + Err(err) => { + tracing::warn!( + target: "rustfs::heal::mrf", + error = err, + "MRF channel initialization failed; intents will be dropped at producers" + ); + return; + } + }; tokio::spawn(async move { - run_mrf_consumer(manager, receiver, durable_receiver).await; + run_mrf_consumer(manager, receiver, durable_receiver, lifecycle_receiver).await; }); tracing::info!(target: "rustfs::heal::mrf", "MRF intent consumer started"); } @@ -969,7 +1497,7 @@ struct ReplayOutcome { journal_on_disk: bool, retain_journal_for_replay: bool, durable_replay_anchors: Vec, - partial_writes: Vec, + partial_writes: Vec<(MrfIntent, Option)>, cleanup: Option, next_checkpoint_sequence: u64, } @@ -992,16 +1520,39 @@ enum ReplayCleanup { struct ReplaySource { data: Vec, cleanup: ReplayCleanup, + lifecycle: HashMap<[u8; 32], VecDeque>, } -async fn read_replay_source(max_bytes: usize) -> Result, snapshot::SnapshotError> { +fn take_lifecycle_record( + lifecycle: &mut HashMap<[u8; 32], VecDeque>, + intent_digest: [u8; 32], +) -> Option { + let records = lifecycle.get_mut(&intent_digest)?; + let record = records.pop_front(); + if records.is_empty() { + lifecycle.remove(&intent_digest); + } + record +} + +async fn read_replay_source( + max_bytes: usize, + lifecycle_max_bytes: usize, +) -> Result, snapshot::SnapshotError> { if let Some(committed) = snapshot::inspect_local_committed_snapshot(max_bytes).await? { + let owner = committed.owner(); + let sequence = committed.sequence(); + let records = read_mrf_lifecycle_checkpoint(owner, sequence, max_bytes, lifecycle_max_bytes) + .await + .unwrap_or_default(); + let mut lifecycle: HashMap<[u8; 32], VecDeque> = HashMap::with_capacity(records.len()); + for record in records { + lifecycle.entry(record.intent_digest).or_default().push_back(record); + } return Ok(Some(ReplaySource { data: committed.payload().to_vec(), - cleanup: ReplayCleanup::Committed { - owner: committed.owner(), - sequence: committed.sequence(), - }, + cleanup: ReplayCleanup::Committed { owner, sequence }, + lifecycle, })); } // The scoped file is a complete authoritative legacy snapshot. Fall back @@ -1017,6 +1568,7 @@ async fn read_replay_source(max_bytes: usize) -> Result, sn Ok(Some(ReplaySource { data, cleanup: ReplayCleanup::Legacy, + lifecycle: HashMap::new(), })) } @@ -1049,7 +1601,7 @@ async fn replay_into( queue: &mut MrfQueue, backoff_until: &mut Option, ) -> ReplayOutcome { - let source = match read_replay_source(queue.byte_budget).await { + let source = match read_replay_source(queue.byte_budget, queue.byte_budget.saturating_mul(4)).await { Ok(Some(source)) => source, Ok(None) => { return ReplayOutcome { @@ -1085,6 +1637,7 @@ async fn replay_into( ReplayCleanup::Committed { sequence, .. } => sequence.saturating_add(1), }; let data = source.data; + let mut lifecycle = source.lifecycle; let (decoded, truncated) = decode_journal(&data); let replayed = decoded.len(); let intents = decoded; @@ -1108,7 +1661,23 @@ async fn replay_into( let mut accepted_without_durable_anchor = false; let mut durable_replay_anchors = Vec::new(); let mut partial_writes = Vec::new(); - for intent in intents { + for mut intent in intents { + if intent.kind.is_durable() { + if !matches!( + rustfs_common::mrf_channel::try_rearm_mrf_replay_intent(&mut intent), + MrfIngressResult::Enqueued + ) { + rearm_incomplete = true; + rustfs_common::mrf_channel::release_mrf_intent(&intent); + continue; + } + if let Some(anchor) = manager.durable_mrf_repair_anchor(&intent).await { + durable_replay_anchors.push(anchor); + } + let restored = intent_digest(&intent).and_then(|digest| take_lifecycle_record(&mut lifecycle, digest)); + partial_writes.push((intent, restored)); + continue; + } let result = queue.try_push_typed(intent.clone()); match result { MrfQueuePushResult::Enqueued => {} @@ -1119,6 +1688,9 @@ async fn replay_into( } } } + if lifecycle.values().any(|records| !records.is_empty()) { + rearm_incomplete = true; + } // Drain the replayed intents immediately; whatever the manager refuses // stays armed in `queue` for the consumer's retry loop. @@ -1140,7 +1712,8 @@ async fn replay_into( if let Some(anchor) = manager.durable_mrf_repair_anchor(&intent).await { durable_replay_anchors.push(anchor.clone()); } - partial_writes.push(intent); + let restored = intent_digest(&intent).and_then(|digest| take_lifecycle_record(&mut lifecycle, digest)); + partial_writes.push((intent, restored)); continue; } match submit_mrf_heal_request(manager, &intent, None).await { @@ -1210,6 +1783,7 @@ async fn run_mrf_consumer( manager: Arc, mut receiver: mpsc::Receiver, mut durable_receiver: mpsc::Receiver, + mut lifecycle_receiver: mpsc::Receiver, ) { let config = MrfConsumerConfig::default(); let mut runtime = MrfRuntime { @@ -1264,7 +1838,7 @@ async fn run_mrf_consumer( } let mut responses = Vec::with_capacity(received); for submission in durable_batch.drain(..) { - match runtime.admit_partial_write(submission.intent) { + match runtime.admit_partial_write(submission.intent, submission.source_bucket_incarnation_id) { Ok(()) => responses.push(submission.response), Err(err) => { let _ = submission.response.send(Err(err)); } } @@ -1277,6 +1851,49 @@ async fn run_mrf_consumer( } runtime.dispatch(manager.as_ref()).await; } + command = lifecycle_receiver.recv() => { + let Some(command) = command else { + return; + }; + match command { + MrfLifecycleCommand::List { after, limit, response } => { + let _ = response.send(runtime.list_unverified_legacy(after, limit)); + } + MrfLifecycleCommand::Recheck { + responsibility_id, + expected_bucket_incarnation_id, + response, + } => { + let result = runtime + .recheck_legacy_generation( + manager.as_ref(), + responsibility_id, + expected_bucket_incarnation_id, + ) + .await; + let _ = response.send(result); + } + MrfLifecycleCommand::Refresh { + responsibility_id, + expected_bucket_incarnation_id, + response, + } => { + let result = runtime + .refresh_legacy_generation( + manager.as_ref(), + responsibility_id, + expected_bucket_incarnation_id, + ) + .await; + let _ = response.send(result); + } + MrfLifecycleCommand::AcceptUnverifiedRisk { request, response } => { + let result = runtime.accept_legacy_risk(manager.as_ref(), request).await; + let _ = response.send(result); + } + } + runtime.publish_metrics(); + } received = receiver.recv_many(&mut batch, runtime.config.replay_batch) => { if received == 0 { // Channel closed: flush once more unless the snapshot is @@ -1294,7 +1911,7 @@ async fn run_mrf_consumer( } for intent in batch.drain(..) { if intent.kind.is_durable() { - if runtime.admit_partial_write(intent.clone()).is_err() { + if runtime.admit_partial_write(intent.clone(), None).is_err() { rustfs_common::mrf_channel::release_mrf_intent(&intent); counter!("rustfs_heal_mrf_dropped_total", "reason" => "queue_overflow").increment(1); } @@ -1466,6 +2083,62 @@ mod tests { payload } + #[test] + fn lifecycle_checkpoint_is_bound_checksummed_and_strictly_decoded() { + let owner = Uuid::new_v4(); + let held_incarnation = Uuid::new_v4(); + let record = ResponsibilityCheckpoint { + intent_digest: [8; 32], + responsibility_id: Uuid::new_v4(), + source_bucket_incarnation_id: Some(held_incarnation), + last_operator_acceptance: None, + state: partial_write::ResponsibilityState::HeldUnverifiedLegacy { + bucket_incarnation_id: held_incarnation, + since_ms: 1_700_000_000_000, + }, + }; + let accepted_at_ms = 1_700_000_000_100; + let accepted_request_id = Uuid::new_v4(); + let accepted_audit = partial_write::MrfOperatorAcceptance { + accepted_at_ms, + actor: "operator-a".to_string(), + reason: "canonical source reviewed".to_string(), + reference: "INC-1234".to_string(), + request_id: accepted_request_id, + acknowledged_unknown_source_incarnation: true, + acknowledged_incarnation_mismatch: false, + }; + let accepted = ResponsibilityCheckpoint { + intent_digest: [9; 32], + responsibility_id: Uuid::new_v4(), + source_bucket_incarnation_id: None, + last_operator_acceptance: Some(accepted_audit), + state: partial_write::ResponsibilityState::OperatorAcceptedUnverified { + bucket_incarnation_id: Uuid::new_v4(), + acknowledged_unknown_source_incarnation: true, + acknowledged_incarnation_mismatch: false, + accepted_at_ms, + actor: "operator-a".to_string(), + reason: "canonical source reviewed".to_string(), + reference: "INC-1234".to_string(), + request_id: accepted_request_id, + }, + }; + let records = vec![record, accepted]; + let encoded = encode_mrf_lifecycle_checkpoint(owner, 42, records.clone(), 8192) + .expect("lifecycle checkpoint should fit the configured bound"); + let decoded = decode_mrf_lifecycle_checkpoint(&encoded, 8192).expect("valid lifecycle checkpoint"); + assert_eq!(decoded.checkpoint_owner, owner); + assert_eq!(decoded.checkpoint_sequence, 42); + assert_eq!(decoded.records, records); + + let mut torn = encoded.clone(); + *torn.last_mut().expect("non-empty checkpoint") ^= 0xff; + assert!(decode_mrf_lifecycle_checkpoint(&torn, 8192).is_none()); + assert!(decode_mrf_lifecycle_checkpoint(&encoded, encoded.len() - 1).is_none()); + assert!(encode_mrf_lifecycle_checkpoint(Uuid::nil(), 42, Vec::new(), 8192).is_none()); + } + fn w13_timestamp() -> String { chrono::Utc::now().to_rfc3339_opts(chrono::SecondsFormat::Secs, true) } @@ -3026,35 +3699,41 @@ mod tests { #[test] fn heal_request_mapping_follows_priority_matrix() { - let decode = build_heal_request(&intent("b", "o", 0)); + let decode = build_heal_request(&intent("b", "o", 0), None); assert!(matches!(decode.heal_type, HealType::ECDecode { .. })); assert_eq!(decode.priority, HealPriority::Urgent); - let metadata = build_heal_request(&MrfIntent { - bucket: StdArc::from("b"), - object: StdArc::from("o"), - version_id: None, - kind: MrfKind::MetadataCorruption, - delete_marker_purge: None, - scope: None, - lease: None, - enqueued_at_ms: 0, - attempts: 0, - }); + let metadata = build_heal_request( + &MrfIntent { + bucket: StdArc::from("b"), + object: StdArc::from("o"), + version_id: None, + kind: MrfKind::MetadataCorruption, + delete_marker_purge: None, + scope: None, + lease: None, + enqueued_at_ms: 0, + attempts: 0, + }, + None, + ); assert!(matches!(metadata.heal_type, HealType::Metadata { .. })); assert_eq!(metadata.priority, HealPriority::High); - let partial = build_heal_request(&MrfIntent { - bucket: StdArc::from("b"), - object: StdArc::from("o"), - version_id: None, - kind: MrfKind::PartialWrite, - delete_marker_purge: None, - scope: None, - lease: None, - enqueued_at_ms: 0, - attempts: 0, - }); + let partial = build_heal_request( + &MrfIntent { + bucket: StdArc::from("b"), + object: StdArc::from("o"), + version_id: None, + kind: MrfKind::PartialWrite, + delete_marker_purge: None, + scope: None, + lease: None, + enqueued_at_ms: 0, + attempts: 0, + }, + None, + ); assert!(matches!(partial.heal_type, HealType::Object { .. })); assert_eq!(partial.priority, HealPriority::Normal); assert_eq!(partial.options.scan_mode, rustfs_heal_contracts::heal_channel::HealScanMode::Deep); diff --git a/crates/heal/src/heal/mrf_queue/partial_write.rs b/crates/heal/src/heal/mrf_queue/partial_write.rs index 2b4a1aeff..e1286ac96 100644 --- a/crates/heal/src/heal/mrf_queue/partial_write.rs +++ b/crates/heal/src/heal/mrf_queue/partial_write.rs @@ -12,19 +12,189 @@ // See the License for the specific language governing permissions and // limitations under the License. -use super::{HealManager, MrfConsumerConfig, MrfDurableRepairAnchor, MrfIntent, MrfQueueKey, queue_key, submit_mrf_heal_request}; +use super::{ + HealManager, MrfConsumerConfig, MrfDurableRepairAnchor, MrfIntent, MrfLegacyRiskAcceptanceRequest, MrfQueueKey, queue_key, + submit_mrf_heal_request, +}; use rustfs_common::mrf_channel::{MrfDurableAdmissionError, MrfIngressResult, release_mrf_intent, try_rearm_mrf_replay_intent}; -use std::collections::{HashMap, HashSet, VecDeque}; +use serde::{Deserialize, Serialize}; +use std::collections::{BTreeMap, HashMap, HashSet, VecDeque}; +use std::time::{SystemTime, UNIX_EPOCH}; use tokio::time::Instant; +use uuid::Uuid; + +const MAX_OPERATOR_REASON_BYTES: usize = 1024; +const MAX_OPERATOR_ACTOR_BYTES: usize = 256; +const MAX_OPERATOR_REFERENCE_BYTES: usize = 256; +const LIFECYCLE_CHECKPOINT_RECORD_OVERHEAD_BYTES: usize = 256; + +#[derive(Clone, Debug, PartialEq, Eq, Hash)] +struct PartialWriteKey { + identity: MrfQueueKey, + source_bucket_incarnation_id: Option, +} + +impl PartialWriteKey { + fn new(intent: &MrfIntent, source_bucket_incarnation_id: Option) -> Self { + Self { + identity: queue_key(intent), + source_bucket_incarnation_id, + } + } + + fn from_anchor(anchor: &MrfDurableRepairAnchor) -> Self { + Self { + identity: MrfQueueKey { + kind: anchor.kind, + bucket: anchor.bucket.clone(), + object: anchor.object.clone(), + version_id: anchor.version_id, + scope: anchor.scope, + delete_marker_purge: anchor.delete_marker_purge, + }, + source_bucket_incarnation_id: Some(anchor.bucket_incarnation_id), + } + } +} + +#[derive(Clone, Debug, PartialEq, Eq, Serialize, Deserialize)] +#[serde(tag = "state", rename_all = "snake_case", deny_unknown_fields)] +pub(super) enum ResponsibilityState { + Active, + HeldUnverifiedLegacy { + bucket_incarnation_id: Uuid, + since_ms: u64, + }, + LegacyGenerationUnknown { + observed_bucket_incarnation_id: Uuid, + detected_at_ms: u64, + }, + BucketIncarnationChanged { + source_bucket_incarnation_id: Uuid, + observed_bucket_incarnation_id: Uuid, + detected_at_ms: u64, + }, + OperatorAcceptedUnverified { + bucket_incarnation_id: Uuid, + acknowledged_unknown_source_incarnation: bool, + acknowledged_incarnation_mismatch: bool, + accepted_at_ms: u64, + actor: String, + reason: String, + reference: String, + request_id: Uuid, + }, +} + +impl ResponsibilityState { + fn is_parked(&self) -> bool { + !matches!(self, Self::Active) + } + + fn estimated_bytes(&self) -> usize { + match self { + Self::Active => 0, + Self::HeldUnverifiedLegacy { .. } => std::mem::size_of::() + std::mem::size_of::(), + Self::LegacyGenerationUnknown { .. } => std::mem::size_of::() + std::mem::size_of::(), + Self::BucketIncarnationChanged { .. } => 2 * std::mem::size_of::() + std::mem::size_of::(), + Self::OperatorAcceptedUnverified { + actor, + reason, + reference, + .. + } => std::mem::size_of::() * 2 + std::mem::size_of::() + actor.len() + reason.len() + reference.len(), + } + } + + pub(super) fn bucket_incarnation_id(&self) -> Option { + match self { + Self::Active => None, + Self::HeldUnverifiedLegacy { + bucket_incarnation_id, .. + } + | Self::OperatorAcceptedUnverified { + bucket_incarnation_id, .. + } => Some(*bucket_incarnation_id), + Self::LegacyGenerationUnknown { + observed_bucket_incarnation_id, + .. + } => Some(*observed_bucket_incarnation_id), + Self::BucketIncarnationChanged { + observed_bucket_incarnation_id, + .. + } => Some(*observed_bucket_incarnation_id), + } + } +} + +#[derive(Clone, Debug, PartialEq, Eq, Serialize, Deserialize)] +#[serde(deny_unknown_fields)] +pub(super) struct ResponsibilityCheckpoint { + pub intent_digest: [u8; 32], + pub responsibility_id: Uuid, + pub source_bucket_incarnation_id: Option, + pub last_operator_acceptance: Option, + pub state: ResponsibilityState, +} + +impl ResponsibilityCheckpoint { + pub(super) fn is_valid(&self) -> bool { + if self.responsibility_id.is_nil() || self.source_bucket_incarnation_id.is_some_and(|value| value.is_nil()) { + return false; + } + if self.last_operator_acceptance.as_ref().is_some_and(|audit| { + validate_operator_audit_fields(&audit.actor, &audit.reason, &audit.reference, audit.request_id).is_err() + || (self.source_bucket_incarnation_id.is_none() && !audit.acknowledged_unknown_source_incarnation) + }) { + return false; + } + match &self.state { + ResponsibilityState::Active => true, + ResponsibilityState::HeldUnverifiedLegacy { + bucket_incarnation_id, .. + } => !bucket_incarnation_id.is_nil() && self.source_bucket_incarnation_id == Some(*bucket_incarnation_id), + ResponsibilityState::LegacyGenerationUnknown { + observed_bucket_incarnation_id, + .. + } => !observed_bucket_incarnation_id.is_nil() && self.source_bucket_incarnation_id.is_none(), + ResponsibilityState::BucketIncarnationChanged { + source_bucket_incarnation_id, + observed_bucket_incarnation_id, + .. + } => { + !source_bucket_incarnation_id.is_nil() + && !observed_bucket_incarnation_id.is_nil() + && self.source_bucket_incarnation_id == Some(*source_bucket_incarnation_id) + } + ResponsibilityState::OperatorAcceptedUnverified { + bucket_incarnation_id, + acknowledged_unknown_source_incarnation, + acknowledged_incarnation_mismatch, + actor, + reason, + reference, + request_id, + .. + } => { + !bucket_incarnation_id.is_nil() + && self + .source_bucket_incarnation_id + .is_none_or(|source| source == *bucket_incarnation_id || *acknowledged_incarnation_mismatch) + && (self.source_bucket_incarnation_id.is_some() || *acknowledged_unknown_source_incarnation) + && validate_operator_audit_fields(actor, reason, reference, *request_id).is_ok() + } + } + } +} struct Responsibility { intent: MrfIntent, anchor: Option, persisted: bool, - /// A completed check identified a healthy legacy object that cannot - /// discharge its durable intent without an independent payload proof. - /// This is process-local only: the unchanged journal rechecks it on restart. - unverified_legacy: bool, + responsibility_id: Uuid, + source_bucket_incarnation_id: Option, + last_operator_acceptance: Option, + state: ResponsibilityState, retry_queued: bool, next_attempt: Instant, } @@ -33,30 +203,66 @@ struct Responsibility { /// Admission, task failure and retry exhaustion cannot release their records. #[derive(Default)] pub(super) struct PartialWrites { - entries: HashMap, - retry_order: VecDeque, + entries: HashMap, + retry_order: VecDeque, + retry_index: HashSet, bytes: usize, } impl PartialWrites { pub(super) fn cost(intent: &MrfIntent) -> usize { - intent.estimated_bytes() + std::mem::size_of::() + 2 * std::mem::size_of::() + Self::cost_with_state(intent, &ResponsibilityState::Active) } + pub(super) fn cost_with_state(intent: &MrfIntent, state: &ResponsibilityState) -> usize { + intent.estimated_bytes() + + std::mem::size_of::() + + 2 * std::mem::size_of::() + + LIFECYCLE_CHECKPOINT_RECORD_OVERHEAD_BYTES + + state.estimated_bytes() + } + + pub(super) fn cost_with_state_and_audit( + intent: &MrfIntent, + state: &ResponsibilityState, + audit: Option<&MrfOperatorAcceptance>, + ) -> usize { + Self::cost_with_state(intent, state) + + audit.map_or(0, |audit| audit.actor.len() + audit.reason.len() + audit.reference.len()) + } + + fn entry_cost(entry: &Responsibility) -> usize { + Self::cost_with_state_and_audit(&entry.intent, &entry.state, entry.last_operator_acceptance.as_ref()) + } + + #[cfg(test)] pub(super) fn admit( &mut self, - mut intent: MrfIntent, + intent: MrfIntent, capacity: usize, byte_budget: usize, ) -> Result<(), MrfDurableAdmissionError> { + self.admit_with_source_incarnation(intent, None, capacity, byte_budget) + } + + pub(super) fn admit_with_source_incarnation( + &mut self, + mut intent: MrfIntent, + source_bucket_incarnation_id: Option, + capacity: usize, + byte_budget: usize, + ) -> Result<(), MrfDurableAdmissionError> { + if source_bucket_incarnation_id.is_some_and(|incarnation| incarnation.is_nil()) { + return Err(MrfDurableAdmissionError::InvalidIdentity); + } if try_rearm_mrf_replay_intent(&mut intent) != MrfIngressResult::Enqueued { return Err(MrfDurableAdmissionError::InvalidIdentity); } - let key = queue_key(&intent); + let key = PartialWriteKey::new(&intent, source_bucket_incarnation_id); let previous = self.entries.get(&key); - let previous_was_held = previous.is_some_and(|entry| entry.unverified_legacy); + let previous_was_held = previous.is_some_and(|entry| entry.state.is_parked()); let mut retry_queued = previous.is_some_and(|entry| entry.retry_queued); - let old_cost = previous.map_or(0, |entry| Self::cost(&entry.intent)); + let old_cost = previous.map_or(0, Self::entry_cost); let next_bytes = self.bytes.saturating_sub(old_cost).saturating_add(Self::cost(&intent)); if (previous.is_none() && self.entries.len() >= capacity) || next_bytes > byte_budget { return Err(MrfDurableAdmissionError::Full); @@ -65,18 +271,21 @@ impl PartialWrites { return Ok(()); } if previous.is_none() || (previous_was_held && !retry_queued) { - self.retry_order.push_back(key.clone()); + if self.retry_index.insert(key.clone()) { + self.retry_order.push_back(key.clone()); + } retry_queued = true; } - // Replacing the generation preserves the logical repair obligation, - // but requires a new checkpoint and proof before it can be released. self.entries.insert( key, Responsibility { intent, anchor: None, persisted: false, - unverified_legacy: false, + responsibility_id: Uuid::new_v4(), + source_bucket_incarnation_id, + last_operator_acceptance: None, + state: ResponsibilityState::Active, retry_queued, next_attempt: Instant::now(), }, @@ -102,63 +311,486 @@ impl PartialWrites { } pub(super) fn park_unverified_legacy(&mut self, anchor: &MrfDurableRepairAnchor) -> bool { - let key = MrfQueueKey { - kind: anchor.kind, - bucket: anchor.bucket.clone(), - object: anchor.object.clone(), - version_id: anchor.version_id, - scope: anchor.scope, - delete_marker_purge: anchor.delete_marker_purge, - }; + let key = PartialWriteKey::from_anchor(anchor); let Some(entry) = self.entries.get_mut(&key) else { return false; }; - if entry.anchor.as_ref() != Some(anchor) || entry.unverified_legacy { + if entry.anchor.as_ref() != Some(anchor) + || entry.state.is_parked() + || anchor.bucket_incarnation_id.is_nil() + || entry.source_bucket_incarnation_id != Some(anchor.bucket_incarnation_id) + { return false; } - entry.unverified_legacy = true; + let old_cost = Self::entry_cost(entry); + entry.state = ResponsibilityState::HeldUnverifiedLegacy { + bucket_incarnation_id: anchor.bucket_incarnation_id, + since_ms: unix_now_ms(), + }; + self.bytes = self.bytes.saturating_sub(old_cost).saturating_add(Self::entry_cost(entry)); true } pub(super) fn unverified_legacy_count(&self) -> usize { - self.entries.values().filter(|entry| entry.unverified_legacy).count() + self.entries + .values() + .filter(|entry| matches!(&entry.state, ResponsibilityState::HeldUnverifiedLegacy { .. })) + .count() + } + + pub(super) fn bucket_incarnation_changed_count(&self) -> usize { + self.entries + .values() + .filter(|entry| matches!(&entry.state, ResponsibilityState::BucketIncarnationChanged { .. })) + .count() + } + + pub(super) fn legacy_generation_unknown_count(&self) -> usize { + self.entries + .values() + .filter(|entry| matches!(&entry.state, ResponsibilityState::LegacyGenerationUnknown { .. })) + .count() + } + + pub(super) fn operator_accepted_unverified_count(&self) -> usize { + self.entries + .values() + .filter(|entry| matches!(&entry.state, ResponsibilityState::OperatorAcceptedUnverified { .. })) + .count() } pub(super) fn oldest_unverified_legacy_enqueued_at_ms(&self) -> Option { self.entries .values() - .filter(|entry| entry.unverified_legacy) - .map(|entry| entry.intent.enqueued_at_ms) + .filter_map(|entry| match &entry.state { + ResponsibilityState::HeldUnverifiedLegacy { since_ms, .. } => Some(*since_ms), + _ => None, + }) .min() } + pub(super) fn oldest_operator_accepted_at_ms(&self) -> Option { + self.entries + .values() + .filter_map(|entry| match &entry.state { + ResponsibilityState::OperatorAcceptedUnverified { accepted_at_ms, .. } => Some(*accepted_at_ms), + _ => None, + }) + .min() + } + + pub(super) fn checkpoint_records( + &self, + digest_for: impl Fn(&MrfIntent) -> Option<[u8; 32]>, + ) -> Vec { + self.entries + .values() + .filter_map(|entry| { + Some(ResponsibilityCheckpoint { + intent_digest: digest_for(&entry.intent)?, + responsibility_id: entry.responsibility_id, + source_bucket_incarnation_id: entry.source_bucket_incarnation_id, + last_operator_acceptance: entry.last_operator_acceptance.clone(), + state: entry.state.clone(), + }) + }) + .collect() + } + + pub(super) fn restore_state(&mut self, intent: &MrfIntent, record: &ResponsibilityCheckpoint) -> bool { + let key = PartialWriteKey::new(intent, record.source_bucket_incarnation_id); + let Some(entry) = self.entries.get_mut(&key) else { + return false; + }; + if !matches!(&entry.state, ResponsibilityState::Active) { + return false; + } + let old_cost = Self::entry_cost(entry); + entry.responsibility_id = record.responsibility_id; + entry.source_bucket_incarnation_id = record.source_bucket_incarnation_id; + entry.last_operator_acceptance = record.last_operator_acceptance.clone(); + entry.state = record.state.clone(); + self.bytes = self.bytes.saturating_sub(old_cost).saturating_add(Self::entry_cost(entry)); + if entry.state.is_parked() { + entry.retry_queued = false; + } + true + } + + pub(super) fn intent_for_responsibility( + &self, + responsibility_id: Uuid, + ) -> Option<(MrfIntent, ResponsibilityState, Option)> { + self.entries + .values() + .find(|entry| entry.responsibility_id == responsibility_id) + .map(|entry| (entry.intent.clone(), entry.state.clone(), entry.source_bucket_incarnation_id)) + } + + pub(super) fn is_active_responsibility(&self, responsibility_id: Uuid) -> bool { + self.entries + .values() + .find(|entry| entry.responsibility_id == responsibility_id) + .is_some_and(|entry| matches!(&entry.state, ResponsibilityState::Active)) + } + + pub(super) fn last_operator_acceptance(&self, responsibility_id: Uuid) -> Option { + self.entries + .values() + .find(|entry| entry.responsibility_id == responsibility_id) + .and_then(|entry| entry.last_operator_acceptance.clone()) + } + + pub(super) fn record_operator_acceptance( + &mut self, + byte_budget: usize, + request: MrfLegacyRiskAcceptanceRequest, + ) -> Result<(ResponsibilityState, Option), &'static str> { + let MrfLegacyRiskAcceptanceRequest { + responsibility_id, + expected_bucket_incarnation_id, + acknowledge_unknown_source_incarnation, + acknowledge_incarnation_mismatch, + actor, + reason, + reference, + request_id, + } = request; + validate_operator_audit_fields(&actor, &reason, &reference, request_id)?; + let entry = self + .entries + .values_mut() + .find(|entry| entry.responsibility_id == responsibility_id) + .ok_or("responsibility not found or generation changed")?; + let (current_incarnation, has_incarnation_mismatch) = match &entry.state { + ResponsibilityState::HeldUnverifiedLegacy { + bucket_incarnation_id, .. + } => (*bucket_incarnation_id, false), + ResponsibilityState::LegacyGenerationUnknown { + observed_bucket_incarnation_id, + .. + } => (*observed_bucket_incarnation_id, false), + ResponsibilityState::BucketIncarnationChanged { + observed_bucket_incarnation_id, + .. + } => (*observed_bucket_incarnation_id, true), + ResponsibilityState::OperatorAcceptedUnverified { + bucket_incarnation_id: existing_bucket_incarnation_id, + actor: existing_actor, + reason: existing_reason, + reference: existing_reference, + request_id: existing_request_id, + acknowledged_unknown_source_incarnation: existing_unknown_ack, + acknowledged_incarnation_mismatch: existing_mismatch_ack, + .. + } if *existing_request_id == request_id + && *existing_bucket_incarnation_id == expected_bucket_incarnation_id + && existing_actor == &actor + && existing_reason == &reason + && existing_reference == &reference + && *existing_unknown_ack == acknowledge_unknown_source_incarnation + && *existing_mismatch_ack == acknowledge_incarnation_mismatch => + { + return Ok((entry.state.clone(), entry.last_operator_acceptance.clone())); + } + ResponsibilityState::OperatorAcceptedUnverified { + request_id: existing_request_id, + .. + } if *existing_request_id == request_id => return Err("request ID was reused with different audit fields"), + ResponsibilityState::OperatorAcceptedUnverified { .. } => { + return Err("a different operator disposition already exists"); + } + ResponsibilityState::Active => return Err("responsibility is not held as unverified legacy"), + }; + if current_incarnation != expected_bucket_incarnation_id { + return Err("bucket incarnation changed; refresh the responsibility listing"); + } + if entry.source_bucket_incarnation_id.is_none() && !acknowledge_unknown_source_incarnation { + return Err("the original bucket incarnation is unknown; explicit acknowledgment is required"); + } + if has_incarnation_mismatch && !acknowledge_incarnation_mismatch { + return Err("the source and current bucket incarnations differ; explicit acknowledgment is required"); + } + let previous = entry.state.clone(); + let previous_acceptance = entry.last_operator_acceptance.clone(); + let old_cost = Self::entry_cost(entry); + let accepted_at_ms = unix_now_ms(); + let acceptance = MrfOperatorAcceptance { + accepted_at_ms, + actor: actor.clone(), + reason: reason.clone(), + reference: reference.clone(), + request_id, + acknowledged_unknown_source_incarnation: acknowledge_unknown_source_incarnation, + acknowledged_incarnation_mismatch: acknowledge_incarnation_mismatch, + }; + let next_state = ResponsibilityState::OperatorAcceptedUnverified { + bucket_incarnation_id: current_incarnation, + acknowledged_unknown_source_incarnation: acknowledge_unknown_source_incarnation, + acknowledged_incarnation_mismatch: acknowledge_incarnation_mismatch, + accepted_at_ms, + actor, + reason, + reference, + request_id, + }; + let next_cost = Self::cost_with_state_and_audit(&entry.intent, &next_state, Some(&acceptance)); + let next_bytes = self.bytes.saturating_sub(old_cost).saturating_add(next_cost); + if next_bytes > byte_budget { + return Err("durable responsibility byte budget is exhausted"); + } + entry.state = next_state; + entry.last_operator_acceptance = Some(acceptance); + self.bytes = next_bytes; + Ok((previous, previous_acceptance)) + } + + pub(super) fn recheck( + &mut self, + responsibility_id: Uuid, + expected_bucket_incarnation_id: Uuid, + ) -> Result<(MrfIntent, ResponsibilityState), &'static str> { + let entry = self + .entries + .values_mut() + .find(|entry| entry.responsibility_id == responsibility_id) + .ok_or("responsibility not found or generation changed")?; + let current_incarnation = entry + .state + .bucket_incarnation_id() + .ok_or("responsibility is not held as unverified legacy")?; + if entry.source_bucket_incarnation_id.is_none() + || matches!(&entry.state, ResponsibilityState::LegacyGenerationUnknown { .. }) + { + return Err("the source bucket incarnation is unknown; targeted recheck is not safe"); + } + if entry.source_bucket_incarnation_id != Some(expected_bucket_incarnation_id) { + return Err("source and current bucket incarnations differ; targeted recheck is not safe"); + } + if matches!(&entry.state, ResponsibilityState::BucketIncarnationChanged { .. }) { + return Err("source and current bucket incarnations differ; recheck is not safe"); + } + if matches!(&entry.state, ResponsibilityState::LegacyGenerationUnknown { .. }) { + return Err("source bucket incarnation is unknown; automatic recheck is not safe"); + } + if current_incarnation != expected_bucket_incarnation_id { + return Err("bucket incarnation changed; refresh the responsibility listing"); + } + let previous = entry.state.clone(); + let old_cost = Self::entry_cost(entry); + entry.state = ResponsibilityState::Active; + self.bytes = self.bytes.saturating_sub(old_cost).saturating_add(Self::entry_cost(entry)); + entry.persisted = false; + entry.next_attempt = Instant::now(); + if !entry.retry_queued { + let key = PartialWriteKey::new(&entry.intent, entry.source_bucket_incarnation_id); + if self.retry_index.insert(key.clone()) { + self.retry_order.push_back(key); + } + entry.retry_queued = true; + } + Ok((entry.intent.clone(), previous)) + } + + pub(super) fn restore_operator_state( + &mut self, + responsibility_id: Uuid, + state: ResponsibilityState, + last_operator_acceptance: Option, + ) -> bool { + let Some(entry) = self + .entries + .values_mut() + .find(|entry| entry.responsibility_id == responsibility_id) + else { + return false; + }; + let old_cost = Self::entry_cost(entry); + entry.state = state; + entry.last_operator_acceptance = last_operator_acceptance; + self.bytes = self.bytes.saturating_sub(old_cost).saturating_add(Self::entry_cost(entry)); + entry.persisted = true; + if entry.state.is_parked() { + entry.retry_queued = false; + } + true + } + + pub(super) fn list_unverified_legacy( + &self, + after: Option, + limit: usize, + ) -> (Vec, Option) { + let mut selected = BTreeMap::new(); + for entry in self.entries.values() { + let (status, bucket_incarnation_id, held_since_ms) = match &entry.state { + ResponsibilityState::Active => continue, + ResponsibilityState::HeldUnverifiedLegacy { + bucket_incarnation_id, + since_ms, + } => ("held_unverified_legacy", *bucket_incarnation_id, Some(*since_ms)), + ResponsibilityState::LegacyGenerationUnknown { + observed_bucket_incarnation_id, + detected_at_ms, + } => ("legacy_generation_unknown", *observed_bucket_incarnation_id, Some(*detected_at_ms)), + ResponsibilityState::BucketIncarnationChanged { + observed_bucket_incarnation_id, + detected_at_ms, + .. + } => ("bucket_incarnation_changed", *observed_bucket_incarnation_id, Some(*detected_at_ms)), + ResponsibilityState::OperatorAcceptedUnverified { + bucket_incarnation_id, .. + } => ("operator_accepted_unverified", *bucket_incarnation_id, None), + }; + if after.is_some_and(|after| entry.responsibility_id <= after) { + continue; + } + if selected.len() > limit + && selected + .last_key_value() + .is_some_and(|(largest_selected, _)| entry.responsibility_id >= *largest_selected) + { + continue; + } + let row = MrfLegacyResponsibility { + responsibility_id: entry.responsibility_id, + bucket: entry.intent.bucket.to_string(), + object: entry.intent.object.to_string(), + source_bucket_incarnation_id: entry.source_bucket_incarnation_id, + version_id: entry + .intent + .version_id + .and_then(|bytes| Uuid::from_slice(&bytes).ok()) + .filter(|version| !version.is_nil()) + .map(|version| version.to_string()), + pool_index: entry.intent.scope.map(|scope| scope.pool_index), + set_index: entry.intent.scope.map(|scope| scope.set_index), + bucket_incarnation_id, + enqueued_at_ms: entry.intent.enqueued_at_ms, + status, + held_since_ms, + accepted: entry.last_operator_acceptance.clone(), + }; + selected.insert(entry.responsibility_id, row); + if selected.len() > limit + 1 { + selected.pop_last(); + } + } + let next_cursor = if selected.len() > limit { + selected.keys().nth(limit - 1).copied() + } else { + None + }; + if selected.len() > limit { + selected.pop_last(); + } + (selected.into_values().collect(), next_cursor) + } + + pub(super) fn replace_observed_incarnation( + &mut self, + responsibility_id: Uuid, + expected_bucket_incarnation_id: Uuid, + observed_bucket_incarnation_id: Uuid, + source_bucket_incarnation_id: Option, + ) -> Result<(ResponsibilityState, Option), &'static str> { + let entry = self + .entries + .values_mut() + .find(|entry| entry.responsibility_id == responsibility_id) + .ok_or("responsibility not found or generation changed")?; + if entry.state.bucket_incarnation_id() != Some(expected_bucket_incarnation_id) { + return Err("responsibility changed; refresh the listing"); + } + let previous = entry.state.clone(); + let previous_acceptance = entry.last_operator_acceptance.clone(); + if source_bucket_incarnation_id.is_some_and(|source| source != observed_bucket_incarnation_id) { + let old_cost = Self::entry_cost(entry); + entry.state = ResponsibilityState::BucketIncarnationChanged { + source_bucket_incarnation_id: match source_bucket_incarnation_id { + Some(source) => source, + None => return Err("source incarnation changed while refreshing responsibility"), + }, + observed_bucket_incarnation_id, + detected_at_ms: unix_now_ms(), + }; + self.bytes = self.bytes.saturating_sub(old_cost).saturating_add(Self::entry_cost(entry)); + entry.retry_queued = false; + } else if source_bucket_incarnation_id.is_none() { + let old_cost = Self::entry_cost(entry); + entry.state = ResponsibilityState::LegacyGenerationUnknown { + observed_bucket_incarnation_id, + detected_at_ms: unix_now_ms(), + }; + self.bytes = self.bytes.saturating_sub(old_cost).saturating_add(Self::entry_cost(entry)); + entry.retry_queued = false; + } + Ok((previous, previous_acceptance)) + } + pub(super) fn mark_persisted(&mut self) { for entry in self.entries.values_mut() { entry.persisted = true; } } - pub(super) async fn dispatch(&mut self, manager: &HealManager, config: &MrfConsumerConfig) { + pub(super) async fn dispatch_collecting_lifecycle_change( + &mut self, + manager: &HealManager, + config: &MrfConsumerConfig, + ) -> bool { let now = Instant::now(); + let mut lifecycle_changed = false; for key in self.ready_keys(now, config.replay_batch) { let Some(entry) = self.entries.get_mut(&key) else { continue; }; entry.next_attempt = now + config.admission_backoff; - if entry.anchor.is_none() { - entry.anchor = manager.durable_mrf_repair_anchor(&entry.intent).await; - } + // The bucket can be deleted and recreated while an obligation is + // parked, so every dispatch must compare against a fresh identity. + entry.anchor = manager.durable_mrf_repair_anchor(&entry.intent).await; if let Some(anchor) = entry.anchor.clone() { - // Every outcome retains responsibility until a verified proof. - // A healthy legacy object without an independent identity proof - // is parked in memory after one check; its disk journal remains - // unchanged and startup replay checks it again. + if entry.source_bucket_incarnation_id.is_none() { + let old_cost = Self::entry_cost(entry); + let detected_at_ms = SystemTime::now() + .duration_since(UNIX_EPOCH) + .map_or(0, |duration| u64::try_from(duration.as_millis()).unwrap_or(u64::MAX)); + entry.state = ResponsibilityState::LegacyGenerationUnknown { + observed_bucket_incarnation_id: anchor.bucket_incarnation_id, + detected_at_ms, + }; + self.bytes = self.bytes.saturating_sub(old_cost).saturating_add(Self::entry_cost(entry)); + entry.retry_queued = false; + lifecycle_changed = true; + continue; + } + if let Some(source_bucket_incarnation_id) = entry.source_bucket_incarnation_id + && source_bucket_incarnation_id != anchor.bucket_incarnation_id + { + let old_cost = Self::entry_cost(entry); + let detected_at_ms = SystemTime::now() + .duration_since(UNIX_EPOCH) + .map_or(0, |duration| u64::try_from(duration.as_millis()).unwrap_or(u64::MAX)); + entry.state = ResponsibilityState::BucketIncarnationChanged { + source_bucket_incarnation_id, + observed_bucket_incarnation_id: anchor.bucket_incarnation_id, + detected_at_ms, + }; + self.bytes = self.bytes.saturating_sub(old_cost).saturating_add(Self::entry_cost(entry)); + entry.retry_queued = false; + lifecycle_changed = true; + continue; + } + // Proofless legacy results stay owned by the durable journal; + // the lifecycle checkpoint determines whether this process + // retries or waits for an explicit operator action. let _ = submit_mrf_heal_request(manager, &entry.intent, Some(anchor)).await; } } + lifecycle_changed } - fn ready_keys(&mut self, now: Instant, limit: usize) -> Vec { + fn ready_keys(&mut self, now: Instant, limit: usize) -> Vec { let mut ready = Vec::with_capacity(limit.min(self.entries.len())); // Rotation prevents a permanently failing prefix from starving the // rest of a backlog larger than one retry interval's batch budget. @@ -169,17 +801,19 @@ impl PartialWrites { let Some(key) = self.retry_order.pop_front() else { break; }; + self.retry_index.remove(&key); let Some(entry) = self.entries.get_mut(&key) else { continue; }; entry.retry_queued = false; - if entry.unverified_legacy { + if entry.state.is_parked() { // Held obligations remain in the durable responsibility map, // but leave the hot retry index until a new generation arrives. continue; } let due = entry.persisted && entry.next_attempt <= now; entry.retry_queued = true; + self.retry_index.insert(key.clone()); self.retry_order.push_back(key.clone()); if due { ready.push(key); @@ -200,16 +834,68 @@ impl PartialWrites { if before == self.entries.len() { return false; } - self.bytes = self.entries.values().map(|entry| Self::cost(&entry.intent)).sum(); + self.bytes = self.entries.values().map(Self::entry_cost).sum(); self.retry_order.retain(|key| self.entries.contains_key(key)); + self.retry_index.retain(|key| self.entries.contains_key(key)); true } } +fn validate_operator_audit_fields(actor: &str, reason: &str, reference: &str, request_id: Uuid) -> Result<(), &'static str> { + if request_id.is_nil() || actor.trim().is_empty() || actor.len() > MAX_OPERATOR_ACTOR_BYTES { + return Err("operator identity or request id is invalid"); + } + if reason.trim().is_empty() || reason.len() > MAX_OPERATOR_REASON_BYTES { + return Err("a reason of at most 1024 bytes is required"); + } + if reference.trim().is_empty() || reference.len() > MAX_OPERATOR_REFERENCE_BYTES { + return Err("an audit reference of at most 256 bytes is required"); + } + Ok(()) +} + +#[derive(Clone, Debug, Serialize)] +#[serde(rename_all = "camelCase")] +pub struct MrfLegacyResponsibility { + pub responsibility_id: Uuid, + pub bucket: String, + pub object: String, + pub source_bucket_incarnation_id: Option, + pub version_id: Option, + pub pool_index: Option, + pub set_index: Option, + pub bucket_incarnation_id: Uuid, + pub enqueued_at_ms: u64, + pub status: &'static str, + pub held_since_ms: Option, + pub accepted: Option, +} + +#[derive(Clone, Debug, PartialEq, Eq, Serialize, Deserialize)] +#[serde(rename_all = "camelCase")] +pub struct MrfOperatorAcceptance { + pub accepted_at_ms: u64, + pub actor: String, + pub reason: String, + pub reference: String, + pub request_id: Uuid, + pub acknowledged_unknown_source_incarnation: bool, + pub acknowledged_incarnation_mismatch: bool, +} + +fn unix_now_ms() -> u64 { + SystemTime::now() + .duration_since(UNIX_EPOCH) + .map_or(0, |duration| u64::try_from(duration.as_millis()).unwrap_or(u64::MAX)) +} + #[cfg(test)] mod tests { use super::*; - use rustfs_common::mrf_channel::{MrfKind, MrfScope}; + use crate::heal::mrf_queue::MrfLegacyRiskAcceptanceRequest; + use crate::heal::storage::{ECStoreHealStorage, HealStorageAPI}; + use rustfs_common::mrf_channel::{MrfKind, MrfScope, MrfVerifiedRepairDisposition, MrfVerifiedRepairEvent}; + use serial_test::serial; use std::sync::Arc; use std::time::Duration; use uuid::Uuid; @@ -233,6 +919,26 @@ mod tests { intent } + fn risk_acceptance( + responsibility_id: Uuid, + expected_bucket_incarnation_id: Uuid, + acknowledge_unknown_source_incarnation: bool, + acknowledge_incarnation_mismatch: bool, + reason: &str, + reference: &str, + ) -> MrfLegacyRiskAcceptanceRequest { + MrfLegacyRiskAcceptanceRequest { + responsibility_id, + expected_bucket_incarnation_id, + acknowledge_unknown_source_incarnation, + acknowledge_incarnation_mismatch, + actor: "operator-a".to_string(), + reason: reason.to_string(), + reference: reference.to_string(), + request_id: Uuid::new_v4(), + } + } + #[test] fn partial_write_retention_bounds_count_and_bytes_without_evicting_responsibility() { let first = intent("a"); @@ -258,26 +964,31 @@ mod tests { let now = Instant::now(); assert!(writes.ready_keys(now, 2).is_empty(), "uncommitted responsibility must not be dispatched"); writes.mark_persisted(); - let first: Vec<_> = writes.ready_keys(now, 2).into_iter().map(|key| key.object).collect(); + let first: Vec<_> = writes.ready_keys(now, 2).into_iter().map(|key| key.identity.object).collect(); assert_eq!(first, vec![Arc::::from("a"), Arc::::from("b")]); let second = writes.ready_keys(now, 2); assert_eq!( - second[0].object.as_ref(), + second[0].identity.object.as_ref(), "c", "the old prefix becoming due again cannot starve the next member" ); } #[test] - fn unverified_legacy_responsibility_is_held_until_restart_without_being_released() { + fn unverified_legacy_without_matching_lifecycle_checkpoint_rechecks_after_restart() { let original = intent("legacy"); + let incarnation = Uuid::new_v4(); let mut writes = PartialWrites::default(); writes - .admit(original.clone(), 1, 8192) + .admit_with_source_incarnation(original.clone(), Some(incarnation), 1, 8192) .expect("durable intent should be retained"); writes.mark_persisted(); - let anchor = MrfDurableRepairAnchor::from_intent(&original, Uuid::new_v4()).expect("exact durable anchor"); - writes.entries.get_mut(&queue_key(&original)).expect("resident intent").anchor = Some(anchor.clone()); + let anchor = MrfDurableRepairAnchor::from_intent(&original, incarnation).expect("exact durable anchor"); + writes + .entries + .get_mut(&PartialWriteKey::new(&original, Some(incarnation))) + .expect("resident intent") + .anchor = Some(anchor.clone()); assert!(writes.park_unverified_legacy(&anchor)); assert!(!writes.park_unverified_legacy(&anchor), "duplicate notices are idempotent"); @@ -288,7 +999,7 @@ mod tests { let replacement = intent("legacy"); writes - .admit(replacement, 1, 8192) + .admit_with_source_incarnation(replacement, Some(incarnation), 1, 8192) .expect("a new generation should become retryable"); writes.mark_persisted(); assert_eq!(writes.unverified_legacy_count(), 0); @@ -296,13 +1007,481 @@ mod tests { let mut restarted = PartialWrites::default(); restarted - .admit(original, 1, 8192) + .admit_with_source_incarnation(original, Some(incarnation), 1, 8192) .expect("the unchanged journal re-arms the same responsibility after restart"); restarted.mark_persisted(); assert_eq!(restarted.unverified_legacy_count(), 0); assert_eq!(restarted.ready_keys(Instant::now() + Duration::from_secs(60), 1).len(), 1); } + #[test] + fn lifecycle_checkpoint_restores_hold_and_risk_acceptance_without_claiming_proof() { + let original = intent("legacy-lifecycle"); + let incarnation = Uuid::new_v4(); + let mut writes = PartialWrites::default(); + writes + .admit_with_source_incarnation(original.clone(), Some(incarnation), 1, 8192) + .expect("retain durable intent"); + writes.mark_persisted(); + let responsibility_id = writes + .entries + .get(&PartialWriteKey::new(&original, Some(incarnation))) + .expect("resident responsibility") + .responsibility_id; + let anchor = MrfDurableRepairAnchor::from_intent(&original, incarnation).expect("exact durable anchor"); + writes + .entries + .get_mut(&PartialWriteKey::new(&original, Some(incarnation))) + .expect("resident responsibility") + .anchor = Some(anchor.clone()); + assert!(writes.park_unverified_legacy(&anchor)); + + let checkpoint = writes + .checkpoint_records(|_| Some([7; 32])) + .into_iter() + .next() + .expect("hold state checkpoint"); + let mut restored = PartialWrites::default(); + restored + .admit_with_source_incarnation(original, Some(incarnation), 1, 8192) + .expect("replay journal intent before applying lifecycle state"); + assert!(restored.restore_state(&intent("legacy-lifecycle"), &checkpoint)); + restored.mark_persisted(); + assert_eq!(restored.unverified_legacy_count(), 1); + assert!(restored.ready_keys(Instant::now() + Duration::from_secs(60), 1).is_empty()); + + let (previous, _) = restored + .record_operator_acceptance( + 8192, + risk_acceptance( + responsibility_id, + incarnation, + true, + false, + "Reviewed against trusted backup; accepting unresolved identity risk", + "INC-1234", + ), + ) + .expect("explicit risk acceptance"); + assert!(matches!(previous, ResponsibilityState::HeldUnverifiedLegacy { .. })); + assert_eq!(restored.unverified_legacy_count(), 0); + assert_eq!(restored.operator_accepted_unverified_count(), 1); + let accepted_entry = restored + .entries + .get(&PartialWriteKey::new(&intent("legacy-lifecycle"), Some(incarnation))) + .expect("accepted entry"); + assert_eq!( + restored.bytes(), + PartialWrites::cost_with_state_and_audit( + &accepted_entry.intent, + &accepted_entry.state, + accepted_entry.last_operator_acceptance.as_ref(), + ) + ); + let (mut items, next_cursor) = restored.list_unverified_legacy(None, 10); + let item = items.pop().expect("operator state remains visible"); + assert!(next_cursor.is_none()); + assert_eq!(item.status, "operator_accepted_unverified"); + + let (_, accepted) = restored + .recheck(responsibility_id, incarnation) + .expect("explicit targeted recheck"); + assert!(matches!(accepted, ResponsibilityState::OperatorAcceptedUnverified { .. })); + assert_eq!(restored.operator_accepted_unverified_count(), 0); + let active_entry = restored + .entries + .get(&PartialWriteKey::new(&intent("legacy-lifecycle"), Some(incarnation))) + .expect("active entry"); + assert_eq!( + restored.bytes(), + PartialWrites::cost_with_state_and_audit( + &active_entry.intent, + &active_entry.state, + active_entry.last_operator_acceptance.as_ref(), + ), + "recheck must continue accounting for the retained audit record" + ); + restored.mark_persisted(); + assert_eq!(restored.ready_keys(Instant::now() + Duration::from_secs(60), 1).len(), 1); + } + + #[test] + fn restored_held_entry_can_be_refreshed_and_retry_index_cleanup_is_linear() { + let original = intent("refresh-after-recreate"); + let original_incarnation = Uuid::new_v4(); + let recreated_incarnation = Uuid::new_v4(); + let mut writes = PartialWrites::default(); + writes + .admit_with_source_incarnation(original.clone(), Some(original_incarnation), 1, 8192) + .expect("durable intent should be retained"); + writes.mark_persisted(); + let anchor = MrfDurableRepairAnchor::from_intent(&original, original_incarnation).expect("anchor"); + writes + .entries + .get_mut(&PartialWriteKey::new(&original, Some(original_incarnation))) + .expect("entry") + .anchor = Some(anchor.clone()); + assert!(writes.park_unverified_legacy(&anchor)); + let checkpoint = writes.checkpoint_records(|_| Some([4; 32])).remove(0); + + let mut restored = PartialWrites::default(); + restored + .admit_with_source_incarnation(original.clone(), Some(original_incarnation), 1, 8192) + .expect("replay creates retry index entry"); + assert!(restored.restore_state(&original, &checkpoint)); + let (previous, _) = restored + .replace_observed_incarnation( + checkpoint.responsibility_id, + original_incarnation, + recreated_incarnation, + Some(original_incarnation), + ) + .expect("refresh should reclassify the changed generation"); + assert!(matches!( + restored.intent_for_responsibility(checkpoint.responsibility_id), + Some((_, ResponsibilityState::BucketIncarnationChanged { observed_bucket_incarnation_id, .. }, _)) + if observed_bucket_incarnation_id == recreated_incarnation + )); + assert!( + restored + .retry_index + .contains(&PartialWriteKey::new(&original, Some(original_incarnation))) + ); + assert_eq!(restored.retry_order.len(), 1, "restoration must not scan/reinsert the full retry queue"); + restored.restore_operator_state(checkpoint.responsibility_id, previous, None); + assert!( + restored + .retry_index + .contains(&PartialWriteKey::new(&original, Some(original_incarnation))) + ); + assert_eq!(restored.retry_order.len(), 1, "rollback leaves one lazily cleaned queue key"); + } + + #[test] + fn lifecycle_actions_reject_stale_generation_and_incarnation() { + let original = intent("legacy-stale"); + let incarnation = Uuid::new_v4(); + let mut writes = PartialWrites::default(); + writes + .admit_with_source_incarnation(original.clone(), Some(incarnation), 1, 8192) + .expect("retain durable intent"); + let anchor = MrfDurableRepairAnchor::from_intent(&original, incarnation).expect("exact durable anchor"); + writes + .entries + .get_mut(&PartialWriteKey::new(&original, Some(incarnation))) + .expect("resident responsibility") + .anchor = Some(anchor.clone()); + assert!(writes.park_unverified_legacy(&anchor)); + let id = writes + .entries + .get(&PartialWriteKey::new(&original, Some(incarnation))) + .expect("entry") + .responsibility_id; + + assert!( + writes + .record_operator_acceptance(8192, risk_acceptance(id, Uuid::new_v4(), true, false, "reason", "INC-1234")) + .is_err() + ); + assert!( + writes + .record_operator_acceptance(8192, risk_acceptance(Uuid::new_v4(), incarnation, true, false, "reason", "INC-1234")) + .is_err() + ); + assert!( + writes + .record_operator_acceptance(8192, risk_acceptance(id, incarnation, true, false, " ", "INC-1234")) + .is_err() + ); + assert_eq!(writes.unverified_legacy_count(), 1); + } + + #[test] + fn source_bucket_incarnation_mismatch_is_parked_until_explicit_risk_ack() { + let source = Uuid::new_v4(); + let observed = Uuid::new_v4(); + let item = intent("recreated-bucket"); + let mut writes = PartialWrites::default(); + writes + .admit_with_source_incarnation(item.clone(), Some(source), 1, 8192) + .expect("retain generation-bound durable intent"); + let key = PartialWriteKey::new(&item, Some(source)); + let entry = writes.entries.get_mut(&key).expect("resident intent"); + entry.state = ResponsibilityState::BucketIncarnationChanged { + source_bucket_incarnation_id: source, + observed_bucket_incarnation_id: observed, + detected_at_ms: unix_now_ms(), + }; + let responsibility_id = entry.responsibility_id; + + assert!( + writes + .record_operator_acceptance( + 8192, + risk_acceptance( + responsibility_id, + observed, + false, + false, + "The old bucket generation is no longer available", + "INC-5678", + ), + ) + .is_err() + ); + let (accepted, _) = writes + .record_operator_acceptance( + 8192, + risk_acceptance( + responsibility_id, + observed, + false, + true, + "The old bucket generation is no longer available", + "INC-5678", + ), + ) + .expect("generation mismatch requires and accepts explicit acknowledgment"); + assert!(matches!(accepted, ResponsibilityState::BucketIncarnationChanged { .. })); + let (items, _) = writes.list_unverified_legacy(None, 10); + let item = items.into_iter().next().expect("mismatch disposition remains visible"); + assert_eq!(item.source_bucket_incarnation_id, Some(source)); + assert!(item.accepted.is_some_and(|audit| audit.acknowledged_incarnation_mismatch)); + } + + #[tokio::test(flavor = "multi_thread", worker_threads = 2)] + #[serial] + async fn same_key_bucket_incarnations_keep_separate_lifecycle_records_through_replay() { + use crate::heal::mrf_queue::{MrfConsumerConfig, MrfQueue, MrfRuntime}; + + let env = rustfs_test_utils::TestECStoreEnv::builder() + .prefix("rustfs_mrf_same_key_incarnation_replay") + .build() + .await; + let bucket = "partial-write-retention"; + env.make_bucket(bucket, false).await; + let storage: Arc = Arc::new(ECStoreHealStorage::new(env.ecstore.clone())); + let manager = Arc::new(HealManager::new_without_root_recovery_for_test(storage, None)); + let disks = super::super::journal_disks().await; + assert!(!disks.is_empty(), "test environment must register local MRF disks"); + + let old_incarnation = Uuid::new_v4(); + let current_incarnation = env.ecstore.pools[0] + .get_disks(0) + .bucket_incarnation_id_from_disk(bucket) + .await + .expect("current bucket incarnation must be available"); + assert_ne!(old_incarnation, current_incarnation); + + let old_intent = intent("same-object"); + let mut writes = PartialWrites::default(); + writes + .admit_with_source_incarnation(old_intent.clone(), Some(old_incarnation), 2, 64 * 1024) + .expect("retain the old bucket's partial-write responsibility"); + let old_key = PartialWriteKey::new(&old_intent, Some(old_incarnation)); + let old_entry = writes.entries.get(&old_key).expect("old responsibility should be resident"); + let old_id = old_entry.responsibility_id; + let retained_old_intent = old_entry.intent.clone(); + let old_anchor = + super::super::MrfDurableRepairAnchor::from_intent(&retained_old_intent, old_incarnation).expect("old proof anchor"); + writes + .entries + .get_mut(&old_key) + .expect("old responsibility remains resident") + .anchor = Some(old_anchor.clone()); + assert!(writes.park_unverified_legacy(&old_anchor)); + let old_request_id = Uuid::new_v4(); + writes + .record_operator_acceptance( + 64 * 1024, + MrfLegacyRiskAcceptanceRequest { + responsibility_id: old_id, + expected_bucket_incarnation_id: old_incarnation, + acknowledge_unknown_source_incarnation: false, + acknowledge_incarnation_mismatch: false, + actor: "integration-test-operator".to_string(), + reason: "Keep the old bucket generation's disposition attached to its own responsibility".to_string(), + reference: "TEST-ISSUE-8192-G1".to_string(), + request_id: old_request_id, + }, + ) + .expect("old-generation risk disposition should be recorded"); + + let current_intent = intent("same-object"); + assert_eq!( + super::super::intent_digest(&retained_old_intent), + super::super::intent_digest(¤t_intent), + "the two bucket incarnations must exercise the same journal identity digest" + ); + writes + .admit_with_source_incarnation(current_intent.clone(), Some(current_incarnation), 2, 64 * 1024) + .expect("retain the replacement bucket's separate responsibility"); + writes.mark_persisted(); + assert_eq!(writes.depth(), 2, "different bucket incarnations are distinct responsibilities"); + let current_key = PartialWriteKey::new(¤t_intent, Some(current_incarnation)); + let current_entry = writes + .entries + .get(¤t_key) + .expect("current responsibility should be resident"); + let current_id = current_entry.responsibility_id; + assert_ne!(old_id, current_id); + assert!(matches!(current_entry.state, ResponsibilityState::Active)); + assert!(current_entry.last_operator_acceptance.is_none()); + assert!( + !writes.park_unverified_legacy(&old_anchor), + "a delayed G1 hold event must not alter the G2 responsibility" + ); + assert!(matches!( + writes.intent_for_responsibility(current_id), + Some((_, ResponsibilityState::Active, Some(source))) if source == current_incarnation + )); + + let mut journal = Vec::new(); + for pending in writes.intents() { + assert!( + super::super::encode_intent(pending, &mut journal), + "both generations must fit the journal" + ); + } + let (decoded_intents, truncated) = super::super::decode_journal(&journal); + assert_eq!(truncated, 0); + assert_eq!(decoded_intents.len(), 2, "the journal must retain both incarnations"); + assert_eq!( + super::super::intent_digest(&decoded_intents[0]), + super::super::intent_digest(&decoded_intents[1]) + ); + + let owner = Uuid::new_v4(); + let sequence = 7; + let config = MrfConsumerConfig::default(); + let lifecycle_limit = config.journal_max_bytes.saturating_mul(4); + let records = writes.checkpoint_records(super::super::intent_digest); + assert_eq!(records.len(), 2); + assert_eq!(records[0].intent_digest, records[1].intent_digest); + let lifecycle = super::super::encode_mrf_lifecycle_checkpoint(owner, sequence, records, lifecycle_limit) + .expect("same-key generations need distinct lifecycle records"); + super::super::snapshot::publish_committed_snapshot_with_companion( + &disks, + owner, + sequence, + &journal, + config.journal_max_bytes, + Some((&super::super::MRF_LIFECYCLE_PATHS, &lifecycle, lifecycle_limit)), + ) + .await + .expect("publish journal and both lifecycle records together"); + + let mut queue = MrfQueue::new(config.queue_capacity, config.journal_max_bytes); + let replay = super::super::replay_into(&manager, &mut queue, &mut None).await; + assert_eq!(replay.replayed, 2); + assert_eq!(queue.depth(), 0, "durable generations bypass ordinary MRF coalescing"); + assert_eq!(replay.partial_writes.len(), 2, "replay must preserve both responsibilities"); + + let mut runtime = MrfRuntime { + partial_writes: PartialWrites::default(), + queue, + config, + checkpoint_owner: Uuid::new_v4(), + next_checkpoint_sequence: replay.next_checkpoint_sequence, + new_since_flush: 0, + dirty: false, + journal_on_disk: replay.journal_on_disk, + retain_replay_journal: replay.retain_journal_for_replay, + durable_replay_anchors: replay.durable_replay_anchors, + replay_cleanup: replay.cleanup, + runtime_checkpoint: None, + backoff_until: None, + }; + runtime.adopt_replayed_partial_writes(replay.partial_writes); + assert_eq!(runtime.partial_writes.depth(), 2); + assert!(matches!( + runtime.partial_writes.intent_for_responsibility(old_id), + Some((_, ResponsibilityState::OperatorAcceptedUnverified { request_id, .. }, Some(source))) + if source == old_incarnation && request_id == old_request_id + )); + assert!(matches!( + runtime.partial_writes.intent_for_responsibility(current_id), + Some((_, ResponsibilityState::Active, Some(source))) if source == current_incarnation + )); + assert!(runtime.partial_writes.last_operator_acceptance(current_id).is_none()); + + let replayed_old_intent = runtime + .partial_writes + .intent_for_responsibility(old_id) + .expect("old lifecycle responsibility should remain addressable") + .0; + let replayed_current_intent = runtime + .partial_writes + .intent_for_responsibility(current_id) + .expect("current lifecycle responsibility should remain addressable") + .0; + let replayed_old_anchor = super::super::MrfDurableRepairAnchor::from_intent(&replayed_old_intent, old_incarnation) + .expect("replayed old-generation proof anchor"); + let replayed_current_anchor = + super::super::MrfDurableRepairAnchor::from_intent(&replayed_current_intent, current_incarnation) + .expect("replayed current-generation proof anchor"); + runtime + .partial_writes + .entries + .get_mut(&PartialWriteKey::new(&replayed_old_intent, Some(old_incarnation))) + .expect("old-generation entry after replay") + .anchor = Some(replayed_old_anchor.clone()); + runtime + .partial_writes + .entries + .get_mut(&PartialWriteKey::new(&replayed_current_intent, Some(current_incarnation))) + .expect("current-generation entry after replay") + .anchor = Some(replayed_current_anchor.clone()); + + let verified_event = |anchor: &super::super::MrfDurableRepairAnchor| MrfVerifiedRepairEvent { + kind: anchor.kind, + bucket: anchor.bucket.clone(), + object: anchor.object.clone(), + version_id: anchor.version_id, + scope: anchor.scope, + delete_marker_purge: anchor.delete_marker_purge, + lease: Some(anchor.lease), + bucket_incarnation_id: anchor.bucket_incarnation_id, + disposition: MrfVerifiedRepairDisposition::Repaired, + }; + rustfs_common::mrf_channel::note_mrf_verified_repair(verified_event(&replayed_old_anchor)); + runtime.discharge_durable_replay_anchors(); + assert!(runtime.partial_writes.intent_for_responsibility(old_id).is_none()); + assert!( + runtime.partial_writes.intent_for_responsibility(current_id).is_some(), + "a matching G1 proof must not release G2" + ); + rustfs_common::mrf_channel::note_mrf_verified_repair(verified_event(&replayed_current_anchor)); + runtime.discharge_durable_replay_anchors(); + assert!(runtime.partial_writes.intent_for_responsibility(current_id).is_none()); + } + + #[test] + fn lifecycle_listing_uses_stable_bounded_cursors() { + let mut writes = PartialWrites::default(); + for (index, object) in ["a", "b", "c", "d"].into_iter().enumerate() { + let item = intent(object); + let incarnation = Uuid::from_u128(100 + u128::try_from(index).expect("small test index")); + writes + .admit_with_source_incarnation(item.clone(), Some(incarnation), 4, 8192) + .expect("retain durable intent"); + let key = PartialWriteKey::new(&item, Some(incarnation)); + let entry = writes.entries.get_mut(&key).expect("resident intent"); + entry.responsibility_id = Uuid::from_u128(u128::try_from(index + 1).expect("small test index")); + let anchor = MrfDurableRepairAnchor::from_intent(&item, incarnation).expect("exact durable anchor"); + entry.anchor = Some(anchor.clone()); + assert!(writes.park_unverified_legacy(&anchor)); + } + + let (first, cursor) = writes.list_unverified_legacy(None, 2); + assert_eq!(first.len(), 2); + let cursor = cursor.expect("more entries should expose a continuation cursor"); + let (second, next_cursor) = writes.list_unverified_legacy(Some(cursor), 2); + assert_eq!(second.len(), 2); + assert!(next_cursor.is_none(), "the last page must omit its cursor"); + assert!(first.last().expect("first page").responsibility_id < second[0].responsibility_id); + } + #[test] fn partial_write_retention_adopts_replay_anchor_before_generation_replacement() { use super::super::{MrfQueue, MrfRuntime}; @@ -324,14 +1503,14 @@ mod tests { runtime_checkpoint: None, backoff_until: None, }; - runtime.adopt_replayed_partial_writes(vec![old]); + runtime.adopt_replayed_partial_writes(vec![(old, None)]); assert!( runtime.durable_replay_anchors.is_empty(), "the live record must be the single proof owner" ); assert!(runtime.retained_replay_journal(), "adoption cannot release the startup checkpoint"); runtime - .admit_partial_write(intent("replayed")) + .admit_partial_write(intent("replayed"), None) .expect("new generation should be retained"); assert!( runtime.retained_replay_journal(), @@ -345,7 +1524,7 @@ mod tests { fn partial_write_retention_new_generation_rejects_old_proof_and_requires_checkpoint() { let mut writes = PartialWrites::default(); let first = intent("same-key"); - let key = queue_key(&first); + let key = PartialWriteKey::new(&first, None); let incarnation = Uuid::new_v4(); let old_anchor = MrfDurableRepairAnchor::from_intent(&first, incarnation).expect("first anchor should be complete"); writes.admit(first, 1, 4096).expect("first write should fit"); diff --git a/crates/heal/src/heal/mrf_queue/snapshot.rs b/crates/heal/src/heal/mrf_queue/snapshot.rs index a5514df5e..15144a0e9 100644 --- a/crates/heal/src/heal/mrf_queue/snapshot.rs +++ b/crates/heal/src/heal/mrf_queue/snapshot.rs @@ -365,6 +365,19 @@ pub async fn publish_committed_snapshot( sequence: u64, payload: &[u8], limit: usize, +) -> Result { + publish_committed_snapshot_with_companion(disks, owner, sequence, payload, limit, None).await +} + +/// Publish a companion lifecycle record on the same disk before the commit +/// manifest. A replica is committed only after both payloads have been stored. +pub async fn publish_committed_snapshot_with_companion( + disks: &[EcstoreDiskStore], + owner: Uuid, + sequence: u64, + payload: &[u8], + limit: usize, + companion: Option<(&[&str; 2], &[u8], usize)>, ) -> Result { if disks.is_empty() { return Err(SnapshotError::NoWritableReplica); @@ -413,6 +426,18 @@ pub async fn publish_committed_snapshot( continue; } } + if let Some((companion_paths, companion_bytes, companion_limit)) = companion { + match cas_replace(disk, companion_paths[slot], companion_bytes, companion_limit).await { + Ok(EcstoreConditionalFileUpdate::Updated) => {} + Ok(EcstoreConditionalFileUpdate::Missing | EcstoreConditionalFileUpdate::Mismatch) => continue, + Err(error) => { + if first_error.is_none() { + first_error = Some(error); + } + continue; + } + } + } match cas_replace_expected(disk, MANIFEST_PATHS[slot], expected_manifest.map(EcstoreDiskBytes::from), &manifest).await { Ok(EcstoreConditionalFileUpdate::Updated) => manifest_replicas += 1, Ok(EcstoreConditionalFileUpdate::Missing | EcstoreConditionalFileUpdate::Mismatch) => {} @@ -436,6 +461,80 @@ pub async fn publish_committed_snapshot( }) } +/// Read a companion only from replicas that also contain its matching +/// committed checkpoint. Differing companions for the same checkpoint fail +/// closed rather than selecting one replica arbitrarily. +pub async fn read_committed_companion( + disks: &[EcstoreDiskStore], + owner: Uuid, + sequence: u64, + companion_paths: &[&str; 2], + checkpoint_limit: usize, + companion_limit: usize, +) -> Result>, SnapshotError> { + let mut selected: Option> = None; + let mut first_error = None; + for disk in disks { + let mut checkpoint_slot = None; + for (slot, (manifest_path, payload_path)) in MANIFEST_PATHS.into_iter().zip(PAYLOAD_PATHS).enumerate() { + let manifest_bytes = match read_bounded(disk, manifest_path, MANIFEST_LEN).await { + Ok(Some(bytes)) => bytes, + Ok(None) => continue, + Err(error @ SnapshotError::Unsupported) => return Err(error), + Err(error) => { + if first_error.is_none() { + first_error = Some(error); + } + continue; + } + }; + let manifest = match Manifest::decode(&manifest_bytes, checkpoint_limit) { + Ok(manifest) if manifest.owner == owner && manifest.sequence == sequence => manifest, + Ok(_) | Err(SnapshotError::Corrupt) | Err(SnapshotError::TooLarge) => continue, + Err(error) => return Err(error), + }; + let payload = match read_bounded(disk, payload_path, manifest.payload_len).await { + Ok(Some(payload)) => payload, + Ok(None) => continue, + Err(error) => { + if first_error.is_none() { + first_error = Some(error); + } + continue; + } + }; + if CommittedSnapshot::decode(slot, &manifest_bytes, payload, checkpoint_limit).is_ok() { + checkpoint_slot = Some(slot); + break; + } + } + let Some(checkpoint_slot) = checkpoint_slot else { + continue; + }; + let companion = match read_bounded(disk, companion_paths[checkpoint_slot], companion_limit).await { + Ok(Some(companion)) => companion, + Ok(None) => continue, + Err(error) => { + if first_error.is_none() { + first_error = Some(error); + } + continue; + } + }; + if selected.as_ref().is_some_and(|existing| existing != &companion) { + return Err(SnapshotError::Conflict); + } + selected = Some(companion); + } + if selected.is_some() { + Ok(selected) + } else if let Some(error) = first_error { + Err(error) + } else { + Ok(None) + } +} + /// Reclaim the slot superseded by an already committed checkpoint. /// /// This is a narrow cleanup primitive: it first reads back the current @@ -842,6 +941,81 @@ mod tests { } } + #[tokio::test] + async fn lifecycle_companion_is_read_only_with_its_matching_checkpoint_replica() { + let root = TempDir::new().expect("test directory"); + let checkpoint_disk = disk(&root, "checkpoint").await; + let sidecar_disk = disk(&root, "sidecar").await; + let owner = Uuid::new_v4(); + let checkpoint_payload = payload("responsibility"); + let sidecar = b"lifecycle state"; + commit(&checkpoint_disk, 0, owner, 7, &checkpoint_payload).await; + install(&sidecar_disk, ".heal-mrf-lifecycle.0.bin", sidecar).await; + + let absent = read_committed_companion( + &[checkpoint_disk.clone(), sidecar_disk.clone()], + owner, + 7, + &[".heal-mrf-lifecycle.0.bin", ".heal-mrf-lifecycle.1.bin"], + 4096, + 4096, + ) + .await + .expect("unpaired lifecycle sidecar is ignored"); + assert!(absent.is_none()); + + publish_committed_snapshot_with_companion( + &[checkpoint_disk.clone(), sidecar_disk.clone()], + owner, + 8, + &checkpoint_payload, + 4096, + Some((&[".heal-mrf-lifecycle.0.bin", ".heal-mrf-lifecycle.1.bin"], sidecar, 4096)), + ) + .await + .expect("paired checkpoint and lifecycle state publish"); + let paired = read_committed_companion( + &[checkpoint_disk, sidecar_disk], + owner, + 8, + &[".heal-mrf-lifecycle.0.bin", ".heal-mrf-lifecycle.1.bin"], + 4096, + 4096, + ) + .await + .expect("read paired lifecycle sidecar") + .expect("paired sidecar exists"); + assert_eq!(paired, sidecar); + } + + #[tokio::test] + async fn lifecycle_companion_recovers_from_a_healthy_replica_after_peer_read_error() { + let root = TempDir::new().expect("test directory"); + let failing_disk = disk(&root, "failing").await; + let healthy_disk = disk(&root, "healthy").await; + let owner = Uuid::new_v4(); + let checkpoint_payload = payload("responsibility"); + let sidecar = b"operator audit and source incarnation"; + commit(&healthy_disk, 0, owner, 9, &checkpoint_payload).await; + install(&healthy_disk, ".heal-mrf-lifecycle.0.bin", sidecar).await; + std::fs::create_dir(root.path().join("failing").join(RUSTFS_META_BUCKET).join(MANIFEST_PATHS[0])) + .expect("simulate one replica read error"); + + let recovered = read_committed_companion( + &[failing_disk, healthy_disk], + owner, + 9, + &[".heal-mrf-lifecycle.0.bin", ".heal-mrf-lifecycle.1.bin"], + 4096, + 4096, + ) + .await + .expect("one unreadable replica must not hide a healthy paired sidecar") + .expect("healthy paired sidecar must be recovered"); + + assert_eq!(recovered, sidecar); + } + #[tokio::test] async fn divergent_commits_at_same_sequence_fail_closed() { let root = TempDir::new().expect("test directory"); diff --git a/crates/heal/src/heal/task.rs b/crates/heal/src/heal/task.rs index 87f038e30..55a999e3f 100644 --- a/crates/heal/src/heal/task.rs +++ b/crates/heal/src/heal/task.rs @@ -323,6 +323,10 @@ pub struct HealRequest { pub heal_type: HealType, /// Admission identity for an explicit administrator bucket heal. Never rebound on replay. pub bucket_incarnation_id: Option, + /// Source bucket generation captured for a durable MRF object repair. + /// Unlike the admin admission fence above, this must survive queueing and + /// retries so execution cannot target a later bucket incarnation. + pub expected_mrf_bucket_incarnation_id: Option, /// Heal options pub options: HealOptions, /// Priority @@ -351,6 +355,7 @@ impl HealRequest { id: Uuid::new_v4().to_string(), heal_type, bucket_incarnation_id: None, + expected_mrf_bucket_incarnation_id: None, options, priority, source: HealRequestSource::Internal, @@ -417,6 +422,7 @@ pub struct HealTask { /// Heal type pub heal_type: HealType, pub bucket_incarnation_id: Option, + pub expected_mrf_bucket_incarnation_id: Option, /// Heal options pub options: HealOptions, /// Priority inherited from the request @@ -494,6 +500,7 @@ impl HealTask { id: request.id, heal_type: request.heal_type, bucket_incarnation_id: request.bucket_incarnation_id, + expected_mrf_bucket_incarnation_id: request.expected_mrf_bucket_incarnation_id, options: request.options, priority: request.priority, source: request.source, @@ -530,6 +537,7 @@ impl HealTask { id: self.id.clone(), heal_type: self.heal_type.clone(), bucket_incarnation_id: self.bucket_incarnation_id, + expected_mrf_bucket_incarnation_id: self.expected_mrf_bucket_incarnation_id, options: self.options.clone(), priority: self.priority, source: self.source, diff --git a/crates/heal/src/heal/task/heal_object.rs b/crates/heal/src/heal/task/heal_object.rs index 239333f4e..81d22ae7e 100644 --- a/crates/heal/src/heal/task/heal_object.rs +++ b/crates/heal/src/heal/task/heal_object.rs @@ -551,12 +551,22 @@ impl HealTask { set: self.options.set_index, }; let mut expected = self.outcome_identity(bucket, object, version_id, self.options.pool_index, self.options.set_index); - let bucket_incarnation_id = self + let current_bucket_incarnation_id = self .outcome_bucket_incarnation_id(bucket, self.options.dry_run) .await? .ok_or_else(|| Error::TaskExecutionFailed { message: format!("Missing bucket incarnation for durable MRF repair {bucket}/{object}"), })?; + let bucket_incarnation_id = self + .expected_mrf_bucket_incarnation_id + .ok_or_else(|| Error::TaskExecutionFailed { + message: format!("Missing source bucket incarnation for durable MRF repair {bucket}/{object}"), + })?; + if current_bucket_incarnation_id != bucket_incarnation_id { + return Err(Error::TaskExecutionFailed { + message: format!("Bucket incarnation changed before durable MRF repair {bucket}/{object}"), + }); + } expected.bucket_incarnation_id = Some(bucket_incarnation_id); let storage_result = self diff --git a/crates/heal/src/heal/task/tests.rs b/crates/heal/src/heal/task/tests.rs index b79555201..500903e93 100644 --- a/crates/heal/src/heal/task/tests.rs +++ b/crates/heal/src/heal/task/tests.rs @@ -2678,6 +2678,25 @@ async fn read_repair_object_heal_sets_read_repair_option() { assert!(!opts[0].no_lock); } +#[tokio::test] +async fn durable_mrf_heal_rejects_a_recreated_bucket_before_storage_heal() { + let original_incarnation = Uuid::new_v4(); + let recreated_incarnation = Uuid::new_v4(); + let storage = Arc::new(MockStorage { + bucket_incarnation_id: Mutex::new(Some(recreated_incarnation)), + ..Default::default() + }); + let mut request = HealRequest::object("bucket".to_string(), "object".to_string(), None); + request.source = HealRequestSource::Mrf; + request.expected_mrf_bucket_incarnation_id = Some(original_incarnation); + let task = HealTask::from_request(request, storage.clone()); + + let result = task.heal_object("bucket", "object", None).await; + + assert!(matches!(result, Err(Error::TaskExecutionFailed { .. }))); + assert!(storage.heal_object_calls.lock().unwrap().is_empty()); +} + #[tokio::test(start_paused = true)] async fn read_repair_object_heal_is_not_failed_by_flat_task_timeout() { let storage = Arc::new(MockStorage { @@ -4002,6 +4021,7 @@ async fn mrf_recreate_missing_object_records_exact_absence_receipt_with_scope() HealPriority::Normal, ); request.source = HealRequestSource::Mrf; + request.expected_mrf_bucket_incarnation_id = Some(incarnation); let task = HealTask::from_request(request, storage.clone()); task.execute() diff --git a/crates/heal/tests/mrf_partial_write_test.rs b/crates/heal/tests/mrf_partial_write_test.rs index 440fb4bf1..c522ea130 100644 --- a/crates/heal/tests/mrf_partial_write_test.rs +++ b/crates/heal/tests/mrf_partial_write_test.rs @@ -91,6 +91,201 @@ async fn partial_write_persistence_failure_is_reported_and_retained_for_retry() ); } +#[test] +fn legacy_unbound_generation_is_parked_and_requires_explicit_risk_acceptance() { + const STACK_SIZE: usize = 8 * 1024 * 1024; + std::thread::Builder::new() + .name("mrf-unbound-lifecycle".to_owned()) + .stack_size(STACK_SIZE) + .spawn(|| { + let runtime = tokio::runtime::Builder::new_current_thread() + .thread_stack_size(STACK_SIZE) + .enable_all() + .build() + .expect("unbound lifecycle runtime should build"); + runtime.block_on(legacy_unbound_generation_is_parked_and_requires_explicit_risk_acceptance_inner()); + }) + .expect("unbound lifecycle test thread should spawn") + .join() + .expect("unbound lifecycle test thread should finish"); +} + +async fn legacy_unbound_generation_is_parked_and_requires_explicit_risk_acceptance_inner() { + use rustfs_common::mrf_channel::{MrfScope, persist_partial_write_intent}; + + temp_env::async_with_vars([("RUSTFS_HEAL_MRF_ENABLE", Some("true"))], async { + let root = tempfile::tempdir().expect("unbound lifecycle fixture directory"); + let env = TestECStoreEnv::builder().base_dir(root.path()).build().await; + env.make_bucket("unbound-lifecycle", false).await; + let manager = manager(&env); + manager.start().await.expect("manager should start"); + mrf_queue::spawn_mrf_consumer(manager.clone()); + persist_partial_write_intent( + "unbound-lifecycle", + "legacy.bin", + None, + MrfScope { + pool_index: 0, + set_index: 0, + }, + ) + .await + .expect("old-format intent should still be durably admitted"); + + assert!( + wait_until(|| async { + mrf_queue::list_legacy_responsibilities(None, 8).await.is_ok_and(|snapshot| { + snapshot + .responsibilities + .iter() + .any(|item| item.bucket == "unbound-lifecycle" && item.object == "legacy.bin") + }) + }) + .await, + "unbound old journal intent must become visible" + ); + let snapshot = mrf_queue::list_legacy_responsibilities(None, 8) + .await + .expect("listing should be available"); + let entry = snapshot + .responsibilities + .iter() + .find(|item| item.bucket == "unbound-lifecycle" && item.object == "legacy.bin") + .expect("visible generation-unknown entry"); + assert_eq!(entry.status, "legacy_generation_unknown"); + assert!(entry.source_bucket_incarnation_id.is_none()); + assert!( + snapshot_contains("legacy.bin").await, + "generation uncertainty must preserve journal responsibility" + ); + let operations = manager.operations_snapshot().await; + assert_eq!( + operations.queue_length, 0, + "unbound old intent must not be sent to the current bucket generation" + ); + + assert!(matches!( + mrf_queue::recheck_legacy_responsibility(entry.responsibility_id, entry.bucket_incarnation_id).await, + Err(rustfs_heal::heal::mrf_queue::MrfLifecycleControlError::InvalidAction(_)) + )); + mrf_queue::accept_unverified_legacy_risk(mrf_queue::MrfLegacyRiskAcceptanceRequest { + responsibility_id: entry.responsibility_id, + expected_bucket_incarnation_id: entry.bucket_incarnation_id, + acknowledge_unknown_source_incarnation: true, + acknowledge_incarnation_mismatch: false, + actor: "integration-test-operator".to_string(), + reason: "The old journal has no source bucket-generation binding".to_string(), + reference: "TEST-ISSUE-2682-UNBOUND".to_string(), + request_id: uuid::Uuid::new_v4(), + }) + .await + .expect("unbound risk disposition requires explicit acknowledgment and must be durable"); + assert!( + snapshot_contains("legacy.bin").await, + "risk acknowledgment must not delete the intent record" + ); + let accepted = mrf_queue::list_legacy_responsibilities(None, 8) + .await + .expect("accepted status should be readable"); + let accepted = accepted + .responsibilities + .iter() + .find(|item| item.responsibility_id == entry.responsibility_id) + .expect("accepted unbound generation remains visible"); + assert_eq!(accepted.status, "operator_accepted_unverified"); + assert!( + accepted + .accepted + .as_ref() + .is_some_and(|audit| audit.acknowledged_unknown_source_incarnation) + ); + manager.stop().await.expect("manager should stop"); + }) + .await; +} + +#[test] +fn unversioned_deleted_partial_write_is_discharged_by_an_absence_proof() { + const STACK_SIZE: usize = 8 * 1024 * 1024; + std::thread::Builder::new() + .name("mrf-partial-write-absence".to_owned()) + .stack_size(STACK_SIZE) + .spawn(|| { + let runtime = tokio::runtime::Builder::new_current_thread() + .thread_stack_size(STACK_SIZE) + .enable_all() + .build() + .expect("partial-write absence runtime should build"); + runtime.block_on(unversioned_deleted_partial_write_is_discharged_by_an_absence_proof_inner()); + }) + .expect("partial-write absence test thread should spawn") + .join() + .expect("partial-write absence test thread should finish"); +} + +async fn unversioned_deleted_partial_write_is_discharged_by_an_absence_proof_inner() { + use rustfs_common::mrf_channel::{MrfScope, persist_partial_write_intent_with_incarnation}; + + temp_env::async_with_vars([("RUSTFS_HEAL_MRF_ENABLE", Some("true"))], async { + let root = tempfile::tempdir().expect("partial-write absence fixture directory"); + let env = TestECStoreEnv::builder().disk_count(16).base_dir(root.path()).build().await; + env.make_bucket("partial-absence", false).await; + let mut coordinator_pool = env.endpoint_pools.as_ref()[0].clone(); + let mut endpoints = coordinator_pool.endpoints.as_ref().to_vec(); + for endpoint in endpoints.iter_mut().skip(4) { + endpoint.is_local = false; + } + coordinator_pool.endpoints = Endpoints::from(endpoints); + init_local_disks(EndpointServerPools::from(vec![coordinator_pool])) + .await + .expect("coordinator journal disks"); + + let manager = manager(&env); + mrf_queue::spawn_mrf_consumer(manager.clone()); + let source_bucket_incarnation_id = env.ecstore.pools[0] + .get_disks(0) + .bucket_incarnation_id_from_disk("partial-absence") + .await + .expect("fixture bucket source incarnation"); + for (object, version_id) in [ + ("deleted-unversioned.bin", None), + ("deleted-versioned.bin", Some(uuid::Uuid::new_v4())), + ] { + persist_partial_write_intent_with_incarnation( + "partial-absence", + object, + version_id, + MrfScope { + pool_index: 0, + set_index: 0, + }, + Some(source_bucket_incarnation_id), + ) + .await + .expect("durable partial-write responsibility must commit before scheduling"); + assert!(snapshot_contains(object).await, "committed responsibility must exist before repair runs"); + } + + manager.start().await.expect("MRF scheduler should start"); + for object in ["deleted-unversioned.bin", "deleted-versioned.bin"] { + assert!( + wait_until(|| async { !snapshot_contains(object).await }).await, + "complete absence proof must discharge the durable responsibility" + ); + } + assert!( + wait_until(|| async { + let snapshot = manager.operations_snapshot().await; + snapshot.queue_length == 0 && snapshot.active_tasks == 0 + }) + .await, + "discharged absence repair must leave no queued work" + ); + manager.stop().await.expect("absence manager should stop"); + }) + .await; +} + #[test] fn degraded_deleted_partial_write_is_discharged_by_an_absence_proof() { const STACK_SIZE: usize = 8 * 1024 * 1024; @@ -703,9 +898,12 @@ async fn partial_write_sigkill_replay_scenario(protected: bool) { mrf_queue::spawn_mrf_consumer(manager.clone()); assert!(snapshot_contains("crash.bin").await, "restart must find durable responsibility"); *set.disks.write().await = all.iter().cloned().map(Some).collect(); + let healed = wait_until(|| async { replicas(&all, "partial-crash", "crash.bin", None, false).await == 4 }).await; assert!( - wait_until(|| async { replicas(&all, "partial-crash", "crash.bin", None, false).await == 4 }).await, - "replayed responsibility must heal the returning member" + healed, + "replayed responsibility must heal the returning member; manager={:?}; responsibility={:?}", + manager.operations_snapshot().await, + mrf_queue::list_legacy_responsibilities(None, 16).await ); assert_payload(&env, "partial-crash", "crash.bin", None, b"durable partial write across SIGKILL").await; if protected { @@ -734,10 +932,158 @@ async fn partial_write_sigkill_replay_scenario(protected: bool) { while tokio::time::Instant::now() < retry_window { let snapshot = manager.operations_snapshot().await; assert_eq!(snapshot.queue_length, 0, "an unverified legacy result must not refill the manager queue"); - assert_eq!(snapshot.active_tasks, 0, "a held intent must not stay active"); + assert_eq!( + snapshot.active_tasks, + 0, + "a held intent must not stay active; lifecycle state: {:?}", + mrf_queue::list_legacy_responsibilities(None, 32) + .await + .expect("lifecycle state should be inspectable") + .responsibilities + ); tokio::time::sleep(Duration::from_millis(100)).await; } assert!(snapshot_contains("crash.bin").await, "unverified legacy responsibility must remain"); + + let before = mrf_queue::list_legacy_responsibilities(None, 32) + .await + .expect("held lifecycle listing should be available"); + let held = before + .responsibilities + .iter() + .find(|entry| entry.bucket == "partial-crash" && entry.object == "crash.bin") + .expect("legacy durable intent must be visible with its exact identity"); + assert_eq!(held.status, "held_unverified_legacy"); + assert_eq!( + held.source_bucket_incarnation_id, + Some( + env.ecstore.pools[0] + .get_disks(0) + .bucket_incarnation_id_from_disk("partial-crash") + .await + .expect("source bucket incarnation should remain stable") + ), + "the producer-bound bucket generation must survive journal replay" + ); + let responsibility_id = held.responsibility_id; + let bucket_incarnation_id = held.bucket_incarnation_id; + let request_id = uuid::Uuid::new_v4(); + mrf_queue::accept_unverified_legacy_risk(mrf_queue::MrfLegacyRiskAcceptanceRequest { + responsibility_id, + expected_bucket_incarnation_id: bucket_incarnation_id, + acknowledge_unknown_source_incarnation: true, + acknowledge_incarnation_mismatch: false, + actor: "integration-test-operator".to_string(), + reason: "The operator has accepted that legacy object identity cannot be proven automatically".to_string(), + reference: "TEST-ISSUE-2682".to_string(), + request_id, + }) + .await + .expect("explicit risk acceptance must persist before success"); + assert!(snapshot_contains("crash.bin").await, "risk acceptance must retain the MRF responsibility"); + let accepted = mrf_queue::list_legacy_responsibilities(None, 32) + .await + .expect("accepted lifecycle state should remain queryable"); + let accepted = accepted + .responsibilities + .iter() + .find(|entry| entry.responsibility_id == responsibility_id) + .expect("accepted responsibility must remain visible"); + assert_eq!(accepted.status, "operator_accepted_unverified"); + assert_eq!(accepted.accepted.as_ref().map(|audit| audit.request_id), Some(request_id)); + assert_eq!( + accepted.accepted.as_ref().map(|audit| audit.actor.as_str()), + Some("integration-test-operator") + ); + + manager + .stop() + .await + .expect("first manager should stop before restart verification"); + let log = std::fs::File::create(root.path().join("lifecycle-restart.log")).expect("restart child log"); + let mut child = Command::new(std::env::current_exe().expect("integration test executable")) + .args(["--exact", "mrf_legacy_lifecycle_restore_fixture", "--nocapture"]) + .env("RUSTFS_TEST_MRF_LIFECYCLE_ROOT", root.path()) + .env("RUSTFS_TEST_MRF_LIFECYCLE_ID", responsibility_id.to_string()) + .stdout(Stdio::from(log.try_clone().expect("clone child log"))) + .stderr(Stdio::from(log)) + .spawn() + .expect("lifecycle restart fixture should start"); + let restored = wait_until(|| async { root.path().join("lifecycle-restored").exists() }).await; + let status = child.wait().expect("lifecycle restart fixture should exit"); + assert!( + restored && status.success(), + "operator-accepted state should survive process restart: {}", + std::fs::read_to_string(root.path().join("lifecycle-restart.log")).expect("read child evidence") + ); + return; } manager.stop().await.expect("restarted manager should stop"); } + +#[test] +fn mrf_legacy_lifecycle_restore_fixture() { + const STACK_SIZE: usize = 8 * 1024 * 1024; + std::thread::Builder::new() + .name("mrf-lifecycle-restore".to_owned()) + .stack_size(STACK_SIZE) + .spawn(|| { + let runtime = tokio::runtime::Builder::new_current_thread() + .thread_stack_size(STACK_SIZE) + .enable_all() + .build() + .expect("lifecycle restore runtime should build"); + runtime.block_on(mrf_legacy_lifecycle_restore_fixture_inner()); + }) + .expect("lifecycle restore thread should spawn") + .join() + .expect("lifecycle restore thread should finish"); +} + +async fn mrf_legacy_lifecycle_restore_fixture_inner() { + let Ok(root) = std::env::var("RUSTFS_TEST_MRF_LIFECYCLE_ROOT") else { + return; + }; + let expected_id = std::env::var("RUSTFS_TEST_MRF_LIFECYCLE_ID") + .expect("lifecycle child responsibility ID") + .parse::() + .expect("valid lifecycle child responsibility ID"); + let root = std::path::PathBuf::from(root); + let env = TestECStoreEnv::builder().base_dir(&root).build().await; + let manager = manager(&env); + manager.start().await.expect("restarted heal manager should start"); + mrf_queue::spawn_mrf_consumer(manager.clone()); + assert!( + wait_until(|| async { + mrf_queue::list_legacy_responsibilities(None, 32).await.is_ok_and(|snapshot| { + snapshot + .responsibilities + .iter() + .any(|entry| entry.responsibility_id == expected_id) + }) + }) + .await, + "durable operator-accepted lifecycle state must replay" + ); + let restored = mrf_queue::list_legacy_responsibilities(None, 32) + .await + .expect("restored lifecycle listing should be available"); + let entry = restored + .responsibilities + .iter() + .find(|entry| entry.responsibility_id == expected_id) + .expect("restart listing matched the stable generation"); + assert_eq!(entry.status, "operator_accepted_unverified"); + assert_eq!( + entry.accepted.as_ref().map(|audit| audit.actor.as_str()), + Some("integration-test-operator") + ); + assert!( + snapshot_contains("crash.bin").await, + "risk-accepted responsibility remains in the durable journal" + ); + tokio::fs::write(root.join("lifecycle-restored"), b"restored") + .await + .expect("signal lifecycle restore"); + manager.stop().await.expect("restarted manager should stop"); +} diff --git a/crates/protos/src/heal_control.rs b/crates/protos/src/heal_control.rs index d197349a3..1189c0749 100644 --- a/crates/protos/src/heal_control.rs +++ b/crates/protos/src/heal_control.rs @@ -153,6 +153,7 @@ impl StartCommand { bucket: self.bucket, object_prefix: self.object_prefix, object_version_id: self.object_version_id, + expected_bucket_incarnation_id: None, force_start: self.force_start, priority: self.priority.into(), pool_index: self diff --git a/docs/architecture/compat-cleanup-register.md b/docs/architecture/compat-cleanup-register.md index c02543f42..e1d0cb5f5 100644 --- a/docs/architecture/compat-cleanup-register.md +++ b/docs/architecture/compat-cleanup-register.md @@ -18,6 +18,7 @@ - `backlog-2519` retained admin heal reports: keep the schema-1 terminal as the commit and replay fence, with a bounded versioned report in a separate namespace on the same disk. Older rollback readers can still query the terminal and suppress replay, but cannot expose its outcome; newer readers mark missing outcomes as unavailable. Remove the legacy marker and missing-report adapter only after all supported direct-upgrade and rollback readers understand the report format and retained schema-1-only receipts have expired. - `odm-list-bare-envelope` historical ODM continuation tokens: preserve complete bare v1/v2 envelopes. Framed issuance defaults on for the deployed framed-only generation; upgrades from older bare-only readers must explicitly disable it before starting new nodes and keep it off until reader convergence. Remove the legacy classifier and framing issuance override only after every supported reader accepts framing and outstanding bare listings have drained or clients explicitly restarted them; tokens have no automatic expiry. Exact full-envelope object keys remain intrinsically ambiguous during this compatibility period. - `backlog-2263` legacy heal MRF inspection: retained per-record journals remain readable while committed-snapshot ownership and writer activation are staged. Remove legacy import only after all supported direct-upgrade and rollback readers understand committed snapshots and migration tooling confirms that no retained or restorable legacy journal requires it. This does not enable a new writer or change the automatic legacy consumer. +- `backlog-2682` MRF lifecycle downgrade fence: the lifecycle sidecar binds durable responsibilities to a source bucket incarnation and records operator disposition, but pre-lifecycle readers ignore it and may replay the compatibility journal against a same-name recreated bucket. Downgrade to those readers is unsupported while any durable MRF responsibility remains. Remove the restriction only after every supported rollback reader enforces the source-incarnation fence and operators have verified that no old-format or lifecycle responsibility remains. - `backlog-1337` legacy restore orphan recovery: releases that predate the restore worker-lock marker can leave a valid operation-id and `ongoing-request="true"` after cancellation or process failure, with no durable liveness proof. New servers allow an exact, non-nil legacy generation to be superseded only when its consistently parsed request date is at least 24 hours old. Remove the clock-based legacy fallback after the minimum supported direct-upgrade release writes the v1 worker-lock marker on every restore and operators have resolved every retained pre-v1 ongoing generation. - `backlog-2133-tier-delete-chunk-parent` bounded tier-delete dispatch compatibility: prefixes at or below the legacy manifest limit keep the byte-compatible v1 single-manifest protocol, while larger prefixes place a chunk-parent sentinel at the original deterministic root path and use operation-scoped child manifests. Older binaries reject the sentinel and child paths, preserving the v6 sole-owner downgrade fence instead of starting a competing local delete. Remove the v1 reader and fail-closed mixed-version sentinel only after every supported rollback release validates the parent/child protocol and migration tooling confirms that no retained v1 dispatch manifest remains. - `backlog-2102` rc.2/rc.3 empty scanner usage floor recovery: old DeleteBucket cleanup could synthesize an empty incomplete v2 usage primary/backup before leadership added an epoch, while newer scanners require a durable authoritative baseline identity. New scanners recognize only that exact serialized empty-fence shape, preserve its epoch through a CAS-protected recovery marker, and rebuild namespace coverage without treating zero usage as authoritative. Remove this recovery path and marker after rc.2 and rc.3 are no longer supported direct-upgrade sources. diff --git a/docs/operations/scanner-runtime-controls.md b/docs/operations/scanner-runtime-controls.md index 996464081..c8c13d8f4 100644 --- a/docs/operations/scanner-runtime-controls.md +++ b/docs/operations/scanner-runtime-controls.md @@ -288,11 +288,35 @@ Durable partial-write responsibilities also have node-local MRF metrics: | Metric | Meaning | |---|---| -| `rustfs_heal_mrf_queue_depth` | In-memory MRF work plus retained durable responsibilities, including held legacy entries. | +| `rustfs_heal_mrf_queue_depth` | In-memory MRF work plus retained durable responsibilities, including held and operator-accepted unverified entries. | | `rustfs_heal_mrf_unverified_legacy` | Durable `PartialWrite` entries whose completed deep check found healthy legacy data without an independent payload identity proof. The journal entry remains intact; the current process pauses automatic re-dispatch for that entry. | | `rustfs_heal_mrf_unverified_legacy_oldest_age_seconds` | Age of the oldest such retained responsibility on this node. | +| `rustfs_heal_mrf_legacy_generation_unknown` | Durable entries replayed without a source bucket-incarnation binding; they are not dispatched until an operator chooses an explicit disposition. | +| `rustfs_heal_mrf_bucket_incarnation_changed` | Durable entries whose source bucket incarnation differs from the current bucket; the old intent is parked rather than sent to the recreated bucket. | +| `rustfs_heal_mrf_operator_accepted_unverified` | Responsibilities for which an administrator explicitly accepted the unresolved data risk. They remain durable and are never reported as repaired or verified. | +| `rustfs_heal_mrf_operator_accepted_unverified_oldest_age_seconds` | Age of the oldest durable operator-accepted, still-unverified responsibility on this node. | -These gauges are emitted per node through the configured OTLP metrics exporter (`RUSTFS_OBS_ENDPOINT` or `RUSTFS_OBS_METRIC_ENDPOINT`). The background-heal status endpoint reports execution tasks; it does not include durable MRF responsibilities. A process restart replays unchanged journal records and checks them again; it does not delete or certify an entry, and it replays all other durable intents on that node as well. For an unversioned object, restore the expected content from a trusted canonical source with protected shard integrity enabled, then restart the node that owns the journal so its intent can obtain an exact receipt. For a versioned object, a protected rewrite creates a new version and does not resolve a responsibility for the old version; preserve the source and only retire that exact version when the intended data is backed up and an authoritative absence proof is appropriate. A successful receipt may discharge only the matching responsibility; an unverified result remains retained and becomes held again. If no trusted copy or expected checksum is available, preserve the held responsibility and investigate its source of truth before retrying. The protected-copy migration in the [shard-integrity audit workflow](shard-integrity-audit.md) writes a new key and preserves its source; completing that migration alone does not prove or discharge an intent for the original key. Never remove MRF journal files manually. +These gauges are emitted per node through the configured OTLP metrics exporter (`RUSTFS_OBS_ENDPOINT` or `RUSTFS_OBS_METRIC_ENDPOINT`). The background-heal status endpoint reports execution tasks; it does not include durable MRF responsibilities. Use the authenticated node-local `GET /rustfs/admin/v4/heal/mrf/responsibilities?limit=100&cursor=` endpoint to list held, generation-unknown, incarnation-mismatched, or operator-accepted entries, including the stable responsibility ID, exact object/version/scope, source/current bucket incarnation, checkpoint owner/sequence, and any risk-acceptance record. `limit` is bounded to 1–256 and `nextCursor` is absent on the last page. If the listed bucket incarnation may have changed since the responsibility was parked, call the action endpoint with `action: "refresh"`, the listed `responsibilityId`, and its current `expectedBucketIncarnationId`. Refresh reads the live bucket identity, persists a changed-generation classification, and never dispatches a heal. Then review the fresh listing before taking another action. Use `action: "recheck"` to resume one exact source-bound held generation after restoring a trusted source or preparing an exact-version deletion. A recheck is persisted before the task is dispatched. + +An operator may instead choose `action: "acceptUnverifiedRisk"` only with `acknowledgeUnverifiedDataRisk: true`, explicit booleans for both source-incarnation acknowledgments, a non-empty reason, an audit reference, and a non-nil request ID. When `sourceBucketIncarnationId` is absent, set `acknowledgeUnknownSourceIncarnation: true`; when it differs from the current `bucketIncarnationId`, set `acknowledgeBucketIncarnationMismatch: true`. A generation with a known source/current mismatch is parked before dispatch, so its old intent cannot be applied to a recreated bucket. Legacy journal records without a source incarnation are listed as `legacy_generation_unknown` and are not automatically dispatched; refresh can record the live generation but does not bind the old responsibility to it. This does not create a payload proof, delete the object, or remove the durable intent. It persists an `operator_accepted_unverified` disposition that stops automatic retries while leaving the responsibility visible. The actor, reason, reference, time, request ID, target identity, source/current bucket incarnations, and acknowledgment flags are kept in a checksummed lifecycle record in the same disk replica and alternating slot as its committed MRF checkpoint. If that lifecycle record is missing, corrupt, or belongs to another checkpoint, unbound records return to `legacy_generation_unknown`; they are not treated as proof or dispatched against an unbound bucket. Pre-lifecycle binaries ignore this state and can heal a compatibility-journal object into a recreated bucket generation; downgrading while any durable MRF responsibility remains is unsupported. See `backlog-2682` in the compatibility cleanup register. + +Example risk-acceptance body (substitute values from the node-local listing): + +```json +{ + "action": "acceptUnverifiedRisk", + "responsibilityId": "", + "expectedBucketIncarnationId": "", + "acknowledgeUnverifiedDataRisk": true, + "acknowledgeUnknownSourceIncarnation": false, + "acknowledgeBucketIncarnationMismatch": false, + "reason": "The source was reviewed and the remaining identity risk is accepted", + "reference": "INC-1234", + "requestId": "" +} +``` + +A process restart restores lifecycle state only when the checksummed sidecar matches the committed checkpoint. Otherwise the durable intent remains visible and generation-unknown; no record is deleted or certified, and other durable intents on that node replay normally. For source-bound unversioned objects, restore expected content from a trusted canonical source with protected shard integrity enabled, then request a targeted recheck so the exact intent can obtain a receipt. For a legacy record without a source generation, the operator must first establish which bucket generation the responsibility belongs to; if that cannot be established, keep it generation-unknown or explicitly accept the risk. For a versioned object, a protected rewrite creates a new version and does not resolve a responsibility for the old version; preserve the source and only retire that exact version when the intended data is backed up and an authoritative absence proof is appropriate. A successful receipt may discharge only the matching responsibility; an unverified result remains retained and becomes held again. If no trusted copy or expected checksum is available, keep the responsibility held or explicitly accept the risk with the node-local admin operation. The protected-copy migration in the [shard-integrity audit workflow](shard-integrity-audit.md) writes a new key and preserves its source; completing that migration alone does not prove or discharge an intent for the original key. Never remove MRF journal files manually. ## Heal runtime controls @@ -382,7 +406,7 @@ Rediscovery and admission observations update the recorded result but do not pos With MRF enabled, a partial commit waits up to ten seconds for its checkpoint receipt. Fully converged writes do not enter this path or add MRF journal writes. The consumer batches available submissions and uses the existing count/byte budgets; an offline member or exhausted hint retry count does not evict an admitted partial-write obligation. New responsibility for the same unversioned object invalidates the previous lease, and a task that already started cannot accept that new responsibility as a merged proof target. -Disabled/unavailable delivery, full queues, invalid identities, persistence failures and receipt timeouts are not durable admission. An already committed object is not rolled back: the producer falls back to the existing in-memory Heal channel. When MRF is enabled, a failed admission attempt also records `mrf_durable_admission_failed`. An admission timeout does not cancel a record already retained by the consumer. These failure cases therefore do not promise that every successful S3 response has a durable repair obligation. Early-ACK PUTs can also return before their rename tail settles; the tail submits responsibility when its final outcome requires repair. The MRF switch is independent of automatic disk scanning, so returning members can be repaired with scanner and auto-heal disabled. +Disabled/unavailable delivery, full queues, invalid identities, persistence failures and receipt timeouts are not durable admission. An already committed object is not rolled back: when its source bucket incarnation is known, the producer falls back to the in-memory Heal channel with that same generation fence. If the source incarnation is unavailable, it does not dispatch an unfenced heal that could mutate a recreated bucket. When MRF is enabled, a failed admission attempt also records `mrf_durable_admission_failed`. An admission timeout does not cancel a record already retained by the consumer. These failure cases therefore do not promise that every successful S3 response has a durable repair obligation. Early-ACK PUTs can also return before their rename tail settles; the tail submits responsibility when its final outcome requires repair. The MRF switch is independent of automatic disk scanning, so returning members can be repaired with scanner and auto-heal disabled. Legacy notices carry only bucket/object/version, not a verified storage disposition, incarnation, scope, or durable responsibility generation. They are drained without clearing hints. Terminal callbacks release only their exact node-local ingress lease so rediscovery remains possible; lease generations are not durable successor receipts. Pending migration staging is not activated, and this change does not enable durable tombstones or garbage collection. Positive cleanup requires a storage-owner receipt with the complete responsibility identity and validated commit/fence evidence; neither task status nor the bounded diagnostic outcome window supplies it. diff --git a/rustfs/src/admin/handlers/heal.rs b/rustfs/src/admin/handlers/heal.rs index 021a27291..f6c02aa0c 100644 --- a/rustfs/src/admin/handlers/heal.rs +++ b/rustfs/src/admin/handlers/heal.rs @@ -57,7 +57,11 @@ const EVENT_ADMIN_REQUEST_FAILED: &str = "admin_request_failed"; const EVENT_ADMIN_RESPONSE_EMITTED: &str = "admin_response_emitted"; const LEGACY_ROOT_HEAL_RESPONSE_ID: &str = "."; const PEER_HEAL_STATUS_TIMEOUT: Duration = Duration::from_secs(5); +const MRF_RESPONSIBILITY_LIST_DEFAULT_LIMIT: usize = 100; +const MRF_RESPONSIBILITY_LIST_MAX_LIMIT: usize = 256; pub(crate) const REPLACEMENT_RECOVERY_STATUS_ROUTE_SUFFIX: &str = "/v4/heal/replacement-recovery"; +pub(crate) const MRF_LEGACY_RESPONSIBILITIES_ROUTE_SUFFIX: &str = "/v4/heal/mrf/responsibilities"; +pub(crate) const MRF_LEGACY_RESPONSIBILITIES_ACTIONS_ROUTE_SUFFIX: &str = "/v4/heal/mrf/responsibilities/actions"; const REPLACEMENT_RECOVERY_STATUS_CONTRACT_VERSION: u32 = 2; #[derive(Debug, Default, Serialize, Deserialize)] @@ -221,6 +225,18 @@ pub fn register_heal_route(r: &mut S3Router) -> std::io::Result< AdminOperation(&BackgroundHealStatusHandler {}), )?; + r.insert( + Method::GET, + format!("{}{}", ADMIN_PREFIX, MRF_LEGACY_RESPONSIBILITIES_ROUTE_SUFFIX).as_str(), + AdminOperation(&MrfLegacyResponsibilitiesHandler {}), + )?; + + r.insert( + Method::POST, + format!("{}{}", ADMIN_PREFIX, MRF_LEGACY_RESPONSIBILITIES_ACTIONS_ROUTE_SUFFIX).as_str(), + AdminOperation(&MrfLegacyResponsibilitiesActionHandler {}), + )?; + r.insert( Method::GET, format!("{}{}", ADMIN_PREFIX, REPLACEMENT_RECOVERY_STATUS_ROUTE_SUFFIX).as_str(), @@ -1414,7 +1430,7 @@ fn encode_replacement_recovery_status(response: &ReplacementRecoveryStatusRespon }) } -async fn validate_heal_admin_request(req: &S3Request) -> S3Result<()> { +async fn authenticate_heal_admin_request(req: &S3Request) -> S3Result { let Some(input_cred) = req.credentials.as_ref() else { return Err(s3_error!(InvalidRequest, "authentication required")); }; @@ -1429,7 +1445,12 @@ async fn validate_heal_admin_request(req: &S3Request) -> S3Result<()> { vec![Action::AdminAction(AdminAction::HealAdminAction)], req.extensions.get::>().and_then(|opt| opt.map(|a| a.0)), ) - .await + .await?; + Ok(cred.access_key) +} + +async fn validate_heal_admin_request(req: &S3Request) -> S3Result<()> { + authenticate_heal_admin_request(req).await.map(|_| ()) } pub struct HealHandler {} @@ -1576,6 +1597,248 @@ impl Operation for HealHandler { pub struct BackgroundHealStatusHandler {} +#[derive(Debug, Deserialize)] +#[serde(rename_all = "camelCase", deny_unknown_fields)] +struct MrfLegacyResponsibilityActionRequest { + action: MrfLegacyResponsibilityAction, + responsibility_id: uuid::Uuid, + expected_bucket_incarnation_id: uuid::Uuid, + #[serde(default)] + acknowledge_unverified_data_risk: Option, + #[serde(default)] + acknowledge_unknown_source_incarnation: Option, + #[serde(default)] + acknowledge_bucket_incarnation_mismatch: Option, + #[serde(default)] + reason: Option, + #[serde(default)] + reference: Option, + #[serde(default)] + request_id: Option, +} + +#[derive(Clone, Copy, Debug, Deserialize)] +#[serde(rename_all = "camelCase")] +enum MrfLegacyResponsibilityAction { + Refresh, + Recheck, + AcceptUnverifiedRisk, +} + +fn validate_mrf_legacy_responsibility_action(action: &MrfLegacyResponsibilityActionRequest) -> Result<(), &'static str> { + if action.responsibility_id.is_nil() || action.expected_bucket_incarnation_id.is_nil() { + return Err("responsibility and bucket incarnation IDs must be non-nil"); + } + match action.action { + MrfLegacyResponsibilityAction::Refresh | MrfLegacyResponsibilityAction::Recheck => { + if action.acknowledge_unverified_data_risk.is_some() + || action.acknowledge_unknown_source_incarnation.is_some() + || action.acknowledge_bucket_incarnation_mismatch.is_some() + || action.reason.is_some() + || action.reference.is_some() + { + return Err("recheck does not accept risk-acknowledgment fields"); + } + } + MrfLegacyResponsibilityAction::AcceptUnverifiedRisk => { + if action.acknowledge_unverified_data_risk != Some(true) { + return Err("acknowledgeUnverifiedDataRisk must be true"); + } + if action.acknowledge_unknown_source_incarnation.is_none() || action.acknowledge_bucket_incarnation_mismatch.is_none() + { + return Err("both source-incarnation acknowledgment fields are required"); + } + if action + .reason + .as_deref() + .is_none_or(|value| value.trim().is_empty() || value.len() > 1024) + { + return Err("reason is required and limited to 1024 UTF-8 bytes"); + } + if action + .reference + .as_deref() + .is_none_or(|value| value.trim().is_empty() || value.len() > 256) + { + return Err("reference is required and limited to 256 UTF-8 bytes"); + } + if action.request_id.is_none_or(|request_id| request_id.is_nil()) { + return Err("requestId must be non-nil"); + } + } + } + Ok(()) +} + +fn parse_mrf_legacy_responsibility_query(uri: &Uri) -> S3Result<(Option, usize)> { + let mut cursor = None; + let mut limit = MRF_RESPONSIBILITY_LIST_DEFAULT_LIMIT; + let mut seen = HashSet::with_capacity(2); + if let Some(query) = uri.query() { + for (key, value) in url::form_urlencoded::parse(query.as_bytes()) { + match key.as_ref() { + "cursor" if seen.insert("cursor") => { + cursor = Some( + value + .parse() + .map_err(|_| admin_error(S3ErrorCode::InvalidArgument, "cursor must be a UUID"))?, + ); + } + "limit" if seen.insert("limit") => { + limit = value + .parse::() + .map_err(|_| admin_error(S3ErrorCode::InvalidArgument, "limit must be an integer between 1 and 256"))?; + if limit == 0 || limit > MRF_RESPONSIBILITY_LIST_MAX_LIMIT { + return Err(admin_error(S3ErrorCode::InvalidArgument, "limit must be between 1 and 256")); + } + } + "cursor" | "limit" => { + return Err(admin_error(S3ErrorCode::InvalidArgument, "duplicate MRF responsibility query parameter")); + } + _ => return Err(admin_error(S3ErrorCode::InvalidArgument, "unknown MRF responsibility query parameter")), + } + } + } + Ok((cursor, limit)) +} + +#[derive(Debug, Serialize)] +#[serde(rename_all = "camelCase")] +struct MrfLegacyResponsibilityActionResponse { + responsibility_id: uuid::Uuid, + state: &'static str, + data_verified: bool, + durable_responsibility_retained: bool, +} + +fn map_mrf_lifecycle_control_error(error: rustfs_heal::heal::mrf_queue::MrfLifecycleControlError) -> s3s::S3Error { + use rustfs_heal::heal::mrf_queue::MrfLifecycleControlError; + match error { + MrfLifecycleControlError::Unavailable | MrfLifecycleControlError::Persistence => { + admin_error(S3ErrorCode::ServiceUnavailable, "MRF lifecycle operation could not be completed") + } + MrfLifecycleControlError::StaleGeneration => { + admin_error(S3ErrorCode::InvalidRequest, "MRF responsibility generation changed") + } + MrfLifecycleControlError::NotHeld => { + admin_error(S3ErrorCode::InvalidRequest, "MRF responsibility is not held as unverified legacy") + } + MrfLifecycleControlError::IncarnationChanged => admin_error( + S3ErrorCode::InvalidRequest, + "bucket incarnation changed; refresh the MRF responsibility listing", + ), + MrfLifecycleControlError::InvalidAction(reason) => admin_error(S3ErrorCode::InvalidRequest, reason), + } +} + +pub struct MrfLegacyResponsibilitiesHandler {} + +#[async_trait::async_trait] +impl Operation for MrfLegacyResponsibilitiesHandler { + async fn call(&self, req: S3Request, _params: Params<'_, '_>) -> S3Result> { + validate_heal_admin_request(&req).await?; + let (cursor, limit) = parse_mrf_legacy_responsibility_query(&req.uri)?; + let snapshot = timeout( + Duration::from_secs(5), + rustfs_heal::heal::mrf_queue::list_legacy_responsibilities(cursor, limit), + ) + .await + .map_err(|_| admin_error(S3ErrorCode::ServiceUnavailable, "MRF responsibility listing timed out"))? + .map_err(map_mrf_lifecycle_control_error)?; + let body = serde_json::to_vec(&snapshot) + .map_err(|_| admin_error(S3ErrorCode::InternalError, "failed to encode MRF responsibility listing"))?; + Ok(json_response(StatusCode::OK, body)) + } +} + +pub struct MrfLegacyResponsibilitiesActionHandler {} + +#[async_trait::async_trait] +impl Operation for MrfLegacyResponsibilitiesActionHandler { + async fn call(&self, mut req: S3Request, _params: Params<'_, '_>) -> S3Result> { + let actor = authenticate_heal_admin_request(&req).await?; + let bytes = req + .input + .store_all_limited(rustfs_config::MAX_ADMIN_REQUEST_BODY_SIZE) + .await + .map_err(|_| admin_error(S3ErrorCode::InvalidRequest, "MRF responsibility action body is too large or unreadable"))?; + let action: MrfLegacyResponsibilityActionRequest = serde_json::from_slice(&bytes) + .map_err(|_| admin_error(S3ErrorCode::InvalidRequest, "invalid MRF responsibility action body"))?; + validate_mrf_legacy_responsibility_action(&action).map_err(|reason| admin_error(S3ErrorCode::InvalidRequest, reason))?; + let (state, data_verified, durable_responsibility_retained) = match action.action { + MrfLegacyResponsibilityAction::Refresh => { + timeout( + Duration::from_secs(30), + rustfs_heal::heal::mrf_queue::refresh_legacy_responsibility( + action.responsibility_id, + action.expected_bucket_incarnation_id, + ), + ) + .await + .map_err(|_| admin_error(S3ErrorCode::ServiceUnavailable, "MRF responsibility refresh timed out"))? + .map_err(map_mrf_lifecycle_control_error)?; + ("refreshed", false, true) + } + MrfLegacyResponsibilityAction::Recheck => { + timeout( + Duration::from_secs(30), + rustfs_heal::heal::mrf_queue::recheck_legacy_responsibility( + action.responsibility_id, + action.expected_bucket_incarnation_id, + ), + ) + .await + .map_err(|_| admin_error(S3ErrorCode::ServiceUnavailable, "MRF responsibility recheck timed out"))? + .map_err(map_mrf_lifecycle_control_error)?; + ("active", false, true) + } + MrfLegacyResponsibilityAction::AcceptUnverifiedRisk => { + let reason = action + .reason + .ok_or_else(|| admin_error(S3ErrorCode::InvalidRequest, "reason is required"))?; + let reference = action + .reference + .ok_or_else(|| admin_error(S3ErrorCode::InvalidRequest, "reference is required"))?; + let request_id = action + .request_id + .ok_or_else(|| admin_error(S3ErrorCode::InvalidRequest, "requestId is required"))?; + timeout( + Duration::from_secs(30), + rustfs_heal::heal::mrf_queue::accept_unverified_legacy_risk( + rustfs_heal::heal::mrf_queue::MrfLegacyRiskAcceptanceRequest { + responsibility_id: action.responsibility_id, + expected_bucket_incarnation_id: action.expected_bucket_incarnation_id, + acknowledge_unknown_source_incarnation: action.acknowledge_unknown_source_incarnation == Some(true), + acknowledge_incarnation_mismatch: action.acknowledge_bucket_incarnation_mismatch == Some(true), + actor, + reason, + reference, + request_id, + }, + ), + ) + .await + .map_err(|_| { + admin_error( + S3ErrorCode::ServiceUnavailable, + "MRF risk disposition timed out; read status before retrying", + ) + })? + .map_err(map_mrf_lifecycle_control_error)?; + ("operatorAcceptedUnverified", false, true) + } + }; + let body = serde_json::to_vec(&MrfLegacyResponsibilityActionResponse { + responsibility_id: action.responsibility_id, + state, + data_verified, + durable_responsibility_retained, + }) + .map_err(|_| admin_error(S3ErrorCode::InternalError, "failed to encode MRF responsibility action response"))?; + Ok(json_response(StatusCode::OK, body)) + } +} + #[async_trait::async_trait] impl Operation for BackgroundHealStatusHandler { async fn call(&self, req: S3Request, _params: Params<'_, '_>) -> S3Result> { @@ -1667,13 +1930,14 @@ mod tests { use super::extract_heal_init_params; use super::{ BackgroundHealCoverage, BackgroundHealCoverageReason, BackgroundHealProgress, HealInitParams, HealResp, HealRuntimeState, - aggregate_cluster_heal_status, aggregate_replacement_recovery_cluster_status, background_heal_runtime_state, - build_heal_channel_request, build_replacement_recovery_status_response, encode_background_heal_status, - encode_heal_control_path, encode_heal_start_success, encode_heal_task_status, execute_after_heal_start_preflight, - heal_channel_response_items, heal_channel_response_progress, heal_channel_response_summary, heal_control_response_id, - json_response, map_heal_response, merge_peer_heal_statuses, peer_topology_complete, query_peer_heal_status, - query_peer_replacement_recovery_status, read_cluster_heal_status, reject_heal_admission, validate_heal_request_mode, - validate_heal_target, + MrfLegacyResponsibilityAction, MrfLegacyResponsibilityActionRequest, aggregate_cluster_heal_status, + aggregate_replacement_recovery_cluster_status, background_heal_runtime_state, build_heal_channel_request, + build_replacement_recovery_status_response, encode_background_heal_status, encode_heal_control_path, + encode_heal_start_success, encode_heal_task_status, execute_after_heal_start_preflight, heal_channel_response_items, + heal_channel_response_progress, heal_channel_response_summary, heal_control_response_id, json_response, + map_heal_response, merge_peer_heal_statuses, parse_mrf_legacy_responsibility_query, peer_topology_complete, + query_peer_heal_status, query_peer_replacement_recovery_status, read_cluster_heal_status, reject_heal_admission, + validate_heal_request_mode, validate_heal_target, validate_mrf_legacy_responsibility_action, }; use crate::storage::rpc::node_service::heal::{ NodeHealProgress, NodeHealStatusSnapshot, NodeReplacementRecoveryStatusSnapshot, encode_node_replacement_recovery_status, @@ -1692,6 +1956,92 @@ mod tests { }; use serde_json::json; use std::sync::atomic::{AtomicBool, Ordering}; + use uuid::Uuid; + + #[test] + fn mrf_legacy_risk_action_requires_explicit_acknowledgment_and_audit_fields() { + let base = json!({ + "action": "acceptUnverifiedRisk", + "responsibilityId": Uuid::new_v4(), + "expectedBucketIncarnationId": Uuid::new_v4(), + "acknowledgeUnverifiedDataRisk": true, + "acknowledgeUnknownSourceIncarnation": false, + "acknowledgeBucketIncarnationMismatch": false, + "reason": "Verified against the retained canonical backup", + "reference": "INC-1234", + "requestId": Uuid::new_v4(), + }); + let action: MrfLegacyResponsibilityActionRequest = serde_json::from_value(base.clone()).expect("valid risk action"); + assert!(validate_mrf_legacy_responsibility_action(&action).is_ok()); + + let mut missing_ack = base.clone(); + missing_ack["acknowledgeUnverifiedDataRisk"] = json!(false); + let missing_ack: MrfLegacyResponsibilityActionRequest = serde_json::from_value(missing_ack).expect("well-formed request"); + assert!(validate_mrf_legacy_responsibility_action(&missing_ack).is_err()); + + let mut missing_reference = base.clone(); + missing_reference.as_object_mut().expect("request object").remove("reference"); + let missing_reference: MrfLegacyResponsibilityActionRequest = + serde_json::from_value(missing_reference).expect("well-formed request"); + assert!(validate_mrf_legacy_responsibility_action(&missing_reference).is_err()); + + let mut missing_source_ack = base.clone(); + missing_source_ack + .as_object_mut() + .expect("request object") + .remove("acknowledgeUnknownSourceIncarnation"); + let missing_source_ack: MrfLegacyResponsibilityActionRequest = + serde_json::from_value(missing_source_ack).expect("well-formed request"); + assert!(validate_mrf_legacy_responsibility_action(&missing_source_ack).is_err()); + + let mut oversized_reason = base.clone(); + oversized_reason["reason"] = json!("x".repeat(1025)); + let oversized_reason: MrfLegacyResponsibilityActionRequest = + serde_json::from_value(oversized_reason).expect("well-formed request"); + assert!(validate_mrf_legacy_responsibility_action(&oversized_reason).is_err()); + + let mut unknown_field = base; + unknown_field["clearJournal"] = json!(true); + assert!(serde_json::from_value::(unknown_field).is_err()); + + let recheck = MrfLegacyResponsibilityActionRequest { + action: MrfLegacyResponsibilityAction::Recheck, + responsibility_id: Uuid::new_v4(), + expected_bucket_incarnation_id: Uuid::new_v4(), + acknowledge_unverified_data_risk: None, + acknowledge_unknown_source_incarnation: None, + acknowledge_bucket_incarnation_mismatch: None, + reason: None, + reference: None, + request_id: None, + }; + assert!(validate_mrf_legacy_responsibility_action(&recheck).is_ok()); + let refresh = MrfLegacyResponsibilityActionRequest { + action: MrfLegacyResponsibilityAction::Refresh, + ..recheck + }; + assert!(validate_mrf_legacy_responsibility_action(&refresh).is_ok()); + } + + #[test] + fn mrf_legacy_listing_query_is_bounded_and_rejects_ambiguous_parameters() { + let cursor = Uuid::new_v4(); + let uri = Uri::try_from(format!("/rustfs/admin/v4/heal/mrf/responsibilities?cursor={cursor}&limit=256")) + .expect("valid responsibility URI"); + assert_eq!(parse_mrf_legacy_responsibility_query(&uri).expect("valid query"), (Some(cursor), 256)); + + for query in [ + "limit=0", + "limit=257", + "limit=-1", + "cursor=bad", + "limit=1&limit=2", + "object=secret", + ] { + let uri = Uri::try_from(format!("/rustfs/admin/v4/heal/mrf/responsibilities?{query}")).expect("well-formed URI"); + assert!(parse_mrf_legacy_responsibility_query(&uri).is_err(), "accepted invalid query {query}"); + } + } use time::{OffsetDateTime, format_description::well_known::Rfc3339}; use tokio::sync::mpsc; use tokio::time::Duration; diff --git a/rustfs/src/admin/route_policy.rs b/rustfs/src/admin/route_policy.rs index 33fa52db9..6666d3e73 100644 --- a/rustfs/src/admin/route_policy.rs +++ b/rustfs/src/admin/route_policy.rs @@ -354,6 +354,18 @@ pub const ADMIN_ROUTE_POLICY_SPECS: &[AdminRouteSpec] = &[ admin(HttpMethod::Post, "/rustfs/admin/v3/heal/{bucket}", HEAL, RouteRiskLevel::High), admin(HttpMethod::Post, "/rustfs/admin/v3/heal/{bucket}/{*prefix}", HEAL, RouteRiskLevel::High), admin(HttpMethod::Post, "/rustfs/admin/v3/background-heal/status", HEAL, RouteRiskLevel::High), + admin( + HttpMethod::Get, + "/rustfs/admin/v4/heal/mrf/responsibilities", + HEAL, + RouteRiskLevel::Sensitive, + ), + admin( + HttpMethod::Post, + "/rustfs/admin/v4/heal/mrf/responsibilities/actions", + HEAL, + RouteRiskLevel::High, + ), admin( HttpMethod::Get, "/rustfs/admin/v4/heal/replacement-recovery", diff --git a/rustfs/src/admin/route_registration_test.rs b/rustfs/src/admin/route_registration_test.rs index 99440d07b..79f4b6a05 100644 --- a/rustfs/src/admin/route_registration_test.rs +++ b/rustfs/src/admin/route_registration_test.rs @@ -206,6 +206,8 @@ fn expected_admin_route_matrix() -> Vec { admin_route_sample(Method::POST, "/v3/heal/{bucket}", "/v3/heal/test-bucket"), admin_route_sample(Method::POST, "/v3/heal/{bucket}/{*prefix}", "/v3/heal/test-bucket/prefix"), admin_route(Method::POST, "/v3/background-heal/status"), + admin_route(Method::GET, "/v4/heal/mrf/responsibilities"), + admin_route(Method::POST, "/v4/heal/mrf/responsibilities/actions"), admin_route(Method::GET, "/v4/heal/replacement-recovery"), admin_route(Method::GET, "/v3/tier"), admin_route(Method::GET, "/v3/tier-stats"), @@ -1356,6 +1358,8 @@ fn test_register_routes_cover_representative_admin_paths() { assert_route(&router, Method::POST, &admin_path("/v3/heal/test-bucket")); assert_route(&router, Method::POST, &admin_path("/v3/heal/test-bucket/prefix")); assert_route(&router, Method::POST, &admin_path("/v3/background-heal/status")); + assert_route(&router, Method::GET, &admin_path("/v4/heal/mrf/responsibilities")); + assert_route(&router, Method::POST, &admin_path("/v4/heal/mrf/responsibilities/actions")); assert_route(&router, Method::GET, &admin_path("/v4/heal/replacement-recovery")); assert_route(&router, Method::GET, &admin_path("/v3/tier")); @@ -1464,6 +1468,8 @@ fn test_admin_alias_paths_match_existing_admin_routes() { (Method::POST, compat_admin_alias_path("/v3/heal/test-bucket")), (Method::POST, compat_admin_alias_path("/v3/heal/test-bucket/prefix")), (Method::POST, compat_admin_alias_path("/v3/background-heal/status")), + (Method::GET, compat_admin_alias_path("/v4/heal/mrf/responsibilities")), + (Method::POST, compat_admin_alias_path("/v4/heal/mrf/responsibilities/actions")), (Method::GET, compat_admin_alias_path("/v3/tier/HOT")), (Method::GET, compat_admin_alias_path("/v3/export-bucket-metadata")), (Method::PUT, compat_admin_alias_path("/v3/import-bucket-metadata")), diff --git a/rustfs/src/app/object/on_demand_migration_put.rs b/rustfs/src/app/object/on_demand_migration_put.rs index 33069d3e3..ca1a697ec 100644 --- a/rustfs/src/app/object/on_demand_migration_put.rs +++ b/rustfs/src/app/object/on_demand_migration_put.rs @@ -517,36 +517,38 @@ mod tests { assert!(!local.delete_marker); } - #[tokio::test] + #[test] #[serial_test::serial] - async fn write_back_rejects_unsupported_topology_before_any_mutation() { - let (_dir, _paths, store) = crate::app::gating_test_env::isolated_multi_pool_ecstore().await; - crate::app::runtime_sources::install_test_app_context(Arc::clone(&store)).await; - let bucket = "odm-unsupported"; - store - .make_bucket(bucket, &MakeBucketOptions::default()) - .await - .expect("bucket"); - let write_back = OnDemandMigrationWriteBack::new(); - let req = request( - bucket, - store.bucket_incarnation_id(bucket).await.expect("bucket incarnation"), - "key", - source_head(b"source"), - ); - assert!(matches!( - write_back.put_object(&req, body_stream(b"source")).await, - Err(WriteBackError::Unsupported(_)) - )); - assert!(matches!( - write_back.create_multipart_upload(&req).await, - Err(WriteBackError::Unsupported(_)) - )); - assert!(matches!( - write_back.complete_multipart_upload(&req, "no-session", Vec::new()).await, - Err(WriteBackError::Unsupported(_)) - )); - assert_nothing_left(&store, bucket, "key").await; + fn write_back_rejects_unsupported_topology_before_any_mutation() { + crate::app::gating_test_env::run_large_stack_test("odm-unsupported-topology", || async { + let (_dir, _paths, store) = crate::app::gating_test_env::isolated_multi_pool_ecstore().await; + crate::app::runtime_sources::install_test_app_context(Arc::clone(&store)).await; + let bucket = "odm-unsupported"; + store + .make_bucket(bucket, &MakeBucketOptions::default()) + .await + .expect("bucket"); + let write_back = OnDemandMigrationWriteBack::new(); + let req = request( + bucket, + store.bucket_incarnation_id(bucket).await.expect("bucket incarnation"), + "key", + source_head(b"source"), + ); + assert!(matches!( + write_back.put_object(&req, body_stream(b"source")).await, + Err(WriteBackError::Unsupported(_)) + )); + assert!(matches!( + write_back.create_multipart_upload(&req).await, + Err(WriteBackError::Unsupported(_)) + )); + assert!(matches!( + write_back.complete_multipart_upload(&req, "no-session", Vec::new()).await, + Err(WriteBackError::Unsupported(_)) + )); + assert_nothing_left(&store, bucket, "key").await; + }); } #[tokio::test]