From 899f81f3ad34fab4588473f5e1d11835976ce5c2 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?=E5=94=90=E5=B0=8F=E9=B8=AD?= Date: Mon, 7 Sep 2026 19:23:32 +0800 Subject: [PATCH] fix(replication): resolve drifted replicas via a target version ledger (#7368) * fix(replication): resolve drifted replicas via a target version ledger A replication target that mints its own version ids (Wasabi, AWS S3) never answers to the source uuid, so every version-addressed mutation after the initial PUT failed forever: permanent version deletes answered NoSuchVersion every heal cycle, and tag / retention / legal-hold updates re-PUT the object, minting one more target version per update (rustfs/backlog#2340). Record the id the target assigned as a per-target ledger on the source version (replication-target-version-, written through the existing status writeback) and resolve every later mutation through it: version deletes DELETE the ledger id, metadata updates go through the metadata-only Object Lock and tagging APIs. Replicas written before the ledger existed are located by exact key and ETag, minus the candidates other generations of the key already claim through their own ledgers; an ambiguous remainder is refused with a backoff instead of guessed, since a wrong pick would destroy a live generation. A fresh write never consults content identity. NoSuchVersion on a version-addressed DELETE counts as purged. The fake target gains the Wasabi shape (404 NoSuchVersion on an unknown id, per-version Object Lock APIs) and the matrix covers the three mutation classes plus the same-bytes generation case. * fix(scanner): drop the unused Digest import Same one-line change as rustfs/rustfs#7366 (main is red with it under -D warnings); carried here so the stacked PRs' merge commits compile until that fix lands. * fix(admin): probe replication-check mutations by the assigned version id (#7373) On a target that mints its own version ids the DeleteMarker and VersionDelete phases of ?replication-check were skipped: they addressed the source id, which such a target never had. The replication worker now addresses the id the target assigned (the target-version ledger), and the probe already holds that id from its own PUT, so run both phases against it. VersionFidelity keeps failing with the mismatch code and the target stays FAILED; the phases report whether ledger-addressed purges work against this endpoint (rustfs/backlog#2340). * fix(replication): abandon purges to targets the bucket no longer names (#7377) * fix(admin): probe replication-check mutations by the assigned version id On a target that mints its own version ids the DeleteMarker and VersionDelete phases of ?replication-check were skipped: they addressed the source id, which such a target never had. The replication worker now addresses the id the target assigned (the target-version ledger), and the probe already holds that id from its own PUT, so run both phases against it. VersionFidelity keeps failing with the mismatch code and the target stays FAILED; the phases report whether ledger-addressed purges work against this endpoint (rustfs/backlog#2340). * fix(replication): abandon purges to targets the bucket no longer names A permanent version delete whose replication keeps failing stays in xl.meta as a PENDING purge, hidden from listings, until every target confirms it. Once the operator removes the replication configuration or the rule naming that target nothing ever confirms it: the heal path derived its delete decision from the configuration (the decision string is not persisted) and skipped the version forever, so DeleteBucket answered BucketNotEmpty for a residue the client could neither list nor remove (rustfs/backlog#2340). Owe a version purge to the targets its purge state names, let the heal path through without a configuration, and have the delete worker settle a target the configuration no longer names as abandoned: the purge is reported complete locally through the normal writeback, the replica on the former target is left alone, and the event replication_purge_abandoned plus a counter are the record. * fix(admin): send replication-check marker creation without a version id Running the DeleteMarker / VersionDelete phases on a target that mints its own version ids exposed two probe-shape bugs on real Wasabi: - the DeleteMarker phase put the assigned version id on its DELETE. A RustFS peer reads the source-deletemarker header and creates a marker, but a generic S3 target executes it as a permanent delete of the probe version, so VersionDelete then answered NoSuchVersion. Use the same wire shape as live delete replication: no versionId on a marker creation. - cleanup treated NoSuchVersion on the version the VersionDelete phase had already removed as a failure (RustFS/MinIO answer 204 there). Also gate the no-configuration heal pass-through for pending purges on a purge state that actually names targets, so a purge without a recorded target keeps the ordinary skip (scanner unit test), and merge origin/main (#7365 settles the pool-metadata probe test that failed in CI). --------- Co-authored-by: houseme --- crates/e2e_test/src/fake_s3_target/mod.rs | 212 +- .../src/replication_extension_test.rs | 23 +- .../src/replication_target_matrix_test.rs | 445 ++++- crates/ecstore/src/api/mod.rs | 1 + .../ecstore/src/bucket/bucket_target_sys.rs | 174 +- .../bucket/replication/replication_pool.rs | 34 +- .../replication/replication_resyncer.rs | 1726 +++++++++++++++-- .../replication_target_boundary.rs | 2 +- crates/filemeta/src/replication.rs | 4 + crates/replication/src/config.rs | 17 + crates/replication/src/filemeta.rs | 6 + .../src/scanner_folder/item_actions.rs | 13 +- crates/utils/src/http/metadata_compat.rs | 52 +- docs/operations/replication-check.md | 2 +- .../replication-outbound-transport.md | 2 +- rustfs/src/admin/router.rs | 106 +- rustfs/src/admin/storage_api.rs | 1 + 17 files changed, 2559 insertions(+), 261 deletions(-) diff --git a/crates/e2e_test/src/fake_s3_target/mod.rs b/crates/e2e_test/src/fake_s3_target/mod.rs index aa8792038..641255fa6 100644 --- a/crates/e2e_test/src/fake_s3_target/mod.rs +++ b/crates/e2e_test/src/fake_s3_target/mod.rs @@ -34,12 +34,14 @@ use s3s::dto::{ AbortMultipartUploadInput, AbortMultipartUploadOutput, CommonPrefix, CompleteMultipartUploadInput, CompleteMultipartUploadOutput, CreateMultipartUploadInput, CreateMultipartUploadOutput, DeleteMarkerEntry, DeleteObjectInput, DeleteObjectOutput, DeleteObjectTaggingInput, DeleteObjectTaggingOutput, ETag, GetBucketVersioningInput, - GetBucketVersioningOutput, GetObjectInput, GetObjectLockConfigurationInput, GetObjectLockConfigurationOutput, - GetObjectOutput, GetObjectTaggingInput, GetObjectTaggingOutput, HeadBucketInput, HeadBucketOutput, HeadObjectInput, + GetBucketVersioningOutput, GetObjectInput, GetObjectLegalHoldInput, GetObjectLegalHoldOutput, + GetObjectLockConfigurationInput, GetObjectLockConfigurationOutput, GetObjectOutput, GetObjectRetentionInput, + GetObjectRetentionOutput, GetObjectTaggingInput, GetObjectTaggingOutput, HeadBucketInput, HeadBucketOutput, HeadObjectInput, HeadObjectOutput, ListObjectVersionsInput, ListObjectVersionsOutput, ListObjectsV2Input, ListObjectsV2Output, Object, - ObjectLockConfiguration, ObjectLockEnabled, ObjectStorageClass, ObjectVersionId, PutObjectInput, PutObjectOutput, - PutObjectTaggingInput, PutObjectTaggingOutput, Range, StreamingBlob, Tag, TagSet, Timestamp, TimestampFormat, - UploadPartInput, UploadPartOutput, + ObjectLockConfiguration, ObjectLockEnabled, ObjectLockLegalHold, ObjectLockLegalHoldStatus, ObjectLockMode, + ObjectLockRetention, ObjectLockRetentionMode, ObjectStorageClass, ObjectVersionId, PutObjectInput, PutObjectLegalHoldInput, + PutObjectLegalHoldOutput, PutObjectOutput, PutObjectRetentionInput, PutObjectRetentionOutput, PutObjectTaggingInput, + PutObjectTaggingOutput, Range, StreamingBlob, Tag, TagSet, Timestamp, TimestampFormat, UploadPartInput, UploadPartOutput, }; use s3s::service::{S3Service, S3ServiceBuilder}; use s3s::validation::{AwsNameValidation, NameValidation}; @@ -127,6 +129,10 @@ pub enum Operation { GetObjectTagging, PutObjectTagging, DeleteObjectTagging, + GetObjectRetention, + PutObjectRetention, + GetObjectLegalHold, + PutObjectLegalHold, ListObjectVersions, ListObjectsV2, CreateMultipartUpload, @@ -501,6 +507,10 @@ struct StoreState { /// PutObject carrying any `x-amz-object-lock-*` header must also carry /// `Content-MD5` or an `x-amz-checksum-*` header. require_checksum_for_object_lock: bool, + /// Models Wasabi (rustfs/backlog#2340): a version-addressed DELETE of a + /// version id the target never had answers 404 `NoSuchVersion` instead of + /// the idempotent 204 RustFS/MinIO give. + reject_unknown_version_deletes: bool, limits: StoreLimits, buckets: HashMap, uploads: HashMap, @@ -565,6 +575,41 @@ struct ObjectVersion { /// SSE-C passthrough transport headers stored with the version (RustFS /// target behavior); empty when the drop mode discarded them. replication_sse_headers: Vec<(String, String)>, + /// Object Lock state of the version: retention (mode, retain-until) from + /// the PUT / CreateMultipartUpload headers or PutObjectRetention, and the + /// legal hold flag; replayed on HEAD. + lock: VersionLock, +} + +#[derive(Clone, Default)] +struct VersionLock { + retention: Option<(String, Timestamp)>, + /// `None` until a legal hold status was ever set; like S3, HEAD then + /// reports nothing, while an explicit OFF is reported as `OFF`. + legal_hold: Option, +} + +impl VersionLock { + fn from_headers( + mode: Option, + retain_until: Option, + legal_hold: Option, + ) -> Self { + Self { + retention: mode.zip(retain_until).map(|(mode, until)| (mode.as_str().to_string(), until)), + legal_hold: legal_hold.map(|status| status.as_str().eq_ignore_ascii_case("ON")), + } + } + + fn legal_hold_status(&self) -> Option { + self.legal_hold.map(|on| { + ObjectLockLegalHoldStatus::from_static(if on { + ObjectLockLegalHoldStatus::ON + } else { + ObjectLockLegalHoldStatus::OFF + }) + }) + } } #[derive(Clone)] @@ -576,6 +621,7 @@ struct MultipartState { metadata: Option>, standard_headers: StandardHeaders, replication_sse_headers: Vec<(String, String)>, + lock: VersionLock, parts: BTreeMap, } @@ -849,6 +895,7 @@ impl FakeS3Target { standard_headers: seed.standard_headers.clone(), tags: Vec::new(), replication_sse_headers: Vec::new(), + lock: VersionLock::default(), }; upsert_version(&mut state, bucket, key.into(), version).expect("seed object must fit the storage budget"); e_tag @@ -922,6 +969,12 @@ impl FakeS3Target { /// PutObject that carries Object Lock parameters (AWS S3 / MinIO rule, /// rustfs#7082). `Content-MD5`, when present, is always verified against /// the body regardless of this mode. + /// Wasabi-like mode: DELETE of an unknown version id answers 404 + /// `NoSuchVersion` (the default 204 models RustFS/MinIO). + pub fn reject_unknown_version_deletes(&self, enabled: bool) { + lock(&self.backend.store).reject_unknown_version_deletes = enabled; + } + pub fn require_checksum_for_object_lock(&self, enabled: bool) { lock(&self.backend.store).require_checksum_for_object_lock = enabled; } @@ -1156,6 +1209,10 @@ fn operation_from_s3_name(name: &str) -> Operation { "GetObjectTagging" => Operation::GetObjectTagging, "PutObjectTagging" => Operation::PutObjectTagging, "DeleteObjectTagging" => Operation::DeleteObjectTagging, + "GetObjectRetention" => Operation::GetObjectRetention, + "PutObjectRetention" => Operation::PutObjectRetention, + "GetObjectLegalHold" => Operation::GetObjectLegalHold, + "PutObjectLegalHold" => Operation::PutObjectLegalHold, "ListObjectsV2" => Operation::ListObjectsV2, "CreateMultipartUpload" => Operation::CreateMultipartUpload, "UploadPart" => Operation::UploadPart, @@ -1293,6 +1350,18 @@ fn parse_request(method: &Method, uri: &Uri) -> ParsedRequest { (&Method::DELETE, true) if query.contains_key("tagging") && only_query_keys(&["tagging", "versionId"]) => { Operation::DeleteObjectTagging } + (&Method::GET, true) if query.contains_key("retention") && only_query_keys(&["retention", "versionId"]) => { + Operation::GetObjectRetention + } + (&Method::PUT, true) if query.contains_key("retention") && only_query_keys(&["retention", "versionId"]) => { + Operation::PutObjectRetention + } + (&Method::GET, true) if query.contains_key("legal-hold") && only_query_keys(&["legal-hold", "versionId"]) => { + Operation::GetObjectLegalHold + } + (&Method::PUT, true) if query.contains_key("legal-hold") && only_query_keys(&["legal-hold", "versionId"]) => { + Operation::PutObjectLegalHold + } // A replication PUT addresses the source version via `?versionId=`. (&Method::PUT, true) if only_query_keys(&["versionId"]) => Operation::PutObject, (&Method::GET, true) if only_query_keys(&["versionId"]) => Operation::GetObject, @@ -1846,6 +1915,28 @@ fn set_version_tags( Ok(resolved) } +fn update_version_lock( + state: &mut StoreState, + bucket: &str, + key: &str, + version_id: Option<&str>, + update: impl FnOnce(&mut VersionLock), +) -> S3Result { + let resolved = find_version(state, bucket, key, version_id)?.version_id; + let version = state + .buckets + .get_mut(bucket) + .expect("bucket existence checked by find_version") + .objects + .get_mut(key) + .expect("key existence checked by find_version") + .iter_mut() + .find(|version| version.version_id == resolved) + .expect("version existence checked by find_version"); + update(&mut version.lock); + Ok(resolved) +} + /// Whether version ids are surfaced for this bucket. Unknown buckets report /// `true`; the caller's lookup raises `NoSuchBucket` first. fn bucket_versioned(state: &StoreState, bucket: &str) -> bool { @@ -2285,6 +2376,11 @@ impl S3 for FakeBackend { standard_headers, tags: Vec::new(), replication_sse_headers: captured_replication_sse_headers(&headers, drop_unlisted), + lock: VersionLock::from_headers( + input.object_lock_mode, + input.object_lock_retain_until_date, + input.object_lock_legal_hold_status, + ), }; upsert_version(&mut lock(&self.store), &input.bucket, input.key, version)?; Ok(apply_response_fault( @@ -2343,6 +2439,13 @@ impl S3 for FakeBackend { last_modified: Some(version.last_modified.clone()), version_id: versioned.then_some(version.version_id), sse_customer_algorithm, + object_lock_mode: version + .lock + .retention + .as_ref() + .map(|(mode, _)| ObjectLockMode::from(mode.clone())), + object_lock_retain_until_date: version.lock.retention.as_ref().map(|(_, until)| until.clone()), + object_lock_legal_hold_status: version.lock.legal_hold_status(), ..Default::default() }); response.status = served.status; @@ -2377,6 +2480,13 @@ impl S3 for FakeBackend { last_modified: Some(version.last_modified.clone()), version_id: versioned.then_some(version.version_id), sse_customer_algorithm, + object_lock_mode: version + .lock + .retention + .as_ref() + .map(|(mode, _)| ObjectLockMode::from(mode.clone())), + object_lock_retain_until_date: version.lock.retention.as_ref().map(|(_, until)| until.clone()), + object_lock_legal_hold_status: version.lock.legal_hold_status(), ..Default::default() }); response.status = served.status; @@ -2436,6 +2546,82 @@ impl S3 for FakeBackend { )) } + async fn get_object_retention( + &self, + req: S3Request, + ) -> S3Result> { + let fault = request_fault(&req); + apply_non_body_fault(fault.as_ref(), &self.control).await?; + let input = req.input; + let version = find_version(&lock(&self.store), &input.bucket, &input.key, input.version_id.as_deref())?; + Ok(apply_response_fault( + S3Response::new(GetObjectRetentionOutput { + retention: version.lock.retention.map(|(mode, until)| ObjectLockRetention { + mode: Some(ObjectLockRetentionMode::from(mode)), + retain_until_date: Some(until), + }), + }), + fault.as_ref(), + )) + } + + async fn put_object_retention( + &self, + req: S3Request, + ) -> S3Result> { + let fault = request_fault(&req); + apply_non_body_fault(fault.as_ref(), &self.control).await?; + let input = req.input; + let retention = input + .retention + .and_then(|retention| retention.mode.zip(retention.retain_until_date)) + .map(|(mode, until)| (mode.as_str().to_string(), until)); + update_version_lock(&mut lock(&self.store), &input.bucket, &input.key, input.version_id.as_deref(), |lock| { + lock.retention = retention; + })?; + Ok(apply_response_fault(S3Response::new(PutObjectRetentionOutput::default()), fault.as_ref())) + } + + async fn get_object_legal_hold( + &self, + req: S3Request, + ) -> S3Result> { + let fault = request_fault(&req); + apply_non_body_fault(fault.as_ref(), &self.control).await?; + let input = req.input; + let version = find_version(&lock(&self.store), &input.bucket, &input.key, input.version_id.as_deref())?; + Ok(apply_response_fault( + S3Response::new(GetObjectLegalHoldOutput { + legal_hold: Some(ObjectLockLegalHold { + status: Some( + version + .lock + .legal_hold_status() + .unwrap_or_else(|| ObjectLockLegalHoldStatus::from_static(ObjectLockLegalHoldStatus::OFF)), + ), + }), + }), + fault.as_ref(), + )) + } + + async fn put_object_legal_hold( + &self, + req: S3Request, + ) -> S3Result> { + let fault = request_fault(&req); + apply_non_body_fault(fault.as_ref(), &self.control).await?; + let input = req.input; + let legal_hold_on = input + .legal_hold + .and_then(|hold| hold.status) + .is_some_and(|status| status.as_str().eq_ignore_ascii_case("ON")); + update_version_lock(&mut lock(&self.store), &input.bucket, &input.key, input.version_id.as_deref(), |lock| { + lock.legal_hold = Some(legal_hold_on); + })?; + Ok(apply_response_fault(S3Response::new(PutObjectLegalHoldOutput::default()), fault.as_ref())) + } + async fn delete_object_tagging( &self, req: S3Request, @@ -2489,6 +2675,7 @@ impl S3 for FakeBackend { return Ok(apply_response_fault(S3Response::new(DeleteObjectOutput::default()), fault.as_ref())); } if let Some(version_id) = input.version_id { + let reject_unknown = state.reject_unknown_version_deletes; let (removed_bytes, removed_versions, delete_marker, remove_key) = { let Some(versions) = state .buckets @@ -2497,6 +2684,9 @@ impl S3 for FakeBackend { .objects .get_mut(&input.key) else { + if reject_unknown { + return Err(s3s::s3_error!(NoSuchVersion, "The specified version does not exist.")); + } return Ok(apply_response_fault( S3Response::new(DeleteObjectOutput { version_id: Some(version_id), @@ -2505,6 +2695,9 @@ impl S3 for FakeBackend { fault.as_ref(), )); }; + if reject_unknown && !versions.iter().any(|version| version.version_id == version_id) { + return Err(s3s::s3_error!(NoSuchVersion, "The specified version does not exist.")); + } let mut removed_bytes = 0usize; let mut removed_versions = 0usize; let mut delete_marker = None; @@ -2558,6 +2751,7 @@ impl S3 for FakeBackend { standard_headers: StandardHeaders::default(), tags: Vec::new(), replication_sse_headers: Vec::new(), + lock: VersionLock::default(), }, )?; Ok(apply_response_fault( @@ -2612,6 +2806,11 @@ impl S3 for FakeBackend { metadata: input.metadata, standard_headers, replication_sse_headers: captured_replication_sse_headers(&headers, drop_unlisted), + lock: VersionLock::from_headers( + input.object_lock_mode, + input.object_lock_retain_until_date, + input.object_lock_legal_hold_status, + ), parts: BTreeMap::new(), }, ); @@ -2760,6 +2959,7 @@ impl S3 for FakeBackend { metadata: upload.metadata.clone(), standard_headers: upload.standard_headers.clone(), replication_sse_headers: upload.replication_sse_headers.clone(), + lock: upload.lock.clone(), parts: BTreeMap::new(), }, selected, @@ -2790,6 +2990,7 @@ impl S3 for FakeBackend { standard_headers: upload.standard_headers, tags: Vec::new(), replication_sse_headers: upload.replication_sse_headers, + lock: upload.lock, }; let mut state = lock(&self.store); let versioned = bucket_versioned(&state, &input.bucket); @@ -4621,6 +4822,7 @@ mod tests { metadata: None, standard_headers: StandardHeaders::default(), replication_sse_headers: Vec::new(), + lock: VersionLock::default(), parts: BTreeMap::new(), }, ); diff --git a/crates/e2e_test/src/replication_extension_test.rs b/crates/e2e_test/src/replication_extension_test.rs index 799126792..10a106975 100644 --- a/crates/e2e_test/src/replication_extension_test.rs +++ b/crates/e2e_test/src/replication_extension_test.rs @@ -512,7 +512,7 @@ pub(crate) async fn put_bucket_replication( put_bucket_replication_with_delete_statuses(env, bucket, target_arn, "Enabled", None).await } -async fn put_bucket_replication_with_delete_statuses( +pub(crate) async fn put_bucket_replication_with_delete_statuses( env: &RustFSTestEnvironment, bucket: &str, target_arn: &str, @@ -627,7 +627,7 @@ async fn put_bucket_replication_rules( Ok(()) } -async fn delete_bucket_replication( +pub(crate) async fn delete_bucket_replication( env: &RustFSTestEnvironment, bucket: &str, ) -> Result> { @@ -9055,9 +9055,11 @@ async fn test_replication_check_flags_multipart_only_version_minting_target() -> .is_some_and(|error| error.contains("CreateMultipartUpload")), "the failure must name the multipart path: {payload}" ); - // The PutObject leg mirrored, so it is the multipart probe that failed. + // The PutObject leg mirrored, so it is the multipart probe that failed; + // the mutation phases address the id the PUT reported and still run. assert_eq!(target_report["Phases"]["Put"]["Status"], "OK", "{payload}"); - assert_eq!(target_report["Phases"]["DeleteMarker"]["Status"], "SKIPPED", "{payload}"); + assert_eq!(target_report["Phases"]["DeleteMarker"]["Status"], "OK", "{payload}"); + assert_eq!(target_report["Phases"]["VersionDelete"]["Status"], "OK", "{payload}"); assert_eq!(target_report["Phases"]["Cleanup"]["Status"], "OK", "{payload}"); let probe_key = target @@ -9245,6 +9247,9 @@ async fn test_replication_check_flags_version_minting_target() -> TestResult { let target_bucket = "version-fidelity-dst"; target.create_bucket(target_bucket); target.assign_own_version_ids(true); + // Wasabi shape: the probe version the VersionDelete phase removed answers + // NoSuchVersion to cleanup's second DELETE, which must count as clean. + target.reject_unknown_version_deletes(true); let mut source_env = RustFSTestEnvironment::new().await?; let mut env_vars = replication_fast_env(); @@ -9289,11 +9294,13 @@ async fn test_replication_check_flags_version_minting_target() -> TestResult { fidelity["Code"], "BucketRemoteTargetVersionMismatch", "the failure must carry a machine-readable code: {payload}" ); - // The probe PUT itself succeeded (fidelity is judged from its response); - // the later mutation phases are pointless against a drifting target and - // must be skipped, but cleanup still runs. + // The probe PUT itself succeeded (fidelity is judged from its response). + // The mutation phases address the id the target assigned — the ledger + // the worker records per object (rustfs/backlog#2340) — so they run and + // pass on a drifting target, and cleanup uses the same id. assert_eq!(target_report["Phases"]["Put"]["Status"], "OK", "{payload}"); - assert_eq!(target_report["Phases"]["DeleteMarker"]["Status"], "SKIPPED", "{payload}"); + assert_eq!(target_report["Phases"]["DeleteMarker"]["Status"], "OK", "{payload}"); + assert_eq!(target_report["Phases"]["VersionDelete"]["Status"], "OK", "{payload}"); assert_eq!(target_report["Phases"]["Cleanup"]["Status"], "OK", "{payload}"); // The probe PUT must carry the source version as `?versionId=` — the diff --git a/crates/e2e_test/src/replication_target_matrix_test.rs b/crates/e2e_test/src/replication_target_matrix_test.rs index bace5a08e..2a3cb973e 100644 --- a/crates/e2e_test/src/replication_target_matrix_test.rs +++ b/crates/e2e_test/src/replication_target_matrix_test.rs @@ -36,14 +36,16 @@ use crate::fake_s3_target::{BucketMode, FAKE_ACCESS_KEY, FAKE_SECRET_KEY}; use crate::fake_s3_target::{FakeS3Target, FaultAction as FakeTargetFault, Operation as FakeTargetOperation, RequestRecord}; use crate::on_demand_migration::common::{OdmEnvOptions, OdmTestEnv, fake_source_client}; use crate::replication_extension_test::{ - LOOPBACK_REPLICATION_TARGET_ENV, ReplicationTargetOptions, enable_bucket_versioning, get_replication_reset_status, - put_bucket_replication, set_replication_target_with_options, start_bucket_replication_reset, + LOOPBACK_REPLICATION_TARGET_ENV, ReplicationTargetOptions, delete_bucket_replication, enable_bucket_versioning, + get_replication_reset_status, put_bucket_replication, put_bucket_replication_with_delete_statuses, + set_replication_target_with_options, start_bucket_replication_reset, }; use aws_sdk_s3::Client; +use aws_sdk_s3::error::ProvideErrorMetadata; use aws_sdk_s3::primitives::{ByteStream, DateTime}; use aws_sdk_s3::types::{ - Checksum, ChecksumAlgorithm, CompletedMultipartUpload, CompletedPart, ObjectAttributes, ObjectLockLegalHoldStatus, - ObjectLockMode, + Checksum, ChecksumAlgorithm, CompletedMultipartUpload, CompletedPart, ObjectAttributes, ObjectLockLegalHold, + ObjectLockLegalHoldStatus, ObjectLockMode, ObjectLockRetention, ObjectLockRetentionMode, Tag, Tagging, }; use bytes::Bytes; use std::error::Error; @@ -66,7 +68,10 @@ enum TargetMode { /// Object Lock parameters must carry `Content-MD5` or `x-amz-checksum-*`. RequireChecksumWithObjectLock, /// AWS S3 / Wasabi / Impossible Cloud: mints its own version ids - /// (rustfs/backlog#2085). Data must still land. + /// (rustfs/backlog#2085) and, like Wasabi, answers NoSuchVersion to a + /// DELETE of an id it never had (rustfs/backlog#2340). Data must still + /// land, and every version-addressed mutation must resolve the replica + /// through the target-version ledger. MintOwnVersionIds, } @@ -83,7 +88,10 @@ impl TargetMode { TargetMode::Baseline => {} TargetMode::RejectAwsChunked => target.reject_aws_chunked_uploads(true), TargetMode::RequireChecksumWithObjectLock => target.require_checksum_for_object_lock(true), - TargetMode::MintOwnVersionIds => target.assign_own_version_ids(true), + TargetMode::MintOwnVersionIds => { + target.assign_own_version_ids(true); + target.reject_unknown_version_deletes(true); + } } } @@ -384,6 +392,431 @@ async fn matrix_mint_own_version_ids_redrive_does_not_duplicate() -> TestResult Ok(()) } +/// rustfs/backlog#2340 (target-version ledger): on a target that mints its own +/// version ids and answers NoSuchVersion to an unknown id (the Wasabi shape), +/// every version-addressed mutation must land on the version the target +/// assigned, which the replication PUT recorded on the source: +/// - a tag update changes the existing target version, no new version; +/// - a retention extension and legal hold ON/OFF change that version too; +/// - a permanent delete of the older of two same-content generations removes +/// exactly that replica and keeps the live one (content identity alone +/// could not tell them apart). +#[tokio::test] +async fn matrix_mint_own_version_ids_addresses_mutations_through_the_ledger() -> TestResult { + init_logging(); + + let target = FakeS3Target::start().await?; + let target_bucket = "matrix-mint-own-ledger-dst".to_string(); + target.create_bucket_with_object_lock(target_bucket.clone()); + TargetMode::MintOwnVersionIds.apply(&target); + + let mut env_vars = replication_fast_env(); + env_vars.extend_from_slice(LOOPBACK_REPLICATION_TARGET_ENV); + env_vars.extend_from_slice(&[ + ("NO_PROXY", "127.0.0.1,localhost"), + ("HTTP_PROXY", ""), + ("HTTPS_PROXY", ""), + // The scanner heal pass retries a purge the first attempt lost. + ("RUSTFS_SCANNER_CYCLE", "1"), + ("RUSTFS_SCANNER_START_DELAY_SECS", "1"), + ]); + let env = OdmTestEnv::start_with(OdmEnvOptions { + env: env_vars, + ..OdmEnvOptions::default() + }) + .await?; + let source_env = &env.rustfs; + + let source_bucket = "matrix-mint-own-ledger-src"; + let source_client = source_env.create_s3_client(); + source_client + .create_bucket() + .bucket(source_bucket) + .object_lock_enabled_for_bucket(true) + .send() + .await?; + enable_bucket_versioning(source_env, source_bucket).await?; + let target_arn = set_replication_target_with_options( + source_env, + source_bucket, + ReplicationTargetOptions { + endpoint: &target.address(), + access_key: FAKE_ACCESS_KEY, + secret_key: FAKE_SECRET_KEY, + target_bucket: &target_bucket, + secure: false, + skip_tls_verify: false, + ca_cert_pem: None, + }, + ) + .await?; + put_bucket_replication_with_delete_statuses(source_env, source_bucket, &target_arn, "Enabled", Some("Enabled")).await?; + let target_client = fake_source_client(&target); + + // Tag update on an existing version. + let tag_key = "ledger/tags.bin"; + let tagged = source_client + .put_object() + .bucket(source_bucket) + .key(tag_key) + .body(ByteStream::from(payload(4 * 1024, 0x01))) + .send() + .await?; + let tag_source_version = tagged.version_id().ok_or("source PUT returned no version id")?.to_string(); + assert_eq!( + wait_for_terminal_replication_status(&source_client, source_bucket, tag_key).await?, + "COMPLETED" + ); + let tag_target_version = single_target_version(&target, &target_bucket, tag_key)?; + source_client + .put_object_tagging() + .bucket(source_bucket) + .key(tag_key) + .version_id(&tag_source_version) + .tagging( + Tagging::builder() + .tag_set(Tag::builder().key("phase").value("after").build()?) + .build()?, + ) + .send() + .await?; + wait_until("tag update on the existing target version", || async { + let tags = target_client + .get_object_tagging() + .bucket(&target_bucket) + .key(tag_key) + .version_id(&tag_target_version) + .send() + .await?; + Ok(tags + .tag_set() + .iter() + .any(|tag| tag.key() == "phase" && tag.value() == "after")) + }) + .await?; + assert_stable_single_version(&target, &target_bucket, tag_key, &tag_target_version).await?; + + // Retention extension and legal hold on an existing version. + let lock_key = "ledger/lock.bin"; + let locked = source_client + .put_object() + .bucket(source_bucket) + .key(lock_key) + .body(ByteStream::from(payload(4 * 1024, 0x02))) + .object_lock_mode(ObjectLockMode::Governance) + .object_lock_retain_until_date(retain_until()) + .send() + .await?; + let lock_source_version = locked.version_id().ok_or("source PUT returned no version id")?.to_string(); + assert_eq!( + wait_for_terminal_replication_status(&source_client, source_bucket, lock_key).await?, + "COMPLETED" + ); + let lock_target_version = single_target_version(&target, &target_bucket, lock_key)?; + let extended = DateTime::from_secs(retain_until().secs() + 86_400); + source_client + .put_object_retention() + .bucket(source_bucket) + .key(lock_key) + .version_id(&lock_source_version) + .retention( + ObjectLockRetention::builder() + .mode(ObjectLockRetentionMode::Governance) + .retain_until_date(extended) + .build(), + ) + .send() + .await?; + source_client + .put_object_legal_hold() + .bucket(source_bucket) + .key(lock_key) + .version_id(&lock_source_version) + .legal_hold(ObjectLockLegalHold::builder().status(ObjectLockLegalHoldStatus::On).build()) + .send() + .await?; + wait_until("retention extension and legal hold on the existing target version", || async { + let head = target_client + .head_object() + .bucket(&target_bucket) + .key(lock_key) + .version_id(&lock_target_version) + .send() + .await?; + Ok(head.object_lock_retain_until_date().map(|date| date.secs()) == Some(extended.secs()) + && head.object_lock_legal_hold_status() == Some(&ObjectLockLegalHoldStatus::On)) + }) + .await?; + source_client + .put_object_legal_hold() + .bucket(source_bucket) + .key(lock_key) + .version_id(&lock_source_version) + .legal_hold(ObjectLockLegalHold::builder().status(ObjectLockLegalHoldStatus::Off).build()) + .send() + .await?; + wait_until("legal hold removal on the existing target version", || async { + let head = target_client + .head_object() + .bucket(&target_bucket) + .key(lock_key) + .version_id(&lock_target_version) + .send() + .await?; + Ok(head.object_lock_legal_hold_status() == Some(&ObjectLockLegalHoldStatus::Off)) + }) + .await?; + assert_stable_single_version(&target, &target_bucket, lock_key, &lock_target_version).await?; + + // Permanent delete of the older of two same-content generations. + let generations_key = "ledger/generations.bin"; + let body = payload(4 * 1024, 0x03); + let older = source_client + .put_object() + .bucket(source_bucket) + .key(generations_key) + .body(ByteStream::from(body.clone())) + .send() + .await?; + let older_version = older.version_id().ok_or("source PUT returned no version id")?.to_string(); + assert_eq!( + wait_for_terminal_replication_status(&source_client, source_bucket, generations_key).await?, + "COMPLETED" + ); + let older_replica = single_target_version(&target, &target_bucket, generations_key)?; + source_client + .put_object() + .bucket(source_bucket) + .key(generations_key) + .body(ByteStream::from(body)) + .send() + .await?; + assert_eq!( + wait_for_terminal_replication_status(&source_client, source_bucket, generations_key).await?, + "COMPLETED" + ); + wait_until("both generations replicated", || async { + Ok(target.stored_versions(&target_bucket, generations_key).len() == 2) + }) + .await?; + let newer_replica = target + .stored_versions(&target_bucket, generations_key) + .into_iter() + .map(|(version_id, _)| version_id) + .find(|version_id| version_id != &older_replica) + .ok_or("the second generation must have its own target version")?; + + source_client + .delete_object() + .bucket(source_bucket) + .key(generations_key) + .version_id(&older_version) + .send() + .await?; + wait_until("permanent delete of the older generation's replica", || async { + let versions: Vec = target + .stored_versions(&target_bucket, generations_key) + .into_iter() + .map(|(version_id, _)| version_id) + .collect(); + Ok(versions == [newer_replica.clone()]) + }) + .await?; + assert_stable_single_version(&target, &target_bucket, generations_key, &newer_replica).await?; + + // No mutation above may have gone out as a re-PUT: one upload per key. + for key in [tag_key, lock_key] { + let puts = target + .requests() + .iter() + .filter(|record| record.key.as_deref() == Some(key) && record.operation == FakeTargetOperation::PutObject) + .count(); + assert_eq!( + puts, 1, + "{key}: a metadata update must not re-PUT the object on a target that mints its own ids" + ); + } + + target.shutdown().await; + Ok(()) +} + +/// rustfs/backlog#2340 (pending purge lifecycle): a permanent delete whose +/// replication keeps failing leaves the version in xl.meta as a PENDING purge, +/// hidden from listings. Once the bucket's replication configuration is +/// removed nothing can ever confirm that purge remotely, so the delete worker +/// must settle it locally (abandoned, with the replica left on the former +/// target) — otherwise the bucket stays `BucketNotEmpty` forever with a +/// residue the client cannot see. +#[tokio::test] +async fn matrix_removed_replication_config_abandons_pending_purge() -> TestResult { + init_logging(); + + let target = FakeS3Target::start().await?; + let target_bucket = "matrix-abandoned-purge-dst".to_string(); + target.create_bucket_with_object_lock(target_bucket.clone()); + TargetMode::MintOwnVersionIds.apply(&target); + + let mut env_vars = replication_fast_env(); + env_vars.extend_from_slice(LOOPBACK_REPLICATION_TARGET_ENV); + env_vars.extend_from_slice(&[ + ("NO_PROXY", "127.0.0.1,localhost"), + ("HTTP_PROXY", ""), + ("HTTPS_PROXY", ""), + // The scanner heal pass is what revisits a pending purge. + ("RUSTFS_SCANNER_CYCLE", "1"), + ("RUSTFS_SCANNER_START_DELAY_SECS", "1"), + ]); + let env = OdmTestEnv::start_with(OdmEnvOptions { + env: env_vars, + ..OdmEnvOptions::default() + }) + .await?; + let source_env = &env.rustfs; + + let source_bucket = "matrix-abandoned-purge-src"; + let source_client = source_env.create_s3_client(); + source_client + .create_bucket() + .bucket(source_bucket) + .object_lock_enabled_for_bucket(true) + .send() + .await?; + enable_bucket_versioning(source_env, source_bucket).await?; + let target_arn = set_replication_target_with_options( + source_env, + source_bucket, + ReplicationTargetOptions { + endpoint: &target.address(), + access_key: FAKE_ACCESS_KEY, + secret_key: FAKE_SECRET_KEY, + target_bucket: &target_bucket, + secure: false, + skip_tls_verify: false, + ca_cert_pem: None, + }, + ) + .await?; + put_bucket_replication_with_delete_statuses(source_env, source_bucket, &target_arn, "Enabled", Some("Enabled")).await?; + + let key = "purge/orphaned.bin"; + let put = source_client + .put_object() + .bucket(source_bucket) + .key(key) + .body(ByteStream::from(payload(4 * 1024, 0x07))) + .send() + .await?; + let source_version = put.version_id().ok_or("source PUT returned no version id")?.to_string(); + assert_eq!( + wait_for_terminal_replication_status(&source_client, source_bucket, key).await?, + "COMPLETED" + ); + let replica = single_target_version(&target, &target_bucket, key)?; + + // The target refuses every purge: the version stays a pending purge. + // More refusals than any scanner cycle can consume within the test. + target.inject_for_key(FakeTargetOperation::DeleteObject, key, FakeTargetFault::ResponseStatus(503), 4_000); + source_client + .delete_object() + .bucket(source_bucket) + .key(key) + .version_id(&source_version) + .send() + .await?; + wait_until("the refused purge to reach the target at least once", || async { + Ok(target.count_requests(FakeTargetOperation::DeleteObject, key) >= 1) + }) + .await?; + let listed = source_client.list_object_versions().bucket(source_bucket).send().await?; + assert!( + listed.versions().is_empty() && listed.delete_markers().is_empty(), + "a pending purge is hidden from listings: {listed:?}" + ); + let blocked = source_client.delete_bucket().bucket(source_bucket).send().await; + assert!( + blocked + .as_ref() + .err() + .and_then(|err| err.as_service_error()) + .is_some_and(|err| err.code() == Some("BucketNotEmpty")), + "the hidden pending purge must block DeleteBucket while the target is still configured: {blocked:?}" + ); + + // Removing the replication configuration orphans the purge; the scanner + // heal pass must settle it locally so the bucket becomes deletable. + let response = delete_bucket_replication(source_env, source_bucket).await?; + assert!(response.status().is_success(), "DeleteBucketReplication: {}", response.status()); + wait_until("DeleteBucket to succeed once the orphaned purge is abandoned", || async { + match source_client.delete_bucket().bucket(source_bucket).send().await { + Ok(_) => Ok(true), + Err(err) if err.as_service_error().is_some_and(|err| err.code() == Some("BucketNotEmpty")) => Ok(false), + Err(err) => Err(err.into()), + } + }) + .await?; + // Abandoned means abandoned: the replica stays on the former target and, + // once the attempts in flight at removal time have drained, no further + // purge attempts are sent to it. + assert_eq!( + single_target_version(&target, &target_bucket, key)?, + replica, + "an abandoned purge must not touch the replica on the former target" + ); + sleep(Duration::from_secs(3)).await; + let settled = target.count_requests(FakeTargetOperation::DeleteObject, key); + sleep(Duration::from_secs(3)).await; + assert_eq!( + target.count_requests(FakeTargetOperation::DeleteObject, key), + settled, + "purge attempts must stop once the target is no longer configured" + ); + + target.shutdown().await; + Ok(()) +} + +fn single_target_version(target: &FakeS3Target, target_bucket: &str, key: &str) -> Result> { + let versions = target.stored_versions(target_bucket, key); + match versions.as_slice() { + [(version_id, false)] => Ok(version_id.clone()), + other => Err(format!("{key}: expected exactly one live target version, got {other:?}").into()), + } +} + +/// The target keeps holding exactly `version_id` for a few scanner cycles: a +/// re-driven PUT or a wrong delete would show up here. +async fn assert_stable_single_version(target: &FakeS3Target, target_bucket: &str, key: &str, version_id: &str) -> TestResult { + for _ in 0..8 { + let versions = target.stored_versions(target_bucket, key); + if versions.len() != 1 || versions[0].0 != version_id { + return Err( + format!("{key}: target versions drifted from the single expected replica {version_id}: {versions:?}").into(), + ); + } + sleep(Duration::from_millis(500)).await; + } + Ok(()) +} + +async fn wait_until(what: &str, mut probe: F) -> TestResult +where + F: FnMut() -> Fut, + Fut: std::future::Future>>, +{ + let wait = async { + loop { + if probe().await? { + return Ok::<_, Box>(()); + } + sleep(Duration::from_millis(250)).await; + } + }; + timeout(Duration::from_secs(90), wait) + .await + .map_err(|_| format!("{what} did not happen within 90 seconds"))? +} + /// Wait until `key` is COMPLETED on the source and, for the observation /// window after that, the target still holds exactly one live version of it. async fn wait_for_replication_status_and_single_version( diff --git a/crates/ecstore/src/api/mod.rs b/crates/ecstore/src/api/mod.rs index 9356f0ff0..7f592685a 100644 --- a/crates/ecstore/src/api/mod.rs +++ b/crates/ecstore/src/api/mod.rs @@ -33,6 +33,7 @@ pub mod bucket { pub use crate::bucket::bucket_target_sys::{ AdvancedPutOptions, BucketTargetError, BucketTargetSys, PutObjectOptions, RemoveObjectOptions, S3ClientError, SsecPassthroughCapability, TargetClient, VersionIdentityCapability, append_version_id_query, + resolve_delete_api_version_id, }; } diff --git a/crates/ecstore/src/bucket/bucket_target_sys.rs b/crates/ecstore/src/bucket/bucket_target_sys.rs index b0634c9ce..99ed2ccb4 100644 --- a/crates/ecstore/src/bucket/bucket_target_sys.rs +++ b/crates/ecstore/src/bucket/bucket_target_sys.rs @@ -33,6 +33,8 @@ use aws_sdk_s3::operation::get_object::{GetObjectError, GetObjectOutput}; use aws_sdk_s3::operation::get_object_tagging::{GetObjectTaggingError, GetObjectTaggingOutput}; use aws_sdk_s3::operation::head_bucket::HeadBucketError; use aws_sdk_s3::operation::head_object::HeadObjectError; +use aws_sdk_s3::operation::put_object_legal_hold::{PutObjectLegalHoldError, PutObjectLegalHoldOutput}; +use aws_sdk_s3::operation::put_object_retention::{PutObjectRetentionError, PutObjectRetentionOutput}; use aws_sdk_s3::operation::put_object_tagging::{PutObjectTaggingError, PutObjectTaggingOutput}; use aws_sdk_s3::operation::upload_part::UploadPartOutput; use aws_sdk_s3::primitives::ByteStream; @@ -42,6 +44,7 @@ use aws_sdk_s3::types::{ ChecksumAlgorithm, ChecksumMode, CompletedMultipartUpload, CompletedPart, ObjectLockLegalHoldStatus, ObjectLockRetentionMode, ServerSideEncryption, }; +use aws_sdk_s3::types::{ObjectLockLegalHold, ObjectLockRetention}; use aws_sdk_s3::{Client as S3Client, operation::head_object::HeadObjectOutput}; use aws_smithy_runtime_api::client::orchestrator::HttpRequest; use futures::{StreamExt, stream}; @@ -139,9 +142,12 @@ fn same_replication_service(edited: &BucketTarget, previous: &BucketTarget) -> b && access_key(edited) == access_key(previous) } -/// Page size and page budget for [`TargetClient::find_version_by_etag`]. +/// Page size and page budget for [`TargetClient::locate_replica_by_etag`]. const FIND_VERSION_BY_ETAG_PAGE_SIZE: i32 = 1000; const FIND_VERSION_BY_ETAG_MAX_PAGES: usize = 8; +/// Candidate cap for [`TargetClient::replica_candidates_by_etag`]: more than +/// this many same-content versions of one key is ambiguity by any measure. +const FIND_VERSION_BY_ETAG_MAX_MATCHES: usize = 16; pub type GetObjectSdkError = Box>; pub type GetObjectTaggingSdkError = Box>; pub type PutObjectTaggingSdkError = Box>; @@ -1363,6 +1369,7 @@ fn generate_arn(t: &BucketTarget, depl_id: &str) -> String { arn.to_string() } +#[derive(Debug, Clone)] pub struct RemoveObjectOptions { pub force_delete: bool, pub governance_bypass: bool, @@ -1418,7 +1425,12 @@ fn build_remove_object_headers(version_id: Option<&str>, opts: &RemoveObjectOpti /// and silently creates a delete marker instead of removing the version, while /// the source stamps `VersionPurgeStatus=Complete` (backlog#799 B8 / #857). /// Non-replication callers always pass the version through unchanged. -fn resolve_delete_api_version_id(version_id: Option, opts: &RemoveObjectOptions) -> Option { +/// The `versionId` a replicated DELETE puts on the wire: none for a +/// delete-marker creation (the target mints the marker; the source version +/// travels in the internal headers for RustFS peers), the addressed version +/// otherwise. A generic S3 target given the version id on a marker-creation +/// DELETE would permanently delete that version instead. +pub fn resolve_delete_api_version_id(version_id: Option, opts: &RemoveObjectOptions) -> Option { if opts.replication_request && opts.replication_delete_marker { None } else { @@ -1971,22 +1983,29 @@ impl TargetClient { .map_err(Box::new) } - /// Locate a replica by content identity on a target that mints its own - /// version ids: page `ListObjectVersions` under the exact key and return - /// the newest live version whose ETag matches `source_etag`. Delete - /// markers and prefix siblings never match. Bounded to - /// [`FIND_VERSION_BY_ETAG_MAX_PAGES`] pages so a key with a very deep - /// history cannot turn one convergence check into an unbounded scan; a - /// replica beyond that window reads as missing, which only costs a + /// Candidate replicas by content identity on a target that mints its own + /// version ids: page `ListObjectVersions` under the exact key and report + /// the live versions whose ETag matches `source_etag`, newest first. + /// Delete markers and prefix siblings never match. Bounded to + /// [`FIND_VERSION_BY_ETAG_MAX_PAGES`] pages and + /// [`FIND_VERSION_BY_ETAG_MAX_MATCHES`] candidates so a key with a very + /// deep history cannot turn one convergence check into an unbounded scan; + /// a replica beyond that window reads as missing, which only costs a /// re-PUT (today's behaviour), never a lost object. - pub async fn find_version_by_etag( + /// + /// Content identity is not version identity: two source generations with + /// the same bytes have the same ETag. Callers drop the candidates other + /// source versions already claim through their ledgers and refuse an + /// [`ReplicaLocation::Ambiguous`] remainder before mutating or deleting. + pub async fn replica_candidates_by_etag( &self, bucket: &str, object: &str, source_etag: &str, - ) -> Result, Box>> { + ) -> Result, Box>> { let mut key_marker: Option = None; let mut version_id_marker: Option = None; + let mut matches: Vec = Vec::new(); for _ in 0..FIND_VERSION_BY_ETAG_MAX_PAGES { let page = self .client @@ -1999,32 +2018,88 @@ impl TargetClient { .send() .await .map_err(Box::new)?; - if let Some(version) = page.versions().iter().find(|version| { - version.key() == Some(object) - && version.version_id().is_some_and(|id| !id.is_empty()) - && replication_etags_match(Some(source_etag), version.e_tag()) - }) { - return Ok(version.version_id().map(str::to_string)); - } - // Every listed key is >= the prefix; once the listing moved past - // the exact key there is nothing left to find. - if page - .versions() - .iter() - .any(|version| version.key().is_some_and(|key| key > object)) + matches.extend( + page.versions() + .iter() + .filter(|version| { + version.key() == Some(object) + && version.version_id().is_some_and(|id| !id.is_empty()) + && replication_etags_match(Some(source_etag), version.e_tag()) + }) + .filter_map(|version| version.version_id().map(str::to_string)), + ); + // A listing that moved past the exact key (every listed key is >= + // the prefix), ended, or already filled the candidate cap decides. + if matches.len() >= FIND_VERSION_BY_ETAG_MAX_MATCHES + || page + .versions() + .iter() + .any(|version| version.key().is_some_and(|key| key > object)) + || !page.is_truncated().unwrap_or(false) { - return Ok(None); - } - if !page.is_truncated().unwrap_or(false) { - return Ok(None); + break; } key_marker = page.next_key_marker().map(str::to_string); version_id_marker = page.next_version_id_marker().map(str::to_string); if key_marker.is_none() { - return Ok(None); + break; } } - Ok(None) + matches.truncate(FIND_VERSION_BY_ETAG_MAX_MATCHES); + Ok(matches) + } + + /// PutObjectRetention against a replica version on a target that does not + /// take retention through the replication PUT's own headers (it mints its + /// own version ids, so a re-PUT would create another version instead of + /// updating this one). Anti-loop marker always added. + pub async fn put_object_retention( + &self, + bucket: &str, + object: &str, + version_id: Option, + mode: ObjectLockRetentionMode, + retain_until: aws_sdk_s3::primitives::DateTime, + ) -> Result>> { + let headers = proxy_outbound_headers(HeaderMap::new()); + self.client + .put_object_retention() + .bucket(bucket) + .key(object) + .set_version_id(resolve_read_api_version_id(version_id)) + .retention( + ObjectLockRetention::builder() + .mode(mode) + .retain_until_date(retain_until) + .build(), + ) + .customize() + .map_request(move |req| apply_extra_headers(req, &headers)) + .send() + .await + .map_err(Box::new) + } + + /// PutObjectLegalHold counterpart of [`Self::put_object_retention`]. + pub async fn put_object_legal_hold( + &self, + bucket: &str, + object: &str, + version_id: Option, + status: ObjectLockLegalHoldStatus, + ) -> Result>> { + let headers = proxy_outbound_headers(HeaderMap::new()); + self.client + .put_object_legal_hold() + .bucket(bucket) + .key(object) + .set_version_id(resolve_read_api_version_id(version_id)) + .legal_hold(ObjectLockLegalHold::builder().status(status).build()) + .customize() + .map_request(move |req| apply_extra_headers(req, &headers)) + .send() + .await + .map_err(Box::new) } /// HEAD used by the read-proxy path (GET/HEAD of an object not yet @@ -2478,6 +2553,45 @@ impl TargetClient { } } +/// Where a replica stands on a target that mints its own version ids, by +/// content identity (exact key + ETag) after the candidates other source +/// versions claim were removed. See +/// [`TargetClient::replica_candidates_by_etag`]. +#[derive(Debug, Clone, PartialEq, Eq)] +pub enum ReplicaLocation { + /// No live version under the key carries the source ETag. + Missing, + /// Exactly one live version carries it: safe to address. + Unique(String), + /// More than one live version carries it (same bytes replicated for + /// several source generations). `newest` is the most recently listed + /// one — good enough to prove the replica exists, never good enough to + /// pick which one to mutate or delete. + Ambiguous { newest: String }, +} + +impl ReplicaLocation { + /// `matches` newest first, as the target listed them. + pub fn from_matches(mut matches: Vec) -> Self { + match matches.len() { + 0 => Self::Missing, + 1 => Self::Unique(matches.remove(0)), + _ => Self::Ambiguous { + newest: matches.remove(0), + }, + } + } + + /// The version to read for existence/ETag checks, where an ambiguous + /// match is still a located replica. + pub fn any_version_id(&self) -> Option<&str> { + match self { + Self::Missing => None, + Self::Unique(version_id) | Self::Ambiguous { newest: version_id } => Some(version_id), + } + } +} + #[derive(Debug)] pub enum BucketTargetError { BucketRemoteTargetNotFound { diff --git a/crates/ecstore/src/bucket/replication/replication_pool.rs b/crates/ecstore/src/bucket/replication/replication_pool.rs index d733a3134..c1f5ae916 100644 --- a/crates/ecstore/src/bucket/replication/replication_pool.rs +++ b/crates/ecstore/src/bucket/replication/replication_pool.rs @@ -3171,7 +3171,11 @@ pub async fn queue_replication_heal(bucket: &str, oi: ObjectInfo, retry_count: u } let rcfg = match ReplicationMetadataStore::optional_replication_config(bucket).await { - Ok(Some(config)) => config, + Ok(Some(config)) => Some(config), + // A bucket without a configuration still owes its pending purges an + // answer: the delete worker finishes them locally as abandoned, which + // is what makes the bucket deletable again (rustfs/backlog#2340). + Ok(None) if owes_version_purge(&oi) => None, Ok(None) => return ReplicationQueueAdmission::Skipped, Err(err) => { debug!( @@ -3221,7 +3225,7 @@ pub async fn queue_replication_heal(bucket: &str, oi: ObjectInfo, retry_count: u } }; - let rcfg_wrapper = ReplicationConfig::new(Some(rcfg), tgts); + let rcfg_wrapper = ReplicationConfig::new(rcfg, tgts); queue_replication_heal_internal(bucket, oi, rcfg_wrapper, retry_count) .await .admission @@ -3249,6 +3253,17 @@ pub async fn queue_replication_metadata(bucket: &str, oi: ObjectInfo, retry_coun } } +/// A version purge the persisted state still owes to named targets. Without +/// the target list nothing can be settled, so such a version keeps the +/// ordinary "no configuration, nothing to heal" skip. +fn owes_version_purge(oi: &ObjectInfo) -> bool { + !oi.version_purge_status.is_empty() + && oi + .version_purge_status_internal + .as_deref() + .is_some_and(|statuses| !statuses.trim().is_empty()) +} + /// queue_replication_heal_internal enqueues objects that failed replication OR eligible for resyncing through /// an ongoing resync operation or via existing objects replication configuration setting. pub(crate) async fn queue_replication_heal_internal( @@ -3267,7 +3282,11 @@ pub(crate) async fn queue_replication_heal_internal( }; } - if rcfg.config.is_none() || rcfg.remotes.is_none() { + // Without a configuration or targets there is nothing to replicate — + // except a version purge the bucket still owes: its stored decision names + // the targets, and the delete worker settles the ones no longer + // configured as abandoned (rustfs/backlog#2340). + if (rcfg.config.is_none() || rcfg.remotes.is_none()) && !owes_version_purge(&oi) { return ReplicationHealQueueResult { object_info: roi, admission: ReplicationQueueAdmission::Skipped, @@ -3312,12 +3331,15 @@ pub(crate) async fn queue_replication_heal_internal( } ReplicationHealQueueAction::QueueDelete(dv) => { // A purge the peer denied under object lock cannot succeed until - // the lock lapses (#6850); requeuing it every heal cycle only + // the lock lapses (#6850), and one whose replica cannot be told + // apart on a target that mints its own version ids cannot + // succeed until the ledger or an operator resolves it + // (rustfs/backlog#2340); requeuing either every heal cycle only // burns bandwidth and failure counters. The backoff expires on // its own, so the purge is probed again — and converges — once - // the retention window has a chance of being over. + // the condition has a chance of being over. if super::replication_object_decision_boundary::is_version_delete_replication(&dv.delete_object) - && super::replication_resyncer::object_lock_denied_purge_backoff_active(&dv) + && super::replication_resyncer::purge_backoff_active(&dv) { return ReplicationHealQueueResult { object_info: roi, diff --git a/crates/ecstore/src/bucket/replication/replication_resyncer.rs b/crates/ecstore/src/bucket/replication/replication_resyncer.rs index 5d23fc6eb..39d20d8df 100644 --- a/crates/ecstore/src/bucket/replication/replication_resyncer.rs +++ b/crates/ecstore/src/bucket/replication/replication_resyncer.rs @@ -22,7 +22,8 @@ use super::replication_filemeta_boundary::ReplicationGenerationSnapshot; use super::replication_filemeta_boundary::{ REPLICATE_EXISTING, ReplicateDecision, ReplicateObjectInfo, ReplicatedInfos, ReplicatedTargetInfo, ReplicationAction, ReplicationState, ReplicationStatusType, ReplicationType, VersionPurgeStatusType, get_replication_state, - parse_replicate_decision, replication_statuses_map, target_reset_header, version_purge_statuses_map, + parse_replicate_decision, replicate_decision_for_admitted_targets, replication_statuses_map, target_reset_header, + version_purge_statuses_map, }; use super::replication_lock_boundary::ReplicationLockTiming; use super::replication_logging::{EVENT_RESYNC_CONFIG_LOOKUP_SKIPPED, LOG_COMPONENT_ECSTORE, LOG_SUBSYSTEM_REPLICATION_RESYNC}; @@ -56,23 +57,23 @@ use super::replication_storage_boundary::{ }; #[cfg(test)] use super::replication_storage_boundary::{NamespaceLockFence, NamespaceLockSignalTestFence, ReplicationDeletedObject}; -#[cfg(test)] -use super::replication_target_boundary::VersionIdentityCapability; +use super::replication_tagging_boundary::ReplicationTagFilter; use super::replication_target_boundary::{ ERR_REPLICATION_SSEC_PASSTHROUGH_UNSUPPORTED, HeadObjectSdkError, PutObjectOptions, PutObjectPartOptions, - RemotePutObjectResponse, ReplicationTargetStore, S3ClientError, SsecPassthroughCapability, SsecPassthroughGate, TargetClient, - is_replication_target_offline_error, replication_action_for_target_head, replication_complete_multipart_options, - replication_delete_marker_purge_remove_options, replication_delete_remove_options, replication_force_delete_remove_options, - replication_object_is_ssec_encrypted, replication_put_object_header_size, replication_put_object_options, - replication_target_head_is_newer_null_version, resolve_read_api_version_id, ssec_passthrough_evidence_present, - ssec_passthrough_gate, version_identity_capability_from_put, version_identity_drifted, + RemotePutObjectResponse, ReplicaLocation, ReplicationTargetStore, S3ClientError, SsecPassthroughCapability, + SsecPassthroughGate, TargetClient, VersionIdentityCapability, is_replication_target_offline_error, + replication_action_for_target_head, replication_complete_multipart_options, replication_delete_marker_purge_remove_options, + replication_delete_remove_options, replication_force_delete_remove_options, replication_object_is_ssec_encrypted, + replication_put_object_header_size, replication_put_object_options, replication_target_head_is_newer_null_version, + resolve_read_api_version_id, ssec_passthrough_evidence_present, ssec_passthrough_gate, version_identity_capability_from_put, + version_identity_drifted, }; use super::replication_versioning_boundary::ReplicationVersioningStore; use super::runtime_boundary as runtime_sources; use aws_sdk_s3::error::{ProvideErrorMetadata, SdkError}; use aws_sdk_s3::operation::head_object::{HeadObjectError, HeadObjectOutput}; use aws_sdk_s3::primitives::ByteStream; -use aws_sdk_s3::types::CompletedPart; +use aws_sdk_s3::types::{CompletedPart, ObjectLockLegalHoldStatus, ObjectLockRetentionMode, Tag, Tagging}; use aws_smithy_types::body::SdkBody; use futures::future::join_all; use futures::stream::StreamExt; @@ -84,8 +85,9 @@ use metrics::counter; use rmp_serde; use rustfs_s3_types::EventName; use rustfs_utils::http::{ - AMZ_BUCKET_REPLICATION_STATUS, AMZ_TAGGING_DIRECTIVE, SUFFIX_REPLICATION_RESET, SUFFIX_REPLICATION_STATUS, - has_internal_suffix, insert_str, + AMZ_BUCKET_REPLICATION_STATUS, AMZ_OBJECT_LOCK_LEGAL_HOLD, AMZ_OBJECT_LOCK_MODE, AMZ_OBJECT_LOCK_RETAIN_UNTIL_DATE, + AMZ_TAGGING_DIRECTIVE, SUFFIX_REPLICATION_RESET, SUFFIX_REPLICATION_STATUS, SUFFIX_REPLICATION_TARGET_VERSION_ARN_PREFIX, + has_internal_suffix, insert_str, replication_target_versions, }; use rustfs_utils::{DEFAULT_SIP_HASH_KEY, get_env_usize, sip_hash}; #[cfg(test)] @@ -128,6 +130,11 @@ const EVENT_REPLICATION_VERSION_IDENTITY_DRIFT: &str = "replication_version_iden const EVENT_REPLICATION_DRIFTED_REPLICA_LOCATED: &str = "replication_drifted_replica_located"; const EVENT_REPLICATION_OBJECT_FAILED: &str = "replication_object_failed"; const EVENT_REPLICATION_PURGE_OBJECT_LOCK_DENIED: &str = "replication_purge_object_lock_denied"; +const EVENT_REPLICATION_PURGE_REPLICA_UNRESOLVED: &str = "replication_purge_replica_unresolved"; +const EVENT_REPLICATION_PURGE_ABANDONED: &str = "replication_purge_abandoned"; +const METRIC_VERSION_PURGE_ABANDONED_TOTAL: &str = "rustfs_replication_version_purge_abandoned_total"; +const EVENT_REPLICATION_DRIFTED_REPLICA_METADATA_SYNCED: &str = "replication_drifted_replica_metadata_synced"; +const METRIC_VERSION_PURGE_REPLICA_TOTAL: &str = "rustfs_replication_version_purge_replica_total"; #[allow( dead_code, @@ -213,30 +220,39 @@ const VERSION_IDENTITY_DRIFT_LOG_INTERVAL: TokioDuration = TokioDuration::from_s static VERSION_IDENTITY_WARNED_ARNS: LazyLock>> = LazyLock::new(|| StdMutex::new(HashMap::new())); -/// Version purges the peer denied under object lock (#6850). A RustFS peer -/// with the replicated-purge GOVERNANCE exemption -/// (`replication_delete_may_bypass_governance`) no longer produces this for -/// governance retention, but COMPLIANCE retention, legal hold, and targets -/// without the exemption (older RustFS, MinIO, generic S3) still deny — and -/// such a purge cannot succeed until the lock on the replica lapses, so -/// retrying every heal cycle only burns bandwidth and failure counters. -/// Entries suppress heal requeues for the backoff window; after it expires -/// one probe runs again, so the purge still converges on its own once -/// retention ends. In-process only: a restart costs at most one extra probe -/// per entry. -const OBJECT_LOCK_DENIED_PURGE_BACKOFF: std::time::Duration = std::time::Duration::from_secs(60 * 60); -const OBJECT_LOCK_DENIED_PURGE_CACHE_MAX: usize = 4096; -type ObjectLockDeniedPurgeKey = (String, String, String); +/// Version purges a target refused for a reason a retry cannot change on its +/// own. Two shapes today: +/// +/// - the peer denied the purge under object lock (#6850). A RustFS peer with +/// the replicated-purge GOVERNANCE exemption +/// (`replication_delete_may_bypass_governance`) no longer produces this for +/// governance retention, but COMPLIANCE retention, legal hold, and targets +/// without the exemption (older RustFS, MinIO, generic S3) still deny — and +/// such a purge cannot succeed until the lock on the replica lapses; +/// - the replica cannot be identified on a target that mints its own +/// version ids (rustfs/backlog#2340): no ledger entry and more than one +/// target version carries the source ETag, so any pick could destroy a +/// live generation. Only an operator (or the ledger catching up through +/// heal) changes that. +/// +/// Retrying either every heal cycle only burns bandwidth and failure +/// counters. Entries suppress heal requeues for the backoff window; after it +/// expires one probe runs again, so the purge still converges on its own +/// once the condition lifts. In-process only: a restart costs at most one +/// extra probe per entry. +const PURGE_BACKOFF: std::time::Duration = std::time::Duration::from_secs(60 * 60); +const PURGE_BACKOFF_CACHE_MAX: usize = 4096; +type PurgeBackoffKey = (String, String, String); -struct ObjectLockDeniedPurge { - denied_at: std::time::Instant, - denied_arns: HashSet, +struct DeferredPurge { + deferred_at: std::time::Instant, + deferred_arns: HashSet, } -static OBJECT_LOCK_DENIED_PURGES: LazyLock>> = +static DEFERRED_PURGES: LazyLock>> = LazyLock::new(|| StdMutex::new(HashMap::new())); -fn object_lock_denied_purge_key(dobj: &DeletedObjectReplicationInfo) -> ObjectLockDeniedPurgeKey { +fn purge_backoff_key(dobj: &DeletedObjectReplicationInfo) -> PurgeBackoffKey { let version_id = dobj .delete_object .delete_marker_version_id @@ -245,42 +261,38 @@ fn object_lock_denied_purge_key(dobj: &DeletedObjectReplicationInfo) -> ObjectLo (dobj.bucket.clone(), dobj.delete_object.object_name.clone(), version_id.to_string()) } -fn record_object_lock_denied_purge(dobj: &DeletedObjectReplicationInfo, arn: &str) { - let mut denied = OBJECT_LOCK_DENIED_PURGES - .lock() - .unwrap_or_else(|poisoned| poisoned.into_inner()); - if denied.len() >= OBJECT_LOCK_DENIED_PURGE_CACHE_MAX { - denied.retain(|_, entry| entry.denied_at.elapsed() < OBJECT_LOCK_DENIED_PURGE_BACKOFF); +fn record_purge_backoff(dobj: &DeletedObjectReplicationInfo, arn: &str) { + let mut deferred = DEFERRED_PURGES.lock().unwrap_or_else(|poisoned| poisoned.into_inner()); + if deferred.len() >= PURGE_BACKOFF_CACHE_MAX { + deferred.retain(|_, entry| entry.deferred_at.elapsed() < PURGE_BACKOFF); } - let key = object_lock_denied_purge_key(dobj); - if denied.len() < OBJECT_LOCK_DENIED_PURGE_CACHE_MAX || denied.contains_key(&key) { - let entry = denied.entry(key).or_insert_with(|| ObjectLockDeniedPurge { - denied_at: std::time::Instant::now(), - denied_arns: HashSet::new(), + let key = purge_backoff_key(dobj); + if deferred.len() < PURGE_BACKOFF_CACHE_MAX || deferred.contains_key(&key) { + let entry = deferred.entry(key).or_insert_with(|| DeferredPurge { + deferred_at: std::time::Instant::now(), + deferred_arns: HashSet::new(), }); - entry.denied_at = std::time::Instant::now(); - entry.denied_arns.insert(arn.to_string()); + entry.deferred_at = std::time::Instant::now(); + entry.deferred_arns.insert(arn.to_string()); } // Still full after dropping expired entries: skip recording — the purge // then simply keeps retrying, which is the pre-#6850 behavior. } -/// Whether a heal requeue of this delete can only reach targets that denied -/// it under object lock within the backoff window. A target the entry does -/// not cover (another peer, or one whose denial expired) keeps the requeue -/// flowing — suppressing it would delay a purge that could succeed there. -pub(crate) fn object_lock_denied_purge_backoff_active(dobj: &DeletedObjectReplicationInfo) -> bool { - let key = object_lock_denied_purge_key(dobj); - let mut denied = OBJECT_LOCK_DENIED_PURGES - .lock() - .unwrap_or_else(|poisoned| poisoned.into_inner()); - match denied.get(&key) { - Some(entry) if entry.denied_at.elapsed() < OBJECT_LOCK_DENIED_PURGE_BACKOFF => { +/// Whether a heal requeue of this delete can only reach targets that +/// deferred it within the backoff window. A target the entry does not cover +/// (another peer, or one whose deferral expired) keeps the requeue flowing — +/// suppressing it would delay a purge that could succeed there. +pub(crate) fn purge_backoff_active(dobj: &DeletedObjectReplicationInfo) -> bool { + let key = purge_backoff_key(dobj); + let mut deferred = DEFERRED_PURGES.lock().unwrap_or_else(|poisoned| poisoned.into_inner()); + match deferred.get(&key) { + Some(entry) if entry.deferred_at.elapsed() < PURGE_BACKOFF => { let admitted = dobj.admitted_target_arns(); - !admitted.is_empty() && admitted.iter().all(|arn| entry.denied_arns.contains(arn)) + !admitted.is_empty() && admitted.iter().all(|arn| entry.deferred_arns.contains(arn)) } Some(_) => { - denied.remove(&key); + deferred.remove(&key); false } None => false, @@ -410,73 +422,420 @@ async fn mark_replication_target_offline_if_needed(target_client: &Arc std::result::Result, HeadObjectSdkError> { +) -> std::result::Result, HeadObjectSdkError> { match head_object_for_worker(tgt_client, &tgt_client.bucket, object, None).await { - Ok(oi) => Ok(Some(oi)), + Ok(head) => Ok(Some(LocatedReplica { + version_id: head.version_id.clone(), + head, + })), Err(e) if head_object_not_found(&e) => Ok(None), Err(e) => Err(e), } } +/// A replica reached through [`replica_head_fallback`], with the version id +/// it was addressed by: the ledger or located id on a drifting target, or +/// whatever the current-version HEAD reported. Kept apart from the HEAD +/// output because a target may withhold `x-amz-version-id` on HEAD. +struct LocatedReplica { + head: HeadObjectOutput, + version_id: Option, +} + fn head_object_not_found(err: &HeadObjectSdkError) -> bool { err.as_service_error().is_some_and(|se| se.is_not_found()) || has_raw_status(err, 404) } +/// Target version ids other versions of the same source key already claim +/// through their ledgers, for one target. Content identity alone cannot tell +/// two same-bytes generations apart on a target that mints its own ids; the +/// sibling ledgers can — a candidate a sibling recorded is that sibling's +/// replica, never this version's. +#[async_trait::async_trait] +pub(crate) trait SiblingLedger: Send + Sync { + async fn claimed_target_versions(&self, bucket: &str, object: &str, exclude: Option, arn: &str) -> HashSet; +} + +/// No sibling knowledge: every content match stays a candidate. For paths +/// that only count replicas and never mutate them (resync verification), and +/// for tests. +pub(crate) struct NoSiblingLedger; + +#[async_trait::async_trait] +impl SiblingLedger for NoSiblingLedger { + async fn claimed_target_versions(&self, _bucket: &str, _object: &str, _exclude: Option, _arn: &str) -> HashSet { + HashSet::new() + } +} + +/// Source-side sibling ledgers: one listing of the exact key. Bounded to a +/// page; a key with more generations than that keeps every candidate, which +/// only ever makes the caller refuse (ambiguous), never guess. +const SIBLING_LEDGER_MAX_VERSIONS: i32 = 1000; + +#[async_trait::async_trait] +impl SiblingLedger for Arc { + async fn claimed_target_versions(&self, bucket: &str, object: &str, exclude: Option, arn: &str) -> HashSet { + let listed = match self + .clone() + .list_object_versions(bucket, object, None, None, None, SIBLING_LEDGER_MAX_VERSIONS) + .await + { + Ok(listed) => listed, + Err(error) => { + debug!( + event = EVENT_RESYNC_RUNTIME_SKIPPED, + component = LOG_COMPONENT_ECSTORE, + subsystem = LOG_SUBSYSTEM_REPLICATION_RESYNC, + bucket, + object, + error = %error, + reason = "sibling_ledger_listing_failed", + "Could not list sibling versions for the target-version ledger" + ); + return HashSet::new(); + } + }; + listed + .objects + .iter() + .filter(|info| info.name == object && info.version_id != exclude) + .filter_map(|info| ledger_target_version_id(&info.user_defined, arn)) + .collect() + } +} + +/// The version id a drifting target assigned to this object version, as the +/// source's persisted ledger records it for `arn` (see +/// `SUFFIX_REPLICATION_TARGET_VERSION_ARN_PREFIX`). `None` when nothing is +/// recorded or the record is inconsistent across the dual internal prefixes +/// (callers then fall back to content identity, never to a guess). +fn ledger_target_version_id(user_defined: &HashMap, arn: &str) -> Option { + let (ledger, corrupt) = replication_target_versions(user_defined); + if corrupt { + return None; + } + ledger.get(arn).cloned() +} + /// Second look at a replica whose version-addressed HEAD failed, for the two /// target shapes where that failure is not a verdict on the replica: /// /// - AWS-style 400/403 (the RustFS uuid is rejected as malformed): HEAD the /// current version without a version id; callers compare ETags. -/// - 404 on a target known to mint its own version ids (the Wasabi shape, -/// rustfs/backlog#2340): the source id never existed there, so locate the -/// replica by exact key and ETag through ListObjectVersions and HEAD the id -/// the target assigned. Without this, every heal, MRF retry and -/// existing-object resync re-drive PUTs the object again and mints one -/// more target version. +/// - 404 on a target that mints its own version ids (the Wasabi shape, +/// rustfs/backlog#2340): the source id never existed there. HEAD the id +/// the source ledger recorded for this target when it has one; otherwise, +/// once the target is known to mint ids, locate the replica by exact key +/// and ETag through ListObjectVersions and HEAD the id the target assigned. +/// Without this, every heal, MRF retry and existing-object resync re-drive +/// PUTs the object again and mints one more target version. +/// +/// `content_identity` is the sibling-ledger view used to disown candidates +/// other source versions already claim; `None` disables content identity +/// altogether — a fresh write (`ReplicationType::Object`) has no replica to +/// find, and a same-bytes older generation would be mistaken for one. /// /// `None` when the error stands as-is: a real miss on an adopting target, or /// a target whose identity contract is still unknown. A failed lookup is /// returned as a HEAD-shaped error so callers keep their "target operation /// failed" handling (retry later) instead of re-driving the PUT. +/// The source side of a replica lookup: which version is being matched and +/// what the source already recorded about it on this target. +struct ReplicaSource<'a> { + bucket: &'a str, + object: &'a str, + version_id: Option, + etag: Option<&'a str>, + ledger_version_id: Option<&'a str>, +} + async fn replica_head_fallback( tgt_client: &TargetClient, - object: &str, - source_etag: Option<&str>, + source: ReplicaSource<'_>, + content_identity: Option<&dyn SiblingLedger>, err: &HeadObjectSdkError, -) -> Option, HeadObjectSdkError>> { +) -> Option, HeadObjectSdkError>> { + let ReplicaSource { + bucket: source_bucket, + object, + version_id: source_version_id, + etag: source_etag, + ledger_version_id, + } = source; if is_version_id_format_mismatch(err) { return Some(head_object_fallback(tgt_client, object).await); } - if !head_object_not_found(err) - || !ReplicationTargetStore::version_identity_capability(&tgt_client.arn).version_addressing_unreliable() - { + if !head_object_not_found(err) { return None; } + if let Some(ledger_version_id) = ledger_version_id { + // A ledger entry is direct evidence of the identity contract; it + // survives a restart while the runtime verdict cache does not. + ReplicationTargetStore::record_version_identity_capability(&tgt_client.arn, VersionIdentityCapability::MintsOwn); + match head_object_for_worker(tgt_client, &tgt_client.bucket, object, Some(ledger_version_id.to_string())).await { + Ok(head) => { + return Some(Ok(Some(LocatedReplica { + head, + version_id: Some(ledger_version_id.to_string()), + }))); + } + // The recorded version is gone: fall through to content identity + // before concluding that the replica is missing. + Err(e) if head_object_not_found(&e) => {} + Err(e) => return Some(Err(e)), + } + } + if !ReplicationTargetStore::version_identity_capability(&tgt_client.arn).version_addressing_unreliable() { + return None; + } + let siblings = content_identity?; let etag = source_etag.filter(|etag| !etag.trim().is_empty())?; - Some(match tgt_client.find_version_by_etag(&tgt_client.bucket, object, etag).await { - Ok(Some(assigned_version_id)) => { - debug!( - event = EVENT_REPLICATION_DRIFTED_REPLICA_LOCATED, - component = LOG_COMPONENT_ECSTORE, - subsystem = LOG_SUBSYSTEM_REPLICATION_RESYNC, - bucket = %tgt_client.bucket, - object = %object, - arn = %tgt_client.arn, - assigned_version_id = %assigned_version_id, - "Located replica by content identity on a target that mints its own version ids" - ); - match head_object_for_worker(tgt_client, &tgt_client.bucket, object, Some(assigned_version_id)).await { - Ok(oi) => Ok(Some(oi)), - // The located version disappeared between LIST and HEAD. - Err(e) if head_object_not_found(&e) => Ok(None), - Err(e) => Err(e), + Some(match tgt_client.replica_candidates_by_etag(&tgt_client.bucket, object, etag).await { + Ok(candidates) => { + let location = + disowned_replica_location(siblings, &tgt_client.arn, source_bucket, object, source_version_id, candidates).await; + match location.any_version_id() { + Some(assigned_version_id) => { + debug!( + event = EVENT_REPLICATION_DRIFTED_REPLICA_LOCATED, + component = LOG_COMPONENT_ECSTORE, + subsystem = LOG_SUBSYSTEM_REPLICATION_RESYNC, + bucket = %tgt_client.bucket, + object = %object, + arn = %tgt_client.arn, + assigned_version_id = %assigned_version_id, + ambiguous = matches!(location, ReplicaLocation::Ambiguous { .. }), + "Located replica by content identity on a target that mints its own version ids" + ); + match head_object_for_worker(tgt_client, &tgt_client.bucket, object, Some(assigned_version_id.to_string())) + .await + { + Ok(head) => Ok(Some(LocatedReplica { + head, + version_id: Some(assigned_version_id.to_string()), + })), + // The located version disappeared between LIST and HEAD. + Err(e) if head_object_not_found(&e) => Ok(None), + Err(e) => Err(e), + } + } + None => Ok(None), } } - Ok(None) => Ok(None), Err(list_err) => Err(Box::new(SdkError::construction_failure(*list_err))), }) } +/// Content-identity candidates minus those other source versions of the key +/// already claim through their ledgers. The sibling listing is only paid for +/// when there is a candidate to disown. +async fn disowned_replica_location( + siblings: &dyn SiblingLedger, + arn: &str, + source_bucket: &str, + object: &str, + source_version_id: Option, + mut candidates: Vec, +) -> ReplicaLocation { + if candidates.is_empty() { + return ReplicaLocation::Missing; + } + let claimed = siblings + .claimed_target_versions(source_bucket, object, source_version_id, arn) + .await; + candidates.retain(|candidate| !claimed.contains(candidate)); + ReplicaLocation::from_matches(candidates) +} + +/// The id the target assigned to a located replica when it is not the source +/// id, i.e. the ledger entry this replica needs. `None` on an adopting target +/// (nothing to record) or when no version id is known. +fn drifted_replica_version_id(replica: &LocatedReplica, source_version_id: Option) -> Option<&str> { + let assigned = replica.version_id.as_deref().filter(|id| !id.is_empty())?; + let source_version_id = source_version_id.map(|version_id| version_id.to_string()); + (source_version_id.as_deref() != Some(assigned)).then_some(assigned) +} + +/// One failed metadata-only operation against a located replica, named by +/// the S3 operation so same-cause failures bucket together downstream. +#[derive(Debug)] +struct ReplicaMetadataSyncError { + operation: &'static str, + source: String, +} + +impl ReplicaMetadataSyncError { + fn io(operation: &'static str, source: impl Display) -> std::io::Error { + std::io::Error::other(Self { + operation, + source: source.to_string(), + }) + } +} + +impl Display for ReplicaMetadataSyncError { + fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result { + write!(f, "{} failed: {}", self.operation, self.source) + } +} + +impl std::error::Error for ReplicaMetadataSyncError {} + +fn user_defined_value<'a>(user_defined: &'a HashMap, name: &str) -> Option<&'a str> { + user_defined + .iter() + .find(|(key, _)| key.eq_ignore_ascii_case(name)) + .map(|(_, value)| value.as_str()) + .filter(|value| !value.is_empty()) +} + +/// Bring a located replica's tags and Object Lock state in line with the +/// source version through the metadata-only S3 operations, addressing the +/// version the target assigned. +/// +/// The regular metadata transport re-PUTs the object with the source version +/// id, which an id-adopting target applies in place. A target that mints its +/// own ids would answer that PUT with one more version (and keep the old +/// one), so the replica's tags never changed and the target grew a duplicate +/// per update (rustfs/backlog#2340). A `Metadata` attempt always applies the +/// source tags; retention and legal hold are applied when the HEAD shows a +/// difference (retention is only ever extended here — shortening or clearing +/// it needs a governance bypass the replication client does not hold, so the +/// replica keeps its longer window and the difference is logged). Heal and +/// resync attempts only touch what the HEAD proves diverged. +async fn sync_drifted_replica_metadata( + tgt_client: &TargetClient, + object: &str, + source: &ObjectInfo, + located: &LocatedReplica, + op_type: ReplicationType, +) -> std::io::Result { + let version_id = located.version_id.clone(); + let replica = &located.head; + let mut applied = false; + + let source_tags = ReplicationTagFilter::decode_tags_to_map(&source.user_tags); + let source_tag_count = i32::try_from(source_tags.len()).unwrap_or(i32::MAX); + if op_type == ReplicationType::Metadata || replica.tag_count.unwrap_or_default() != source_tag_count { + if source_tags.is_empty() { + if replica.tag_count.unwrap_or_default() > 0 { + tgt_client + .delete_object_tagging(&tgt_client.bucket, object, version_id.clone()) + .await + .map_err(|e| ReplicaMetadataSyncError::io("delete_object_tagging", e))?; + applied = true; + } + } else { + let mut tag_set: Vec<(String, String)> = source_tags.into_iter().collect(); + tag_set.sort(); + let tagging = Tagging::builder() + .set_tag_set(Some( + tag_set + .into_iter() + .map(|(key, value)| Tag::builder().key(key).value(value).build()) + .collect::, _>>() + .map_err(|e| ReplicaMetadataSyncError::io("build_tag", e))?, + )) + .build() + .map_err(|e| ReplicaMetadataSyncError::io("build_tag_set", e))?; + tgt_client + .put_object_tagging(&tgt_client.bucket, object, version_id.clone(), tagging) + .await + .map_err(|e| ReplicaMetadataSyncError::io("put_object_tagging", e))?; + applied = true; + } + } + + let source_mode = user_defined_value(&source.user_defined, AMZ_OBJECT_LOCK_MODE); + let source_retain_until = user_defined_value(&source.user_defined, AMZ_OBJECT_LOCK_RETAIN_UNTIL_DATE) + .and_then(|value| OffsetDateTime::parse(value, &Rfc3339).ok()); + match (source_mode, source_retain_until) { + (Some(mode), Some(retain_until)) => { + let replica_mode = replica.object_lock_mode.as_ref().map(|mode| mode.as_str()); + let replica_retain_until = replica.object_lock_retain_until_date.as_ref().map(|date| date.secs()); + let retention_differs = !replica_mode.is_some_and(|replica_mode| replica_mode.eq_ignore_ascii_case(mode)) + || replica_retain_until != Some(retain_until.unix_timestamp()); + if retention_differs { + if replica_retain_until.is_some_and(|current| current > retain_until.unix_timestamp()) { + debug!( + event = EVENT_REPLICATION_DRIFTED_REPLICA_METADATA_SYNCED, + component = LOG_COMPONENT_ECSTORE, + subsystem = LOG_SUBSYSTEM_REPLICATION_RESYNC, + bucket = %tgt_client.bucket, + object = %object, + arn = %tgt_client.arn, + reason = "retention_shortening_requires_bypass", + "Replica retention is longer than the source's; left unchanged" + ); + } else { + let mode = ObjectLockRetentionMode::from(mode.to_ascii_uppercase().as_str()); + tgt_client + .put_object_retention( + &tgt_client.bucket, + object, + version_id.clone(), + mode, + aws_sdk_s3::primitives::DateTime::from_secs(retain_until.unix_timestamp()), + ) + .await + .map_err(|e| ReplicaMetadataSyncError::io("put_object_retention", e))?; + applied = true; + } + } + } + _ => { + if replica.object_lock_mode.is_some() { + debug!( + event = EVENT_REPLICATION_DRIFTED_REPLICA_METADATA_SYNCED, + component = LOG_COMPONENT_ECSTORE, + subsystem = LOG_SUBSYSTEM_REPLICATION_RESYNC, + bucket = %tgt_client.bucket, + object = %object, + arn = %tgt_client.arn, + reason = "retention_clear_requires_bypass", + "Replica retention has no source counterpart; left unchanged" + ); + } + } + } + + let source_legal_hold = user_defined_value(&source.user_defined, AMZ_OBJECT_LOCK_LEGAL_HOLD) + .is_some_and(|value| value.eq_ignore_ascii_case("ON")); + let replica_legal_hold = replica + .object_lock_legal_hold_status + .as_ref() + .is_some_and(|status| status.as_str().eq_ignore_ascii_case("ON")); + if source_legal_hold != replica_legal_hold { + let status = if source_legal_hold { + ObjectLockLegalHoldStatus::On + } else { + ObjectLockLegalHoldStatus::Off + }; + tgt_client + .put_object_legal_hold(&tgt_client.bucket, object, version_id.clone(), status) + .await + .map_err(|e| ReplicaMetadataSyncError::io("put_object_legal_hold", e))?; + applied = true; + } + + if applied { + debug!( + event = EVENT_REPLICATION_DRIFTED_REPLICA_METADATA_SYNCED, + component = LOG_COMPONENT_ECSTORE, + subsystem = LOG_SUBSYSTEM_REPLICATION_RESYNC, + bucket = %tgt_client.bucket, + object = %object, + arn = %tgt_client.arn, + assigned_version_id = version_id.as_deref().unwrap_or(""), + "Synced replica metadata in place on a target that mints its own version ids" + ); + Ok(ReplicationAction::Metadata) + } else { + Ok(ReplicationAction::None) + } +} + /// Resolve the N2 fail-closed gate for an SSE-C passthrough attempt against /// this target. Returns `Some(audit_required)` when replication may proceed; /// on a freshly-flagged header-dropping target it settles `rinfo` as FAILED @@ -1472,7 +1831,20 @@ async fn verify_resync_head_result( // A version-addressed HEAD is not the last word on every target: // re-verify through the fallback before counting a well-replicated // object as failed (see `replica_head_fallback`). - match replica_head_fallback(target_client.as_ref(), &roi.name, roi.etag.as_deref(), &err).await { + match replica_head_fallback( + target_client.as_ref(), + ReplicaSource { + bucket: &roi.bucket, + object: &roi.name, + version_id: roi.version_id, + etag: roi.etag.as_deref(), + ledger_version_id: None, + }, + Some(&NoSiblingLedger), + &err, + ) + .await + { Some(Ok(Some(_))) => { st.replicated_count += 1; st.replicated_size += roi.size; @@ -1683,6 +2055,26 @@ pub async fn get_heal_replicate_object_info(oi: &ObjectInfo, rcfg: &ReplicationC let target_statuses = replication_statuses_map(&oi.replication_status_internal.clone().unwrap_or_default()); let target_purge_statuses = version_purge_statuses_map(&oi.version_purge_status_internal.clone().unwrap_or_default()); + // A version purge is owed to the targets its purge state names, whatever + // the configuration says now: the decision string is not persisted, so a + // heal after restart (or after the configuration was removed or edited) + // would otherwise never revisit the purge and the hidden version would + // block DeleteBucket forever (rustfs/backlog#2340). The delete worker + // settles a target the configuration no longer names as abandoned. + let dsc = if delete_path && !dsc.replicate_any() { + let owed: Vec = target_purge_statuses + .iter() + .filter(|(_, status)| matches!(status, VersionPurgeStatusType::Pending | VersionPurgeStatusType::Failed)) + .map(|(arn, _)| arn.clone()) + .collect(); + if owed.is_empty() { + dsc + } else { + replicate_decision_for_admitted_targets(&owed) + } + } else { + dsc + }; let existing_obj_resync = if delete_path && !has_stored_delete_decision && !delete_state.0 && !delete_state.1 { Default::default() } else { @@ -2038,6 +2430,13 @@ pub(crate) async fn replicate_delete_with_outcome( targets: Vec::with_capacity(dsc.targets_map.len()), }; + let purge_source = version_purge_source(&storage, &bucket, &dobj, &dsc).await.map(Arc::new); + let configured_arns = if is_version_delete_replication(&dobj.delete_object) { + configured_replication_arns(&bucket).await + } else { + None + }; + let mut join_set = JoinSet::new(); // Process each target @@ -2058,6 +2457,17 @@ pub(crate) async fn replicate_delete_with_outcome( continue; } + // The bucket no longer replicates to this target: nothing can ever + // confirm the purge remotely, so finish it locally as abandoned. + if let Some(configured) = configured_arns.as_ref() + && !configured.contains(&tgt_entry.arn) + { + rinfos + .targets + .push(abandoned_purge_target_info(&bucket, &dobj, &tgt_entry.arn)); + continue; + } + // Get the remote target client let Some(tgt_client) = ReplicationTargetStore::remote_target_client(&bucket, &tgt_entry.arn).await else { debug!( @@ -2074,9 +2484,10 @@ pub(crate) async fn replicate_delete_with_outcome( }; let dobj_clone = dobj.clone(); + let purge_source = purge_source.clone(); // Spawn task in the join set - join_set.spawn(async move { replicate_delete_to_target(&dobj_clone, tgt_client.clone()).await }); + join_set.spawn(async move { replicate_delete_to_target(&dobj_clone, tgt_client.clone(), purge_source).await }); } // Collect all results @@ -2827,7 +3238,181 @@ fn unavailable_delete_target_info(dobj: &DeletedObjectReplicationInfo, arn: &str rinfo } -async fn replicate_delete_to_target(dobj: &DeletedObjectReplicationInfo, tgt_client: Arc) -> ReplicatedTargetInfo { +/// The target ARNs the bucket's replication configuration still names, or +/// `None` when that cannot be decided right now (unreadable/invalid +/// configuration): a purge is only abandoned on positive evidence. No +/// configuration at all names no target. +async fn configured_replication_arns(bucket: &str) -> Option> { + match get_replication_config(bucket).await { + Ok(Some(config)) => Some(config.configured_target_arns()), + Ok(None) => Some(HashSet::new()), + Err(_) => None, + } +} + +/// Finish a version purge locally for a target the bucket no longer +/// replicates to (the rule or the whole configuration was removed). +/// +/// The source keeps a purged version in xl.meta, hidden from listings, until +/// every target confirms the purge — and once the operator removed the +/// target nothing ever will: the version stayed PENDING forever, blocking +/// `DeleteBucket` with a residue the client could neither see nor remove +/// (rustfs/backlog#2340). Reporting the purge as complete lets the normal +/// writeback drop the version. The replica, if any, stays on the former +/// target: that is the operator's data now, and this event is the record. +fn abandoned_purge_target_info(bucket: &str, dobj: &DeletedObjectReplicationInfo, arn: &str) -> ReplicatedTargetInfo { + let mut rinfo = dobj + .delete_object + .replication_state + .clone() + .unwrap_or_default() + .target_state(arn); + rinfo.op_type = dobj.op_type; + if rinfo.version_purge_status == VersionPurgeStatusType::Complete { + return rinfo; + } + warn!( + event = EVENT_REPLICATION_PURGE_ABANDONED, + component = LOG_COMPONENT_ECSTORE, + subsystem = LOG_SUBSYSTEM_REPLICATION_RESYNC, + bucket, + object = dobj.delete_object.object_name, + version_id = ?dobj.delete_object.version_id.or(dobj.delete_object.delete_marker_version_id), + arn, + reason = "target_not_configured", + "Replicated version purge abandoned: the bucket no longer replicates to this target, so the version is purged locally" + ); + counter!(METRIC_VERSION_PURGE_ABANDONED_TOTAL).increment(1); + rinfo.version_purge_status = VersionPurgeStatusType::Complete; + rinfo.error = None; + rinfo +} + +/// What the source still knows about a data version being purged, read once +/// per delete: the version stays in xl.meta with a PENDING purge status until +/// every target confirms, so its ETag and target-version ledger are available +/// to resolve the replica on a target that mints its own version ids. +struct VersionPurgeSource { + etag: Option, + ledger: HashMap, + ledger_corrupt: bool, + /// Target versions other generations of the key claim, per target ARN: + /// never a candidate for this version's purge. + claimed_by_siblings: HashMap>, +} + +async fn version_purge_source( + storage: &Arc, + bucket: &str, + dobj: &DeletedObjectReplicationInfo, + dsc: &ReplicateDecision, +) -> Option { + // Delete-marker purges resolve through the marker ledger recorded when + // the marker was created (`delete_replication_target_version_id`). + if dobj.delete_object.delete_marker || dobj.delete_object.delete_marker_version_id.is_some() { + return None; + } + let version_id = dobj.delete_object.version_id.filter(|version_id| !version_id.is_nil())?; + let object_name = &dobj.delete_object.object_name; + let info = storage + .get_object_info( + bucket, + object_name, + &ObjectOptions { + version_id: Some(version_id.to_string()), + versioned: ReplicationVersioningStore::prefix_enabled(bucket, object_name).await, + version_suspended: ReplicationVersioningStore::prefix_suspended(bucket, object_name).await, + ..Default::default() + }, + ) + .await + .ok()?; + let (ledger, ledger_corrupt) = replication_target_versions(&info.user_defined); + // Only a target without a ledger entry falls back to content identity, + // and only then are the sibling ledgers worth a listing. + let mut claimed_by_siblings = HashMap::new(); + for target in dsc.targets_map.values() { + if target.replicate && !ledger.contains_key(&target.arn) { + let claimed = storage + .claimed_target_versions(bucket, object_name, Some(version_id), &target.arn) + .await; + claimed_by_siblings.insert(target.arn.clone(), claimed); + } + } + Some(VersionPurgeSource { + etag: info.etag, + ledger, + ledger_corrupt, + claimed_by_siblings, + }) +} + +/// Which version a data-version purge addresses on `arn` once the source +/// ledger and, failing that, content identity have been consulted. +enum PurgeReplicaResolution { + /// Address the version the caller already derived (id-adopting target). + SourceVersion, + /// Address the version the target assigned. + Resolved(String), + /// The target holds no live version with the source content: the purge + /// goal is already met there. + Absent, + /// Refuse: no ledger entry and more than one target version carries the + /// source ETag, so any pick could destroy a live generation. + Ambiguous, + /// Refuse: the persisted ledger is inconsistent. + LedgerCorrupt, + /// The content-identity lookup itself failed (transport/service). + LookupFailed(String), +} + +/// `force_lookup` consults content identity even on a target whose identity +/// contract is unknown: used after a source-id DELETE answered +/// NoSuchVersion, where "already gone" and "never had that id" (Wasabi +/// after a restart, before any PUT re-taught the verdict) look the same. +async fn resolve_purge_replica( + tgt_client: &TargetClient, + object: &str, + source: &VersionPurgeSource, + force_lookup: bool, +) -> PurgeReplicaResolution { + if source.ledger_corrupt { + return PurgeReplicaResolution::LedgerCorrupt; + } + if let Some(assigned) = source.ledger.get(&tgt_client.arn) { + ReplicationTargetStore::record_version_identity_capability(&tgt_client.arn, VersionIdentityCapability::MintsOwn); + return PurgeReplicaResolution::Resolved(assigned.clone()); + } + if !force_lookup && !ReplicationTargetStore::version_identity_capability(&tgt_client.arn).version_addressing_unreliable() { + return PurgeReplicaResolution::SourceVersion; + } + let Some(etag) = source.etag.as_deref().filter(|etag| !etag.trim().is_empty()) else { + return PurgeReplicaResolution::SourceVersion; + }; + match tgt_client.replica_candidates_by_etag(&tgt_client.bucket, object, etag).await { + Ok(mut candidates) => { + if let Some(claimed) = source.claimed_by_siblings.get(&tgt_client.arn) { + candidates.retain(|candidate| !claimed.contains(candidate)); + } + match ReplicaLocation::from_matches(candidates) { + ReplicaLocation::Unique(assigned) => PurgeReplicaResolution::Resolved(assigned), + ReplicaLocation::Missing => PurgeReplicaResolution::Absent, + ReplicaLocation::Ambiguous { .. } => PurgeReplicaResolution::Ambiguous, + } + } + Err(error) => PurgeReplicaResolution::LookupFailed(error.to_string()), + } +} + +fn purge_target_already_clean(error: &S3ClientError) -> bool { + matches!(error.code.as_deref(), Some("NoSuchKey" | "NoSuchVersion")) +} + +async fn replicate_delete_to_target( + dobj: &DeletedObjectReplicationInfo, + tgt_client: Arc, + purge_source: Option>, +) -> ReplicatedTargetInfo { let mut rinfo = dobj .delete_object .replication_state @@ -2864,7 +3449,7 @@ async fn replicate_delete_to_target(dobj: &DeletedObjectReplicationInfo, tgt_cli // assigned (recorded when the marker was created there); see // `delete_replication_target_version_id`. A corrupt record is a failure, // not a guess: the entry stays visible until the metadata is repaired. - let Some(version_id) = delete_replication_target_version_id(&dobj.delete_object, &tgt_client.arn) else { + let Some(mut version_id) = delete_replication_target_version_id(&dobj.delete_object, &tgt_client.arn) else { warn!( event = EVENT_DELETE_MARKER_PURGE_FAILED, component = LOG_COMPONENT_ECSTORE, @@ -2880,6 +3465,77 @@ async fn replicate_delete_to_target(dobj: &DeletedObjectReplicationInfo, tgt_cli return rinfo; }; + // A data-version purge on a target that mints its own version ids must + // address the id the target assigned: the source uuid answers + // NoSuchVersion there forever (rustfs/backlog#2340). The ledger recorded + // at replication time is exact; content identity is the fallback for + // replicas written before the ledger existed, and it refuses to guess. + let mut addressed_source_version = true; + if let Some(source) = purge_source.as_deref() { + let object = &dobj.delete_object.object_name; + let refuse = |rinfo: &mut ReplicatedTargetInfo, reason: &str, detail: String, backoff: bool| { + if backoff { + record_purge_backoff(dobj, &tgt_client.arn); + } + warn!( + event = EVENT_REPLICATION_PURGE_REPLICA_UNRESOLVED, + component = LOG_COMPONENT_ECSTORE, + subsystem = LOG_SUBSYSTEM_REPLICATION_RESYNC, + bucket = tgt_client.bucket, + object = %object, + arn = %tgt_client.arn, + reason, + detail = %detail, + "Replicated version purge could not resolve the replica on a target that mints its own version ids" + ); + counter!(METRIC_VERSION_PURGE_REPLICA_TOTAL, "resolution" => reason.to_string()).increment(1); + rinfo.version_purge_status = VersionPurgeStatusType::Failed; + rinfo.error = Some(detail); + }; + match resolve_purge_replica(&tgt_client, object, source, false).await { + PurgeReplicaResolution::SourceVersion => {} + PurgeReplicaResolution::Resolved(assigned) => { + counter!(METRIC_VERSION_PURGE_REPLICA_TOTAL, "resolution" => "resolved").increment(1); + version_id = Some(assigned); + addressed_source_version = false; + } + PurgeReplicaResolution::Absent => { + debug!( + bucket = tgt_client.bucket, + object = %object, + arn = %tgt_client.arn, + "replicate_delete_to_target: no replica with the source content on the target; purge already satisfied" + ); + counter!(METRIC_VERSION_PURGE_REPLICA_TOTAL, "resolution" => "absent").increment(1); + rinfo.version_purge_status = VersionPurgeStatusType::Complete; + return rinfo; + } + PurgeReplicaResolution::Ambiguous => { + refuse( + &mut rinfo, + "ambiguous", + "replica identity is ambiguous on the target: several versions carry the source content and no target version is recorded".to_string(), + true, + ); + return rinfo; + } + PurgeReplicaResolution::LedgerCorrupt => { + refuse( + &mut rinfo, + "ledger_corrupt", + "recorded target version metadata is inconsistent".to_string(), + true, + ); + return rinfo; + } + PurgeReplicaResolution::LookupFailed(error) => { + refuse(&mut rinfo, "lookup_failed", format!("replica lookup failed: {error}"), false); + mark_replication_target_offline_if_needed(&tgt_client, &error).await; + return rinfo; + } + } + } + if dobj.delete_object.delete_marker && dobj.delete_object.delete_marker_version_id.is_some() { match head_object_for_worker( tgt_client.as_ref(), @@ -2908,22 +3564,66 @@ async fn replicate_delete_to_target(dobj: &DeletedObjectReplicationInfo, tgt_cli } } - match tgt_client + let remove_options = replication_delete_remove_options( + delete_replication_creates_marker(&dobj.delete_object), + dobj.delete_object.delete_marker_mtime, + ); + // A version purge must keep the versionId on the DELETE even when the + // purged version is a delete marker: marker-creation semantics would drop + // it and a generic S3 target would mint a fresh marker on every retry + // (rustfs#6823). + let mut removed = tgt_client .remove_object( &tgt_client.bucket, &dobj.delete_object.object_name, version_id.clone(), - // A version purge must keep the versionId on the DELETE even when - // the purged version is a delete marker: marker-creation semantics - // would drop it and a generic S3 target would mint a fresh marker - // on every retry (rustfs#6823). - replication_delete_remove_options( - delete_replication_creates_marker(&dobj.delete_object), - dobj.delete_object.delete_marker_mtime, - ), + remove_options.clone(), ) - .await + .await; + // NoSuchVersion for the SOURCE id on a target whose identity contract is + // not known: "already gone" and "never had that id" look the same, so + // resolve by content identity once before concluding either. + if is_version_purge + && addressed_source_version + && removed.as_ref().is_err_and(purge_target_already_clean) + && let Some(source) = purge_source.as_deref() { + let object = &dobj.delete_object.object_name; + match resolve_purge_replica(&tgt_client, object, source, true).await { + PurgeReplicaResolution::Resolved(assigned) => { + ReplicationTargetStore::record_version_identity_capability(&tgt_client.arn, VersionIdentityCapability::MintsOwn); + counter!(METRIC_VERSION_PURGE_REPLICA_TOTAL, "resolution" => "resolved").increment(1); + version_id = Some(assigned); + removed = tgt_client + .remove_object(&tgt_client.bucket, object, version_id.clone(), remove_options) + .await; + } + PurgeReplicaResolution::Ambiguous => { + record_purge_backoff(dobj, &tgt_client.arn); + warn!( + event = EVENT_REPLICATION_PURGE_REPLICA_UNRESOLVED, + component = LOG_COMPONENT_ECSTORE, + subsystem = LOG_SUBSYSTEM_REPLICATION_RESYNC, + bucket = tgt_client.bucket, + object = %object, + arn = %tgt_client.arn, + reason = "ambiguous", + "Replicated version purge could not resolve the replica on a target that mints its own version ids" + ); + counter!(METRIC_VERSION_PURGE_REPLICA_TOTAL, "resolution" => "ambiguous").increment(1); + rinfo.version_purge_status = VersionPurgeStatusType::Failed; + rinfo.error = Some( + "replica identity is ambiguous on the target: several versions carry the source content and no target version is recorded" + .to_string(), + ); + return rinfo; + } + // Missing, a lookup failure, a corrupt ledger or no content + // identity to compare: the NoSuchVersion answer stands. + _ => {} + } + } + match removed { Ok(assigned_version_id) => { debug!( bucket = tgt_client.bucket, @@ -2947,6 +3647,19 @@ async fn replicate_delete_to_target(dobj: &DeletedObjectReplicationInfo, tgt_cli rinfo.version_purge_status = VersionPurgeStatusType::Complete; } } + // The version is already gone on the target: the purge goal is met. + // Strict S3 targets answer NoSuchVersion here (RustFS/MinIO answer + // 204); failing would retry a delete that can never do more. + Err(e) if is_version_purge && purge_target_already_clean(&e) => { + debug!( + bucket = tgt_client.bucket, + object = dobj.delete_object.object_name, + version_id = ?version_id, + error = %e, + "replicate_delete_to_target: version already absent on the target" + ); + rinfo.version_purge_status = VersionPurgeStatusType::Complete; + } Err(e) => { let object_lock_denied = is_version_purge && is_object_lock_denied_delete(e.code.as_deref(), e.message.as_deref()); if object_lock_denied { @@ -2957,7 +3670,7 @@ async fn replicate_delete_to_target(dobj: &DeletedObjectReplicationInfo, tgt_cli // lock on the replica lapses. Surface it loudly instead of // letting a silent failed counter and a hot heal-retry loop // stand in for the divergence. - record_object_lock_denied_purge(dobj, &tgt_client.arn); + record_purge_backoff(dobj, &tgt_client.arn); error!( event = EVENT_REPLICATION_PURGE_OBJECT_LOCK_DENIED, component = LOG_COMPONENT_ECSTORE, @@ -3080,12 +3793,23 @@ fn replication_status_writeback_options( roi: &ReplicateObjectInfo, replication_lock_guard: &rustfs_lock::NamespaceLockGuard, new_replication_internal: Option<&String>, + target_version_ledger: &[(String, String)], mode: ReplicationStatusWritebackMode, ) -> ObjectOptions { let mut eval_metadata = HashMap::new(); if let Some(status) = new_replication_internal { insert_str(&mut eval_metadata, SUFFIX_REPLICATION_STATUS, status.clone()); } + // One durable key per drifting target: the version id it assigned to this + // object version, so a later purge, tag or lock update addresses that id + // instead of the source uuid the target never had. + for (arn, assigned_version_id) in target_version_ledger { + insert_str( + &mut eval_metadata, + &format!("{SUFFIX_REPLICATION_TARGET_VERSION_ARN_PREFIX}{arn}"), + assigned_version_id.clone(), + ); + } let mut write_opts = ObjectOptions { version_id: roi.version_id.map(|version_id| version_id.to_string()), eval_metadata: Some(eval_metadata), @@ -3108,10 +3832,12 @@ async fn persist_replication_state_if_current( storage: &Arc, replication_lock_guard: &rustfs_lock::NamespaceLockGuard, new_replication_internal: Option<&String>, + target_version_ledger: &[(String, String)], mode: ReplicationStatusWritebackMode, object_info: &mut ObjectInfo, ) -> Result { - let write_opts = replication_status_writeback_options(roi, replication_lock_guard, new_replication_internal, mode); + let write_opts = + replication_status_writeback_options(roi, replication_lock_guard, new_replication_internal, target_version_ledger, mode); match storage.put_object_metadata(&roi.bucket, &roi.name, &write_opts).await { Ok(updated) => { *object_info = updated; @@ -3264,7 +3990,15 @@ pub(crate) async fn replicate_object_with_outcome( let mut object_info = roi.to_object_info(); let mut disposition = ReplicationAttemptDisposition::Persisted; let mut suppress_terminal_publication = false; - let state_update_needed = roi.replication_status_internal != new_replication_internal || rinfos.replication_resynced(); + let target_version_ledger: Vec<(String, String)> = rinfos + .targets + .iter() + .filter(|target| !target.is_empty()) + .filter_map(|target| Some((target.arn.clone(), target.target_version_id.clone()?))) + .collect(); + let state_update_needed = roi.replication_status_internal != new_replication_internal + || rinfos.replication_resynced() + || !target_version_ledger.is_empty(); let writeback_mode = replication_status_writeback_mode(state_update_needed); match persist_replication_state_if_current( @@ -3272,6 +4006,7 @@ pub(crate) async fn replicate_object_with_outcome( &storage, &obj_lock_guard, new_replication_internal.as_ref(), + &target_version_ledger, writeback_mode, &mut object_info, ) @@ -3472,8 +4207,7 @@ trait ReplicateObjectInfoExt { storage: Arc, tgt_client: Arc, ) -> ReplicatedTargetInfo; - async fn replicate_all(&self, storage: Arc, tgt_client: Arc) - -> ReplicatedTargetInfo; + async fn replicate_all(&self, storage: Arc, tgt_client: Arc) -> ReplicatedTargetInfo; fn to_object_info(&self) -> ObjectInfo; } @@ -3660,14 +4394,40 @@ impl ReplicateObjectInfoExt for ReplicateObjectInfo { } } Err(e) => { - if let Some(fallback) = replica_head_fallback(&tgt_client, &object, object_info.etag.as_deref(), &e).await { + let ledger_version_id = ledger_target_version_id(&object_info.user_defined, &tgt_client.arn); + // A fresh write has no replica to find by content: an older + // generation with the same bytes would be mistaken for one. + // Only the ledger (a lost response of this very version) and + // the AWS-style format fallback apply here. + if let Some(fallback) = replica_head_fallback( + &tgt_client, + ReplicaSource { + bucket: &bucket, + object: &object, + version_id: self.version_id, + etag: object_info.etag.as_deref(), + ledger_version_id: ledger_version_id.as_deref(), + }, + None, + &e, + ) + .await + { match fallback { - Ok(Some(oi)) if replication_etags_match(object_info.etag.as_deref(), oi.e_tag.as_deref()) => { + Ok(Some(located)) + if replication_etags_match(object_info.etag.as_deref(), located.head.e_tag.as_deref()) => + { if ssec_audit_required - && !settle_ssec_passthrough_evidence(&oi, &tgt_client, &bucket, &object, &mut rinfo).await + && !settle_ssec_passthrough_evidence(&located.head, &tgt_client, &bucket, &object, &mut rinfo) + .await { return rinfo; } + if let Some(assigned) = drifted_replica_version_id(&located, self.version_id) + && ledger_version_id.as_deref() != Some(assigned) + { + rinfo.target_version_id = Some(assigned.to_string()); + } rinfo.replication_status = ReplicationStatusType::Completed; rinfo.replication_resynced = true; rinfo.replication_action = ReplicationAction::None; @@ -3774,9 +4534,10 @@ impl ReplicateObjectInfoExt for ReplicateObjectInfo { return rinfo; } - if let Some(err) = if is_multipart { + let source_version_id = self.version_id; + let assigned_version_id = if is_multipart { drop(gr); - let result = replicate_object_with_multipart(MultipartReplicationContext { + replicate_object_with_multipart(MultipartReplicationContext { storage: storage.clone(), cli: tgt_client.clone(), src_bucket: &bucket, @@ -3787,12 +4548,11 @@ impl ReplicateObjectInfoExt for ReplicateObjectInfo { arn: &rinfo.arn, put_opts, }) - .await; - result.err() + .await } else { gr.stream = wrap_with_bandwidth_monitor(gr.stream, &put_opts, &bucket, &rinfo.arn); let byte_stream = async_read_to_bytestream(gr.stream); - let result = tgt_client + tgt_client .put_object(&tgt_client.bucket, &object, transfer_size, byte_stream, &put_opts) .await .map_err(|e| std::io::Error::other(e.to_string())) @@ -3803,27 +4563,32 @@ impl ReplicateObjectInfoExt for ReplicateObjectInfo { response.version_id.as_deref(), ); verify_single_part_replica(&object_info, &response, obj_opts.raw_data_movement_read) - }); - result.err() - } { - rinfo.replication_status = ReplicationStatusType::Failed; - rinfo.error = Some(err.to_string()); - warn!( - event = EVENT_RESYNC_TARGET_OPERATION_FAILED, - component = LOG_COMPONENT_ECSTORE, - subsystem = LOG_SUBSYSTEM_REPLICATION_RESYNC, - bucket = %bucket, - target_bucket = %tgt_client.bucket, - arn = %tgt_client.arn, - object = %object, - operation = "put_object", - error = ?err, - "Replication target operation failed" - ); + .map(|()| response.version_id) + }) + }; + let assigned_version_id = match assigned_version_id { + Ok(assigned_version_id) => assigned_version_id, + Err(err) => { + rinfo.replication_status = ReplicationStatusType::Failed; + rinfo.error = Some(err.to_string()); + warn!( + event = EVENT_RESYNC_TARGET_OPERATION_FAILED, + component = LOG_COMPONENT_ECSTORE, + subsystem = LOG_SUBSYSTEM_REPLICATION_RESYNC, + bucket = %bucket, + target_bucket = %tgt_client.bucket, + arn = %tgt_client.arn, + object = %object, + operation = "put_object", + error = ?err, + "Replication target operation failed" + ); - mark_replication_target_offline_if_needed(&tgt_client, &err).await; - return rinfo; - } + mark_replication_target_offline_if_needed(&tgt_client, &err).await; + return rinfo; + } + }; + rinfo.target_version_id = assigned_target_version_id(assigned_version_id, source_version_id); // First SSE-C passthrough PUT against this target: verify the replica // kept its decryption material before reporting COMPLETED. @@ -3839,11 +4604,7 @@ impl ReplicateObjectInfoExt for ReplicateObjectInfo { rinfo } - async fn replicate_all( - &self, - storage: Arc, - tgt_client: Arc, - ) -> ReplicatedTargetInfo { + async fn replicate_all(&self, storage: Arc, tgt_client: Arc) -> ReplicatedTargetInfo { let start_time = OffsetDateTime::now_utc(); let bucket = self.bucket.clone(); @@ -3911,6 +4672,7 @@ impl ReplicateObjectInfoExt for ReplicateObjectInfo { let Some((replication_action, object_info)) = resolve_replicate_all_action( ReplicateAllActionContext { + siblings: &storage, roi: self, tgt_client: &tgt_client, bucket: &bucket, @@ -3955,7 +4717,7 @@ impl ReplicateObjectInfoExt for ReplicateObjectInfo { } }; - if let Some(err) = replicate_all_payload_to_target( + match replicate_all_payload_to_target( ReplicateAllPayloadContext { storage: &storage, tgt_client: &tgt_client, @@ -3972,8 +4734,13 @@ impl ReplicateObjectInfoExt for ReplicateObjectInfo { ) .await { - fail_replicate_all_put_object(&mut rinfo, &tgt_client, &bucket, &object, &err, start_time).await; - return rinfo; + Ok(assigned_version_id) => { + rinfo.target_version_id = assigned_target_version_id(assigned_version_id, self.version_id); + } + Err(err) => { + fail_replicate_all_put_object(&mut rinfo, &tgt_client, &bucket, &object, &err, start_time).await; + return rinfo; + } } // First SSE-C passthrough PUT against this target: verify the replica @@ -4206,6 +4973,9 @@ fn apply_replication_resync_timestamp(rinfo: &mut ReplicatedTargetInfo, reset_id /// Borrowed inputs for [`resolve_replicate_all_action`]. struct ReplicateAllActionContext<'a> { + /// Sibling ledgers of the source key, to disown content-identity + /// candidates other generations already claim. + siblings: &'a dyn SiblingLedger, roi: &'a ReplicateObjectInfo, tgt_client: &'a Arc, bucket: &'a str, @@ -4227,6 +4997,7 @@ async fn resolve_replicate_all_action( rinfo: &mut ReplicatedTargetInfo, ) -> Option<(ReplicationAction, ObjectInfo)> { let ReplicateAllActionContext { + siblings, roi, tgt_client, bucket, @@ -4292,26 +5063,76 @@ async fn resolve_replicate_all_action( } } Err(e) => { - if let Some(fallback) = replica_head_fallback(tgt_client, object, object_info.etag.as_deref(), &e).await { + let ledger_version_id = ledger_target_version_id(&object_info.user_defined, &tgt_client.arn); + if let Some(fallback) = replica_head_fallback( + tgt_client, + ReplicaSource { + bucket, + object, + version_id: roi.version_id, + etag: object_info.etag.as_deref(), + ledger_version_id: ledger_version_id.as_deref(), + }, + Some(siblings), + &e, + ) + .await + { match fallback { - Ok(Some(oi)) => { + Ok(Some(located)) => { + let oi = &located.head; let etags_match = replication_etags_match(object_info.etag.as_deref(), oi.e_tag.as_deref()); if require_existing_target && !etags_match { rinfo.error = Some("replica metadata target does not contain matching object data".to_string()); rinfo.duration = (OffsetDateTime::now_utc() - start_time).unsigned_abs(); return None; } - replication_action = if etags_match { - if ssec_audit_required - && !settle_ssec_passthrough_evidence(&oi, tgt_client, bucket, object, rinfo).await + if etags_match + && ssec_audit_required + && !settle_ssec_passthrough_evidence(oi, tgt_client, bucket, object, rinfo).await + { + rinfo.duration = (OffsetDateTime::now_utc() - start_time).unsigned_abs(); + return None; + } + if etags_match && let Some(assigned) = drifted_replica_version_id(&located, roi.version_id) { + // The replica lives under an id the target minted: + // record it (the ledger every later mutation + // resolves through) and apply metadata in place — + // the PUT transport below would mint another + // version instead of updating this one. + if ledger_version_id.as_deref() != Some(assigned) { + rinfo.target_version_id = Some(assigned.to_string()); + } + if let Err(err) = + sync_drifted_replica_metadata(tgt_client, object, &object_info, &located, roi.op_type).await { + rinfo.replication_status = ReplicationStatusType::Failed; + rinfo.error = Some(err.to_string()); + warn!( + event = EVENT_RESYNC_TARGET_OPERATION_FAILED, + component = LOG_COMPONENT_ECSTORE, + subsystem = LOG_SUBSYSTEM_REPLICATION_RESYNC, + bucket = %bucket, + object = %object, + arn = %tgt_client.arn, + operation = "sync_replica_metadata", + error = %err, + "Replication target operation failed" + ); + mark_replication_target_offline_if_needed(tgt_client, &err).await; rinfo.duration = (OffsetDateTime::now_utc() - start_time).unsigned_abs(); return None; } - ReplicationAction::None + // Converged in place: the caller's `None` branch + // records the sync without running the PUT. + replication_action = ReplicationAction::None; } else { - ReplicationAction::All - }; + replication_action = if etags_match { + ReplicationAction::None + } else { + ReplicationAction::All + }; + } } Ok(None) => { if require_existing_target { @@ -4435,24 +5256,35 @@ struct ReplicateAllPayloadContext<'a, S: ReplicationObjectIO> { put_opts: PutObjectOptions, } +/// The ledger entry a successful write leaves behind: the id the target +/// answered with when it is not the source id. Adopting targets record +/// nothing, so the ledger only ever grows on drifting targets. +fn assigned_target_version_id(assigned_version_id: Option, source_version_id: Option) -> Option { + let source_version_id = source_version_id.map(|version_id| version_id.to_string()); + assigned_version_id + .filter(|assigned| !assigned.is_empty()) + .filter(|assigned| source_version_id.as_deref() != Some(assigned.as_str())) +} + /// Ship the object payload to the replication target over the multipart or -/// single-put transport, returning the transport error when the upload fails. +/// single-put transport. Returns the version id the target assigned (when it +/// reported one), or the transport error when the upload fails. async fn replicate_all_payload_to_target( ctx: ReplicateAllPayloadContext<'_, S>, mut gr: GetObjectReader, -) -> Option { +) -> std::io::Result> { // Fail before streaming a body the target is required to reject: an S3 // PutObject caps at 5 GiB, and this route is chosen by the source object's // storage shape rather than its size (rustfs#6825). if let Some(reason) = replication_single_put_size_error(ctx.is_multipart, ctx.transfer_size, ctx.object_info.etag.as_deref()) { drop(gr); - return Some(std::io::Error::other(reason)); + return Err(std::io::Error::other(reason)); } if ctx.is_multipart { drop(gr); - let result = replicate_object_with_multipart(MultipartReplicationContext { + replicate_object_with_multipart(MultipartReplicationContext { storage: ctx.storage.clone(), cli: ctx.tgt_client.clone(), src_bucket: ctx.bucket, @@ -4463,13 +5295,11 @@ async fn replicate_all_payload_to_target( arn: ctx.arn, put_opts: ctx.put_opts, }) - .await; - result.err() + .await } else { gr.stream = wrap_with_bandwidth_monitor(gr.stream, &ctx.put_opts, ctx.bucket, ctx.arn); let byte_stream = async_read_to_bytestream(gr.stream); - let result = ctx - .tgt_client + ctx.tgt_client .put_object(&ctx.tgt_client.bucket, ctx.object, ctx.transfer_size, byte_stream, &ctx.put_opts) .await .map_err(|e| std::io::Error::other(e.to_string())) @@ -4480,8 +5310,8 @@ async fn replicate_all_payload_to_target( response.version_id.as_deref(), ); verify_single_part_replica(ctx.object_info, &response, ctx.obj_opts.raw_data_movement_read) - }); - result.err() + .map(|()| response.version_id) + }) } } @@ -4537,7 +5367,9 @@ struct MultipartReplicationContext<'a, S: ReplicationObjectIO> { put_opts: PutObjectOptions, } -async fn replicate_object_with_multipart(ctx: MultipartReplicationContext<'_, S>) -> std::io::Result<()> { +async fn replicate_object_with_multipart( + ctx: MultipartReplicationContext<'_, S>, +) -> std::io::Result> { let mut attempts = 1; let upload_id = loop { match ctx @@ -4680,15 +5512,15 @@ fn target_upload_already_removed(err: &S3ClientError) -> bool { /// invisible incomplete upload on the target that keeps billing for its parts. /// The abort outcome never replaces the transfer error: an abort failure is /// only logged and `result` is returned as-is. -async fn abort_multipart_on_failure( - result: std::io::Result<()>, +async fn abort_multipart_on_failure( + result: std::io::Result, dst_bucket: &str, object: &str, upload_id: &str, arn: &str, abort: F, schedule_abort_retry: R, -) -> std::io::Result<()> +) -> std::io::Result where F: FnOnce() -> Fut, Fut: std::future::Future>, @@ -4772,7 +5604,7 @@ fn multipart_replication_read_plan( async fn replicate_multipart_parts_and_complete( ctx: MultipartReplicationContext<'_, S>, upload_id: &str, -) -> std::io::Result<()> { +) -> std::io::Result> { let MultipartReplicationContext { storage, cli, @@ -4874,7 +5706,7 @@ async fn replicate_multipart_parts_and_complete( // a version that never existed. audit_target_version_identity(&cli, &put_opts.internal.source_version_id, completed.version_id()); - Ok(()) + Ok(completed.version_id().map(str::to_string)) } #[cfg(test)] @@ -5129,6 +5961,7 @@ mod tests { &ReplicateObjectInfo::default(), &guard, None, + &[], ReplicationStatusWritebackMode::ValidateOnly, ); forced_lost.store(true, std::sync::atomic::Ordering::Release); @@ -5390,6 +6223,7 @@ mod tests { let action = resolve_replicate_all_action( ReplicateAllActionContext { + siblings: &NoSiblingLedger, roi: &roi, tgt_client: &target, bucket: &roi.bucket, @@ -5434,6 +6268,7 @@ mod tests { let action = resolve_replicate_all_action( ReplicateAllActionContext { + siblings: &NoSiblingLedger, roi: &roi, tgt_client: &target, bucket: &roi.bucket, @@ -5536,6 +6371,7 @@ mod tests { let action = resolve_replicate_all_action( ReplicateAllActionContext { + siblings: &NoSiblingLedger, roi: &roi, tgt_client: &target, bucket: &roi.bucket, @@ -5553,6 +6389,11 @@ mod tests { "a replica located by content identity must not be re-driven: {action:?}" ); assert!(rinfo.error.is_none(), "{:?}", rinfo.error); + assert_eq!( + rinfo.target_version_id.as_deref(), + Some(DRIFTED_ASSIGNED_VERSION_ID), + "a replica located by content identity must be recorded in the target-version ledger" + ); let seen = server.join().expect("test HTTP server should finish"); assert_eq!(seen.len(), 3, "HEAD by source id, ListObjectVersions, HEAD by assigned id: {seen:?}"); assert!(seen[0].starts_with("HEAD ") && seen[0].contains(&roi.version_id.unwrap().to_string())); @@ -5572,6 +6413,7 @@ mod tests { let action = resolve_replicate_all_action( ReplicateAllActionContext { + siblings: &NoSiblingLedger, roi: &roi, tgt_client: &target, bucket: &roi.bucket, @@ -5605,6 +6447,500 @@ mod tests { server.join().expect("test HTTP server should finish"); } + /// Serves exactly `requests` connections, answering each from + /// `respond(request_line)`, and returns the request lines it saw. Reads + /// the whole request (headers plus `Content-Length` body, honoring + /// `Expect: 100-continue`) so a PUT with a body is not cut off. + fn spawn_scripted_target_server( + requests: usize, + respond: impl Fn(&str) -> String + Send + 'static, + ) -> (String, std::thread::JoinHandle>) { + use std::io::{Read, Write}; + + let listener = std::net::TcpListener::bind(("127.0.0.1", 0)).expect("test HTTP listener should bind"); + let endpoint = format!("http://{}", listener.local_addr().expect("test HTTP listener should have an address")); + let handle = std::thread::spawn(move || { + let mut seen = Vec::new(); + for _ in 0..requests { + let (mut stream, _) = listener.accept().expect("test HTTP client should connect"); + let mut raw = Vec::new(); + let mut chunk = [0_u8; 8192]; + let header_end = loop { + let bytes_read = stream.read(&mut chunk).expect("test HTTP request should be read"); + if bytes_read == 0 { + break raw.len(); + } + raw.extend_from_slice(&chunk[..bytes_read]); + if let Some(position) = raw.windows(4).position(|window| window == b"\r\n\r\n") { + break position + 4; + } + }; + let head = String::from_utf8_lossy(&raw[..header_end]).to_string(); + let content_length = head + .lines() + .find_map(|line| { + line.split_once(':') + .filter(|(name, _)| name.eq_ignore_ascii_case("content-length")) + }) + .and_then(|(_, value)| value.trim().parse::().ok()) + .unwrap_or(0); + if head.lines().any(|line| line.to_ascii_lowercase().starts_with("expect:")) && content_length > 0 { + stream + .write_all(b"HTTP/1.1 100 Continue\r\n\r\n") + .expect("test HTTP continue should be written"); + } + while raw.len() < header_end + content_length { + let bytes_read = stream.read(&mut chunk).expect("test HTTP body should be read"); + if bytes_read == 0 { + break; + } + raw.extend_from_slice(&chunk[..bytes_read]); + } + let request_line = head.lines().next().unwrap_or_default().to_string(); + stream + .write_all(respond(&request_line).as_bytes()) + .expect("test HTTP response should be written"); + seen.push(request_line); + } + seen + }); + (endpoint, handle) + } + + fn empty_response(status: &str) -> String { + format!("HTTP/1.1 {status}\r\nContent-Length: 0\r\nConnection: close\r\n\r\n") + } + + fn xml_response(status: &str, body: String) -> String { + format!( + "HTTP/1.1 {status}\r\nContent-Type: application/xml\r\nContent-Length: {}\r\nConnection: close\r\n\r\n{body}", + body.len() + ) + } + + fn list_versions_response(version_ids: &[&str]) -> String { + let versions = version_ids + .iter() + .enumerate() + .map(|(index, version_id)| { + format!( + "object{version_id}{}2026-09-06T10:00:0{index}.000Z"{DRIFTED_ETAG}"4STANDARD", + index == 0 + ) + }) + .collect::(); + xml_response( + "200 OK", + format!( + "target-bucketobject1000false{versions}" + ), + ) + } + + fn version_purge_dobj(arn: &str) -> DeletedObjectReplicationInfo { + let mut state = ReplicationState::default(); + state.purge_targets.insert(arn.to_string(), VersionPurgeStatusType::Pending); + DeletedObjectReplicationInfo { + bucket: "source".to_string(), + target_arn: arn.to_string(), + delete_object: ReplicationDeletedObject { + object_name: "object".to_string(), + version_id: Some(Uuid::new_v4()), + replication_state: Some(state), + ..Default::default() + }, + ..Default::default() + } + } + + fn purge_source(ledger: &[(&str, &str)]) -> Arc { + Arc::new(VersionPurgeSource { + etag: Some(DRIFTED_ETAG.to_string()), + ledger: ledger + .iter() + .map(|(arn, version_id)| (arn.to_string(), version_id.to_string())) + .collect(), + ledger_corrupt: false, + claimed_by_siblings: HashMap::new(), + }) + } + + #[tokio::test] + async fn version_purge_addresses_the_ledger_version_on_a_target_that_mints_own_ids() { + let (endpoint, server) = spawn_scripted_target_server(1, |line| { + if line.starts_with("DELETE ") && line.contains(&format!("versionId={DRIFTED_ASSIGNED_VERSION_ID}")) { + empty_response("204 No Content") + } else { + empty_response("500 Unexpected") + } + }); + let target = test_target_client(endpoint); + register_test_target(&target).await; + let dobj = version_purge_dobj(&target.arn); + + let rinfo = + replicate_delete_to_target(&dobj, target.clone(), Some(purge_source(&[(&target.arn, DRIFTED_ASSIGNED_VERSION_ID)]))) + .await; + + assert_eq!(rinfo.version_purge_status, VersionPurgeStatusType::Complete, "{:?}", rinfo.error); + let seen = server.join().expect("test HTTP server should finish"); + assert_eq!(seen.len(), 1, "the ledger id is addressed directly, without a lookup: {seen:?}"); + assert!( + !seen[0].contains(&dobj.delete_object.version_id.unwrap().to_string()), + "the source uuid must not be sent to a target that never had it: {}", + seen[0] + ); + assert!( + ReplicationTargetStore::version_identity_capability(&target.arn).version_addressing_unreliable(), + "a ledger entry proves the target mints its own ids" + ); + } + + #[tokio::test] + async fn version_purge_without_ledger_locates_the_unique_replica_by_etag() { + let (endpoint, server) = spawn_scripted_target_server(2, |line| { + if line.starts_with("GET ") && line.contains("versions") { + list_versions_response(&[DRIFTED_ASSIGNED_VERSION_ID]) + } else if line.starts_with("DELETE ") && line.contains(&format!("versionId={DRIFTED_ASSIGNED_VERSION_ID}")) { + empty_response("204 No Content") + } else { + empty_response("500 Unexpected") + } + }); + let target = test_target_client(endpoint); + register_test_target(&target).await; + ReplicationTargetStore::record_version_identity_capability(&target.arn, VersionIdentityCapability::MintsOwn); + let dobj = version_purge_dobj(&target.arn); + + let rinfo = replicate_delete_to_target(&dobj, target.clone(), Some(purge_source(&[]))).await; + + assert_eq!(rinfo.version_purge_status, VersionPurgeStatusType::Complete, "{:?}", rinfo.error); + let seen = server.join().expect("test HTTP server should finish"); + assert_eq!(seen.len(), 2, "ListObjectVersions, then DELETE by the located id: {seen:?}"); + assert!(seen[0].starts_with("GET ") && seen[0].contains("prefix=object"), "{}", seen[0]); + assert!( + seen[1].starts_with("DELETE ") && seen[1].contains(DRIFTED_ASSIGNED_VERSION_ID), + "{}", + seen[1] + ); + } + + #[tokio::test] + async fn version_purge_refuses_an_ambiguous_replica_identity_and_backs_off() { + let (endpoint, server) = spawn_scripted_target_server(1, |line| { + if line.starts_with("GET ") && line.contains("versions") { + list_versions_response(&[DRIFTED_ASSIGNED_VERSION_ID, "001788697733811332140-older-generation"]) + } else { + empty_response("500 Unexpected") + } + }); + let target = test_target_client(endpoint); + register_test_target(&target).await; + ReplicationTargetStore::record_version_identity_capability(&target.arn, VersionIdentityCapability::MintsOwn); + let dobj = version_purge_dobj(&target.arn); + assert!(!purge_backoff_active(&dobj)); + + let rinfo = replicate_delete_to_target(&dobj, target.clone(), Some(purge_source(&[]))).await; + + assert_eq!(rinfo.version_purge_status, VersionPurgeStatusType::Failed); + assert!( + rinfo.error.as_deref().is_some_and(|error| error.contains("ambiguous")), + "the refusal must name the deciding evidence: {:?}", + rinfo.error + ); + let seen = server.join().expect("test HTTP server should finish"); + assert_eq!(seen.len(), 1, "no DELETE may be sent when the replica cannot be told apart: {seen:?}"); + assert!(purge_backoff_active(&dobj), "an unresolvable purge must not be requeued every heal cycle"); + } + + #[tokio::test] + async fn version_purge_treats_a_missing_replica_as_already_purged() { + let (endpoint, server) = spawn_scripted_target_server(1, |line| { + if line.starts_with("GET ") && line.contains("versions") { + list_versions_response(&[]) + } else { + empty_response("500 Unexpected") + } + }); + let target = test_target_client(endpoint); + register_test_target(&target).await; + ReplicationTargetStore::record_version_identity_capability(&target.arn, VersionIdentityCapability::MintsOwn); + let dobj = version_purge_dobj(&target.arn); + + let rinfo = replicate_delete_to_target(&dobj, target.clone(), Some(purge_source(&[]))).await; + + assert_eq!(rinfo.version_purge_status, VersionPurgeStatusType::Complete, "{:?}", rinfo.error); + assert_eq!(server.join().expect("test HTTP server should finish").len(), 1); + } + + #[tokio::test] + async fn version_purge_treats_no_such_version_as_purged() { + let (endpoint, server) = spawn_scripted_target_server(1, |line| { + if line.starts_with("DELETE ") { + xml_response( + "404 Not Found", + "NoSuchVersionThe specified version does not exist.".to_string(), + ) + } else { + empty_response("500 Unexpected") + } + }); + let target = test_target_client(endpoint); + register_test_target(&target).await; + let dobj = version_purge_dobj(&target.arn); + + let rinfo = + replicate_delete_to_target(&dobj, target.clone(), Some(purge_source(&[(&target.arn, DRIFTED_ASSIGNED_VERSION_ID)]))) + .await; + + assert_eq!(rinfo.version_purge_status, VersionPurgeStatusType::Complete, "{:?}", rinfo.error); + assert!(!purge_backoff_active(&dobj)); + server.join().expect("test HTTP server should finish"); + } + + #[tokio::test] + async fn version_purge_re_resolves_by_etag_when_the_source_id_answers_no_such_version() { + // Identity contract unknown (fresh process), no ledger: the first + // DELETE goes out by the source uuid. NoSuchVersion there must not + // be read as "already gone" while a replica with the source content + // still exists under a minted id. + let no_such_version = xml_response( + "404 Not Found", + "NoSuchVersionThe specified version does not exist.".to_string(), + ); + let (endpoint, server) = spawn_scripted_target_server(3, move |line| { + if line.starts_with("DELETE ") && line.contains(&format!("versionId={DRIFTED_ASSIGNED_VERSION_ID}")) { + empty_response("204 No Content") + } else if line.starts_with("DELETE ") { + no_such_version.clone() + } else if line.starts_with("GET ") && line.contains("versions") { + list_versions_response(&[DRIFTED_ASSIGNED_VERSION_ID]) + } else { + empty_response("500 Unexpected") + } + }); + let target = test_target_client(endpoint); + register_test_target(&target).await; + let dobj = version_purge_dobj(&target.arn); + + let rinfo = replicate_delete_to_target(&dobj, target.clone(), Some(purge_source(&[]))).await; + + assert_eq!(rinfo.version_purge_status, VersionPurgeStatusType::Complete, "{:?}", rinfo.error); + let seen = server.join().expect("test HTTP server should finish"); + assert_eq!( + seen.len(), + 3, + "DELETE by source id, ListObjectVersions, DELETE by the located id: {seen:?}" + ); + assert!(seen[0].contains(&dobj.delete_object.version_id.unwrap().to_string()), "{}", seen[0]); + assert!(seen[1].starts_with("GET "), "{}", seen[1]); + assert!(seen[2].contains(DRIFTED_ASSIGNED_VERSION_ID), "{}", seen[2]); + assert!( + ReplicationTargetStore::version_identity_capability(&target.arn).version_addressing_unreliable(), + "the located replica proves the target mints its own ids" + ); + } + + /// Two same-bytes generations, the older one's replica already claimed + /// by its ledger: a purge of the newer (pre-ledger) version must not + /// touch it — with the sibling's claim removed the target holds no + /// candidate, so the purge is already satisfied there. + #[tokio::test] + async fn version_purge_disowns_replicas_claimed_by_sibling_ledgers() { + let (endpoint, server) = spawn_scripted_target_server(1, |line| { + if line.starts_with("GET ") && line.contains("versions") { + list_versions_response(&[DRIFTED_ASSIGNED_VERSION_ID]) + } else { + empty_response("500 Unexpected") + } + }); + let target = test_target_client(endpoint); + register_test_target(&target).await; + ReplicationTargetStore::record_version_identity_capability(&target.arn, VersionIdentityCapability::MintsOwn); + let dobj = version_purge_dobj(&target.arn); + let source = Arc::new(VersionPurgeSource { + etag: Some(DRIFTED_ETAG.to_string()), + ledger: HashMap::new(), + ledger_corrupt: false, + claimed_by_siblings: HashMap::from([(target.arn.clone(), HashSet::from([DRIFTED_ASSIGNED_VERSION_ID.to_string()]))]), + }); + + let rinfo = replicate_delete_to_target(&dobj, target.clone(), Some(source)).await; + + assert_eq!(rinfo.version_purge_status, VersionPurgeStatusType::Complete, "{:?}", rinfo.error); + let seen = server.join().expect("test HTTP server should finish"); + assert_eq!(seen.len(), 1, "a sibling's replica must never be deleted: {seen:?}"); + } + + struct ClaimedBySibling(&'static str); + + #[async_trait::async_trait] + impl SiblingLedger for ClaimedBySibling { + async fn claimed_target_versions(&self, _: &str, _: &str, _: Option, _: &str) -> HashSet { + HashSet::from([self.0.to_string()]) + } + } + + /// Heal of a same-bytes newer generation whose PUT never landed: the + /// only content match is the older generation's replica (claimed by its + /// ledger), so the object must be replicated, not declared converged. + #[tokio::test] + async fn heal_replicates_a_generation_whose_only_content_match_belongs_to_a_sibling() { + let (endpoint, server) = spawn_scripted_target_server(2, |line| { + if line.starts_with("HEAD ") { + empty_response("404 Not Found") + } else if line.starts_with("GET ") && line.contains("versions") { + list_versions_response(&[DRIFTED_ASSIGNED_VERSION_ID]) + } else { + empty_response("500 Unexpected") + } + }); + let target = test_target_client(endpoint); + ReplicationTargetStore::record_version_identity_capability(&target.arn, VersionIdentityCapability::MintsOwn); + let (roi, object_info) = drifted_roi_and_object(); + let mut rinfo = replicate_all_target_info(&roi, &target); + + let action = resolve_replicate_all_action( + ReplicateAllActionContext { + siblings: &ClaimedBySibling(DRIFTED_ASSIGNED_VERSION_ID), + roi: &roi, + tgt_client: &target, + bucket: &roi.bucket, + object: &roi.name, + start_time: OffsetDateTime::now_utc(), + ssec_audit_required: false, + }, + object_info, + &mut rinfo, + ) + .await; + + assert!(matches!(action, Some((ReplicationAction::All, _))), "{action:?}"); + assert_eq!(server.join().expect("test HTTP server should finish").len(), 2); + } + + #[test] + fn abandoned_purge_completes_locally_and_keeps_a_finished_target_untouched() { + let arn = "arn:rustfs:replication::removed-target"; + let dobj = version_purge_dobj(arn); + let rinfo = abandoned_purge_target_info("source", &dobj, arn); + assert_eq!(rinfo.version_purge_status, VersionPurgeStatusType::Complete); + assert_eq!(rinfo.arn, arn); + assert!(rinfo.error.is_none()); + + let mut finished = version_purge_dobj(arn); + finished + .delete_object + .replication_state + .as_mut() + .expect("purge state") + .purge_targets + .insert(arn.to_string(), VersionPurgeStatusType::Complete); + let rinfo = abandoned_purge_target_info("source", &finished, arn); + assert_eq!(rinfo.version_purge_status, VersionPurgeStatusType::Complete); + } + + #[tokio::test] + async fn version_purge_refuses_a_corrupt_ledger() { + let target = test_target_client("http://127.0.0.1:1".to_string()); + register_test_target(&target).await; + let dobj = version_purge_dobj(&target.arn); + let source = Arc::new(VersionPurgeSource { + etag: Some(DRIFTED_ETAG.to_string()), + ledger: HashMap::new(), + ledger_corrupt: true, + claimed_by_siblings: HashMap::new(), + }); + + let rinfo = replicate_delete_to_target(&dobj, target.clone(), Some(source)).await; + + assert_eq!(rinfo.version_purge_status, VersionPurgeStatusType::Failed); + assert!( + rinfo.error.as_deref().is_some_and(|error| error.contains("inconsistent")), + "{:?}", + rinfo.error + ); + assert!(purge_backoff_active(&dobj)); + } + + #[tokio::test] + async fn metadata_update_on_a_drifted_target_uses_the_tagging_api_instead_of_a_put() { + let (endpoint, server) = spawn_scripted_target_server(3, |line| { + if line.starts_with("HEAD ") && line.contains(&format!("versionId={DRIFTED_ASSIGNED_VERSION_ID}")) { + format!( + "HTTP/1.1 200 OK\r\nETag: \"{DRIFTED_ETAG}\"\r\nContent-Length: 4\r\nx-amz-version-id: {DRIFTED_ASSIGNED_VERSION_ID}\r\nx-amz-tagging-count: 1\r\nLast-Modified: Sun, 06 Sep 2026 10:00:00 GMT\r\nConnection: close\r\n\r\n" + ) + } else if line.starts_with("HEAD ") { + empty_response("404 Not Found") + } else if line.starts_with("PUT ") + && line.contains("tagging") + && line.contains(&format!("versionId={DRIFTED_ASSIGNED_VERSION_ID}")) + { + empty_response("200 OK") + } else { + empty_response("500 Unexpected") + } + }); + let target = test_target_client(endpoint); + let (mut roi, mut object_info) = drifted_roi_and_object(); + roi.op_type = ReplicationType::Metadata; + roi.user_tags = "phase=after".to_string(); + object_info.user_tags = Arc::new("phase=after".to_string()); + let mut user_defined = HashMap::new(); + insert_str( + &mut user_defined, + &format!("{SUFFIX_REPLICATION_TARGET_VERSION_ARN_PREFIX}{}", target.arn), + DRIFTED_ASSIGNED_VERSION_ID.to_string(), + ); + object_info.user_defined = Arc::new(user_defined); + let mut rinfo = replicate_all_target_info(&roi, &target); + + let action = resolve_replicate_all_action( + ReplicateAllActionContext { + siblings: &NoSiblingLedger, + roi: &roi, + tgt_client: &target, + bucket: &roi.bucket, + object: &roi.name, + start_time: OffsetDateTime::now_utc(), + ssec_audit_required: false, + }, + object_info, + &mut rinfo, + ) + .await; + + assert!( + matches!(action, Some((ReplicationAction::None, _))), + "metadata is applied in place; the PUT transport must not run: {action:?}" + ); + assert!(rinfo.error.is_none(), "{:?}", rinfo.error); + assert!( + rinfo.target_version_id.is_none(), + "an id the ledger already records must not trigger another metadata writeback" + ); + let seen = server.join().expect("test HTTP server should finish"); + assert_eq!(seen.len(), 3, "HEAD by source id, HEAD by ledger id, PutObjectTagging: {seen:?}"); + assert!( + seen[1].starts_with("HEAD ") && seen[1].contains(DRIFTED_ASSIGNED_VERSION_ID), + "{}", + seen[1] + ); + assert!(seen[2].starts_with("PUT ") && seen[2].contains("tagging"), "{}", seen[2]); + } + + #[test] + fn ledger_records_only_ids_the_target_minted() { + let source = Uuid::new_v4(); + assert_eq!(assigned_target_version_id(Some(source.to_string()), Some(source)), None); + assert_eq!(assigned_target_version_id(Some(String::new()), Some(source)), None); + assert_eq!(assigned_target_version_id(None, Some(source)), None); + assert_eq!( + assigned_target_version_id(Some(DRIFTED_ASSIGNED_VERSION_ID.to_string()), Some(source)).as_deref(), + Some(DRIFTED_ASSIGNED_VERSION_ID) + ); + } + #[test] fn put_response_audit_records_identity_verdict() { let target = test_target_client("http://127.0.0.1:1".to_string()); @@ -6177,6 +7513,42 @@ mod tests { assert_eq!(roi.target_purge_statuses.get(role), Some(&VersionPurgeStatusType::Pending)); } + /// The decision string is not persisted: after a restart, or once the + /// configuration is gone, a heal of a failed purge must still name the + /// targets the purge state records — that is what lets the delete worker + /// settle a removed target as abandoned instead of skipping forever. + #[tokio::test] + async fn heal_owes_a_failed_purge_to_the_targets_its_purge_state_names_without_a_configuration() { + let bucket = format!("heal-orphaned-purge-{}", Uuid::new_v4()); + let arn = "arn:rustfs:replication:us-east-1:removed:bucket"; + ReplicationVersioningStore::install_prefix_state_test_config( + &bucket, + VersioningConfiguration { + status: Some(BucketVersioningStatus::from_static(BucketVersioningStatus::ENABLED)), + ..Default::default() + }, + ); + let oi = ObjectInfo { + bucket, + name: "purge/orphaned.bin".to_string(), + version_id: Some(Uuid::new_v4()), + mod_time: Some(OffsetDateTime::now_utc()), + version_purge_status: VersionPurgeStatusType::Failed, + version_purge_status_internal: Some(format!("{arn}=FAILED;")), + ..Default::default() + }; + + let mut roi = get_heal_replicate_object_info(&oi, &ReplicationConfig::new(None, None)) + .await + .expect("a purge without a configuration must still classify"); + + assert!(roi.dsc.targets_map.get(arn).is_some_and(|target| target.replicate), "{:?}", roi.dsc); + assert!(matches!( + super::super::replication_queue_boundary::replication_heal_queue_action(&mut roi), + super::super::replication_queue_boundary::ReplicationHealQueueAction::QueueDelete(_) + )); + } + #[tokio::test] async fn heal_pending_purge_reads_one_versioning_generation() { let bucket = format!("heal-versioning-snapshot-{}", Uuid::new_v4()); @@ -6490,7 +7862,7 @@ mod tests { } #[test] - fn object_lock_denied_purge_backoff_tracks_version_and_target() { + fn purge_backoff_tracks_version_and_target() { let denied = DeletedObjectReplicationInfo { bucket: "worm-backoff-test-bucket".to_string(), target_arn: "arn:rustfs:replication::worm-test:t1".to_string(), @@ -6501,21 +7873,21 @@ mod tests { }, ..Default::default() }; - assert!(!object_lock_denied_purge_backoff_active(&denied)); + assert!(!purge_backoff_active(&denied)); - record_object_lock_denied_purge(&denied, "arn:rustfs:replication::worm-test:t1"); - assert!(object_lock_denied_purge_backoff_active(&denied)); + record_purge_backoff(&denied, "arn:rustfs:replication::worm-test:t1"); + assert!(purge_backoff_active(&denied)); // A requeue that can also reach a target this denial does not cover // must keep flowing: the purge may succeed there. let mut other_target = denied.clone(); other_target.target_arn = "arn:rustfs:replication::worm-test:t2".to_string(); - assert!(!object_lock_denied_purge_backoff_active(&other_target)); + assert!(!purge_backoff_active(&other_target)); // A different version of the same object must not be suppressed. let mut other_version = denied; other_version.delete_object.version_id = Some(uuid::Uuid::new_v4()); - assert!(!object_lock_denied_purge_backoff_active(&other_version)); + assert!(!purge_backoff_active(&other_version)); } #[tokio::test] @@ -6555,7 +7927,7 @@ mod tests { // failed abort must hand the upload id to the retry schedule (#6854): // the object itself is re-replicated under a fresh upload id, so // nothing else will ever abort this one. - let result = abort_multipart_on_failure( + let result = abort_multipart_on_failure::<_, _, _, ()>( Err(std::io::Error::other("transfer failed")), "dst-bucket", "obj", @@ -6579,7 +7951,7 @@ mod tests { let retry_scheduled = Arc::new(AtomicBool::new(false)); let retry_flag = retry_scheduled.clone(); - let result = abort_multipart_on_failure( + let result = abort_multipart_on_failure::<_, _, _, ()>( Err(std::io::Error::other("transfer failed")), "dst-bucket", "obj", @@ -7073,7 +8445,7 @@ mod tests { .await; server.abort(); assert!(server.await.expect_err("fixture server is stopped").is_cancelled()); - if let Some(error) = result.expect("replication must finish") { + if let Err(error) = result.expect("replication must finish") { panic!("legacy parts must replicate successfully: {error}"); } assert_eq!( diff --git a/crates/ecstore/src/bucket/replication/replication_target_boundary.rs b/crates/ecstore/src/bucket/replication/replication_target_boundary.rs index 387da2035..eb24d22ce 100644 --- a/crates/ecstore/src/bucket/replication/replication_target_boundary.rs +++ b/crates/ecstore/src/bucket/replication/replication_target_boundary.rs @@ -38,7 +38,7 @@ use time::format_description::well_known::Rfc3339; pub(crate) use crate::bucket::bucket_target_sys::{ AdvancedPutOptions, HeadObjectSdkError, PutObjectOptions, PutObjectPartOptions, RemotePutObjectResponse, RemoveObjectOptions, - S3ClientError, TargetClient, resolve_read_api_version_id, + ReplicaLocation, S3ClientError, TargetClient, resolve_read_api_version_id, }; #[cfg(test)] pub(crate) use crate::bucket::target::BucketTarget; diff --git a/crates/filemeta/src/replication.rs b/crates/filemeta/src/replication.rs index b34c01751..628952c65 100644 --- a/crates/filemeta/src/replication.rs +++ b/crates/filemeta/src/replication.rs @@ -442,6 +442,10 @@ pub struct ReplicatedTargetInfo { pub error: Option, #[serde(default, skip_serializing_if = "Option::is_none")] pub target_delete_marker_version_id: Option, + /// Kept in step with the replication crate's copy: the id a target that + /// mints its own version ids assigned to this object version. + #[serde(default, skip_serializing_if = "Option::is_none")] + pub target_version_id: Option, } impl ReplicatedTargetInfo { diff --git a/crates/replication/src/config.rs b/crates/replication/src/config.rs index e413434cf..c7dba3508 100644 --- a/crates/replication/src/config.rs +++ b/crates/replication/src/config.rs @@ -122,6 +122,10 @@ pub trait ReplicationConfigurationExt { fn has_active_rules(&self, prefix: &str, recursive: bool) -> bool; fn filter_target_arns(&self, obj: &ObjectOpts) -> Vec; fn filter_force_delete_target_arns(&self, prefix: &str) -> Vec; + /// Every target ARN the configuration still names, whatever the rule's + /// status, prefix or filter: the set a pending replication delete may + /// still be owed to. A target outside it was removed by the operator. + fn configured_target_arns(&self) -> HashSet; fn filter_target_replication_decisions(&self, obj: &ObjectOpts) -> Vec<(String, bool)> { self.filter_target_arns(obj) .into_iter() @@ -772,6 +776,19 @@ impl ReplicationConfigurationExt for ReplicationConfiguration { } /// Filter target ARNs and return a slice of the distinct values in the config + fn configured_target_arns(&self) -> HashSet { + let role = self.role.trim(); + if !role.is_empty() { + return HashSet::from([role.to_string()]); + } + self.rules + .iter() + .map(|rule| rule.destination.bucket.trim()) + .filter(|arn| !arn.is_empty()) + .map(str::to_string) + .collect() + } + fn filter_target_arns(&self, obj: &ObjectOpts) -> Vec { let role = self.role.trim(); if !role.is_empty() { diff --git a/crates/replication/src/filemeta.rs b/crates/replication/src/filemeta.rs index 555b58c57..0021dc69d 100644 --- a/crates/replication/src/filemeta.rs +++ b/crates/replication/src/filemeta.rs @@ -438,6 +438,12 @@ pub struct ReplicatedTargetInfo { /// Version the target assigned to the delete marker it just created. #[serde(default, skip_serializing_if = "Option::is_none")] pub target_delete_marker_version_id: Option, + /// Version the target assigned to this object version when it differs + /// from the source id (a target that mints its own ids). Persisted as the + /// per-target ledger every later version-addressed mutation resolves + /// through; `None` on targets that adopt the source id. + #[serde(default, skip_serializing_if = "Option::is_none")] + pub target_version_id: Option, } impl ReplicatedTargetInfo { diff --git a/crates/scanner/src/scanner_folder/item_actions.rs b/crates/scanner/src/scanner_folder/item_actions.rs index 1186e356e..67482603d 100644 --- a/crates/scanner/src/scanner_folder/item_actions.rs +++ b/crates/scanner/src/scanner_folder/item_actions.rs @@ -1036,12 +1036,19 @@ impl ScannerItem { return; } - let Some(replication) = self.replication.clone() else { - return; + let replication = match self.replication.clone() { + Some(replication) => (*replication).clone(), + // No active rules or targets, but a purge the bucket still owes + // must reach the heal path: the delete worker settles it against + // the current configuration (abandoned when the target is gone, + // rustfs/backlog#2340) so the hidden version stops blocking + // DeleteBucket. + None if !oi.version_purge_status.is_empty() => ReplicationConfig::new(None, None), + None => return, }; let done_replication = Metrics::time(Metric::CheckReplication); - let replication_result = queue_replication_heal(&oi.bucket, oi.clone(), (*replication).clone(), 0).await; + let replication_result = queue_replication_heal(&oi.bucket, oi.clone(), replication, 0).await; done_replication(); let roi = replication_result.object_info; record_scanner_replication_admission(global_metrics(), &roi, replication_result.admission); diff --git a/crates/utils/src/http/metadata_compat.rs b/crates/utils/src/http/metadata_compat.rs index f68889468..2a60be1e5 100644 --- a/crates/utils/src/http/metadata_compat.rs +++ b/crates/utils/src/http/metadata_compat.rs @@ -93,6 +93,10 @@ pub const SUFFIX_TIER_SKIP_FV_ID: &str = "tier-skip-fvid"; /// Per-target delete-marker version ids are stored one key per target ARN. pub const SUFFIX_REPLICATION_DELETE_MARKER_VERSION_ARN_PREFIX: &str = "replication-delete-marker-version-"; +/// Per-target data-version ids, one key per target ARN: the version a +/// replication target that mints its own ids assigned to this object version +/// (rustfs/backlog#2340). Absent on targets that adopt the source id. +pub const SUFFIX_REPLICATION_TARGET_VERSION_ARN_PREFIX: &str = "replication-target-version-"; // On-demand migration provenance. Written by the migration write-back onto // every pulled object so operators and later tooling can tell a migrated @@ -314,6 +318,17 @@ pub fn strip_internal_prefix_preserving_case(key: &str) -> Option<&str> { /// Reads the bounded per-target delete-marker version map in one metadata scan. /// The boolean is set when matching metadata is malformed or compatibility keys disagree. pub fn target_delete_marker_versions(map: &HashMap) -> (HashMap, bool) { + internal_versions_by_arn(map, SUFFIX_REPLICATION_DELETE_MARKER_VERSION_ARN_PREFIX) +} + +/// Reads the bounded per-target data-version ledger (the id each drifting +/// target assigned to this object version) in one metadata scan. Same +/// bounds and corruption reporting as [`target_delete_marker_versions`]. +pub fn replication_target_versions(map: &HashMap) -> (HashMap, bool) { + internal_versions_by_arn(map, SUFFIX_REPLICATION_TARGET_VERSION_ARN_PREFIX) +} + +fn internal_versions_by_arn(map: &HashMap, arn_prefix: &str) -> (HashMap, bool) { const MAX_ENTRIES: usize = 1_000; const MAX_ARN_LEN: usize = 1_024; const MAX_VERSION_ID_LEN: usize = 1_024; @@ -324,13 +339,13 @@ pub fn target_delete_marker_versions(map: &HashMap) -> (HashMap< let Some(suffix) = strip_internal_prefix_preserving_case(key) else { continue; }; - let Some(prefix) = suffix.get(..SUFFIX_REPLICATION_DELETE_MARKER_VERSION_ARN_PREFIX.len()) else { + let Some(prefix) = suffix.get(..arn_prefix.len()) else { continue; }; - if !prefix.eq_ignore_ascii_case(SUFFIX_REPLICATION_DELETE_MARKER_VERSION_ARN_PREFIX) { + if !prefix.eq_ignore_ascii_case(arn_prefix) { continue; } - let arn = &suffix[SUFFIX_REPLICATION_DELETE_MARKER_VERSION_ARN_PREFIX.len()..]; + let arn = &suffix[arn_prefix.len()..]; if !arn.starts_with("arn:") || arn.len() > MAX_ARN_LEN || value.is_empty() || value.len() > MAX_VERSION_ID_LEN { corrupt = true; continue; @@ -703,6 +718,37 @@ mod tests { assert!(corrupt); } + #[test] + fn replication_target_versions_are_keyed_apart_from_delete_marker_versions() { + let arn = "arn:rustfs:replication::target"; + let mut metadata = HashMap::new(); + insert_str( + &mut metadata, + &format!("{SUFFIX_REPLICATION_TARGET_VERSION_ARN_PREFIX}{arn}"), + "data-version".to_string(), + ); + insert_str( + &mut metadata, + &format!("{SUFFIX_REPLICATION_DELETE_MARKER_VERSION_ARN_PREFIX}{arn}"), + "marker-version".to_string(), + ); + + let (data_versions, corrupt) = replication_target_versions(&metadata); + assert!(!corrupt); + assert_eq!(data_versions.get(arn).map(String::as_str), Some("data-version")); + let (marker_versions, corrupt) = target_delete_marker_versions(&metadata); + assert!(!corrupt); + assert_eq!(marker_versions.get(arn).map(String::as_str), Some("marker-version")); + + metadata.insert( + format!("{MINIO_INTERNAL_PREFIX}{SUFFIX_REPLICATION_TARGET_VERSION_ARN_PREFIX}{arn}"), + "other-version".to_string(), + ); + let (data_versions, corrupt) = replication_target_versions(&metadata); + assert!(data_versions.is_empty()); + assert!(corrupt); + } + #[test] fn target_delete_marker_versions_bound_distinct_entries_during_scan() { let metadata = (0..=1_000) diff --git a/docs/operations/replication-check.md b/docs/operations/replication-check.md index d102d30db..fcada857f 100644 --- a/docs/operations/replication-check.md +++ b/docs/operations/replication-check.md @@ -62,4 +62,4 @@ The route returns HTTP 200 with JSON after all configured targets have been chec `VersionFidelity` pins the version-identity contract on both write paths. The probe PUT carries a source version id (header plus `?versionId=` query, the exact shape live replication uses) and the target must answer with the same id; a second probe repeats the check through CreateMultipartUpload -> UploadPart -> CompleteMultipartUpload, where the target fixes the version at initiate and only reports it on completion. A target can adopt PutObject ids and still mint its own for multipart; the failure message names the path that drifted. -A target that mints its own version ids breaks every version-addressed operation that follows (version deletes, heal re-drives). The phase therefore fails with `"Code": "BucketRemoteTargetVersionMismatch"`, the later mutation phases are skipped, and cleanup still removes the probe via the version id the target actually assigned. +A target that mints its own version ids never answers to the source version id. The phase therefore fails with `"Code": "BucketRemoteTargetVersionMismatch"` and the target result is `FAILED`. Replication to such a target still converges: the replication worker records the id the target assigned to each object version on the source (the target-version ledger, internal metadata key `replication-target-version-`) and addresses version deletes, tag and Object Lock updates through it. The `DeleteMarker` and `VersionDelete` phases probe exactly that path — they address the id the target assigned to the probe object, not the source id — so on a drifting target they report whether ledger-addressed purges work against this endpoint, and cleanup removes the probe via the same id. They are `SKIPPED` only when the probe `Put` itself failed or reported no version id. diff --git a/docs/operations/replication-outbound-transport.md b/docs/operations/replication-outbound-transport.md index 69996531e..125492735 100644 --- a/docs/operations/replication-outbound-transport.md +++ b/docs/operations/replication-outbound-transport.md @@ -18,7 +18,7 @@ | Rejects or mis-stores `aws-chunked` bodies (SeaweedFS 3.97) | Handled by the plain-payload default above. | Outbound target matrix, `RejectAwsChunked` mode | | Requires `Content-MD5` or `x-amz-checksum-*` on a PutObject with Object Lock parameters (AWS S3, MinIO, Impossible Cloud, most compatible stores) | Satisfied: a locked single PUT carries `Content-MD5` derived from the source ETag (plaintext objects whose ETag is the MD5 of the wire bytes) or an SDK CRC32 checksum (multipart-layout ETags, managed SSE, SSE-C passthrough — this one is an `aws-chunked` trailer, so a target that also rejects that framing cannot take such objects). Releases before this fix (`1.0.0-rc.5`) need `RUSTFS_REPLICATION_STREAMING_CHECKSUMS=true` as a workaround. | Outbound target matrix, `RequireChecksumWithObjectLock` mode | | Stores `x-amz-checksum-*` from a PutObject and returns it on `HEAD ?ChecksumMode=ENABLED` (AWS S3, Wasabi, RustFS) | Satisfied for single-part objects: the replica answers with the source's checksum. Before this fix (`1.0.0-rc.5`) the checksum left the source as `x-amz-meta-` user metadata and no replica carried it (rustfs/backlog#2340). | Outbound target matrix, `Checksummed` shape | -| Mints its own version ids (AWS S3, Wasabi, Impossible Cloud) | Data lands; version-addressed convergence does not. See rustfs/backlog#2085 and `docs/operations/replication-check.md` (VersionFidelity). | `replication-check`, outbound target matrix, `MintOwnVersionIds` mode | +| Mints its own version ids (AWS S3, Wasabi, Impossible Cloud) | Data lands, and later mutations converge through the target-version ledger: the id the target assigned is recorded on the source version (internal key `replication-target-version-`) and version deletes, tag and Object Lock updates address it (tag/retention/legal-hold through the metadata-only APIs, never a re-PUT). A replica written before the ledger existed is located by exact key + ETag, minus the candidates other generations of the key already claim; an ambiguous remainder is refused with a one-hour backoff (`replication_purge_replica_unresolved`). `NoSuchVersion` on a version-addressed DELETE counts as purged. See rustfs/backlog#2340, rustfs/backlog#2085 and `docs/operations/replication-check.md` (VersionFidelity). | `replication-check`, outbound target matrix, `MintOwnVersionIds` mode (also models Wasabi's 404 `NoSuchVersion` on unknown ids) | | Returns an ETag that is not the content MD5 without announcing SSE | Every single-part object fails ETag verification. Set `RUSTFS_REPLICATION_REPLICA_ETAG_VERIFY=false`. | Replication status FAILED with `replica etag mismatch` | ## Environment knobs diff --git a/rustfs/src/admin/router.rs b/rustfs/src/admin/router.rs index bb0e06f72..1945acc1f 100644 --- a/rustfs/src/admin/router.rs +++ b/rustfs/src/admin/router.rs @@ -18,7 +18,7 @@ use super::storage_api::bucket::replication::{self, BucketReplicationResyncStatu use super::storage_api::bucket::target::{BucketTarget, BucketTargetType, BucketTargets}; use super::storage_api::bucket::target_sys::{ BucketTargetSys, PutObjectOptions, RemoveObjectOptions, S3ClientError, SsecPassthroughCapability, TargetClient, - VersionIdentityCapability, append_version_id_query, + VersionIdentityCapability, append_version_id_query, resolve_delete_api_version_id, }; use super::storage_api::bucket::versioning_sys::BucketVersioningSys; use super::storage_api::bucket::{AdminReplicationConfigExt as _, AdminVersioningConfigExt as _}; @@ -2245,12 +2245,13 @@ async fn execute_replication_probe(result: &mut ReplicationCheckTargetStatus, op match operations.put().await { Ok(outcome) => { result.phases.put = ReplicationCheckPhaseStatus::passed(); - // P1-19 version-identity contract: replication only converges on - // targets that adopt the source version id — version-addressed - // deletes and heal re-drives never match a minted id. Judge it - // from the probe PUT's own response; on mismatch the later - // mutation phases are pointless (they address by version id), but - // cleanup still runs against whatever id the target assigned. + // P1-19 version-identity contract: a target that mints its own + // version ids never answers to the source id. Judge it from the + // probe PUT's own response. The mutation phases below still run + // on such a target: they address the id the target assigned — + // the same ledger the replication worker records per object + // (rustfs/backlog#2340) — so they report whether version- + // addressed deletes can converge there at all. match version_fidelity_error("PutObject", &outcome) { None => result.phases.version_fidelity = ReplicationCheckPhaseStatus::passed(), Some(error) => { @@ -2325,7 +2326,12 @@ async fn execute_replication_probe(result: &mut ReplicationCheckTargetStatus, op } } - if result.phases.put.status == "OK" && result.phases.version_fidelity.status == "OK" { + // DeleteMarker / VersionDelete address `probe_version_id`, the id the + // target actually assigned, so they run on a drifting target too. What + // they cannot prove there is the source-id addressing VersionFidelity + // already failed; what they do prove is that the ledger-addressed purge + // path the worker uses on such a target works against this endpoint. + if result.phases.put.status == "OK" && probe_version_id.is_some() { match operations.create_delete_marker(probe_version_id.as_deref()).await { Ok(version_id) => { delete_marker_version_id = version_id; @@ -2739,12 +2745,22 @@ async fn delete_replication_probe_object( insert_header(&mut headers, SUFFIX_SOURCE_REPLICATION_CHECK, "true"); } + if let Some(version_id) = version_id { + insert_header(&mut headers, SUFFIX_SOURCE_VERSION_ID, version_id); + } + + // Same wire shape as live delete replication: a marker creation carries + // no `versionId` (the target mints the marker), a version delete does. A + // generic S3 target handed the version id on the marker step would + // permanently delete the probe version instead, and the VersionDelete + // phase would then find nothing (seen on Wasabi). + let api_version_id = resolve_delete_api_version_id(version_id.map(ToOwned::to_owned), &options); target_client .client .delete_object() .bucket(target_bucket) .key(probe_key) - .set_version_id(version_id.map(ToOwned::to_owned)) + .set_version_id(api_version_id) .customize() .map_request(move |mut req| { for (key, value) in headers.clone() { @@ -2776,6 +2792,14 @@ async fn delete_replication_probe_version( .map_err(S3ClientError::from) } +/// The VersionDelete phase already removed the probe version the cleanup is +/// handed, and a strict S3 target (Wasabi) answers a second DELETE of that +/// id with `NoSuchVersion` where RustFS/MinIO answer 204: the goal is met +/// either way. +fn probe_version_already_gone(err: &S3ClientError) -> bool { + matches!(err.code.as_deref(), Some("NoSuchKey" | "NoSuchVersion")) +} + async fn cleanup_replication_probe<'a>( target_client: &TargetClient, target_bucket: &str, @@ -2787,6 +2811,7 @@ async fn cleanup_replication_probe<'a>( for version_id in known_version_ids.into_iter().flatten() { if deleted_ids.insert(version_id.to_string()) && let Err(err) = delete_replication_probe_version(target_client, target_bucket, probe_key, version_id).await + && !probe_version_already_gone(&err) { errors.push(format_replication_check_client_error( &err, @@ -2832,6 +2857,7 @@ async fn cleanup_replication_probe<'a>( for version_id in discovered_ids { if deleted_ids.insert(version_id.clone()) && let Err(err) = delete_replication_probe_version(target_client, target_bucket, probe_key, &version_id).await + && !probe_version_already_gone(&err) { errors.push(format_replication_check_client_error( &err, @@ -3894,6 +3920,8 @@ mod tests { version_delete_error: Option<&'static str>, cleanup_error: Option<&'static str>, calls: Vec<&'static str>, + /// Version ids the delete-marker and version-delete phases addressed. + mutation_ids: Vec>, cleanup_ids: Vec>, } @@ -3938,16 +3966,18 @@ mod tests { } } - async fn create_delete_marker(&mut self, _version_id: Option<&str>) -> Result, S3ClientError> { + async fn create_delete_marker(&mut self, version_id: Option<&str>) -> Result, S3ClientError> { self.calls.push("delete-marker"); + self.mutation_ids.push(version_id.map(ToOwned::to_owned)); match self.delete_marker_error { Some(code) => Err(scripted_probe_error(code)), None => Ok(Some("marker-version".to_string())), } } - async fn delete_version(&mut self, _version_id: Option<&str>) -> Result<(), S3ClientError> { + async fn delete_version(&mut self, version_id: Option<&str>) -> Result<(), S3ClientError> { self.calls.push("version-delete"); + self.mutation_ids.push(version_id.map(ToOwned::to_owned)); match self.version_delete_error { Some(code) => Err(scripted_probe_error(code)), None => Ok(()), @@ -3968,12 +3998,12 @@ mod tests { } /// P1-19: a target that mints its own version ids must fail the - /// VersionFidelity phase with the machine-readable mismatch code, skip - /// the version-addressed mutation phases (they cannot mean anything on a - /// drifting target), and still clean up using the id the target actually - /// assigned — the source-derived id would never match. + /// VersionFidelity phase with the machine-readable mismatch code. The + /// mutation phases still run, addressing the id the target assigned + /// (the ledger the worker records per object, rustfs/backlog#2340), and + /// cleanup uses that id too — the source-derived id would never match. #[tokio::test] - async fn replication_probe_flags_version_minting_target() { + async fn replication_probe_flags_version_minting_target_and_probes_mutations_by_assigned_id() { let mut result = replication_check_target("arn:a", "OK", None); let mut operations = ScriptedReplicationProbe { minted_version_id: Some("target-minted-version"), @@ -3982,16 +4012,52 @@ mod tests { execute_replication_probe(&mut result, &mut operations).await; - assert_eq!(operations.calls, ["put", "cleanup"]); + assert_eq!(operations.calls, ["put", "delete-marker", "version-delete", "cleanup"]); assert_eq!(result.status, "FAILED"); assert_eq!(result.phases.put.status, "OK"); assert_eq!(result.phases.version_fidelity.status, "FAILED"); assert_eq!(result.phases.version_fidelity.code, Some(REPLICATION_CHECK_CODE_VERSION_MISMATCH)); - assert_eq!(result.phases.delete_marker.status, "SKIPPED"); - assert_eq!(result.phases.version_delete.status, "SKIPPED"); + assert_eq!(result.phases.delete_marker.status, "OK"); + assert_eq!(result.phases.version_delete.status, "OK"); + assert_eq!( + operations.mutation_ids, + [ + Some("target-minted-version".to_string()), + Some("target-minted-version".to_string()) + ], + "both mutation phases must address the id the target assigned" + ); assert_eq!(result.phases.ssec_passthrough.status, "SKIPPED"); assert_eq!(result.phases.cleanup.status, "OK"); - assert_eq!(operations.cleanup_ids, [Some("target-minted-version".to_string()), None, None, None]); + assert_eq!( + operations.cleanup_ids, + [ + Some("target-minted-version".to_string()), + None, + None, + Some("marker-version".to_string()) + ] + ); + } + + /// A drifting target that also refuses the ledger-addressed delete keeps + /// the phase-level evidence: VersionDelete fails on its own, apart from + /// the identity verdict. + #[tokio::test] + async fn replication_probe_reports_version_delete_failure_on_a_drifting_target() { + let mut result = replication_check_target("arn:a", "OK", None); + let mut operations = ScriptedReplicationProbe { + minted_version_id: Some("target-minted-version"), + version_delete_error: Some("AccessDenied"), + ..Default::default() + }; + + execute_replication_probe(&mut result, &mut operations).await; + + assert_eq!(result.phases.version_fidelity.status, "FAILED"); + assert_eq!(result.phases.delete_marker.status, "OK"); + assert_eq!(result.phases.version_delete.status, "FAILED"); + assert_eq!(result.phases.cleanup.status, "OK"); } #[tokio::test] diff --git a/rustfs/src/admin/storage_api.rs b/rustfs/src/admin/storage_api.rs index 6b6b4179d..f257a9ce5 100644 --- a/rustfs/src/admin/storage_api.rs +++ b/rustfs/src/admin/storage_api.rs @@ -204,6 +204,7 @@ pub(crate) mod bandwidth { pub(crate) mod bucket_target_sys { pub(crate) use super::ecstore_bucket::bucket_target_sys::append_version_id_query; + pub(crate) use super::ecstore_bucket::bucket_target_sys::resolve_delete_api_version_id; pub(crate) type AdvancedPutOptions = super::ecstore_bucket::bucket_target_sys::AdvancedPutOptions; pub(crate) type BucketTargetError = super::ecstore_bucket::bucket_target_sys::BucketTargetError; pub(crate) type BucketTargetSys = super::ecstore_bucket::bucket_target_sys::BucketTargetSys;