diff --git a/crates/ecstore/src/bucket/durability.rs b/crates/ecstore/src/bucket/durability.rs index 1b3bc0176..a54448431 100644 --- a/crates/ecstore/src/bucket/durability.rs +++ b/crates/ecstore/src/bucket/durability.rs @@ -64,25 +64,40 @@ impl BucketDurabilityConfig { } } -/// Default durability tier seeded into a newly created bucket's metadata -/// (rustfs/backlog#1811). `relaxed` aligns new buckets with MinIO's default -/// posture: object data is still fdatasynced, while xl.meta and directory-entry -/// fsyncs follow the relaxed durability gate. +/// Default durability policy for newly created buckets. An unset value must +/// inherit the process-wide mode so a default single-node deployment remains +/// strict. Relaxed and none remain explicit operator choices. pub const ENV_NEW_BUCKET_DURABILITY_MODE: &str = "RUSTFS_NEW_BUCKET_DURABILITY_MODE"; -pub const DEFAULT_NEW_BUCKET_DURABILITY_MODE: &str = BUCKET_DURABILITY_MODE_RELAXED; +pub const DEFAULT_NEW_BUCKET_DURABILITY_MODE: &str = "inherit"; + +const EVENT_NEW_BUCKET_DURABILITY_MODE: &str = "new_bucket_durability_mode"; +const LOG_COMPONENT_ECSTORE: &str = "ecstore"; +const LOG_SUBSYSTEM_BUCKET_DURABILITY: &str = "bucket_durability"; /// The `durability.json` bytes to seed into a freshly created bucket's metadata. /// Empty means "no override" (the bucket then follows the global -/// `RUSTFS_DURABILITY_MODE`); otherwise the serialized chosen tier. Operators -/// can set `inherit` to disable the new-bucket override. Invalid values also -/// fail closed to inherit the global mode instead of seeding a surprising tier. -pub fn new_bucket_durability_config_json() -> Vec { +/// `RUSTFS_DURABILITY_MODE`); otherwise the serialized chosen tier. Invalid +/// values also fail closed to inherit the global mode instead of seeding a +/// surprising tier. +pub fn new_bucket_durability_config_json(bucket: &str) -> Vec { let raw = std::env::var(ENV_NEW_BUCKET_DURABILITY_MODE).unwrap_or_else(|_| DEFAULT_NEW_BUCKET_DURABILITY_MODE.to_string()); let mode = raw.trim(); if mode.eq_ignore_ascii_case("inherit") || mode.is_empty() || !BucketDurabilityConfig::is_valid_mode(mode) { return Vec::new(); } - serde_json::to_vec(&BucketDurabilityConfig::new(mode)).expect("BucketDurabilityConfig serialization cannot fail") + let config = BucketDurabilityConfig::new(mode); + if let Some(mode) = config.normalized_mode().filter(|mode| mode != BUCKET_DURABILITY_MODE_STRICT) { + tracing::warn!( + event = EVENT_NEW_BUCKET_DURABILITY_MODE, + component = LOG_COMPONENT_ECSTORE, + subsystem = LOG_SUBSYSTEM_BUCKET_DURABILITY, + state = "non_strict_new_bucket_override", + bucket = %bucket, + mode = %mode, + "New bucket durability is explicitly configured below strict" + ); + } + serde_json::to_vec(&config).expect("BucketDurabilityConfig serialization cannot fail") } #[cfg(test)] @@ -90,7 +105,7 @@ mod tests { use super::*; fn new_bucket_seeded_mode() -> Option { - let json = new_bucket_durability_config_json(); + let json = new_bucket_durability_config_json("test-bucket"); if json.is_empty() { return None; } @@ -132,9 +147,9 @@ mod tests { } #[test] - fn new_bucket_default_seeds_relaxed_when_unset() { + fn new_bucket_default_inherits_when_unset() { temp_env::with_var_unset(ENV_NEW_BUCKET_DURABILITY_MODE, || { - assert_eq!(new_bucket_seeded_mode().as_deref(), Some(BUCKET_DURABILITY_MODE_RELAXED)); + assert_eq!(new_bucket_seeded_mode(), None); }); } diff --git a/crates/ecstore/src/bucket/metadata.rs b/crates/ecstore/src/bucket/metadata.rs index cc75e2d7f..ff9d16890 100644 --- a/crates/ecstore/src/bucket/metadata.rs +++ b/crates/ecstore/src/bucket/metadata.rs @@ -593,7 +593,7 @@ impl BucketMetadata { /// durability posture. pub fn new_with_default_durability(name: &str) -> Self { let mut metadata = Self::new(name); - metadata.durability_config_json = super::durability::new_bucket_durability_config_json(); + metadata.durability_config_json = super::durability::new_bucket_durability_config_json(name); metadata } @@ -1732,21 +1732,15 @@ mod test { } #[test] - fn new_bucket_metadata_constructor_seeds_default_durability() { + fn new_bucket_metadata_constructor_inherits_default_durability() { temp_env::with_var_unset(crate::bucket::durability::ENV_NEW_BUCKET_DURABILITY_MODE, || { let metadata = BucketMetadata::new_with_default_durability("new-user-bucket"); - assert_eq!( - metadata.durability_config().and_then(|cfg| cfg.normalized_mode()).as_deref(), - Some(crate::bucket::durability::BUCKET_DURABILITY_MODE_RELAXED) - ); + assert!(metadata.durability_config().is_none()); let encoded = metadata.marshal_msg().expect("marshal metadata"); let decoded = BucketMetadata::unmarshal(&encoded).expect("unmarshal metadata"); assert_eq!(decoded.durability_config_json, metadata.durability_config_json); - assert_eq!( - decoded.durability_config().and_then(|cfg| cfg.normalized_mode()).as_deref(), - Some(crate::bucket::durability::BUCKET_DURABILITY_MODE_RELAXED) - ); + assert!(decoded.durability_config().is_none()); }); } diff --git a/crates/ecstore/src/disk/local.rs b/crates/ecstore/src/disk/local.rs index aa8a2c39a..3cc2aac1a 100644 --- a/crates/ecstore/src/disk/local.rs +++ b/crates/ecstore/src/disk/local.rs @@ -1372,13 +1372,24 @@ pub(crate) fn durability_mode() -> DurabilityMode { rustfs_utils::get_env_opt_str(ENV_RUSTFS_DURABILITY_MODE), rustfs_utils::get_env_bool(ENV_RUSTFS_DRIVE_SYNC_ENABLE, DEFAULT_RUSTFS_DRIVE_SYNC_ENABLE), ); - info!( - event = EVENT_DISK_LOCAL_DURABILITY_MODE, - component = LOG_COMPONENT_ECSTORE, - subsystem = LOG_SUBSYSTEM_DISK_LOCAL, - mode = mode.as_str(), - "Storage durability mode resolved" - ); + if mode == DurabilityMode::Strict { + info!( + event = EVENT_DISK_LOCAL_DURABILITY_MODE, + component = LOG_COMPONENT_ECSTORE, + subsystem = LOG_SUBSYSTEM_DISK_LOCAL, + mode = mode.as_str(), + "Storage durability mode resolved" + ); + } else { + warn!( + event = EVENT_DISK_LOCAL_DURABILITY_MODE, + component = LOG_COMPONENT_ECSTORE, + subsystem = LOG_SUBSYSTEM_DISK_LOCAL, + state = "non_strict_mode_configured", + mode = mode.as_str(), + "Storage durability mode does not provide strict power-loss durability" + ); + } mode }) } @@ -6915,7 +6926,9 @@ impl LocalDisk { let (buf, mtime) = res?; if buf.is_empty() { - return Err(DiskError::FileNotFound); + // A missing xl.meta is mapped by the open/read error above. A file + // that exists but has no metadata bytes is corruption, not absence. + return Err(DiskError::FileCorrupt); } Ok((buf, mtime)) @@ -11697,6 +11710,39 @@ mod test { assert_eq!(err, DiskError::FileCorrupt); } + #[tokio::test] + async fn read_version_reports_empty_xl_meta_as_corrupt_not_missing() { + use tempfile::tempdir; + + let dir = tempdir().expect("test directory should be created"); + let endpoint = Endpoint::try_from(dir.path().to_str().expect("test path should be utf8")).expect("endpoint should parse"); + let disk = LocalDisk::new(&endpoint, false).await.expect("local disk should be created"); + let bucket = "bucket"; + let object = "empty-metadata"; + ensure_test_volume(&disk, bucket).await; + + let missing_err = disk + .read_version("", bucket, "missing-object", "", &ReadOptions::default()) + .await + .expect_err("a missing xl.meta remains not found"); + assert_eq!(missing_err, DiskError::FileNotFound); + + let object_dir = dir.path().join(bucket).join(object); + fs::create_dir_all(&object_dir) + .await + .expect("object directory should be created"); + fs::write(object_dir.join(STORAGE_FORMAT_FILE), b"") + .await + .expect("empty metadata fixture should be written"); + + let err = disk + .read_version("", bucket, object, "", &ReadOptions::default()) + .await + .expect_err("an existing zero-length xl.meta is corrupt, not an absent object"); + + assert_eq!(err, DiskError::FileCorrupt); + } + #[tokio::test] async fn read_version_delete_marker_never_enters_inline_shard_math() { use tempfile::tempdir; diff --git a/crates/ecstore/src/store/bucket.rs b/crates/ecstore/src/store/bucket.rs index 9debdafbc..4695fee4a 100644 --- a/crates/ecstore/src/store/bucket.rs +++ b/crates/ecstore/src/store/bucket.rs @@ -3394,7 +3394,7 @@ mod tests { #[tokio::test(flavor = "multi_thread")] #[serial] - async fn make_bucket_seeds_new_bucket_durability_override() { + async fn make_bucket_inherits_default_durability() { temp_env::async_with_vars([(crate::bucket::durability::ENV_NEW_BUCKET_DURABILITY_MODE, None::<&str>)], async { let (_disk_paths, ecstore) = setup_bucket_delete_test_env().await; let bucket = format!("bucket-default-durability-{}", Uuid::new_v4().simple()); @@ -3407,10 +3407,7 @@ mod tests { let metadata = metadata_sys::get_in(&ecstore.ctx, &bucket) .await .expect("metadata should load for the new bucket"); - assert_eq!( - metadata.durability_config().and_then(|cfg| cfg.normalized_mode()).as_deref(), - Some(crate::bucket::durability::BUCKET_DURABILITY_MODE_RELAXED) - ); + assert!(metadata.durability_config().is_none()); }) .await; } diff --git a/crates/scanner/src/scanner_folder.rs b/crates/scanner/src/scanner_folder.rs index ff1d80d4d..ae0a6440c 100644 --- a/crates/scanner/src/scanner_folder.rs +++ b/crates/scanner/src/scanner_folder.rs @@ -718,6 +718,8 @@ pub struct FolderScanner { failed_object_ttl_secs: u64, failed_objects_max: usize, + failed_object_paths_seen: HashSet, + resolved_failed_object_paths: HashSet, sleeper: DynamicSleeper, // should_heal: Arc bool + Send + Sync>, @@ -927,7 +929,7 @@ impl FolderScanner { into.add_child(child_hash); } - fn should_skip_failed(&self, path: &str) -> bool { + fn failed_retry_suppressed(&self, path: &str) -> bool { let ttl = self.failed_object_ttl_secs; if ttl == 0 { return false; @@ -947,6 +949,7 @@ impl FolderScanner { return; } + self.resolved_failed_object_paths.remove(path); let now = Self::now_secs(); self.new_cache.info.failed_objects.insert(path.to_string(), now); @@ -954,6 +957,43 @@ impl FolderScanner { if max_entries > 0 && self.new_cache.info.failed_objects.len() > max_entries { self.prune_failed_objects(now, ttl); } + if self.new_cache.info.failed_objects.contains_key(path) { + self.failed_object_paths_seen.insert(path.to_string()); + } + } + + fn mark_failed_path_seen(&mut self, path: &str) -> bool { + if self.new_cache.info.failed_objects.contains_key(path) { + return !self.failed_object_paths_seen.insert(path.to_string()); + } + false + } + + fn clear_failed_path(&mut self, path: &str) { + if self.new_cache.info.failed_objects.contains_key(path) { + self.resolved_failed_object_paths.insert(path.to_string()); + } + self.failed_object_paths_seen.remove(path); + } + + fn reconcile_failed_objects_after_full_scan(&mut self) { + self.new_cache + .info + .failed_objects + .retain(|path, _| !self.resolved_failed_object_paths.contains(path)); + self.resolved_failed_object_paths.clear(); + + let mixed_coverage = self + .new_cache + .info + .scan_progress + .is_some_and(|progress| progress.started_plan != progress.requested_plan); + if self.prefix_scan_scope.is_some() || self.resume_frontier.is_some() || self.coverage_gap || mixed_coverage { + return; + } + + let seen = &self.failed_object_paths_seen; + self.new_cache.info.failed_objects.retain(|path, _| seen.contains(path)); } fn prune_failed_objects_cache(&mut self) { @@ -1795,12 +1835,14 @@ impl FolderScanner { file_type: entry_type, }; - // If this path is already known as failed, just skip it. - // We intentionally do NOT call `record_failed` or bump `failed_objects` here, - // because the failure was recorded when the original error occurred - // (e.g. in the get_size error branch below). This branch only accounts - // for subsequent skips of already-failed paths. - if self.should_skip_failed(&item.path) { + // Keep failed paths visible to this scan so repair success can + // clear stale failure state and a complete walk can discard + // paths that were removed since the previous cycle. + let repeated_failed_path = self.mark_failed_path_seen(&item.path); + if repeated_failed_path && self.failed_retry_suppressed(&item.path) { + // A new FolderScanner starts with an empty seen set, so + // cached paths are still rechecked once per cycle. Skip + // duplicate listings within this same scanner pass. self.coverage_gap |= self.old_cache.info.scan_progress.is_some(); continue; } @@ -1813,14 +1855,17 @@ impl FolderScanner { Ok(sz) => sz, Err(e) => { let failure_action = classify_get_size_failure(&item, &e); + let retry_suppressed = self.failed_retry_suppressed(&item.path); if failure_action != GetSizeFailureAction::Skip { + self.resolved_failed_object_paths.remove(&item.path); self.coverage_gap |= self.old_cache.info.scan_progress.is_some(); - // Track failed objects to prevent infinite retry loops - into.failed_objects += 1; - self.record_failed(&item.path); + if !retry_suppressed { + into.failed_objects += 1; + self.record_failed(&item.path); + } - if should_log_failed_object(into.failed_objects) { + if !retry_suppressed && should_log_failed_object(into.failed_objects) { if let GetSizeFailureAction::HealMetadata { object } = &failure_action { error!( target: "rustfs::scanner::folder", @@ -1850,9 +1895,13 @@ impl FolderScanner { ); } } + } else { + self.clear_failed_path(&item.path); } - if let GetSizeFailureAction::HealMetadata { object } = failure_action { + if let GetSizeFailureAction::HealMetadata { object } = failure_action + && !retry_suppressed + { // Single-flight (backlog#1894 axis A) — the // recording mode and its guarantees are pinned by // corrupt_metadata_recording below. @@ -1901,6 +1950,7 @@ impl FolderScanner { } }; + self.clear_failed_path(&item.path); found_object_metadata = true; item.transform_meta_dir(); @@ -1967,8 +2017,10 @@ impl FolderScanner { self.coverage_gap |= self.old_cache.info.scan_progress.is_some(); found_object_metadata = true; let metadata_path = path_join_buf(&[&dir_path, STORAGE_FORMAT_FILE]); + self.mark_failed_path_seen(&metadata_path); + let retry_suppressed = self.failed_retry_suppressed(&metadata_path); - if !self.should_skip_failed(&metadata_path) { + if !retry_suppressed { into.failed_objects = into.failed_objects.saturating_add(1); self.record_failed(&metadata_path); @@ -1985,7 +2037,9 @@ impl FolderScanner { "Scanner found erasure object data without metadata" ); } + } + if !retry_suppressed { let (bucket, object) = path2_bucket_object_with_base_path(&self.root, &folder.name); if !bucket.is_empty() && !object.is_empty() { self.send_required_scanner_heal_request( @@ -2769,6 +2823,8 @@ pub(crate) async fn scan_data_folder_scoped( prefix_scan_scope, failed_object_ttl_secs: failed_object_ttl, failed_objects_max, + failed_object_paths_seen: HashSet::new(), + resolved_failed_object_paths: HashSet::new(), sleeper, disks, disks_quorum, @@ -2823,6 +2879,7 @@ pub(crate) async fn scan_data_folder_scoped( Ok(()) => { // Get the new cache and finalize it let coverage_gap = scanner.coverage_gap; + scanner.reconcile_failed_objects_after_full_scan(); let new_cache = scanner.as_mut_new_cache(); new_cache.force_compact(DATA_SCANNER_COMPACT_AT_CHILDREN); new_cache.info.last_update = Some(SystemTime::now()); diff --git a/crates/scanner/src/scanner_folder/tests.rs b/crates/scanner/src/scanner_folder/tests.rs index fea7f6499..1cd050f44 100644 --- a/crates/scanner/src/scanner_folder/tests.rs +++ b/crates/scanner/src/scanner_folder/tests.rs @@ -336,6 +336,8 @@ async fn build_test_scanner() -> (FolderScanner, std::path::PathBuf) { prefix_scan_scope: None, failed_object_ttl_secs: u64::MAX, failed_objects_max: usize::MAX, + failed_object_paths_seen: HashSet::new(), + resolved_failed_object_paths: HashSet::new(), sleeper: SCANNER_SLEEPER.clone(), disks: Vec::new(), disks_quorum: 0, @@ -454,7 +456,7 @@ impl Drop for TestGuard { #[tokio::test] #[serial] -async fn test_should_skip_failed_respects_ttl() { +async fn test_failed_retry_suppression_respects_ttl() { let (mut scanner, temp_dir) = build_test_scanner().await; let _guard = TestGuard::new(60, 100, &mut scanner, temp_dir); let now = FolderScanner::now_secs(); @@ -470,8 +472,15 @@ async fn test_should_skip_failed_respects_ttl() { .failed_objects .insert("expired".to_string(), now.saturating_sub(120)); - assert!(scanner.should_skip_failed("recent")); - assert!(!scanner.should_skip_failed("expired")); + assert!(!scanner.mark_failed_path_seen("recent"), "first observation gets one retry"); + assert!( + scanner.mark_failed_path_seen("recent"), + "duplicate observation is eligible for TTL suppression" + ); + assert!(!scanner.mark_failed_path_seen("expired"), "expired failure is observed for a fresh retry"); + assert!(scanner.mark_failed_path_seen("expired"), "duplicate expired entry is marked in this pass"); + assert!(scanner.failed_retry_suppressed("recent")); + assert!(!scanner.failed_retry_suppressed("expired")); } #[tokio::test] @@ -485,7 +494,7 @@ async fn test_record_failed_ttl_zero_noop() { let now = FolderScanner::now_secs(); scanner.new_cache.info.failed_objects.insert("path2".to_string(), now); - assert!(!scanner.should_skip_failed("path2")); + assert!(!scanner.failed_retry_suppressed("path2")); } #[tokio::test] @@ -1105,6 +1114,72 @@ async fn test_prune_failed_objects_cache_drops_expired() { assert!(scanner.new_cache.info.failed_objects.contains_key("fresh")); } +#[tokio::test] +#[serial] +async fn test_failed_objects_clear_on_recovery_and_full_scan_reconciliation() { + let (mut scanner, temp_dir) = build_test_scanner().await; + let _guard = TestGuard::new(60, 100, &mut scanner, temp_dir); + let now = FolderScanner::now_secs(); + scanner + .new_cache + .info + .failed_objects + .insert("bucket/recovered/xl.meta".to_string(), now); + scanner + .new_cache + .info + .failed_objects + .insert("bucket/removed/xl.meta".to_string(), now); + scanner + .new_cache + .info + .failed_objects + .insert("bucket/still-failed/xl.meta".to_string(), now); + + scanner.clear_failed_path("bucket/recovered/xl.meta"); + assert!( + scanner.new_cache.info.failed_objects.contains_key("bucket/recovered/xl.meta"), + "recovery is committed only after the scan completes" + ); + scanner.mark_failed_path_seen("bucket/still-failed/xl.meta"); + scanner.reconcile_failed_objects_after_full_scan(); + + assert_eq!( + scanner + .new_cache + .info + .failed_objects + .keys() + .map(String::as_str) + .collect::>(), + ["bucket/still-failed/xl.meta"] + ); +} + +#[tokio::test] +#[serial] +async fn test_checkpoint_resume_preserves_failed_paths_outside_visited_suffix() { + let (mut scanner, temp_dir) = build_test_scanner().await; + let _guard = TestGuard::new(60, 100, &mut scanner, temp_dir); + scanner.resume_frontier = Some("bucket/resume-after".to_string()); + scanner + .new_cache + .info + .failed_objects + .insert("bucket/prefix-failure/xl.meta".to_string(), FolderScanner::now_secs()); + + scanner.reconcile_failed_objects_after_full_scan(); + + assert!( + scanner + .new_cache + .info + .failed_objects + .contains_key("bucket/prefix-failure/xl.meta"), + "a resumed suffix scan cannot prune failure state for its unvisited prefix" + ); +} + #[tokio::test] #[serial] async fn test_prune_failed_objects_max_zero_keeps_fresh() { @@ -2813,6 +2888,57 @@ async fn test_scan_data_folder_returns_partial_cache_on_budget_cancel() { assert_eq!(budget.reason(), Some(crate::scanner_budget::ScannerCycleBudgetReason::Directories)); } +#[tokio::test] +#[serial] +async fn full_scan_clears_failed_cache_for_manually_removed_object() { + let (scanner, temp_dir) = build_test_scanner().await; + let _guard = TestGuard { + temp_dir: Some(temp_dir.clone()), + }; + tokio::fs::create_dir_all(temp_dir.join("bucket")) + .await + .expect("bucket directory should be created"); + write_test_object_metadata(&temp_dir, "bucket", "repaired").await; + + let mut info = crate::data_usage_define::DataUsageCacheInfo { + name: "bucket".to_string(), + ..Default::default() + }; + let canonical_root = temp_dir + .canonicalize() + .expect("test disk root should resolve to the path used by scanner"); + info.failed_objects.insert( + canonical_root.join("bucket/removed/xl.meta").to_string_lossy().into_owned(), + FolderScanner::now_secs(), + ); + info.failed_objects.insert( + canonical_root.join("bucket/repaired/xl.meta").to_string_lossy().into_owned(), + FolderScanner::now_secs(), + ); + let cache = DataUsageCache { + info, + ..Default::default() + }; + let parent = CancellationToken::new(); + let budget = ScannerCycleBudget::new(&parent, Default::default()); + + let scanned = scan_data_folder( + budget.token(), + budget, + vec![scanner.local_disk.clone()], + scanner.local_disk.clone(), + cache, + None, + HealScanMode::Normal, + SCANNER_SLEEPER.clone(), + ) + .await + .expect("complete bucket scan should succeed"); + + assert!(scanned.info.snapshot_complete); + assert!(scanned.info.failed_objects.is_empty()); +} + #[tokio::test] #[serial] async fn test_scan_data_folder_returns_raw_cursor_on_enumeration_cancel_without_root_progress() { @@ -3486,11 +3612,12 @@ async fn test_scan_data_folder_keeps_unresolved_objects_partial() { let _guard = TestGuard { temp_dir: Some(temp_dir.clone()), }; - write_test_object_metadata(&temp_dir, "bucket", "object").await; + write_test_object_metadata_bytes(&temp_dir, "bucket", "object", &[]).await; let failed_path = temp_dir - .join("bucket") - .join("object") + .canonicalize() + .expect("test disk root should resolve to the path used by scanner") + .join("bucket/object") .join(STORAGE_FORMAT_FILE) .to_string_lossy() .into_owned(); @@ -3502,7 +3629,10 @@ async fn test_scan_data_folder_keeps_unresolved_objects_partial() { }, ..Default::default() }; - cache.info.failed_objects.insert(failed_path, FolderScanner::now_secs()); + cache + .info + .failed_objects + .insert(failed_path.clone(), FolderScanner::now_secs()); let parent = CancellationToken::new(); let budget = ScannerCycleBudget::new(&parent, Default::default()); @@ -3523,7 +3653,11 @@ async fn test_scan_data_folder_keeps_unresolved_objects_partial() { other => panic!("expected unresolved object to keep the cache partial, got {other:?}"), }; assert!(!partial.info.snapshot_complete); - assert!(!partial.info.failed_objects.is_empty()); + assert!( + partial.info.failed_objects.contains_key(&failed_path), + "corrupt object failure should remain recorded: {:?}", + partial.info.failed_objects + ); } #[tokio::test] diff --git a/crates/scanner/src/scanner_io/tests.rs b/crates/scanner/src/scanner_io/tests.rs index de1438fe8..8b776eb0c 100644 --- a/crates/scanner/src/scanner_io/tests.rs +++ b/crates/scanner/src/scanner_io/tests.rs @@ -3654,10 +3654,9 @@ async fn get_size_marks_corrupt_metadata_for_heal() { tokio::fs::create_dir_all(&object_dir) .await .expect("failed to create object directory"); - tokio::fs::write(&metadata_path, b"not-valid-filemeta") + tokio::fs::write(&metadata_path, b"") .await - .expect("failed to write corrupt metadata"); - + .expect("empty metadata fixture should be created"); let endpoint = Endpoint::try_from(temp_dir.to_string_lossy().as_ref()).expect("failed to create endpoint"); let disk = new_disk( &endpoint, @@ -3690,11 +3689,16 @@ async fn get_size_marks_corrupt_metadata_for_heal() { debug: false, }; - let err = disk - .get_size(item) - .await - .expect_err("corrupt metadata should be surfaced as scanner-heal work"); - assert!(is_scanner_metadata_corrupt_error(&err)); + for contents in [b"".as_slice(), b"not-valid-filemeta".as_slice()] { + tokio::fs::write(&metadata_path, contents) + .await + .expect("corrupt metadata fixture should be written"); + let err = disk + .get_size(item.clone()) + .await + .expect_err("corrupt metadata should be surfaced as scanner-heal work"); + assert!(is_scanner_metadata_corrupt_error(&err)); + } let _ = tokio::fs::remove_dir_all(&temp_dir).await; } diff --git a/docs/operations/durability-modes.md b/docs/operations/durability-modes.md index a543237cc..dc6b714cb 100644 --- a/docs/operations/durability-modes.md +++ b/docs/operations/durability-modes.md @@ -5,11 +5,9 @@ RustFS lets operators choose how much fsync work runs on the object write path. The default (`strict`) preserves the fully synced behavior RustFS has -always shipped; the relaxed tiers are **opt-in** trades of power-loss -durability for latency/IOPS. Note that **newly created buckets default to -`relaxed`** (see [New-bucket default](#new-bucket-default)) — a gradual -migration that leaves the process-wide default and all pre-existing buckets -on `strict`. +always shipped. `relaxed` and `none` are **opt-in** trades of power-loss +durability for latency/IOPS. Newly created buckets inherit the process-wide +mode unless `RUSTFS_NEW_BUCKET_DURABILITY_MODE` explicitly sets an override. ## Configuration @@ -21,7 +19,7 @@ RUSTFS_DURABILITY_MODE=strict|relaxed|none # default: strict RUSTFS_DRIVE_SYNC_ENABLE=true|false # default: true # Tier seeded into a NEWLY CREATED bucket's own override (see "New-bucket default") -RUSTFS_NEW_BUCKET_DURABILITY_MODE=relaxed|strict|none|inherit # default: relaxed +RUSTFS_NEW_BUCKET_DURABILITY_MODE=relaxed|strict|none|inherit # default: inherit process mode ``` Resolution rules: @@ -93,6 +91,20 @@ power simultaneously, `relaxed` can lose recently acknowledged objects cluster-wide. Single-node deployments must stay on `strict`. +## Recovery from corrupt object metadata + +A zero-length or unreadable `xl.meta` is corruption, not an absent object. +RustFS reports it as a metadata error and keeps usage snapshots incomplete; +quota admission can fail closed until a complete snapshot is available. On a +multi-node deployment, attempt heal from healthy replicas and verify the +recovered versions before cleaning any drive. If no healthy metadata copy or +backup exists, the version index cannot be reconstructed from shard data +alone. Preserve the affected directory for recovery before any operator-led +cleanup. S3 `DeleteObject` is not a physical cleanup mechanism for corrupt +metadata. After a successful repair or cleanup, a complete scanner cycle +reconciles the failed-path cache; confirm the scanner reports a completed +cycle before relying on a newly published usage snapshot. + **`none`.** No fsync on the object data path at all; acknowledged objects can vanish wholesale on power loss, payload included. System-critical writes are still pinned (below). This is the tier equivalent of the old escape hatch, @@ -149,18 +161,21 @@ configuration plane. ### New-bucket default -A newly created bucket gets a `relaxed` override **seeded into its own -metadata** at creation time, so it opts -into MinIO's default posture (object data still fdatasynced; xl.meta and -directory-entry fsyncs left to the page cache) without touching the -process-wide default. This is a gradual migration: +A newly created bucket inherits the process-wide mode by default. On an +otherwise unconfigured deployment this means `strict`. Operators can explicitly +set `RUSTFS_NEW_BUCKET_DURABILITY_MODE=relaxed` to seed a relaxed override into +new buckets, or choose `strict`, `none`, or `inherit`. A non-strict override is +logged when the bucket is created; an explicitly non-strict process-wide mode +is warned at startup. Relaxed remains appropriate only for the multi-node, +independent-power-domain deployment described above. -- **Pre-existing buckets are unaffected.** A bucket with no `durability.json` - entry keeps following the process-wide mode (`strict` by default), exactly - as before. -- **The process-wide default stays `strict`.** `RUSTFS_DURABILITY_MODE` and - `RUSTFS_DRIVE_SYNC_ENABLE` are unchanged; only newly created buckets carry - their own `relaxed` override. +- **Pre-existing buckets are unaffected by this default change.** Existing + per-bucket overrides remain stored in metadata, and buckets without an + override continue to follow the process-wide mode. +- **Explicit global configuration still applies.** If the process-wide mode is + explicitly relaxed (or the legacy full-off mode is selected), an inherited + new bucket follows that mode. Keep the process-wide mode strict on a + single-node deployment. - **System-critical namespaces stay pinned to `strict`** regardless of any override (see [System-critical pinning](#system-critical-pinning)). @@ -168,17 +183,21 @@ The seeded tier is controlled by an env var read once per bucket creation (bucket creation is not a hot path): ```bash -RUSTFS_NEW_BUCKET_DURABILITY_MODE=relaxed # default: seed `relaxed` +RUSTFS_NEW_BUCKET_DURABILITY_MODE=relaxed # explicitly seed `relaxed` RUSTFS_NEW_BUCKET_DURABILITY_MODE=strict # seed `strict` instead RUSTFS_NEW_BUCKET_DURABILITY_MODE=none # seed `none` instead RUSTFS_NEW_BUCKET_DURABILITY_MODE=inherit # seed nothing: follow the global mode ``` -`inherit` (and any unrecognized value, which fails closed to `inherit`) means -the new bucket gets no override and follows the process-wide mode. Set this -cluster-wide to opt out of the new default entirely. The seed only applies at -bucket creation; it never retroactively rewrites existing buckets. To change -an existing bucket's tier, use the per-bucket admin API above. +`inherit` (the default) and any unrecognized value, which fails closed to +`inherit`, mean the new bucket gets no override and follows the process-wide +mode. The seed only applies at bucket creation; it never retroactively rewrites +existing buckets. To correct an existing bucket created with a relaxed +override, use the admin API above to set `strict` or delete the override to +inherit the process-wide mode. Stored overrides do not record whether `relaxed` +came from an earlier default or an explicit operator choice, so RustFS does not +rewrite them automatically. Read and correct each affected bucket through the +admin API; changing the env var alone is not a migration. ### Resolution order