fix(storage): default new bucket durability to strict (#8253)

* fix(storage): align new bucket durability with strict defaults

Co-Authored-By: heihutu <heihutu@gmail.com>

Co-Authored-By: zhi22915 <qiuzgang@gmail.com>

* fix(scanner): defer failure recovery until scan completion

Co-Authored-By: heihutu <heihutu@gmail.com>

Co-Authored-By: zhi22915 <qiuzgang@gmail.com>

---------

Co-authored-by: zhi22915 <qiuzgang@gmail.com>
This commit is contained in:
Hauser
2026-09-30 09:23:32 +08:00
committed by GitHub
parent 296854c3d2
commit 498080dec4
8 changed files with 355 additions and 89 deletions
+28 -13
View File
@@ -64,25 +64,40 @@ impl BucketDurabilityConfig {
}
}
/// Default durability tier seeded into a newly created bucket's metadata
/// (rustfs/backlog#1811). `relaxed` aligns new buckets with MinIO's default
/// posture: object data is still fdatasynced, while xl.meta and directory-entry
/// fsyncs follow the relaxed durability gate.
/// Default durability policy for newly created buckets. An unset value must
/// inherit the process-wide mode so a default single-node deployment remains
/// strict. Relaxed and none remain explicit operator choices.
pub const ENV_NEW_BUCKET_DURABILITY_MODE: &str = "RUSTFS_NEW_BUCKET_DURABILITY_MODE";
pub const DEFAULT_NEW_BUCKET_DURABILITY_MODE: &str = BUCKET_DURABILITY_MODE_RELAXED;
pub const DEFAULT_NEW_BUCKET_DURABILITY_MODE: &str = "inherit";
const EVENT_NEW_BUCKET_DURABILITY_MODE: &str = "new_bucket_durability_mode";
const LOG_COMPONENT_ECSTORE: &str = "ecstore";
const LOG_SUBSYSTEM_BUCKET_DURABILITY: &str = "bucket_durability";
/// The `durability.json` bytes to seed into a freshly created bucket's metadata.
/// Empty means "no override" (the bucket then follows the global
/// `RUSTFS_DURABILITY_MODE`); otherwise the serialized chosen tier. Operators
/// can set `inherit` to disable the new-bucket override. Invalid values also
/// fail closed to inherit the global mode instead of seeding a surprising tier.
pub fn new_bucket_durability_config_json() -> Vec<u8> {
/// `RUSTFS_DURABILITY_MODE`); otherwise the serialized chosen tier. Invalid
/// values also fail closed to inherit the global mode instead of seeding a
/// surprising tier.
pub fn new_bucket_durability_config_json(bucket: &str) -> Vec<u8> {
let raw = std::env::var(ENV_NEW_BUCKET_DURABILITY_MODE).unwrap_or_else(|_| DEFAULT_NEW_BUCKET_DURABILITY_MODE.to_string());
let mode = raw.trim();
if mode.eq_ignore_ascii_case("inherit") || mode.is_empty() || !BucketDurabilityConfig::is_valid_mode(mode) {
return Vec::new();
}
serde_json::to_vec(&BucketDurabilityConfig::new(mode)).expect("BucketDurabilityConfig serialization cannot fail")
let config = BucketDurabilityConfig::new(mode);
if let Some(mode) = config.normalized_mode().filter(|mode| mode != BUCKET_DURABILITY_MODE_STRICT) {
tracing::warn!(
event = EVENT_NEW_BUCKET_DURABILITY_MODE,
component = LOG_COMPONENT_ECSTORE,
subsystem = LOG_SUBSYSTEM_BUCKET_DURABILITY,
state = "non_strict_new_bucket_override",
bucket = %bucket,
mode = %mode,
"New bucket durability is explicitly configured below strict"
);
}
serde_json::to_vec(&config).expect("BucketDurabilityConfig serialization cannot fail")
}
#[cfg(test)]
@@ -90,7 +105,7 @@ mod tests {
use super::*;
fn new_bucket_seeded_mode() -> Option<String> {
let json = new_bucket_durability_config_json();
let json = new_bucket_durability_config_json("test-bucket");
if json.is_empty() {
return None;
}
@@ -132,9 +147,9 @@ mod tests {
}
#[test]
fn new_bucket_default_seeds_relaxed_when_unset() {
fn new_bucket_default_inherits_when_unset() {
temp_env::with_var_unset(ENV_NEW_BUCKET_DURABILITY_MODE, || {
assert_eq!(new_bucket_seeded_mode().as_deref(), Some(BUCKET_DURABILITY_MODE_RELAXED));
assert_eq!(new_bucket_seeded_mode(), None);
});
}
+4 -10
View File
@@ -593,7 +593,7 @@ impl BucketMetadata {
/// durability posture.
pub fn new_with_default_durability(name: &str) -> Self {
let mut metadata = Self::new(name);
metadata.durability_config_json = super::durability::new_bucket_durability_config_json();
metadata.durability_config_json = super::durability::new_bucket_durability_config_json(name);
metadata
}
@@ -1732,21 +1732,15 @@ mod test {
}
#[test]
fn new_bucket_metadata_constructor_seeds_default_durability() {
fn new_bucket_metadata_constructor_inherits_default_durability() {
temp_env::with_var_unset(crate::bucket::durability::ENV_NEW_BUCKET_DURABILITY_MODE, || {
let metadata = BucketMetadata::new_with_default_durability("new-user-bucket");
assert_eq!(
metadata.durability_config().and_then(|cfg| cfg.normalized_mode()).as_deref(),
Some(crate::bucket::durability::BUCKET_DURABILITY_MODE_RELAXED)
);
assert!(metadata.durability_config().is_none());
let encoded = metadata.marshal_msg().expect("marshal metadata");
let decoded = BucketMetadata::unmarshal(&encoded).expect("unmarshal metadata");
assert_eq!(decoded.durability_config_json, metadata.durability_config_json);
assert_eq!(
decoded.durability_config().and_then(|cfg| cfg.normalized_mode()).as_deref(),
Some(crate::bucket::durability::BUCKET_DURABILITY_MODE_RELAXED)
);
assert!(decoded.durability_config().is_none());
});
}
+54 -8
View File
@@ -1372,13 +1372,24 @@ pub(crate) fn durability_mode() -> DurabilityMode {
rustfs_utils::get_env_opt_str(ENV_RUSTFS_DURABILITY_MODE),
rustfs_utils::get_env_bool(ENV_RUSTFS_DRIVE_SYNC_ENABLE, DEFAULT_RUSTFS_DRIVE_SYNC_ENABLE),
);
info!(
event = EVENT_DISK_LOCAL_DURABILITY_MODE,
component = LOG_COMPONENT_ECSTORE,
subsystem = LOG_SUBSYSTEM_DISK_LOCAL,
mode = mode.as_str(),
"Storage durability mode resolved"
);
if mode == DurabilityMode::Strict {
info!(
event = EVENT_DISK_LOCAL_DURABILITY_MODE,
component = LOG_COMPONENT_ECSTORE,
subsystem = LOG_SUBSYSTEM_DISK_LOCAL,
mode = mode.as_str(),
"Storage durability mode resolved"
);
} else {
warn!(
event = EVENT_DISK_LOCAL_DURABILITY_MODE,
component = LOG_COMPONENT_ECSTORE,
subsystem = LOG_SUBSYSTEM_DISK_LOCAL,
state = "non_strict_mode_configured",
mode = mode.as_str(),
"Storage durability mode does not provide strict power-loss durability"
);
}
mode
})
}
@@ -6915,7 +6926,9 @@ impl LocalDisk {
let (buf, mtime) = res?;
if buf.is_empty() {
return Err(DiskError::FileNotFound);
// A missing xl.meta is mapped by the open/read error above. A file
// that exists but has no metadata bytes is corruption, not absence.
return Err(DiskError::FileCorrupt);
}
Ok((buf, mtime))
@@ -11697,6 +11710,39 @@ mod test {
assert_eq!(err, DiskError::FileCorrupt);
}
#[tokio::test]
async fn read_version_reports_empty_xl_meta_as_corrupt_not_missing() {
use tempfile::tempdir;
let dir = tempdir().expect("test directory should be created");
let endpoint = Endpoint::try_from(dir.path().to_str().expect("test path should be utf8")).expect("endpoint should parse");
let disk = LocalDisk::new(&endpoint, false).await.expect("local disk should be created");
let bucket = "bucket";
let object = "empty-metadata";
ensure_test_volume(&disk, bucket).await;
let missing_err = disk
.read_version("", bucket, "missing-object", "", &ReadOptions::default())
.await
.expect_err("a missing xl.meta remains not found");
assert_eq!(missing_err, DiskError::FileNotFound);
let object_dir = dir.path().join(bucket).join(object);
fs::create_dir_all(&object_dir)
.await
.expect("object directory should be created");
fs::write(object_dir.join(STORAGE_FORMAT_FILE), b"")
.await
.expect("empty metadata fixture should be written");
let err = disk
.read_version("", bucket, object, "", &ReadOptions::default())
.await
.expect_err("an existing zero-length xl.meta is corrupt, not an absent object");
assert_eq!(err, DiskError::FileCorrupt);
}
#[tokio::test]
async fn read_version_delete_marker_never_enters_inline_shard_math() {
use tempfile::tempdir;
+2 -5
View File
@@ -3394,7 +3394,7 @@ mod tests {
#[tokio::test(flavor = "multi_thread")]
#[serial]
async fn make_bucket_seeds_new_bucket_durability_override() {
async fn make_bucket_inherits_default_durability() {
temp_env::async_with_vars([(crate::bucket::durability::ENV_NEW_BUCKET_DURABILITY_MODE, None::<&str>)], async {
let (_disk_paths, ecstore) = setup_bucket_delete_test_env().await;
let bucket = format!("bucket-default-durability-{}", Uuid::new_v4().simple());
@@ -3407,10 +3407,7 @@ mod tests {
let metadata = metadata_sys::get_in(&ecstore.ctx, &bucket)
.await
.expect("metadata should load for the new bucket");
assert_eq!(
metadata.durability_config().and_then(|cfg| cfg.normalized_mode()).as_deref(),
Some(crate::bucket::durability::BUCKET_DURABILITY_MODE_RELAXED)
);
assert!(metadata.durability_config().is_none());
})
.await;
}
+70 -13
View File
@@ -718,6 +718,8 @@ pub struct FolderScanner {
failed_object_ttl_secs: u64,
failed_objects_max: usize,
failed_object_paths_seen: HashSet<String>,
resolved_failed_object_paths: HashSet<String>,
sleeper: DynamicSleeper,
// should_heal: Arc<dyn Fn() -> bool + Send + Sync>,
@@ -927,7 +929,7 @@ impl FolderScanner {
into.add_child(child_hash);
}
fn should_skip_failed(&self, path: &str) -> bool {
fn failed_retry_suppressed(&self, path: &str) -> bool {
let ttl = self.failed_object_ttl_secs;
if ttl == 0 {
return false;
@@ -947,6 +949,7 @@ impl FolderScanner {
return;
}
self.resolved_failed_object_paths.remove(path);
let now = Self::now_secs();
self.new_cache.info.failed_objects.insert(path.to_string(), now);
@@ -954,6 +957,43 @@ impl FolderScanner {
if max_entries > 0 && self.new_cache.info.failed_objects.len() > max_entries {
self.prune_failed_objects(now, ttl);
}
if self.new_cache.info.failed_objects.contains_key(path) {
self.failed_object_paths_seen.insert(path.to_string());
}
}
fn mark_failed_path_seen(&mut self, path: &str) -> bool {
if self.new_cache.info.failed_objects.contains_key(path) {
return !self.failed_object_paths_seen.insert(path.to_string());
}
false
}
fn clear_failed_path(&mut self, path: &str) {
if self.new_cache.info.failed_objects.contains_key(path) {
self.resolved_failed_object_paths.insert(path.to_string());
}
self.failed_object_paths_seen.remove(path);
}
fn reconcile_failed_objects_after_full_scan(&mut self) {
self.new_cache
.info
.failed_objects
.retain(|path, _| !self.resolved_failed_object_paths.contains(path));
self.resolved_failed_object_paths.clear();
let mixed_coverage = self
.new_cache
.info
.scan_progress
.is_some_and(|progress| progress.started_plan != progress.requested_plan);
if self.prefix_scan_scope.is_some() || self.resume_frontier.is_some() || self.coverage_gap || mixed_coverage {
return;
}
let seen = &self.failed_object_paths_seen;
self.new_cache.info.failed_objects.retain(|path, _| seen.contains(path));
}
fn prune_failed_objects_cache(&mut self) {
@@ -1795,12 +1835,14 @@ impl FolderScanner {
file_type: entry_type,
};
// If this path is already known as failed, just skip it.
// We intentionally do NOT call `record_failed` or bump `failed_objects` here,
// because the failure was recorded when the original error occurred
// (e.g. in the get_size error branch below). This branch only accounts
// for subsequent skips of already-failed paths.
if self.should_skip_failed(&item.path) {
// Keep failed paths visible to this scan so repair success can
// clear stale failure state and a complete walk can discard
// paths that were removed since the previous cycle.
let repeated_failed_path = self.mark_failed_path_seen(&item.path);
if repeated_failed_path && self.failed_retry_suppressed(&item.path) {
// A new FolderScanner starts with an empty seen set, so
// cached paths are still rechecked once per cycle. Skip
// duplicate listings within this same scanner pass.
self.coverage_gap |= self.old_cache.info.scan_progress.is_some();
continue;
}
@@ -1813,14 +1855,17 @@ impl FolderScanner {
Ok(sz) => sz,
Err(e) => {
let failure_action = classify_get_size_failure(&item, &e);
let retry_suppressed = self.failed_retry_suppressed(&item.path);
if failure_action != GetSizeFailureAction::Skip {
self.resolved_failed_object_paths.remove(&item.path);
self.coverage_gap |= self.old_cache.info.scan_progress.is_some();
// Track failed objects to prevent infinite retry loops
into.failed_objects += 1;
self.record_failed(&item.path);
if !retry_suppressed {
into.failed_objects += 1;
self.record_failed(&item.path);
}
if should_log_failed_object(into.failed_objects) {
if !retry_suppressed && should_log_failed_object(into.failed_objects) {
if let GetSizeFailureAction::HealMetadata { object } = &failure_action {
error!(
target: "rustfs::scanner::folder",
@@ -1850,9 +1895,13 @@ impl FolderScanner {
);
}
}
} else {
self.clear_failed_path(&item.path);
}
if let GetSizeFailureAction::HealMetadata { object } = failure_action {
if let GetSizeFailureAction::HealMetadata { object } = failure_action
&& !retry_suppressed
{
// Single-flight (backlog#1894 axis A) — the
// recording mode and its guarantees are pinned by
// corrupt_metadata_recording below.
@@ -1901,6 +1950,7 @@ impl FolderScanner {
}
};
self.clear_failed_path(&item.path);
found_object_metadata = true;
item.transform_meta_dir();
@@ -1967,8 +2017,10 @@ impl FolderScanner {
self.coverage_gap |= self.old_cache.info.scan_progress.is_some();
found_object_metadata = true;
let metadata_path = path_join_buf(&[&dir_path, STORAGE_FORMAT_FILE]);
self.mark_failed_path_seen(&metadata_path);
let retry_suppressed = self.failed_retry_suppressed(&metadata_path);
if !self.should_skip_failed(&metadata_path) {
if !retry_suppressed {
into.failed_objects = into.failed_objects.saturating_add(1);
self.record_failed(&metadata_path);
@@ -1985,7 +2037,9 @@ impl FolderScanner {
"Scanner found erasure object data without metadata"
);
}
}
if !retry_suppressed {
let (bucket, object) = path2_bucket_object_with_base_path(&self.root, &folder.name);
if !bucket.is_empty() && !object.is_empty() {
self.send_required_scanner_heal_request(
@@ -2769,6 +2823,8 @@ pub(crate) async fn scan_data_folder_scoped(
prefix_scan_scope,
failed_object_ttl_secs: failed_object_ttl,
failed_objects_max,
failed_object_paths_seen: HashSet::new(),
resolved_failed_object_paths: HashSet::new(),
sleeper,
disks,
disks_quorum,
@@ -2823,6 +2879,7 @@ pub(crate) async fn scan_data_folder_scoped(
Ok(()) => {
// Get the new cache and finalize it
let coverage_gap = scanner.coverage_gap;
scanner.reconcile_failed_objects_after_full_scan();
let new_cache = scanner.as_mut_new_cache();
new_cache.force_compact(DATA_SCANNER_COMPACT_AT_CHILDREN);
new_cache.info.last_update = Some(SystemTime::now());
+143 -9
View File
@@ -336,6 +336,8 @@ async fn build_test_scanner() -> (FolderScanner, std::path::PathBuf) {
prefix_scan_scope: None,
failed_object_ttl_secs: u64::MAX,
failed_objects_max: usize::MAX,
failed_object_paths_seen: HashSet::new(),
resolved_failed_object_paths: HashSet::new(),
sleeper: SCANNER_SLEEPER.clone(),
disks: Vec::new(),
disks_quorum: 0,
@@ -454,7 +456,7 @@ impl Drop for TestGuard {
#[tokio::test]
#[serial]
async fn test_should_skip_failed_respects_ttl() {
async fn test_failed_retry_suppression_respects_ttl() {
let (mut scanner, temp_dir) = build_test_scanner().await;
let _guard = TestGuard::new(60, 100, &mut scanner, temp_dir);
let now = FolderScanner::now_secs();
@@ -470,8 +472,15 @@ async fn test_should_skip_failed_respects_ttl() {
.failed_objects
.insert("expired".to_string(), now.saturating_sub(120));
assert!(scanner.should_skip_failed("recent"));
assert!(!scanner.should_skip_failed("expired"));
assert!(!scanner.mark_failed_path_seen("recent"), "first observation gets one retry");
assert!(
scanner.mark_failed_path_seen("recent"),
"duplicate observation is eligible for TTL suppression"
);
assert!(!scanner.mark_failed_path_seen("expired"), "expired failure is observed for a fresh retry");
assert!(scanner.mark_failed_path_seen("expired"), "duplicate expired entry is marked in this pass");
assert!(scanner.failed_retry_suppressed("recent"));
assert!(!scanner.failed_retry_suppressed("expired"));
}
#[tokio::test]
@@ -485,7 +494,7 @@ async fn test_record_failed_ttl_zero_noop() {
let now = FolderScanner::now_secs();
scanner.new_cache.info.failed_objects.insert("path2".to_string(), now);
assert!(!scanner.should_skip_failed("path2"));
assert!(!scanner.failed_retry_suppressed("path2"));
}
#[tokio::test]
@@ -1105,6 +1114,72 @@ async fn test_prune_failed_objects_cache_drops_expired() {
assert!(scanner.new_cache.info.failed_objects.contains_key("fresh"));
}
#[tokio::test]
#[serial]
async fn test_failed_objects_clear_on_recovery_and_full_scan_reconciliation() {
let (mut scanner, temp_dir) = build_test_scanner().await;
let _guard = TestGuard::new(60, 100, &mut scanner, temp_dir);
let now = FolderScanner::now_secs();
scanner
.new_cache
.info
.failed_objects
.insert("bucket/recovered/xl.meta".to_string(), now);
scanner
.new_cache
.info
.failed_objects
.insert("bucket/removed/xl.meta".to_string(), now);
scanner
.new_cache
.info
.failed_objects
.insert("bucket/still-failed/xl.meta".to_string(), now);
scanner.clear_failed_path("bucket/recovered/xl.meta");
assert!(
scanner.new_cache.info.failed_objects.contains_key("bucket/recovered/xl.meta"),
"recovery is committed only after the scan completes"
);
scanner.mark_failed_path_seen("bucket/still-failed/xl.meta");
scanner.reconcile_failed_objects_after_full_scan();
assert_eq!(
scanner
.new_cache
.info
.failed_objects
.keys()
.map(String::as_str)
.collect::<Vec<_>>(),
["bucket/still-failed/xl.meta"]
);
}
#[tokio::test]
#[serial]
async fn test_checkpoint_resume_preserves_failed_paths_outside_visited_suffix() {
let (mut scanner, temp_dir) = build_test_scanner().await;
let _guard = TestGuard::new(60, 100, &mut scanner, temp_dir);
scanner.resume_frontier = Some("bucket/resume-after".to_string());
scanner
.new_cache
.info
.failed_objects
.insert("bucket/prefix-failure/xl.meta".to_string(), FolderScanner::now_secs());
scanner.reconcile_failed_objects_after_full_scan();
assert!(
scanner
.new_cache
.info
.failed_objects
.contains_key("bucket/prefix-failure/xl.meta"),
"a resumed suffix scan cannot prune failure state for its unvisited prefix"
);
}
#[tokio::test]
#[serial]
async fn test_prune_failed_objects_max_zero_keeps_fresh() {
@@ -2813,6 +2888,57 @@ async fn test_scan_data_folder_returns_partial_cache_on_budget_cancel() {
assert_eq!(budget.reason(), Some(crate::scanner_budget::ScannerCycleBudgetReason::Directories));
}
#[tokio::test]
#[serial]
async fn full_scan_clears_failed_cache_for_manually_removed_object() {
let (scanner, temp_dir) = build_test_scanner().await;
let _guard = TestGuard {
temp_dir: Some(temp_dir.clone()),
};
tokio::fs::create_dir_all(temp_dir.join("bucket"))
.await
.expect("bucket directory should be created");
write_test_object_metadata(&temp_dir, "bucket", "repaired").await;
let mut info = crate::data_usage_define::DataUsageCacheInfo {
name: "bucket".to_string(),
..Default::default()
};
let canonical_root = temp_dir
.canonicalize()
.expect("test disk root should resolve to the path used by scanner");
info.failed_objects.insert(
canonical_root.join("bucket/removed/xl.meta").to_string_lossy().into_owned(),
FolderScanner::now_secs(),
);
info.failed_objects.insert(
canonical_root.join("bucket/repaired/xl.meta").to_string_lossy().into_owned(),
FolderScanner::now_secs(),
);
let cache = DataUsageCache {
info,
..Default::default()
};
let parent = CancellationToken::new();
let budget = ScannerCycleBudget::new(&parent, Default::default());
let scanned = scan_data_folder(
budget.token(),
budget,
vec![scanner.local_disk.clone()],
scanner.local_disk.clone(),
cache,
None,
HealScanMode::Normal,
SCANNER_SLEEPER.clone(),
)
.await
.expect("complete bucket scan should succeed");
assert!(scanned.info.snapshot_complete);
assert!(scanned.info.failed_objects.is_empty());
}
#[tokio::test]
#[serial]
async fn test_scan_data_folder_returns_raw_cursor_on_enumeration_cancel_without_root_progress() {
@@ -3486,11 +3612,12 @@ async fn test_scan_data_folder_keeps_unresolved_objects_partial() {
let _guard = TestGuard {
temp_dir: Some(temp_dir.clone()),
};
write_test_object_metadata(&temp_dir, "bucket", "object").await;
write_test_object_metadata_bytes(&temp_dir, "bucket", "object", &[]).await;
let failed_path = temp_dir
.join("bucket")
.join("object")
.canonicalize()
.expect("test disk root should resolve to the path used by scanner")
.join("bucket/object")
.join(STORAGE_FORMAT_FILE)
.to_string_lossy()
.into_owned();
@@ -3502,7 +3629,10 @@ async fn test_scan_data_folder_keeps_unresolved_objects_partial() {
},
..Default::default()
};
cache.info.failed_objects.insert(failed_path, FolderScanner::now_secs());
cache
.info
.failed_objects
.insert(failed_path.clone(), FolderScanner::now_secs());
let parent = CancellationToken::new();
let budget = ScannerCycleBudget::new(&parent, Default::default());
@@ -3523,7 +3653,11 @@ async fn test_scan_data_folder_keeps_unresolved_objects_partial() {
other => panic!("expected unresolved object to keep the cache partial, got {other:?}"),
};
assert!(!partial.info.snapshot_complete);
assert!(!partial.info.failed_objects.is_empty());
assert!(
partial.info.failed_objects.contains_key(&failed_path),
"corrupt object failure should remain recorded: {:?}",
partial.info.failed_objects
);
}
#[tokio::test]
+12 -8
View File
@@ -3654,10 +3654,9 @@ async fn get_size_marks_corrupt_metadata_for_heal() {
tokio::fs::create_dir_all(&object_dir)
.await
.expect("failed to create object directory");
tokio::fs::write(&metadata_path, b"not-valid-filemeta")
tokio::fs::write(&metadata_path, b"")
.await
.expect("failed to write corrupt metadata");
.expect("empty metadata fixture should be created");
let endpoint = Endpoint::try_from(temp_dir.to_string_lossy().as_ref()).expect("failed to create endpoint");
let disk = new_disk(
&endpoint,
@@ -3690,11 +3689,16 @@ async fn get_size_marks_corrupt_metadata_for_heal() {
debug: false,
};
let err = disk
.get_size(item)
.await
.expect_err("corrupt metadata should be surfaced as scanner-heal work");
assert!(is_scanner_metadata_corrupt_error(&err));
for contents in [b"".as_slice(), b"not-valid-filemeta".as_slice()] {
tokio::fs::write(&metadata_path, contents)
.await
.expect("corrupt metadata fixture should be written");
let err = disk
.get_size(item.clone())
.await
.expect_err("corrupt metadata should be surfaced as scanner-heal work");
assert!(is_scanner_metadata_corrupt_error(&err));
}
let _ = tokio::fs::remove_dir_all(&temp_dir).await;
}
+42 -23
View File
@@ -5,11 +5,9 @@
RustFS lets operators choose how much fsync work runs on the object write
path. The default (`strict`) preserves the fully synced behavior RustFS has
always shipped; the relaxed tiers are **opt-in** trades of power-loss
durability for latency/IOPS. Note that **newly created buckets default to
`relaxed`** (see [New-bucket default](#new-bucket-default)) — a gradual
migration that leaves the process-wide default and all pre-existing buckets
on `strict`.
always shipped. `relaxed` and `none` are **opt-in** trades of power-loss
durability for latency/IOPS. Newly created buckets inherit the process-wide
mode unless `RUSTFS_NEW_BUCKET_DURABILITY_MODE` explicitly sets an override.
## Configuration
@@ -21,7 +19,7 @@ RUSTFS_DURABILITY_MODE=strict|relaxed|none # default: strict
RUSTFS_DRIVE_SYNC_ENABLE=true|false # default: true
# Tier seeded into a NEWLY CREATED bucket's own override (see "New-bucket default")
RUSTFS_NEW_BUCKET_DURABILITY_MODE=relaxed|strict|none|inherit # default: relaxed
RUSTFS_NEW_BUCKET_DURABILITY_MODE=relaxed|strict|none|inherit # default: inherit process mode
```
Resolution rules:
@@ -93,6 +91,20 @@ power simultaneously, `relaxed` can lose recently acknowledged objects
cluster-wide. Single-node
deployments must stay on `strict`.
## Recovery from corrupt object metadata
A zero-length or unreadable `xl.meta` is corruption, not an absent object.
RustFS reports it as a metadata error and keeps usage snapshots incomplete;
quota admission can fail closed until a complete snapshot is available. On a
multi-node deployment, attempt heal from healthy replicas and verify the
recovered versions before cleaning any drive. If no healthy metadata copy or
backup exists, the version index cannot be reconstructed from shard data
alone. Preserve the affected directory for recovery before any operator-led
cleanup. S3 `DeleteObject` is not a physical cleanup mechanism for corrupt
metadata. After a successful repair or cleanup, a complete scanner cycle
reconciles the failed-path cache; confirm the scanner reports a completed
cycle before relying on a newly published usage snapshot.
**`none`.** No fsync on the object data path at all; acknowledged objects can
vanish wholesale on power loss, payload included. System-critical writes are
still pinned (below). This is the tier equivalent of the old escape hatch,
@@ -149,18 +161,21 @@ configuration plane.
### New-bucket default
A newly created bucket gets a `relaxed` override **seeded into its own
metadata** at creation time, so it opts
into MinIO's default posture (object data still fdatasynced; xl.meta and
directory-entry fsyncs left to the page cache) without touching the
process-wide default. This is a gradual migration:
A newly created bucket inherits the process-wide mode by default. On an
otherwise unconfigured deployment this means `strict`. Operators can explicitly
set `RUSTFS_NEW_BUCKET_DURABILITY_MODE=relaxed` to seed a relaxed override into
new buckets, or choose `strict`, `none`, or `inherit`. A non-strict override is
logged when the bucket is created; an explicitly non-strict process-wide mode
is warned at startup. Relaxed remains appropriate only for the multi-node,
independent-power-domain deployment described above.
- **Pre-existing buckets are unaffected.** A bucket with no `durability.json`
entry keeps following the process-wide mode (`strict` by default), exactly
as before.
- **The process-wide default stays `strict`.** `RUSTFS_DURABILITY_MODE` and
`RUSTFS_DRIVE_SYNC_ENABLE` are unchanged; only newly created buckets carry
their own `relaxed` override.
- **Pre-existing buckets are unaffected by this default change.** Existing
per-bucket overrides remain stored in metadata, and buckets without an
override continue to follow the process-wide mode.
- **Explicit global configuration still applies.** If the process-wide mode is
explicitly relaxed (or the legacy full-off mode is selected), an inherited
new bucket follows that mode. Keep the process-wide mode strict on a
single-node deployment.
- **System-critical namespaces stay pinned to `strict`** regardless of any
override (see [System-critical pinning](#system-critical-pinning)).
@@ -168,17 +183,21 @@ The seeded tier is controlled by an env var read once per bucket creation
(bucket creation is not a hot path):
```bash
RUSTFS_NEW_BUCKET_DURABILITY_MODE=relaxed # default: seed `relaxed`
RUSTFS_NEW_BUCKET_DURABILITY_MODE=relaxed # explicitly seed `relaxed`
RUSTFS_NEW_BUCKET_DURABILITY_MODE=strict # seed `strict` instead
RUSTFS_NEW_BUCKET_DURABILITY_MODE=none # seed `none` instead
RUSTFS_NEW_BUCKET_DURABILITY_MODE=inherit # seed nothing: follow the global mode
```
`inherit` (and any unrecognized value, which fails closed to `inherit`) means
the new bucket gets no override and follows the process-wide mode. Set this
cluster-wide to opt out of the new default entirely. The seed only applies at
bucket creation; it never retroactively rewrites existing buckets. To change
an existing bucket's tier, use the per-bucket admin API above.
`inherit` (the default) and any unrecognized value, which fails closed to
`inherit`, mean the new bucket gets no override and follows the process-wide
mode. The seed only applies at bucket creation; it never retroactively rewrites
existing buckets. To correct an existing bucket created with a relaxed
override, use the admin API above to set `strict` or delete the override to
inherit the process-wide mode. Stored overrides do not record whether `relaxed`
came from an earlier default or an explicit operator choice, so RustFS does not
rewrite them automatically. Read and correct each affected bucket through the
admin API; changing the env var alone is not a migration.
### Resolution order