mirror of
https://github.com/rustfs/rustfs.git
synced 2026-10-06 13:13:03 +00:00
fix(storage): default new bucket durability to strict (#8253)
* fix(storage): align new bucket durability with strict defaults Co-Authored-By: heihutu <heihutu@gmail.com> Co-Authored-By: zhi22915 <qiuzgang@gmail.com> * fix(scanner): defer failure recovery until scan completion Co-Authored-By: heihutu <heihutu@gmail.com> Co-Authored-By: zhi22915 <qiuzgang@gmail.com> --------- Co-authored-by: zhi22915 <qiuzgang@gmail.com>
This commit is contained in:
@@ -64,25 +64,40 @@ impl BucketDurabilityConfig {
|
||||
}
|
||||
}
|
||||
|
||||
/// Default durability tier seeded into a newly created bucket's metadata
|
||||
/// (rustfs/backlog#1811). `relaxed` aligns new buckets with MinIO's default
|
||||
/// posture: object data is still fdatasynced, while xl.meta and directory-entry
|
||||
/// fsyncs follow the relaxed durability gate.
|
||||
/// Default durability policy for newly created buckets. An unset value must
|
||||
/// inherit the process-wide mode so a default single-node deployment remains
|
||||
/// strict. Relaxed and none remain explicit operator choices.
|
||||
pub const ENV_NEW_BUCKET_DURABILITY_MODE: &str = "RUSTFS_NEW_BUCKET_DURABILITY_MODE";
|
||||
pub const DEFAULT_NEW_BUCKET_DURABILITY_MODE: &str = BUCKET_DURABILITY_MODE_RELAXED;
|
||||
pub const DEFAULT_NEW_BUCKET_DURABILITY_MODE: &str = "inherit";
|
||||
|
||||
const EVENT_NEW_BUCKET_DURABILITY_MODE: &str = "new_bucket_durability_mode";
|
||||
const LOG_COMPONENT_ECSTORE: &str = "ecstore";
|
||||
const LOG_SUBSYSTEM_BUCKET_DURABILITY: &str = "bucket_durability";
|
||||
|
||||
/// The `durability.json` bytes to seed into a freshly created bucket's metadata.
|
||||
/// Empty means "no override" (the bucket then follows the global
|
||||
/// `RUSTFS_DURABILITY_MODE`); otherwise the serialized chosen tier. Operators
|
||||
/// can set `inherit` to disable the new-bucket override. Invalid values also
|
||||
/// fail closed to inherit the global mode instead of seeding a surprising tier.
|
||||
pub fn new_bucket_durability_config_json() -> Vec<u8> {
|
||||
/// `RUSTFS_DURABILITY_MODE`); otherwise the serialized chosen tier. Invalid
|
||||
/// values also fail closed to inherit the global mode instead of seeding a
|
||||
/// surprising tier.
|
||||
pub fn new_bucket_durability_config_json(bucket: &str) -> Vec<u8> {
|
||||
let raw = std::env::var(ENV_NEW_BUCKET_DURABILITY_MODE).unwrap_or_else(|_| DEFAULT_NEW_BUCKET_DURABILITY_MODE.to_string());
|
||||
let mode = raw.trim();
|
||||
if mode.eq_ignore_ascii_case("inherit") || mode.is_empty() || !BucketDurabilityConfig::is_valid_mode(mode) {
|
||||
return Vec::new();
|
||||
}
|
||||
serde_json::to_vec(&BucketDurabilityConfig::new(mode)).expect("BucketDurabilityConfig serialization cannot fail")
|
||||
let config = BucketDurabilityConfig::new(mode);
|
||||
if let Some(mode) = config.normalized_mode().filter(|mode| mode != BUCKET_DURABILITY_MODE_STRICT) {
|
||||
tracing::warn!(
|
||||
event = EVENT_NEW_BUCKET_DURABILITY_MODE,
|
||||
component = LOG_COMPONENT_ECSTORE,
|
||||
subsystem = LOG_SUBSYSTEM_BUCKET_DURABILITY,
|
||||
state = "non_strict_new_bucket_override",
|
||||
bucket = %bucket,
|
||||
mode = %mode,
|
||||
"New bucket durability is explicitly configured below strict"
|
||||
);
|
||||
}
|
||||
serde_json::to_vec(&config).expect("BucketDurabilityConfig serialization cannot fail")
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
@@ -90,7 +105,7 @@ mod tests {
|
||||
use super::*;
|
||||
|
||||
fn new_bucket_seeded_mode() -> Option<String> {
|
||||
let json = new_bucket_durability_config_json();
|
||||
let json = new_bucket_durability_config_json("test-bucket");
|
||||
if json.is_empty() {
|
||||
return None;
|
||||
}
|
||||
@@ -132,9 +147,9 @@ mod tests {
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn new_bucket_default_seeds_relaxed_when_unset() {
|
||||
fn new_bucket_default_inherits_when_unset() {
|
||||
temp_env::with_var_unset(ENV_NEW_BUCKET_DURABILITY_MODE, || {
|
||||
assert_eq!(new_bucket_seeded_mode().as_deref(), Some(BUCKET_DURABILITY_MODE_RELAXED));
|
||||
assert_eq!(new_bucket_seeded_mode(), None);
|
||||
});
|
||||
}
|
||||
|
||||
|
||||
@@ -593,7 +593,7 @@ impl BucketMetadata {
|
||||
/// durability posture.
|
||||
pub fn new_with_default_durability(name: &str) -> Self {
|
||||
let mut metadata = Self::new(name);
|
||||
metadata.durability_config_json = super::durability::new_bucket_durability_config_json();
|
||||
metadata.durability_config_json = super::durability::new_bucket_durability_config_json(name);
|
||||
metadata
|
||||
}
|
||||
|
||||
@@ -1732,21 +1732,15 @@ mod test {
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn new_bucket_metadata_constructor_seeds_default_durability() {
|
||||
fn new_bucket_metadata_constructor_inherits_default_durability() {
|
||||
temp_env::with_var_unset(crate::bucket::durability::ENV_NEW_BUCKET_DURABILITY_MODE, || {
|
||||
let metadata = BucketMetadata::new_with_default_durability("new-user-bucket");
|
||||
assert_eq!(
|
||||
metadata.durability_config().and_then(|cfg| cfg.normalized_mode()).as_deref(),
|
||||
Some(crate::bucket::durability::BUCKET_DURABILITY_MODE_RELAXED)
|
||||
);
|
||||
assert!(metadata.durability_config().is_none());
|
||||
|
||||
let encoded = metadata.marshal_msg().expect("marshal metadata");
|
||||
let decoded = BucketMetadata::unmarshal(&encoded).expect("unmarshal metadata");
|
||||
assert_eq!(decoded.durability_config_json, metadata.durability_config_json);
|
||||
assert_eq!(
|
||||
decoded.durability_config().and_then(|cfg| cfg.normalized_mode()).as_deref(),
|
||||
Some(crate::bucket::durability::BUCKET_DURABILITY_MODE_RELAXED)
|
||||
);
|
||||
assert!(decoded.durability_config().is_none());
|
||||
});
|
||||
}
|
||||
|
||||
|
||||
@@ -1372,13 +1372,24 @@ pub(crate) fn durability_mode() -> DurabilityMode {
|
||||
rustfs_utils::get_env_opt_str(ENV_RUSTFS_DURABILITY_MODE),
|
||||
rustfs_utils::get_env_bool(ENV_RUSTFS_DRIVE_SYNC_ENABLE, DEFAULT_RUSTFS_DRIVE_SYNC_ENABLE),
|
||||
);
|
||||
info!(
|
||||
event = EVENT_DISK_LOCAL_DURABILITY_MODE,
|
||||
component = LOG_COMPONENT_ECSTORE,
|
||||
subsystem = LOG_SUBSYSTEM_DISK_LOCAL,
|
||||
mode = mode.as_str(),
|
||||
"Storage durability mode resolved"
|
||||
);
|
||||
if mode == DurabilityMode::Strict {
|
||||
info!(
|
||||
event = EVENT_DISK_LOCAL_DURABILITY_MODE,
|
||||
component = LOG_COMPONENT_ECSTORE,
|
||||
subsystem = LOG_SUBSYSTEM_DISK_LOCAL,
|
||||
mode = mode.as_str(),
|
||||
"Storage durability mode resolved"
|
||||
);
|
||||
} else {
|
||||
warn!(
|
||||
event = EVENT_DISK_LOCAL_DURABILITY_MODE,
|
||||
component = LOG_COMPONENT_ECSTORE,
|
||||
subsystem = LOG_SUBSYSTEM_DISK_LOCAL,
|
||||
state = "non_strict_mode_configured",
|
||||
mode = mode.as_str(),
|
||||
"Storage durability mode does not provide strict power-loss durability"
|
||||
);
|
||||
}
|
||||
mode
|
||||
})
|
||||
}
|
||||
@@ -6915,7 +6926,9 @@ impl LocalDisk {
|
||||
|
||||
let (buf, mtime) = res?;
|
||||
if buf.is_empty() {
|
||||
return Err(DiskError::FileNotFound);
|
||||
// A missing xl.meta is mapped by the open/read error above. A file
|
||||
// that exists but has no metadata bytes is corruption, not absence.
|
||||
return Err(DiskError::FileCorrupt);
|
||||
}
|
||||
|
||||
Ok((buf, mtime))
|
||||
@@ -11697,6 +11710,39 @@ mod test {
|
||||
assert_eq!(err, DiskError::FileCorrupt);
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
async fn read_version_reports_empty_xl_meta_as_corrupt_not_missing() {
|
||||
use tempfile::tempdir;
|
||||
|
||||
let dir = tempdir().expect("test directory should be created");
|
||||
let endpoint = Endpoint::try_from(dir.path().to_str().expect("test path should be utf8")).expect("endpoint should parse");
|
||||
let disk = LocalDisk::new(&endpoint, false).await.expect("local disk should be created");
|
||||
let bucket = "bucket";
|
||||
let object = "empty-metadata";
|
||||
ensure_test_volume(&disk, bucket).await;
|
||||
|
||||
let missing_err = disk
|
||||
.read_version("", bucket, "missing-object", "", &ReadOptions::default())
|
||||
.await
|
||||
.expect_err("a missing xl.meta remains not found");
|
||||
assert_eq!(missing_err, DiskError::FileNotFound);
|
||||
|
||||
let object_dir = dir.path().join(bucket).join(object);
|
||||
fs::create_dir_all(&object_dir)
|
||||
.await
|
||||
.expect("object directory should be created");
|
||||
fs::write(object_dir.join(STORAGE_FORMAT_FILE), b"")
|
||||
.await
|
||||
.expect("empty metadata fixture should be written");
|
||||
|
||||
let err = disk
|
||||
.read_version("", bucket, object, "", &ReadOptions::default())
|
||||
.await
|
||||
.expect_err("an existing zero-length xl.meta is corrupt, not an absent object");
|
||||
|
||||
assert_eq!(err, DiskError::FileCorrupt);
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
async fn read_version_delete_marker_never_enters_inline_shard_math() {
|
||||
use tempfile::tempdir;
|
||||
|
||||
@@ -3394,7 +3394,7 @@ mod tests {
|
||||
|
||||
#[tokio::test(flavor = "multi_thread")]
|
||||
#[serial]
|
||||
async fn make_bucket_seeds_new_bucket_durability_override() {
|
||||
async fn make_bucket_inherits_default_durability() {
|
||||
temp_env::async_with_vars([(crate::bucket::durability::ENV_NEW_BUCKET_DURABILITY_MODE, None::<&str>)], async {
|
||||
let (_disk_paths, ecstore) = setup_bucket_delete_test_env().await;
|
||||
let bucket = format!("bucket-default-durability-{}", Uuid::new_v4().simple());
|
||||
@@ -3407,10 +3407,7 @@ mod tests {
|
||||
let metadata = metadata_sys::get_in(&ecstore.ctx, &bucket)
|
||||
.await
|
||||
.expect("metadata should load for the new bucket");
|
||||
assert_eq!(
|
||||
metadata.durability_config().and_then(|cfg| cfg.normalized_mode()).as_deref(),
|
||||
Some(crate::bucket::durability::BUCKET_DURABILITY_MODE_RELAXED)
|
||||
);
|
||||
assert!(metadata.durability_config().is_none());
|
||||
})
|
||||
.await;
|
||||
}
|
||||
|
||||
@@ -718,6 +718,8 @@ pub struct FolderScanner {
|
||||
|
||||
failed_object_ttl_secs: u64,
|
||||
failed_objects_max: usize,
|
||||
failed_object_paths_seen: HashSet<String>,
|
||||
resolved_failed_object_paths: HashSet<String>,
|
||||
|
||||
sleeper: DynamicSleeper,
|
||||
// should_heal: Arc<dyn Fn() -> bool + Send + Sync>,
|
||||
@@ -927,7 +929,7 @@ impl FolderScanner {
|
||||
into.add_child(child_hash);
|
||||
}
|
||||
|
||||
fn should_skip_failed(&self, path: &str) -> bool {
|
||||
fn failed_retry_suppressed(&self, path: &str) -> bool {
|
||||
let ttl = self.failed_object_ttl_secs;
|
||||
if ttl == 0 {
|
||||
return false;
|
||||
@@ -947,6 +949,7 @@ impl FolderScanner {
|
||||
return;
|
||||
}
|
||||
|
||||
self.resolved_failed_object_paths.remove(path);
|
||||
let now = Self::now_secs();
|
||||
self.new_cache.info.failed_objects.insert(path.to_string(), now);
|
||||
|
||||
@@ -954,6 +957,43 @@ impl FolderScanner {
|
||||
if max_entries > 0 && self.new_cache.info.failed_objects.len() > max_entries {
|
||||
self.prune_failed_objects(now, ttl);
|
||||
}
|
||||
if self.new_cache.info.failed_objects.contains_key(path) {
|
||||
self.failed_object_paths_seen.insert(path.to_string());
|
||||
}
|
||||
}
|
||||
|
||||
fn mark_failed_path_seen(&mut self, path: &str) -> bool {
|
||||
if self.new_cache.info.failed_objects.contains_key(path) {
|
||||
return !self.failed_object_paths_seen.insert(path.to_string());
|
||||
}
|
||||
false
|
||||
}
|
||||
|
||||
fn clear_failed_path(&mut self, path: &str) {
|
||||
if self.new_cache.info.failed_objects.contains_key(path) {
|
||||
self.resolved_failed_object_paths.insert(path.to_string());
|
||||
}
|
||||
self.failed_object_paths_seen.remove(path);
|
||||
}
|
||||
|
||||
fn reconcile_failed_objects_after_full_scan(&mut self) {
|
||||
self.new_cache
|
||||
.info
|
||||
.failed_objects
|
||||
.retain(|path, _| !self.resolved_failed_object_paths.contains(path));
|
||||
self.resolved_failed_object_paths.clear();
|
||||
|
||||
let mixed_coverage = self
|
||||
.new_cache
|
||||
.info
|
||||
.scan_progress
|
||||
.is_some_and(|progress| progress.started_plan != progress.requested_plan);
|
||||
if self.prefix_scan_scope.is_some() || self.resume_frontier.is_some() || self.coverage_gap || mixed_coverage {
|
||||
return;
|
||||
}
|
||||
|
||||
let seen = &self.failed_object_paths_seen;
|
||||
self.new_cache.info.failed_objects.retain(|path, _| seen.contains(path));
|
||||
}
|
||||
|
||||
fn prune_failed_objects_cache(&mut self) {
|
||||
@@ -1795,12 +1835,14 @@ impl FolderScanner {
|
||||
file_type: entry_type,
|
||||
};
|
||||
|
||||
// If this path is already known as failed, just skip it.
|
||||
// We intentionally do NOT call `record_failed` or bump `failed_objects` here,
|
||||
// because the failure was recorded when the original error occurred
|
||||
// (e.g. in the get_size error branch below). This branch only accounts
|
||||
// for subsequent skips of already-failed paths.
|
||||
if self.should_skip_failed(&item.path) {
|
||||
// Keep failed paths visible to this scan so repair success can
|
||||
// clear stale failure state and a complete walk can discard
|
||||
// paths that were removed since the previous cycle.
|
||||
let repeated_failed_path = self.mark_failed_path_seen(&item.path);
|
||||
if repeated_failed_path && self.failed_retry_suppressed(&item.path) {
|
||||
// A new FolderScanner starts with an empty seen set, so
|
||||
// cached paths are still rechecked once per cycle. Skip
|
||||
// duplicate listings within this same scanner pass.
|
||||
self.coverage_gap |= self.old_cache.info.scan_progress.is_some();
|
||||
continue;
|
||||
}
|
||||
@@ -1813,14 +1855,17 @@ impl FolderScanner {
|
||||
Ok(sz) => sz,
|
||||
Err(e) => {
|
||||
let failure_action = classify_get_size_failure(&item, &e);
|
||||
let retry_suppressed = self.failed_retry_suppressed(&item.path);
|
||||
|
||||
if failure_action != GetSizeFailureAction::Skip {
|
||||
self.resolved_failed_object_paths.remove(&item.path);
|
||||
self.coverage_gap |= self.old_cache.info.scan_progress.is_some();
|
||||
// Track failed objects to prevent infinite retry loops
|
||||
into.failed_objects += 1;
|
||||
self.record_failed(&item.path);
|
||||
if !retry_suppressed {
|
||||
into.failed_objects += 1;
|
||||
self.record_failed(&item.path);
|
||||
}
|
||||
|
||||
if should_log_failed_object(into.failed_objects) {
|
||||
if !retry_suppressed && should_log_failed_object(into.failed_objects) {
|
||||
if let GetSizeFailureAction::HealMetadata { object } = &failure_action {
|
||||
error!(
|
||||
target: "rustfs::scanner::folder",
|
||||
@@ -1850,9 +1895,13 @@ impl FolderScanner {
|
||||
);
|
||||
}
|
||||
}
|
||||
} else {
|
||||
self.clear_failed_path(&item.path);
|
||||
}
|
||||
|
||||
if let GetSizeFailureAction::HealMetadata { object } = failure_action {
|
||||
if let GetSizeFailureAction::HealMetadata { object } = failure_action
|
||||
&& !retry_suppressed
|
||||
{
|
||||
// Single-flight (backlog#1894 axis A) — the
|
||||
// recording mode and its guarantees are pinned by
|
||||
// corrupt_metadata_recording below.
|
||||
@@ -1901,6 +1950,7 @@ impl FolderScanner {
|
||||
}
|
||||
};
|
||||
|
||||
self.clear_failed_path(&item.path);
|
||||
found_object_metadata = true;
|
||||
|
||||
item.transform_meta_dir();
|
||||
@@ -1967,8 +2017,10 @@ impl FolderScanner {
|
||||
self.coverage_gap |= self.old_cache.info.scan_progress.is_some();
|
||||
found_object_metadata = true;
|
||||
let metadata_path = path_join_buf(&[&dir_path, STORAGE_FORMAT_FILE]);
|
||||
self.mark_failed_path_seen(&metadata_path);
|
||||
let retry_suppressed = self.failed_retry_suppressed(&metadata_path);
|
||||
|
||||
if !self.should_skip_failed(&metadata_path) {
|
||||
if !retry_suppressed {
|
||||
into.failed_objects = into.failed_objects.saturating_add(1);
|
||||
self.record_failed(&metadata_path);
|
||||
|
||||
@@ -1985,7 +2037,9 @@ impl FolderScanner {
|
||||
"Scanner found erasure object data without metadata"
|
||||
);
|
||||
}
|
||||
}
|
||||
|
||||
if !retry_suppressed {
|
||||
let (bucket, object) = path2_bucket_object_with_base_path(&self.root, &folder.name);
|
||||
if !bucket.is_empty() && !object.is_empty() {
|
||||
self.send_required_scanner_heal_request(
|
||||
@@ -2769,6 +2823,8 @@ pub(crate) async fn scan_data_folder_scoped(
|
||||
prefix_scan_scope,
|
||||
failed_object_ttl_secs: failed_object_ttl,
|
||||
failed_objects_max,
|
||||
failed_object_paths_seen: HashSet::new(),
|
||||
resolved_failed_object_paths: HashSet::new(),
|
||||
sleeper,
|
||||
disks,
|
||||
disks_quorum,
|
||||
@@ -2823,6 +2879,7 @@ pub(crate) async fn scan_data_folder_scoped(
|
||||
Ok(()) => {
|
||||
// Get the new cache and finalize it
|
||||
let coverage_gap = scanner.coverage_gap;
|
||||
scanner.reconcile_failed_objects_after_full_scan();
|
||||
let new_cache = scanner.as_mut_new_cache();
|
||||
new_cache.force_compact(DATA_SCANNER_COMPACT_AT_CHILDREN);
|
||||
new_cache.info.last_update = Some(SystemTime::now());
|
||||
|
||||
@@ -336,6 +336,8 @@ async fn build_test_scanner() -> (FolderScanner, std::path::PathBuf) {
|
||||
prefix_scan_scope: None,
|
||||
failed_object_ttl_secs: u64::MAX,
|
||||
failed_objects_max: usize::MAX,
|
||||
failed_object_paths_seen: HashSet::new(),
|
||||
resolved_failed_object_paths: HashSet::new(),
|
||||
sleeper: SCANNER_SLEEPER.clone(),
|
||||
disks: Vec::new(),
|
||||
disks_quorum: 0,
|
||||
@@ -454,7 +456,7 @@ impl Drop for TestGuard {
|
||||
|
||||
#[tokio::test]
|
||||
#[serial]
|
||||
async fn test_should_skip_failed_respects_ttl() {
|
||||
async fn test_failed_retry_suppression_respects_ttl() {
|
||||
let (mut scanner, temp_dir) = build_test_scanner().await;
|
||||
let _guard = TestGuard::new(60, 100, &mut scanner, temp_dir);
|
||||
let now = FolderScanner::now_secs();
|
||||
@@ -470,8 +472,15 @@ async fn test_should_skip_failed_respects_ttl() {
|
||||
.failed_objects
|
||||
.insert("expired".to_string(), now.saturating_sub(120));
|
||||
|
||||
assert!(scanner.should_skip_failed("recent"));
|
||||
assert!(!scanner.should_skip_failed("expired"));
|
||||
assert!(!scanner.mark_failed_path_seen("recent"), "first observation gets one retry");
|
||||
assert!(
|
||||
scanner.mark_failed_path_seen("recent"),
|
||||
"duplicate observation is eligible for TTL suppression"
|
||||
);
|
||||
assert!(!scanner.mark_failed_path_seen("expired"), "expired failure is observed for a fresh retry");
|
||||
assert!(scanner.mark_failed_path_seen("expired"), "duplicate expired entry is marked in this pass");
|
||||
assert!(scanner.failed_retry_suppressed("recent"));
|
||||
assert!(!scanner.failed_retry_suppressed("expired"));
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
@@ -485,7 +494,7 @@ async fn test_record_failed_ttl_zero_noop() {
|
||||
|
||||
let now = FolderScanner::now_secs();
|
||||
scanner.new_cache.info.failed_objects.insert("path2".to_string(), now);
|
||||
assert!(!scanner.should_skip_failed("path2"));
|
||||
assert!(!scanner.failed_retry_suppressed("path2"));
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
@@ -1105,6 +1114,72 @@ async fn test_prune_failed_objects_cache_drops_expired() {
|
||||
assert!(scanner.new_cache.info.failed_objects.contains_key("fresh"));
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
#[serial]
|
||||
async fn test_failed_objects_clear_on_recovery_and_full_scan_reconciliation() {
|
||||
let (mut scanner, temp_dir) = build_test_scanner().await;
|
||||
let _guard = TestGuard::new(60, 100, &mut scanner, temp_dir);
|
||||
let now = FolderScanner::now_secs();
|
||||
scanner
|
||||
.new_cache
|
||||
.info
|
||||
.failed_objects
|
||||
.insert("bucket/recovered/xl.meta".to_string(), now);
|
||||
scanner
|
||||
.new_cache
|
||||
.info
|
||||
.failed_objects
|
||||
.insert("bucket/removed/xl.meta".to_string(), now);
|
||||
scanner
|
||||
.new_cache
|
||||
.info
|
||||
.failed_objects
|
||||
.insert("bucket/still-failed/xl.meta".to_string(), now);
|
||||
|
||||
scanner.clear_failed_path("bucket/recovered/xl.meta");
|
||||
assert!(
|
||||
scanner.new_cache.info.failed_objects.contains_key("bucket/recovered/xl.meta"),
|
||||
"recovery is committed only after the scan completes"
|
||||
);
|
||||
scanner.mark_failed_path_seen("bucket/still-failed/xl.meta");
|
||||
scanner.reconcile_failed_objects_after_full_scan();
|
||||
|
||||
assert_eq!(
|
||||
scanner
|
||||
.new_cache
|
||||
.info
|
||||
.failed_objects
|
||||
.keys()
|
||||
.map(String::as_str)
|
||||
.collect::<Vec<_>>(),
|
||||
["bucket/still-failed/xl.meta"]
|
||||
);
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
#[serial]
|
||||
async fn test_checkpoint_resume_preserves_failed_paths_outside_visited_suffix() {
|
||||
let (mut scanner, temp_dir) = build_test_scanner().await;
|
||||
let _guard = TestGuard::new(60, 100, &mut scanner, temp_dir);
|
||||
scanner.resume_frontier = Some("bucket/resume-after".to_string());
|
||||
scanner
|
||||
.new_cache
|
||||
.info
|
||||
.failed_objects
|
||||
.insert("bucket/prefix-failure/xl.meta".to_string(), FolderScanner::now_secs());
|
||||
|
||||
scanner.reconcile_failed_objects_after_full_scan();
|
||||
|
||||
assert!(
|
||||
scanner
|
||||
.new_cache
|
||||
.info
|
||||
.failed_objects
|
||||
.contains_key("bucket/prefix-failure/xl.meta"),
|
||||
"a resumed suffix scan cannot prune failure state for its unvisited prefix"
|
||||
);
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
#[serial]
|
||||
async fn test_prune_failed_objects_max_zero_keeps_fresh() {
|
||||
@@ -2813,6 +2888,57 @@ async fn test_scan_data_folder_returns_partial_cache_on_budget_cancel() {
|
||||
assert_eq!(budget.reason(), Some(crate::scanner_budget::ScannerCycleBudgetReason::Directories));
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
#[serial]
|
||||
async fn full_scan_clears_failed_cache_for_manually_removed_object() {
|
||||
let (scanner, temp_dir) = build_test_scanner().await;
|
||||
let _guard = TestGuard {
|
||||
temp_dir: Some(temp_dir.clone()),
|
||||
};
|
||||
tokio::fs::create_dir_all(temp_dir.join("bucket"))
|
||||
.await
|
||||
.expect("bucket directory should be created");
|
||||
write_test_object_metadata(&temp_dir, "bucket", "repaired").await;
|
||||
|
||||
let mut info = crate::data_usage_define::DataUsageCacheInfo {
|
||||
name: "bucket".to_string(),
|
||||
..Default::default()
|
||||
};
|
||||
let canonical_root = temp_dir
|
||||
.canonicalize()
|
||||
.expect("test disk root should resolve to the path used by scanner");
|
||||
info.failed_objects.insert(
|
||||
canonical_root.join("bucket/removed/xl.meta").to_string_lossy().into_owned(),
|
||||
FolderScanner::now_secs(),
|
||||
);
|
||||
info.failed_objects.insert(
|
||||
canonical_root.join("bucket/repaired/xl.meta").to_string_lossy().into_owned(),
|
||||
FolderScanner::now_secs(),
|
||||
);
|
||||
let cache = DataUsageCache {
|
||||
info,
|
||||
..Default::default()
|
||||
};
|
||||
let parent = CancellationToken::new();
|
||||
let budget = ScannerCycleBudget::new(&parent, Default::default());
|
||||
|
||||
let scanned = scan_data_folder(
|
||||
budget.token(),
|
||||
budget,
|
||||
vec![scanner.local_disk.clone()],
|
||||
scanner.local_disk.clone(),
|
||||
cache,
|
||||
None,
|
||||
HealScanMode::Normal,
|
||||
SCANNER_SLEEPER.clone(),
|
||||
)
|
||||
.await
|
||||
.expect("complete bucket scan should succeed");
|
||||
|
||||
assert!(scanned.info.snapshot_complete);
|
||||
assert!(scanned.info.failed_objects.is_empty());
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
#[serial]
|
||||
async fn test_scan_data_folder_returns_raw_cursor_on_enumeration_cancel_without_root_progress() {
|
||||
@@ -3486,11 +3612,12 @@ async fn test_scan_data_folder_keeps_unresolved_objects_partial() {
|
||||
let _guard = TestGuard {
|
||||
temp_dir: Some(temp_dir.clone()),
|
||||
};
|
||||
write_test_object_metadata(&temp_dir, "bucket", "object").await;
|
||||
write_test_object_metadata_bytes(&temp_dir, "bucket", "object", &[]).await;
|
||||
|
||||
let failed_path = temp_dir
|
||||
.join("bucket")
|
||||
.join("object")
|
||||
.canonicalize()
|
||||
.expect("test disk root should resolve to the path used by scanner")
|
||||
.join("bucket/object")
|
||||
.join(STORAGE_FORMAT_FILE)
|
||||
.to_string_lossy()
|
||||
.into_owned();
|
||||
@@ -3502,7 +3629,10 @@ async fn test_scan_data_folder_keeps_unresolved_objects_partial() {
|
||||
},
|
||||
..Default::default()
|
||||
};
|
||||
cache.info.failed_objects.insert(failed_path, FolderScanner::now_secs());
|
||||
cache
|
||||
.info
|
||||
.failed_objects
|
||||
.insert(failed_path.clone(), FolderScanner::now_secs());
|
||||
|
||||
let parent = CancellationToken::new();
|
||||
let budget = ScannerCycleBudget::new(&parent, Default::default());
|
||||
@@ -3523,7 +3653,11 @@ async fn test_scan_data_folder_keeps_unresolved_objects_partial() {
|
||||
other => panic!("expected unresolved object to keep the cache partial, got {other:?}"),
|
||||
};
|
||||
assert!(!partial.info.snapshot_complete);
|
||||
assert!(!partial.info.failed_objects.is_empty());
|
||||
assert!(
|
||||
partial.info.failed_objects.contains_key(&failed_path),
|
||||
"corrupt object failure should remain recorded: {:?}",
|
||||
partial.info.failed_objects
|
||||
);
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
|
||||
@@ -3654,10 +3654,9 @@ async fn get_size_marks_corrupt_metadata_for_heal() {
|
||||
tokio::fs::create_dir_all(&object_dir)
|
||||
.await
|
||||
.expect("failed to create object directory");
|
||||
tokio::fs::write(&metadata_path, b"not-valid-filemeta")
|
||||
tokio::fs::write(&metadata_path, b"")
|
||||
.await
|
||||
.expect("failed to write corrupt metadata");
|
||||
|
||||
.expect("empty metadata fixture should be created");
|
||||
let endpoint = Endpoint::try_from(temp_dir.to_string_lossy().as_ref()).expect("failed to create endpoint");
|
||||
let disk = new_disk(
|
||||
&endpoint,
|
||||
@@ -3690,11 +3689,16 @@ async fn get_size_marks_corrupt_metadata_for_heal() {
|
||||
debug: false,
|
||||
};
|
||||
|
||||
let err = disk
|
||||
.get_size(item)
|
||||
.await
|
||||
.expect_err("corrupt metadata should be surfaced as scanner-heal work");
|
||||
assert!(is_scanner_metadata_corrupt_error(&err));
|
||||
for contents in [b"".as_slice(), b"not-valid-filemeta".as_slice()] {
|
||||
tokio::fs::write(&metadata_path, contents)
|
||||
.await
|
||||
.expect("corrupt metadata fixture should be written");
|
||||
let err = disk
|
||||
.get_size(item.clone())
|
||||
.await
|
||||
.expect_err("corrupt metadata should be surfaced as scanner-heal work");
|
||||
assert!(is_scanner_metadata_corrupt_error(&err));
|
||||
}
|
||||
|
||||
let _ = tokio::fs::remove_dir_all(&temp_dir).await;
|
||||
}
|
||||
|
||||
@@ -5,11 +5,9 @@
|
||||
|
||||
RustFS lets operators choose how much fsync work runs on the object write
|
||||
path. The default (`strict`) preserves the fully synced behavior RustFS has
|
||||
always shipped; the relaxed tiers are **opt-in** trades of power-loss
|
||||
durability for latency/IOPS. Note that **newly created buckets default to
|
||||
`relaxed`** (see [New-bucket default](#new-bucket-default)) — a gradual
|
||||
migration that leaves the process-wide default and all pre-existing buckets
|
||||
on `strict`.
|
||||
always shipped. `relaxed` and `none` are **opt-in** trades of power-loss
|
||||
durability for latency/IOPS. Newly created buckets inherit the process-wide
|
||||
mode unless `RUSTFS_NEW_BUCKET_DURABILITY_MODE` explicitly sets an override.
|
||||
|
||||
## Configuration
|
||||
|
||||
@@ -21,7 +19,7 @@ RUSTFS_DURABILITY_MODE=strict|relaxed|none # default: strict
|
||||
RUSTFS_DRIVE_SYNC_ENABLE=true|false # default: true
|
||||
|
||||
# Tier seeded into a NEWLY CREATED bucket's own override (see "New-bucket default")
|
||||
RUSTFS_NEW_BUCKET_DURABILITY_MODE=relaxed|strict|none|inherit # default: relaxed
|
||||
RUSTFS_NEW_BUCKET_DURABILITY_MODE=relaxed|strict|none|inherit # default: inherit process mode
|
||||
```
|
||||
|
||||
Resolution rules:
|
||||
@@ -93,6 +91,20 @@ power simultaneously, `relaxed` can lose recently acknowledged objects
|
||||
cluster-wide. Single-node
|
||||
deployments must stay on `strict`.
|
||||
|
||||
## Recovery from corrupt object metadata
|
||||
|
||||
A zero-length or unreadable `xl.meta` is corruption, not an absent object.
|
||||
RustFS reports it as a metadata error and keeps usage snapshots incomplete;
|
||||
quota admission can fail closed until a complete snapshot is available. On a
|
||||
multi-node deployment, attempt heal from healthy replicas and verify the
|
||||
recovered versions before cleaning any drive. If no healthy metadata copy or
|
||||
backup exists, the version index cannot be reconstructed from shard data
|
||||
alone. Preserve the affected directory for recovery before any operator-led
|
||||
cleanup. S3 `DeleteObject` is not a physical cleanup mechanism for corrupt
|
||||
metadata. After a successful repair or cleanup, a complete scanner cycle
|
||||
reconciles the failed-path cache; confirm the scanner reports a completed
|
||||
cycle before relying on a newly published usage snapshot.
|
||||
|
||||
**`none`.** No fsync on the object data path at all; acknowledged objects can
|
||||
vanish wholesale on power loss, payload included. System-critical writes are
|
||||
still pinned (below). This is the tier equivalent of the old escape hatch,
|
||||
@@ -149,18 +161,21 @@ configuration plane.
|
||||
|
||||
### New-bucket default
|
||||
|
||||
A newly created bucket gets a `relaxed` override **seeded into its own
|
||||
metadata** at creation time, so it opts
|
||||
into MinIO's default posture (object data still fdatasynced; xl.meta and
|
||||
directory-entry fsyncs left to the page cache) without touching the
|
||||
process-wide default. This is a gradual migration:
|
||||
A newly created bucket inherits the process-wide mode by default. On an
|
||||
otherwise unconfigured deployment this means `strict`. Operators can explicitly
|
||||
set `RUSTFS_NEW_BUCKET_DURABILITY_MODE=relaxed` to seed a relaxed override into
|
||||
new buckets, or choose `strict`, `none`, or `inherit`. A non-strict override is
|
||||
logged when the bucket is created; an explicitly non-strict process-wide mode
|
||||
is warned at startup. Relaxed remains appropriate only for the multi-node,
|
||||
independent-power-domain deployment described above.
|
||||
|
||||
- **Pre-existing buckets are unaffected.** A bucket with no `durability.json`
|
||||
entry keeps following the process-wide mode (`strict` by default), exactly
|
||||
as before.
|
||||
- **The process-wide default stays `strict`.** `RUSTFS_DURABILITY_MODE` and
|
||||
`RUSTFS_DRIVE_SYNC_ENABLE` are unchanged; only newly created buckets carry
|
||||
their own `relaxed` override.
|
||||
- **Pre-existing buckets are unaffected by this default change.** Existing
|
||||
per-bucket overrides remain stored in metadata, and buckets without an
|
||||
override continue to follow the process-wide mode.
|
||||
- **Explicit global configuration still applies.** If the process-wide mode is
|
||||
explicitly relaxed (or the legacy full-off mode is selected), an inherited
|
||||
new bucket follows that mode. Keep the process-wide mode strict on a
|
||||
single-node deployment.
|
||||
- **System-critical namespaces stay pinned to `strict`** regardless of any
|
||||
override (see [System-critical pinning](#system-critical-pinning)).
|
||||
|
||||
@@ -168,17 +183,21 @@ The seeded tier is controlled by an env var read once per bucket creation
|
||||
(bucket creation is not a hot path):
|
||||
|
||||
```bash
|
||||
RUSTFS_NEW_BUCKET_DURABILITY_MODE=relaxed # default: seed `relaxed`
|
||||
RUSTFS_NEW_BUCKET_DURABILITY_MODE=relaxed # explicitly seed `relaxed`
|
||||
RUSTFS_NEW_BUCKET_DURABILITY_MODE=strict # seed `strict` instead
|
||||
RUSTFS_NEW_BUCKET_DURABILITY_MODE=none # seed `none` instead
|
||||
RUSTFS_NEW_BUCKET_DURABILITY_MODE=inherit # seed nothing: follow the global mode
|
||||
```
|
||||
|
||||
`inherit` (and any unrecognized value, which fails closed to `inherit`) means
|
||||
the new bucket gets no override and follows the process-wide mode. Set this
|
||||
cluster-wide to opt out of the new default entirely. The seed only applies at
|
||||
bucket creation; it never retroactively rewrites existing buckets. To change
|
||||
an existing bucket's tier, use the per-bucket admin API above.
|
||||
`inherit` (the default) and any unrecognized value, which fails closed to
|
||||
`inherit`, mean the new bucket gets no override and follows the process-wide
|
||||
mode. The seed only applies at bucket creation; it never retroactively rewrites
|
||||
existing buckets. To correct an existing bucket created with a relaxed
|
||||
override, use the admin API above to set `strict` or delete the override to
|
||||
inherit the process-wide mode. Stored overrides do not record whether `relaxed`
|
||||
came from an earlier default or an explicit operator choice, so RustFS does not
|
||||
rewrite them automatically. Read and correct each affected bucket through the
|
||||
admin API; changing the env var alone is not a migration.
|
||||
|
||||
### Resolution order
|
||||
|
||||
|
||||
Reference in New Issue
Block a user