perf(ecstore): add strict/relaxed/none durability modes over the drive-sync switch (#4397)

This commit is contained in:
houseme
2026-07-08 08:45:33 +08:00
committed by GitHub
parent 92bf55ce62
commit eaff17cade
2 changed files with 713 additions and 34 deletions
+584 -34
View File
@@ -233,13 +233,19 @@ const ENV_RUSTFS_OBJECT_MMAP_READ_METHOD: &str = "RUSTFS_OBJECT_MMAP_READ_METHOD
const RUSTFS_OBJECT_MMAP_READ_METHOD_MMAP_COPY: &str = "mmap_copy";
const RUSTFS_OBJECT_MMAP_READ_METHOD_DIRECT_READ_COPY: &str = "direct_read_copy";
/// Fsync writes and commit-point renames so acknowledged data survives power loss.
/// Disabling trades durability for latency: data acknowledged with 200 OK may only
/// live in the page cache and be lost on a whole-node power failure.
/// Legacy binary switch for commit-point durability (fsync writes and renames).
/// Kept for compatibility: `true` maps to the `strict` durability mode (the
/// default), `false` keeps its historical semantics of disabling every fsync
/// on this disk, system-critical metadata included. Superseded by
/// `RUSTFS_DURABILITY_MODE`, which takes precedence when both are set.
/// Default: true.
const ENV_RUSTFS_DRIVE_SYNC_ENABLE: &str = "RUSTFS_DRIVE_SYNC_ENABLE";
const DEFAULT_RUSTFS_DRIVE_SYNC_ENABLE: bool = true;
/// Durability tier for object data-path writes: `strict` (default) | `relaxed` | `none`.
/// See docs/operations/durability-modes.md for the power-loss guarantee matrix.
const ENV_RUSTFS_DURABILITY_MODE: &str = "RUSTFS_DURABILITY_MODE";
#[derive(Clone, Copy, Debug, PartialEq, Eq)]
enum LocalReadCopyMethod {
MmapCopy,
@@ -251,9 +257,202 @@ fn is_direct_io_read_enabled() -> bool {
rustfs_utils::get_env_bool(ENV_RUSTFS_OBJECT_DIRECT_IO_READ_ENABLE, DEFAULT_RUSTFS_OBJECT_DIRECT_IO_READ_ENABLE)
}
/// Check if durable (fsync) writes are enabled.
pub(crate) fn drive_sync_enabled() -> bool {
rustfs_utils::get_env_bool(ENV_RUSTFS_DRIVE_SYNC_ENABLE, DEFAULT_RUSTFS_DRIVE_SYNC_ENABLE)
const EVENT_DISK_LOCAL_DURABILITY_MODE: &str = "disk_local_durability_mode";
/// Process-wide durability tier for commit-point fsync work on the local disk.
///
/// `Strict` is the default and preserves the historical (fully synced) write
/// path bit for bit. The other tiers are opt-in and only relax the object
/// data path; writes committing into system-critical namespaces stay pinned
/// to `Strict` (see [`effective_durability`]), except under `LegacyOff`,
/// which keeps the exact historical semantics of
/// `RUSTFS_DRIVE_SYNC_ENABLE=false` (no fsync anywhere, system metadata
/// included) so existing deployments keep their behavior unchanged.
#[derive(Clone, Copy, Debug, PartialEq, Eq)]
pub(crate) enum DurabilityMode {
/// Every commit point is fsynced: shard/part contents, xl.meta contents,
/// rollback backups, and the directory entries of commit renames.
Strict,
/// Object payload bytes (erasure shard files, multipart part files) are
/// still fdatasynced before the commit rename, but metadata commits
/// (xl.meta contents, rollback backups, directory entries) are left to
/// the page cache. Aligned with MinIO's default durability posture.
Relaxed,
/// No fsync on the object data path at all. System-critical namespaces
/// are still pinned to `Strict`.
None,
/// Historical semantics of `RUSTFS_DRIVE_SYNC_ENABLE=false`: no fsync
/// anywhere, without the system-critical pinning. Only reachable through
/// the legacy switch; not exposed by `RUSTFS_DURABILITY_MODE`.
LegacyOff,
}
impl DurabilityMode {
fn parse(value: &str) -> Option<Self> {
match value.trim().to_ascii_lowercase().as_str() {
"strict" => Some(Self::Strict),
"relaxed" => Some(Self::Relaxed),
"none" => Some(Self::None),
_ => Option::None,
}
}
fn as_str(self) -> &'static str {
match self {
Self::Strict => "strict",
Self::Relaxed => "relaxed",
Self::None => "none",
Self::LegacyOff => "legacy-off",
}
}
/// Whether object payload bytes (erasure shard files, multipart part
/// files) must be fdatasynced at commit points.
fn syncs_data_shards(self) -> bool {
matches!(self, Self::Strict | Self::Relaxed)
}
/// Whether metadata commits must be fsynced: xl.meta contents, rollback
/// backups, and the directory entries created by commit renames.
fn syncs_commit_metadata(self) -> bool {
matches!(self, Self::Strict)
}
}
/// Pure resolution of the durability mode from configuration values.
///
/// `RUSTFS_DURABILITY_MODE` wins when set to a valid value; otherwise the
/// legacy `RUSTFS_DRIVE_SYNC_ENABLE` switch keeps its historical mapping
/// (`true` -> strict, `false` -> the old full-off semantics). The default is
/// strict.
fn resolve_durability_mode(mode_env: Option<String>, legacy_drive_sync_enabled: bool) -> DurabilityMode {
if let Some(raw) = mode_env {
if let Some(mode) = DurabilityMode::parse(&raw) {
return mode;
}
warn!(
event = EVENT_DISK_LOCAL_DURABILITY_MODE,
component = LOG_COMPONENT_ECSTORE,
subsystem = LOG_SUBSYSTEM_DISK_LOCAL,
value = %raw,
"Invalid RUSTFS_DURABILITY_MODE value; expected strict|relaxed|none, falling back to the legacy drive-sync switch"
);
}
if legacy_drive_sync_enabled {
DurabilityMode::Strict
} else {
DurabilityMode::LegacyOff
}
}
/// The configured durability mode, resolved from the environment once per
/// process and cached (the previous binary switch re-read the environment on
/// every call, i.e. a dozen times per PUT, and could even flip mid-operation).
pub(crate) fn durability_mode() -> DurabilityMode {
#[cfg(test)]
if let Some(mode) = durability_mode_override::get() {
return mode;
}
static MODE: OnceLock<DurabilityMode> = OnceLock::new();
*MODE.get_or_init(|| {
let mode = resolve_durability_mode(
rustfs_utils::get_env_opt_str(ENV_RUSTFS_DURABILITY_MODE),
rustfs_utils::get_env_bool(ENV_RUSTFS_DRIVE_SYNC_ENABLE, DEFAULT_RUSTFS_DRIVE_SYNC_ENABLE),
);
info!(
event = EVENT_DISK_LOCAL_DURABILITY_MODE,
component = LOG_COMPONENT_ECSTORE,
subsystem = LOG_SUBSYSTEM_DISK_LOCAL,
mode = mode.as_str(),
"Storage durability mode resolved"
);
mode
})
}
/// Test-only override for [`durability_mode`].
///
/// The production value is resolved from the environment once per process, so
/// tests exercising non-default tiers need a process-level override hook.
/// Setting an override serializes callers on a mutex so a relaxed-tier test
/// can never leak its mode into a parallel strict-tier test.
#[cfg(test)]
pub(crate) mod durability_mode_override {
use super::DurabilityMode;
use std::sync::{Mutex, MutexGuard, PoisonError, RwLock};
static OVERRIDE: RwLock<Option<DurabilityMode>> = RwLock::new(None);
static SERIAL: Mutex<()> = Mutex::new(());
pub(crate) fn get() -> Option<DurabilityMode> {
*OVERRIDE.read().unwrap_or_else(PoisonError::into_inner)
}
/// Holds the override (and the serialization lock) until dropped.
pub(crate) struct OverrideGuard {
_serial: MutexGuard<'static, ()>,
}
impl Drop for OverrideGuard {
fn drop(&mut self) {
*OVERRIDE.write().unwrap_or_else(PoisonError::into_inner) = None;
}
}
pub(crate) fn set(mode: DurabilityMode) -> OverrideGuard {
let serial = SERIAL.lock().unwrap_or_else(PoisonError::into_inner);
*OVERRIDE.write().unwrap_or_else(PoisonError::into_inner) = Some(mode);
OverrideGuard { _serial: serial }
}
}
/// Whether `volume` stages in-flight user object data (`.rustfs.sys/tmp`,
/// `.rustfs.sys/multipart`, and their subtrees). These namespaces follow the
/// configured durability mode: their contents commit into user buckets and
/// are exactly the writes the relaxed tiers exist for.
fn is_scratch_volume(volume: &str) -> bool {
for scratch in [super::RUSTFS_META_TMP_BUCKET, super::RUSTFS_META_MULTIPART_BUCKET] {
if volume == scratch || volume.strip_prefix(scratch).is_some_and(|rest| rest.starts_with('/')) {
return true;
}
}
false
}
/// Whether writes committing into `volume` carry system-critical state:
/// format.json, IAM and cluster config, bucket metadata, and everything else
/// under `.rustfs.sys` (or `.minio.sys` during migration) outside the scratch
/// namespaces. Losing these can take out the whole deployment and they are
/// far off the object hot path, so they never follow a relaxed tier.
fn is_system_critical_volume(volume: &str) -> bool {
if is_scratch_volume(volume) {
return false;
}
for meta in [super::RUSTFS_META_BUCKET, super::MIGRATING_META_BUCKET] {
if volume == meta || volume.strip_prefix(meta).is_some_and(|rest| rest.starts_with('/')) {
return true;
}
}
false
}
/// Effective durability for writes that commit into `volume`.
///
/// System-critical volumes are pinned to `Strict` regardless of the
/// configured tier. The legacy full-off switch keeps its historical
/// semantics and is never pinned.
pub(crate) fn effective_durability(volume: &str) -> DurabilityMode {
let mode = durability_mode();
match mode {
DurabilityMode::Strict | DurabilityMode::LegacyOff => mode,
DurabilityMode::Relaxed | DurabilityMode::None => {
if is_system_critical_volume(volume) {
DurabilityMode::Strict
} else {
mode
}
}
}
}
/// Get the O_DIRECT read threshold size.
@@ -2486,18 +2685,25 @@ impl LocalDisk {
let tmp_volume_dir = self.get_bucket_path(super::RUSTFS_META_TMP_BUCKET)?;
let tmp_file_path = self.get_object_path(super::RUSTFS_META_TMP_BUCKET, Uuid::new_v4().to_string().as_str())?;
let durability = effective_durability(volume);
// The tmp file is renamed to its final location right below, so only
// its contents must be durable here (SyncMode::FileOnly): the rename
// drops the tmp directory entry, and the destination parent directory
// is fsynced after the rename.
let tmp_sync = if sync { SyncMode::FileOnly } else { SyncMode::None };
// is fsynced after the rename. Both are metadata commits, so relaxed
// tiers skip them.
let tmp_sync = if sync && durability.syncs_commit_metadata() {
SyncMode::FileOnly
} else {
SyncMode::None
};
self.write_all_internal(&tmp_file_path, InternalBuf::Ref(buf), tmp_sync, &tmp_volume_dir)
.await?;
rename_all(tmp_file_path, &file_path, volume_dir).await?;
if sync
&& drive_sync_enabled()
&& durability.syncs_commit_metadata()
&& let Some(parent) = file_path.parent()
{
os::fsync_dir(parent).await.map_err(to_file_error)?;
@@ -2517,8 +2723,14 @@ impl LocalDisk {
// Files written here (format.json, ...) stay where they land — no
// rename follows — so the new directory entry must be fsynced too.
self.write_all_private(volume, path, data, SyncMode::FileAndDir, &volume_dir)
.await?;
// System-critical volumes are pinned to strict by effective_durability;
// only the legacy full-off switch (historical semantics) skips this.
let sync = if effective_durability(volume).syncs_commit_metadata() {
SyncMode::FileAndDir
} else {
SyncMode::None
};
self.write_all_private(volume, path, data, sync, &volume_dir).await?;
Ok(())
}
@@ -2534,7 +2746,9 @@ impl LocalDisk {
Ok(())
}
// write_all_internal do write file
// write_all_internal do write file.
// Executes the given SyncMode verbatim: durability policy (tier gating,
// system-critical pinning) is resolved by callers via effective_durability.
async fn write_all_internal(
&self,
file_path: &Path,
@@ -2548,8 +2762,6 @@ impl LocalDisk {
skip_parent
};
let sync = if drive_sync_enabled() { sync } else { SyncMode::None };
match data {
InternalBuf::Ref(buf) => {
let mut f = self.open_file(file_path, O_CREATE | O_WRONLY | O_TRUNC, skip_parent).await?;
@@ -3759,9 +3971,10 @@ impl DiskAPI for LocalDisk {
}
// UploadPart is acknowledged once this rename lands, so the part data and
// its directory entry must be durable before we return.
let sync = drive_sync_enabled();
if sync && !src_is_dir {
// its directory entry must be durable before we return. Relaxed keeps the
// part payload fdatasync but leaves the directory entry to the page cache.
let durability = effective_durability(dst_volume);
if durability.syncs_data_shards() && !src_is_dir {
let src = src_file_path.clone();
tokio::task::spawn_blocking(move || std::fs::File::open(&src)?.sync_data())
.await
@@ -3771,7 +3984,9 @@ impl DiskAPI for LocalDisk {
rename_all(&src_file_path, &dst_file_path, &dst_volume_dir).await?;
if sync && let Some(parent) = dst_file_path.parent() {
if durability.syncs_commit_metadata()
&& let Some(parent) = dst_file_path.parent()
{
os::fsync_dir(parent).await.map_err(to_file_error)?;
}
@@ -4109,6 +4324,14 @@ impl DiskAPI for LocalDisk {
let no_inline = fi.data.is_none() && fi.size > 0;
// Resolved once for the whole commit so a concurrent configuration
// change can never leave a single rename_data half-synced. The tier is
// keyed on the destination volume: user data staged in scratch
// namespaces follows the configured tier, while commits into
// system-critical namespaces (IAM, config, bucket metadata) stay
// pinned to strict.
let durability = effective_durability(dst_volume);
if no_inline {
// Non-inline: read xl.meta, parse, write, rename data dir, rename xl.meta
let has_dst_buf = match super::fs::read_file(&dst_file_path).await {
@@ -4151,12 +4374,18 @@ impl DiskAPI for LocalDisk {
// point below, so only its contents must be durable before the
// rename (SyncMode::FileOnly); the dst parent directory is fsynced
// after the commit rename, and a crash before the rename means the
// PUT was never acknowledged.
// PUT was never acknowledged. A metadata commit: relaxed tiers
// leave it to the page cache.
let tmp_meta_sync = if durability.syncs_commit_metadata() {
SyncMode::FileOnly
} else {
SyncMode::None
};
self.write_all_private(
src_volume,
&format!("{}/{}", &src_path, STORAGE_FORMAT_FILE),
new_dst_buf.into(),
SyncMode::FileOnly,
tmp_meta_sync,
src_file_parent,
)
.await?;
@@ -4166,7 +4395,8 @@ impl DiskAPI for LocalDisk {
// page cache. Multipart parts were already synced during rename_part, so
// their fdatasync here is a cheap no-op. A missing source dir is left for
// the rename below to report through the existing rollback path.
if drive_sync_enabled()
// Payload durability: kept by both strict and relaxed.
if durability.syncs_data_shards()
&& let Some((src_data_path, _)) = has_data_dir_path.as_ref()
&& let Err(err) = os::sync_dir_files(src_data_path).await
&& err.kind() != ErrorKind::NotFound
@@ -4206,8 +4436,15 @@ impl DiskAPI for LocalDisk {
}
// The rollback backup stays where it is written (no rename) and is
// the sole restore source for a later undo_write, so it keeps
// SyncMode::FileAndDir: contents and directory entry both durable.
// the sole restore source for a later undo_write, so under strict
// it keeps SyncMode::FileAndDir: contents and directory entry both
// durable. It is part of the metadata commit machinery, so relaxed
// tiers leave it to the page cache like the xl.meta it mirrors.
let backup_sync = if durability.syncs_commit_metadata() {
SyncMode::FileAndDir
} else {
SyncMode::None
};
if let Some(old_data_dir) = has_old_data_dir
&& let Some(dst_buf) = has_dst_buf.as_ref()
&& let Err(err) = self
@@ -4215,7 +4452,7 @@ impl DiskAPI for LocalDisk {
dst_volume,
&format!("{}/{}/{}", &dst_path, &old_data_dir, STORAGE_FORMAT_FILE_BACKUP),
dst_buf.clone().into(),
SyncMode::FileAndDir,
backup_sync,
&skip_parent,
)
.await
@@ -4252,8 +4489,9 @@ impl DiskAPI for LocalDisk {
}
// Persist the directory entries for both the data dir and xl.meta renames;
// without this the commit itself can vanish on power loss.
if drive_sync_enabled()
// without this the commit itself can vanish on power loss. Relaxed tiers
// accept that window (documented in docs/operations/durability-modes.md).
if durability.syncs_commit_metadata()
&& let Some(parent) = dst_file_path.parent()
&& let Err(err) = os::fsync_dir(parent).await
{
@@ -4309,11 +4547,15 @@ impl DiskAPI for LocalDisk {
let version_signature = rename_data_versions_signature(&xlmeta);
let new_buf = xlmeta.marshal_msg()?;
// Write new xl.meta + rename
// Write new xl.meta + rename. Inline objects carry their data
// inside xl.meta, so this whole sequence is a metadata commit:
// relaxed tiers do no per-object fsync here at all (aligned
// with MinIO's default), trading a documented power-loss
// window for latency.
if let Some(parent) = src.parent() {
std::fs::create_dir_all(parent)?;
}
let sync = drive_sync_enabled();
let sync = durability.syncs_commit_metadata();
let mut f = std::fs::OpenOptions::new()
.create(true)
.write(true)
@@ -5123,15 +5365,16 @@ mod test {
}
#[tokio::test]
#[allow(clippy::await_holding_lock)]
async fn test_write_all_meta_skips_tmp_parent_dir_fsync_but_fsyncs_dst_parent() {
use tempfile::tempdir;
let _mode = durability_mode_override::set(DurabilityMode::Strict);
let dir = tempdir().expect("temp dir should be created");
let endpoint = Endpoint::try_from(dir.path().to_str().expect("temp dir should be utf8")).expect("endpoint should parse");
let disk = LocalDisk::new(&endpoint, false).await.expect("local disk should be created");
assert!(drive_sync_enabled(), "test requires the default drive-sync configuration");
let bucket = "sync-meta-bucket";
ensure_test_volume(&disk, bucket).await;
ensure_test_volume(&disk, RUSTFS_META_TMP_BUCKET).await;
@@ -5164,15 +5407,16 @@ mod test {
}
#[tokio::test]
#[allow(clippy::await_holding_lock)]
async fn test_write_all_public_still_fsyncs_parent_dir() {
use tempfile::tempdir;
let _mode = durability_mode_override::set(DurabilityMode::Strict);
let dir = tempdir().expect("temp dir should be created");
let endpoint = Endpoint::try_from(dir.path().to_str().expect("temp dir should be utf8")).expect("endpoint should parse");
let disk = LocalDisk::new(&endpoint, false).await.expect("local disk should be created");
assert!(drive_sync_enabled(), "test requires the default drive-sync configuration");
let bucket = "sync-public-bucket";
ensure_test_volume(&disk, bucket).await;
@@ -5191,15 +5435,16 @@ mod test {
}
#[tokio::test]
#[allow(clippy::await_holding_lock)]
async fn test_rename_data_non_inline_skips_tmp_parent_dir_fsync() {
use tempfile::tempdir;
let _mode = durability_mode_override::set(DurabilityMode::Strict);
let dir = tempdir().expect("temp dir should be created");
let endpoint = Endpoint::try_from(dir.path().to_str().expect("temp dir should be utf8")).expect("endpoint should parse");
let disk = LocalDisk::new(&endpoint, false).await.expect("local disk should be created");
assert!(drive_sync_enabled(), "test requires the default drive-sync configuration");
let bucket = "sync-rename-bucket";
let object = "dir/object";
let tmp_object = "tmp-sync-write";
@@ -5276,6 +5521,311 @@ mod test {
);
}
#[test]
fn test_resolve_durability_mode_mapping() {
// Default: nothing set -> strict (current main behavior).
assert_eq!(resolve_durability_mode(None, DEFAULT_RUSTFS_DRIVE_SYNC_ENABLE), DurabilityMode::Strict);
// Legacy switch compatibility mapping.
assert_eq!(resolve_durability_mode(None, true), DurabilityMode::Strict);
assert_eq!(resolve_durability_mode(None, false), DurabilityMode::LegacyOff);
// Explicit mode wins over the legacy switch.
assert_eq!(resolve_durability_mode(Some("strict".into()), false), DurabilityMode::Strict);
assert_eq!(resolve_durability_mode(Some("relaxed".into()), true), DurabilityMode::Relaxed);
assert_eq!(resolve_durability_mode(Some("none".into()), true), DurabilityMode::None);
// Case- and whitespace-tolerant.
assert_eq!(resolve_durability_mode(Some(" RELAXED ".into()), true), DurabilityMode::Relaxed);
// Invalid values fall back to the legacy switch, then the default.
assert_eq!(resolve_durability_mode(Some("bogus".into()), true), DurabilityMode::Strict);
assert_eq!(resolve_durability_mode(Some("bogus".into()), false), DurabilityMode::LegacyOff);
assert_eq!(resolve_durability_mode(Some(String::new()), true), DurabilityMode::Strict);
}
#[test]
fn test_durability_mode_sync_gates() {
// Strict = current main behavior: every commit point synced.
assert!(DurabilityMode::Strict.syncs_data_shards());
assert!(DurabilityMode::Strict.syncs_commit_metadata());
// Relaxed keeps payload durability, drops metadata-commit fsyncs.
assert!(DurabilityMode::Relaxed.syncs_data_shards());
assert!(!DurabilityMode::Relaxed.syncs_commit_metadata());
// None and the legacy full-off switch sync nothing on the data path.
assert!(!DurabilityMode::None.syncs_data_shards());
assert!(!DurabilityMode::None.syncs_commit_metadata());
assert!(!DurabilityMode::LegacyOff.syncs_data_shards());
assert!(!DurabilityMode::LegacyOff.syncs_commit_metadata());
}
#[test]
fn test_system_critical_volume_classification() {
// System namespaces are pinned.
assert!(is_system_critical_volume(RUSTFS_META_BUCKET));
assert!(is_system_critical_volume(&format!("{RUSTFS_META_BUCKET}/buckets")));
assert!(is_system_critical_volume(&format!("{RUSTFS_META_BUCKET}/config")));
assert!(is_system_critical_volume(super::super::MIGRATING_META_BUCKET));
// Scratch namespaces stage user object data and follow the tier.
assert!(!is_system_critical_volume(RUSTFS_META_TMP_BUCKET));
assert!(!is_system_critical_volume(RUSTFS_META_TMP_DELETED_BUCKET));
assert!(!is_system_critical_volume(super::super::RUSTFS_META_MULTIPART_BUCKET));
// User buckets follow the tier; similarly-prefixed names are not meta.
assert!(!is_system_critical_volume("my-bucket"));
assert!(!is_system_critical_volume(".rustfs.sys-lookalike"));
}
#[test]
fn test_effective_durability_pins_system_volumes() {
{
let _mode = durability_mode_override::set(DurabilityMode::Relaxed);
assert_eq!(effective_durability("user-bucket"), DurabilityMode::Relaxed);
assert_eq!(effective_durability(super::super::RUSTFS_META_MULTIPART_BUCKET), DurabilityMode::Relaxed);
assert_eq!(effective_durability(RUSTFS_META_TMP_BUCKET), DurabilityMode::Relaxed);
assert_eq!(effective_durability(RUSTFS_META_BUCKET), DurabilityMode::Strict);
}
{
let _mode = durability_mode_override::set(DurabilityMode::None);
assert_eq!(effective_durability("user-bucket"), DurabilityMode::None);
assert_eq!(effective_durability(RUSTFS_META_BUCKET), DurabilityMode::Strict);
}
{
// The legacy full-off switch keeps its historical semantics:
// nothing is pinned, not even system-critical namespaces.
let _mode = durability_mode_override::set(DurabilityMode::LegacyOff);
assert_eq!(effective_durability("user-bucket"), DurabilityMode::LegacyOff);
assert_eq!(effective_durability(RUSTFS_META_BUCKET), DurabilityMode::LegacyOff);
}
}
#[tokio::test]
#[allow(clippy::await_holding_lock)]
async fn test_write_all_meta_relaxed_skips_dst_parent_dir_fsync() {
use tempfile::tempdir;
let _mode = durability_mode_override::set(DurabilityMode::Relaxed);
let dir = tempdir().expect("temp dir should be created");
let endpoint = Endpoint::try_from(dir.path().to_str().expect("temp dir should be utf8")).expect("endpoint should parse");
let disk = LocalDisk::new(&endpoint, false).await.expect("local disk should be created");
let bucket = "relaxed-meta-bucket";
ensure_test_volume(&disk, bucket).await;
ensure_test_volume(&disk, RUSTFS_META_TMP_BUCKET).await;
let meta_path = format!("dir/object/{STORAGE_FORMAT_FILE}");
disk.write_all_meta(bucket, &meta_path, b"payload", true)
.await
.expect("write_all_meta should succeed");
let dst_file_path = disk.get_object_path(bucket, &meta_path).expect("dst path should resolve");
assert_eq!(
tokio::fs::read(&dst_file_path).await.expect("xl.meta should exist"),
b"payload",
"relaxed mode must not change what gets written, only what gets fsynced"
);
let dst_parent = dst_file_path.parent().expect("dst file should have a parent").to_path_buf();
assert!(
!os::fsync_dir_recorder::was_fsynced(&dst_parent),
"relaxed mode must skip the metadata-commit dir fsync on user volumes"
);
}
#[tokio::test]
#[allow(clippy::await_holding_lock)]
async fn test_write_all_public_relaxed_pins_system_volume() {
use tempfile::tempdir;
let _mode = durability_mode_override::set(DurabilityMode::Relaxed);
let dir = tempdir().expect("temp dir should be created");
let endpoint = Endpoint::try_from(dir.path().to_str().expect("temp dir should be utf8")).expect("endpoint should parse");
let disk = LocalDisk::new(&endpoint, false).await.expect("local disk should be created");
let user_bucket = "relaxed-public-bucket";
ensure_test_volume(&disk, user_bucket).await;
// User volume: the direct write follows the relaxed tier.
disk.write_all(user_bucket, "config/settings.json", Bytes::from_static(b"payload"))
.await
.expect("write_all should succeed");
let user_parent = disk
.get_object_path(user_bucket, "config/settings.json")
.expect("file path should resolve")
.parent()
.expect("file should have a parent")
.to_path_buf();
assert!(
!os::fsync_dir_recorder::was_fsynced(&user_parent),
"relaxed mode must skip the dir fsync for direct writes into user volumes"
);
// System-critical volume: pinned to strict regardless of the tier.
disk.write_all(RUSTFS_META_BUCKET, "buckets/test/bucket-metadata", Bytes::from_static(b"payload"))
.await
.expect("write_all should succeed");
let meta_parent = disk
.get_object_path(RUSTFS_META_BUCKET, "buckets/test/bucket-metadata")
.expect("meta path should resolve")
.parent()
.expect("meta file should have a parent")
.to_path_buf();
assert!(
os::fsync_dir_recorder::was_fsynced(&meta_parent),
"system-critical writes must stay fully synced under relaxed mode"
);
}
#[tokio::test]
#[allow(clippy::await_holding_lock)]
async fn test_rename_data_relaxed_keeps_shard_sync_skips_commit_fsyncs() {
use tempfile::tempdir;
let _mode = durability_mode_override::set(DurabilityMode::Relaxed);
let dir = tempdir().expect("temp dir should be created");
let endpoint = Endpoint::try_from(dir.path().to_str().expect("temp dir should be utf8")).expect("endpoint should parse");
let disk = LocalDisk::new(&endpoint, false).await.expect("local disk should be created");
let bucket = "relaxed-rename-bucket";
let object = "dir/object";
let tmp_object = "tmp-relaxed-write";
let version_id = Uuid::parse_str("77777777-7777-7777-7777-777777777777").expect("version id should parse");
let old_data_dir = Uuid::parse_str("88888888-8888-8888-8888-888888888888").expect("old data dir should parse");
let new_data_dir = Uuid::parse_str("99999999-9999-9999-9999-999999999999").expect("new data dir should parse");
ensure_test_volume(&disk, bucket).await;
ensure_test_volume(&disk, RUSTFS_META_TMP_BUCKET).await;
let old_fi = test_file_info(object, version_id, Some(old_data_dir), None);
let dst_object_dir = dir.path().join(bucket).join(object);
fs::create_dir_all(dst_object_dir.join(old_data_dir.to_string()))
.await
.expect("old data dir should be created");
fs::write(dst_object_dir.join(STORAGE_FORMAT_FILE), test_meta(old_fi))
.await
.expect("old metadata should be written");
let tmp_data_dir = dir
.path()
.join(RUSTFS_META_TMP_BUCKET)
.join(tmp_object)
.join(new_data_dir.to_string());
fs::create_dir_all(&tmp_data_dir)
.await
.expect("new tmp data dir should be created");
fs::write(tmp_data_dir.join("part.1"), b"new-data")
.await
.expect("new tmp data should be written");
let new_fi = test_file_info(object, version_id, Some(new_data_dir), None);
let resp = disk
.rename_data(RUSTFS_META_TMP_BUCKET, tmp_object, new_fi, bucket, object)
.await
.expect("rename_data should commit");
assert_eq!(resp.old_data_dir, Some(old_data_dir), "relaxed mode must not change commit semantics");
// Compare against root-resolved paths: the recorder stores the paths
// the disk actually fsyncs, which go through the canonicalized root.
let resolved_tmp_data_dir = disk
.get_object_path(RUSTFS_META_TMP_BUCKET, &format!("{tmp_object}/{new_data_dir}"))
.expect("tmp data dir path should resolve");
let resolved_dst_object_dir = disk.get_object_path(bucket, object).expect("dst object dir should resolve");
// Payload durability is kept: sync_dir_files fdatasyncs the shard
// files and fsyncs the staged data dir before the commit rename.
assert!(
os::fsync_dir_recorder::was_fsynced(&resolved_tmp_data_dir),
"relaxed mode must keep the shard-data sync before the commit rename"
);
// Metadata-commit fsyncs are skipped: neither the destination parent
// dir nor the rollback backup parent dir is fsynced.
assert!(
!os::fsync_dir_recorder::was_fsynced(&resolved_dst_object_dir),
"relaxed mode must skip the commit-rename dir fsync"
);
let resolved_backup_parent = resolved_dst_object_dir.join(old_data_dir.to_string());
assert!(
!os::fsync_dir_recorder::was_fsynced(&resolved_backup_parent),
"relaxed mode must skip the rollback-backup dir fsync"
);
assert!(
dst_object_dir
.join(old_data_dir.to_string())
.join(STORAGE_FORMAT_FILE_BACKUP)
.exists(),
"the rollback backup itself must still be written"
);
}
#[tokio::test]
#[allow(clippy::await_holding_lock)]
async fn test_rename_data_legacy_off_skips_all_fsyncs() {
use tempfile::tempdir;
let _mode = durability_mode_override::set(DurabilityMode::LegacyOff);
let dir = tempdir().expect("temp dir should be created");
let endpoint = Endpoint::try_from(dir.path().to_str().expect("temp dir should be utf8")).expect("endpoint should parse");
let disk = LocalDisk::new(&endpoint, false).await.expect("local disk should be created");
let bucket = "legacy-off-bucket";
let object = "dir/object";
let tmp_object = "tmp-legacy-write";
let version_id = Uuid::parse_str("aaaaaaaa-aaaa-aaaa-aaaa-aaaaaaaaaaaa").expect("version id should parse");
let new_data_dir = Uuid::parse_str("bbbbbbbb-bbbb-bbbb-bbbb-bbbbbbbbbbbb").expect("new data dir should parse");
ensure_test_volume(&disk, bucket).await;
ensure_test_volume(&disk, RUSTFS_META_TMP_BUCKET).await;
let tmp_data_dir = dir
.path()
.join(RUSTFS_META_TMP_BUCKET)
.join(tmp_object)
.join(new_data_dir.to_string());
fs::create_dir_all(&tmp_data_dir)
.await
.expect("new tmp data dir should be created");
fs::write(tmp_data_dir.join("part.1"), b"new-data")
.await
.expect("new tmp data should be written");
let new_fi = test_file_info(object, version_id, Some(new_data_dir), None);
disk.rename_data(RUSTFS_META_TMP_BUCKET, tmp_object, new_fi, bucket, object)
.await
.expect("rename_data should commit");
// Historical RUSTFS_DRIVE_SYNC_ENABLE=false semantics: no fsync at
// all, not even the shard-data sync relaxed keeps. Assertions use the
// root-resolved paths the disk actually passes to fsync.
let resolved_tmp_data_dir = disk
.get_object_path(RUSTFS_META_TMP_BUCKET, &format!("{tmp_object}/{new_data_dir}"))
.expect("tmp data dir path should resolve");
assert!(
!os::fsync_dir_recorder::was_fsynced(&resolved_tmp_data_dir),
"legacy-off must not sync staged shard data"
);
let resolved_dst_object_dir = disk.get_object_path(bucket, object).expect("dst object dir should resolve");
assert!(
!os::fsync_dir_recorder::was_fsynced(&resolved_dst_object_dir),
"legacy-off must not fsync the commit-rename dir"
);
// And system-critical volumes are NOT pinned: the old full-off
// behavior is preserved bit for bit for existing deployments.
disk.write_all(RUSTFS_META_BUCKET, "buckets/legacy/bucket-metadata", Bytes::from_static(b"payload"))
.await
.expect("write_all should succeed");
let meta_parent = disk
.get_object_path(RUSTFS_META_BUCKET, "buckets/legacy/bucket-metadata")
.expect("meta path should resolve")
.parent()
.expect("meta file should have a parent")
.to_path_buf();
assert!(
!os::fsync_dir_recorder::was_fsynced(&meta_parent),
"legacy-off keeps the historical semantics: system metadata is not synced either"
);
}
#[tokio::test]
async fn test_rename_data_writes_old_metadata_backup_for_inline_overwrite() {
use tempfile::tempdir;
+129
View File
@@ -0,0 +1,129 @@
# Durability modes (drive sync tiers)
RustFS lets operators choose how much fsync work runs on the object write
path. The default (`strict`) preserves the fully synced behavior RustFS has
always shipped; the relaxed tiers are **opt-in** trades of power-loss
durability for latency/IOPS.
## Configuration
```bash
# New tiered switch (wins when set to a valid value)
RUSTFS_DURABILITY_MODE=strict|relaxed|none # default: strict
# Legacy binary switch (kept for compatibility, superseded by the above)
RUSTFS_DRIVE_SYNC_ENABLE=true|false # default: true
```
Resolution rules:
| `RUSTFS_DURABILITY_MODE` | `RUSTFS_DRIVE_SYNC_ENABLE` | Effective mode |
| --- | --- | --- |
| unset | unset | `strict` (default) |
| unset | `true` | `strict` |
| unset | `false` | `legacy-off` (the historical "everything off" semantics) |
| `strict` / `relaxed` / `none` | anything | the named mode |
| invalid value | any | logged warning, falls back to the legacy switch, then the default |
Values are case-insensitive and whitespace-tolerant. The mode is resolved
**once per process** and cached (it also removes the per-call `getenv` the
old switch performed a dozen times per PUT); changing the environment
requires a restart. The resolved mode is logged at startup under the
`disk_local_durability_mode` event.
`legacy-off` is not a value of `RUSTFS_DURABILITY_MODE`; it is only reachable
through `RUSTFS_DRIVE_SYNC_ENABLE=false` so that existing deployments keep
their exact current behavior. It is deprecated and will be retained for at
least one major version.
## What each mode fsyncs
Write points on the object path and how each mode treats them:
| Write point | `strict` | `relaxed` | `none` | `legacy-off` |
| --- | --- | --- | --- | --- |
| Erasure shard files (fdatasync before the commit rename) | yes | **yes** | no | no |
| Multipart part payload (fdatasync before `rename_part` commit) | yes | **yes** | no | no |
| xl.meta contents (tmp write before the commit rename) | yes | no | no | no |
| Inline objects (data embedded in xl.meta) | yes | no | no | no |
| Old-metadata rollback backups | yes | no | no | no |
| Directory entries of commit renames (fsync of the parent dir) | yes | no | no | no |
| System-critical writes (see pinning below) | yes | yes (pinned) | yes (pinned) | **no** |
## Power-loss guarantees, honestly stated
**`strict` (default).** Every acknowledged write (PUT, UploadPart,
CompleteMultipartUpload, delete markers, metadata updates) is durable on the
individual drive before the 200 OK: payload bytes, xl.meta, rollback backups,
and the directory entries of the commit renames are all fsynced. A whole-node
(or whole-cluster) power failure does not lose acknowledged data. This is the
current mainline behavior, unchanged.
**`relaxed`.** Payload bytes of non-inline objects and multipart parts are
fdatasynced to the device before the acknowledgement, but the metadata
commits — xl.meta contents, rollback backups, and the directory entries of
the commit renames — are left to the page cache. Consequences on a power
failure:
- On the affected drive, a **recently acknowledged version can be lost
entirely** — not merely "the directory entry rolls back". When neither the
xl.meta bytes nor the rename's directory entry are synced, the commit
itself can vanish; surviving shard bytes become unreferenced orphans on
that drive.
- **Inline (small) objects receive no per-object fsync at all** in this mode:
their data lives inside xl.meta, and xl.meta is not synced. This matches
MinIO's default posture (no per-object fsync) but means small objects have
the widest loss window.
- Durability of acknowledged writes therefore rests on **erasure-coded
redundancy across other nodes** plus the unclean-shutdown heal introduced
in PR #4221 converging the affected drive afterwards.
Deployment rule for `relaxed`: only multi-node clusters whose nodes sit in
**independent power domains** (separate feeds/UPS). If all nodes can lose
power simultaneously — the exact incident class that motivated PR #4221
`relaxed` can lose recently acknowledged objects cluster-wide. Single-node
deployments must stay on `strict`.
**`none`.** No fsync on the object data path at all; acknowledged objects can
vanish wholesale on power loss, payload included. System-critical writes are
still pinned (below). This is the tier equivalent of the old escape hatch,
useful for throwaway/benchmark data only.
**`legacy-off`.** The historical semantics of
`RUSTFS_DRIVE_SYNC_ENABLE=false`, preserved bit for bit for existing
deployments: nothing is fsynced anywhere, **including system-critical
metadata** such as `format.json`. Prefer `RUSTFS_DURABILITY_MODE=none`, which
keeps the system-critical writes safe.
## System-critical pinning
Writes that commit into system namespaces are pinned to `strict` regardless
of the configured tier (except under `legacy-off`, see above):
- `.rustfs.sys``format.json`, IAM and cluster configuration, bucket
metadata, and everything else outside the scratch namespaces;
- `.minio.sys` — the same namespace during MinIO migration.
The scratch namespaces `.rustfs.sys/tmp` and `.rustfs.sys/multipart` stage
in-flight **user object data** and follow the configured tier — they are
exactly the writes the relaxed tiers exist for. Their durability is decided
by the destination volume at commit time, so an IAM or bucket-metadata object
staged in tmp still commits with full `strict` durability.
The durability mode is server-side configuration only; it cannot be raised or
lowered by any request header.
## Performance expectations
The often-quoted 26x PUT throughput delta was measured on macOS with the old
binary switch fully **off** (equivalent to `none`/`legacy-off`), where
`F_FULLFSYNC` heavily amplifies sync cost. `relaxed` keeps the per-shard
fdatasync, so its gain is necessarily smaller and must be measured on the
target platform (Linux ext4/xfs) before being relied on. Do not use `none`
numbers to size `relaxed`.
## Scope
This is phase 1 of rustfs/backlog#926: a global, per-process tier configured
by environment variable. Per-bucket durability tiers (bucket metadata +
admin API) are a separate follow-up phase.