Files
rustfs/crates/ecstore/src/disk/local.rs
T
houseme f7d2b25638 fix(ecstore): propagate disk delete/rename failures instead of swallowing them (#4546)
Two disk-layer correctness fixes in crates/ecstore/src/disk/local.rs.

move_to_trash (#948, ECA-07) previously handled only DiskFull in its
tail error block and let every other rename failure (EIO/EACCES/ENOTDIR/
cross-device) fall through to `return Ok(())`, so a delete that never
landed on a faulty disk was reported as success and the drive fault was
hidden from heal/offline logic. It now keeps the DiskFull in-place-remove
fallback, treats a missing source (FileNotFound) as benign, and
propagates every other error (already mapped by to_file_error, e.g.
I/O -> FaultyDisk, permission -> FileAccessDenied), matching MinIO's
deleteFile.

rename_file and rename_part (#960, ECA-19) directory (trailing-slash)
branch unconditionally removed the destination before rename and
propagated the resulting NotFound, so renaming a directory to a new
location always failed. Both now tolerate ErrorKind::NotFound on that
pre-rename remove and continue, matching MinIO's RenameFile.

Adds regression tests covering benign-missing-source, real-error
propagation, the unchanged move happy path, and directory rename to a
missing destination for both rename_file and rename_part.

Co-authored-by: heihutu <heihutu@gmail.com>
2026-07-08 18:27:11 +00:00

9458 lines
380 KiB
Rust
Raw Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
// Copyright 2024 RustFS Team
//
// Licensed under the Apache License, Version 2.0 (the "License");
// you may not use this file except in compliance with the License.
// You may obtain a copy of the License at
//
// http://www.apache.org/licenses/LICENSE-2.0
//
// Unless required by applicable law or agreed to in writing, software
// distributed under the License is distributed on an "AS IS" BASIS,
// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
// See the License for the specific language governing permissions and
// limitations under the License.
use crate::config::storageclass::DEFAULT_INLINE_BLOCK;
use crate::data_usage::local_snapshot::ensure_data_usage_layout;
use crate::disk::disk_store::get_object_disk_read_timeout;
use crate::disk::{
BUCKET_META_PREFIX, CHECK_PART_FILE_CORRUPT, CHECK_PART_FILE_NOT_FOUND, CHECK_PART_SUCCESS, CHECK_PART_UNKNOWN,
CHECK_PART_VOLUME_NOT_FOUND, CheckPartsResp, DeleteOptions, DiskAPI, DiskInfo, DiskInfoOptions, DiskLocation, DiskMetrics,
FileInfoVersions, FileReader, FileWriter, MmapCopyStageMetrics, RUSTFS_META_BUCKET, RUSTFS_META_TMP_BUCKET,
RUSTFS_META_TMP_DELETED_BUCKET, ReadMultipleReq, ReadMultipleResp, ReadOptions, RenameDataResp, STORAGE_FORMAT_FILE,
STORAGE_FORMAT_FILE_BACKUP, UpdateMetadataOpts, VolumeInfo, WalkDirOptions, conv_part_err_to_int,
endpoint::Endpoint,
error::{DiskError, Error, FileAccessDeniedWithContext, Result},
error_conv::{to_access_error, to_file_error, to_unformatted_disk_error, to_volume_error},
format::FormatV3,
fs::{
O_APPEND, O_CREATE, O_RDONLY, O_TRUNC, O_WRONLY, access, access_std, lstat, lstat_std, remove, remove_all_std,
remove_std, rename,
},
os,
os::{check_path_length, is_empty_dir, is_root_disk, rename_all, rename_all_ignore_missing_source},
};
use crate::erasure::coding::bitrot_verify;
use crate::runtime::sources as runtime_sources;
use bytes::Bytes;
use metrics::counter;
use parking_lot::{Mutex as ParkingLotMutex, RwLock as ParkingLotRwLock};
use rustfs_filemeta::{
Cache, FileInfo, FileInfoOpts, FileMeta, MetaCacheEntry, MetacacheWriter, ObjectPartInfo, Opts, RawFileInfo, UpdateFn,
get_file_info, read_xl_meta_no_data,
};
use rustfs_utils::HashAlgorithm;
use rustfs_utils::os::get_info;
use rustfs_utils::path::{
GLOBAL_DIR_SUFFIX, GLOBAL_DIR_SUFFIX_WITH_SLASH, SLASH_SEPARATOR, clean, decode_dir_object, encode_dir_object, has_suffix,
path_join, path_join_buf,
};
use std::collections::HashMap;
use std::collections::HashSet;
use std::fmt::Debug;
use std::io::{Error as IoError, SeekFrom};
#[cfg(target_os = "linux")]
use std::sync::atomic::AtomicBool;
use std::sync::atomic::{AtomicU32, Ordering};
use std::sync::{Arc, OnceLock};
use std::time::Duration;
use std::{
fs::Metadata,
path::{Path, PathBuf},
};
use time::OffsetDateTime;
use tokio::fs::{self, File};
use tokio::io::{AsyncRead, AsyncReadExt, AsyncSeekExt, AsyncWrite, AsyncWriteExt, ErrorKind, ReadBuf};
use tokio::sync::{Notify, RwLock};
use tokio::time::{Instant, Sleep, interval_at, timeout};
use tracing::{debug, error, info, warn};
use uuid::Uuid;
const DELETED_OBJECTS_CLEANUP_INTERVAL: Duration = Duration::from_secs(60 * 5);
const STALE_TMP_OBJECT_EXPIRY: Duration = Duration::from_secs(24 * 60 * 60);
const RUSTFS_META_TMP_OLD_BUCKET: &str = ".rustfs.sys/tmp-old";
const STARTUP_CLEANUP_WAIT_TIMEOUT: Duration = Duration::from_secs(2);
const ENV_BITROT_SIZE_MISMATCH_RETRY_COUNT: &str = "RUSTFS_BITROT_SIZE_MISMATCH_RETRY_COUNT";
const ENV_BITROT_SIZE_MISMATCH_RETRY_DELAY_MS: &str = "RUSTFS_BITROT_SIZE_MISMATCH_RETRY_DELAY_MS";
const DEFAULT_BITROT_SIZE_MISMATCH_RETRY_COUNT: u64 = 2;
const DEFAULT_BITROT_SIZE_MISMATCH_RETRY_DELAY_MS: u64 = 100;
const LOG_COMPONENT_ECSTORE: &str = "ecstore";
const LOG_SUBSYSTEM_DISK_LOCAL: &str = "disk_local";
const EVENT_DISK_LOCAL_STARTUP_CLEANUP: &str = "disk_local_startup_cleanup";
const EVENT_DISK_LOCAL_BACKGROUND_CLEANUP: &str = "disk_local_background_cleanup";
const EVENT_DISK_LOCAL_SCAN_FAILED: &str = "disk_local_scan_failed";
const EVENT_DISK_LOCAL_RENAME_REJECTED: &str = "disk_local_rename_rejected";
const EVENT_DISK_LOCAL_READ_VERSION_FALLBACK: &str = "disk_local_read_version_fallback";
#[cfg(target_os = "linux")]
const EVENT_DISK_LOCAL_DIRECT_IO_FALLBACK: &str = "disk_local_direct_io_fallback";
const EVENT_DISK_LOCAL_DELETE_FAILED: &str = "disk_local_delete_failed";
const EVENT_DISK_LOCAL_CHECK_PARTS: &str = "disk_local_check_parts";
const EVENT_DISK_LOCAL_ACCESS_FAILED: &str = "disk_local_access_failed";
const EVENT_DISK_LOCAL_VOLUME_SETUP_FAILED: &str = "disk_local_volume_setup_failed";
const EVENT_DISK_LOCAL_FORMAT_DECODE_FAILED: &str = "disk_local_format_decode_failed";
const METRIC_GET_OBJECT_MMAP_PAGE_FAULTS_TOTAL: &str = "rustfs_io_get_object_mmap_page_faults_total";
const METRIC_GET_OBJECT_DIRECT_READ_PAGE_FAULTS_TOTAL: &str = "rustfs_io_get_object_direct_read_page_faults_total";
#[inline(always)]
fn record_mmap_copy_stage(metrics: MmapCopyStageMetrics, stage: &'static str, started_at: Option<std::time::Instant>) {
if let Some(started_at) = started_at {
rustfs_io_metrics::record_get_object_stage_duration(metrics.path, stage, started_at.elapsed().as_secs_f64());
}
}
#[cfg(unix)]
#[derive(Clone, Copy, Debug, Default, PartialEq, Eq)]
struct MmapPageFaultCounts {
minor: libc::c_long,
major: libc::c_long,
}
#[cfg(unix)]
#[derive(Clone, Copy, Debug, Default, PartialEq, Eq)]
struct MmapPageFaultDelta {
minor: u64,
major: u64,
}
#[cfg(all(unix, any(target_os = "linux", target_os = "android")))]
fn mmap_rusage_who() -> libc::c_int {
libc::RUSAGE_THREAD
}
#[cfg(all(unix, not(any(target_os = "linux", target_os = "android"))))]
fn mmap_rusage_who() -> libc::c_int {
libc::RUSAGE_SELF
}
#[cfg(unix)]
// SAFETY: this allowance is limited to reading kernel-provided rusage data via
// libc; each unsafe operation below documents pointer validity and initialization.
#[allow(unsafe_code)]
fn read_mmap_page_fault_counts(enabled: bool) -> Option<MmapPageFaultCounts> {
if !enabled {
return None;
}
let mut usage = std::mem::MaybeUninit::<libc::rusage>::uninit();
// SAFETY: `getrusage` writes to the provided `rusage` pointer when it
// returns 0. The pointer is valid for writes and initialized only on success.
let rc = unsafe { libc::getrusage(mmap_rusage_who(), usage.as_mut_ptr()) };
if rc != 0 {
return None;
}
// SAFETY: `getrusage` returned success, so `usage` has been initialized.
let usage = unsafe { usage.assume_init() };
Some(MmapPageFaultCounts {
minor: usage.ru_minflt,
major: usage.ru_majflt,
})
}
#[cfg(unix)]
fn non_negative_fault_delta(before: libc::c_long, after: libc::c_long) -> u64 {
if after <= before {
return 0;
}
u64::try_from(after - before).unwrap_or(u64::MAX)
}
#[cfg(unix)]
fn mmap_page_fault_delta(before: Option<MmapPageFaultCounts>, after: Option<MmapPageFaultCounts>) -> MmapPageFaultDelta {
match (before, after) {
(Some(before), Some(after)) => MmapPageFaultDelta {
minor: non_negative_fault_delta(before.minor, after.minor),
major: non_negative_fault_delta(before.major, after.major),
},
_ => MmapPageFaultDelta::default(),
}
}
#[cfg(unix)]
fn record_mmap_page_fault_delta(path: &'static str, stage: &'static str, delta: MmapPageFaultDelta) {
if delta.minor > 0 {
counter!(
METRIC_GET_OBJECT_MMAP_PAGE_FAULTS_TOTAL,
"path" => path,
"stage" => stage,
"kind" => "minor",
)
.increment(delta.minor);
}
if delta.major > 0 {
counter!(
METRIC_GET_OBJECT_MMAP_PAGE_FAULTS_TOTAL,
"path" => path,
"stage" => stage,
"kind" => "major",
)
.increment(delta.major);
}
}
#[cfg(unix)]
fn record_direct_read_page_fault_delta(path: &'static str, stage: &'static str, delta: MmapPageFaultDelta) {
if delta.minor > 0 {
counter!(
METRIC_GET_OBJECT_DIRECT_READ_PAGE_FAULTS_TOTAL,
"path" => path,
"stage" => stage,
"kind" => "minor",
)
.increment(delta.minor);
}
if delta.major > 0 {
counter!(
METRIC_GET_OBJECT_DIRECT_READ_PAGE_FAULTS_TOTAL,
"path" => path,
"stage" => stage,
"kind" => "major",
)
.increment(delta.major);
}
}
/// Enable O_DIRECT for large sequential reads.
/// When enabled, shard reads bypass the page cache using O_DIRECT flag.
/// Requires aligned buffers (typically 512 bytes or 4096 bytes).
/// Default: false (uses page cache via mmap/pread).
const ENV_RUSTFS_OBJECT_DIRECT_IO_READ_ENABLE: &str = "RUSTFS_OBJECT_DIRECT_IO_READ_ENABLE";
const DEFAULT_RUSTFS_OBJECT_DIRECT_IO_READ_ENABLE: bool = false;
/// Minimum shard size threshold for O_DIRECT reads.
/// Only shards larger than this threshold will use O_DIRECT.
/// Default: 4MB.
const ENV_RUSTFS_OBJECT_DIRECT_IO_READ_THRESHOLD: &str = "RUSTFS_OBJECT_DIRECT_IO_READ_THRESHOLD";
const DEFAULT_RUSTFS_OBJECT_DIRECT_IO_READ_THRESHOLD: usize = 4 * 1024 * 1024;
/// Enable O_DIRECT for erasure shard / multipart part data writes (Linux only).
/// When enabled, `create_file` streams shard bytes straight to the device with
/// O_DIRECT, so the commit-point `sync_dir_files` fdatasync no longer flushes
/// ~2 MiB of dirty pages inside the `rename_data` critical section (it degrades
/// to a cheap metadata/device FLUSH). Aligned whole blocks are written direct;
/// the trailing sub-alignment remainder falls back to a buffered write after
/// clearing O_DIRECT (MinIO's recipe). Durability is unchanged: the file is
/// still fdatasynced at the commit point by the unchanged `sync_dir_files`.
/// EINVAL/EOPNOTSUPP (tmpfs, overlayfs, 9p, ...) latch the path off and fall
/// back to buffered writes for the whole disk. Non-Linux always falls back.
/// Default: false (buffered writes via the page cache, as before).
const ENV_RUSTFS_OBJECT_DIRECT_IO_WRITE_ENABLE: &str = "RUSTFS_OBJECT_DIRECT_IO_WRITE_ENABLE";
const DEFAULT_RUSTFS_OBJECT_DIRECT_IO_WRITE_ENABLE: bool = false;
const ENV_RUSTFS_OBJECT_MMAP_POPULATE_ENABLE: &str = "RUSTFS_OBJECT_MMAP_POPULATE_ENABLE";
const DEFAULT_RUSTFS_OBJECT_MMAP_POPULATE_ENABLE: bool = false;
const ENV_RUSTFS_OBJECT_MMAP_READ_METHOD: &str = "RUSTFS_OBJECT_MMAP_READ_METHOD";
const RUSTFS_OBJECT_MMAP_READ_METHOD_MMAP_COPY: &str = "mmap_copy";
const RUSTFS_OBJECT_MMAP_READ_METHOD_DIRECT_READ_COPY: &str = "direct_read_copy";
/// Legacy binary switch for commit-point durability (fsync writes and renames).
/// Kept for compatibility: `true` maps to the `strict` durability mode (the
/// default), `false` keeps its historical semantics of disabling every fsync
/// on this disk, system-critical metadata included. Superseded by
/// `RUSTFS_DURABILITY_MODE`, which takes precedence when both are set.
/// Default: true.
const ENV_RUSTFS_DRIVE_SYNC_ENABLE: &str = "RUSTFS_DRIVE_SYNC_ENABLE";
const DEFAULT_RUSTFS_DRIVE_SYNC_ENABLE: bool = true;
/// Durability tier for object data-path writes: `strict` (default) | `relaxed` | `none`.
/// See docs/operations/durability-modes.md for the power-loss guarantee matrix.
const ENV_RUSTFS_DURABILITY_MODE: &str = "RUSTFS_DURABILITY_MODE";
#[derive(Clone, Copy, Debug, PartialEq, Eq)]
enum LocalReadCopyMethod {
MmapCopy,
DirectReadCopy,
}
/// Check if O_DIRECT reads are enabled.
fn is_direct_io_read_enabled() -> bool {
rustfs_utils::get_env_bool(ENV_RUSTFS_OBJECT_DIRECT_IO_READ_ENABLE, DEFAULT_RUSTFS_OBJECT_DIRECT_IO_READ_ENABLE)
}
/// Check if O_DIRECT shard/part data writes are enabled.
fn is_direct_io_write_enabled() -> bool {
rustfs_utils::get_env_bool(ENV_RUSTFS_OBJECT_DIRECT_IO_WRITE_ENABLE, DEFAULT_RUSTFS_OBJECT_DIRECT_IO_WRITE_ENABLE)
}
const EVENT_DISK_LOCAL_DURABILITY_MODE: &str = "disk_local_durability_mode";
/// Process-wide durability tier for commit-point fsync work on the local disk.
///
/// `Strict` is the default and preserves the historical (fully synced) write
/// path bit for bit. The other tiers are opt-in and only relax the object
/// data path; writes committing into system-critical namespaces stay pinned
/// to `Strict` (see [`effective_durability`]), except under `LegacyOff`,
/// which keeps the exact historical semantics of
/// `RUSTFS_DRIVE_SYNC_ENABLE=false` (no fsync anywhere, system metadata
/// included) so existing deployments keep their behavior unchanged.
#[derive(Clone, Copy, Debug, PartialEq, Eq)]
pub(crate) enum DurabilityMode {
/// Every commit point is fsynced: shard/part contents, xl.meta contents,
/// rollback backups, and the directory entries of commit renames.
Strict,
/// Object payload bytes (erasure shard files, multipart part files) are
/// still fdatasynced before the commit rename, but metadata commits
/// (xl.meta contents, rollback backups, directory entries) are left to
/// the page cache. Aligned with MinIO's default durability posture.
Relaxed,
/// No fsync on the object data path at all. System-critical namespaces
/// are still pinned to `Strict`.
None,
/// Historical semantics of `RUSTFS_DRIVE_SYNC_ENABLE=false`: no fsync
/// anywhere, without the system-critical pinning. Only reachable through
/// the legacy switch; not exposed by `RUSTFS_DURABILITY_MODE`.
LegacyOff,
}
impl DurabilityMode {
pub(crate) fn parse(value: &str) -> Option<Self> {
match value.trim().to_ascii_lowercase().as_str() {
"strict" => Some(Self::Strict),
"relaxed" => Some(Self::Relaxed),
"none" => Some(Self::None),
_ => Option::None,
}
}
fn as_str(self) -> &'static str {
match self {
Self::Strict => "strict",
Self::Relaxed => "relaxed",
Self::None => "none",
Self::LegacyOff => "legacy-off",
}
}
/// Whether object payload bytes (erasure shard files, multipart part
/// files) must be fdatasynced at commit points.
fn syncs_data_shards(self) -> bool {
matches!(self, Self::Strict | Self::Relaxed)
}
/// Whether metadata commits must be fsynced: xl.meta contents, rollback
/// backups, and the directory entries created by commit renames.
fn syncs_commit_metadata(self) -> bool {
matches!(self, Self::Strict)
}
}
/// Pure resolution of the durability mode from configuration values.
///
/// `RUSTFS_DURABILITY_MODE` wins when set to a valid value; otherwise the
/// legacy `RUSTFS_DRIVE_SYNC_ENABLE` switch keeps its historical mapping
/// (`true` -> strict, `false` -> the old full-off semantics). The default is
/// strict.
fn resolve_durability_mode(mode_env: Option<String>, legacy_drive_sync_enabled: bool) -> DurabilityMode {
if let Some(raw) = mode_env {
if let Some(mode) = DurabilityMode::parse(&raw) {
return mode;
}
warn!(
event = EVENT_DISK_LOCAL_DURABILITY_MODE,
component = LOG_COMPONENT_ECSTORE,
subsystem = LOG_SUBSYSTEM_DISK_LOCAL,
value = %raw,
"Invalid RUSTFS_DURABILITY_MODE value; expected strict|relaxed|none, falling back to the legacy drive-sync switch"
);
}
if legacy_drive_sync_enabled {
DurabilityMode::Strict
} else {
DurabilityMode::LegacyOff
}
}
/// The configured durability mode, resolved from the environment once per
/// process and cached (the previous binary switch re-read the environment on
/// every call, i.e. a dozen times per PUT, and could even flip mid-operation).
pub(crate) fn durability_mode() -> DurabilityMode {
#[cfg(test)]
if let Some(mode) = durability_mode_override::get() {
return mode;
}
static MODE: OnceLock<DurabilityMode> = OnceLock::new();
*MODE.get_or_init(|| {
let mode = resolve_durability_mode(
rustfs_utils::get_env_opt_str(ENV_RUSTFS_DURABILITY_MODE),
rustfs_utils::get_env_bool(ENV_RUSTFS_DRIVE_SYNC_ENABLE, DEFAULT_RUSTFS_DRIVE_SYNC_ENABLE),
);
info!(
event = EVENT_DISK_LOCAL_DURABILITY_MODE,
component = LOG_COMPONENT_ECSTORE,
subsystem = LOG_SUBSYSTEM_DISK_LOCAL,
mode = mode.as_str(),
"Storage durability mode resolved"
);
mode
})
}
/// Test-only override for [`durability_mode`].
///
/// The production value is resolved from the environment once per process, so
/// tests exercising non-default tiers need a process-level override hook.
/// Setting an override serializes callers on a mutex so a relaxed-tier test
/// can never leak its mode into a parallel strict-tier test.
#[cfg(test)]
pub(crate) mod durability_mode_override {
use super::DurabilityMode;
use std::sync::{Mutex, MutexGuard, PoisonError, RwLock};
static OVERRIDE: RwLock<Option<DurabilityMode>> = RwLock::new(None);
static SERIAL: Mutex<()> = Mutex::new(());
pub(crate) fn get() -> Option<DurabilityMode> {
*OVERRIDE.read().unwrap_or_else(PoisonError::into_inner)
}
/// Holds the override (and the serialization lock) until dropped.
pub(crate) struct OverrideGuard {
_serial: MutexGuard<'static, ()>,
}
impl Drop for OverrideGuard {
fn drop(&mut self) {
*OVERRIDE.write().unwrap_or_else(PoisonError::into_inner) = None;
}
}
pub(crate) fn set(mode: DurabilityMode) -> OverrideGuard {
let serial = SERIAL.lock().unwrap_or_else(PoisonError::into_inner);
*OVERRIDE.write().unwrap_or_else(PoisonError::into_inner) = Some(mode);
OverrideGuard { _serial: serial }
}
}
/// Per-bucket durability overrides (HP-5 phase 2, rustfs/backlog#938).
///
/// The disk layer never loads bucket metadata itself: the bucket metadata
/// subsystem publishes the parsed override here whenever a bucket's cached
/// metadata is set, refreshed, or removed, so this registry follows exactly
/// the existing bucket-metadata cache invalidation semantics (immediate on
/// the node applying a config change, peer reload notification plus the
/// periodic refresh loop elsewhere). Lookups sit on the commit hot path, so
/// the empty-registry case (no bucket overrides configured anywhere — the
/// default) is a single relaxed atomic load and the phase 1 behavior is
/// preserved bit for bit.
pub(crate) mod bucket_durability {
use super::{
DurabilityMode, EVENT_DISK_LOCAL_DURABILITY_MODE, LOG_COMPONENT_ECSTORE, LOG_SUBSYSTEM_DISK_LOCAL, is_scratch_volume,
is_system_critical_volume,
};
use std::collections::HashMap;
use std::sync::atomic::{AtomicBool, Ordering};
use std::sync::{OnceLock, PoisonError, RwLock};
use tracing::{info, warn};
static OVERRIDES: OnceLock<RwLock<HashMap<String, DurabilityMode>>> = OnceLock::new();
/// Fast-path gate: false means "no override registered anywhere", which
/// keeps default deployments off the map lookup entirely.
static NON_EMPTY: AtomicBool = AtomicBool::new(false);
fn overrides() -> &'static RwLock<HashMap<String, DurabilityMode>> {
OVERRIDES.get_or_init(|| RwLock::new(HashMap::new()))
}
/// Publish (or clear, with `None`) the durability override for `bucket`.
///
/// System namespaces can never carry an override: they are pinned to
/// `strict` by [`super::effective_durability`], and any attempt to
/// register one is rejected here as defense in depth.
pub(crate) fn set(bucket: &str, mode: Option<DurabilityMode>) {
if bucket.is_empty() {
return;
}
if is_system_critical_volume(bucket) || is_scratch_volume(bucket) {
if mode.is_some() {
warn!(
event = EVENT_DISK_LOCAL_DURABILITY_MODE,
component = LOG_COMPONENT_ECSTORE,
subsystem = LOG_SUBSYSTEM_DISK_LOCAL,
bucket = %bucket,
"Rejected per-bucket durability override for a system namespace; it stays pinned to strict"
);
}
return;
}
// The legacy full-off switch is process-wide only and deliberately
// unreachable per bucket (`DurabilityMode::parse` never returns it).
let mode = mode.filter(|m| *m != DurabilityMode::LegacyOff);
let mut map = overrides().write().unwrap_or_else(PoisonError::into_inner);
let changed = match mode {
Some(mode) => map.insert(bucket.to_string(), mode) != Some(mode),
None => map.remove(bucket).is_some(),
};
NON_EMPTY.store(!map.is_empty(), Ordering::Release);
drop(map);
if changed {
info!(
event = EVENT_DISK_LOCAL_DURABILITY_MODE,
component = LOG_COMPONENT_ECSTORE,
subsystem = LOG_SUBSYSTEM_DISK_LOCAL,
bucket = %bucket,
mode = mode.map_or("inherit", |m| m.as_str()),
"Per-bucket durability override updated"
);
}
}
/// The override registered for `volume`, if any. `volume` is the commit
/// destination, so user buckets resolve by name while scratch and system
/// namespaces never match (they are refused by [`set`]).
pub(crate) fn lookup(volume: &str) -> Option<DurabilityMode> {
if !NON_EMPTY.load(Ordering::Acquire) {
return None;
}
overrides()
.read()
.unwrap_or_else(PoisonError::into_inner)
.get(volume)
.copied()
}
}
/// Whether `volume` stages in-flight user object data (`.rustfs.sys/tmp`,
/// `.rustfs.sys/multipart`, and their subtrees). These namespaces follow the
/// configured durability mode: their contents commit into user buckets and
/// are exactly the writes the relaxed tiers exist for.
fn is_scratch_volume(volume: &str) -> bool {
for scratch in [super::RUSTFS_META_TMP_BUCKET, super::RUSTFS_META_MULTIPART_BUCKET] {
if volume == scratch || volume.strip_prefix(scratch).is_some_and(|rest| rest.starts_with('/')) {
return true;
}
}
false
}
/// Whether writes committing into `volume` carry system-critical state:
/// format.json, IAM and cluster config, bucket metadata, and everything else
/// under `.rustfs.sys` (or `.minio.sys` during migration) outside the scratch
/// namespaces. Losing these can take out the whole deployment and they are
/// far off the object hot path, so they never follow a relaxed tier.
fn is_system_critical_volume(volume: &str) -> bool {
if is_scratch_volume(volume) {
return false;
}
for meta in [super::RUSTFS_META_BUCKET, super::MIGRATING_META_BUCKET] {
if volume == meta || volume.strip_prefix(meta).is_some_and(|rest| rest.starts_with('/')) {
return true;
}
}
false
}
/// Effective durability for writes that commit into `volume`.
///
/// Resolution order: system-critical volumes are pinned to `Strict`
/// regardless of any configuration; otherwise a per-bucket override
/// (published by the bucket metadata subsystem, see [`bucket_durability`])
/// wins over the process-wide mode; otherwise the process-wide mode applies.
/// The legacy full-off switch keeps its historical semantics: it is never
/// pinned and per-bucket overrides do not apply under it.
pub(crate) fn effective_durability(volume: &str) -> DurabilityMode {
let global = durability_mode();
if global == DurabilityMode::LegacyOff {
return global;
}
if is_system_critical_volume(volume) {
return DurabilityMode::Strict;
}
bucket_durability::lookup(volume).unwrap_or(global)
}
/// Get the O_DIRECT read threshold size.
fn get_direct_io_read_threshold() -> usize {
rustfs_utils::get_env_usize(ENV_RUSTFS_OBJECT_DIRECT_IO_READ_THRESHOLD, DEFAULT_RUSTFS_OBJECT_DIRECT_IO_READ_THRESHOLD)
}
fn should_populate_mmap_read(length: usize) -> bool {
length > 0 && rustfs_utils::get_env_bool(ENV_RUSTFS_OBJECT_MMAP_POPULATE_ENABLE, DEFAULT_RUSTFS_OBJECT_MMAP_POPULATE_ENABLE)
}
fn local_read_copy_method() -> LocalReadCopyMethod {
let method = rustfs_utils::get_env_str(ENV_RUSTFS_OBJECT_MMAP_READ_METHOD, RUSTFS_OBJECT_MMAP_READ_METHOD_MMAP_COPY);
match method.as_str() {
RUSTFS_OBJECT_MMAP_READ_METHOD_DIRECT_READ_COPY => LocalReadCopyMethod::DirectReadCopy,
_ => LocalReadCopyMethod::MmapCopy,
}
}
/// Runtime state for the true O_DIRECT read path (Linux only).
///
/// `supported` starts true and latches false on the first EINVAL/EOPNOTSUPP
/// from an O_DIRECT open/read (tmpfs, overlayfs, and 9p commonly reject the
/// flag); the path then permanently falls back to the buffered read methods
/// for this disk. `align` caches the DIO alignment probed from the backing
/// filesystem. O_DIRECT errors must never surface to callers: EINVAL maps to
/// `FileNotFound` in `to_file_error`, which would masquerade as a missing
/// shard and trigger spurious EC rebuilds.
#[cfg(target_os = "linux")]
#[derive(Debug)]
struct DirectIoReadState {
supported: AtomicBool,
align: OnceLock<usize>,
fallback_logged: AtomicBool,
}
#[cfg(target_os = "linux")]
impl DirectIoReadState {
fn new() -> Self {
Self {
supported: AtomicBool::new(true),
align: OnceLock::new(),
fallback_logged: AtomicBool::new(false),
}
}
}
#[cfg(target_os = "linux")]
const DEFAULT_DIRECT_IO_ALIGN: usize = 4096;
/// Probe the DIO alignment requirement for the file's filesystem via
/// statx STATX_DIOALIGN (kernel >= 6.1). Falls back to 4096, a safe upper
/// bound for 512e/4Kn devices, when the kernel or filesystem does not
/// report it.
#[cfg(target_os = "linux")]
fn probe_direct_io_align(file: &std::fs::File) -> usize {
use rustix::fs::{AtFlags, StatxFlags};
match rustix::fs::statx(file, "", AtFlags::EMPTY_PATH, StatxFlags::DIOALIGN) {
Ok(stx) => {
if StatxFlags::from_bits_retain(stx.stx_mask).contains(StatxFlags::DIOALIGN) {
let align = stx.stx_dio_mem_align.max(stx.stx_dio_offset_align) as usize;
if align.is_power_of_two() && align >= 512 {
return align;
}
}
DEFAULT_DIRECT_IO_ALIGN
}
Err(_) => DEFAULT_DIRECT_IO_ALIGN,
}
}
/// Heap buffer with explicit alignment for O_DIRECT reads.
#[cfg(target_os = "linux")]
struct AlignedBuf {
ptr: std::ptr::NonNull<u8>,
len: usize,
layout: std::alloc::Layout,
}
#[cfg(target_os = "linux")]
#[allow(unsafe_code)]
impl AlignedBuf {
fn new(len: usize, align: usize) -> std::io::Result<Self> {
debug_assert!(len > 0, "AlignedBuf must not be zero-sized");
let layout = std::alloc::Layout::from_size_align(len, align)
.map_err(|e| std::io::Error::new(std::io::ErrorKind::InvalidInput, e))?;
// SAFETY: `layout` has non-zero size (callers guarantee len > 0) and a
// valid power-of-two alignment enforced by Layout::from_size_align.
let ptr = unsafe { std::alloc::alloc_zeroed(layout) };
let ptr = std::ptr::NonNull::new(ptr).ok_or(std::io::ErrorKind::OutOfMemory)?;
Ok(Self { ptr, len, layout })
}
fn as_slice(&self) -> &[u8] {
// SAFETY: `ptr` is a live allocation of exactly `len` bytes owned by
// self, initialized to zero at allocation and only written via
// `as_mut_slice`.
unsafe { std::slice::from_raw_parts(self.ptr.as_ptr(), self.len) }
}
fn as_mut_slice(&mut self) -> &mut [u8] {
// SAFETY: as in `as_slice`, plus `&mut self` guarantees exclusivity.
unsafe { std::slice::from_raw_parts_mut(self.ptr.as_ptr(), self.len) }
}
}
#[cfg(target_os = "linux")]
#[allow(unsafe_code)]
impl Drop for AlignedBuf {
fn drop(&mut self) {
// SAFETY: `ptr`/`layout` come from the successful alloc_zeroed in new().
unsafe { std::alloc::dealloc(self.ptr.as_ptr(), self.layout) }
}
}
#[cfg(target_os = "linux")]
fn is_direct_io_unsupported(err: &std::io::Error) -> bool {
matches!(err.raw_os_error(), Some(libc::EINVAL) | Some(libc::EOPNOTSUPP))
}
/// True O_DIRECT positioned read: open with O_DIRECT, read the aligned
/// superset range into an aligned bounce buffer, then slice out the exact
/// logical range. Alignment padding never leaks to callers — BitrotReader
/// reads exact shard_size and would flag padded output as corruption.
///
/// Short reads are legal for O_DIRECT; the loop stops at EOF (res == 0).
/// A read that ends before covering the logical range is an error (the
/// caller has already validated `offset + length <= file size`, so this
/// only happens on concurrent truncation) and makes the caller fall back
/// to the buffered path.
#[cfg(target_os = "linux")]
fn pread_direct_aligned(file_path: &Path, offset: u64, length: usize, state: &DirectIoReadState) -> std::io::Result<Bytes> {
use std::os::unix::fs::{FileExt, OpenOptionsExt};
let file = std::fs::OpenOptions::new()
.read(true)
.custom_flags(rustix::fs::OFlags::DIRECT.bits() as i32)
.open(file_path)?;
let align = *state.align.get_or_init(|| probe_direct_io_align(&file));
let align_u64 = align as u64;
let aligned_offset = offset - (offset % align_u64);
let logical_start =
usize::try_from(offset - aligned_offset).map_err(|_| std::io::Error::from(std::io::ErrorKind::InvalidInput))?;
let logical_end = logical_start.checked_add(length).ok_or(std::io::ErrorKind::InvalidInput)?;
let aligned_len = logical_end.checked_add(align - 1).ok_or(std::io::ErrorKind::InvalidInput)? / align * align;
let mut buf = AlignedBuf::new(aligned_len, align)?;
let mut filled = 0usize;
while filled < aligned_len {
// `filled` stays a multiple of `align` except possibly at EOF, so
// both the buffer address and the file offset remain aligned.
let n = file.read_at(&mut buf.as_mut_slice()[filled..], aligned_offset + filled as u64)?;
if n == 0 {
break;
}
filled += n;
}
if filled < logical_end {
return Err(std::io::Error::new(std::io::ErrorKind::UnexpectedEof, "short O_DIRECT read"));
}
Ok(Bytes::copy_from_slice(&buf.as_slice()[logical_start..logical_end]))
}
// `AlignedBuf` uniquely owns a single heap allocation reached only through
// `&self`/`&mut self`; there is no interior mutability and no aliasing, so it
// is sound to move it across threads (into a `spawn_blocking` flush closure)
// and to share `&AlignedBuf` between threads.
#[cfg(target_os = "linux")]
#[allow(unsafe_code)]
// SAFETY: exclusive heap ownership, no aliasing (see the note above).
unsafe impl Send for AlignedBuf {}
#[cfg(target_os = "linux")]
#[allow(unsafe_code)]
// SAFETY: `&AlignedBuf` only exposes read-only access to an immutable buffer.
unsafe impl Sync for AlignedBuf {}
/// Runtime state for the true O_DIRECT write path (Linux only), mirroring
/// [`DirectIoReadState`].
///
/// `supported` starts true and latches false on the first EINVAL/EOPNOTSUPP
/// from an O_DIRECT open (tmpfs, overlayfs, and 9p commonly reject the flag);
/// `create_file` then permanently opens shard files buffered for this disk.
/// `align` caches the DIO alignment probed from the backing filesystem. As on
/// the read path, an O_DIRECT open error must never surface to callers: EINVAL
/// maps to `FileNotFound` in `to_file_error`, which would masquerade as a
/// missing shard and trigger spurious EC rebuilds.
#[cfg(target_os = "linux")]
#[derive(Debug)]
struct DirectIoWriteState {
supported: AtomicBool,
align: OnceLock<usize>,
fallback_logged: AtomicBool,
}
#[cfg(target_os = "linux")]
impl DirectIoWriteState {
fn new() -> Self {
Self {
supported: AtomicBool::new(true),
align: OnceLock::new(),
fallback_logged: AtomicBool::new(false),
}
}
}
/// Target staging size for O_DIRECT writes, rounded up to the DIO alignment.
/// Bounds the per-writer aligned bounce buffer and batches many shard blocks
/// into one positioned write to keep the syscall count low.
const DIRECT_WRITE_STAGING_BYTES: usize = 1024 * 1024;
/// Aligned bounce-buffer capacity for a given DIO alignment: the target staging
/// size rounded up to a whole multiple of `align` so the buffer address, every
/// flushed batch length, and every write offset stay alignment-correct.
/// Platform-independent (no O_DIRECT), so it is unit-tested on any host.
fn direct_write_staging_capacity(align: usize) -> usize {
debug_assert!(align.is_power_of_two() && align >= 512);
DIRECT_WRITE_STAGING_BYTES.div_ceil(align) * align
}
/// Split `filled` staged bytes into the alignment-sized prefix written with
/// O_DIRECT and the sub-alignment tail written buffered. Platform-independent,
/// so the tail-boundary math is unit-tested on any host.
fn direct_write_tail_split(filled: usize, align: usize) -> (usize, usize) {
let aligned = filled - (filled % align);
(aligned, filled - aligned)
}
/// Positioned write-all helper: retries short writes at increasing offsets.
///
/// Under O_DIRECT the buffer address, `offset`, and length must all be aligned;
/// callers guarantee that. After O_DIRECT has been cleared (tail path) there is
/// no alignment requirement.
#[cfg(target_os = "linux")]
fn pwrite_all(file: &std::fs::File, mut buf: &[u8], mut offset: u64) -> std::io::Result<()> {
use std::os::unix::fs::FileExt;
while !buf.is_empty() {
let n = file.write_at(buf, offset)?;
if n == 0 {
return Err(std::io::Error::new(
std::io::ErrorKind::WriteZero,
"O_DIRECT positioned write wrote 0 bytes",
));
}
buf = &buf[n..];
offset += n as u64;
}
Ok(())
}
/// Never let an O_DIRECT write error reach `to_file_error` as `InvalidInput`:
/// that maps to `FileNotFound` and would masquerade as a missing shard,
/// triggering a spurious EC rebuild (backlog#897 / issue correction #2). Any
/// EINVAL/EOPNOTSUPP surfacing from a flush is remapped to a generic error so
/// the write-quorum machinery treats it as the real write failure it is.
#[cfg(target_os = "linux")]
fn sanitize_direct_write_error(err: std::io::Error) -> std::io::Error {
if is_direct_io_unsupported(&err) {
std::io::Error::other(format!("O_DIRECT shard write failed: {err}"))
} else {
err
}
}
/// Owned O_DIRECT write state moved in and out of the `spawn_blocking` flush
/// closures so the reactor is never blocked on synchronous device I/O.
#[cfg(target_os = "linux")]
struct DirectWriteInner {
file: std::fs::File,
/// Aligned bounce buffer; its capacity (`buf.len`) is a whole multiple of
/// `align`.
buf: AlignedBuf,
/// Bytes currently staged in `buf` and not yet written to the device.
filled: usize,
/// Next file offset for an O_DIRECT positioned write; always a multiple of
/// `align` because every batch flushed before the tail is a whole multiple.
write_offset: u64,
align: usize,
direct_cleared: bool,
}
#[cfg(target_os = "linux")]
impl DirectWriteInner {
/// Flush a full staging batch (`filled == buf capacity`, a multiple of
/// `align`) straight to the device with O_DIRECT.
fn flush_batch(&mut self) -> std::io::Result<()> {
if self.filled == 0 {
return Ok(());
}
debug_assert_eq!(self.filled % self.align, 0, "batch flush must be alignment-sized");
pwrite_all(&self.file, &self.buf.as_slice()[..self.filled], self.write_offset).map_err(sanitize_direct_write_error)?;
self.write_offset += self.filled as u64;
self.filled = 0;
Ok(())
}
/// Final flush at shutdown: write the aligned prefix with O_DIRECT, then the
/// sub-alignment tail buffered after clearing O_DIRECT (MinIO's recipe; the
/// tail is not separately fsynced — the commit-point `sync_dir_files`
/// fdatasync covers the whole file, issue correction #5).
fn finish(&mut self) -> std::io::Result<()> {
let (aligned, remainder) = direct_write_tail_split(self.filled, self.align);
if aligned > 0 {
pwrite_all(&self.file, &self.buf.as_slice()[..aligned], self.write_offset).map_err(sanitize_direct_write_error)?;
self.write_offset += aligned as u64;
}
if remainder > 0 {
self.clear_direct()?;
// Snapshot the slice bounds first to avoid borrowing `self.buf`
// while `self.file` is borrowed immutably below.
let start = aligned;
let end = self.filled;
pwrite_all(&self.file, &self.buf.as_slice()[start..end], self.write_offset)?;
self.write_offset += remainder as u64;
}
self.filled = 0;
Ok(())
}
/// Drop the O_DIRECT flag from the open file so the unaligned tail can be
/// written through the page cache without an alignment fault.
fn clear_direct(&mut self) -> std::io::Result<()> {
if self.direct_cleared {
return Ok(());
}
let flags = rustix::fs::fcntl_getfl(&self.file).map_err(std::io::Error::from)?;
rustix::fs::fcntl_setfl(&self.file, flags - rustix::fs::OFlags::DIRECT).map_err(std::io::Error::from)?;
self.direct_cleared = true;
Ok(())
}
}
#[cfg(target_os = "linux")]
type DirectFlushHandle = tokio::task::JoinHandle<(DirectWriteInner, std::io::Result<()>)>;
#[cfg(target_os = "linux")]
enum DirectWriteState {
Idle(Option<DirectWriteInner>),
Busy(DirectFlushHandle),
}
/// Streaming O_DIRECT writer returned by `create_file` on Linux when the path
/// is enabled and supported.
///
/// Incoming bytes are memcpy'd into an aligned bounce buffer (cheap, on the
/// reactor); each full aligned batch and the shutdown tail are flushed on the
/// blocking pool so the reactor never stalls on synchronous device I/O — the
/// same offloading posture as the buffered `tokio::fs::File` writer it
/// replaces. Durability is unchanged: no fsync happens here; the commit-point
/// `sync_dir_files` fdatasync persists the file.
#[cfg(target_os = "linux")]
struct DirectWriter {
state: DirectWriteState,
shutdown_started: bool,
shutdown_done: bool,
}
#[cfg(target_os = "linux")]
impl DirectWriter {
fn new(inner: DirectWriteInner) -> Self {
Self {
state: DirectWriteState::Idle(Some(inner)),
shutdown_started: false,
shutdown_done: false,
}
}
/// Build a writer over an already-open plain file with a caller-chosen
/// alignment and staging capacity. Exercises the streaming/tail state
/// machine deterministically on CI filesystems that reject O_DIRECT
/// (tmpfs/overlayfs), where the production `open_direct_writer` would latch
/// off; `write_at`/`fcntl` behave identically on a buffered file.
#[cfg(test)]
fn from_std_file_for_test(file: std::fs::File, align: usize, capacity: usize) -> Self {
assert_eq!(capacity % align, 0, "test capacity must be an alignment multiple");
let buf = AlignedBuf::new(capacity, align).expect("aligned buffer allocation");
Self::new(DirectWriteInner {
file,
buf,
filled: 0,
write_offset: 0,
align,
direct_cleared: false,
})
}
/// Drive an in-flight flush to completion, returning the recovered inner
/// state. Returns `Pending`/errors verbatim; on success the state is left
/// `Idle`.
fn poll_drive_busy(&mut self, cx: &mut std::task::Context<'_>) -> std::task::Poll<std::io::Result<()>> {
if let DirectWriteState::Busy(handle) = &mut self.state {
let (inner, res) = match std::task::ready!(std::pin::Pin::new(handle).poll(cx)) {
Ok(pair) => pair,
Err(join_err) => {
return std::task::Poll::Ready(Err(std::io::Error::other(format!("O_DIRECT flush task failed: {join_err}"))));
}
};
self.state = DirectWriteState::Idle(Some(inner));
if self.shutdown_started {
self.shutdown_done = true;
}
res?;
}
std::task::Poll::Ready(Ok(()))
}
}
#[cfg(target_os = "linux")]
impl AsyncWrite for DirectWriter {
fn poll_write(
self: std::pin::Pin<&mut Self>,
cx: &mut std::task::Context<'_>,
buf: &[u8],
) -> std::task::Poll<std::io::Result<usize>> {
let this = self.get_mut();
loop {
match &mut this.state {
DirectWriteState::Busy(_) => {
std::task::ready!(this.poll_drive_busy(cx))?;
}
DirectWriteState::Idle(inner_opt) => {
if buf.is_empty() {
return std::task::Poll::Ready(Ok(0));
}
let inner = inner_opt.as_mut().expect("idle direct writer must hold inner state");
let capacity = inner.buf.len;
let space = capacity - inner.filled;
let n = space.min(buf.len());
let start = inner.filled;
inner.buf.as_mut_slice()[start..start + n].copy_from_slice(&buf[..n]);
inner.filled += n;
if inner.filled == capacity {
let mut inner = inner_opt.take().expect("idle direct writer must hold inner state");
let handle = tokio::task::spawn_blocking(move || {
let res = inner.flush_batch();
(inner, res)
});
this.state = DirectWriteState::Busy(handle);
}
return std::task::Poll::Ready(Ok(n));
}
}
}
}
fn poll_flush(self: std::pin::Pin<&mut Self>, cx: &mut std::task::Context<'_>) -> std::task::Poll<std::io::Result<()>> {
// Only drive an in-flight batch to completion. Sub-alignment staged
// bytes cannot be flushed mid-stream (they would misalign the next
// O_DIRECT offset); they are written by `poll_shutdown`.
self.get_mut().poll_drive_busy(cx)
}
fn poll_shutdown(self: std::pin::Pin<&mut Self>, cx: &mut std::task::Context<'_>) -> std::task::Poll<std::io::Result<()>> {
let this = self.get_mut();
loop {
match &mut this.state {
DirectWriteState::Busy(_) => {
std::task::ready!(this.poll_drive_busy(cx))?;
}
DirectWriteState::Idle(inner_opt) => {
if this.shutdown_done {
return std::task::Poll::Ready(Ok(()));
}
let mut inner = inner_opt.take().expect("idle direct writer must hold inner state");
this.shutdown_started = true;
let handle = tokio::task::spawn_blocking(move || {
let res = inner.finish();
(inner, res)
});
this.state = DirectWriteState::Busy(handle);
}
}
}
}
}
/// Open a shard file for an O_DIRECT streaming write, probing DIO alignment on
/// the freshly created file. Returns `Ok(None)` (with the state latched off and
/// a one-time warning) when the filesystem rejects O_DIRECT, so the caller can
/// fall back to the buffered writer without ever surfacing EINVAL.
#[cfg(target_os = "linux")]
fn open_direct_writer(file_path: &Path, state: &DirectIoWriteState) -> Result<Option<DirectWriter>> {
use std::os::unix::fs::OpenOptionsExt;
let open_result = std::fs::OpenOptions::new()
.create(true)
.write(true)
.truncate(true)
.custom_flags(libc::O_DIRECT)
.open(file_path);
let file = match open_result {
Ok(file) => file,
Err(err) => {
if is_direct_io_unsupported(&err) {
state.supported.store(false, Ordering::Relaxed);
if !state.fallback_logged.swap(true, Ordering::Relaxed) {
warn!(
event = EVENT_DISK_LOCAL_DIRECT_IO_FALLBACK,
component = LOG_COMPONENT_ECSTORE,
subsystem = LOG_SUBSYSTEM_DISK_LOCAL,
path = %file_path.display(),
error = ?err,
"O_DIRECT write unavailable; falling back to buffered writes"
);
}
return Ok(None);
}
// A genuine open failure (permissions, disk full, ...): map it as a
// normal file error so callers see the real cause.
return Err(to_file_error(err).into());
}
};
let align = *state.align.get_or_init(|| probe_direct_io_align(&file));
let capacity = direct_write_staging_capacity(align);
let buf = match AlignedBuf::new(capacity, align) {
Ok(buf) => buf,
Err(err) => return Err(to_file_error(err).into()),
};
Ok(Some(DirectWriter::new(DirectWriteInner {
file,
buf,
filled: 0,
write_offset: 0,
align,
direct_cleared: false,
})))
}
#[cfg(unix)]
#[allow(unsafe_code)]
fn mmap_page_size() -> Result<u64> {
static PAGE_SIZE: OnceLock<Option<u64>> = OnceLock::new();
PAGE_SIZE
.get_or_init(|| {
// SAFETY: `sysconf(_SC_PAGESIZE)` has no pointer arguments and only
// queries process-global OS configuration.
let page_size = unsafe { libc::sysconf(libc::_SC_PAGESIZE) };
if page_size <= 0 {
return None;
}
u64::try_from(page_size).ok()
})
.ok_or_else(|| DiskError::other("failed to determine system page size"))
}
#[cfg(test)]
static RENAME_DATA_FAIL_BEFORE_OLD_METADATA_BACKUP: std::sync::Mutex<Option<String>> = std::sync::Mutex::new(None);
#[cfg(test)]
fn set_rename_data_fail_before_old_metadata_backup(dst_path: &str) {
*RENAME_DATA_FAIL_BEFORE_OLD_METADATA_BACKUP
.lock()
.expect("test failpoint lock should not be poisoned") = Some(dst_path.to_string());
}
#[cfg(test)]
fn should_fail_before_old_metadata_backup(dst_path: &str) -> bool {
let mut target = RENAME_DATA_FAIL_BEFORE_OLD_METADATA_BACKUP
.lock()
.expect("test failpoint lock should not be poisoned");
if target.as_deref() == Some(dst_path) {
target.take();
true
} else {
false
}
}
#[cfg(not(test))]
fn should_fail_before_old_metadata_backup(_dst_path: &str) -> bool {
false
}
/// Commit-sequence points where the rename_data crash-consistency harness
/// (rustfs/backlog#935, test plan in rustfs/backlog#896) can simulate an
/// abrupt power loss.
///
/// Unlike [`should_fail_before_old_metadata_backup`], which exercises the
/// graceful in-process rollback (delete the staged data dir, return an
/// error), a crash point models a hard power loss: the commit sequence stops
/// dead at the armed step with **no** cleanup, leaving the on-disk state
/// exactly as the preceding steps left it. The harness then reopens the disk
/// and asserts the raw state is coherent — the object reads back as either the
/// old version or the new version, never a mixed or corrupt one — without any
/// rollback code having run.
///
/// The variants are constructed at the real commit-path call sites in every
/// build, but the arming static and [`should_crash_rename_data_at`] are
/// `#[cfg(test)]`; in production the guard is a const-`false` no-op, so the
/// injection points compile away to nothing.
#[derive(Clone, Copy, Debug, PartialEq, Eq)]
pub(crate) enum RenameDataCrashPoint {
/// After the data dir has been renamed into its destination but before the
/// old-metadata rollback backup is written. xl.meta has not been committed
/// yet, so a crash here must leave the object readable as the old version.
AfterDataRename,
/// After the rollback backup is persisted, immediately before the xl.meta
/// commit rename that makes the new version visible. Still pre-commit, so a
/// crash here must also leave the object readable as the old version.
AfterBackupBeforeMetaCommit,
}
#[cfg(test)]
static RENAME_DATA_CRASH_POINT: std::sync::Mutex<Option<(RenameDataCrashPoint, String)>> = std::sync::Mutex::new(None);
/// Arm a one-shot crash injection: the next `rename_data` committing into
/// `dst_path` stops at `point`. Consumed on the first match so it never leaks
/// into an unrelated commit.
#[cfg(test)]
fn arm_rename_data_crash(point: RenameDataCrashPoint, dst_path: &str) {
*RENAME_DATA_CRASH_POINT
.lock()
.expect("test crash point lock should not be poisoned") = Some((point, dst_path.to_string()));
}
#[cfg(test)]
fn should_crash_rename_data_at(point: RenameDataCrashPoint, dst_path: &str) -> bool {
let mut armed = RENAME_DATA_CRASH_POINT
.lock()
.expect("test crash point lock should not be poisoned");
if armed.as_ref().is_some_and(|(p, path)| *p == point && path == dst_path) {
armed.take();
true
} else {
false
}
}
#[cfg(not(test))]
#[inline(always)]
fn should_crash_rename_data_at(_point: RenameDataCrashPoint, _dst_path: &str) -> bool {
false
}
fn log_startup_disk_io_error(stage: &str, path: &Path, err: &IoError) {
warn!(
event = EVENT_DISK_LOCAL_STARTUP_CLEANUP,
component = LOG_COMPONENT_ECSTORE,
subsystem = LOG_SUBSYSTEM_DISK_LOCAL,
stage,
path = %path.display(),
error_kind = ?err.kind(),
raw_os_error = ?err.raw_os_error(),
error = ?err,
state = "io_failed",
"Disk local startup filesystem operation failed"
);
}
fn log_startup_disk_error(stage: &str, path: &Path, err: &DiskError) {
if let DiskError::Io(io_err) = err {
log_startup_disk_io_error(stage, path, io_err);
return;
}
warn!(
event = EVENT_DISK_LOCAL_STARTUP_CLEANUP,
component = LOG_COMPONENT_ECSTORE,
subsystem = LOG_SUBSYSTEM_DISK_LOCAL,
stage,
path = %path.display(),
error = ?err,
state = "failed",
"Disk local startup operation failed"
);
}
#[derive(Debug, Clone)]
pub struct FormatInfo {
pub id: Option<Uuid>,
pub data: Bytes,
pub file_info: Option<Metadata>,
pub last_check: Option<OffsetDateTime>,
}
/// A helper enum to handle internal buffer types for writing data.
pub enum InternalBuf<'a> {
Ref(&'a [u8]),
Owned(Bytes),
}
/// Durability mode for `write_all_internal`.
///
/// `FileOnly` is reserved for tmp files the caller immediately renames away.
/// The safe-rename recipe (file content fdatasync -> rename -> fsync of the
/// destination parent directory) never needs the tmp directory entry to be
/// durable: the rename removes it, and a crash before the rename means the
/// operation was never acknowledged, so there is nothing to recover. Files
/// that stay where they are written (format.json via `write_all_public`, the
/// old-metadata rollback backup in `rename_data`, ...) must use `FileAndDir`
/// so both the contents and the new directory entry survive power loss.
#[derive(Clone, Copy, Debug, PartialEq, Eq)]
enum SyncMode {
/// No fsync; durability is not required (or drive sync is disabled).
None,
/// fdatasync the file contents, then fsync its parent directory.
FileAndDir,
/// fdatasync only the file contents. Only valid when the caller renames
/// the file away right after the write and fsyncs the rename
/// destination's parent directory before acknowledging.
FileOnly,
}
struct FileCacheReclaimWriter {
inner: File,
reclaim_len: usize,
reclaim_on_shutdown: bool,
reclaimed: bool,
}
struct FileCacheReclaimReader {
inner: File,
reclaim_offset: u64,
reclaim_len: usize,
reclaim_on_drop: bool,
reclaimed: bool,
}
struct StallTimeoutReader<R> {
inner: R,
timeout: Duration,
timer: Option<std::pin::Pin<Box<Sleep>>>,
}
impl<R> StallTimeoutReader<R> {
fn new(inner: R, timeout: Duration) -> Self {
Self {
inner,
timeout,
timer: None,
}
}
}
impl<R: AsyncRead + Unpin> AsyncRead for StallTimeoutReader<R> {
fn poll_read(
mut self: std::pin::Pin<&mut Self>,
cx: &mut std::task::Context<'_>,
buf: &mut ReadBuf<'_>,
) -> std::task::Poll<std::io::Result<()>> {
let filled_before = buf.filled().len();
match std::pin::Pin::new(&mut self.inner).poll_read(cx, buf) {
std::task::Poll::Ready(result) => {
self.timer = None;
std::task::Poll::Ready(result)
}
std::task::Poll::Pending => {
if self.timeout.is_zero() {
return std::task::Poll::Pending;
}
if self.timer.is_none() {
self.timer = Some(Box::pin(tokio::time::sleep(self.timeout)));
}
if let Some(timer) = self.timer.as_mut()
&& std::future::Future::poll(timer.as_mut(), cx).is_ready()
{
self.timer = None;
return std::task::Poll::Ready(Err(std::io::Error::new(
ErrorKind::TimedOut,
"local disk read stall timeout",
)));
}
if buf.filled().len() > filled_before {
self.timer = None;
}
std::task::Poll::Pending
}
}
}
}
fn record_file_cache_reclaim_success(kind: &'static str, reclaim_len: usize, started: std::time::Instant) {
counter!("rustfs_page_cache_reclaim_requests_total", "kind" => kind.to_string(), "result" => "ok".to_string()).increment(1);
counter!("rustfs_page_cache_reclaim_bytes_total", "kind" => kind.to_string()).increment(reclaim_len as u64);
metrics::histogram!("rustfs_page_cache_reclaim_duration_seconds", "kind" => kind.to_string())
.record(started.elapsed().as_secs_f64());
}
fn record_file_cache_reclaim_error(kind: &'static str) {
counter!("rustfs_page_cache_reclaim_requests_total", "kind" => kind.to_string(), "result" => "err".to_string()).increment(1);
}
fn bitrot_size_mismatch_retry_count() -> usize {
rustfs_utils::get_env_u64(ENV_BITROT_SIZE_MISMATCH_RETRY_COUNT, DEFAULT_BITROT_SIZE_MISMATCH_RETRY_COUNT) as usize
}
fn bitrot_size_mismatch_retry_delay() -> Duration {
Duration::from_millis(rustfs_utils::get_env_u64(
ENV_BITROT_SIZE_MISMATCH_RETRY_DELAY_MS,
DEFAULT_BITROT_SIZE_MISMATCH_RETRY_DELAY_MS,
))
}
fn is_bitrot_size_mismatch_error(err: &std::io::Error) -> bool {
err.to_string().contains("bitrot shard file size mismatch")
}
fn is_bitrot_verification_error(err: &std::io::Error) -> bool {
is_bitrot_size_mismatch_error(err) || err.to_string().contains("bitrot hash mismatch")
}
fn metacache_write_error(err: rustfs_filemeta::Error) -> DiskError {
let err = DiskError::from(err);
if err.contains_io_error_kind(ErrorKind::BrokenPipe) {
DiskError::metacache_output_stream_closed()
} else {
err
}
}
async fn write_metacache_obj<W>(out: &mut MetacacheWriter<W>, obj: &MetaCacheEntry) -> Result<()>
where
W: AsyncWrite + Unpin,
{
out.write_obj(obj).await.map_err(metacache_write_error)
}
impl FileCacheReclaimReader {
fn new(inner: File, reclaim_offset: u64, reclaim_len: usize, reclaim_on_drop: bool) -> Self {
#[cfg(target_os = "macos")]
if reclaim_on_drop {
let _ = set_fd_nocache(&inner);
}
Self {
inner,
reclaim_offset,
reclaim_len,
reclaim_on_drop,
reclaimed: false,
}
}
#[cfg(target_os = "linux")]
fn reclaim_file_cache(&mut self) -> std::io::Result<()> {
use core::num::NonZeroU64;
use rustix::fs::{Advice, fadvise};
if !self.reclaim_on_drop || self.reclaimed || self.reclaim_len == 0 {
return Ok(());
}
let started = std::time::Instant::now();
let reclaim_len =
NonZeroU64::new(self.reclaim_len as u64).expect("reclaim_len is guaranteed non-zero by the early return");
fadvise(&self.inner, self.reclaim_offset, Some(reclaim_len), Advice::DontNeed).map_err(std::io::Error::from)?;
self.reclaimed = true;
record_file_cache_reclaim_success("read", self.reclaim_len, started);
Ok(())
}
#[cfg(not(target_os = "linux"))]
fn reclaim_file_cache(&mut self) -> std::io::Result<()> {
Ok(())
}
}
#[cfg(target_os = "macos")]
#[allow(unsafe_code)]
fn set_fd_nocache(file: &File) -> std::io::Result<()> {
use std::os::fd::AsRawFd;
// SAFETY: `fcntl` is called on a valid file descriptor owned by `file`.
let ret = unsafe { libc::fcntl(file.as_raw_fd(), libc::F_NOCACHE, 1) };
if ret == -1 {
return Err(std::io::Error::last_os_error());
}
Ok(())
}
#[cfg(target_os = "macos")]
#[allow(unsafe_code)]
fn set_std_fd_nocache(file: &std::fs::File) -> std::io::Result<()> {
use std::os::fd::AsRawFd;
// SAFETY: `fcntl` is called on a valid file descriptor owned by `file`.
let ret = unsafe { libc::fcntl(file.as_raw_fd(), libc::F_NOCACHE, 1) };
if ret == -1 {
return Err(std::io::Error::last_os_error());
}
Ok(())
}
impl Drop for FileCacheReclaimReader {
fn drop(&mut self) {
if let Err(err) = self.reclaim_file_cache() {
record_file_cache_reclaim_error("read");
debug!(error = ?err, reclaim_offset = self.reclaim_offset, reclaim_len = self.reclaim_len, "failed to reclaim file cache after read");
}
}
}
impl AsyncRead for FileCacheReclaimReader {
fn poll_read(
mut self: std::pin::Pin<&mut Self>,
cx: &mut std::task::Context<'_>,
buf: &mut ReadBuf<'_>,
) -> std::task::Poll<std::io::Result<()>> {
std::pin::Pin::new(&mut self.inner).poll_read(cx, buf)
}
}
impl FileCacheReclaimWriter {
fn new(inner: File, reclaim_len: usize, reclaim_on_shutdown: bool) -> Self {
#[cfg(target_os = "macos")]
if reclaim_on_shutdown {
let _ = set_fd_nocache(&inner);
}
Self {
inner,
reclaim_len,
reclaim_on_shutdown,
reclaimed: false,
}
}
#[cfg(target_os = "linux")]
fn reclaim_file_cache(&mut self) -> std::io::Result<()> {
use core::num::NonZeroU64;
use rustix::fs::{Advice, fadvise};
if !self.reclaim_on_shutdown || self.reclaimed || self.reclaim_len == 0 {
return Ok(());
}
let started = std::time::Instant::now();
let reclaim_len =
NonZeroU64::new(self.reclaim_len as u64).expect("reclaim_len is guaranteed non-zero by the early return");
fadvise(&self.inner, 0, Some(reclaim_len), Advice::DontNeed).map_err(std::io::Error::from)?;
self.reclaimed = true;
record_file_cache_reclaim_success("write", self.reclaim_len, started);
Ok(())
}
#[cfg(not(target_os = "linux"))]
fn reclaim_file_cache(&mut self) -> std::io::Result<()> {
Ok(())
}
}
impl AsyncWrite for FileCacheReclaimWriter {
fn poll_write(
mut self: std::pin::Pin<&mut Self>,
cx: &mut std::task::Context<'_>,
buf: &[u8],
) -> std::task::Poll<std::io::Result<usize>> {
std::pin::Pin::new(&mut self.inner).poll_write(cx, buf)
}
fn poll_flush(mut self: std::pin::Pin<&mut Self>, cx: &mut std::task::Context<'_>) -> std::task::Poll<std::io::Result<()>> {
std::pin::Pin::new(&mut self.inner).poll_flush(cx)
}
fn poll_shutdown(
mut self: std::pin::Pin<&mut Self>,
cx: &mut std::task::Context<'_>,
) -> std::task::Poll<std::io::Result<()>> {
match std::pin::Pin::new(&mut self.inner).poll_shutdown(cx) {
std::task::Poll::Ready(Ok(())) => {
if let Err(err) = self.reclaim_file_cache() {
record_file_cache_reclaim_error("write");
debug!(error = ?err, reclaim_len = self.reclaim_len, "failed to reclaim file cache after write");
}
std::task::Poll::Ready(Ok(()))
}
other => other,
}
}
fn poll_write_vectored(
mut self: std::pin::Pin<&mut Self>,
cx: &mut std::task::Context<'_>,
bufs: &[std::io::IoSlice<'_>],
) -> std::task::Poll<std::io::Result<usize>> {
std::pin::Pin::new(&mut self.inner).poll_write_vectored(cx, bufs)
}
fn is_write_vectored(&self) -> bool {
self.inner.is_write_vectored()
}
}
fn should_reclaim_file_cache_after_write(file_size: i64) -> bool {
if file_size <= 0 {
return false;
}
if !rustfs_utils::get_env_bool(
rustfs_config::ENV_OBJECT_FILE_CACHE_RECLAIM_WRITE_ENABLE,
rustfs_config::DEFAULT_OBJECT_FILE_CACHE_RECLAIM_WRITE_ENABLE,
) {
return false;
}
let threshold = rustfs_utils::get_env_usize(
rustfs_config::ENV_OBJECT_FILE_CACHE_RECLAIM_THRESHOLD,
rustfs_config::DEFAULT_OBJECT_FILE_CACHE_RECLAIM_THRESHOLD,
);
file_size as usize >= threshold
}
fn should_reclaim_file_cache_after_read(length: usize) -> bool {
if length == 0 {
return false;
}
if !rustfs_utils::get_env_bool(
rustfs_config::ENV_OBJECT_FILE_CACHE_RECLAIM_READ_ENABLE,
rustfs_config::DEFAULT_OBJECT_FILE_CACHE_RECLAIM_READ_ENABLE,
) {
return false;
}
let threshold = rustfs_utils::get_env_usize(
rustfs_config::ENV_OBJECT_FILE_CACHE_RECLAIM_THRESHOLD,
rustfs_config::DEFAULT_OBJECT_FILE_CACHE_RECLAIM_THRESHOLD,
);
length >= threshold
}
/// Write-open semantics for [`LocalIoBackend::open_write`].
///
/// `Truncate` mirrors `DiskAPI::create_file` (O_CREATE|O_WRONLY|O_TRUNC, no
/// volume access check, cache-reclaim writer); `Append` mirrors
/// `DiskAPI::append_file` (O_CREATE|O_APPEND|O_WRONLY, volume access check).
/// The access-check asymmetry is preserved historical behavior.
#[derive(Clone, Copy, Debug)]
pub(crate) enum WriteMode {
Truncate { size_hint: i64 },
Append,
}
/// Local-disk file I/O backend behind [`LocalDisk`].
///
/// Models the real per-file operations of the `DiskAPI` hot path so an
/// alternative backend (e.g. a runtime-probed io_uring implementation) can be
/// swapped in without touching callers. The default [`StdBackend`] preserves
/// the pre-trait behavior byte-for-byte. Commit-point durability
/// (fdatasync -> rename -> fsync-dir in `rename_data`) is deliberately NOT
/// part of this trait.
#[async_trait::async_trait]
pub(crate) trait LocalIoBackend: Send + Sync + Debug + 'static {
/// Positioned whole-range read returning owned bytes
/// (mirrors `read_file_mmap_copy_with_metrics`).
async fn pread_bytes(
&self,
volume: &str,
path: &str,
offset: usize,
length: usize,
metrics: Option<MmapCopyStageMetrics>,
) -> Result<Bytes>;
/// Open a bounded streaming reader over `offset..offset+length`
/// (mirrors `read_file_stream`).
async fn open_read_stream(&self, volume: &str, path: &str, offset: usize, length: usize) -> Result<FileReader>;
/// Open a whole-file streaming reader (mirrors `read_file`).
async fn open_full_read(&self, volume: &str, path: &str) -> Result<FileReader>;
/// Open a writer (mirrors `create_file`/`append_file` per [`WriteMode`]).
async fn open_write(&self, volume: &str, path: &str, mode: WriteMode) -> Result<FileWriter>;
}
/// Default [`LocalIoBackend`]: tokio blocking-pool file I/O plus the
/// mmap-copy / direct-read-copy positioned read, moved verbatim from the
/// former `DiskAPI` method bodies on `LocalDisk`.
#[derive(Debug)]
pub(crate) struct StdBackend {
root: PathBuf,
#[cfg(target_os = "linux")]
direct_io: Arc<DirectIoReadState>,
#[cfg(target_os = "linux")]
direct_io_write: Arc<DirectIoWriteState>,
}
impl StdBackend {
pub(crate) fn new(root: PathBuf) -> Self {
Self {
root,
#[cfg(target_os = "linux")]
direct_io: Arc::new(DirectIoReadState::new()),
#[cfg(target_os = "linux")]
direct_io_write: Arc::new(DirectIoWriteState::new()),
}
}
async fn open_file(&self, path: impl AsRef<Path>, mode: usize, skip_parent: impl AsRef<Path>) -> Result<File> {
let mut skip_parent = skip_parent.as_ref();
if skip_parent.as_os_str().is_empty() {
skip_parent = self.root.as_path();
}
if let Some(parent) = path.as_ref().parent()
&& parent != skip_parent
{
os::make_dir_all(parent, skip_parent).await?;
}
let f = super::fs::open_file(path.as_ref(), mode).await.map_err(to_file_error)?;
Ok(f)
}
async fn open_file_read_only(&self, path: impl AsRef<Path>) -> Result<File> {
let f = super::fs::open_file(path.as_ref(), O_RDONLY).await.map_err(to_file_error)?;
Ok(f)
}
}
#[async_trait::async_trait]
impl LocalIoBackend for StdBackend {
/// File read using mmap-then-copy on Unix or efficient read on non-Unix.
// SAFETY: Unix unsafe calls in this function only query page size and mmap
// a read-only file region after bounds and alignment are validated.
#[allow(unsafe_code)]
async fn pread_bytes(
&self,
volume: &str,
path: &str,
offset: usize,
length: usize,
metrics: Option<MmapCopyStageMetrics>,
) -> Result<Bytes> {
let metrics = metrics.filter(|_| rustfs_io_metrics::get_stage_metrics_enabled());
let metrics_enabled = metrics.is_some();
let metadata_validate_start = metrics_enabled.then(std::time::Instant::now);
let Some(end_offset) = offset.checked_add(length) else {
if let Some(metrics) = metrics {
record_mmap_copy_stage(metrics, metrics.metadata_validate_stage, metadata_validate_start);
}
return Err(DiskError::FileCorrupt);
};
// Unix: use mmap to read the data (copies into Bytes for safe ownership)
// Non-Unix: fall back to efficient read
#[cfg(unix)]
{
use memmap2::MmapOptions;
use std::time::{Duration as StdDuration, Instant as StdInstant};
struct MmapCopyReadResult {
bytes: Bytes,
access_check_duration: StdDuration,
path_resolve_duration: StdDuration,
metadata_lookup_duration: StdDuration,
metadata_validate_duration: StdDuration,
file_open_duration: StdDuration,
mmap_map_duration: StdDuration,
mmap_copy_duration: StdDuration,
direct_read_copy_duration: StdDuration,
mmap_map_fault_delta: MmapPageFaultDelta,
mmap_copy_fault_delta: MmapPageFaultDelta,
direct_read_copy_fault_delta: MmapPageFaultDelta,
blocking_task_duration: StdDuration,
used_direct_io: bool,
}
enum MmapCopyReadError {
Disk(DiskError),
OutOfBounds { actual_size: u64 },
}
impl From<DiskError> for MmapCopyReadError {
fn from(err: DiskError) -> Self {
Self::Disk(err)
}
}
let start = StdInstant::now();
let root = self.root.clone();
let volume_owned = volume.to_owned();
let path_owned = path.to_owned();
let should_reclaim_after_read = should_reclaim_file_cache_after_read(length);
let should_populate_mmap_read = should_populate_mmap_read(length);
let read_copy_method = local_read_copy_method();
#[cfg(target_os = "linux")]
let direct_io_eligible = is_direct_io_read_enabled() && length > 0 && length >= get_direct_io_read_threshold();
#[cfg(target_os = "linux")]
let direct_io_state = self.direct_io.clone();
let offset_u64 = u64::try_from(offset).map_err(|_| DiskError::FileCorrupt)?;
let end_offset_u64 = u64::try_from(end_offset).map_err(|_| DiskError::FileCorrupt)?;
let blocking_wait_start = metrics_enabled.then(std::time::Instant::now);
let read_result = tokio::task::spawn_blocking(move || {
let blocking_task_start = metrics_enabled.then(StdInstant::now);
let access_check_start = metrics_enabled.then(StdInstant::now);
let volume_dir = local_disk_bucket_path(&root, &volume_owned)?;
if !skip_access_checks(&volume_owned) {
access_std(&volume_dir).map_err(|e| DiskError::from(to_access_error(e, DiskError::VolumeAccessDenied)))?;
}
let access_check_duration = access_check_start.map_or(StdDuration::ZERO, |started_at| started_at.elapsed());
let path_resolve_start = metrics_enabled.then(StdInstant::now);
let file_path = local_disk_object_path(&root, &volume_owned, &path_owned)?;
check_path_length(file_path.to_string_lossy().as_ref())?;
let path_resolve_duration = path_resolve_start.map_or(StdDuration::ZERO, |started_at| started_at.elapsed());
let file_open_start = metrics_enabled.then(StdInstant::now);
let mut file = std::fs::File::open(&file_path).map_err(DiskError::from)?;
let file_open_duration = file_open_start.map_or(StdDuration::ZERO, |started_at| started_at.elapsed());
let metadata_lookup_start = metrics_enabled.then(StdInstant::now);
let meta = file.metadata().map_err(DiskError::from)?;
let metadata_lookup_duration = metadata_lookup_start.map_or(StdDuration::ZERO, |started_at| started_at.elapsed());
let metadata_validate_start = metrics_enabled.then(StdInstant::now);
if meta.len() < end_offset_u64 {
return Err(MmapCopyReadError::OutOfBounds { actual_size: meta.len() });
}
let metadata_validate_duration =
metadata_validate_start.map_or(StdDuration::ZERO, |started_at| started_at.elapsed());
#[cfg(target_os = "macos")]
if should_reclaim_after_read {
let _ = set_std_fd_nocache(&file);
}
let mut mmap_map_duration = StdDuration::ZERO;
let mut mmap_copy_duration = StdDuration::ZERO;
let mut direct_read_copy_duration = StdDuration::ZERO;
let mut mmap_map_fault_delta = MmapPageFaultDelta::default();
let mut mmap_copy_fault_delta = MmapPageFaultDelta::default();
let mut direct_read_copy_fault_delta = MmapPageFaultDelta::default();
let mut _reclaim_offset = offset_u64;
let mut _reclaim_len = length;
#[cfg(target_os = "linux")]
let mut direct_io_bytes: Option<Bytes> = None;
#[cfg(not(target_os = "linux"))]
let direct_io_bytes: Option<Bytes> = None;
#[cfg(target_os = "linux")]
if direct_io_eligible && direct_io_state.supported.load(Ordering::Relaxed) {
let direct_start = metrics_enabled.then(StdInstant::now);
let direct_faults_before = read_mmap_page_fault_counts(metrics_enabled);
match pread_direct_aligned(&file_path, offset_u64, length, &direct_io_state) {
Ok(bytes) => {
let direct_faults_after = read_mmap_page_fault_counts(metrics_enabled);
direct_read_copy_duration = direct_start.map_or(StdDuration::ZERO, |started_at| started_at.elapsed());
direct_read_copy_fault_delta = mmap_page_fault_delta(direct_faults_before, direct_faults_after);
direct_io_bytes = Some(bytes);
}
Err(err) => {
// Never surface O_DIRECT errors: EINVAL maps to
// FileNotFound in to_file_error and would trigger a
// spurious EC rebuild. Latch off on unsupported
// filesystems; otherwise retry buffered this once.
if is_direct_io_unsupported(&err) {
direct_io_state.supported.store(false, Ordering::Relaxed);
}
if !direct_io_state.fallback_logged.swap(true, Ordering::Relaxed) {
warn!(
event = EVENT_DISK_LOCAL_DIRECT_IO_FALLBACK,
component = LOG_COMPONENT_ECSTORE,
subsystem = LOG_SUBSYSTEM_DISK_LOCAL,
path = %file_path.display(),
error = ?err,
"O_DIRECT read unavailable; falling back to buffered reads"
);
}
}
}
}
let used_direct_io = direct_io_bytes.is_some();
let bytes = if let Some(bytes) = direct_io_bytes {
bytes
} else {
match read_copy_method {
LocalReadCopyMethod::MmapCopy => {
// mmap offsets on Unix must be page-size aligned. Align the
// mapping down to the nearest page boundary, then slice out the
// originally requested logical range.
let page_size = mmap_page_size()?;
let aligned_offset = offset_u64 - (offset_u64 % page_size);
let logical_offset = usize::try_from(offset_u64 - aligned_offset)
.map_err(|_| DiskError::other("mmap offset overflow"))?;
let map_len = logical_offset
.checked_add(length)
.ok_or_else(|| DiskError::other("mmap length overflow"))?;
_reclaim_offset = aligned_offset;
_reclaim_len = map_len;
// SAFETY: The file is opened as read-only, and we're mapping a region
// that we've already verified exists and is within file bounds. The
// file offset passed to mmap is page-size aligned as required on Unix.
let mmap_map_start = metrics_enabled.then(StdInstant::now);
let mmap_map_faults_before = read_mmap_page_fault_counts(metrics_enabled);
let mut mmap_options = MmapOptions::new();
mmap_options.offset(aligned_offset).len(map_len);
if should_populate_mmap_read {
mmap_options.populate();
}
let mmap = unsafe { mmap_options.map(&file) }.map_err(DiskError::other)?;
let mmap_map_faults_after = read_mmap_page_fault_counts(metrics_enabled);
mmap_map_duration = mmap_map_start.map_or(StdDuration::ZERO, |started_at| started_at.elapsed());
mmap_map_fault_delta = mmap_page_fault_delta(mmap_map_faults_before, mmap_map_faults_after);
// Copy only the requested logical range into a Bytes buffer. This
// avoids undefined behavior from treating OS-managed mmap memory as
// allocator-managed Vec storage, at the cost of an extra copy.
let end = logical_offset
.checked_add(length)
.ok_or_else(|| DiskError::other("mmap slice length overflow"))?;
let mmap_copy_start = metrics_enabled.then(StdInstant::now);
let mmap_copy_faults_before = read_mmap_page_fault_counts(metrics_enabled);
let bytes = Bytes::copy_from_slice(&mmap[logical_offset..end]);
let mmap_copy_faults_after = read_mmap_page_fault_counts(metrics_enabled);
mmap_copy_duration = mmap_copy_start.map_or(StdDuration::ZERO, |started_at| started_at.elapsed());
mmap_copy_fault_delta = mmap_page_fault_delta(mmap_copy_faults_before, mmap_copy_faults_after);
bytes
}
LocalReadCopyMethod::DirectReadCopy => {
use std::io::{Read as _, Seek as _};
let direct_read_copy_start = metrics_enabled.then(StdInstant::now);
let direct_read_copy_faults_before = read_mmap_page_fault_counts(metrics_enabled);
file.seek(SeekFrom::Start(offset_u64)).map_err(DiskError::from)?;
let mut buffer = vec![0; length];
file.read_exact(&mut buffer).map_err(DiskError::from)?;
let direct_read_copy_faults_after = read_mmap_page_fault_counts(metrics_enabled);
direct_read_copy_duration =
direct_read_copy_start.map_or(StdDuration::ZERO, |started_at| started_at.elapsed());
direct_read_copy_fault_delta =
mmap_page_fault_delta(direct_read_copy_faults_before, direct_read_copy_faults_after);
Bytes::from(buffer)
}
}
};
#[cfg(target_os = "linux")]
if should_reclaim_after_read && _reclaim_len > 0 {
use core::num::NonZeroU64;
use rustix::fs::{Advice, fadvise};
let reclaim_len = NonZeroU64::new(
u64::try_from(_reclaim_len).map_err(|_| DiskError::other("read reclaim length overflow"))?,
)
.ok_or_else(|| DiskError::other("read reclaim length overflow"))?;
fadvise(&file, _reclaim_offset, Some(reclaim_len), Advice::DontNeed)
.map_err(std::io::Error::from)
.map_err(DiskError::from)?;
}
let blocking_task_duration = blocking_task_start.map_or(StdDuration::ZERO, |started_at| started_at.elapsed());
Ok::<MmapCopyReadResult, MmapCopyReadError>(MmapCopyReadResult {
bytes,
access_check_duration,
path_resolve_duration,
metadata_lookup_duration,
metadata_validate_duration,
file_open_duration,
mmap_map_duration,
mmap_copy_duration,
direct_read_copy_duration,
mmap_map_fault_delta,
mmap_copy_fault_delta,
direct_read_copy_fault_delta,
blocking_task_duration,
used_direct_io,
})
})
.await
.map_err(DiskError::from)
.map_err(MmapCopyReadError::Disk)
.and_then(|result| result);
if let Some(metrics) = metrics {
record_mmap_copy_stage(metrics, metrics.blocking_wait_stage, blocking_wait_start);
}
let read_result = match read_result {
Ok(read_result) => read_result,
Err(MmapCopyReadError::Disk(err)) => return Err(err),
Err(MmapCopyReadError::OutOfBounds { actual_size }) => {
error!(
event = EVENT_DISK_LOCAL_READ_VERSION_FALLBACK,
component = LOG_COMPONENT_ECSTORE,
subsystem = LOG_SUBSYSTEM_DISK_LOCAL,
volume,
path,
offset,
length,
actual_size,
reason = "read_file_mmap_copy_out_of_bounds",
"Disk local read fallback failed"
);
return Err(DiskError::FileCorrupt);
}
};
if metrics_enabled && let Some(metrics) = metrics {
rustfs_io_metrics::record_get_object_stage_duration(
metrics.path,
metrics.blocking_task_stage,
read_result.blocking_task_duration.as_secs_f64(),
);
rustfs_io_metrics::record_get_object_stage_duration(
metrics.path,
metrics.access_check_stage,
read_result.access_check_duration.as_secs_f64(),
);
rustfs_io_metrics::record_get_object_stage_duration(
metrics.path,
metrics.path_resolve_stage,
read_result.path_resolve_duration.as_secs_f64(),
);
rustfs_io_metrics::record_get_object_stage_duration(
metrics.path,
metrics.metadata_lookup_stage,
read_result.metadata_lookup_duration.as_secs_f64(),
);
rustfs_io_metrics::record_get_object_stage_duration(
metrics.path,
metrics.metadata_validate_stage,
read_result.metadata_validate_duration.as_secs_f64(),
);
rustfs_io_metrics::record_get_object_stage_duration(
metrics.path,
metrics.file_open_stage,
read_result.file_open_duration.as_secs_f64(),
);
if read_result.used_direct_io {
rustfs_io_metrics::record_get_object_stage_duration(
metrics.path,
metrics.direct_read_copy_stage,
read_result.direct_read_copy_duration.as_secs_f64(),
);
record_direct_read_page_fault_delta(
metrics.path,
metrics.direct_read_copy_stage,
read_result.direct_read_copy_fault_delta,
);
} else {
match read_copy_method {
LocalReadCopyMethod::MmapCopy => {
rustfs_io_metrics::record_get_object_stage_duration(
metrics.path,
metrics.mmap_map_stage,
read_result.mmap_map_duration.as_secs_f64(),
);
rustfs_io_metrics::record_get_object_stage_duration(
metrics.path,
metrics.mmap_copy_stage,
read_result.mmap_copy_duration.as_secs_f64(),
);
record_mmap_page_fault_delta(metrics.path, metrics.mmap_map_stage, read_result.mmap_map_fault_delta);
record_mmap_page_fault_delta(
metrics.path,
metrics.mmap_copy_stage,
read_result.mmap_copy_fault_delta,
);
}
LocalReadCopyMethod::DirectReadCopy => {
rustfs_io_metrics::record_get_object_stage_duration(
metrics.path,
metrics.direct_read_copy_stage,
read_result.direct_read_copy_duration.as_secs_f64(),
);
record_direct_read_page_fault_delta(
metrics.path,
metrics.direct_read_copy_stage,
read_result.direct_read_copy_fault_delta,
);
}
}
}
}
let bytes = read_result.bytes;
// Log successful mmap read metrics
let duration_ms = start.elapsed().as_secs_f64() * 1000.0;
// Record mmap read metrics
rustfs_io_metrics::record_zero_copy_read(length, duration_ms);
debug!(
size = length,
duration_ms = duration_ms,
mmap_populate = should_populate_mmap_read,
read_copy_method = ?read_copy_method,
"mmap_read_success"
);
return Ok(bytes);
}
// Non-Unix fallback: efficient read into Bytes
#[cfg(not(unix))]
{
// Record zero-copy fallback
rustfs_io_metrics::record_zero_copy_fallback("non_unix_platform");
debug!(reason = "non_unix_platform", "zero_copy_fallback");
let access_check_start = metrics_enabled.then(std::time::Instant::now);
let volume_dir = local_disk_bucket_path(&self.root, volume)?;
if !skip_access_checks(volume) {
access(&volume_dir)
.await
.map_err(|e| to_access_error(e, DiskError::VolumeAccessDenied))?;
}
if let Some(metrics) = metrics {
record_mmap_copy_stage(metrics, metrics.access_check_stage, access_check_start);
}
let path_resolve_start = metrics_enabled.then(std::time::Instant::now);
let file_path = local_disk_object_path(&self.root, volume, path)?;
check_path_length(file_path.to_string_lossy().as_ref())?;
if let Some(metrics) = metrics {
record_mmap_copy_stage(metrics, metrics.path_resolve_stage, path_resolve_start);
}
let file_path_clone = file_path.clone();
let metadata_lookup_start = metrics_enabled.then(std::time::Instant::now);
let meta_result = tokio::task::spawn_blocking(move || std::fs::metadata(&file_path_clone).map_err(DiskError::from))
.await
.map_err(DiskError::from)
.and_then(|result| result);
if let Some(metrics) = metrics {
record_mmap_copy_stage(metrics, metrics.metadata_lookup_stage, metadata_lookup_start);
}
let meta = meta_result?;
let metadata_validate_start = metrics_enabled.then(std::time::Instant::now);
if meta.len() < end_offset as u64 {
if let Some(metrics) = metrics {
record_mmap_copy_stage(metrics, metrics.metadata_validate_stage, metadata_validate_start);
}
error!(
event = EVENT_DISK_LOCAL_READ_VERSION_FALLBACK,
component = LOG_COMPONENT_ECSTORE,
subsystem = LOG_SUBSYSTEM_DISK_LOCAL,
volume,
path,
offset,
length,
actual_size = meta.len(),
reason = "read_file_mmap_copy_out_of_bounds",
"Disk local read fallback failed"
);
return Err(DiskError::FileCorrupt);
}
if let Some(metrics) = metrics {
record_mmap_copy_stage(metrics, metrics.metadata_validate_stage, metadata_validate_start);
}
let mut f = self.open_file(file_path, O_RDONLY, volume_dir).await?;
if offset > 0 {
f.seek(SeekFrom::Start(offset as u64)).await?;
}
let mut buffer = vec![0; length];
f.read_exact(&mut buffer).await?;
Ok(Bytes::from(buffer))
}
}
async fn open_read_stream(&self, volume: &str, path: &str, offset: usize, length: usize) -> Result<FileReader> {
let volume_dir = local_disk_bucket_path(&self.root, volume)?;
if !skip_access_checks(volume) {
access(&volume_dir)
.await
.map_err(|e| to_access_error(e, DiskError::VolumeAccessDenied))?;
}
let file_path = local_disk_object_path(&self.root, volume, path)?;
check_path_length(file_path.to_string_lossy().as_ref())?;
let mut f = self.open_file_read_only(file_path).await?;
let meta = f.metadata().await?;
let end_offset = offset.checked_add(length).ok_or(DiskError::FileCorrupt)?;
if meta.len() < end_offset as u64 {
error!(
event = EVENT_DISK_LOCAL_READ_VERSION_FALLBACK,
component = LOG_COMPONENT_ECSTORE,
subsystem = LOG_SUBSYSTEM_DISK_LOCAL,
volume,
path,
offset,
length,
actual_size = meta.len(),
reason = "read_file_stream_out_of_bounds",
"Disk local read fallback failed"
);
return Err(DiskError::FileCorrupt);
}
if offset > 0 {
f.seek(SeekFrom::Start(offset as u64)).await?;
}
let reclaim_on_drop = should_reclaim_file_cache_after_read(length);
let reader = FileCacheReclaimReader::new(f, offset as u64, length, reclaim_on_drop);
Ok(Box::new(StallTimeoutReader::new(reader, get_object_disk_read_timeout())))
}
async fn open_full_read(&self, volume: &str, path: &str) -> Result<FileReader> {
let volume_dir = local_disk_bucket_path(&self.root, volume)?;
if !skip_access_checks(volume) {
access(&volume_dir)
.await
.map_err(|e| to_access_error(e, DiskError::VolumeAccessDenied))?;
}
let file_path = local_disk_object_path(&self.root, volume, path)?;
check_path_length(file_path.to_string_lossy().as_ref())?;
let f = self.open_file_read_only(file_path).await?;
Ok(Box::new(f))
}
async fn open_write(&self, volume: &str, path: &str, mode: WriteMode) -> Result<FileWriter> {
match mode {
WriteMode::Truncate { size_hint } => {
let volume_dir = local_disk_bucket_path(&self.root, volume)?;
let file_path = local_disk_object_path(&self.root, volume, path)?;
check_path_length(file_path.to_string_lossy().as_ref())?;
if let Some(parent) = file_path.parent() {
os::make_dir_all(parent, &volume_dir).await?;
}
// O_DIRECT streaming write (Linux, opt-in): shard bytes stream
// straight to the device so the commit-point fdatasync no longer
// flushes the whole shard's dirty pages inside the rename_data
// critical section. Latches off and falls back to the buffered
// writer on filesystems that reject O_DIRECT; never surfaces the
// EINVAL (which would masquerade as a missing shard).
#[cfg(target_os = "linux")]
if is_direct_io_write_enabled() && self.direct_io_write.supported.load(Ordering::Relaxed) {
let write_state = self.direct_io_write.clone();
let direct_path = file_path.clone();
let direct = tokio::task::spawn_blocking(move || open_direct_writer(&direct_path, &write_state))
.await
.map_err(|err| DiskError::other(format!("O_DIRECT open task failed: {err}")))??;
if let Some(writer) = direct {
return Ok(Box::new(writer));
}
}
// O_TRUNC: if a file already exists at this path, stale trailing bytes past
// the new content would otherwise survive and mismatch the metadata size.
let f = super::fs::open_file(&file_path, O_CREATE | O_WRONLY | O_TRUNC)
.await
.map_err(to_file_error)?;
let reclaim_on_shutdown = should_reclaim_file_cache_after_write(size_hint);
Ok(Box::new(FileCacheReclaimWriter::new(f, size_hint.max(0) as usize, reclaim_on_shutdown)))
}
WriteMode::Append => {
let volume_dir = local_disk_bucket_path(&self.root, volume)?;
if !skip_access_checks(volume) {
access(&volume_dir)
.await
.map_err(|e| to_access_error(e, DiskError::VolumeAccessDenied))?;
}
let file_path = local_disk_object_path(&self.root, volume, path)?;
check_path_length(file_path.to_string_lossy().as_ref())?;
let f = self.open_file(file_path, O_CREATE | O_APPEND | O_WRONLY, volume_dir).await?;
Ok(Box::new(f))
}
}
}
}
pub struct LocalDisk {
pub root: PathBuf,
pub format_path: PathBuf,
pub format_info: RwLock<FormatInfo>,
pub endpoint: Endpoint,
pub disk_info_cache: Arc<Cache<DiskInfo>>,
pub scanning: Arc<AtomicU32>,
pub rotational: bool,
pub fstype: String,
pub major: u64,
pub minor: u64,
pub nrrequests: u64,
scan_locks: Arc<ParkingLotMutex<HashSet<(String, String)>>>,
// Performance optimization fields
path_cache: Arc<ParkingLotRwLock<HashMap<String, PathBuf>>>,
current_dir: Arc<OnceLock<PathBuf>>,
// pub id: Mutex<Option<Uuid>>,
// pub format_data: Mutex<Vec<u8>>,
// pub format_file_info: Mutex<Option<Metadata>>,
// pub format_last_check: Mutex<Option<OffsetDateTime>>,
startup_cleanup_ready: Arc<AtomicU32>,
startup_cleanup_notify: Arc<Notify>,
exit_signal: Option<tokio::sync::broadcast::Sender<()>>,
io_backend: Arc<dyn LocalIoBackend>,
}
impl Drop for LocalDisk {
fn drop(&mut self) {
if let Some(exit_signal) = self.exit_signal.take() {
let _ = exit_signal.send(());
}
}
}
impl Debug for LocalDisk {
fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result {
f.debug_struct("LocalDisk")
.field("root", &self.root)
.field("format_path", &self.format_path)
.field("format_info", &self.format_info)
.field("endpoint", &self.endpoint)
.finish()
}
}
/// Resolve the local disk root path from an endpoint path.
///
/// Tries `canonicalize` first (fast path). On Windows, if canonicalization reports
/// `NotFound` for paths that may still be valid mount roots, falls back to
/// `absolutize` + metadata check to accept valid local directory roots that
/// don't support full canonicalization.
fn resolve_local_disk_root(ep_path: &str) -> Result<PathBuf> {
match rustfs_utils::canonicalize(ep_path) {
Ok(path) => Ok(path),
Err(err) => {
if err.kind() != ErrorKind::NotFound {
return Err(to_file_error(err).into());
}
#[cfg(windows)]
{
// On Windows, canonicalize can fail for ZFS volumes, junction points,
// subst drives, and other non-standard filesystem mounts. Try a fallback
// path resolution using absolutize + metadata check.
let absolute = match crate::disk::endpoint::windows_fallback_local_path(ep_path, &err, "local disk root") {
Ok(path) => path,
Err(_) => {
return Err(DiskError::VolumeNotFound);
}
};
match std::fs::metadata(&absolute) {
Ok(metadata) => {
if !metadata.is_dir() {
return Err(DiskError::DiskNotDir);
}
return Ok(absolute);
}
Err(meta_err) => {
if meta_err.kind() == ErrorKind::NotFound {
return Err(DiskError::VolumeNotFound);
}
return Err(to_file_error(meta_err).into());
}
}
}
#[cfg(not(windows))]
{
Err(DiskError::VolumeNotFound)
}
}
}
}
impl LocalDisk {
pub async fn new(ep: &Endpoint, cleanup: bool) -> Result<Self> {
debug!(
event = EVENT_DISK_LOCAL_STARTUP_CLEANUP,
component = LOG_COMPONENT_ECSTORE,
subsystem = LOG_SUBSYSTEM_DISK_LOCAL,
endpoint = %ep,
state = "create_started",
cleanup,
"Local disk creation started"
);
let endpoint_path = ep.get_file_path();
let root = resolve_local_disk_root(&endpoint_path).inspect_err(|err| {
log_startup_disk_error("resolve_local_disk_root", Path::new(&endpoint_path), err);
})?;
ensure_data_usage_layout(&root)
.await
.map_err(DiskError::from)
.inspect_err(|err| {
log_startup_disk_error("ensure_data_usage_layout", &root, err);
})?;
let startup_cleanup_ready = Arc::new(AtomicU32::new(u32::from(!cleanup)));
let startup_cleanup_notify = Arc::new(Notify::new());
if cleanup
&& let Err(err) =
Self::cleanup_tmp_on_startup(&root, startup_cleanup_ready.clone(), startup_cleanup_notify.clone()).await
{
startup_cleanup_ready.store(1, Ordering::Release);
startup_cleanup_notify.notify_waiters();
warn!(
event = EVENT_DISK_LOCAL_STARTUP_CLEANUP,
component = LOG_COMPONENT_ECSTORE,
subsystem = LOG_SUBSYSTEM_DISK_LOCAL,
root = ?root,
state = "failed",
error = ?err,
"Local disk startup cleanup failed"
);
}
// Use optimized path resolution instead of absolutize_virtually
let format_path = root.join(RUSTFS_META_BUCKET).join(super::FORMAT_CONFIG_FILE);
debug!(
event = EVENT_DISK_LOCAL_STARTUP_CLEANUP,
component = LOG_COMPONENT_ECSTORE,
subsystem = LOG_SUBSYSTEM_DISK_LOCAL,
root = ?root,
format_path = ?format_path,
state = "format_path_resolved",
"Local disk format path resolved"
);
let (format_data, format_meta) = read_file_exists(&format_path).await.inspect_err(|err| {
log_startup_disk_error("read_format_json", &format_path, err);
})?;
let mut id = None;
// let mut format_legacy = false;
let mut format_last_check = None;
if !format_data.is_empty() {
let s = format_data.as_ref();
let fm = FormatV3::try_from(s).map_err(Error::other)?;
let (set_idx, disk_idx) = fm.find_disk_index_by_disk_id(fm.erasure.this)?;
if set_idx as i32 != ep.set_idx || disk_idx as i32 != ep.disk_idx {
return Err(DiskError::InconsistentDisk);
}
id = Some(fm.erasure.this);
// format_legacy = fm.erasure.distribution_algo == DistributionAlgoVersion::V1;
format_last_check = Some(OffsetDateTime::now_utc());
}
let format_info = FormatInfo {
id,
data: format_data,
file_info: format_meta,
last_check: format_last_check,
};
let root_clone = root.clone();
let update_fn: UpdateFn<DiskInfo> = Box::new(move || {
let disk_id = id;
let root = root_clone.clone();
Box::pin(async move {
match get_disk_info(root.clone()).await {
Ok((info, is_root_disk)) => {
let physical_device_ids = match rustfs_utils::os::get_physical_device_ids(root.to_string_lossy().as_ref())
{
Ok(ids) => ids,
Err(err) => {
warn!(
event = EVENT_DISK_LOCAL_STARTUP_CLEANUP,
component = LOG_COMPONENT_ECSTORE,
subsystem = LOG_SUBSYSTEM_DISK_LOCAL,
root = ?root,
state = "physical_device_id_lookup_failed",
error = ?err,
"Disk local startup metadata lookup failed"
);
Vec::new()
}
};
// An erasure-set heal drops a marker on the disks it is
// rebuilding (see rustfs-heal); surface it so scanner
// coordination, lock selection and admin/metrics see
// the rebuild. Refreshed with this cache (~1s).
let healing =
tokio::fs::try_exists(root.join(super::RUSTFS_META_BUCKET).join(super::HEALING_MARKER_PATH))
.await
.unwrap_or(false);
let disk_info = DiskInfo {
total: info.total,
free: info.free,
used: info.used,
used_inodes: info.files.saturating_sub(info.ffree),
free_inodes: info.ffree,
major: info.major,
minor: info.minor,
fs_type: info.fstype,
root_disk: is_root_disk,
physical_device_ids,
id: disk_id,
healing,
..Default::default()
};
// if root {
// return Err(Error::new(DiskError::DriveIsRoot));
// }
Ok(disk_info)
}
Err(err) => Err(err.into()),
}
})
});
let cache = Cache::new(update_fn, Duration::from_secs(1), Opts::default());
// TODO: DIRECT support
// TODD: DiskInfo
let mut disk = Self {
root: root.clone(),
endpoint: ep.clone(),
format_path,
format_info: RwLock::new(format_info),
disk_info_cache: Arc::new(cache),
scanning: Arc::new(AtomicU32::new(0)),
rotational: Default::default(),
fstype: Default::default(),
minor: Default::default(),
major: Default::default(),
nrrequests: Default::default(),
scan_locks: Arc::new(ParkingLotMutex::new(HashSet::new())),
// // format_legacy,
// format_file_info: Mutex::new(format_meta),
// format_data: Mutex::new(format_data),
// format_last_check: Mutex::new(format_last_check),
path_cache: Arc::new(ParkingLotRwLock::new(HashMap::with_capacity(2048))),
current_dir: Arc::new(OnceLock::new()),
startup_cleanup_ready,
startup_cleanup_notify,
exit_signal: None,
io_backend: Arc::new(StdBackend::new(root.clone())),
};
let (info, _root) = get_disk_info(root.clone()).await.inspect_err(|err| {
log_startup_disk_error("get_disk_info", &root, err);
})?;
disk.major = info.major;
disk.minor = info.minor;
disk.fstype = info.fstype;
// if root {
// return Err(Error::new(DiskError::DriveIsRoot));
// }
if info.nrrequests > 0 {
disk.nrrequests = info.nrrequests;
}
if info.rotational {
disk.rotational = true;
}
disk.make_meta_volumes().await.inspect_err(|err| {
log_startup_disk_error("make_meta_volumes", &disk.root, err);
})?;
let (exit_tx, exit_rx) = tokio::sync::broadcast::channel(1);
disk.exit_signal = Some(exit_tx);
let root = disk.root.clone();
tokio::spawn(Self::cleanup_deleted_objects_loop(root, exit_rx));
debug!(
event = EVENT_DISK_LOCAL_STARTUP_CLEANUP,
component = LOG_COMPONENT_ECSTORE,
subsystem = LOG_SUBSYSTEM_DISK_LOCAL,
endpoint = %disk.endpoint,
root = ?disk.root,
state = "created",
"Local disk created"
);
Ok(disk)
}
async fn cleanup_deleted_objects_loop(root: PathBuf, mut exit_rx: tokio::sync::broadcast::Receiver<()>) {
let start_at = Instant::now() + DELETED_OBJECTS_CLEANUP_INTERVAL;
let mut interval = interval_at(start_at, DELETED_OBJECTS_CLEANUP_INTERVAL);
loop {
tokio::select! {
_ = interval.tick() => {
if let Err(err) = Self::cleanup_deleted_objects(root.clone()).await {
error!(
event = EVENT_DISK_LOCAL_BACKGROUND_CLEANUP,
component = LOG_COMPONENT_ECSTORE,
subsystem = LOG_SUBSYSTEM_DISK_LOCAL,
task = "deleted_objects",
state = "failed",
error = ?err,
"Disk local background cleanup failed"
);
}
if let Err(err) = Self::cleanup_stale_tmp_objects(root.clone()).await {
error!(
event = EVENT_DISK_LOCAL_BACKGROUND_CLEANUP,
component = LOG_COMPONENT_ECSTORE,
subsystem = LOG_SUBSYSTEM_DISK_LOCAL,
task = "stale_tmp_objects",
state = "failed",
error = ?err,
"Disk local background cleanup failed"
);
}
}
_ = exit_rx.recv() => {
info!(
event = EVENT_DISK_LOCAL_BACKGROUND_CLEANUP,
component = LOG_COMPONENT_ECSTORE,
subsystem = LOG_SUBSYSTEM_DISK_LOCAL,
task = "deleted_objects_loop",
state = "stopped",
"Disk local background cleanup loop stopped"
);
break;
}
}
}
}
fn meta_path(root: &Path, meta_path: &str) -> PathBuf {
#[cfg(windows)]
let meta_path = meta_path.replace('/', "\\");
#[cfg(not(windows))]
let meta_path = meta_path.to_string();
root.join(meta_path)
}
async fn cleanup_tmp_on_startup(
root: &Path,
startup_cleanup_ready: Arc<AtomicU32>,
startup_cleanup_notify: Arc<Notify>,
) -> Result<()> {
let tmp_path = Self::meta_path(root, RUSTFS_META_TMP_BUCKET);
let tmp_old_path = Self::meta_path(root, RUSTFS_META_TMP_OLD_BUCKET).join(Uuid::new_v4().to_string());
rename_all(&tmp_path, &tmp_old_path, root).await.inspect_err(|err| {
log_startup_disk_error("cleanup_tmp_rename_all", &tmp_path, err);
})?;
let tmp_deleted_path = Self::meta_path(root, RUSTFS_META_TMP_DELETED_BUCKET);
tokio::fs::create_dir_all(&tmp_deleted_path).await.inspect_err(|err| {
log_startup_disk_io_error("cleanup_tmp_create_deleted_dir", &tmp_deleted_path, err);
})?;
let tmp_old_root = Self::meta_path(root, RUSTFS_META_TMP_OLD_BUCKET);
tokio::spawn(async move {
if let Err(err) = tokio::fs::remove_dir_all(&tmp_old_root).await
&& err.kind() != ErrorKind::NotFound
{
log_startup_disk_io_error("cleanup_tmp_remove_old_dir", &tmp_old_root, &err);
}
startup_cleanup_ready.store(1, Ordering::Release);
startup_cleanup_notify.notify_waiters();
});
Ok(())
}
async fn wait_for_startup_cleanup(&self) {
if self.startup_cleanup_ready.load(Ordering::Acquire) != 0 {
return;
}
if wait_for_startup_cleanup_signal(
self.startup_cleanup_ready.as_ref(),
self.startup_cleanup_notify.as_ref(),
STARTUP_CLEANUP_WAIT_TIMEOUT,
)
.await
{
debug!(disk = %self.endpoint, "startup cleanup barrier released before walk_dir");
} else {
warn!(
event = EVENT_DISK_LOCAL_STARTUP_CLEANUP,
component = LOG_COMPONENT_ECSTORE,
subsystem = LOG_SUBSYSTEM_DISK_LOCAL,
disk = %self.endpoint,
timeout_ms = STARTUP_CLEANUP_WAIT_TIMEOUT.as_millis(),
state = "timed_out",
"Disk local startup cleanup barrier timed out"
);
}
}
async fn cleanup_stale_tmp_objects(root: PathBuf) -> Result<()> {
Self::cleanup_stale_tmp_objects_with_expiry(root, STALE_TMP_OBJECT_EXPIRY).await
}
async fn cleanup_stale_tmp_objects_with_expiry(root: PathBuf, expiry: Duration) -> Result<()> {
let tmp_path = Self::meta_path(&root, RUSTFS_META_TMP_BUCKET);
let mut entries = match fs::read_dir(&tmp_path).await {
Ok(entries) => entries,
Err(e) => {
if e.kind() == ErrorKind::NotFound {
return Ok(());
}
return Err(e.into());
}
};
while let Some(entry) = entries.next_entry().await? {
let name = entry.file_name().to_string_lossy().to_string();
if name.is_empty() || name == "." || name == ".." || name == ".trash" {
continue;
}
let file_type = entry.file_type().await?;
if !file_type.is_dir() {
continue;
}
let Some(age) = entry
.metadata()
.await?
.modified()
.ok()
.and_then(|modified| modified.elapsed().ok())
else {
continue;
};
if age <= expiry {
continue;
}
let target_path = Self::meta_path(&root, RUSTFS_META_TMP_DELETED_BUCKET).join(Uuid::new_v4().to_string());
rename_all(entry.path(), target_path, Self::meta_path(&root, RUSTFS_META_BUCKET)).await?;
}
Ok(())
}
async fn cleanup_deleted_objects(root: PathBuf) -> Result<()> {
let trash = Self::meta_path(&root, RUSTFS_META_TMP_DELETED_BUCKET);
let mut entries = match fs::read_dir(&trash).await {
Ok(entries) => entries,
Err(e) => {
if e.kind() == ErrorKind::NotFound {
return Ok(());
}
return Err(e.into());
}
};
while let Some(entry) = entries.next_entry().await? {
let name = entry.file_name().to_string_lossy().to_string();
if name.is_empty() || name == "." || name == ".." {
continue;
}
let file_type = entry.file_type().await?;
let path = trash.join(name);
if file_type.is_dir() {
if let Err(e) = tokio::fs::remove_dir_all(path).await
&& e.kind() != ErrorKind::NotFound
{
return Err(e.into());
}
} else if let Err(e) = tokio::fs::remove_file(path).await
&& e.kind() != ErrorKind::NotFound
{
return Err(e.into());
}
}
Ok(())
}
fn is_valid_volname(volname: &str) -> bool {
if volname.len() < 3 {
return false;
}
if cfg!(target_os = "windows") {
// Windows volume names must not include reserved characters.
// This regular expression matches disallowed characters.
if volname.contains('|')
|| volname.contains('<')
|| volname.contains('>')
|| volname.contains('?')
|| volname.contains('*')
|| volname.contains(':')
|| volname.contains('"')
|| volname.contains('\\')
{
return false;
}
} else {
// Non-Windows systems may require additional validation rules.
}
true
}
#[tracing::instrument(level = "debug", skip(self))]
async fn check_format_json(&self) -> Result<Metadata> {
let md = fs::metadata(&self.format_path).await.map_err(to_unformatted_disk_error)?;
Ok(md)
}
async fn make_meta_volumes(&self) -> Result<()> {
let buckets = format!("{RUSTFS_META_BUCKET}/{BUCKET_META_PREFIX}");
let multipart = format!("{}/{}", RUSTFS_META_BUCKET, "multipart");
let config = format!("{}/{}", RUSTFS_META_BUCKET, "config");
let tmp = format!("{}/{}", RUSTFS_META_BUCKET, "tmp");
let defaults = vec![
buckets.as_str(),
multipart.as_str(),
config.as_str(),
tmp.as_str(),
RUSTFS_META_TMP_DELETED_BUCKET,
];
self.make_volumes(defaults).await
}
// Optimized path resolution with caching
pub fn resolve_abs_path(&self, path: impl AsRef<Path>) -> Result<PathBuf> {
let path_ref = path.as_ref();
let path_str = path_ref.to_string_lossy();
// Fast cache read
{
let cache = self.path_cache.read();
if let Some(cached_path) = cache.get(path_str.as_ref()) {
return Ok(cached_path.clone());
}
}
// Calculate absolute path without using path_absolutize for better performance
let abs_path = if path_ref.is_absolute() {
path_ref.to_path_buf()
} else {
#[cfg(windows)]
{
self.root.join(path_str.replace('/', "\\"))
}
#[cfg(not(windows))]
{
self.root.join(path_ref)
}
};
// Normalize path components to avoid filesystem calls
let normalized = normalize_path_components(abs_path.as_path());
// Cache the result
{
let mut cache = self.path_cache.write();
// Simple cache size control
if cache.len() >= 4096 {
// Clear half the cache - simple eviction strategy
let keys_to_remove: Vec<_> = cache.keys().take(cache.len() / 2).cloned().collect();
for key in keys_to_remove {
cache.remove(&key);
}
}
cache.insert(path_str.into_owned(), normalized.clone());
}
Ok(normalized)
}
// Get the absolute path of an object
pub fn get_object_path(&self, bucket: &str, key: &str) -> Result<PathBuf> {
local_disk_object_path(&self.root, bucket, key)
}
// Get the absolute path of a bucket
pub fn get_bucket_path(&self, bucket: &str) -> Result<PathBuf> {
local_disk_bucket_path(&self.root, bucket)
}
// Check if a path is valid
fn check_valid_path<P: AsRef<Path>>(&self, path: P) -> Result<()> {
check_local_disk_valid_path(&self.root, path)
}
fn reject_symlink_components(&self, path: &Path) -> Result<()> {
reject_local_disk_symlink_components(&self.root, path)
}
fn try_acquire_scan_lock(&self, opts: &WalkDirOptions) -> Result<LocalScanLockGuard> {
let key = local_disk_scan_lock_key(&opts.bucket, &opts.base_dir, opts.filter_prefix.as_deref());
let mut scan_locks = self.scan_locks.lock();
if !scan_locks.insert(key.clone()) {
return Err(DiskError::DiskOngoingReq);
}
Ok(LocalScanLockGuard {
scan_locks: Arc::clone(&self.scan_locks),
key,
})
}
// Batch path generation with single lock acquisition
fn get_object_paths_batch(&self, requests: &[(String, String)]) -> Result<Vec<PathBuf>> {
let mut results = Vec::with_capacity(requests.len());
let mut cache_misses = Vec::new();
// First attempt to get all paths from cache
{
let cache = self.path_cache.read();
for (i, (bucket, key)) in requests.iter().enumerate() {
let cache_key = path_join_buf(&[bucket, key]);
if let Some(cached_path) = cache.get(&cache_key) {
results.push((i, cached_path.clone()));
} else {
cache_misses.push((i, bucket, key, cache_key));
}
}
}
// Handle cache misses
if !cache_misses.is_empty() {
let mut new_entries = Vec::new();
for (i, _bucket, _key, cache_key) in cache_misses {
#[cfg(windows)]
let path = self.root.join(cache_key.replace('/', "\\"));
#[cfg(not(windows))]
let path = self.root.join(&cache_key);
results.push((i, path.clone()));
new_entries.push((cache_key, path));
}
// Batch update cache
{
let mut cache = self.path_cache.write();
for (key, path) in new_entries {
cache.insert(key, path);
}
}
}
// Sort results back to original order
results.sort_by_key(|(i, _)| *i);
Ok(results.into_iter().map(|(_, path)| path).collect())
}
// /// Write to the filesystem atomically.
// /// This is done by first writing to a temporary location and then moving the file.
// pub(crate) async fn prepare_file_write<'a>(&self, path: &'a PathBuf) -> Result<FileWriter<'a>> {
// let tmp_path = self.get_object_path(RUSTFS_META_TMP_BUCKET, Uuid::new_v4().to_string().as_str())?;
// debug!("prepare_file_write tmp_path:{:?}, path:{:?}", &tmp_path, &path);
// let file = File::create(&tmp_path).await?;
// let writer = BufWriter::new(file);
// Ok(FileWriter {
// tmp_path,
// dest_path: path,
// writer,
// clean_tmp: true,
// })
// }
async fn move_to_trash(&self, delete_path: &PathBuf, recursive: bool, immediate_purge: bool) -> Result<()> {
// if recursive {
// remove_all_std(delete_path).map_err(to_volume_error)?;
// } else {
// remove_std(delete_path).map_err(to_file_error)?;
// }
// return Ok(());
// TODO: async notifications for disk space checks and trash cleanup
let trash_path = self.get_object_path(RUSTFS_META_TMP_DELETED_BUCKET, Uuid::new_v4().to_string().as_str())?;
// if let Some(parent) = trash_path.parent() {
// if !parent.exists() {
// fs::create_dir_all(parent).await?;
// }
// }
let err = if recursive {
rename_all_ignore_missing_source(delete_path, trash_path, self.get_bucket_path(RUSTFS_META_TMP_DELETED_BUCKET)?)
.await
.err()
} else {
match rename(&delete_path, &trash_path).await {
Ok(()) => None,
Err(err) if err.kind() == ErrorKind::NotFound => None,
Err(err) => Some(to_file_error(err).into()),
}
};
if immediate_purge || delete_path.to_string_lossy().ends_with(SLASH_SEPARATOR) {
let trash_path2 = self.get_object_path(RUSTFS_META_TMP_DELETED_BUCKET, Uuid::new_v4().to_string().as_str())?;
let _ = rename_all_ignore_missing_source(
encode_dir_object(delete_path.to_string_lossy().as_ref()),
trash_path2,
self.get_bucket_path(RUSTFS_META_TMP_DELETED_BUCKET)?,
)
.await;
}
if let Some(err) = err {
if err == Error::DiskFull {
// Out of space to stage the trash rename: fall back to an in-place
// remove and propagate any failure from that remove.
if recursive {
remove_all_std(delete_path).map_err(to_volume_error)?;
} else {
remove_std(delete_path).map_err(to_file_error)?;
}
return Ok(());
}
// A missing source is benign (the object is already gone). Both the
// recursive path (rename_all_ignore_missing_source) and the
// non-recursive NotFound arm above already fold that case into `None`,
// but keep the guard explicit so a genuine rename failure is never
// reported as success. Every other error is a real failure: propagate
// it (already mapped by to_file_error, e.g. I/O -> FaultyDisk,
// permission -> FileAccessDenied) so callers can surface a faulty disk
// and trigger heal, matching MinIO's deleteFile.
if err == Error::FileNotFound {
return Ok(());
}
return Err(err);
}
Ok(())
}
#[tracing::instrument(level = "debug", skip(self))]
#[async_recursion::async_recursion]
async fn delete_file(
&self,
base_path: &PathBuf,
delete_path: &PathBuf,
recursive: bool,
immediate_purge: bool,
) -> Result<()> {
// debug!("delete_file {:?}\n base_path:{:?}", &delete_path, &base_path);
if is_root_path(base_path) || is_root_path(delete_path) {
// debug!("delete_file skip {:?}", &delete_path);
return Ok(());
}
if !delete_path.starts_with(base_path) || base_path == delete_path {
// debug!("delete_file skip {:?}", &delete_path);
return Ok(());
}
if recursive {
self.move_to_trash(delete_path, recursive, immediate_purge).await?;
} else if delete_path.is_dir() {
// debug!("delete_file remove_dir {:?}", &delete_path);
if let Err(err) = fs::remove_dir(&delete_path).await {
// debug!("remove_dir err {:?} when {:?}", &err, &delete_path);
match err.kind() {
ErrorKind::NotFound => (),
ErrorKind::DirectoryNotEmpty => (),
kind => {
warn!(
event = EVENT_DISK_LOCAL_DELETE_FAILED,
component = LOG_COMPONENT_ECSTORE,
subsystem = LOG_SUBSYSTEM_DISK_LOCAL,
path = ?delete_path,
operation = "remove_dir",
error_kind = %kind,
"Disk local delete failed"
);
return Err(Error::other(FileAccessDeniedWithContext {
path: delete_path.clone(),
source: err,
}));
}
}
}
// debug!("delete_file remove_dir done {:?}", &delete_path);
} else if let Err(err) = fs::remove_file(&delete_path).await {
// debug!("remove_file err {:?} when {:?}", &err, &delete_path);
match err.kind() {
ErrorKind::NotFound => (),
_ => {
warn!(
event = EVENT_DISK_LOCAL_DELETE_FAILED,
component = LOG_COMPONENT_ECSTORE,
subsystem = LOG_SUBSYSTEM_DISK_LOCAL,
path = ?delete_path,
operation = "remove_file",
error = ?err,
"Disk local delete failed"
);
return Err(Error::other(FileAccessDeniedWithContext {
path: delete_path.clone(),
source: err,
}));
}
}
}
if let Some(dir_path) = delete_path.parent() {
Box::pin(self.delete_file(base_path, &PathBuf::from(dir_path), false, false)).await?;
}
// debug!("delete_file done {:?}", &delete_path);
Ok(())
}
/// read xl.meta raw data
#[tracing::instrument(level = "debug", skip(self, volume_dir, file_path))]
async fn read_raw(
&self,
bucket: &str,
volume_dir: impl AsRef<Path>,
file_path: impl AsRef<Path>,
read_data: bool,
) -> Result<(Vec<u8>, Option<OffsetDateTime>)> {
if file_path.as_ref().as_os_str().is_empty() {
return Err(DiskError::FileNotFound);
}
let meta_path = path_join(&[file_path.as_ref(), Path::new(STORAGE_FORMAT_FILE)]);
let res = {
if read_data {
self.read_all_data_with_dmtime(bucket, volume_dir, meta_path).await
} else {
match self.read_metadata_with_dmtime(meta_path).await {
Ok(res) => Ok(res),
Err(err) => {
if err == Error::FileNotFound
&& !skip_access_checks(volume_dir.as_ref().to_string_lossy().to_string().as_str())
&& let Err(e) = access(volume_dir.as_ref()).await
&& e.kind() == ErrorKind::NotFound
{
// warn!("read_metadata_with_dmtime os err {:?}", &aerr);
return Err(DiskError::VolumeNotFound);
}
Err(err)
}
}
}
};
let (buf, mtime) = res?;
if buf.is_empty() {
return Err(DiskError::FileNotFound);
}
Ok((buf, mtime))
}
#[cfg_attr(feature = "hotpath", hotpath::measure)]
async fn read_metadata_with_dmtime(&self, file_path: impl AsRef<Path>) -> Result<(Vec<u8>, Option<OffsetDateTime>)> {
check_path_length(file_path.as_ref().to_string_lossy().as_ref())?;
let mut f = super::fs::open_file(file_path.as_ref(), O_RDONLY)
.await
.map_err(to_file_error)?;
let meta = f.metadata().await.map_err(to_file_error)?;
if meta.is_dir() {
// fix use io::Error
return Err(Error::FileNotFound);
}
let size = meta.len() as usize;
let data = read_xl_meta_no_data(&mut f, size).await?;
let modtime = match meta.modified() {
Ok(md) => Some(OffsetDateTime::from(md)),
Err(_) => None,
};
Ok((data, modtime))
}
#[cfg_attr(feature = "hotpath", hotpath::measure)]
async fn read_all_data(&self, volume: &str, volume_dir: impl AsRef<Path>, file_path: impl AsRef<Path>) -> Result<Vec<u8>> {
// TODO: timeout support
let (data, _) = self.read_all_data_with_dmtime(volume, volume_dir, file_path).await?;
Ok(data)
}
#[tracing::instrument(level = "debug", skip(self, volume_dir, file_path))]
async fn read_all_data_with_dmtime(
&self,
volume: &str,
volume_dir: impl AsRef<Path>,
file_path: impl AsRef<Path>,
) -> Result<(Vec<u8>, Option<OffsetDateTime>)> {
let mut f = match super::fs::open_file(file_path.as_ref(), O_RDONLY).await {
Ok(f) => f,
Err(e) => {
if e.kind() == ErrorKind::NotFound
&& !skip_access_checks(volume)
&& let Err(er) = access(volume_dir.as_ref()).await
&& er.kind() == ErrorKind::NotFound
{
warn!(
event = EVENT_DISK_LOCAL_READ_VERSION_FALLBACK,
component = LOG_COMPONENT_ECSTORE,
subsystem = LOG_SUBSYSTEM_DISK_LOCAL,
reason = "read_all_data_with_dmtime_volume_not_found",
error = ?er,
"Disk local read fallback failed"
);
return Err(DiskError::VolumeNotFound);
}
return Err(to_file_error(e).into());
}
};
let meta = f.metadata().await.map_err(to_file_error)?;
if meta.is_dir() {
return Err(DiskError::FileNotFound);
}
let size = meta.len() as usize;
let mut bytes = Vec::new();
bytes.try_reserve_exact(size).map_err(Error::other)?;
f.read_to_end(&mut bytes).await.map_err(to_file_error)?;
let modtime = match meta.modified() {
Ok(md) => Some(OffsetDateTime::from(md)),
Err(_) => None,
};
Ok((bytes, modtime))
}
async fn delete_versions_internal(&self, volume: &str, path: &str, fis: &[FileInfo]) -> Result<()> {
let volume_dir = self.get_bucket_path(volume)?;
let xlpath = self.get_object_path(volume, format!("{path}/{STORAGE_FORMAT_FILE}").as_str())?;
let (data, _) = self.read_all_data_with_dmtime(volume, volume_dir.as_path(), &xlpath).await?;
if data.is_empty() {
return Err(DiskError::FileNotFound);
}
let mut fm = FileMeta::default();
fm.unmarshal_msg(&data)?;
for fi in fis.iter() {
let data_dir = match fm.delete_version(fi) {
Ok(res) => res,
Err(err) => {
let err: DiskError = err.into();
if !fi.deleted && (err == DiskError::FileNotFound || err == DiskError::FileVersionNotFound) {
continue;
}
return Err(err);
}
};
if let Some(dir) = data_dir {
let vid = fi.version_id.unwrap_or_default();
let _ = fm.data.remove(vec![vid, dir]);
let dir_path = self.get_object_path(volume, format!("{path}/{dir}").as_str())?;
if let Err(err) = self.move_to_trash(&dir_path, true, false).await
&& !(err == DiskError::FileNotFound || err == DiskError::VolumeNotFound)
{
return Err(err);
};
}
}
// Remove xl.meta when no versions remain
if fm.versions.is_empty() {
self.delete_file(&volume_dir, &xlpath, true, false).await?;
return Ok(());
}
// Update xl.meta atomically: a concurrent reader or crash mid-write must
// never observe a truncated xl.meta for versions that were not deleted.
let buf = fm.marshal_msg()?;
self.write_all_meta(volume, format!("{path}/{STORAGE_FORMAT_FILE}").as_str(), &buf, true)
.await?;
Ok(())
}
async fn write_all_meta(&self, volume: &str, path: &str, buf: &[u8], sync: bool) -> Result<()> {
let volume_dir = self.get_bucket_path(volume)?;
let file_path = self.get_object_path(volume, path)?;
check_path_length(file_path.to_string_lossy().as_ref())?;
let tmp_volume_dir = self.get_bucket_path(super::RUSTFS_META_TMP_BUCKET)?;
let tmp_file_path = self.get_object_path(super::RUSTFS_META_TMP_BUCKET, Uuid::new_v4().to_string().as_str())?;
let durability = effective_durability(volume);
// The tmp file is renamed to its final location right below, so only
// its contents must be durable here (SyncMode::FileOnly): the rename
// drops the tmp directory entry, and the destination parent directory
// is fsynced after the rename. Both are metadata commits, so relaxed
// tiers skip them.
let tmp_sync = if sync && durability.syncs_commit_metadata() {
SyncMode::FileOnly
} else {
SyncMode::None
};
self.write_all_internal(&tmp_file_path, InternalBuf::Ref(buf), tmp_sync, &tmp_volume_dir)
.await?;
rename_all(tmp_file_path, &file_path, volume_dir).await?;
if sync
&& durability.syncs_commit_metadata()
&& let Some(parent) = file_path.parent()
{
os::fsync_dir(parent).await.map_err(to_file_error)?;
}
Ok(())
}
// write_all_public for trail
async fn write_all_public(&self, volume: &str, path: &str, data: Bytes) -> Result<()> {
if volume == RUSTFS_META_BUCKET && path == super::FORMAT_CONFIG_FILE {
let mut format_info = self.format_info.write().await;
format_info.data.clone_from(&data);
}
let volume_dir = self.get_bucket_path(volume)?;
// Files written here (format.json, ...) stay where they land — no
// rename follows — so the new directory entry must be fsynced too.
// System-critical volumes are pinned to strict by effective_durability;
// only the legacy full-off switch (historical semantics) skips this.
let sync = if effective_durability(volume).syncs_commit_metadata() {
SyncMode::FileAndDir
} else {
SyncMode::None
};
self.write_all_private(volume, path, data, sync, &volume_dir).await?;
Ok(())
}
// write_all_private with check_path_length
#[tracing::instrument(level = "debug", skip_all)]
async fn write_all_private(&self, volume: &str, path: &str, buf: Bytes, sync: SyncMode, skip_parent: &Path) -> Result<()> {
let file_path = self.get_object_path(volume, path)?;
check_path_length(file_path.to_string_lossy().as_ref())?;
self.write_all_internal(&file_path, InternalBuf::Owned(buf), sync, skip_parent)
.await?;
Ok(())
}
// write_all_internal do write file.
// Executes the given SyncMode verbatim: durability policy (tier gating,
// system-critical pinning) is resolved by callers via effective_durability.
async fn write_all_internal(
&self,
file_path: &Path,
data: InternalBuf<'_>,
sync: SyncMode,
skip_parent: &Path,
) -> Result<()> {
let skip_parent = if skip_parent.as_os_str().is_empty() {
self.root.as_path()
} else {
skip_parent
};
match data {
InternalBuf::Ref(buf) => {
let mut f = self.open_file(file_path, O_CREATE | O_WRONLY | O_TRUNC, skip_parent).await?;
f.write_all(buf).await.map_err(to_file_error)?;
if sync != SyncMode::None {
f.sync_data().await.map_err(to_file_error)?;
// Persist the directory entry too, so a freshly created file
// survives power loss along with its contents. Skipped for
// FileOnly: the caller renames the file away immediately.
if sync == SyncMode::FileAndDir
&& let Some(parent) = file_path.parent()
{
os::fsync_dir(parent).await.map_err(to_file_error)?;
}
}
}
InternalBuf::Owned(buf) => {
let path = file_path.to_path_buf();
if let Some(parent) = path.parent()
&& parent != skip_parent
{
os::make_dir_all(parent, skip_parent).await?;
}
tokio::task::spawn_blocking(move || {
let mut f = std::fs::OpenOptions::new()
.create(true)
.write(true)
.truncate(true)
.open(&path)
.map_err(to_file_error)?;
std::io::Write::write_all(&mut f, buf.as_ref()).map_err(to_file_error)?;
if sync != SyncMode::None {
f.sync_data().map_err(to_file_error)?;
// See the Ref branch above: FileOnly callers rename the
// file away, so the tmp directory entry never needs to
// become durable.
if sync == SyncMode::FileAndDir
&& let Some(parent) = path.parent()
{
os::fsync_dir_std(parent).map_err(to_file_error)?;
}
}
Ok::<(), std::io::Error>(())
})
.await
.map_err(DiskError::from)??;
}
}
Ok(())
}
async fn open_file(&self, path: impl AsRef<Path>, mode: usize, skip_parent: impl AsRef<Path>) -> Result<File> {
let mut skip_parent = skip_parent.as_ref();
if skip_parent.as_os_str().is_empty() {
skip_parent = self.root.as_path();
}
if let Some(parent) = path.as_ref().parent()
&& parent != skip_parent
{
os::make_dir_all(parent, skip_parent).await?;
}
let f = super::fs::open_file(path.as_ref(), mode).await.map_err(to_file_error)?;
Ok(f)
}
async fn open_file_read_only(&self, path: impl AsRef<Path>) -> Result<File> {
let f = super::fs::open_file(path.as_ref(), O_RDONLY).await.map_err(to_file_error)?;
Ok(f)
}
#[allow(dead_code)]
fn get_metrics(&self) -> DiskMetrics {
DiskMetrics::default()
}
async fn bitrot_verify(&self, part_path: &PathBuf, part_size: usize, algo: HashAlgorithm, shard_size: usize) -> Result<()> {
let retry_count = bitrot_size_mismatch_retry_count();
let retry_delay = bitrot_size_mismatch_retry_delay();
for attempt in 0..=retry_count {
let file = super::fs::open_file(part_path, O_RDONLY).await.map_err(to_file_error)?;
let meta = file.metadata().await.map_err(to_file_error)?;
let file_size = meta.len() as usize;
match bitrot_verify(Box::new(file), file_size, part_size, algo.clone(), shard_size).await {
Ok(()) => return Ok(()),
Err(err) if attempt < retry_count && is_bitrot_size_mismatch_error(&err) => {
info!(
event = EVENT_DISK_LOCAL_CHECK_PARTS,
component = LOG_COMPONENT_ECSTORE,
subsystem = LOG_SUBSYSTEM_DISK_LOCAL,
path = %part_path.display(),
expected_size = part_size,
actual_size = file_size,
retry_attempt = attempt + 1,
retry_count,
retry_delay_ms = retry_delay.as_millis(),
state = "bitrot_retry",
"Disk local check_parts state changed"
);
tokio::time::sleep(retry_delay).await;
}
Err(err) if is_bitrot_verification_error(&err) => return Err(DiskError::FileCorrupt),
Err(err) => return Err(to_file_error(err).into()),
}
}
Err(DiskError::FileCorrupt)
}
#[async_recursion::async_recursion]
#[allow(clippy::too_many_arguments)]
async fn scan_dir<W>(
&self,
mut current: String,
mut prefix: String,
opts: &WalkDirOptions,
out: &mut MetacacheWriter<W>,
objs_returned: &mut i32,
skip_current_dir_object: bool,
multipart_dir_to_skip: Option<HashSet<String>>,
) -> Result<()>
where
W: AsyncWrite + Unpin + Send,
{
let forward = {
opts.forward_to
.as_ref()
.and_then(|v| v.strip_prefix(&current))
.map(|forward| {
if let Some(idx) = forward.find('/') {
forward[..idx].to_owned()
} else {
forward.to_owned()
}
})
};
if opts.limit > 0 && *objs_returned >= opts.limit {
return Ok(());
}
// TODO: add lock
let mut entries = match self.list_dir("", &opts.bucket, &current, -1).await {
Ok(res) => res,
Err(e) => {
if e != DiskError::VolumeNotFound && e != Error::FileNotFound {
error!(
event = EVENT_DISK_LOCAL_SCAN_FAILED,
component = LOG_COMPONENT_ECSTORE,
subsystem = LOG_SUBSYSTEM_DISK_LOCAL,
path = %current,
operation = "list_dir",
error = ?e,
"Disk local scan failed"
);
return Err(e);
}
if opts.report_notfound && e == Error::FileNotFound && current == opts.base_dir {
return Err(DiskError::FileNotFound);
}
return Ok(());
}
};
if entries.is_empty() {
return Ok(());
}
current = current.trim_matches('/').to_owned();
let bucket = opts.bucket.as_str();
let mut dir_objes = HashSet::new();
// First-level filtering
for item in entries.iter_mut() {
let entry = item.clone();
// check limit
if opts.limit > 0 && *objs_returned >= opts.limit {
return Ok(());
}
// check multipart dir
if skip_current_dir_object
&& let Some(ref dir_to_skip) = multipart_dir_to_skip
&& dir_to_skip.contains(entry.trim_end_matches(SLASH_SEPARATOR))
{
*item = "".to_owned();
continue;
}
// check prefix
if !prefix.is_empty() && !entry.starts_with(prefix.as_str()) {
*item = "".to_owned();
continue;
}
if let Some(forward) = &forward
&& &entry < forward
{
*item = "".to_owned();
continue;
}
if entry.ends_with(SLASH_SEPARATOR) {
if entry.ends_with(GLOBAL_DIR_SUFFIX_WITH_SLASH) {
let entry = format!("{}{}", entry.as_str().trim_end_matches(GLOBAL_DIR_SUFFIX_WITH_SLASH), SLASH_SEPARATOR);
dir_objes.insert(entry.clone());
*item = entry;
continue;
}
*item = entry.trim_end_matches(SLASH_SEPARATOR).to_owned();
continue;
}
*item = "".to_owned();
if entry.ends_with(STORAGE_FORMAT_FILE) {
if skip_current_dir_object {
continue;
}
let metadata = self
.read_metadata(bucket, format!("{}/{}", &current, &entry).as_str())
.await?;
let entry = entry.strip_suffix(STORAGE_FORMAT_FILE).unwrap_or_default().to_owned();
let name = entry.trim_end_matches(SLASH_SEPARATOR);
let name = decode_dir_object(format!("{}/{}", &current, &name).as_str());
if opts.limit <= 0 || metadata_counts_toward_limit(&metadata) {
*objs_returned += 1;
}
write_metacache_obj(
out,
&MetaCacheEntry {
name: name.clone(),
metadata: metadata.to_vec(),
..Default::default()
},
)
.await?;
continue;
}
}
entries.sort();
if let Some(forward) = &forward {
for (i, entry) in entries.iter().enumerate() {
if entry >= forward || forward.starts_with(entry.as_str()) {
entries.drain(..i);
break;
}
}
}
let mut dir_stack: Vec<(String, bool, Option<HashSet<String>>)> = Vec::with_capacity(5);
// Explicit directory markers and real directories can resolve to the same logical path.
let schedule_dir = |dir_stack: &mut Vec<(String, bool, Option<HashSet<String>>)>,
dir_name: String,
skip_object: bool,
dir_to_skip: Option<HashSet<String>>| {
if let Some((last_dir_name, existing_skip_object, existing_dir_to_skip)) = dir_stack.last_mut()
&& *last_dir_name == dir_name
{
*existing_skip_object |= skip_object;
if let Some(existing_dir_to_skip) = existing_dir_to_skip {
if let Some(new_dir_to_skip) = &dir_to_skip {
existing_dir_to_skip.extend(new_dir_to_skip.iter().cloned());
}
} else {
*existing_dir_to_skip = dir_to_skip;
}
} else {
dir_stack.push((dir_name, skip_object, dir_to_skip));
}
};
prefix = "".to_owned();
for entry in entries.iter() {
if opts.limit > 0 && *objs_returned >= opts.limit {
return Ok(());
}
if entry.is_empty() {
continue;
}
let name = path_join_buf(&[current.as_str(), entry.as_str()]);
while let Some((last_name, _, _)) = dir_stack.last()
&& *last_name < name
{
let (pop, skip_object, dir_to_skip) = dir_stack.pop().expect("operation should succeed");
write_metacache_obj(
out,
&MetaCacheEntry {
name: pop.clone(),
..Default::default()
},
)
.await?;
let scan_path = pop.clone();
if opts.recursive
&& let Err(er) =
Box::pin(self.scan_dir(pop, prefix.clone(), opts, out, objs_returned, skip_object, dir_to_skip)).await
{
if !er.is_metacache_output_stream_closed() {
error!(
event = EVENT_DISK_LOCAL_SCAN_FAILED,
component = LOG_COMPONENT_ECSTORE,
subsystem = LOG_SUBSYSTEM_DISK_LOCAL,
path = %scan_path,
operation = "scan_dir",
error = ?er,
"Disk local scan failed"
);
}
return Err(er);
}
}
let mut meta = MetaCacheEntry {
name,
..Default::default()
};
let mut is_dir_obj = false;
if let Some(_dir) = dir_objes.get(entry) {
is_dir_obj = true;
meta.name
.truncate(meta.name.len() - meta.name.chars().last().expect("operation should succeed").len_utf8());
meta.name.push_str(GLOBAL_DIR_SUFFIX_WITH_SLASH);
}
let fname = format!("{}/{}", &meta.name, STORAGE_FORMAT_FILE);
match self.read_metadata(&opts.bucket, fname.as_str()).await {
Ok(res) => {
if is_dir_obj {
meta.name = meta.name.trim_end_matches(GLOBAL_DIR_SUFFIX_WITH_SLASH).to_owned();
meta.name.push_str(SLASH_SEPARATOR);
}
meta.metadata = res.to_vec();
write_metacache_obj(out, &meta).await?;
let file_meta = if opts.limit > 0 || opts.recursive {
FileMeta::load(&res).ok()
} else {
None
};
if opts.limit <= 0 || file_meta.as_ref().is_none_or(file_meta_counts_toward_limit) {
*objs_returned += 1;
}
if opts.recursive {
let mut dir_to_skip = HashSet::new();
if let Some(file_meta) = file_meta.as_ref()
&& let Ok(data_dirs) = file_meta.get_data_dirs()
{
for data_dir in data_dirs.iter().flatten() {
dir_to_skip.insert(data_dir.to_string());
}
}
let mut dir_name = meta.name.clone();
if !dir_name.ends_with(SLASH_SEPARATOR) {
dir_name.push_str(SLASH_SEPARATOR);
}
schedule_dir(
&mut dir_stack,
dir_name,
true,
if dir_to_skip.is_empty() { None } else { Some(dir_to_skip) },
);
}
}
Err(err) => {
if err == Error::FileNotFound || err == Error::IsNotRegular {
// NOT an object, append to stack (with slash)
// If dirObject, but no metadata (which is unexpected) we skip it.
if !is_dir_obj && !is_empty_dir(self.get_object_path(&opts.bucket, &meta.name)?).await {
meta.name.push_str(SLASH_SEPARATOR);
if opts.recursive
|| opts.incl_deleted
|| self.directory_has_visible_listing_entry(&opts.bucket, &meta.name).await?
{
schedule_dir(&mut dir_stack, meta.name, false, None);
}
}
continue;
}
error!(
event = EVENT_DISK_LOCAL_SCAN_FAILED,
component = LOG_COMPONENT_ECSTORE,
subsystem = LOG_SUBSYSTEM_DISK_LOCAL,
path = %fname,
operation = "read_metadata",
error = ?err,
"Disk local scan failed"
);
return Err(err);
}
};
}
while let Some((dir, skip_object, dir_to_skip)) = dir_stack.pop() {
if opts.limit > 0 && *objs_returned >= opts.limit {
return Ok(());
}
write_metacache_obj(
out,
&MetaCacheEntry {
name: dir.clone(),
..Default::default()
},
)
.await?;
let scan_path = dir.clone();
if opts.recursive
&& let Err(er) =
Box::pin(self.scan_dir(dir, prefix.clone(), opts, out, objs_returned, skip_object, dir_to_skip)).await
{
if !er.is_metacache_output_stream_closed() {
error!(
event = EVENT_DISK_LOCAL_SCAN_FAILED,
component = LOG_COMPONENT_ECSTORE,
subsystem = LOG_SUBSYSTEM_DISK_LOCAL,
path = %scan_path,
operation = "scan_dir",
error = ?er,
"Disk local recursive scan failed"
);
}
return Err(er);
}
}
Ok(())
}
async fn directory_has_visible_listing_entry(&self, bucket: &str, dir_name: &str) -> Result<bool> {
let mut stack = vec![dir_name.trim_matches('/').to_owned()];
while let Some(current) = stack.pop() {
if current.is_empty() {
continue;
}
let entries = match self.list_dir("", bucket, &current, -1).await {
Ok(entries) => entries,
Err(err) => {
if err == DiskError::VolumeNotFound || err == Error::FileNotFound {
continue;
}
return Err(err);
}
};
let mut data_dirs_to_skip = HashSet::new();
let mut child_dirs = Vec::new();
for entry in entries {
if entry == STORAGE_FORMAT_FILE {
let metadata_path = path_join_buf(&[current.as_str(), STORAGE_FORMAT_FILE]);
match self.read_metadata(bucket, metadata_path.as_str()).await {
Ok(metadata) => {
let file_meta = match FileMeta::load(&metadata) {
Ok(file_meta) => file_meta,
Err(_) => return Ok(true),
};
if file_meta_counts_toward_limit(&file_meta) {
return Ok(true);
}
if let Ok(data_dirs) = file_meta.get_data_dirs() {
for data_dir in data_dirs.iter().flatten() {
data_dirs_to_skip.insert(data_dir.to_string());
}
}
}
Err(err) => {
if err != Error::FileNotFound && err != Error::IsNotRegular {
return Err(err);
}
}
}
continue;
}
if entry.ends_with(SLASH_SEPARATOR) {
let child = entry.trim_end_matches(SLASH_SEPARATOR);
if !child.is_empty() {
child_dirs.push(child.to_owned());
}
}
}
for child in child_dirs {
if !data_dirs_to_skip.contains(&child) {
stack.push(path_join_buf(&[current.as_str(), child.as_str()]));
}
}
}
Ok(false)
}
}
pub struct ScanGuard(pub Arc<AtomicU32>);
impl Drop for ScanGuard {
fn drop(&mut self) {
self.0.fetch_sub(1, Ordering::Release);
}
}
struct LocalScanLockGuard {
scan_locks: Arc<ParkingLotMutex<HashSet<(String, String)>>>,
key: (String, String),
}
impl Drop for LocalScanLockGuard {
fn drop(&mut self) {
self.scan_locks.lock().remove(&self.key);
}
}
fn local_disk_scan_lock_key(bucket: &str, base_dir: &str, filter_prefix: Option<&str>) -> (String, String) {
let mut prefix = base_dir.trim_matches('/').to_owned();
if let Some(filter_prefix) = filter_prefix
.map(|prefix| prefix.trim_matches('/'))
.filter(|prefix| !prefix.is_empty())
{
if prefix.is_empty() {
prefix.push_str(filter_prefix);
} else {
prefix.push_str(SLASH_SEPARATOR);
prefix.push_str(filter_prefix);
}
}
(bucket.to_owned(), prefix)
}
fn rename_data_versions_signature(meta: &FileMeta) -> Option<Vec<u8>> {
if meta.versions.len() > 10 {
return None;
}
let mut signature = Vec::with_capacity(meta.versions.len() * 16);
for version in meta.versions.iter() {
signature.extend_from_slice(version.header.version_id.unwrap_or_default().as_bytes());
}
Some(signature)
}
fn is_root_path(path: impl AsRef<Path>) -> bool {
path.as_ref().components().count() == 1 && path.as_ref().has_root()
}
fn metadata_counts_toward_limit(metadata: &[u8]) -> bool {
FileMeta::load(metadata).map_or(true, |meta| file_meta_counts_toward_limit(&meta))
}
fn file_meta_counts_toward_limit(meta: &FileMeta) -> bool {
meta.into_fileinfo("", "", "", false, true, false)
.map_or_else(|_| !meta.all_hidden(true), |latest| !latest.deleted && !latest.tier_free_version())
}
// Filter std::io::ErrorKind::NotFound
async fn read_file_exists(path: impl AsRef<Path>) -> Result<(Bytes, Option<Metadata>)> {
let p = path.as_ref();
let (data, meta) = match read_file_all(&p).await {
Ok((data, meta)) => (data, Some(meta)),
Err(e) => {
if e == Error::FileNotFound {
(Bytes::new(), None)
} else {
return Err(e);
}
}
};
// let mut data = Vec::new();
// if meta.is_some() {
// data = fs::read(&p).await?;
// }
Ok((data, meta))
}
async fn read_file_all(path: impl AsRef<Path>) -> Result<(Bytes, Metadata)> {
let p = path.as_ref();
let meta = read_file_metadata(&path).await?;
let data = fs::read(&p)
.await
.inspect_err(|err| {
log_startup_disk_io_error("read_file_all", p, err);
})
.map_err(to_file_error)?;
Ok((data.into(), meta))
}
async fn read_file_metadata(p: impl AsRef<Path>) -> Result<Metadata> {
let path = p.as_ref();
let meta = fs::metadata(path)
.await
.inspect_err(|err| {
if err.kind() != ErrorKind::NotFound {
log_startup_disk_io_error("read_file_metadata", path, err);
}
})
.map_err(to_file_error)?;
Ok(meta)
}
fn skip_access_checks(p: impl AsRef<str>) -> bool {
let vols = [
RUSTFS_META_TMP_DELETED_BUCKET,
super::RUSTFS_META_TMP_BUCKET,
super::RUSTFS_META_MULTIPART_BUCKET,
RUSTFS_META_BUCKET,
];
for v in vols.iter() {
if p.as_ref().starts_with(v) {
return true;
}
}
false
}
fn local_disk_object_path(root: &Path, bucket: &str, key: &str) -> Result<PathBuf> {
let cache_key = if key.is_empty() {
bucket.to_string()
} else {
path_join_buf(&[bucket, key])
};
#[cfg(windows)]
let path = root.join(cache_key.replace('/', "\\"));
#[cfg(not(windows))]
let path = root.join(cache_key);
check_local_disk_valid_path(root, &path)?;
Ok(path)
}
fn local_disk_bucket_path(root: &Path, bucket: &str) -> Result<PathBuf> {
#[cfg(windows)]
let bucket_path = root.join(bucket.replace('/', "\\"));
#[cfg(not(windows))]
let bucket_path = root.join(bucket);
check_local_disk_valid_path(root, &bucket_path)?;
Ok(bucket_path)
}
fn check_local_disk_valid_path(root: &Path, path: impl AsRef<Path>) -> Result<()> {
let path = normalize_path_components(path);
if !path.starts_with(root) {
return Err(DiskError::InvalidPath);
}
reject_local_disk_symlink_components(root, &path)
}
fn reject_local_disk_symlink_components(root: &Path, path: &Path) -> Result<()> {
let relative = path.strip_prefix(root).map_err(|_| DiskError::InvalidPath)?;
let mut current = root.to_path_buf();
for component in relative.components() {
current.push(component.as_os_str());
match lstat_std(&current) {
Ok(metadata) => {
if metadata.file_type().is_symlink() {
return Err(DiskError::InvalidPath);
}
}
Err(err) if err.kind() == ErrorKind::NotFound => break,
Err(err) => return Err(to_file_error(err).into()),
}
}
Ok(())
}
// Lightweight path normalization without filesystem calls
fn normalize_path_components(path: impl AsRef<Path>) -> PathBuf {
let path = path.as_ref();
let mut result = PathBuf::new();
for component in path.components() {
match component {
std::path::Component::Normal(name) => {
result.push(name);
}
std::path::Component::ParentDir => {
result.pop();
}
std::path::Component::CurDir => {
// Ignore current directory components
}
std::path::Component::RootDir => {
result.push(component);
}
std::path::Component::Prefix(_prefix) => {
result.push(component);
}
}
}
result
}
#[async_trait::async_trait]
impl DiskAPI for LocalDisk {
fn to_string(&self) -> String {
self.root.to_string_lossy().to_string()
}
fn is_local(&self) -> bool {
true
}
fn host_name(&self) -> String {
self.endpoint.host_port()
}
async fn is_online(&self) -> bool {
true
}
fn endpoint(&self) -> Endpoint {
self.endpoint.clone()
}
async fn close(&self) -> Result<()> {
Ok(())
}
fn path(&self) -> PathBuf {
self.root.clone()
}
fn get_disk_location(&self) -> DiskLocation {
DiskLocation {
pool_idx: {
if self.endpoint.pool_idx < 0 {
None
} else {
Some(self.endpoint.pool_idx as usize)
}
},
set_idx: {
if self.endpoint.set_idx < 0 {
None
} else {
Some(self.endpoint.set_idx as usize)
}
},
disk_idx: {
if self.endpoint.disk_idx < 0 {
None
} else {
Some(self.endpoint.disk_idx as usize)
}
},
}
}
#[tracing::instrument(level = "debug", skip(self))]
async fn get_disk_id(&self) -> Result<Option<Uuid>> {
let format_info = {
let format_info = self.format_info.read().await;
format_info.clone()
};
let id = format_info.id;
if format_info.file_info.is_some() && id.is_some() {
// Reuse the cached disk id only when the cached format check is fresh.
if let Some(last_check) = format_info.last_check
&& last_check.unix_timestamp() + 1 >= OffsetDateTime::now_utc().unix_timestamp()
{
return Ok(id);
}
}
let file_meta = match self.check_format_json().await {
Ok(meta) => meta,
Err(err) => {
if matches!(err, DiskError::UnformattedDisk | DiskError::DiskNotFound) {
let mut format_info = self.format_info.write().await;
format_info.id = None;
format_info.data = Bytes::new();
format_info.file_info = None;
format_info.last_check = None;
}
return Err(err);
}
};
if let Some(file_info) = &format_info.file_info
&& super::fs::same_file(&file_meta, file_info)
{
let mut format_info = self.format_info.write().await;
format_info.last_check = Some(OffsetDateTime::now_utc());
drop(format_info);
return Ok(id);
}
debug!("get_disk_id: read format.json");
let b = fs::read(&self.format_path).await.map_err(to_unformatted_disk_error)?;
let fm = FormatV3::try_from(b.as_slice()).map_err(|e| {
warn!(
event = EVENT_DISK_LOCAL_FORMAT_DECODE_FAILED,
component = LOG_COMPONENT_ECSTORE,
subsystem = LOG_SUBSYSTEM_DISK_LOCAL,
error = ?e,
"Disk local format decode failed"
);
DiskError::CorruptedBackend
})?;
let (m, n) = fm.find_disk_index_by_disk_id(fm.erasure.this)?;
let disk_id = fm.erasure.this;
if m as i32 != self.endpoint.set_idx || n as i32 != self.endpoint.disk_idx {
return Err(DiskError::InconsistentDisk);
}
let mut format_info = self.format_info.write().await;
format_info.id = Some(disk_id);
format_info.file_info = Some(file_meta);
format_info.data = b.into();
format_info.last_check = Some(OffsetDateTime::now_utc());
drop(format_info);
Ok(Some(disk_id))
}
async fn set_disk_id(&self, _id: Option<Uuid>) -> Result<()> {
// No setup is required locally
Ok(())
}
#[tracing::instrument(skip(self))]
async fn read_all(&self, volume: &str, path: &str) -> Result<Bytes> {
crate::hp_guard!("LocalDisk::read_all");
if volume == RUSTFS_META_BUCKET && path == super::FORMAT_CONFIG_FILE {
let format_info = self.format_info.read().await;
if !format_info.data.is_empty() {
return Ok(format_info.data.clone());
}
}
let p = self.get_object_path(volume, path)?;
let (data, _) = read_file_all(&p).await?;
Ok(data)
}
#[tracing::instrument(level = "debug", skip_all)]
async fn write_all(&self, volume: &str, path: &str, data: Bytes) -> Result<()> {
crate::hp_guard!("LocalDisk::write_all");
self.write_all_public(volume, path, data).await
}
#[tracing::instrument(skip(self))]
async fn delete(&self, volume: &str, path: &str, opt: DeleteOptions) -> Result<()> {
crate::hp_guard!("LocalDisk::delete");
let volume_dir = self.get_bucket_path(volume)?;
if !skip_access_checks(volume)
&& let Err(e) = access(&volume_dir).await
{
return Err(to_access_error(e, DiskError::VolumeAccessDenied).into());
}
let file_path = self.get_object_path(volume, path)?;
check_path_length(file_path.to_string_lossy().to_string().as_str())?;
self.delete_file(&volume_dir, &file_path, opt.recursive, opt.immediate)
.await?;
Ok(())
}
#[tracing::instrument(skip(self))]
async fn verify_file(&self, volume: &str, path: &str, fi: &FileInfo) -> Result<CheckPartsResp> {
let volume_dir = self.get_bucket_path(volume)?;
if !skip_access_checks(volume)
&& let Err(e) = access(&volume_dir).await
{
return Err(to_access_error(e, DiskError::VolumeAccessDenied).into());
}
let mut resp = CheckPartsResp {
results: vec![0; fi.parts.len()],
};
let erasure = &fi.erasure;
for (i, part) in fi.parts.iter().enumerate() {
let checksum_info = erasure.get_checksum_info(part.number);
let checksum_algo =
if fi.uses_legacy_checksum && checksum_info.algorithm == rustfs_utils::HashAlgorithm::HighwayHash256S {
rustfs_utils::HashAlgorithm::HighwayHash256SLegacy
} else {
checksum_info.algorithm
};
let part_path = self.get_object_path(
volume,
path_join_buf(&[
path,
&fi.data_dir.map_or_else(|| "".to_string(), |dir| dir.to_string()),
&format!("part.{}", part.number),
])
.as_str(),
)?;
let err = self
.bitrot_verify(
&part_path,
erasure.shard_file_size(part.size as i64) as usize,
checksum_algo,
erasure.shard_size(),
)
.await
.err();
resp.results[i] = conv_part_err_to_int(&err);
if resp.results[i] == CHECK_PART_UNKNOWN
&& let Some(err) = err
{
error!(
event = EVENT_DISK_LOCAL_CHECK_PARTS,
component = LOG_COMPONENT_ECSTORE,
subsystem = LOG_SUBSYSTEM_DISK_LOCAL,
path = ?part_path,
part_number = part.number,
state = "bitrot_verify_failed",
error = ?err,
"Disk local check_parts state changed"
);
if err == DiskError::FileAccessDenied {
continue;
}
info!(
event = EVENT_DISK_LOCAL_CHECK_PARTS,
component = LOG_COMPONENT_ECSTORE,
subsystem = LOG_SUBSYSTEM_DISK_LOCAL,
endpoint = %self.endpoint,
path = ?part_path,
part_number = part.number,
state = "unknown",
"Disk local check_parts state changed"
);
}
}
Ok(resp)
}
#[tracing::instrument(skip(self))]
async fn read_parts(&self, bucket: &str, paths: &[String]) -> Result<Vec<ObjectPartInfo>> {
let volume_dir = self.get_bucket_path(bucket)?;
let mut ret = vec![ObjectPartInfo::default(); paths.len()];
for (i, path_str) in paths.iter().enumerate() {
let path = Path::new(path_str);
let file_name = path.file_name().and_then(|v| v.to_str()).unwrap_or_default();
let num = file_name
.strip_prefix("part.")
.and_then(|v| v.strip_suffix(".meta"))
.and_then(|v| v.parse::<usize>().ok())
.unwrap_or_default();
if let Err(err) = access(
self.get_object_path(
bucket,
path_join_buf(&[
path.parent().unwrap_or_else(|| Path::new("")).to_string_lossy().as_ref(),
&format!("part.{num}"),
])
.as_str(),
)?,
)
.await
{
ret[i] = ObjectPartInfo {
number: num,
error: Some(err.to_string()),
..Default::default()
};
continue;
}
let data = match self
.read_all_data(bucket, volume_dir.clone(), self.get_object_path(bucket, path.to_string_lossy().as_ref())?)
.await
{
Ok(data) => data,
Err(err) => {
ret[i] = ObjectPartInfo {
number: num,
error: Some(err.to_string()),
..Default::default()
};
continue;
}
};
match ObjectPartInfo::unmarshal(&data) {
Ok(meta) => {
ret[i] = meta;
}
Err(err) => {
ret[i] = ObjectPartInfo {
number: num,
error: Some(err.to_string()),
..Default::default()
};
}
};
}
Ok(ret)
}
#[tracing::instrument(skip(self))]
async fn check_parts(&self, volume: &str, path: &str, fi: &FileInfo) -> Result<CheckPartsResp> {
let volume_dir = self.get_bucket_path(volume)?;
let file_path = self.get_object_path(volume, path)?;
check_path_length(file_path.to_string_lossy().as_ref())?;
let mut resp = CheckPartsResp {
results: vec![0; fi.parts.len()],
};
for (i, part) in fi.parts.iter().enumerate() {
let part_path = self.get_object_path(
volume,
path_join_buf(&[
path,
&fi.data_dir.map_or_else(|| "".to_string(), |dir| dir.to_string()),
&format!("part.{}", part.number),
])
.as_str(),
)?;
debug!(
event = EVENT_DISK_LOCAL_CHECK_PARTS,
component = LOG_COMPONENT_ECSTORE,
subsystem = LOG_SUBSYSTEM_DISK_LOCAL,
path = ?part_path,
part_number = part.number,
state = "checking",
"Disk local check_parts state changed"
);
match lstat(&part_path).await {
Ok(st) => {
if st.is_dir() {
resp.results[i] = CHECK_PART_FILE_NOT_FOUND;
continue;
}
if (st.len() as i64) < fi.erasure.shard_file_size(part.size as i64) {
resp.results[i] = CHECK_PART_FILE_CORRUPT;
continue;
}
resp.results[i] = CHECK_PART_SUCCESS;
}
Err(err) => {
debug!(
event = EVENT_DISK_LOCAL_CHECK_PARTS,
component = LOG_COMPONENT_ECSTORE,
subsystem = LOG_SUBSYSTEM_DISK_LOCAL,
path = ?part_path,
part_number = part.number,
state = "part_stat_failed",
error = ?err,
"Disk local check_parts state changed"
);
let e: DiskError = to_file_error(err).into();
if e == DiskError::FileNotFound {
if !skip_access_checks(volume)
&& let Err(err) = access(&volume_dir).await
&& err.kind() == ErrorKind::NotFound
{
resp.results[i] = CHECK_PART_VOLUME_NOT_FOUND;
continue;
}
resp.results[i] = CHECK_PART_FILE_NOT_FOUND;
} else {
error!(
event = EVENT_DISK_LOCAL_CHECK_PARTS,
component = LOG_COMPONENT_ECSTORE,
subsystem = LOG_SUBSYSTEM_DISK_LOCAL,
path = ?file_path,
part_number = part.number,
state = "file_stat_failed",
error = ?e,
"Disk local check_parts state changed"
);
}
continue;
}
}
}
Ok(resp)
}
#[tracing::instrument(level = "debug", skip(self))]
async fn rename_part(&self, src_volume: &str, src_path: &str, dst_volume: &str, dst_path: &str, meta: Bytes) -> Result<()> {
let src_volume_dir = self.get_bucket_path(src_volume)?;
let dst_volume_dir = self.get_bucket_path(dst_volume)?;
if !skip_access_checks(src_volume) {
super::fs::access_std(&src_volume_dir).map_err(|e| to_access_error(e, DiskError::VolumeAccessDenied))?
}
if !skip_access_checks(dst_volume) {
super::fs::access_std(&dst_volume_dir).map_err(|e| to_access_error(e, DiskError::VolumeAccessDenied))?
}
let src_is_dir = has_suffix(src_path, SLASH_SEPARATOR);
let dst_is_dir = has_suffix(dst_path, SLASH_SEPARATOR);
if !src_is_dir && dst_is_dir || src_is_dir && !dst_is_dir {
warn!(
event = EVENT_DISK_LOCAL_RENAME_REJECTED,
component = LOG_COMPONENT_ECSTORE,
subsystem = LOG_SUBSYSTEM_DISK_LOCAL,
reason = "src_dst_type_mismatch",
src_is_dir,
dst_is_dir,
"Disk local rename rejected"
);
return Err(DiskError::FileAccessDenied);
}
let src_file_path = self.get_object_path(src_volume, src_path)?;
let dst_file_path = self.get_object_path(dst_volume, dst_path)?;
// warn!("rename_part src_file_path:{:?}, dst_file_path:{:?}", &src_file_path, &dst_file_path);
check_path_length(src_file_path.to_string_lossy().as_ref())?;
check_path_length(dst_file_path.to_string_lossy().as_ref())?;
if src_is_dir {
let meta_op = match lstat_std(&src_file_path).map_err(|e| to_file_error(e).into()) {
Ok(meta) => Some(meta),
Err(e) => {
return Err(e);
}
};
if let Some(meta) = meta_op
&& !meta.is_dir()
{
warn!(
event = EVENT_DISK_LOCAL_RENAME_REJECTED,
component = LOG_COMPONENT_ECSTORE,
subsystem = LOG_SUBSYSTEM_DISK_LOCAL,
reason = "src_expected_dir_missing",
path = ?src_file_path,
"Disk local rename rejected"
);
return Err(DiskError::FileAccessDenied);
}
// Clear any stale destination before the directory rename. An absent
// destination is the normal case when renaming a directory to a new
// location, so tolerate NotFound instead of aborting the whole rename
// (MinIO's RenameFile ignores osIsNotExist here).
if let Err(e) = remove_std(&dst_file_path)
&& e.kind() != ErrorKind::NotFound
{
return Err(to_file_error(e).into());
}
} else {
let meta = lstat_std(&src_file_path).map_err(|e| -> DiskError { to_file_error(e).into() })?;
if meta.is_dir() {
warn!(
event = EVENT_DISK_LOCAL_RENAME_REJECTED,
component = LOG_COMPONENT_ECSTORE,
subsystem = LOG_SUBSYSTEM_DISK_LOCAL,
reason = "src_unexpected_dir",
path = ?src_file_path,
"Disk local rename rejected"
);
return Err(DiskError::FileAccessDenied);
}
}
// UploadPart is acknowledged once this rename lands, so the part data and
// its directory entry must be durable before we return. Relaxed keeps the
// part payload fdatasync but leaves the directory entry to the page cache.
let durability = effective_durability(dst_volume);
if durability.syncs_data_shards() && !src_is_dir {
let src = src_file_path.clone();
tokio::task::spawn_blocking(move || std::fs::File::open(&src)?.sync_data())
.await
.map_err(DiskError::from)?
.map_err(to_file_error)?;
}
rename_all(&src_file_path, &dst_file_path, &dst_volume_dir).await?;
if durability.syncs_commit_metadata()
&& let Some(parent) = dst_file_path.parent()
{
os::fsync_dir(parent).await.map_err(to_file_error)?;
}
let dst_meta = lstat_std(&dst_file_path).map_err(|e| -> DiskError { to_file_error(e).into() })?;
if src_is_dir != dst_meta.is_dir() {
warn!(
event = EVENT_DISK_LOCAL_RENAME_REJECTED,
component = LOG_COMPONENT_ECSTORE,
subsystem = LOG_SUBSYSTEM_DISK_LOCAL,
reason = "dst_type_changed_after_rename",
path = ?dst_file_path,
"Disk local rename rejected"
);
return Err(DiskError::FileAccessDenied);
}
self.write_all(dst_volume, format!("{dst_path}.meta").as_str(), meta).await?;
if let Some(parent) = src_file_path.parent() {
self.delete_file(&src_volume_dir, &parent.to_path_buf(), false, false).await?;
}
Ok(())
}
#[tracing::instrument(skip(self))]
async fn rename_file(&self, src_volume: &str, src_path: &str, dst_volume: &str, dst_path: &str) -> Result<()> {
crate::hp_guard!("LocalDisk::rename_file");
let src_volume_dir = self.get_bucket_path(src_volume)?;
let dst_volume_dir = self.get_bucket_path(dst_volume)?;
if !skip_access_checks(src_volume) {
access(&src_volume_dir)
.await
.map_err(|e| to_access_error(e, DiskError::VolumeAccessDenied))?;
}
if !skip_access_checks(dst_volume) {
access(&dst_volume_dir)
.await
.map_err(|e| to_access_error(e, DiskError::VolumeAccessDenied))?;
}
let src_is_dir = has_suffix(src_path, SLASH_SEPARATOR);
let dst_is_dir = has_suffix(dst_path, SLASH_SEPARATOR);
if (dst_is_dir || src_is_dir) && (!dst_is_dir || !src_is_dir) {
return Err(Error::from(DiskError::FileAccessDenied));
}
let src_file_path = self.get_object_path(src_volume, src_path)?;
check_path_length(src_file_path.to_string_lossy().as_ref())?;
let dst_file_path = self.get_object_path(dst_volume, dst_path)?;
check_path_length(dst_file_path.to_string_lossy().as_ref())?;
if src_is_dir {
let meta_op = match lstat(&src_file_path).await {
Ok(meta) => Some(meta),
Err(e) => {
let e: DiskError = to_file_error(e).into();
if e != DiskError::FileNotFound {
return Err(e);
} else {
None
}
}
};
if let Some(meta) = meta_op
&& !meta.is_dir()
{
return Err(DiskError::FileAccessDenied);
}
// Clear any stale destination before the directory rename. An absent
// destination is the normal case when renaming a directory to a new
// location, so tolerate NotFound instead of aborting the whole rename
// (MinIO's RenameFile ignores osIsNotExist here).
if let Err(e) = remove(&dst_file_path).await
&& e.kind() != ErrorKind::NotFound
{
return Err(to_file_error(e).into());
}
}
rename_all(&src_file_path, &dst_file_path, &dst_volume_dir).await?;
if let Some(parent) = src_file_path.parent() {
let _ = self.delete_file(&src_volume_dir, &parent.to_path_buf(), false, false).await;
}
Ok(())
}
#[tracing::instrument(level = "debug", skip(self))]
async fn create_file(&self, origvolume: &str, volume: &str, path: &str, _file_size: i64) -> Result<FileWriter> {
crate::hp_guard!("LocalDisk::create_file");
if !origvolume.is_empty() {
let origvolume_dir = self.get_bucket_path(origvolume)?;
if !skip_access_checks(origvolume) {
access(origvolume_dir)
.await
.map_err(|e| to_access_error(e, DiskError::VolumeAccessDenied))?;
}
}
self.io_backend
.open_write(volume, path, WriteMode::Truncate { size_hint: _file_size })
.await
}
#[tracing::instrument(level = "debug", skip(self))]
// async fn append_file(&self, volume: &str, path: &str, mut r: DuplexStream) -> Result<File> {
async fn append_file(&self, volume: &str, path: &str) -> Result<FileWriter> {
self.io_backend.open_write(volume, path, WriteMode::Append).await
}
#[tracing::instrument(level = "debug", skip(self))]
async fn read_file(&self, volume: &str, path: &str) -> Result<FileReader> {
crate::hp_guard!("LocalDisk::read_file");
self.io_backend.open_full_read(volume, path).await
}
#[tracing::instrument(level = "debug", skip(self))]
async fn read_file_stream(&self, volume: &str, path: &str, offset: usize, length: usize) -> Result<FileReader> {
crate::hp_guard!("LocalDisk::read_file_stream");
self.io_backend.open_read_stream(volume, path, offset, length).await
}
/// File read using mmap-then-copy on Unix or efficient read on non-Unix.
// SAFETY: Unix unsafe calls in this function only query page size and mmap
// a read-only file region after bounds and alignment are validated.
#[allow(unsafe_code)]
#[tracing::instrument(level = "debug", skip(self))]
async fn read_file_mmap_copy(&self, volume: &str, path: &str, offset: usize, length: usize) -> Result<Bytes> {
self.read_file_mmap_copy_with_metrics(volume, path, offset, length, None)
.await
}
/// File read using mmap-then-copy on Unix or efficient read on non-Unix.
#[tracing::instrument(level = "debug", skip(self))]
async fn read_file_mmap_copy_with_metrics(
&self,
volume: &str,
path: &str,
offset: usize,
length: usize,
metrics: Option<MmapCopyStageMetrics>,
) -> Result<Bytes> {
self.io_backend.pread_bytes(volume, path, offset, length, metrics).await
}
#[tracing::instrument(level = "debug", skip(self))]
async fn list_dir(&self, origvolume: &str, volume: &str, dir_path: &str, count: i32) -> Result<Vec<String>> {
if !origvolume.is_empty() {
let origvolume_dir = self.get_bucket_path(origvolume)?;
if !skip_access_checks(origvolume)
&& let Err(e) = access(origvolume_dir).await
{
return Err(to_access_error(e, DiskError::VolumeAccessDenied).into());
}
}
let volume_dir = self.get_bucket_path(volume)?;
let dir_path_abs = self.get_object_path(volume, dir_path.trim_start_matches(SLASH_SEPARATOR))?;
let entries = match os::read_dir(&dir_path_abs, count).await {
Ok(res) => res,
Err(e) => {
if e.kind() == ErrorKind::NotFound
&& !skip_access_checks(volume)
&& let Err(e) = access(&volume_dir).await
{
return Err(to_access_error(e, DiskError::VolumeAccessDenied).into());
}
return Err(to_file_error(e).into());
}
};
Ok(entries)
}
// FIXME: TODO: io.writer TODO cancel
#[tracing::instrument(level = "debug", skip(self, wr))]
async fn walk_dir<W: AsyncWrite + Unpin + Send>(&self, opts: WalkDirOptions, wr: &mut W) -> Result<()> {
self.wait_for_startup_cleanup().await;
let volume_dir = self.get_bucket_path(&opts.bucket)?;
if !skip_access_checks(&opts.bucket)
&& let Err(e) = access(&volume_dir).await
{
return Err(to_access_error(e, DiskError::VolumeAccessDenied).into());
}
let _scan_lock = self.try_acquire_scan_lock(&opts)?;
let mut wr = wr;
let mut out = MetacacheWriter::new(&mut wr);
let mut objs_returned = 0;
let mut skip_current_dir_object = false;
let mut multipart_dir_to_skip: HashSet<String> = HashSet::new();
if opts.base_dir.ends_with(SLASH_SEPARATOR) {
if let Ok(data) = self
.read_metadata(
&opts.bucket,
path_join_buf(&[
format!("{}{}", opts.base_dir.trim_end_matches(SLASH_SEPARATOR), GLOBAL_DIR_SUFFIX).as_str(),
STORAGE_FORMAT_FILE,
])
.as_str(),
)
.await
{
let meta = MetaCacheEntry {
name: opts.base_dir.clone(),
metadata: data.to_vec(),
..Default::default()
};
write_metacache_obj(&mut out, &meta).await?;
objs_returned += 1;
} else {
let fpath =
self.get_object_path(&opts.bucket, path_join_buf(&[opts.base_dir.as_str(), STORAGE_FORMAT_FILE]).as_str())?;
if let Ok(meta) = tokio::fs::metadata(&fpath).await
&& meta.is_file()
{
skip_current_dir_object = true;
if let Ok(meta_bytes) = self
.read_metadata(
opts.bucket.as_str(),
path_join_buf(&[opts.base_dir.as_str(), STORAGE_FORMAT_FILE]).as_str(),
)
.await
&& let Ok(file_meta) = FileMeta::load(&meta_bytes)
&& let Ok(data_dirs) = file_meta.get_data_dirs()
{
for data_dir in data_dirs.iter().flatten() {
multipart_dir_to_skip.insert(data_dir.to_string());
}
}
}
}
}
self.scan_dir(
opts.base_dir.clone(),
opts.filter_prefix.clone().unwrap_or_default(),
&opts,
&mut out,
&mut objs_returned,
skip_current_dir_object,
if multipart_dir_to_skip.is_empty() {
None
} else {
Some(multipart_dir_to_skip)
},
)
.await?;
Ok(())
}
#[tracing::instrument(level = "debug", skip(self, fi))]
async fn rename_data(
&self,
src_volume: &str,
src_path: &str,
fi: FileInfo,
dst_volume: &str,
dst_path: &str,
) -> Result<RenameDataResp> {
crate::hp_guard!("LocalDisk::rename_data");
let src_volume_dir = self.get_bucket_path(src_volume)?;
if !skip_access_checks(src_volume)
&& let Err(e) = super::fs::access_std(&src_volume_dir)
{
info!(
event = EVENT_DISK_LOCAL_ACCESS_FAILED,
component = LOG_COMPONENT_ECSTORE,
subsystem = LOG_SUBSYSTEM_DISK_LOCAL,
path = ?src_volume_dir,
operation = "rename_data_src_access",
error = %e,
"Disk local access check failed"
);
return Err(to_access_error(e, DiskError::VolumeAccessDenied).into());
}
let dst_volume_dir = self.get_bucket_path(dst_volume)?;
if !skip_access_checks(dst_volume)
&& let Err(e) = super::fs::access_std(&dst_volume_dir)
{
info!(
event = EVENT_DISK_LOCAL_ACCESS_FAILED,
component = LOG_COMPONENT_ECSTORE,
subsystem = LOG_SUBSYSTEM_DISK_LOCAL,
path = ?dst_volume_dir,
operation = "rename_data_dst_access",
error = %e,
"Disk local access check failed"
);
return Err(to_access_error(e, DiskError::VolumeAccessDenied).into());
}
// xl.meta path
let src_file_path = self.get_object_path(src_volume, format!("{}/{}", &src_path, STORAGE_FORMAT_FILE).as_str())?;
let dst_file_path = self.get_object_path(dst_volume, format!("{}/{}", &dst_path, STORAGE_FORMAT_FILE).as_str())?;
// data_dir path
let has_data_dir_path = {
let has_data_dir = {
if !fi.is_remote() {
fi.data_dir
.map(|dir| rustfs_utils::path::retain_slash(dir.to_string().as_str()))
} else {
None
}
};
if let Some(data_dir) = has_data_dir {
let src_data_path = self.get_object_path(
src_volume,
rustfs_utils::path::retain_slash(format!("{}/{}", &src_path, data_dir).as_str()).as_str(),
)?;
let dst_data_path = self.get_object_path(
dst_volume,
rustfs_utils::path::retain_slash(format!("{}/{}", &dst_path, data_dir).as_str()).as_str(),
)?;
Some((src_data_path, dst_data_path))
} else {
None
}
};
check_path_length(src_file_path.to_string_lossy().to_string().as_str())?;
check_path_length(dst_file_path.to_string_lossy().to_string().as_str())?;
let no_inline = fi.data.is_none() && fi.size > 0;
// Resolved once for the whole commit so a concurrent configuration
// change can never leave a single rename_data half-synced. The tier is
// keyed on the destination volume: user data staged in scratch
// namespaces follows the configured tier, while commits into
// system-critical namespaces (IAM, config, bucket metadata) stay
// pinned to strict.
let durability = effective_durability(dst_volume);
if no_inline {
// Non-inline: read xl.meta, parse, write, rename data dir, rename xl.meta
let has_dst_buf = match super::fs::read_file(&dst_file_path).await {
Ok(res) => Some(res),
Err(e) => {
let e: DiskError = to_file_error(e).into();
if e != DiskError::FileNotFound {
return Err(e);
}
None
}
};
let mut xlmeta = FileMeta::new();
if let Some(dst_buf) = has_dst_buf.as_ref()
&& FileMeta::is_xl2_v1_format(dst_buf)
&& let Ok(nmeta) = FileMeta::load(dst_buf)
{
xlmeta = nmeta
}
let mut skip_parent = dst_volume_dir.clone();
if has_dst_buf.as_ref().is_some()
&& let Some(parent) = dst_file_path.parent()
{
skip_parent = parent.to_path_buf();
}
let version_id = fi.version_id.unwrap_or_default();
let has_old_data_dir = xlmeta.find_unshared_data_dir_for_version(Some(version_id));
if let Some(old_data_dir) = has_old_data_dir.as_ref() {
let _ = xlmeta.data.remove_two(version_id, *old_data_dir);
}
xlmeta.add_version(fi)?;
let version_signature = rename_data_versions_signature(&xlmeta);
let new_dst_buf = xlmeta.marshal_msg()?;
let src_file_parent = src_file_path.parent().unwrap_or(src_volume_dir.as_path());
// This tmp xl.meta is renamed onto dst_file_path at the commit
// point below, so only its contents must be durable before the
// rename (SyncMode::FileOnly); the dst parent directory is fsynced
// after the commit rename, and a crash before the rename means the
// PUT was never acknowledged. A metadata commit: relaxed tiers
// leave it to the page cache.
let tmp_meta_sync = if durability.syncs_commit_metadata() {
SyncMode::FileOnly
} else {
SyncMode::None
};
// The tmp xl.meta write and the shard-file fdatasync are independent
// (disjoint paths) and both only need to be durable before the commit
// renames below, so run them concurrently to drop a blocking
// round-trip from the PUT commit critical path (rustfs/backlog#922
// step 2). The "contents durable -> rename -> dst dir fsync" ordering
// is unchanged — both futures complete before any rename — which the
// rename_data crash-consistency harness (backlog#935) exercises.
//
// Shard durability: once rename_data succeeds the write is
// acknowledged, so data must not live only in the page cache.
// Multipart parts were already synced during rename_part, so their
// fdatasync here is a cheap no-op. A missing source dir is left for the
// rename below to report through the existing rollback path. Payload
// durability is kept by both strict and relaxed.
// Bound to a local so the borrow lives across the join! below.
let tmp_meta_rel_path = format!("{}/{}", &src_path, STORAGE_FORMAT_FILE);
let tmp_meta_write =
self.write_all_private(src_volume, &tmp_meta_rel_path, new_dst_buf.into(), tmp_meta_sync, src_file_parent);
let shard_sync = async {
if durability.syncs_data_shards()
&& let Some((src_data_path, _)) = has_data_dir_path.as_ref()
&& let Err(err) = os::sync_dir_files(src_data_path).await
&& err.kind() != ErrorKind::NotFound
{
return Err::<(), DiskError>(to_file_error(err).into());
}
Ok(())
};
let (tmp_meta_res, shard_sync_res) = tokio::join!(tmp_meta_write, shard_sync);
// Surface a tmp-meta failure first (its prior serial position), then a
// shard-sync failure; either aborts before any rename, exactly as the
// sequential version did.
tmp_meta_res?;
shard_sync_res?;
if let Some((src_data_path, dst_data_path)) = has_data_dir_path.as_ref()
&& let Err(err) = rename_all(src_data_path, dst_data_path, &skip_parent).await
{
let _ = self.delete_file(&dst_volume_dir, dst_data_path, false, false).await;
info!(
event = EVENT_DISK_LOCAL_RENAME_REJECTED,
component = LOG_COMPONENT_ECSTORE,
subsystem = LOG_SUBSYSTEM_DISK_LOCAL,
reason = "rename_all_data_path_failed",
src_path = ?src_data_path,
dst_path = ?dst_data_path,
error = ?err,
"Disk local rename flow failed"
);
return Err(err);
}
// Crash-consistency injection: hard power loss after the data dir
// is in place but before xl.meta commits. No cleanup — the harness
// reopens the disk and asserts the object still reads as the old
// version (the staged data dir is a harmless orphan for GC).
if should_crash_rename_data_at(RenameDataCrashPoint::AfterDataRename, dst_path) {
return Err(DiskError::Unexpected);
}
if should_fail_before_old_metadata_backup(dst_path) {
if let Some((_, dst_data_path)) = has_data_dir_path.as_ref() {
let _ = self.delete_file(&dst_volume_dir, dst_data_path, false, false).await;
}
info!(
event = EVENT_DISK_LOCAL_RENAME_REJECTED,
component = LOG_COMPONENT_ECSTORE,
subsystem = LOG_SUBSYSTEM_DISK_LOCAL,
reason = "test_fail_before_old_metadata_backup",
"Disk local rename flow failed before metadata commit"
);
return Err(DiskError::Unexpected);
}
// The rollback backup stays where it is written (no rename) and is
// the sole restore source for a later undo_write, so under strict
// it keeps SyncMode::FileAndDir: contents and directory entry both
// durable. It is part of the metadata commit machinery, so relaxed
// tiers leave it to the page cache like the xl.meta it mirrors.
let backup_sync = if durability.syncs_commit_metadata() {
SyncMode::FileAndDir
} else {
SyncMode::None
};
if let Some(old_data_dir) = has_old_data_dir
&& let Some(dst_buf) = has_dst_buf.as_ref()
&& let Err(err) = self
.write_all_private(
dst_volume,
&format!("{}/{}/{}", &dst_path, &old_data_dir, STORAGE_FORMAT_FILE_BACKUP),
dst_buf.clone().into(),
backup_sync,
&skip_parent,
)
.await
{
if let Some((_, dst_data_path)) = has_data_dir_path.as_ref() {
let _ = self.delete_file(&dst_volume_dir, dst_data_path, false, false).await;
}
info!(
event = EVENT_DISK_LOCAL_RENAME_REJECTED,
component = LOG_COMPONENT_ECSTORE,
subsystem = LOG_SUBSYSTEM_DISK_LOCAL,
reason = "write_old_metadata_backup_failed",
error = ?err,
"Disk local rename flow failed"
);
return Err(err);
}
// Crash-consistency injection: hard power loss after the rollback
// backup is durable but before the xl.meta commit rename. No
// cleanup — the harness asserts the object still reads as the old
// version, since the destination xl.meta is untouched here.
if should_crash_rename_data_at(RenameDataCrashPoint::AfterBackupBeforeMetaCommit, dst_path) {
return Err(DiskError::Unexpected);
}
if let Err(err) = rename_all(&src_file_path, &dst_file_path, &skip_parent).await {
if let Some((_, dst_data_path)) = has_data_dir_path.as_ref() {
let _ = self.delete_file(&dst_volume_dir, dst_data_path, false, false).await;
}
info!(
event = EVENT_DISK_LOCAL_RENAME_REJECTED,
component = LOG_COMPONENT_ECSTORE,
subsystem = LOG_SUBSYSTEM_DISK_LOCAL,
reason = "rename_all_metadata_failed",
src_path = ?src_file_path,
dst_path = ?dst_file_path,
error = ?err,
"Disk local rename flow failed"
);
return Err(err);
}
// Persist the directory entries for both the data dir and xl.meta renames;
// without this the commit itself can vanish on power loss. Relaxed tiers
// accept that window (documented in docs/operations/durability-modes.md).
if durability.syncs_commit_metadata()
&& let Some(parent) = dst_file_path.parent()
&& let Err(err) = os::fsync_dir(parent).await
{
return Err(to_file_error(err).into());
}
// First PUT of an object creates its directory (and any missing prefix
// dirs) via reliable_mkdir_all, which never fsyncs the parent chain. The
// commit fsync above persists the object dir's *contents*, not its own
// entry in the bucket/prefix dir, so on power loss after ack the whole
// object dir could vanish (rustfs/backlog#922 step 4). For a new object
// (no prior xl.meta) fsync the ancestor chain from the object dir's
// parent up to and including the bucket so those new directory entries
// are durable. Overwrites already have a durable object dir. The
// starts_with guard bounds the walk to the bucket subtree. Relaxed/none
// accept the wider window, like the commit fsync above.
if has_dst_buf.is_none() && durability.syncs_commit_metadata() {
let mut ancestor = dst_file_path.parent().and_then(|object_dir| object_dir.parent());
while let Some(dir) = ancestor {
if !dir.starts_with(&dst_volume_dir) {
break;
}
os::fsync_dir(dir).await.map_err(to_file_error)?;
if dir == dst_volume_dir.as_path() {
break;
}
ancestor = dir.parent();
}
}
if let Some(src_file_path_parent) = src_file_path.parent() {
if src_volume != super::RUSTFS_META_MULTIPART_BUCKET {
let _ = remove_std(src_file_path_parent);
} else {
let _ = self
.delete_file(&dst_volume_dir, &src_file_path_parent.to_path_buf(), true, false)
.await;
}
}
Ok(RenameDataResp {
old_data_dir: has_old_data_dir,
sign: version_signature,
})
} else {
// Inline: merge read + parse + write + rename into single spawn_blocking
let src = src_file_path.clone();
let dst = dst_file_path.clone();
// Captured by the closure to fsync the new object's ancestor dir chain.
let bucket_dir = dst_volume_dir.clone();
let cleanup_path = if src_volume == super::RUSTFS_META_MULTIPART_BUCKET {
src_file_path.parent().map(|p| p.to_path_buf())
} else {
None
};
let (old_data_dir, version_signature) = tokio::task::spawn_blocking(move || {
// Read existing xl.meta
let has_dst_buf = match std::fs::read(&dst) {
Ok(buf) => Some(Bytes::from(buf)),
Err(e) if e.kind() == std::io::ErrorKind::NotFound => None,
Err(e) => return Err(to_file_error(e)),
};
let mut xlmeta = FileMeta::new();
if let Some(ref buf) = has_dst_buf
&& FileMeta::is_xl2_v1_format(buf)
&& let Ok(nmeta) = FileMeta::load(buf)
{
xlmeta = nmeta
}
let version_id = fi.version_id.unwrap_or_default();
let old_data_dir = xlmeta.find_unshared_data_dir_for_version(Some(version_id));
if let Some(d) = old_data_dir.as_ref() {
let _ = xlmeta.data.remove_two(version_id, *d);
}
xlmeta.add_version(fi)?;
let version_signature = rename_data_versions_signature(&xlmeta);
let new_buf = xlmeta.marshal_msg()?;
// Write new xl.meta + rename. Inline objects carry their data
// inside xl.meta, so this whole sequence is a metadata commit:
// relaxed tiers do no per-object fsync here at all (aligned
// with MinIO's default), trading a documented power-loss
// window for latency.
if let Some(parent) = src.parent() {
std::fs::create_dir_all(parent)?;
}
let sync = durability.syncs_commit_metadata();
let mut f = std::fs::OpenOptions::new()
.create(true)
.write(true)
.truncate(true)
.open(&src)?;
std::io::Write::write_all(&mut f, &new_buf)?;
if sync {
f.sync_data()?;
}
if let Some(old_dir) = old_data_dir.as_ref()
&& let Some(ref buf) = has_dst_buf
&& let Some(dst_parent) = dst.parent()
{
let old_path = dst_parent.join(old_dir.to_string()).join(STORAGE_FORMAT_FILE_BACKUP);
let old_parent = old_path.parent().map(|p| p.to_path_buf());
if let Some(ref old_parent) = old_parent {
std::fs::create_dir_all(old_parent)?;
}
// This rollback backup is the sole restore source for a later
// undo_write when the set-level write quorum fails. Persist it as
// durably as the new xl.meta written above (and as the non-inline
// branch does): a bare std::fs::write leaves both the bytes and the
// new directory entry in the page cache, so a crash before a
// rollback could restore a lost or truncated backup.
let mut backup = std::fs::OpenOptions::new()
.create(true)
.write(true)
.truncate(true)
.open(&old_path)
.map_err(to_file_error)?;
std::io::Write::write_all(&mut backup, buf).map_err(to_file_error)?;
if sync {
backup.sync_data().map_err(to_file_error)?;
if let Some(ref old_parent) = old_parent {
os::fsync_dir_std(old_parent).map_err(to_file_error)?;
}
}
}
match std::fs::rename(&src, &dst) {
Ok(()) => Ok(()),
Err(err) if err.kind() == std::io::ErrorKind::NotFound && !src.exists() => Ok(()),
Err(err) if err.kind() == std::io::ErrorKind::NotFound => {
if let Some(parent) = dst.parent() {
std::fs::create_dir_all(parent)?;
}
std::fs::rename(&src, &dst).map_err(to_file_error)?;
Ok(())
}
Err(err) => Err(to_file_error(err)),
}?;
// Persist the commit rename's directory entry across power loss.
if sync && let Some(dst_parent) = dst.parent() {
os::fsync_dir_std(dst_parent)?;
}
// Same power-loss gap as the non-inline path (rustfs/backlog#922
// step 4): a first PUT creates the object dir (and any missing
// prefix dirs) whose entry in the bucket/prefix dir reliable_mkdir_all
// never fsynced. The fsync above persists the object dir's contents,
// not its own entry, so for a new inline object fsync the ancestor
// chain up to and including the bucket. Overwrites already have a
// durable object dir; the starts_with guard bounds the walk.
if sync && has_dst_buf.is_none() {
let mut ancestor = dst.parent().and_then(|object_dir| object_dir.parent());
while let Some(ancestor_dir) = ancestor {
if !ancestor_dir.starts_with(&bucket_dir) {
break;
}
os::fsync_dir_std(ancestor_dir)?;
if ancestor_dir == bucket_dir.as_path() {
break;
}
ancestor = ancestor_dir.parent();
}
}
Ok::<(Option<uuid::Uuid>, Option<Vec<u8>>), std::io::Error>((old_data_dir, version_signature))
})
.await
.map_err(DiskError::from)??;
// Cleanup
if let Some(ref cleanup) = cleanup_path {
let _ = self.delete_file(&dst_volume_dir, cleanup, true, false).await;
} else if let Some(parent) = src_file_path.parent() {
let _ = remove_std(parent);
}
Ok(RenameDataResp {
old_data_dir,
sign: version_signature,
})
}
}
#[tracing::instrument(skip(self))]
async fn make_volumes(&self, volumes: Vec<&str>) -> Result<()> {
for vol in volumes {
if let Err(e) = self.make_volume(vol).await
&& e != DiskError::VolumeExists
{
error!(
event = EVENT_DISK_LOCAL_VOLUME_SETUP_FAILED,
component = LOG_COMPONENT_ECSTORE,
subsystem = LOG_SUBSYSTEM_DISK_LOCAL,
volume = vol,
operation = "make_volumes",
error = %e,
"Disk local volume setup failed"
);
return Err(e);
}
// TODO: health check
}
Ok(())
}
#[tracing::instrument(skip(self))]
async fn make_volume(&self, volume: &str) -> Result<()> {
if !Self::is_valid_volname(volume) {
return Err(Error::other("Invalid arguments specified"));
}
let volume_dir = self.get_bucket_path(volume)?;
if let Err(e) = access(&volume_dir).await {
if e.kind() == ErrorKind::NotFound {
os::make_dir_all(&volume_dir, self.root.as_path()).await?;
return Ok(());
}
error!(
event = EVENT_DISK_LOCAL_VOLUME_SETUP_FAILED,
component = LOG_COMPONENT_ECSTORE,
subsystem = LOG_SUBSYSTEM_DISK_LOCAL,
volume,
operation = "make_volume",
error = %e,
"Disk local volume setup failed"
);
return Err(to_volume_error(e).into());
}
Err(DiskError::VolumeExists)
}
#[tracing::instrument(skip(self))]
async fn list_volumes(&self) -> Result<Vec<VolumeInfo>> {
let mut volumes = Vec::new();
let entries = os::read_dir(&self.root, -1).await.map_err(to_volume_error)?;
for entry in entries {
if !has_suffix(&entry, SLASH_SEPARATOR) || !Self::is_valid_volname(clean(&entry).as_str()) {
continue;
}
volumes.push(VolumeInfo {
name: clean(&entry),
created: None,
});
}
Ok(volumes)
}
#[tracing::instrument(skip(self))]
async fn stat_volume(&self, volume: &str) -> Result<VolumeInfo> {
let volume_dir = self.get_bucket_path(volume)?;
let meta = lstat(&volume_dir).await.map_err(to_volume_error)?;
let modtime = match meta.modified() {
Ok(md) => Some(OffsetDateTime::from(md)),
Err(_) => None,
};
Ok(VolumeInfo {
name: volume.to_string(),
created: modtime,
})
}
#[tracing::instrument(skip(self))]
async fn delete_paths(&self, volume: &str, paths: &[String]) -> Result<()> {
let volume_dir = self.get_bucket_path(volume)?;
if !skip_access_checks(volume) {
access(&volume_dir)
.await
.map_err(|e| to_access_error(e, DiskError::VolumeAccessDenied))?;
}
for path in paths.iter() {
let file_path = self.get_object_path(volume, path)?;
check_path_length(file_path.to_string_lossy().as_ref())?;
self.move_to_trash(&file_path, false, false).await?;
}
Ok(())
}
#[tracing::instrument(skip(self))]
async fn update_metadata(&self, volume: &str, path: &str, fi: FileInfo, opts: &UpdateMetadataOpts) -> Result<()> {
if !fi.metadata.is_empty() {
let file_path = self.get_object_path(volume, path)?;
check_path_length(file_path.to_string_lossy().as_ref())?;
let buf = self
.read_all(volume, format!("{}/{}", &path, STORAGE_FORMAT_FILE).as_str())
.await
.map_err(|e| {
if e == DiskError::FileNotFound && fi.version_id.is_some() {
DiskError::FileVersionNotFound
} else {
e
}
})?;
if !FileMeta::is_xl2_v1_format(buf.as_ref()) {
return Err(DiskError::FileVersionNotFound);
}
let mut xl_meta = FileMeta::load(buf.as_ref())?;
xl_meta.update_object_version_with_opts(fi, opts.replace_user_metadata)?;
let wbuf = xl_meta.marshal_msg()?;
return self
.write_all_meta(volume, format!("{path}/{STORAGE_FORMAT_FILE}").as_str(), &wbuf, !opts.no_persistence)
.await;
}
Err(Error::other("Invalid Argument"))
}
#[tracing::instrument(skip(self))]
async fn write_metadata(&self, _org_volume: &str, volume: &str, path: &str, fi: FileInfo) -> Result<()> {
crate::hp_guard!("LocalDisk::write_metadata");
let p = self.get_object_path(volume, format!("{path}/{STORAGE_FORMAT_FILE}").as_str())?;
let mut meta = FileMeta::new();
if !fi.fresh {
let (buf, _) = read_file_exists(&p).await?;
if !buf.is_empty() {
let _ = meta.unmarshal_msg(&buf).map_err(|_| {
meta = FileMeta::new();
});
}
}
meta.add_version(fi)?;
let fm_data = meta.marshal_msg()?;
// Atomic temp+rename: this path also rewrites live xl.meta (delete markers,
// decommission), where an in-place truncate would expose torn metadata.
self.write_all_meta(volume, format!("{path}/{STORAGE_FORMAT_FILE}").as_str(), &fm_data, true)
.await?;
Ok(())
}
#[tracing::instrument(level = "debug", skip(self))]
async fn read_version(
&self,
org_volume: &str,
volume: &str,
path: &str,
version_id: &str,
opts: &ReadOptions,
) -> Result<FileInfo> {
crate::hp_guard!("LocalDisk::read_version");
if !org_volume.is_empty() {
let org_volume_path = self.get_bucket_path(org_volume)?;
if !skip_access_checks(org_volume) {
access(&org_volume_path)
.await
.map_err(|e| to_access_error(e, DiskError::VolumeAccessDenied))?;
}
}
let file_path = self.get_object_path(volume, path)?;
let volume_dir = self.get_bucket_path(volume)?;
check_path_length(file_path.to_string_lossy().as_ref())?;
let read_data = opts.read_data;
let (data, _) = self
.read_raw(volume, volume_dir.clone(), file_path, read_data)
.await
.map_err(|e| {
if e == DiskError::FileNotFound && !version_id.is_empty() {
DiskError::FileVersionNotFound
} else {
e
}
})?;
let mut fi = get_file_info(
&data,
volume,
path,
version_id,
FileInfoOpts {
data: read_data,
include_free_versions: opts.incl_free_versions,
},
)?;
if opts.read_data {
if fi.data.as_ref().is_some_and(|d| !d.is_empty()) || fi.size == 0 {
if fi.inline_data() {
return Ok(fi);
}
if fi.size == 0 || fi.version_id.is_none_or(|v| v.is_nil()) {
fi.set_inline_data();
return Ok(fi);
};
if let Some(part) = fi.parts.first() {
let part_path = format!("part.{}", part.number);
let part_path = path_join_buf(&[
path,
fi.data_dir.map_or_else(|| "".to_string(), |dir| dir.to_string()).as_str(),
part_path.as_str(),
]);
let part_path = self.get_object_path(volume, part_path.as_str())?;
if lstat(&part_path).await.is_err() {
fi.set_inline_data();
return Ok(fi);
}
}
fi.data = None;
}
let inline = fi.transition_status.is_empty() && fi.data_dir.is_some() && fi.parts.len() == 1;
if inline && fi.shard_file_size(fi.parts[0].actual_size) < DEFAULT_INLINE_BLOCK as i64 {
let part_path = path_join_buf(&[
path,
fi.data_dir.map_or_else(|| "".to_string(), |dir| dir.to_string()).as_str(),
format!("part.{}", fi.parts[0].number).as_str(),
]);
let part_path = self.get_object_path(volume, part_path.as_str())?;
let data = self.read_all_data(volume, volume_dir, part_path.clone()).await.map_err(|e| {
warn!(
event = EVENT_DISK_LOCAL_READ_VERSION_FALLBACK,
component = LOG_COMPONENT_ECSTORE,
subsystem = LOG_SUBSYSTEM_DISK_LOCAL,
path = ?part_path,
reason = "inline_data_read_failed",
error = %e,
"Disk local read_version fallback failed"
);
e
})?;
fi.data = Some(Bytes::from(data));
}
}
Ok(fi)
}
#[tracing::instrument(level = "debug", skip(self))]
async fn read_xl(&self, volume: &str, path: &str, read_data: bool) -> Result<RawFileInfo> {
crate::hp_guard!("LocalDisk::read_xl");
let file_path = self.get_object_path(volume, path)?;
let file_dir = self.get_bucket_path(volume)?;
let (buf, _) = self.read_raw(volume, file_dir, file_path, read_data).await?;
Ok(RawFileInfo { buf })
}
#[tracing::instrument(skip(self))]
async fn delete_version(
&self,
volume: &str,
path: &str,
fi: FileInfo,
force_del_marker: bool,
opts: DeleteOptions,
) -> Result<()> {
if path.starts_with(SLASH_SEPARATOR) {
return self
.delete(
volume,
path,
DeleteOptions {
recursive: false,
immediate: false,
..Default::default()
},
)
.await;
}
let volume_dir = self.get_bucket_path(volume)?;
let file_path = self.get_object_path(volume, path)?;
check_path_length(file_path.to_string_lossy().as_ref())?;
let xl_path = path_join(&[file_path.as_path(), Path::new(STORAGE_FORMAT_FILE)]);
let buf = match self.read_all_data(volume, &volume_dir, &xl_path).await {
Ok(res) => res,
Err(err) => {
if err != DiskError::FileNotFound {
return Err(err);
}
if fi.deleted && force_del_marker {
return self.write_metadata("", volume, path, fi).await;
}
return if fi.version_id.is_some() {
Err(DiskError::FileVersionNotFound)
} else {
Err(DiskError::FileNotFound)
};
}
};
let mut meta = FileMeta::load(&buf)?;
let old_dir = meta.delete_version(&fi)?;
if let Some(uuid) = old_dir {
let vid = fi.version_id.unwrap_or_default();
let _ = meta.data.remove(vec![vid, uuid])?;
let old_path = path_join(&[file_path.as_path(), Path::new(uuid.to_string().as_str())]);
check_path_length(old_path.to_string_lossy().as_ref())?;
if let Err(err) = self.move_to_trash(&old_path, true, false).await
&& err != DiskError::FileNotFound
&& err != DiskError::VolumeNotFound
{
return Err(err);
}
}
if let Some(old_data_dir) = opts.old_data_dir
&& opts.undo_write
{
let src_path = path_join(&[
file_path.as_path(),
Path::new(format!("{old_data_dir}{SLASH_SEPARATOR}{STORAGE_FORMAT_FILE_BACKUP}").as_str()),
]);
let dst_path = path_join(&[file_path.as_path(), Path::new(STORAGE_FORMAT_FILE)]);
return rename_all(&src_path, &dst_path, file_path).await;
}
if !meta.versions.is_empty() {
let buf = meta.marshal_msg()?;
return self
.write_all_meta(volume, format!("{path}{SLASH_SEPARATOR}{STORAGE_FORMAT_FILE}").as_str(), &buf, true)
.await;
}
self.delete_file(&volume_dir, &xl_path, true, false).await
}
#[tracing::instrument(level = "debug", skip(self))]
async fn delete_versions(&self, volume: &str, versions: Vec<FileInfoVersions>, _opts: DeleteOptions) -> Vec<Option<Error>> {
let mut errs = Vec::with_capacity(versions.len());
for _ in 0..versions.len() {
errs.push(None);
}
for (i, ver) in versions.iter().enumerate() {
if let Err(e) = self.delete_versions_internal(volume, ver.name.as_str(), &ver.versions).await {
errs[i] = Some(e);
} else {
errs[i] = None;
}
}
errs
}
#[tracing::instrument(skip(self))]
async fn read_multiple(&self, req: ReadMultipleReq) -> Result<Vec<ReadMultipleResp>> {
let mut results = Vec::new();
let mut found = 0;
for v in req.files.iter() {
let fpath = self.get_object_path(&req.bucket, format!("{}/{}", &req.prefix, v).as_str())?;
let mut res = ReadMultipleResp {
bucket: req.bucket.clone(),
prefix: req.prefix.clone(),
file: v.clone(),
..Default::default()
};
// if req.metadata_only {}
match read_file_all(&fpath).await {
Ok((data, meta)) => {
found += 1;
if req.max_size > 0 && data.len() > req.max_size {
res.exists = true;
res.error = format!("max size ({}) exceeded: {}", req.max_size, data.len());
results.push(res);
break;
}
res.exists = true;
res.data = data.into();
res.mod_time = match meta.modified() {
Ok(md) => Some(OffsetDateTime::from(md)),
Err(_) => {
warn!(
event = EVENT_DISK_LOCAL_FORMAT_DECODE_FAILED,
component = LOG_COMPONENT_ECSTORE,
subsystem = LOG_SUBSYSTEM_DISK_LOCAL,
reason = "modified_time_unsupported",
"Disk local modified time is unsupported on this platform"
);
None
}
};
results.push(res);
if req.max_results > 0 && found >= req.max_results {
break;
}
}
Err(e) => {
if e != DiskError::FileNotFound && e != DiskError::VolumeNotFound {
res.exists = true;
res.error = e.to_string();
}
if req.abort404 && !res.exists {
results.push(res);
break;
}
results.push(res);
}
}
}
Ok(results)
}
#[tracing::instrument(skip(self))]
async fn delete_volume(&self, volume: &str, force_delete: bool) -> Result<()> {
let p = self.get_bucket_path(volume)?;
// Non-force is non-recursive: `remove_dir` (rmdir) fails atomically with
// `DirectoryNotEmpty` -> VolumeNotEmpty if the bucket still holds any
// object data, so a misclassified "dangling" bucket on the heal path
// (or a non-force S3 DeleteBucket on a populated bucket) can never be
// recursively wiped. Only an explicit `force_delete` (e.g. S3 force
// bucket delete) removes recursively. Mirrors MinIO's
// xlStorage.DeleteVol (Remove vs RemoveAll). (backlog#799 B1)
let res = if force_delete {
fs::remove_dir_all(&p).await
} else {
fs::remove_dir(&p).await
};
if let Err(err) = res {
let e: DiskError = to_volume_error(err).into();
if e != DiskError::VolumeNotFound {
return Err(e);
}
}
Ok(())
}
#[tracing::instrument(skip(self))]
async fn disk_info(&self, _: &DiskInfoOptions) -> Result<DiskInfo> {
let mut info = Cache::get(self.disk_info_cache.clone()).await?;
info.nr_requests = self.nrrequests;
info.rotational = self.rotational;
info.mount_path = self.path().to_str().expect("operation should succeed").to_string();
info.endpoint = self.endpoint.to_string();
info.scanning = self.scanning.load(Ordering::Acquire) == 1;
if info.id.is_none() {
info.id = self.get_disk_id().await.unwrap_or(None);
}
Ok(info)
}
#[tracing::instrument(skip(self))]
fn start_scan(&self) -> ScanGuard {
self.scanning.fetch_add(1, Ordering::Release);
ScanGuard(Arc::clone(&self.scanning))
}
#[tracing::instrument(skip(self))]
async fn read_metadata(&self, volume: &str, path: &str) -> Result<Bytes> {
crate::hp_guard!("LocalDisk::read_metadata");
let file_path = self.get_object_path(volume, path)?;
let volume_dir = self.get_bucket_path(volume)?;
let (data, _) = self.read_all_data_with_dmtime(volume, volume_dir, file_path).await?;
Ok(data.into())
}
}
async fn wait_for_startup_cleanup_signal(
startup_cleanup_ready: &AtomicU32,
startup_cleanup_notify: &Notify,
wait_timeout: Duration,
) -> bool {
if startup_cleanup_ready.load(Ordering::Acquire) != 0 {
return true;
}
timeout(wait_timeout, async {
loop {
if startup_cleanup_ready.load(Ordering::Acquire) != 0 {
return;
}
let notified = startup_cleanup_notify.notified();
if startup_cleanup_ready.load(Ordering::Acquire) != 0 {
return;
}
notified.await;
}
})
.await
.is_ok()
}
#[tracing::instrument]
async fn get_disk_info(drive_path: PathBuf) -> Result<(rustfs_utils::os::DiskInfo, bool)> {
let drive_path = drive_path.to_string_lossy().to_string();
check_path_length(&drive_path)?;
let disk_info = get_info(&drive_path).inspect_err(|err| {
log_startup_disk_io_error("get_disk_info_stat", Path::new(&drive_path), err);
})?;
let root_drive = if let Some(root_disk_threshold) = runtime_sources::root_disk_threshold_for_erasure_disk().await {
if root_disk_threshold > 0 {
disk_info.total <= root_disk_threshold
} else {
is_root_disk(&drive_path, SLASH_SEPARATOR).unwrap_or_default()
}
} else {
false
};
Ok((disk_info, root_drive))
}
#[cfg(test)]
mod test {
use super::*;
use std::io;
use std::pin::Pin;
use std::task::{Context, Poll};
use tokio::io::{AsyncReadExt, AsyncWrite, ReadBuf};
fn test_file_info(name: &str, version_id: Uuid, data_dir: Option<Uuid>, data: Option<Bytes>) -> FileInfo {
let size = data
.as_ref()
.map(|data| i64::try_from(data.len()).expect("test data length should fit i64"))
.unwrap_or(1);
FileInfo {
name: name.to_string(),
version_id: Some(version_id),
data_dir,
data,
size,
mod_time: Some(OffsetDateTime::now_utc()),
..Default::default()
}
}
fn test_meta(fi: FileInfo) -> Vec<u8> {
let mut meta = FileMeta::default();
meta.add_version(fi).expect("test metadata should accept file info");
meta.marshal_msg().expect("test metadata should encode")
}
async fn ensure_test_volume(disk: &LocalDisk, volume: &str) {
match disk.make_volume(volume).await {
Ok(()) | Err(DiskError::VolumeExists) => {}
Err(err) => panic!("test volume should be available: {err:?}"),
}
}
/// Regression coverage for the disk-layer delete/rename fixes:
/// - move_to_trash must propagate real rename failures instead of silently
/// reporting success (rustfs/backlog#948, ECA-07).
/// - the directory (trailing-slash) branch of rename_file/rename_part must
/// tolerate a missing destination instead of aborting on NotFound
/// (rustfs/backlog#960, ECA-19).
mod delete_and_rename_regressions {
use super::*;
use tempfile::tempdir;
async fn new_disk() -> (LocalDisk, tempfile::TempDir) {
let dir = tempdir().expect("temp dir should be created");
let endpoint =
Endpoint::try_from(dir.path().to_str().expect("temp dir should be utf8")).expect("endpoint should parse");
let disk = LocalDisk::new(&endpoint, false).await.expect("local disk should be created");
(disk, dir)
}
// #948: a genuinely missing source is benign and must still return Ok.
#[tokio::test]
async fn move_to_trash_missing_source_is_ok() {
let (disk, dir) = new_disk().await;
let missing = dir.path().join("bucket").join("does-not-exist");
disk.move_to_trash(&missing, true, false)
.await
.expect("missing source must be treated as benign");
disk.move_to_trash(&missing, false, false)
.await
.expect("missing source must be treated as benign (non-recursive)");
}
// #948: a real rename failure (here ENOTDIR, because a path component is a
// regular file) must propagate instead of being swallowed as Ok(()). Before
// the fix every non-DiskFull error fell through to `return Ok(())`.
#[tokio::test]
async fn move_to_trash_propagates_real_rename_error() {
let (disk, dir) = new_disk().await;
let bucket_dir = dir.path().join("bucket");
fs::create_dir_all(&bucket_dir).await.expect("bucket dir should be created");
let regular_file = bucket_dir.join("afile");
fs::write(&regular_file, b"x").await.expect("regular file should be written");
// Traversing through the regular file yields ENOTDIR at rename time.
let bad_path = regular_file.join("child");
let err = disk
.move_to_trash(&bad_path, true, false)
.await
.expect_err("a real rename failure must propagate, not be reported as success");
assert_eq!(err, DiskError::FileAccessDenied, "ENOTDIR must map to FileAccessDenied via to_file_error");
}
// #948: the happy path is unchanged — an existing object is moved out of its
// original location and the call succeeds.
#[tokio::test]
async fn move_to_trash_moves_existing_object() {
let (disk, dir) = new_disk().await;
let object_dir = dir.path().join("bucket").join("obj-dir");
fs::create_dir_all(&object_dir).await.expect("object dir should be created");
fs::write(object_dir.join("part.1"), b"data")
.await
.expect("part should be written");
disk.move_to_trash(&object_dir, true, false)
.await
.expect("existing object should move to trash");
assert!(!object_dir.exists(), "object must be gone from its original location");
}
// #960: renaming a directory to a brand-new (non-existent) location must
// succeed. Before the fix the unconditional pre-rename remove returned
// FileNotFound and aborted the whole rename.
#[tokio::test]
async fn rename_file_directory_to_missing_destination_succeeds() {
let (disk, dir) = new_disk().await;
ensure_test_volume(&disk, "vol").await;
let src_dir = dir.path().join("vol").join("a").join("dir");
fs::create_dir_all(&src_dir).await.expect("src dir should be created");
fs::write(src_dir.join("file"), b"payload")
.await
.expect("src file should be written");
assert!(has_suffix("a/dir/", SLASH_SEPARATOR), "src path must carry directory semantics");
disk.rename_file("vol", "a/dir/", "vol", "b/newdir/")
.await
.expect("directory rename to a missing destination must succeed");
let moved = dir.path().join("vol").join("b").join("newdir").join("file");
assert_eq!(fs::read(&moved).await.expect("moved file should be readable"), b"payload");
assert!(!src_dir.exists(), "source directory must be gone after rename");
}
// #960: the same NotFound-tolerance fix applied to rename_part.
#[tokio::test]
async fn rename_part_directory_to_missing_destination_succeeds() {
let (disk, dir) = new_disk().await;
ensure_test_volume(&disk, "vol").await;
let src_dir = dir.path().join("vol").join("a").join("dir");
fs::create_dir_all(&src_dir).await.expect("src dir should be created");
fs::write(src_dir.join("file"), b"payload")
.await
.expect("src file should be written");
disk.rename_part("vol", "a/dir/", "vol", "b/newdir/", Bytes::from_static(b"meta-bytes"))
.await
.expect("directory rename_part to a missing destination must succeed");
let moved = dir.path().join("vol").join("b").join("newdir").join("file");
assert_eq!(fs::read(&moved).await.expect("moved file should be readable"), b"payload");
let meta = dir.path().join("vol").join("b").join("newdir").join(".meta");
assert_eq!(fs::read(&meta).await.expect("meta file should be readable"), b"meta-bytes");
assert!(!src_dir.exists(), "source directory must be gone after rename");
}
}
/// Crash-consistency harness for the rename_data commit sequence
/// (rustfs/backlog#935 HP-14, test plan rustfs/backlog#896; hard rule from
/// rustfs/backlog#878: "partial commit 后对象只能是旧版本或新版本,不能混合").
///
/// For every pre-commit crash point × durability tier, it seeds a committed
/// object, stages a replacement, injects a hard power loss (no in-process
/// rollback runs), reopens the disk to model a restart, and asserts the
/// object still reads back as exactly the old version — or, when there was
/// no old version, does not exist. The un-injected run asserts the commit
/// makes the new version visible. Relaxed is exercised alongside Strict so
/// the durability relaxations landing in HP-1/HP-4/HP-5 are held to the same
/// old-or-new invariant, only with a wider (documented) power-loss window.
mod crash_consistency {
use super::*;
use tempfile::tempdir;
const VERSION_ID: &str = "aaaaaaaa-aaaa-aaaa-aaaa-aaaaaaaaaaaa";
const OLD_DATA_DIR: &str = "bbbbbbbb-bbbb-bbbb-bbbb-bbbbbbbbbbbb";
const NEW_DATA_DIR: &str = "cccccccc-cccc-cccc-cccc-cccccccccccc";
async fn run_scenario(mode: DurabilityMode, crash: Option<RenameDataCrashPoint>, with_old_version: bool) {
// Serializes with every other durability-sensitive test and pins the
// resolved tier for the whole scenario (held until dropped).
let _mode = durability_mode_override::set(mode);
let dir = tempdir().expect("temp dir should be created");
let endpoint =
Endpoint::try_from(dir.path().to_str().expect("temp dir should be utf8")).expect("endpoint should parse");
let disk = LocalDisk::new(&endpoint, false).await.expect("local disk should be created");
let bucket = "bucket";
let object = "crash-object";
let tmp_object = "tmp-crash-object";
let version_id = Uuid::parse_str(VERSION_ID).expect("version id should parse");
let old_data_dir = Uuid::parse_str(OLD_DATA_DIR).expect("old data dir should parse");
let new_data_dir = Uuid::parse_str(NEW_DATA_DIR).expect("new data dir should parse");
ensure_test_volume(&disk, bucket).await;
ensure_test_volume(&disk, RUSTFS_META_TMP_BUCKET).await;
let object_dir = dir.path().join(bucket).join(object);
let meta_path = object_dir.join(STORAGE_FORMAT_FILE);
let old_meta = if with_old_version {
let old_fi = test_file_info(object, version_id, Some(old_data_dir), None);
let old_meta = test_meta(old_fi);
fs::create_dir_all(object_dir.join(old_data_dir.to_string()))
.await
.expect("old data dir should be created");
fs::write(&meta_path, &old_meta)
.await
.expect("old metadata should be written");
Some(old_meta)
} else {
None
};
// Stage the replacement version's shard data under the tmp bucket.
let tmp_data_dir = dir
.path()
.join(RUSTFS_META_TMP_BUCKET)
.join(tmp_object)
.join(new_data_dir.to_string());
fs::create_dir_all(&tmp_data_dir)
.await
.expect("new tmp data dir should be created");
fs::write(tmp_data_dir.join("part.1"), b"new-data")
.await
.expect("new tmp data should be written");
if let Some(point) = crash {
arm_rename_data_crash(point, object);
}
let new_fi = test_file_info(object, version_id, Some(new_data_dir), None);
let result = disk
.rename_data(RUSTFS_META_TMP_BUCKET, tmp_object, new_fi, bucket, object)
.await;
// Reopen the disk to model a process restart after the crash.
drop(disk);
let disk = LocalDisk::new(&endpoint, false).await.expect("local disk should reopen");
let read = disk
.read_version("", bucket, object, &version_id.to_string(), &ReadOptions::default())
.await;
match crash {
Some(point) => {
assert!(result.is_err(), "{mode:?}/{point:?}: an armed crash must surface as an error");
match &old_meta {
Some(old) => {
// The commit rename never ran: xl.meta is byte-for-byte
// the old version and its data dir is intact.
let after = fs::read(&meta_path).await.expect("old metadata must survive the crash");
assert_eq!(&after, old, "{mode:?}/{point:?}: xl.meta must remain the old version");
assert!(
object_dir.join(old_data_dir.to_string()).exists(),
"{mode:?}/{point:?}: old data dir must remain on disk"
);
let fi = read.expect("old version must be readable after the crash");
assert_eq!(
fi.data_dir,
Some(old_data_dir),
"{mode:?}/{point:?}: read must resolve to the old data dir, never the half-committed new one"
);
}
None => {
// No prior version: a pre-commit crash must leave no
// object behind at all.
assert!(
!meta_path.exists(),
"{mode:?}/{point:?}: no old version means no xl.meta after a pre-commit crash"
);
let err = read.expect_err("absent object must not be readable");
assert!(
matches!(err, DiskError::FileNotFound | DiskError::FileVersionNotFound),
"{mode:?}/{point:?}: unexpected error for absent object: {err:?}"
);
}
}
}
None => {
result.expect("un-injected rename_data must commit");
let fi = read.expect("new version must be readable after commit");
assert_eq!(
fi.data_dir,
Some(new_data_dir),
"{mode:?}: read must resolve to the newly committed data dir"
);
assert!(
object_dir.join(new_data_dir.to_string()).exists(),
"{mode:?}: new data dir must be in place after commit"
);
}
}
}
const CRASH_POINTS: [RenameDataCrashPoint; 2] = [
RenameDataCrashPoint::AfterDataRename,
RenameDataCrashPoint::AfterBackupBeforeMetaCommit,
];
#[tokio::test]
async fn overwrite_pre_commit_crash_keeps_old_version_strict() {
for point in CRASH_POINTS {
run_scenario(DurabilityMode::Strict, Some(point), true).await;
}
}
#[tokio::test]
async fn overwrite_pre_commit_crash_keeps_old_version_relaxed() {
for point in CRASH_POINTS {
run_scenario(DurabilityMode::Relaxed, Some(point), true).await;
}
}
#[tokio::test]
async fn fresh_pre_commit_crash_leaves_no_object_strict() {
for point in CRASH_POINTS {
run_scenario(DurabilityMode::Strict, Some(point), false).await;
}
}
#[tokio::test]
async fn fresh_pre_commit_crash_leaves_no_object_relaxed() {
for point in CRASH_POINTS {
run_scenario(DurabilityMode::Relaxed, Some(point), false).await;
}
}
#[tokio::test]
async fn commit_without_crash_makes_new_version_visible() {
run_scenario(DurabilityMode::Strict, None, true).await;
run_scenario(DurabilityMode::Relaxed, None, true).await;
run_scenario(DurabilityMode::Strict, None, false).await;
}
}
struct BlockingScanWriter {
entered_tx: Option<tokio::sync::oneshot::Sender<()>>,
}
impl AsyncWrite for BlockingScanWriter {
fn poll_write(mut self: Pin<&mut Self>, _cx: &mut Context<'_>, _buf: &[u8]) -> Poll<io::Result<usize>> {
if let Some(tx) = self.entered_tx.take() {
let _ = tx.send(());
}
Poll::Pending
}
fn poll_flush(self: Pin<&mut Self>, _cx: &mut Context<'_>) -> Poll<io::Result<()>> {
Poll::Pending
}
fn poll_shutdown(self: Pin<&mut Self>, _cx: &mut Context<'_>) -> Poll<io::Result<()>> {
Poll::Pending
}
}
#[tokio::test]
async fn test_local_disk_scan_rejects_concurrent_same_prefix_and_releases_on_cancel() {
use tempfile::tempdir;
let dir = tempdir().expect("temp dir should be created");
let bucket = "test-bucket";
let object_dir = dir.path().join(bucket).join("prefix/object");
fs::create_dir_all(&object_dir).await.expect("object dir should be created");
fs::write(object_dir.join(STORAGE_FORMAT_FILE), b"meta")
.await
.expect("object metadata should be written");
let endpoint = Endpoint::try_from(dir.path().to_str().expect("temp dir should be utf8")).expect("endpoint should parse");
let disk = Arc::new(LocalDisk::new(&endpoint, false).await.expect("local disk should be created"));
let opts = WalkDirOptions {
bucket: bucket.to_string(),
base_dir: "prefix/".to_string(),
recursive: true,
..Default::default()
};
let (entered_tx, entered_rx) = tokio::sync::oneshot::channel();
let first_disk = Arc::clone(&disk);
let first_opts = opts.clone();
let mut blocking_writer = BlockingScanWriter {
entered_tx: Some(entered_tx),
};
let first_scan = tokio::spawn(async move { first_disk.walk_dir(first_opts, &mut blocking_writer).await });
entered_rx.await.expect("first scan should enter write path");
let mut second_writer = tokio::io::sink();
let second_scan = disk.walk_dir(opts.clone(), &mut second_writer).await;
assert!(
matches!(second_scan, Err(DiskError::DiskOngoingReq)),
"concurrent scan of same bucket and prefix must be rejected, got {second_scan:?}"
);
first_scan.abort();
assert!(
first_scan
.await
.expect_err("first scan task should be cancelled")
.is_cancelled(),
"aborting the blocked scan should cancel the task"
);
let mut after_cancel_writer = tokio::io::sink();
let after_cancel = disk.walk_dir(opts, &mut after_cancel_writer).await;
assert!(
after_cancel.is_ok(),
"cancelled scan must release the bucket/prefix lock, got {after_cancel:?}"
);
}
#[tokio::test]
async fn test_rename_data_writes_old_metadata_backup_before_non_inline_undo() {
use tempfile::tempdir;
let dir = tempdir().expect("temp dir should be created");
let endpoint = Endpoint::try_from(dir.path().to_str().expect("temp dir should be utf8")).expect("endpoint should parse");
let disk = LocalDisk::new(&endpoint, false).await.expect("local disk should be created");
let bucket = "bucket";
let object = "dir/object";
let tmp_object = "tmp-write";
let version_id = Uuid::parse_str("11111111-1111-1111-1111-111111111111").expect("version id should parse");
let old_data_dir = Uuid::parse_str("22222222-2222-2222-2222-222222222222").expect("old data dir should parse");
let new_data_dir = Uuid::parse_str("33333333-3333-3333-3333-333333333333").expect("new data dir should parse");
ensure_test_volume(&disk, bucket).await;
ensure_test_volume(&disk, RUSTFS_META_TMP_BUCKET).await;
let old_fi = test_file_info(object, version_id, Some(old_data_dir), None);
let dst_object_dir = dir.path().join(bucket).join("dir/object");
fs::create_dir_all(dst_object_dir.join(old_data_dir.to_string()))
.await
.expect("old data dir should be created");
fs::write(dst_object_dir.join(STORAGE_FORMAT_FILE), test_meta(old_fi))
.await
.expect("old metadata should be written");
let tmp_data_dir = dir
.path()
.join(RUSTFS_META_TMP_BUCKET)
.join(tmp_object)
.join(new_data_dir.to_string());
fs::create_dir_all(&tmp_data_dir)
.await
.expect("new tmp data dir should be created");
fs::write(tmp_data_dir.join("part.1"), b"new-data")
.await
.expect("new tmp data should be written");
let new_fi = test_file_info(object, version_id, Some(new_data_dir), None);
let resp = disk
.rename_data(RUSTFS_META_TMP_BUCKET, tmp_object, new_fi, bucket, object)
.await
.expect("rename_data should commit");
assert_eq!(resp.old_data_dir, Some(old_data_dir));
assert_eq!(resp.sign, Some(version_id.as_bytes().to_vec()));
assert!(
dst_object_dir
.join(old_data_dir.to_string())
.join(STORAGE_FORMAT_FILE_BACKUP)
.exists()
);
assert!(
!dst_object_dir
.join(old_data_dir.to_string())
.join(STORAGE_FORMAT_FILE)
.exists()
);
}
#[tokio::test]
#[allow(clippy::await_holding_lock)]
async fn test_write_all_meta_skips_tmp_parent_dir_fsync_but_fsyncs_dst_parent() {
use tempfile::tempdir;
let _mode = durability_mode_override::set(DurabilityMode::Strict);
let dir = tempdir().expect("temp dir should be created");
let endpoint = Endpoint::try_from(dir.path().to_str().expect("temp dir should be utf8")).expect("endpoint should parse");
let disk = LocalDisk::new(&endpoint, false).await.expect("local disk should be created");
let bucket = "sync-meta-bucket";
ensure_test_volume(&disk, bucket).await;
ensure_test_volume(&disk, RUSTFS_META_TMP_BUCKET).await;
let meta_path = format!("dir/object/{STORAGE_FORMAT_FILE}");
disk.write_all_meta(bucket, &meta_path, b"payload", true)
.await
.expect("write_all_meta should succeed");
let dst_file_path = disk.get_object_path(bucket, &meta_path).expect("dst path should resolve");
assert_eq!(
tokio::fs::read(&dst_file_path).await.expect("xl.meta should exist"),
b"payload",
"renamed xl.meta must carry the written contents"
);
let tmp_parent = disk
.get_bucket_path(RUSTFS_META_TMP_BUCKET)
.expect("tmp bucket path should resolve");
assert!(
!os::fsync_dir_recorder::was_fsynced(&tmp_parent),
"tmp parent dir must not be fsynced for a write-then-rename tmp file"
);
let dst_parent = dst_file_path.parent().expect("dst file should have a parent").to_path_buf();
assert!(
os::fsync_dir_recorder::was_fsynced(&dst_parent),
"destination parent dir must be fsynced after the commit rename"
);
}
#[tokio::test]
#[allow(clippy::await_holding_lock)]
async fn test_write_all_public_still_fsyncs_parent_dir() {
use tempfile::tempdir;
let _mode = durability_mode_override::set(DurabilityMode::Strict);
let dir = tempdir().expect("temp dir should be created");
let endpoint = Endpoint::try_from(dir.path().to_str().expect("temp dir should be utf8")).expect("endpoint should parse");
let disk = LocalDisk::new(&endpoint, false).await.expect("local disk should be created");
let bucket = "sync-public-bucket";
ensure_test_volume(&disk, bucket).await;
disk.write_all(bucket, "config/settings.json", Bytes::from_static(b"payload"))
.await
.expect("write_all should succeed");
let file_path = disk
.get_object_path(bucket, "config/settings.json")
.expect("file path should resolve");
let parent = file_path.parent().expect("file should have a parent").to_path_buf();
assert!(
os::fsync_dir_recorder::was_fsynced(&parent),
"direct (non-renamed) writes must keep fsyncing their parent dir"
);
}
#[tokio::test]
#[allow(clippy::await_holding_lock)]
async fn test_rename_data_non_inline_skips_tmp_parent_dir_fsync() {
use tempfile::tempdir;
let _mode = durability_mode_override::set(DurabilityMode::Strict);
let dir = tempdir().expect("temp dir should be created");
let endpoint = Endpoint::try_from(dir.path().to_str().expect("temp dir should be utf8")).expect("endpoint should parse");
let disk = LocalDisk::new(&endpoint, false).await.expect("local disk should be created");
let bucket = "sync-rename-bucket";
let object = "dir/object";
let tmp_object = "tmp-sync-write";
let version_id = Uuid::parse_str("44444444-4444-4444-4444-444444444444").expect("version id should parse");
let old_data_dir = Uuid::parse_str("55555555-5555-5555-5555-555555555555").expect("old data dir should parse");
let new_data_dir = Uuid::parse_str("66666666-6666-6666-6666-666666666666").expect("new data dir should parse");
ensure_test_volume(&disk, bucket).await;
ensure_test_volume(&disk, RUSTFS_META_TMP_BUCKET).await;
let old_fi = test_file_info(object, version_id, Some(old_data_dir), None);
let dst_object_dir = dir.path().join(bucket).join(object);
fs::create_dir_all(dst_object_dir.join(old_data_dir.to_string()))
.await
.expect("old data dir should be created");
fs::write(dst_object_dir.join(STORAGE_FORMAT_FILE), test_meta(old_fi))
.await
.expect("old metadata should be written");
let tmp_data_dir = dir
.path()
.join(RUSTFS_META_TMP_BUCKET)
.join(tmp_object)
.join(new_data_dir.to_string());
fs::create_dir_all(&tmp_data_dir)
.await
.expect("new tmp data dir should be created");
fs::write(tmp_data_dir.join("part.1"), b"new-data")
.await
.expect("new tmp data should be written");
let new_fi = test_file_info(object, version_id, Some(new_data_dir), None);
disk.rename_data(RUSTFS_META_TMP_BUCKET, tmp_object, new_fi, bucket, object)
.await
.expect("rename_data should commit");
// The tmp xl.meta write point uses SyncMode::FileOnly: its parent dir
// ({tmp}/{tmp_object}) must not be fsynced.
let tmp_meta_parent = disk
.get_object_path(RUSTFS_META_TMP_BUCKET, &format!("{tmp_object}/{STORAGE_FORMAT_FILE}"))
.expect("tmp meta path should resolve")
.parent()
.expect("tmp meta should have a parent")
.to_path_buf();
assert!(
!os::fsync_dir_recorder::was_fsynced(&tmp_meta_parent),
"tmp xl.meta parent dir must not be fsynced for the write-then-rename tmp file"
);
// The commit sequence itself is untouched: the destination parent dir
// is fsynced after the commit rename, ...
let dst_meta_parent = disk
.get_object_path(bucket, &format!("{object}/{STORAGE_FORMAT_FILE}"))
.expect("dst meta path should resolve")
.parent()
.expect("dst meta should have a parent")
.to_path_buf();
assert!(
os::fsync_dir_recorder::was_fsynced(&dst_meta_parent),
"destination parent dir must be fsynced after the commit rename"
);
// ... and the rollback backup (which stays in place, no rename) still
// fsyncs its parent dir (SyncMode::FileAndDir).
let backup_parent = disk
.get_object_path(bucket, &format!("{object}/{old_data_dir}/{STORAGE_FORMAT_FILE_BACKUP}"))
.expect("backup path should resolve")
.parent()
.expect("backup should have a parent")
.to_path_buf();
assert!(
os::fsync_dir_recorder::was_fsynced(&backup_parent),
"old-metadata rollback backup must keep fsyncing its parent dir"
);
}
// Seed a first PUT of `object` (no prior version) through the non-inline
// rename_data path and return (disk, tempdir). The object dir and any prefix
// dirs are created during the commit.
async fn commit_new_object(mode: DurabilityMode, bucket: &str, object: &str) -> (LocalDisk, tempfile::TempDir) {
use tempfile::tempdir;
let dir = tempdir().expect("temp dir should be created");
let endpoint = Endpoint::try_from(dir.path().to_str().expect("temp dir should be utf8")).expect("endpoint should parse");
let disk = LocalDisk::new(&endpoint, false).await.expect("local disk should be created");
ensure_test_volume(&disk, bucket).await;
ensure_test_volume(&disk, RUSTFS_META_TMP_BUCKET).await;
let tmp_object = "tmp-new-object";
let version_id = Uuid::parse_str("77777777-7777-7777-7777-777777777777").expect("version id should parse");
let new_data_dir = Uuid::parse_str("88888888-8888-8888-8888-888888888888").expect("new data dir should parse");
let tmp_data_dir = dir
.path()
.join(RUSTFS_META_TMP_BUCKET)
.join(tmp_object)
.join(new_data_dir.to_string());
fs::create_dir_all(&tmp_data_dir)
.await
.expect("new tmp data dir should be created");
fs::write(tmp_data_dir.join("part.1"), b"new-data")
.await
.expect("new tmp data should be written");
let _mode = durability_mode_override::set(mode);
let new_fi = test_file_info(object, version_id, Some(new_data_dir), None);
disk.rename_data(RUSTFS_META_TMP_BUCKET, tmp_object, new_fi, bucket, object)
.await
.expect("rename_data should commit the new object");
(disk, dir)
}
#[tokio::test]
async fn test_rename_data_new_object_fsyncs_new_ancestor_dirs() {
// A first PUT under a new prefix must fsync every newly created ancestor
// directory (prefix dir and bucket dir) so the object dir's own entry
// survives power loss after ack (rustfs/backlog#922 step 4).
let bucket = "new-object-bucket";
let (disk, _dir) = commit_new_object(DurabilityMode::Strict, bucket, "prefix/new-object").await;
let bucket_dir = disk.get_bucket_path(bucket).expect("bucket path should resolve");
let prefix_dir = disk.get_object_path(bucket, "prefix").expect("prefix path should resolve");
assert!(
os::fsync_dir_recorder::was_fsynced(&prefix_dir),
"the newly created prefix dir must be fsynced"
);
assert!(
os::fsync_dir_recorder::was_fsynced(&bucket_dir),
"the bucket dir must be fsynced so the new prefix entry survives power loss"
);
}
#[tokio::test]
async fn test_rename_data_relaxed_new_object_skips_ancestor_fsync() {
// Relaxed persists shard payload but leaves metadata/directory commits to
// the page cache, so a new object must not fsync the ancestor chain.
let bucket = "new-object-bucket-relaxed";
let (disk, _dir) = commit_new_object(DurabilityMode::Relaxed, bucket, "prefix/new-object").await;
let bucket_dir = disk.get_bucket_path(bucket).expect("bucket path should resolve");
assert!(
!os::fsync_dir_recorder::was_fsynced(&bucket_dir),
"relaxed durability must not fsync the bucket dir"
);
}
#[tokio::test]
async fn test_rename_data_new_inline_object_fsyncs_new_ancestor_dirs() {
// The inline commit path (fi.data present) has the same mkdir gap as the
// non-inline path: a first PUT under a new prefix must fsync the newly
// created prefix and bucket dirs.
use tempfile::tempdir;
let dir = tempdir().expect("temp dir should be created");
let endpoint = Endpoint::try_from(dir.path().to_str().expect("temp dir should be utf8")).expect("endpoint should parse");
let disk = LocalDisk::new(&endpoint, false).await.expect("local disk should be created");
let bucket = "new-inline-bucket";
let object = "prefix/new-inline-object";
let tmp_object = "tmp-new-inline";
ensure_test_volume(&disk, bucket).await;
ensure_test_volume(&disk, RUSTFS_META_TMP_BUCKET).await;
let _mode = durability_mode_override::set(DurabilityMode::Strict);
let version_id = Uuid::parse_str("99999999-9999-9999-9999-999999999999").expect("version id should parse");
// fi.data present -> no_inline is false -> the inline commit branch runs.
let new_fi = test_file_info(object, version_id, None, Some(Bytes::from_static(b"inline-payload")));
disk.rename_data(RUSTFS_META_TMP_BUCKET, tmp_object, new_fi, bucket, object)
.await
.expect("inline rename_data should commit the new object");
let bucket_dir = disk.get_bucket_path(bucket).expect("bucket path should resolve");
let prefix_dir = disk.get_object_path(bucket, "prefix").expect("prefix path should resolve");
assert!(
os::fsync_dir_recorder::was_fsynced(&prefix_dir),
"the newly created prefix dir must be fsynced on an inline first PUT"
);
assert!(
os::fsync_dir_recorder::was_fsynced(&bucket_dir),
"the bucket dir must be fsynced on an inline first PUT"
);
}
#[test]
fn test_resolve_durability_mode_mapping() {
// Default: nothing set -> strict (current main behavior).
assert_eq!(resolve_durability_mode(None, DEFAULT_RUSTFS_DRIVE_SYNC_ENABLE), DurabilityMode::Strict);
// Legacy switch compatibility mapping.
assert_eq!(resolve_durability_mode(None, true), DurabilityMode::Strict);
assert_eq!(resolve_durability_mode(None, false), DurabilityMode::LegacyOff);
// Explicit mode wins over the legacy switch.
assert_eq!(resolve_durability_mode(Some("strict".into()), false), DurabilityMode::Strict);
assert_eq!(resolve_durability_mode(Some("relaxed".into()), true), DurabilityMode::Relaxed);
assert_eq!(resolve_durability_mode(Some("none".into()), true), DurabilityMode::None);
// Case- and whitespace-tolerant.
assert_eq!(resolve_durability_mode(Some(" RELAXED ".into()), true), DurabilityMode::Relaxed);
// Invalid values fall back to the legacy switch, then the default.
assert_eq!(resolve_durability_mode(Some("bogus".into()), true), DurabilityMode::Strict);
assert_eq!(resolve_durability_mode(Some("bogus".into()), false), DurabilityMode::LegacyOff);
assert_eq!(resolve_durability_mode(Some(String::new()), true), DurabilityMode::Strict);
}
#[test]
fn test_durability_mode_sync_gates() {
// Strict = current main behavior: every commit point synced.
assert!(DurabilityMode::Strict.syncs_data_shards());
assert!(DurabilityMode::Strict.syncs_commit_metadata());
// Relaxed keeps payload durability, drops metadata-commit fsyncs.
assert!(DurabilityMode::Relaxed.syncs_data_shards());
assert!(!DurabilityMode::Relaxed.syncs_commit_metadata());
// None and the legacy full-off switch sync nothing on the data path.
assert!(!DurabilityMode::None.syncs_data_shards());
assert!(!DurabilityMode::None.syncs_commit_metadata());
assert!(!DurabilityMode::LegacyOff.syncs_data_shards());
assert!(!DurabilityMode::LegacyOff.syncs_commit_metadata());
}
#[test]
fn test_system_critical_volume_classification() {
// System namespaces are pinned.
assert!(is_system_critical_volume(RUSTFS_META_BUCKET));
assert!(is_system_critical_volume(&format!("{RUSTFS_META_BUCKET}/buckets")));
assert!(is_system_critical_volume(&format!("{RUSTFS_META_BUCKET}/config")));
assert!(is_system_critical_volume(super::super::MIGRATING_META_BUCKET));
// Scratch namespaces stage user object data and follow the tier.
assert!(!is_system_critical_volume(RUSTFS_META_TMP_BUCKET));
assert!(!is_system_critical_volume(RUSTFS_META_TMP_DELETED_BUCKET));
assert!(!is_system_critical_volume(super::super::RUSTFS_META_MULTIPART_BUCKET));
// User buckets follow the tier; similarly-prefixed names are not meta.
assert!(!is_system_critical_volume("my-bucket"));
assert!(!is_system_critical_volume(".rustfs.sys-lookalike"));
}
#[test]
fn test_effective_durability_pins_system_volumes() {
{
let _mode = durability_mode_override::set(DurabilityMode::Relaxed);
assert_eq!(effective_durability("user-bucket"), DurabilityMode::Relaxed);
assert_eq!(effective_durability(super::super::RUSTFS_META_MULTIPART_BUCKET), DurabilityMode::Relaxed);
assert_eq!(effective_durability(RUSTFS_META_TMP_BUCKET), DurabilityMode::Relaxed);
assert_eq!(effective_durability(RUSTFS_META_BUCKET), DurabilityMode::Strict);
}
{
let _mode = durability_mode_override::set(DurabilityMode::None);
assert_eq!(effective_durability("user-bucket"), DurabilityMode::None);
assert_eq!(effective_durability(RUSTFS_META_BUCKET), DurabilityMode::Strict);
}
{
// The legacy full-off switch keeps its historical semantics:
// nothing is pinned, not even system-critical namespaces.
let _mode = durability_mode_override::set(DurabilityMode::LegacyOff);
assert_eq!(effective_durability("user-bucket"), DurabilityMode::LegacyOff);
assert_eq!(effective_durability(RUSTFS_META_BUCKET), DurabilityMode::LegacyOff);
}
}
/// Removes the bucket's durability override when dropped so a test can
/// never leak its override into another test's lookup.
struct BucketOverrideGuard(&'static str);
impl BucketOverrideGuard {
fn set(bucket: &'static str, mode: DurabilityMode) -> Self {
bucket_durability::set(bucket, Some(mode));
Self(bucket)
}
}
impl Drop for BucketOverrideGuard {
fn drop(&mut self) {
bucket_durability::set(self.0, None);
}
}
#[test]
fn test_effective_durability_bucket_override() {
// Global strict + per-bucket relaxed: only the named bucket drops.
{
let _mode = durability_mode_override::set(DurabilityMode::Strict);
let _guard = BucketOverrideGuard::set("hp5b-override-relaxed", DurabilityMode::Relaxed);
assert_eq!(effective_durability("hp5b-override-relaxed"), DurabilityMode::Relaxed);
assert_eq!(effective_durability("hp5b-other-bucket"), DurabilityMode::Strict);
assert_eq!(effective_durability(RUSTFS_META_BUCKET), DurabilityMode::Strict);
}
// Override cleared: the bucket follows the global mode again (a new
// PUT after a config change resolves the new tier).
{
let _mode = durability_mode_override::set(DurabilityMode::Strict);
assert_eq!(effective_durability("hp5b-override-relaxed"), DurabilityMode::Strict);
}
// Global relaxed + per-bucket strict: overrides can raise durability.
{
let _mode = durability_mode_override::set(DurabilityMode::Relaxed);
let _guard = BucketOverrideGuard::set("hp5b-override-strict", DurabilityMode::Strict);
assert_eq!(effective_durability("hp5b-override-strict"), DurabilityMode::Strict);
assert_eq!(effective_durability("hp5b-other-bucket"), DurabilityMode::Relaxed);
}
}
#[test]
fn test_bucket_durability_refuses_system_namespaces() {
let _mode = durability_mode_override::set(DurabilityMode::Strict);
// System-critical and scratch namespaces can never carry an override.
bucket_durability::set(RUSTFS_META_BUCKET, Some(DurabilityMode::Relaxed));
bucket_durability::set(&format!("{RUSTFS_META_BUCKET}/buckets"), Some(DurabilityMode::None));
bucket_durability::set(RUSTFS_META_TMP_BUCKET, Some(DurabilityMode::Relaxed));
bucket_durability::set(super::super::RUSTFS_META_MULTIPART_BUCKET, Some(DurabilityMode::Relaxed));
bucket_durability::set("", Some(DurabilityMode::Relaxed));
assert_eq!(bucket_durability::lookup(RUSTFS_META_BUCKET), Option::None);
assert_eq!(bucket_durability::lookup(RUSTFS_META_TMP_BUCKET), Option::None);
assert_eq!(effective_durability(RUSTFS_META_BUCKET), DurabilityMode::Strict);
// The legacy full-off tier is process-wide only: registering it per
// bucket is dropped, not stored.
bucket_durability::set("hp5b-legacy-refused", Some(DurabilityMode::LegacyOff));
assert_eq!(bucket_durability::lookup("hp5b-legacy-refused"), Option::None);
}
#[test]
fn test_effective_durability_legacy_off_ignores_bucket_overrides() {
let _mode = durability_mode_override::set(DurabilityMode::LegacyOff);
let _guard = BucketOverrideGuard::set("hp5b-legacy-bucket", DurabilityMode::Strict);
// The legacy switch keeps its historical semantics bit for bit.
assert_eq!(effective_durability("hp5b-legacy-bucket"), DurabilityMode::LegacyOff);
}
/// HP-5b behavior regression: with the global mode at strict (the
/// default), a bucket override to relaxed must skip the metadata-commit
/// dir fsync for that bucket only, and clearing the override must restore
/// the strict behavior for the next write.
#[tokio::test]
#[allow(clippy::await_holding_lock)]
async fn test_write_all_meta_bucket_override_relaxed_then_cleared() {
use tempfile::tempdir;
let _mode = durability_mode_override::set(DurabilityMode::Strict);
let dir = tempdir().expect("temp dir should be created");
let endpoint = Endpoint::try_from(dir.path().to_str().expect("temp dir should be utf8")).expect("endpoint should parse");
let disk = LocalDisk::new(&endpoint, false).await.expect("local disk should be created");
let overridden = "hp5b-relaxed-write-bucket";
let untouched = "hp5b-strict-write-bucket";
ensure_test_volume(&disk, overridden).await;
ensure_test_volume(&disk, untouched).await;
ensure_test_volume(&disk, RUSTFS_META_TMP_BUCKET).await;
let meta_path = format!("dir/object/{STORAGE_FORMAT_FILE}");
{
let _guard = BucketOverrideGuard::set("hp5b-relaxed-write-bucket", DurabilityMode::Relaxed);
disk.write_all_meta(overridden, &meta_path, b"payload", true)
.await
.expect("write_all_meta should succeed");
let overridden_parent = disk
.get_object_path(overridden, &meta_path)
.expect("dst path should resolve")
.parent()
.expect("dst file should have a parent")
.to_path_buf();
assert!(
!os::fsync_dir_recorder::was_fsynced(&overridden_parent),
"bucket override to relaxed must skip the metadata-commit dir fsync"
);
// A bucket without an override keeps the strict default.
disk.write_all_meta(untouched, &meta_path, b"payload", true)
.await
.expect("write_all_meta should succeed");
let untouched_parent = disk
.get_object_path(untouched, &meta_path)
.expect("dst path should resolve")
.parent()
.expect("dst file should have a parent")
.to_path_buf();
assert!(
os::fsync_dir_recorder::was_fsynced(&untouched_parent),
"buckets without an override must keep the strict commit fsyncs"
);
}
// Override cleared (guard dropped): the next write is strict again.
let second_meta_path = format!("dir/object-after-clear/{STORAGE_FORMAT_FILE}");
disk.write_all_meta(overridden, &second_meta_path, b"payload", true)
.await
.expect("write_all_meta should succeed");
let after_clear_parent = disk
.get_object_path(overridden, &second_meta_path)
.expect("dst path should resolve")
.parent()
.expect("dst file should have a parent")
.to_path_buf();
assert!(
os::fsync_dir_recorder::was_fsynced(&after_clear_parent),
"clearing the override must restore strict fsyncs for new writes"
);
}
#[tokio::test]
#[allow(clippy::await_holding_lock)]
async fn test_write_all_meta_relaxed_skips_dst_parent_dir_fsync() {
use tempfile::tempdir;
let _mode = durability_mode_override::set(DurabilityMode::Relaxed);
let dir = tempdir().expect("temp dir should be created");
let endpoint = Endpoint::try_from(dir.path().to_str().expect("temp dir should be utf8")).expect("endpoint should parse");
let disk = LocalDisk::new(&endpoint, false).await.expect("local disk should be created");
let bucket = "relaxed-meta-bucket";
ensure_test_volume(&disk, bucket).await;
ensure_test_volume(&disk, RUSTFS_META_TMP_BUCKET).await;
let meta_path = format!("dir/object/{STORAGE_FORMAT_FILE}");
disk.write_all_meta(bucket, &meta_path, b"payload", true)
.await
.expect("write_all_meta should succeed");
let dst_file_path = disk.get_object_path(bucket, &meta_path).expect("dst path should resolve");
assert_eq!(
tokio::fs::read(&dst_file_path).await.expect("xl.meta should exist"),
b"payload",
"relaxed mode must not change what gets written, only what gets fsynced"
);
let dst_parent = dst_file_path.parent().expect("dst file should have a parent").to_path_buf();
assert!(
!os::fsync_dir_recorder::was_fsynced(&dst_parent),
"relaxed mode must skip the metadata-commit dir fsync on user volumes"
);
}
#[tokio::test]
#[allow(clippy::await_holding_lock)]
async fn test_write_all_public_relaxed_pins_system_volume() {
use tempfile::tempdir;
let _mode = durability_mode_override::set(DurabilityMode::Relaxed);
let dir = tempdir().expect("temp dir should be created");
let endpoint = Endpoint::try_from(dir.path().to_str().expect("temp dir should be utf8")).expect("endpoint should parse");
let disk = LocalDisk::new(&endpoint, false).await.expect("local disk should be created");
let user_bucket = "relaxed-public-bucket";
ensure_test_volume(&disk, user_bucket).await;
// User volume: the direct write follows the relaxed tier.
disk.write_all(user_bucket, "config/settings.json", Bytes::from_static(b"payload"))
.await
.expect("write_all should succeed");
let user_parent = disk
.get_object_path(user_bucket, "config/settings.json")
.expect("file path should resolve")
.parent()
.expect("file should have a parent")
.to_path_buf();
assert!(
!os::fsync_dir_recorder::was_fsynced(&user_parent),
"relaxed mode must skip the dir fsync for direct writes into user volumes"
);
// System-critical volume: pinned to strict regardless of the tier.
disk.write_all(RUSTFS_META_BUCKET, "buckets/test/bucket-metadata", Bytes::from_static(b"payload"))
.await
.expect("write_all should succeed");
let meta_parent = disk
.get_object_path(RUSTFS_META_BUCKET, "buckets/test/bucket-metadata")
.expect("meta path should resolve")
.parent()
.expect("meta file should have a parent")
.to_path_buf();
assert!(
os::fsync_dir_recorder::was_fsynced(&meta_parent),
"system-critical writes must stay fully synced under relaxed mode"
);
}
#[tokio::test]
#[allow(clippy::await_holding_lock)]
async fn test_rename_data_relaxed_keeps_shard_sync_skips_commit_fsyncs() {
use tempfile::tempdir;
let _mode = durability_mode_override::set(DurabilityMode::Relaxed);
let dir = tempdir().expect("temp dir should be created");
let endpoint = Endpoint::try_from(dir.path().to_str().expect("temp dir should be utf8")).expect("endpoint should parse");
let disk = LocalDisk::new(&endpoint, false).await.expect("local disk should be created");
let bucket = "relaxed-rename-bucket";
let object = "dir/object";
let tmp_object = "tmp-relaxed-write";
let version_id = Uuid::parse_str("77777777-7777-7777-7777-777777777777").expect("version id should parse");
let old_data_dir = Uuid::parse_str("88888888-8888-8888-8888-888888888888").expect("old data dir should parse");
let new_data_dir = Uuid::parse_str("99999999-9999-9999-9999-999999999999").expect("new data dir should parse");
ensure_test_volume(&disk, bucket).await;
ensure_test_volume(&disk, RUSTFS_META_TMP_BUCKET).await;
let old_fi = test_file_info(object, version_id, Some(old_data_dir), None);
let dst_object_dir = dir.path().join(bucket).join(object);
fs::create_dir_all(dst_object_dir.join(old_data_dir.to_string()))
.await
.expect("old data dir should be created");
fs::write(dst_object_dir.join(STORAGE_FORMAT_FILE), test_meta(old_fi))
.await
.expect("old metadata should be written");
let tmp_data_dir = dir
.path()
.join(RUSTFS_META_TMP_BUCKET)
.join(tmp_object)
.join(new_data_dir.to_string());
fs::create_dir_all(&tmp_data_dir)
.await
.expect("new tmp data dir should be created");
fs::write(tmp_data_dir.join("part.1"), b"new-data")
.await
.expect("new tmp data should be written");
let new_fi = test_file_info(object, version_id, Some(new_data_dir), None);
let resp = disk
.rename_data(RUSTFS_META_TMP_BUCKET, tmp_object, new_fi, bucket, object)
.await
.expect("rename_data should commit");
assert_eq!(resp.old_data_dir, Some(old_data_dir), "relaxed mode must not change commit semantics");
// Compare against root-resolved paths: the recorder stores the paths
// the disk actually fsyncs, which go through the canonicalized root.
let resolved_tmp_data_dir = disk
.get_object_path(RUSTFS_META_TMP_BUCKET, &format!("{tmp_object}/{new_data_dir}"))
.expect("tmp data dir path should resolve");
let resolved_dst_object_dir = disk.get_object_path(bucket, object).expect("dst object dir should resolve");
// Payload durability is kept: sync_dir_files fdatasyncs the shard
// files and fsyncs the staged data dir before the commit rename.
assert!(
os::fsync_dir_recorder::was_fsynced(&resolved_tmp_data_dir),
"relaxed mode must keep the shard-data sync before the commit rename"
);
// Metadata-commit fsyncs are skipped: neither the destination parent
// dir nor the rollback backup parent dir is fsynced.
assert!(
!os::fsync_dir_recorder::was_fsynced(&resolved_dst_object_dir),
"relaxed mode must skip the commit-rename dir fsync"
);
let resolved_backup_parent = resolved_dst_object_dir.join(old_data_dir.to_string());
assert!(
!os::fsync_dir_recorder::was_fsynced(&resolved_backup_parent),
"relaxed mode must skip the rollback-backup dir fsync"
);
assert!(
dst_object_dir
.join(old_data_dir.to_string())
.join(STORAGE_FORMAT_FILE_BACKUP)
.exists(),
"the rollback backup itself must still be written"
);
}
#[tokio::test]
#[allow(clippy::await_holding_lock)]
async fn test_rename_data_legacy_off_skips_all_fsyncs() {
use tempfile::tempdir;
let _mode = durability_mode_override::set(DurabilityMode::LegacyOff);
let dir = tempdir().expect("temp dir should be created");
let endpoint = Endpoint::try_from(dir.path().to_str().expect("temp dir should be utf8")).expect("endpoint should parse");
let disk = LocalDisk::new(&endpoint, false).await.expect("local disk should be created");
let bucket = "legacy-off-bucket";
let object = "dir/object";
let tmp_object = "tmp-legacy-write";
let version_id = Uuid::parse_str("aaaaaaaa-aaaa-aaaa-aaaa-aaaaaaaaaaaa").expect("version id should parse");
let new_data_dir = Uuid::parse_str("bbbbbbbb-bbbb-bbbb-bbbb-bbbbbbbbbbbb").expect("new data dir should parse");
ensure_test_volume(&disk, bucket).await;
ensure_test_volume(&disk, RUSTFS_META_TMP_BUCKET).await;
let tmp_data_dir = dir
.path()
.join(RUSTFS_META_TMP_BUCKET)
.join(tmp_object)
.join(new_data_dir.to_string());
fs::create_dir_all(&tmp_data_dir)
.await
.expect("new tmp data dir should be created");
fs::write(tmp_data_dir.join("part.1"), b"new-data")
.await
.expect("new tmp data should be written");
let new_fi = test_file_info(object, version_id, Some(new_data_dir), None);
disk.rename_data(RUSTFS_META_TMP_BUCKET, tmp_object, new_fi, bucket, object)
.await
.expect("rename_data should commit");
// Historical RUSTFS_DRIVE_SYNC_ENABLE=false semantics: no fsync at
// all, not even the shard-data sync relaxed keeps. Assertions use the
// root-resolved paths the disk actually passes to fsync.
let resolved_tmp_data_dir = disk
.get_object_path(RUSTFS_META_TMP_BUCKET, &format!("{tmp_object}/{new_data_dir}"))
.expect("tmp data dir path should resolve");
assert!(
!os::fsync_dir_recorder::was_fsynced(&resolved_tmp_data_dir),
"legacy-off must not sync staged shard data"
);
let resolved_dst_object_dir = disk.get_object_path(bucket, object).expect("dst object dir should resolve");
assert!(
!os::fsync_dir_recorder::was_fsynced(&resolved_dst_object_dir),
"legacy-off must not fsync the commit-rename dir"
);
// And system-critical volumes are NOT pinned: the old full-off
// behavior is preserved bit for bit for existing deployments.
disk.write_all(RUSTFS_META_BUCKET, "buckets/legacy/bucket-metadata", Bytes::from_static(b"payload"))
.await
.expect("write_all should succeed");
let meta_parent = disk
.get_object_path(RUSTFS_META_BUCKET, "buckets/legacy/bucket-metadata")
.expect("meta path should resolve")
.parent()
.expect("meta file should have a parent")
.to_path_buf();
assert!(
!os::fsync_dir_recorder::was_fsynced(&meta_parent),
"legacy-off keeps the historical semantics: system metadata is not synced either"
);
}
#[tokio::test]
async fn test_rename_data_writes_old_metadata_backup_for_inline_overwrite() {
use tempfile::tempdir;
let dir = tempdir().expect("temp dir should be created");
let endpoint = Endpoint::try_from(dir.path().to_str().expect("temp dir should be utf8")).expect("endpoint should parse");
let disk = LocalDisk::new(&endpoint, false).await.expect("local disk should be created");
let bucket = "bucket";
let object = "inline-object";
let tmp_object = "tmp-inline-write";
let version_id = Uuid::parse_str("12121212-1212-1212-1212-121212121212").expect("version id should parse");
let old_data_dir = Uuid::parse_str("34343434-3434-3434-3434-343434343434").expect("old data dir should parse");
ensure_test_volume(&disk, bucket).await;
ensure_test_volume(&disk, RUSTFS_META_TMP_BUCKET).await;
let old_fi = test_file_info(object, version_id, Some(old_data_dir), None);
let old_meta = test_meta(old_fi);
let dst_object_dir = dir.path().join(bucket).join(object);
fs::create_dir_all(dst_object_dir.join(old_data_dir.to_string()))
.await
.expect("old data dir should be created");
fs::write(dst_object_dir.join(STORAGE_FORMAT_FILE), &old_meta)
.await
.expect("old metadata should be written");
let tmp_object_dir = dir.path().join(RUSTFS_META_TMP_BUCKET).join(tmp_object);
fs::create_dir_all(&tmp_object_dir)
.await
.expect("tmp object dir should be created");
let new_fi = test_file_info(object, version_id, None, Some(Bytes::from_static(b"inline-new")));
let resp = disk
.rename_data(RUSTFS_META_TMP_BUCKET, tmp_object, new_fi, bucket, object)
.await
.expect("inline rename_data should commit");
assert_eq!(resp.old_data_dir, Some(old_data_dir));
assert_eq!(resp.sign, Some(version_id.as_bytes().to_vec()));
let backup_path = dst_object_dir.join(old_data_dir.to_string()).join(STORAGE_FORMAT_FILE_BACKUP);
assert!(backup_path.exists());
// The rollback backup must contain the previous metadata bytes verbatim so
// that undo_write can restore the prior committed object; guards the inline
// backup write against truncation/corruption regressions.
assert_eq!(
fs::read(&backup_path).await.expect("backup should be readable"),
old_meta,
"inline rollback backup must contain the previous metadata bytes verbatim"
);
assert!(
!dst_object_dir
.join(old_data_dir.to_string())
.join(STORAGE_FORMAT_FILE)
.exists()
);
}
#[tokio::test]
async fn test_delete_version_undo_restores_backup_to_object_root() {
use tempfile::tempdir;
let dir = tempdir().expect("temp dir should be created");
let endpoint = Endpoint::try_from(dir.path().to_str().expect("temp dir should be utf8")).expect("endpoint should parse");
let disk = LocalDisk::new(&endpoint, false).await.expect("local disk should be created");
let bucket = "bucket";
let object = "dir/object";
let version_id = Uuid::parse_str("44444444-4444-4444-4444-444444444444").expect("version id should parse");
let old_data_dir = Uuid::parse_str("55555555-5555-5555-5555-555555555555").expect("old data dir should parse");
let new_data_dir = Uuid::parse_str("66666666-6666-6666-6666-666666666666").expect("new data dir should parse");
ensure_test_volume(&disk, bucket).await;
let object_dir = dir.path().join(bucket).join("dir/object");
fs::create_dir_all(object_dir.join(old_data_dir.to_string()))
.await
.expect("old backup dir should be created");
fs::create_dir_all(object_dir.join(new_data_dir.to_string()))
.await
.expect("new data dir should be created");
let old_fi = test_file_info(object, version_id, Some(old_data_dir), None);
let old_meta = test_meta(old_fi);
let new_fi = test_file_info(object, version_id, Some(new_data_dir), None);
fs::write(
object_dir.join(old_data_dir.to_string()).join(STORAGE_FORMAT_FILE_BACKUP),
old_meta.clone(),
)
.await
.expect("old metadata backup should be written");
fs::write(object_dir.join(STORAGE_FORMAT_FILE), test_meta(new_fi.clone()))
.await
.expect("new metadata should be written");
disk.delete_version(
bucket,
object,
new_fi,
false,
DeleteOptions {
undo_write: true,
old_data_dir: Some(old_data_dir),
..Default::default()
},
)
.await
.expect("undo should restore old metadata");
let restored_meta = fs::read(object_dir.join(STORAGE_FORMAT_FILE))
.await
.expect("restored metadata should be readable");
assert_eq!(restored_meta, old_meta);
assert!(!object_dir.join("dir/object").join(STORAGE_FORMAT_FILE).exists());
}
#[tokio::test]
async fn test_delete_version_undo_restores_backup_when_other_versions_remain() {
use tempfile::tempdir;
let dir = tempdir().expect("temp dir should be created");
let endpoint = Endpoint::try_from(dir.path().to_str().expect("temp dir should be utf8")).expect("endpoint should parse");
let disk = LocalDisk::new(&endpoint, false).await.expect("local disk should be created");
let bucket = "bucket";
let object = "dir/object";
let version_id = Uuid::parse_str("44444444-4444-4444-4444-444444444444").expect("version id should parse");
let other_version_id = Uuid::parse_str("77777777-7777-7777-7777-777777777777").expect("version id should parse");
let old_data_dir = Uuid::parse_str("55555555-5555-5555-5555-555555555555").expect("old data dir should parse");
let new_data_dir = Uuid::parse_str("66666666-6666-6666-6666-666666666666").expect("new data dir should parse");
let other_data_dir = Uuid::parse_str("88888888-8888-8888-8888-888888888888").expect("other data dir should parse");
ensure_test_volume(&disk, bucket).await;
let object_dir = dir.path().join(bucket).join("dir/object");
fs::create_dir_all(object_dir.join(old_data_dir.to_string()))
.await
.expect("old backup dir should be created");
fs::create_dir_all(object_dir.join(new_data_dir.to_string()))
.await
.expect("new data dir should be created");
let old_fi = test_file_info(object, version_id, Some(old_data_dir), None);
let other_fi = test_file_info(object, other_version_id, Some(other_data_dir), None);
let mut old_meta = FileMeta::default();
old_meta
.add_version(old_fi)
.expect("old metadata should accept old file info");
old_meta
.add_version(other_fi.clone())
.expect("old metadata should accept other file info");
let old_meta = old_meta.marshal_msg().expect("old metadata should encode");
let new_fi = test_file_info(object, version_id, Some(new_data_dir), None);
let mut new_meta = FileMeta::default();
new_meta
.add_version(new_fi.clone())
.expect("new metadata should accept new file info");
new_meta
.add_version(other_fi)
.expect("new metadata should accept other file info");
fs::write(
object_dir.join(old_data_dir.to_string()).join(STORAGE_FORMAT_FILE_BACKUP),
old_meta.clone(),
)
.await
.expect("old metadata backup should be written");
fs::write(
object_dir.join(STORAGE_FORMAT_FILE),
new_meta.marshal_msg().expect("new metadata should encode"),
)
.await
.expect("new metadata should be written");
disk.delete_version(
bucket,
object,
new_fi,
false,
DeleteOptions {
undo_write: true,
old_data_dir: Some(old_data_dir),
..Default::default()
},
)
.await
.expect("undo should restore old metadata");
let restored_meta = fs::read(object_dir.join(STORAGE_FORMAT_FILE))
.await
.expect("restored metadata should be readable");
assert_eq!(restored_meta, old_meta);
}
#[tokio::test]
async fn test_rename_data_failure_before_metadata_commit_preserves_old_metadata() {
use tempfile::tempdir;
let dir = tempdir().expect("temp dir should be created");
let endpoint = Endpoint::try_from(dir.path().to_str().expect("temp dir should be utf8")).expect("endpoint should parse");
let disk = LocalDisk::new(&endpoint, false).await.expect("local disk should be created");
let bucket = "bucket";
let object = "failpoint-object";
let tmp_object = "tmp-object";
let version_id = Uuid::parse_str("aaaaaaaa-aaaa-aaaa-aaaa-aaaaaaaaaaaa").expect("version id should parse");
let old_data_dir = Uuid::parse_str("bbbbbbbb-bbbb-bbbb-bbbb-bbbbbbbbbbbb").expect("old data dir should parse");
let new_data_dir = Uuid::parse_str("cccccccc-cccc-cccc-cccc-cccccccccccc").expect("version id should parse");
ensure_test_volume(&disk, bucket).await;
ensure_test_volume(&disk, RUSTFS_META_TMP_BUCKET).await;
let old_fi = test_file_info(object, version_id, Some(old_data_dir), None);
let old_meta = test_meta(old_fi);
let object_dir = dir.path().join(bucket).join(object);
fs::create_dir_all(object_dir.join(old_data_dir.to_string()))
.await
.expect("old data dir should be created");
fs::write(object_dir.join(STORAGE_FORMAT_FILE), old_meta.clone())
.await
.expect("old metadata should be written");
let tmp_data_dir = dir
.path()
.join(RUSTFS_META_TMP_BUCKET)
.join(tmp_object)
.join(new_data_dir.to_string());
fs::create_dir_all(&tmp_data_dir)
.await
.expect("new tmp data dir should be created");
fs::write(tmp_data_dir.join("part.1"), b"new-data")
.await
.expect("new tmp data should be written");
set_rename_data_fail_before_old_metadata_backup(object);
let new_fi = test_file_info(object, version_id, Some(new_data_dir), None);
let result = disk
.rename_data(RUSTFS_META_TMP_BUCKET, tmp_object, new_fi, bucket, object)
.await;
assert!(result.is_err());
let current_meta = fs::read(object_dir.join(STORAGE_FORMAT_FILE))
.await
.expect("old metadata should still be readable");
assert_eq!(current_meta, old_meta);
assert!(!object_dir.join("object").join(STORAGE_FORMAT_FILE).exists());
}
#[tokio::test]
async fn test_skip_access_checks() {
// let arr = Vec::new();
let vols = [
RUSTFS_META_TMP_DELETED_BUCKET,
super::super::RUSTFS_META_TMP_BUCKET,
super::super::RUSTFS_META_MULTIPART_BUCKET,
RUSTFS_META_BUCKET,
];
let paths: Vec<_> = vols.iter().map(|v| path_join(&[Path::new(v), Path::new("test")])).collect();
for p in paths.iter() {
assert!(skip_access_checks(p.to_str().expect("operation should succeed")));
}
}
#[derive(Debug, Default)]
struct PendingTestReader;
impl AsyncRead for PendingTestReader {
fn poll_read(self: Pin<&mut Self>, _cx: &mut Context<'_>, _buf: &mut ReadBuf<'_>) -> Poll<io::Result<()>> {
Poll::Pending
}
}
#[tokio::test(start_paused = true)]
async fn local_read_timeout_reader_times_out_when_inner_stalls() {
let mut reader = StallTimeoutReader::new(PendingTestReader, Duration::from_secs(10));
let mut buf = [0; 1];
let err = reader
.read(&mut buf)
.await
.expect_err("stalled local reader should return a timeout error");
assert_eq!(err.kind(), ErrorKind::TimedOut);
}
#[tokio::test]
async fn test_get_disk_id_invalidates_cache_after_format_removal() {
use crate::disk::FORMAT_CONFIG_FILE;
use crate::disk::format::FormatV3;
use tempfile::tempdir;
let dir = tempdir().expect("operation should succeed");
let mut endpoint =
Endpoint::try_from(dir.path().to_str().expect("operation should succeed")).expect("operation should succeed");
endpoint.set_pool_index(0);
endpoint.set_set_index(0);
endpoint.set_disk_index(0);
let meta_dir = dir.path().join(RUSTFS_META_BUCKET);
fs::create_dir_all(&meta_dir).await.expect("meta dir should be creatable");
let mut format = FormatV3::new(1, 1);
format.erasure.this = format.erasure.sets[0][0];
let format_json = format.to_json().expect("format should serialize");
fs::write(meta_dir.join(FORMAT_CONFIG_FILE), format_json)
.await
.expect("format.json should be writable");
let disk = LocalDisk::new(&endpoint, false)
.await
.expect("local disk should open after seeding format");
let initial_id = disk.get_disk_id().await.expect("disk id lookup should succeed");
assert!(initial_id.is_some(), "new disk should expose a disk id");
fs::remove_file(&disk.format_path)
.await
.expect("format.json should be removable");
tokio::time::sleep(Duration::from_secs(2)).await;
let err = disk
.get_disk_id()
.await
.expect_err("removed format.json should invalidate the cached disk id");
assert!(matches!(err, DiskError::UnformattedDisk));
let format_info = disk.format_info.read().await.clone();
assert!(format_info.id.is_none(), "cached disk id should be cleared");
assert!(format_info.data.is_empty(), "cached format bytes should be cleared");
assert!(format_info.file_info.is_none(), "cached file metadata should be cleared");
assert!(format_info.last_check.is_none(), "cached format timestamp should be cleared");
}
#[tokio::test]
async fn cleanup_tmp_on_startup_moves_existing_tmp_and_recreates_trash() {
use tempfile::tempdir;
let dir = tempdir().expect("operation should succeed");
let tmp = LocalDisk::meta_path(dir.path(), RUSTFS_META_TMP_BUCKET);
let leftover = tmp.join("leftover").join("data");
fs::create_dir_all(leftover.parent().expect("operation should succeed"))
.await
.expect("operation should succeed");
fs::write(&leftover, b"temporary").await.expect("operation should succeed");
LocalDisk::cleanup_tmp_on_startup(dir.path(), Arc::new(AtomicU32::new(0)), Arc::new(Notify::new()))
.await
.expect("operation should succeed");
assert!(!tmp.join("leftover").exists());
assert!(LocalDisk::meta_path(dir.path(), RUSTFS_META_TMP_DELETED_BUCKET).exists());
}
#[tokio::test]
async fn cleanup_stale_tmp_objects_moves_expired_tmp_dirs_to_trash() {
use tempfile::tempdir;
let dir = tempdir().expect("operation should succeed");
let tmp = LocalDisk::meta_path(dir.path(), RUSTFS_META_TMP_BUCKET);
let stale = tmp.join("stale").join("data");
let trash = LocalDisk::meta_path(dir.path(), RUSTFS_META_TMP_DELETED_BUCKET);
fs::create_dir_all(stale.parent().expect("operation should succeed"))
.await
.expect("operation should succeed");
fs::create_dir_all(&trash).await.expect("operation should succeed");
fs::write(&stale, b"temporary").await.expect("operation should succeed");
tokio::time::sleep(Duration::from_millis(2)).await;
LocalDisk::cleanup_stale_tmp_objects_with_expiry(dir.path().to_path_buf(), Duration::ZERO)
.await
.expect("operation should succeed");
assert!(!tmp.join("stale").exists());
assert!(trash.exists());
let mut entries = fs::read_dir(&trash).await.expect("operation should succeed");
assert!(entries.next_entry().await.expect("operation should succeed").is_some());
}
#[tokio::test]
async fn cleanup_stale_tmp_objects_keeps_fresh_dirs_and_regular_files() {
use tempfile::tempdir;
let dir = tempdir().expect("operation should succeed");
let tmp = LocalDisk::meta_path(dir.path(), RUSTFS_META_TMP_BUCKET);
let fresh_dir = tmp.join("fresh").join("data");
let regular_file = tmp.join("note.txt");
let trash = LocalDisk::meta_path(dir.path(), RUSTFS_META_TMP_DELETED_BUCKET);
fs::create_dir_all(fresh_dir.parent().expect("operation should succeed"))
.await
.expect("operation should succeed");
fs::create_dir_all(&trash).await.expect("operation should succeed");
fs::write(&fresh_dir, b"temporary").await.expect("operation should succeed");
fs::write(&regular_file, b"keep").await.expect("operation should succeed");
LocalDisk::cleanup_stale_tmp_objects_with_expiry(dir.path().to_path_buf(), Duration::from_secs(60))
.await
.expect("operation should succeed");
assert!(tmp.join("fresh").exists());
assert!(regular_file.exists());
let mut entries = fs::read_dir(&trash).await.expect("operation should succeed");
assert!(entries.next_entry().await.expect("operation should succeed").is_none());
}
#[tokio::test(start_paused = true)]
async fn cleanup_loop_interval_does_not_tick_immediately() {
let start_at = tokio::time::Instant::now() + DELETED_OBJECTS_CLEANUP_INTERVAL;
let mut interval = interval_at(start_at, DELETED_OBJECTS_CLEANUP_INTERVAL);
assert!(tokio::time::timeout(Duration::from_secs(1), interval.tick()).await.is_err());
tokio::time::advance(DELETED_OBJECTS_CLEANUP_INTERVAL).await;
interval.tick().await;
}
#[tokio::test(start_paused = true)]
async fn startup_cleanup_barrier_waits_for_notification() {
let ready = Arc::new(AtomicU32::new(0));
let notify = Arc::new(Notify::new());
let wait = tokio::spawn({
let ready = ready.clone();
let notify = notify.clone();
async move { wait_for_startup_cleanup_signal(ready.as_ref(), notify.as_ref(), Duration::from_secs(2)).await }
});
tokio::task::yield_now().await;
assert!(!wait.is_finished());
ready.store(1, Ordering::Release);
notify.notify_waiters();
assert!(wait.await.expect("operation should succeed"));
}
#[tokio::test(start_paused = true)]
async fn startup_cleanup_barrier_times_out() {
let ready = Arc::new(AtomicU32::new(0));
let notify = Arc::new(Notify::new());
let wait = tokio::spawn({
let ready = ready.clone();
let notify = notify.clone();
async move { wait_for_startup_cleanup_signal(ready.as_ref(), notify.as_ref(), Duration::from_secs(2)).await }
});
tokio::task::yield_now().await;
tokio::time::advance(Duration::from_secs(2)).await;
assert!(!wait.await.expect("operation should succeed"));
}
#[test]
fn metacache_write_obj_classifies_closed_output_stream() {
struct BrokenPipeWriter;
impl AsyncWrite for BrokenPipeWriter {
fn poll_write(
self: std::pin::Pin<&mut Self>,
_cx: &mut std::task::Context<'_>,
_buf: &[u8],
) -> std::task::Poll<std::io::Result<usize>> {
std::task::Poll::Ready(Err(std::io::Error::new(std::io::ErrorKind::BrokenPipe, "closed")))
}
fn poll_flush(
self: std::pin::Pin<&mut Self>,
_cx: &mut std::task::Context<'_>,
) -> std::task::Poll<std::io::Result<()>> {
std::task::Poll::Ready(Ok(()))
}
fn poll_shutdown(
self: std::pin::Pin<&mut Self>,
_cx: &mut std::task::Context<'_>,
) -> std::task::Poll<std::io::Result<()>> {
std::task::Poll::Ready(Ok(()))
}
}
let mut writer = BrokenPipeWriter;
let mut out = MetacacheWriter::new(&mut writer);
let err = futures::executor::block_on(write_metacache_obj(
&mut out,
&MetaCacheEntry {
name: "object".to_string(),
..Default::default()
},
))
.expect_err("closed metacache output stream should fail");
assert!(err.is_metacache_output_stream_closed());
}
#[tokio::test]
async fn test_scan_dir_includes_nested_object_dirs() {
use rustfs_filemeta::MetacacheReader;
use tempfile::tempdir;
let dir = tempdir().expect("operation should succeed");
let bucket = "test-bucket";
let bucket_dir = dir.path().join(bucket);
fs::create_dir_all(bucket_dir.join("foo/bar/xyzzy"))
.await
.expect("operation should succeed");
fs::create_dir_all(bucket_dir.join("quux/thud"))
.await
.expect("operation should succeed");
fs::create_dir_all(bucket_dir.join("asdf"))
.await
.expect("operation should succeed");
fs::write(bucket_dir.join("foo/bar/xl.meta"), b"meta")
.await
.expect("operation should succeed");
fs::write(bucket_dir.join("foo/bar/xyzzy/xl.meta"), b"meta")
.await
.expect("operation should succeed");
fs::write(bucket_dir.join("quux/thud/xl.meta"), b"meta")
.await
.expect("operation should succeed");
fs::write(bucket_dir.join("asdf/xl.meta"), b"meta")
.await
.expect("operation should succeed");
let endpoint =
Endpoint::try_from(dir.path().to_str().expect("operation should succeed")).expect("operation should succeed");
let disk = LocalDisk::new(&endpoint, false).await.expect("operation should succeed");
let (reader, mut writer) = tokio::io::duplex(4096);
let mut out = MetacacheWriter::new(&mut writer);
let opts = WalkDirOptions {
bucket: bucket.to_string(),
base_dir: "".to_string(),
recursive: true,
..Default::default()
};
let mut objs_returned = 0;
disk.scan_dir("".to_string(), "".to_string(), &opts, &mut out, &mut objs_returned, false, None)
.await
.expect("operation should succeed");
out.close().await.expect("operation should succeed");
let mut reader = MetacacheReader::new(reader);
let entries = reader.read_all().await.expect("operation should succeed");
let names: Vec<String> = entries
.into_iter()
.filter(|entry| !entry.metadata.is_empty())
.map(|entry| entry.name)
.collect();
assert!(names.contains(&"asdf".to_string()));
assert!(names.contains(&"foo/bar".to_string()));
assert!(names.contains(&"foo/bar/xyzzy".to_string()));
assert!(names.contains(&"quux/thud".to_string()));
}
#[tokio::test]
async fn test_scan_dir_deduplicates_explicit_dir_marker_recursion() {
use rustfs_filemeta::MetacacheReader;
use tempfile::tempdir;
let dir = tempdir().expect("operation should succeed");
let bucket = "test-bucket";
let bucket_dir = dir.path().join(bucket);
fs::create_dir_all(bucket_dir.join("marker/file.txt"))
.await
.expect("operation should succeed");
fs::create_dir_all(bucket_dir.join("marker/subdir/file.txt"))
.await
.expect("operation should succeed");
fs::create_dir_all(bucket_dir.join(format!("marker/subdir{GLOBAL_DIR_SUFFIX}")))
.await
.expect("operation should succeed");
fs::write(bucket_dir.join("marker/file.txt/xl.meta"), b"meta")
.await
.expect("operation should succeed");
fs::write(bucket_dir.join("marker/subdir/file.txt/xl.meta"), b"meta")
.await
.expect("operation should succeed");
fs::write(bucket_dir.join(format!("marker/subdir{GLOBAL_DIR_SUFFIX}/xl.meta")), b"meta")
.await
.expect("operation should succeed");
let endpoint =
Endpoint::try_from(dir.path().to_str().expect("operation should succeed")).expect("operation should succeed");
let disk = LocalDisk::new(&endpoint, false).await.expect("operation should succeed");
let (reader, mut writer) = tokio::io::duplex(4096);
let mut out = MetacacheWriter::new(&mut writer);
let opts = WalkDirOptions {
bucket: bucket.to_string(),
base_dir: "marker/".to_string(),
recursive: true,
..Default::default()
};
let mut objs_returned = 0;
disk.scan_dir("marker/".to_string(), "".to_string(), &opts, &mut out, &mut objs_returned, false, None)
.await
.expect("operation should succeed");
out.close().await.expect("operation should succeed");
let mut reader = MetacacheReader::new(reader);
let entries = reader.read_all().await.expect("operation should succeed");
let names: Vec<String> = entries
.into_iter()
.filter(|entry| !entry.metadata.is_empty())
.map(|entry| entry.name)
.collect();
assert_eq!(names.iter().filter(|name| *name == "marker/subdir/file.txt").count(), 1);
assert_eq!(names.iter().filter(|name| *name == "marker/subdir/").count(), 1);
assert_eq!(names.iter().filter(|name| *name == "marker/file.txt").count(), 1);
}
#[tokio::test]
async fn test_scan_dir_forward_to_repeated_prefix_component() {
use rustfs_filemeta::MetacacheReader;
use tempfile::tempdir;
let dir = tempdir().expect("operation should succeed");
let bucket = "test-bucket";
let bucket_dir = dir.path().join(bucket);
for name in [
"different/prefix/prefix/repo-0000",
"different/prefix/prefix/repo-0001",
"different/prefix/prefix/repo-0002",
"engineering/alpha-0000",
"engineering/engineering/engineering/repo-0000",
"engineering/engineering/engineering/repo-0001",
"engineering/engineering/repo-0000",
"engineering/engineering/repo-0001",
"engineering/engineering/repo-0002",
"engineering/zulu-0000",
"unrelated/engineering/repo-0000",
] {
let object_dir = bucket_dir.join(name);
fs::create_dir_all(&object_dir).await.expect("operation should succeed");
fs::write(object_dir.join(STORAGE_FORMAT_FILE), b"meta")
.await
.expect("operation should succeed");
}
let endpoint =
Endpoint::try_from(dir.path().to_str().expect("operation should succeed")).expect("operation should succeed");
let disk = LocalDisk::new(&endpoint, false).await.expect("operation should succeed");
async fn scan_names(disk: &LocalDisk, bucket: &str, base_dir: &str, forward_to: &str) -> (Vec<String>, i32) {
let (reader, mut writer) = tokio::io::duplex(4096);
let mut out = MetacacheWriter::new(&mut writer);
let opts = WalkDirOptions {
bucket: bucket.to_string(),
base_dir: base_dir.to_string(),
recursive: true,
forward_to: Some(forward_to.to_string()),
..Default::default()
};
let mut objs_returned = 0;
disk.scan_dir(base_dir.to_string(), "".to_string(), &opts, &mut out, &mut objs_returned, false, None)
.await
.expect("operation should succeed");
out.close().await.expect("operation should succeed");
drop(out);
drop(writer);
let mut reader = MetacacheReader::new(reader);
let entries = reader.read_all().await.expect("operation should succeed");
let names: Vec<String> = entries
.into_iter()
.filter(|entry| !entry.metadata.is_empty())
.map(|entry| entry.name)
.collect();
(names, objs_returned)
}
let (engineering_names, engineering_count) =
scan_names(&disk, bucket, "engineering/", "engineering/engineering/engineering/repo-0001").await;
assert_eq!(
engineering_names,
vec![
"engineering/engineering/engineering/repo-0001".to_string(),
"engineering/engineering/repo-0000".to_string(),
"engineering/engineering/repo-0001".to_string(),
"engineering/engineering/repo-0002".to_string(),
"engineering/zulu-0000".to_string(),
],
"forward_to must resume at the requested triply repeated prefix and preserve lexicographic order"
);
assert_eq!(engineering_count as usize, engineering_names.len());
let (different_names, different_count) =
scan_names(&disk, bucket, "different/", "different/prefix/prefix/repo-0001").await;
assert_eq!(
different_names,
vec![
"different/prefix/prefix/repo-0001".to_string(),
"different/prefix/prefix/repo-0002".to_string(),
],
"forward_to must also work for repeated components unrelated to the engineering prefix"
);
assert_eq!(different_count as usize, different_names.len());
let (double_names, double_count) = scan_names(&disk, bucket, "engineering/", "engineering/engineering/repo-0001").await;
assert_eq!(
double_names,
vec![
"engineering/engineering/repo-0001".to_string(),
"engineering/engineering/repo-0002".to_string(),
"engineering/zulu-0000".to_string(),
],
"forward_to must not skip a child directory whose name repeats the base prefix"
);
assert_eq!(double_count as usize, double_names.len());
}
#[tokio::test]
async fn test_scan_dir_hidden_delete_markers_do_not_exhaust_limit() {
use rustfs_filemeta::MetacacheReader;
use tempfile::tempdir;
fn delete_marker_metadata(version_id: &str) -> Vec<u8> {
let mut fm = FileMeta::default();
fm.add_version(FileInfo {
deleted: true,
version_id: Some(Uuid::parse_str(version_id).expect("test version id should parse")),
mod_time: Some(OffsetDateTime::now_utc()),
..Default::default()
})
.expect("delete marker metadata should be valid");
fm.marshal_msg().expect("delete marker metadata should encode")
}
fn delete_marker_with_old_object_metadata(delete_version_id: &str, object_version_id: &str) -> Vec<u8> {
let mut fm = FileMeta::default();
fm.add_version({
let mut fi = FileInfo::new("hidden", 1, 1);
fi.version_id = Some(Uuid::parse_str(object_version_id).expect("test version id should parse"));
fi.mod_time = Some(OffsetDateTime::now_utc() - time::Duration::seconds(1));
fi
})
.expect("object metadata should be valid");
fm.add_version(FileInfo {
deleted: true,
version_id: Some(Uuid::parse_str(delete_version_id).expect("test version id should parse")),
mod_time: Some(OffsetDateTime::now_utc()),
..Default::default()
})
.expect("delete marker metadata should be valid");
fm.marshal_msg().expect("delete marker metadata should encode")
}
fn object_metadata(version_id: &str) -> Vec<u8> {
let mut fm = FileMeta::default();
let mut fi = FileInfo::new("visible", 1, 1);
fi.version_id = Some(Uuid::parse_str(version_id).expect("test version id should parse"));
fi.mod_time = Some(OffsetDateTime::now_utc());
fm.add_version(fi).expect("object metadata should be valid");
fm.marshal_msg().expect("object metadata should encode")
}
let dir = tempdir().expect("operation should succeed");
let bucket = "test-bucket";
let bucket_dir = dir.path().join(bucket);
for (name, version_id) in [
("shard/aaa-trash-0000", "11111111-1111-1111-1111-111111111111"),
("shard/aaa-trash-0001", "22222222-2222-2222-2222-222222222222"),
("shard/aaa-trash-0002", "33333333-3333-3333-3333-333333333333"),
] {
let object_dir = bucket_dir.join(name);
fs::create_dir_all(&object_dir).await.expect("operation should succeed");
fs::write(object_dir.join(STORAGE_FORMAT_FILE), delete_marker_metadata(version_id))
.await
.expect("operation should succeed");
}
let hidden_versioned_dir = bucket_dir.join("shard/aaa-trash-0003");
fs::create_dir_all(&hidden_versioned_dir)
.await
.expect("operation should succeed");
fs::write(
hidden_versioned_dir.join(STORAGE_FORMAT_FILE),
delete_marker_with_old_object_metadata(
"44444444-4444-4444-4444-444444444444",
"55555555-5555-5555-5555-555555555555",
),
)
.await
.expect("operation should succeed");
let visible_dir = bucket_dir.join("shard/bbb-visible-0000");
fs::create_dir_all(&visible_dir).await.expect("operation should succeed");
fs::write(
visible_dir.join(STORAGE_FORMAT_FILE),
object_metadata("66666666-6666-6666-6666-666666666666"),
)
.await
.expect("operation should succeed");
let endpoint =
Endpoint::try_from(dir.path().to_str().expect("operation should succeed")).expect("operation should succeed");
let disk = LocalDisk::new(&endpoint, false).await.expect("operation should succeed");
let (reader, mut writer) = tokio::io::duplex(4096);
let mut out = MetacacheWriter::new(&mut writer);
let opts = WalkDirOptions {
bucket: bucket.to_string(),
base_dir: "".to_string(),
recursive: true,
limit: 1,
..Default::default()
};
let mut objs_returned = 0;
disk.scan_dir("".to_string(), "".to_string(), &opts, &mut out, &mut objs_returned, false, None)
.await
.expect("operation should succeed");
out.close().await.expect("operation should succeed");
drop(out);
drop(writer);
let mut reader = MetacacheReader::new(reader);
let has_visible_object = reader
.read_all()
.await
.expect("operation should succeed")
.into_iter()
.any(|entry| !entry.metadata.is_empty() && entry.name == "shard/bbb-visible-0000");
assert!(has_visible_object);
assert_eq!(objs_returned, 1);
}
#[tokio::test]
async fn test_scan_dir_nonrecursive_skips_dirs_with_only_hidden_delete_markers() {
use rustfs_filemeta::MetacacheReader;
use tempfile::tempdir;
fn hidden_versioned_object_metadata(name: &str, delete_version_id: &str, object_version_id: &str) -> Vec<u8> {
let mut fm = FileMeta::default();
fm.add_version({
let mut fi = FileInfo::new(name, 1, 1);
fi.version_id = Some(Uuid::parse_str(object_version_id).expect("test version id should parse"));
fi.mod_time = Some(OffsetDateTime::now_utc() - time::Duration::seconds(1));
fi
})
.expect("object metadata should be valid");
fm.add_version(FileInfo {
name: name.to_owned(),
deleted: true,
version_id: Some(Uuid::parse_str(delete_version_id).expect("test version id should parse")),
mod_time: Some(OffsetDateTime::now_utc()),
..Default::default()
})
.expect("delete marker metadata should be valid");
fm.marshal_msg().expect("hidden metadata should encode")
}
fn visible_object_metadata(name: &str, version_id: &str) -> Vec<u8> {
let mut fm = FileMeta::default();
let mut fi = FileInfo::new(name, 1, 1);
fi.version_id = Some(Uuid::parse_str(version_id).expect("test version id should parse"));
fi.mod_time = Some(OffsetDateTime::now_utc());
fm.add_version(fi).expect("object metadata should be valid");
fm.marshal_msg().expect("visible metadata should encode")
}
async fn scan_names(disk: &LocalDisk, bucket: &str, base_dir: &str, incl_deleted: bool) -> Vec<String> {
let (reader, mut writer) = tokio::io::duplex(4096);
let mut out = MetacacheWriter::new(&mut writer);
let opts = WalkDirOptions {
bucket: bucket.to_string(),
base_dir: base_dir.to_string(),
recursive: false,
incl_deleted,
..Default::default()
};
let mut objs_returned = 0;
disk.scan_dir(base_dir.to_string(), "".to_string(), &opts, &mut out, &mut objs_returned, false, None)
.await
.expect("scan_dir should succeed");
out.close().await.expect("metacache writer should close");
drop(out);
drop(writer);
let mut reader = MetacacheReader::new(reader);
reader
.read_all()
.await
.expect("scan output should decode")
.into_iter()
.map(|entry| entry.name)
.collect()
}
let dir = tempdir().expect("tempdir should be created");
let bucket = "test-bucket";
let bucket_dir = dir.path().join(bucket);
let hidden_object = bucket_dir.join("hidden/deleted.txt");
fs::create_dir_all(&hidden_object)
.await
.expect("hidden object dir should be created");
fs::write(
hidden_object.join(STORAGE_FORMAT_FILE),
hidden_versioned_object_metadata(
"hidden/deleted.txt",
"11111111-1111-1111-1111-111111111111",
"22222222-2222-2222-2222-222222222222",
),
)
.await
.expect("hidden object metadata should be written");
let nested_hidden_object = bucket_dir.join("hidden/nested/deleted.txt");
fs::create_dir_all(&nested_hidden_object)
.await
.expect("nested hidden object dir should be created");
fs::write(
nested_hidden_object.join(STORAGE_FORMAT_FILE),
hidden_versioned_object_metadata(
"hidden/nested/deleted.txt",
"33333333-3333-3333-3333-333333333333",
"44444444-4444-4444-4444-444444444444",
),
)
.await
.expect("nested hidden object metadata should be written");
let visible_object = bucket_dir.join("visible/nested/object.txt");
fs::create_dir_all(&visible_object)
.await
.expect("visible object dir should be created");
fs::write(
visible_object.join(STORAGE_FORMAT_FILE),
visible_object_metadata("visible/nested/object.txt", "55555555-5555-5555-5555-555555555555"),
)
.await
.expect("visible object metadata should be written");
let endpoint =
Endpoint::try_from(dir.path().to_str().expect("tempdir path should be utf8")).expect("endpoint should parse");
let disk = LocalDisk::new(&endpoint, false).await.expect("local disk should be created");
let root_names = scan_names(&disk, bucket, "", false).await;
assert!(!root_names.contains(&"hidden/".to_string()));
assert!(root_names.contains(&"visible/".to_string()));
let hidden_names = scan_names(&disk, bucket, "hidden/", false).await;
assert!(!hidden_names.contains(&"hidden/nested/".to_string()));
let visible_names = scan_names(&disk, bucket, "visible/", false).await;
assert!(visible_names.contains(&"visible/nested/".to_string()));
let versioned_root_names = scan_names(&disk, bucket, "", true).await;
assert!(versioned_root_names.contains(&"hidden/".to_string()));
let versioned_hidden_names = scan_names(&disk, bucket, "hidden/", true).await;
assert!(versioned_hidden_names.contains(&"hidden/nested/".to_string()));
}
#[cfg(unix)]
#[tokio::test]
async fn test_scan_dir_propagates_metadata_read_errors() {
use std::fs::Permissions;
use std::os::unix::fs::PermissionsExt;
use tempfile::tempdir;
let dir = tempdir().expect("operation should succeed");
let bucket = "test-bucket";
let bucket_dir = dir.path().join(bucket);
let object_dir = bucket_dir.join("broken");
let meta_path = object_dir.join(STORAGE_FORMAT_FILE);
fs::create_dir_all(&object_dir).await.expect("operation should succeed");
fs::write(&meta_path, b"meta").await.expect("operation should succeed");
let original_permissions = fs::metadata(&meta_path)
.await
.expect("operation should succeed")
.permissions();
fs::set_permissions(&meta_path, Permissions::from_mode(0o000))
.await
.expect("operation should succeed");
if fs::File::open(&meta_path).await.is_ok() {
fs::set_permissions(&meta_path, original_permissions)
.await
.expect("operation should succeed");
return;
}
let endpoint =
Endpoint::try_from(dir.path().to_str().expect("operation should succeed")).expect("operation should succeed");
let disk = LocalDisk::new(&endpoint, false).await.expect("operation should succeed");
let (_reader, mut writer) = tokio::io::duplex(4096);
let mut out = MetacacheWriter::new(&mut writer);
let opts = WalkDirOptions {
bucket: bucket.to_string(),
base_dir: "".to_string(),
recursive: true,
..Default::default()
};
let mut objs_returned = 0;
let result = disk
.scan_dir("".to_string(), "".to_string(), &opts, &mut out, &mut objs_returned, false, None)
.await;
fs::set_permissions(&meta_path, original_permissions)
.await
.expect("operation should succeed");
assert!(matches!(result, Err(DiskError::FileAccessDenied)));
}
#[tokio::test]
async fn test_walk_dir_ignore_multipart_dirs() {
use rustfs_filemeta::MetacacheReader;
use tempfile::tempdir;
const UUID_MULTIPART_1: &str = "8b262d24-fcf9-473d-a4cd-f9b27f24f60e";
const UUID_MULTIPART_2: &str = "fbf3183c-63be-45cc-b3bf-424ddb7f95f8";
const UUID_OBJ: &str = "db8b9b74-9016-4f9e-83e9-82a772947d28";
const VER_ID_1: &str = "c683f9f8-c0a1-4bc5-8a67-0faafa839a1a";
const VER_ID_2: &str = "a4b84f6e-c8ba-461b-8f9d-43feb0893efb";
const VER_ID_3: &str = "892c9ae7-2bb3-44ee-9a71-bc7ddf08d765";
const BASE_DIR: &str = "dir1/obj/";
const MULTIPART_DIR: &str = "multipart-file";
const DIR_IN_MULTIPART_DIR: &str = "dir-in-multipart";
const EMPTY_STR: &str = "";
let parse_uuid = |s: &str| Uuid::parse_str(s).expect("operation should succeed");
let create_file_info = |version_id: &str, data_dir: &str| FileInfo {
version_id: Some(parse_uuid(version_id)),
data_dir: Some(parse_uuid(data_dir)),
mod_time: Some(OffsetDateTime::now_utc()),
..Default::default()
};
let dir = tempdir().expect("operation should succeed");
let obj_base = dir.path().join("test-bucket").join(BASE_DIR);
let multipart_base = obj_base.join(MULTIPART_DIR);
let dir_in_multipart_base = multipart_base.join(DIR_IN_MULTIPART_DIR);
fs::create_dir_all(&multipart_base).await.expect("operation should succeed");
for uuid in &[UUID_MULTIPART_1, UUID_MULTIPART_2] {
fs::create_dir_all(multipart_base.join(uuid))
.await
.expect("operation should succeed");
fs::write(multipart_base.join(uuid).join("part.1"), b"part")
.await
.expect("operation should succeed");
}
fs::create_dir_all(obj_base.join(UUID_OBJ))
.await
.expect("operation should succeed");
fs::write(obj_base.join(UUID_OBJ).join("part.1"), b"part")
.await
.expect("operation should succeed");
fs::create_dir_all(&dir_in_multipart_base)
.await
.expect("operation should succeed");
fs::write(dir_in_multipart_base.join(STORAGE_FORMAT_FILE), b"meta")
.await
.expect("operation should succeed");
let mut fm = FileMeta::default();
fm.add_version(create_file_info(VER_ID_1, UUID_MULTIPART_1))
.expect("operation should succeed");
fm.add_version(create_file_info(VER_ID_2, UUID_MULTIPART_2))
.expect("operation should succeed");
fs::write(
multipart_base.join(STORAGE_FORMAT_FILE),
fm.marshal_msg().expect("operation should succeed"),
)
.await
.expect("operation should succeed");
let mut fm = FileMeta::default();
fm.add_version(create_file_info(VER_ID_3, UUID_OBJ))
.expect("operation should succeed");
fs::write(obj_base.join(STORAGE_FORMAT_FILE), fm.marshal_msg().expect("operation should succeed"))
.await
.expect("operation should succeed");
let endpoint =
Endpoint::try_from(dir.path().to_str().expect("operation should succeed")).expect("operation should succeed");
let disk = LocalDisk::new(&endpoint, false).await.expect("operation should succeed");
let (reader, mut writer) = tokio::io::duplex(4096);
disk.walk_dir(
WalkDirOptions {
bucket: "test-bucket".to_string(),
base_dir: BASE_DIR.to_string(),
recursive: true,
filter_prefix: Some(EMPTY_STR.to_string()),
..Default::default()
},
&mut writer,
)
.await
.expect("operation should succeed");
MetacacheWriter::new(&mut writer)
.close()
.await
.expect("operation should succeed");
let mut reader = MetacacheReader::new(reader);
let entries = reader.read_all().await.expect("operation should succeed");
let names: Vec<String> = entries.into_iter().map(|entry| entry.name).collect();
assert_eq!(
names
.iter()
.filter(|name| *name == &format!("{}{}", BASE_DIR, MULTIPART_DIR))
.count(),
1
);
assert_eq!(
names
.iter()
.filter(|name| *name == &format!("{}{}/", BASE_DIR, MULTIPART_DIR))
.count(),
1
);
assert_eq!(
names
.iter()
.filter(|name| *name == &format!("{}{}/{}", BASE_DIR, MULTIPART_DIR, DIR_IN_MULTIPART_DIR))
.count(),
1
);
assert_eq!(
names
.iter()
.filter(|name| *name == &format!("{}{}/{}/", BASE_DIR, MULTIPART_DIR, DIR_IN_MULTIPART_DIR))
.count(),
1
);
assert_eq!(
names
.iter()
.filter(|name| *name == &format!("{}{}/{}", BASE_DIR, MULTIPART_DIR, UUID_MULTIPART_1))
.count(),
0
);
assert_eq!(
names
.iter()
.filter(|name| *name == &format!("{}{}/{}", BASE_DIR, MULTIPART_DIR, UUID_MULTIPART_2))
.count(),
0
);
assert_eq!(
names
.iter()
.filter(|name| *name == &format!("{}{}", BASE_DIR, UUID_OBJ))
.count(),
0
);
assert_eq!(
names
.iter()
.filter(|name| *name == &format!("{}{}/{}/", BASE_DIR, MULTIPART_DIR, UUID_MULTIPART_1))
.count(),
0
);
assert_eq!(
names
.iter()
.filter(|name| *name == &format!("{}{}/{}/", BASE_DIR, MULTIPART_DIR, UUID_MULTIPART_2))
.count(),
0
);
assert_eq!(
names
.iter()
.filter(|name| *name == &format!("{}{}/", BASE_DIR, UUID_OBJ))
.count(),
0
);
}
#[tokio::test]
async fn test_make_volume() {
let p = "./testv0";
fs::create_dir_all(&p).await.expect("operation should succeed");
let ep = match Endpoint::try_from(p) {
Ok(e) => e,
Err(e) => {
println!("{e}");
return;
}
};
let disk = LocalDisk::new(&ep, false).await.expect("operation should succeed");
let tmpp = disk
.resolve_abs_path(Path::new(RUSTFS_META_TMP_DELETED_BUCKET))
.expect("operation should succeed");
println!("ppp :{:?}", &tmpp);
let volumes = vec!["a123", "b123", "c123"];
disk.make_volumes(volumes.clone()).await.expect("operation should succeed");
disk.make_volumes(volumes.clone()).await.expect("operation should succeed");
let _ = fs::remove_dir_all(&p).await;
}
#[tokio::test]
async fn test_delete_volume() {
let p = "./testv1";
fs::create_dir_all(&p).await.expect("operation should succeed");
let ep = match Endpoint::try_from(p) {
Ok(e) => e,
Err(e) => {
println!("{e}");
return;
}
};
let disk = LocalDisk::new(&ep, false).await.expect("operation should succeed");
let tmpp = disk
.resolve_abs_path(Path::new(RUSTFS_META_TMP_DELETED_BUCKET))
.expect("operation should succeed");
println!("ppp :{:?}", &tmpp);
let volumes = vec!["a123", "b123", "c123"];
disk.make_volumes(volumes.clone()).await.expect("operation should succeed");
disk.delete_volume("a", true).await.expect("operation should succeed");
let _ = fs::remove_dir_all(&p).await;
}
#[tokio::test]
async fn test_local_disk_basic_operations() {
let test_dir = "./test_local_disk_basic";
fs::create_dir_all(&test_dir).await.expect("operation should succeed");
let endpoint = Endpoint::try_from(test_dir).expect("operation should succeed");
let disk = LocalDisk::new(&endpoint, false).await.expect("operation should succeed");
// Test basic properties
assert!(disk.is_local());
// Note: host_name() for local disks might be empty or contain localhost/hostname
// assert!(!disk.host_name().is_empty());
assert!(!disk.to_string().is_empty());
// Test path resolution
let abs_path = disk.resolve_abs_path("test/path").expect("operation should succeed");
assert!(abs_path.is_absolute());
// Test bucket path
let bucket_path = disk.get_bucket_path("test-bucket").expect("operation should succeed");
assert!(bucket_path.to_string_lossy().contains("test-bucket"));
// Test object path
let object_path = disk
.get_object_path("test-bucket", "test-object")
.expect("operation should succeed");
assert!(object_path.to_string_lossy().contains("test-bucket"));
assert!(object_path.to_string_lossy().contains("test-object"));
// Clean up the test directory
let _ = fs::remove_dir_all(&test_dir).await;
}
#[cfg(unix)]
#[tokio::test]
async fn test_get_bucket_path_rejects_symlink_escape() {
use std::os::unix::fs::symlink;
use tempfile::tempdir;
let root_dir = tempdir().expect("operation should succeed");
let outside_dir = tempdir().expect("operation should succeed");
let link_path = root_dir.path().join("escape-bucket");
symlink(outside_dir.path(), &link_path).expect("operation should succeed");
let endpoint = Endpoint::try_from(root_dir.path().to_string_lossy().as_ref()).expect("operation should succeed");
let disk = LocalDisk::new(&endpoint, false).await.expect("operation should succeed");
assert!(matches!(disk.get_bucket_path("escape-bucket"), Err(DiskError::InvalidPath)));
}
#[cfg(unix)]
#[tokio::test]
async fn test_get_object_path_rejects_symlink_component_escape() {
use std::os::unix::fs::symlink;
use tempfile::tempdir;
let root_dir = tempdir().expect("operation should succeed");
let outside_dir = tempdir().expect("operation should succeed");
let bucket_dir = root_dir.path().join("bucket");
fs::create_dir_all(&bucket_dir).await.expect("operation should succeed");
let link_path = bucket_dir.join("escape");
symlink(outside_dir.path(), &link_path).expect("operation should succeed");
let endpoint = Endpoint::try_from(root_dir.path().to_string_lossy().as_ref()).expect("operation should succeed");
let disk = LocalDisk::new(&endpoint, false).await.expect("operation should succeed");
assert!(matches!(disk.get_object_path("bucket", "escape/object.txt"), Err(DiskError::InvalidPath)));
}
#[tokio::test]
async fn test_local_disk_file_operations() {
let test_dir = "./test_local_disk_file_ops";
fs::create_dir_all(&test_dir).await.expect("operation should succeed");
let endpoint = Endpoint::try_from(test_dir).expect("operation should succeed");
let disk = LocalDisk::new(&endpoint, false).await.expect("operation should succeed");
// Create test volume
disk.make_volume("test-volume").await.expect("operation should succeed");
// Test write and read operations
let test_data: Vec<u8> = vec![1, 2, 3, 4, 5];
disk.write_all("test-volume", "test-file.txt", test_data.clone().into())
.await
.expect("operation should succeed");
let read_data = disk
.read_all("test-volume", "test-file.txt")
.await
.expect("operation should succeed");
assert_eq!(read_data, test_data);
// Test file deletion
let delete_opts = DeleteOptions {
recursive: false,
immediate: true,
undo_write: false,
old_data_dir: None,
};
disk.delete("test-volume", "test-file.txt", delete_opts)
.await
.expect("operation should succeed");
// Clean up
disk.delete_volume("test-volume", true)
.await
.expect("operation should succeed");
let _ = fs::remove_dir_all(&test_dir).await;
}
#[tokio::test]
async fn delete_volume_non_force_refuses_non_empty_bucket() {
// backlog#799 B1: a non-force delete_volume must refuse a bucket that
// still holds object data (VolumeNotEmpty) and leave it intact, so a
// misclassified "dangling" bucket cannot be recursively wiped. Only an
// explicit force delete removes it recursively.
let test_dir = "./test_b1_delete_volume_guard";
let _ = fs::remove_dir_all(&test_dir).await;
fs::create_dir_all(&test_dir).await.expect("operation should succeed");
let endpoint = Endpoint::try_from(test_dir).expect("operation should succeed");
let disk = LocalDisk::new(&endpoint, false).await.expect("operation should succeed");
disk.make_volume("b1-bucket").await.expect("operation should succeed");
let data: Vec<u8> = vec![1, 2, 3];
disk.write_all("b1-bucket", "obj.dat", data.clone().into())
.await
.expect("operation should succeed");
// Non-force must refuse and preserve the data.
let err = disk
.delete_volume("b1-bucket", false)
.await
.expect_err("non-empty bucket must be refused");
assert!(matches!(err, DiskError::VolumeNotEmpty), "expected VolumeNotEmpty, got {err:?}");
assert!(
disk.stat_volume("b1-bucket").await.is_ok(),
"bucket must still exist after a refused non-force delete"
);
assert_eq!(disk.read_all("b1-bucket", "obj.dat").await.expect("data preserved"), data);
// Force removes it recursively.
disk.delete_volume("b1-bucket", true)
.await
.expect("force delete removes non-empty");
assert!(disk.stat_volume("b1-bucket").await.is_err(), "bucket must be gone after force delete");
let _ = fs::remove_dir_all(&test_dir).await;
}
#[tokio::test]
async fn test_local_disk_volume_operations() {
let test_dir = "./test_local_disk_volumes";
fs::create_dir_all(&test_dir).await.expect("operation should succeed");
let endpoint = Endpoint::try_from(test_dir).expect("operation should succeed");
let disk = LocalDisk::new(&endpoint, false).await.expect("operation should succeed");
// Test creating multiple volumes
let volumes = vec!["vol1", "vol2", "vol3"];
disk.make_volumes(volumes.clone()).await.expect("operation should succeed");
// Test listing volumes
let volume_list = disk.list_volumes().await.expect("operation should succeed");
assert!(!volume_list.is_empty());
// Test volume stats
for vol in &volumes {
let vol_info = disk.stat_volume(vol).await.expect("operation should succeed");
assert_eq!(vol_info.name, *vol);
}
// Test deleting volumes
for vol in &volumes {
disk.delete_volume(vol, true).await.expect("operation should succeed");
}
// Clean up the test directory
let _ = fs::remove_dir_all(&test_dir).await;
}
#[tokio::test]
async fn test_local_disk_disk_info() {
let test_dir = "./test_local_disk_info";
fs::create_dir_all(&test_dir).await.expect("operation should succeed");
let endpoint = Endpoint::try_from(test_dir).expect("operation should succeed");
let disk = LocalDisk::new(&endpoint, false).await.expect("operation should succeed");
let disk_info_opts = DiskInfoOptions {
disk_id: "test-disk".to_string(),
metrics: true,
noop: false,
};
let disk_info = disk.disk_info(&disk_info_opts).await.expect("operation should succeed");
// Basic checks on disk info
// Note: On macOS, Windows, and some other systems, fs_type may be empty
// because statvfs does not provide filesystem type information.
// This is a platform limitation, not a bug.
#[cfg(not(any(target_os = "macos", windows)))]
assert!(!disk_info.fs_type.is_empty(), "fs_type should not be empty on this platform");
assert!(disk_info.total > 0);
assert!(disk_info.free <= disk_info.total);
assert_eq!(disk_info.nr_requests, disk.nrrequests);
assert_eq!(disk_info.rotational, disk.rotational);
assert!(!disk_info.mount_path.is_empty());
assert!(!disk_info.endpoint.is_empty());
// Clean up the test directory
let _ = fs::remove_dir_all(&test_dir).await;
}
#[tokio::test]
async fn test_read_file_stream_rejects_offset_length_overflow() {
use tempfile::tempdir;
let dir = tempdir().expect("operation should succeed");
let endpoint =
Endpoint::try_from(dir.path().to_str().expect("operation should succeed")).expect("operation should succeed");
let disk = LocalDisk::new(&endpoint, false).await.expect("operation should succeed");
disk.make_volume("test-volume").await.expect("operation should succeed");
disk.write_all("test-volume", "test-file.txt", Bytes::from_static(b"test"))
.await
.expect("operation should succeed");
let result = disk.read_file_stream("test-volume", "test-file.txt", usize::MAX, 1).await;
assert!(matches!(result, Err(DiskError::FileCorrupt)));
}
#[tokio::test]
async fn test_read_file_mmap_copy_rejects_offset_length_overflow() {
use tempfile::tempdir;
let dir = tempdir().expect("operation should succeed");
let endpoint =
Endpoint::try_from(dir.path().to_str().expect("operation should succeed")).expect("operation should succeed");
let disk = LocalDisk::new(&endpoint, false).await.expect("operation should succeed");
disk.make_volume("test-volume").await.expect("operation should succeed");
disk.write_all("test-volume", "test-file.txt", Bytes::from_static(b"test"))
.await
.expect("operation should succeed");
let result = disk.read_file_mmap_copy("test-volume", "test-file.txt", usize::MAX, 1).await;
assert!(matches!(result, Err(DiskError::FileCorrupt)));
}
#[tokio::test]
#[allow(deprecated)]
async fn test_read_file_zero_copy_legacy_alias_rejects_offset_length_overflow() {
use tempfile::tempdir;
let dir = tempdir().expect("operation should succeed");
let endpoint =
Endpoint::try_from(dir.path().to_str().expect("operation should succeed")).expect("operation should succeed");
let disk = LocalDisk::new(&endpoint, false).await.expect("operation should succeed");
disk.make_volume("test-volume").await.expect("operation should succeed");
disk.write_all("test-volume", "test-file.txt", Bytes::from_static(b"test"))
.await
.expect("operation should succeed");
let result = disk.read_file_zero_copy("test-volume", "test-file.txt", usize::MAX, 1).await;
assert!(matches!(result, Err(DiskError::FileCorrupt)));
}
#[test]
fn test_is_valid_volname() {
// Valid volume names (length >= 3)
assert!(LocalDisk::is_valid_volname("valid-name"));
assert!(LocalDisk::is_valid_volname("test123"));
assert!(LocalDisk::is_valid_volname("my-bucket"));
// Test minimum length requirement
assert!(!LocalDisk::is_valid_volname(""));
assert!(!LocalDisk::is_valid_volname("a"));
assert!(!LocalDisk::is_valid_volname("ab"));
assert!(LocalDisk::is_valid_volname("abc"));
// Note: The current implementation doesn't check for system volume names
// It only checks length and platform-specific special characters
// System volume names are valid according to the current implementation
assert!(LocalDisk::is_valid_volname(RUSTFS_META_BUCKET));
assert!(LocalDisk::is_valid_volname(super::super::RUSTFS_META_TMP_BUCKET));
// Testing platform-specific behavior for special characters
#[cfg(windows)]
{
// On Windows systems, these should be invalid
assert!(!LocalDisk::is_valid_volname("invalid\\name"));
assert!(!LocalDisk::is_valid_volname("invalid:name"));
assert!(!LocalDisk::is_valid_volname("invalid|name"));
assert!(!LocalDisk::is_valid_volname("invalid<name"));
assert!(!LocalDisk::is_valid_volname("invalid>name"));
assert!(!LocalDisk::is_valid_volname("invalid?name"));
assert!(!LocalDisk::is_valid_volname("invalid*name"));
assert!(!LocalDisk::is_valid_volname("invalid\"name"));
}
#[cfg(not(windows))]
{
// On non-Windows systems, the current implementation doesn't check special characters
// So these would be considered valid
assert!(LocalDisk::is_valid_volname("valid/name"));
assert!(LocalDisk::is_valid_volname("valid:name"));
}
}
#[tokio::test]
async fn test_read_file_exists() {
let test_file = "./test_read_exists.txt";
// Test non-existent file
let (data, metadata) = read_file_exists(test_file).await.expect("operation should succeed");
assert!(data.is_empty());
assert!(metadata.is_none());
// Create test file
fs::write(test_file, b"test content").await.expect("operation should succeed");
// Test existing file
let (data, metadata) = read_file_exists(test_file).await.expect("operation should succeed");
assert_eq!(data.as_ref(), b"test content");
assert!(metadata.is_some());
// Clean up
let _ = fs::remove_file(test_file).await;
}
#[tokio::test]
async fn test_read_file_all() {
let test_file = "./test_read_all.txt";
let test_content = b"test content for read_all";
// Create test file
fs::write(test_file, test_content).await.expect("operation should succeed");
// Test reading file
let (data, metadata) = read_file_all(test_file).await.expect("operation should succeed");
assert_eq!(data.as_ref(), test_content);
assert!(metadata.is_file());
assert_eq!(metadata.len(), test_content.len() as u64);
// Clean up
let _ = fs::remove_file(test_file).await;
}
#[tokio::test]
async fn test_read_file_metadata() {
let test_file = "./test_metadata.txt";
// Create test file
fs::write(test_file, b"test").await.expect("operation should succeed");
// Test reading metadata
let metadata = read_file_metadata(test_file).await.expect("operation should succeed");
assert!(metadata.is_file());
assert_eq!(metadata.len(), 4); // "test" is 4 bytes
// Clean up
let _ = fs::remove_file(test_file).await;
}
#[test]
fn test_is_root_path() {
// Unix root path
assert!(is_root_path("/"));
// Windows root path (only on Windows)
#[cfg(windows)]
assert!(is_root_path("\\"));
// Non-root paths
assert!(!is_root_path("/home"));
assert!(!is_root_path("/tmp"));
assert!(!is_root_path("relative/path"));
// On non-Windows systems, backslash is not a root path
#[cfg(not(windows))]
assert!(!is_root_path("\\"));
}
#[test]
fn test_normalize_path_components() {
// Test basic relative path
assert_eq!(normalize_path_components("a/b/c"), PathBuf::from("a/b/c"));
// Test path with current directory components (should be ignored)
assert_eq!(normalize_path_components("a/./b/./c"), PathBuf::from("a/b/c"));
// Test path with parent directory components
assert_eq!(normalize_path_components("a/b/../c"), PathBuf::from("a/c"));
// Test path with multiple parent directory components
assert_eq!(normalize_path_components("a/b/c/../../d"), PathBuf::from("a/d"));
// Test path that goes beyond root
assert_eq!(normalize_path_components("a/../../../b"), PathBuf::from("b"));
// Test absolute path
assert_eq!(normalize_path_components("/a/b/c"), PathBuf::from("/a/b/c"));
// Test absolute path with parent components
assert_eq!(normalize_path_components("/a/b/../c"), PathBuf::from("/a/c"));
// Test complex path with mixed components
assert_eq!(normalize_path_components("a/./b/../c/./d/../e"), PathBuf::from("a/c/e"));
// Test path with only current directory
assert_eq!(normalize_path_components("."), PathBuf::from(""));
// Test path with only parent directory
assert_eq!(normalize_path_components(".."), PathBuf::from(""));
// Test path with multiple current directories
assert_eq!(normalize_path_components("./././a"), PathBuf::from("a"));
// Test path with multiple parent directories
assert_eq!(normalize_path_components("../../a"), PathBuf::from("a"));
// Test empty path
assert_eq!(normalize_path_components(""), PathBuf::from(""));
// Test path starting with current directory
assert_eq!(normalize_path_components("./a/b"), PathBuf::from("a/b"));
// Test path starting with parent directory
assert_eq!(normalize_path_components("../a/b"), PathBuf::from("a/b"));
// Test complex case with multiple levels of parent navigation
assert_eq!(normalize_path_components("a/b/c/../../../d/e/f/../../g"), PathBuf::from("d/g"));
// Test path that completely cancels out
assert_eq!(normalize_path_components("a/b/../../../c/d/../../.."), PathBuf::from(""));
// Test Windows-style paths (if applicable)
#[cfg(windows)]
{
assert_eq!(normalize_path_components("C:\\a\\b\\c"), PathBuf::from("C:\\a\\b\\c"));
assert_eq!(normalize_path_components("C:\\a\\..\\b"), PathBuf::from("C:\\b"));
}
}
#[test]
fn should_reclaim_file_cache_after_write_respects_env_and_threshold() {
temp_env::with_var_unset(rustfs_config::ENV_OBJECT_FILE_CACHE_RECLAIM_WRITE_ENABLE, || {
assert!(!should_reclaim_file_cache_after_write(8 * 1024 * 1024));
});
temp_env::with_var(rustfs_config::ENV_OBJECT_FILE_CACHE_RECLAIM_WRITE_ENABLE, Some("true"), || {
temp_env::with_var(rustfs_config::ENV_OBJECT_FILE_CACHE_RECLAIM_THRESHOLD, Some("4194304"), || {
assert!(should_reclaim_file_cache_after_write(8 * 1024 * 1024));
assert!(!should_reclaim_file_cache_after_write(1024));
});
});
}
#[test]
fn should_reclaim_file_cache_after_read_respects_env_and_threshold() {
temp_env::with_var_unset(rustfs_config::ENV_OBJECT_FILE_CACHE_RECLAIM_READ_ENABLE, || {
temp_env::with_var_unset(rustfs_config::ENV_OBJECT_FILE_CACHE_RECLAIM_THRESHOLD, || {
assert!(should_reclaim_file_cache_after_read(8 * 1024 * 1024));
assert!(!should_reclaim_file_cache_after_read(1024));
});
});
temp_env::with_var(rustfs_config::ENV_OBJECT_FILE_CACHE_RECLAIM_READ_ENABLE, Some("false"), || {
temp_env::with_var(rustfs_config::ENV_OBJECT_FILE_CACHE_RECLAIM_THRESHOLD, Some("4194304"), || {
assert!(!should_reclaim_file_cache_after_read(8 * 1024 * 1024));
});
});
temp_env::with_var(rustfs_config::ENV_OBJECT_FILE_CACHE_RECLAIM_READ_ENABLE, Some("true"), || {
temp_env::with_var(rustfs_config::ENV_OBJECT_FILE_CACHE_RECLAIM_THRESHOLD, Some("4194304"), || {
assert!(should_reclaim_file_cache_after_read(8 * 1024 * 1024));
assert!(!should_reclaim_file_cache_after_read(1024));
});
});
}
#[test]
fn should_populate_mmap_read_respects_env() {
temp_env::with_var_unset(ENV_RUSTFS_OBJECT_MMAP_POPULATE_ENABLE, || {
assert!(!should_populate_mmap_read(512 * 1024));
});
temp_env::with_var(ENV_RUSTFS_OBJECT_MMAP_POPULATE_ENABLE, Some("true"), || {
assert!(should_populate_mmap_read(512 * 1024));
assert!(!should_populate_mmap_read(0));
});
temp_env::with_var(ENV_RUSTFS_OBJECT_MMAP_POPULATE_ENABLE, Some("false"), || {
assert!(!should_populate_mmap_read(512 * 1024));
});
}
#[test]
fn local_read_copy_method_respects_env() {
temp_env::with_var_unset(ENV_RUSTFS_OBJECT_MMAP_READ_METHOD, || {
assert_eq!(local_read_copy_method(), LocalReadCopyMethod::MmapCopy);
});
temp_env::with_var(
ENV_RUSTFS_OBJECT_MMAP_READ_METHOD,
Some(RUSTFS_OBJECT_MMAP_READ_METHOD_DIRECT_READ_COPY),
|| {
assert_eq!(local_read_copy_method(), LocalReadCopyMethod::DirectReadCopy);
},
);
temp_env::with_var(ENV_RUSTFS_OBJECT_MMAP_READ_METHOD, Some("unknown"), || {
assert_eq!(local_read_copy_method(), LocalReadCopyMethod::MmapCopy);
});
}
#[cfg(unix)]
#[test]
fn mmap_page_size_is_cached_positive() {
let first = mmap_page_size().expect("page size should be available");
let second = mmap_page_size().expect("cached page size should be available");
assert!(first > 0);
assert_eq!(first, second);
}
#[cfg(unix)]
#[tokio::test]
async fn read_file_mmap_copy_supports_direct_read_copy_method() {
use tempfile::tempdir;
temp_env::async_with_vars(
[(ENV_RUSTFS_OBJECT_MMAP_READ_METHOD, Some(RUSTFS_OBJECT_MMAP_READ_METHOD_DIRECT_READ_COPY))],
async {
let root_dir = tempdir().expect("operation should succeed");
let endpoint = Endpoint::try_from(root_dir.path().to_string_lossy().as_ref()).expect("operation should succeed");
let disk = LocalDisk::new(&endpoint, false).await.expect("operation should succeed");
disk.make_volume("test-volume").await.expect("operation should succeed");
disk.write_all("test-volume", "test-file.txt", Bytes::from_static(b"0123456789abcdef"))
.await
.expect("operation should succeed");
let data = disk
.read_file_mmap_copy("test-volume", "test-file.txt", 4, 6)
.await
.expect("operation should succeed");
assert_eq!(data, Bytes::from_static(b"456789"));
},
)
.await;
}
#[cfg(unix)]
#[test]
fn mmap_page_fault_delta_clamps_non_monotonic_counts() {
let before = Some(MmapPageFaultCounts { minor: 10, major: 4 });
let after = Some(MmapPageFaultCounts { minor: 7, major: 6 });
assert_eq!(mmap_page_fault_delta(before, after), MmapPageFaultDelta { minor: 0, major: 2 });
assert_eq!(mmap_page_fault_delta(before, None), MmapPageFaultDelta::default());
}
#[test]
fn test_is_bitrot_size_mismatch_error_only_matches_target_message() {
assert!(is_bitrot_size_mismatch_error(&std::io::Error::other("bitrot shard file size mismatch")));
assert!(!is_bitrot_size_mismatch_error(&std::io::Error::other("bitrot hash mismatch")));
}
#[test]
fn test_is_bitrot_verification_error_matches_hash_and_size_mismatch() {
assert!(is_bitrot_verification_error(&std::io::Error::other("bitrot shard file size mismatch")));
assert!(is_bitrot_verification_error(&std::io::Error::other("bitrot hash mismatch")));
assert!(!is_bitrot_verification_error(&std::io::Error::other("unrelated io failure")));
}
#[tokio::test]
async fn local_disk_read_file_verifier_reports_bitrot_mismatch() {
use crate::erasure::coding::BitrotWriter;
use rustfs_filemeta::ChecksumInfo;
use tempfile::tempdir;
let root_dir = tempdir().expect("temp dir should be created");
let endpoint = Endpoint::try_from(root_dir.path().to_string_lossy().as_ref()).expect("endpoint should parse");
let disk = LocalDisk::new(&endpoint, false).await.expect("local disk should be created");
let volume = "verify-volume";
ensure_test_volume(&disk, volume).await;
let payload = Bytes::from_static(b"bitrot-payload!!");
let object = "object.bin";
let data_dir = Uuid::new_v4();
let part_number = 1;
let checksum_algo = HashAlgorithm::HighwayHash256S;
let mut file_info = FileInfo::new(object, 1, 0);
file_info.volume = volume.to_string();
file_info.name = object.to_string();
file_info.size = i64::try_from(payload.len()).expect("test payload length should fit i64");
file_info.data_dir = Some(data_dir);
file_info.erasure.block_size = payload.len();
file_info.erasure.index = 1;
file_info.erasure.checksums = vec![ChecksumInfo {
part_number,
algorithm: checksum_algo.clone(),
hash: Bytes::new(),
}];
file_info.parts = vec![ObjectPartInfo {
number: part_number,
size: payload.len(),
actual_size: i64::try_from(payload.len()).expect("test payload length should fit i64"),
..Default::default()
}];
let mut writer = BitrotWriter::new(std::io::Cursor::new(Vec::new()), file_info.erasure.shard_size(), checksum_algo);
writer
.write(&payload)
.await
.expect("bitrot writer should encode test payload");
writer.shutdown().await.expect("bitrot writer should flush test payload");
let mut encoded = writer.into_inner().into_inner();
let last = encoded.last_mut().expect("encoded part should not be empty");
*last ^= 0xff;
let part_path = path_join_buf(&[object, &data_dir.to_string(), &format!("part.{part_number}")]);
disk.write_all(volume, &part_path, Bytes::from(encoded))
.await
.expect("corrupted encoded part should be written");
let result = disk
.verify_file(volume, object, &file_info)
.await
.expect("verify_file should return per-part status");
assert_eq!(result.results, vec![CHECK_PART_FILE_CORRUPT]);
}
// ----- HP-6: O_DIRECT shard write -----
#[test]
fn direct_io_write_env_gate_defaults_off() {
temp_env::with_var_unset(ENV_RUSTFS_OBJECT_DIRECT_IO_WRITE_ENABLE, || {
assert!(!is_direct_io_write_enabled(), "O_DIRECT write must default to off");
});
temp_env::with_var(ENV_RUSTFS_OBJECT_DIRECT_IO_WRITE_ENABLE, Some("true"), || {
assert!(is_direct_io_write_enabled());
});
temp_env::with_var(ENV_RUSTFS_OBJECT_DIRECT_IO_WRITE_ENABLE, Some("false"), || {
assert!(!is_direct_io_write_enabled());
});
}
#[test]
fn direct_write_staging_capacity_is_smallest_alignment_multiple_covering_target() {
for align in [512usize, 1024, 4096, 8192, 65536] {
let cap = direct_write_staging_capacity(align);
assert_eq!(cap % align, 0, "capacity must be an alignment multiple (align={align})");
assert!(
cap >= DIRECT_WRITE_STAGING_BYTES,
"capacity must cover the target staging size (align={align})"
);
assert!(
cap - DIRECT_WRITE_STAGING_BYTES < align,
"capacity must be the smallest covering multiple (align={align})"
);
}
}
#[test]
fn direct_write_tail_split_separates_aligned_prefix_from_remainder() {
let align = 4096usize;
assert_eq!(direct_write_tail_split(0, align), (0, 0));
assert_eq!(direct_write_tail_split(align, align), (align, 0));
assert_eq!(direct_write_tail_split(3 * align, align), (3 * align, 0));
assert_eq!(direct_write_tail_split(100, align), (0, 100));
let (prefix, remainder) = direct_write_tail_split(2 * align + 123, align);
assert_eq!((prefix, remainder), (2 * align, 123));
assert_eq!(prefix + remainder, 2 * align + 123, "split must reconstruct the input length");
assert_eq!(prefix % align, 0, "prefix must be alignment-sized");
assert!(remainder < align, "remainder must be sub-alignment");
}
/// End-to-end round trip through `create_file`'s writer with O_DIRECT writes
/// enabled. On a block-backed Linux filesystem this drives the true
/// O_DIRECT path; on macOS and on CI filesystems that reject O_DIRECT it
/// exercises the buffered fallback. Either way, enabling the gate must not
/// change the bytes read back, across sizes crossing the alignment and
/// multi-batch staging boundaries (zero-breakage contract).
#[cfg(unix)]
#[tokio::test]
async fn create_file_direct_write_round_trips_all_sizes() {
use tempfile::tempdir;
let root_dir = tempdir().expect("tempdir");
let endpoint = Endpoint::try_from(root_dir.path().to_string_lossy().as_ref()).expect("endpoint");
let disk = LocalDisk::new(&endpoint, false).await.expect("disk");
disk.make_volume("test-volume").await.expect("make_volume");
let sizes = [
1usize,
511,
512,
4095,
4096,
4097,
8192,
12345,
DIRECT_WRITE_STAGING_BYTES - 1,
DIRECT_WRITE_STAGING_BYTES,
DIRECT_WRITE_STAGING_BYTES + 4096,
2 * DIRECT_WRITE_STAGING_BYTES + 777,
];
for (i, size) in sizes.into_iter().enumerate() {
let mut state = 0x1234_5678u64.wrapping_add(i as u64);
let content: Vec<u8> = (0..size)
.map(|_| {
state = state.wrapping_mul(6364136223846793005).wrapping_add(1442695040888963407);
(state >> 33) as u8
})
.collect();
let path = format!("obj-{i}.bin");
temp_env::async_with_vars([(ENV_RUSTFS_OBJECT_DIRECT_IO_WRITE_ENABLE, Some("true"))], async {
let mut writer = disk
.create_file("", "test-volume", &path, size as i64)
.await
.expect("create_file");
// Odd-sized chunks exercise partial staging fills and flushes.
let mut off = 0;
while off < content.len() {
let end = (off + 777).min(content.len());
writer.write_all(&content[off..end]).await.expect("write_all");
off = end;
}
writer.shutdown().await.expect("shutdown");
})
.await;
let got = disk.read_file_mmap_copy("test-volume", &path, 0, size).await.expect("read");
assert_eq!(got.as_ref(), content.as_slice(), "round-trip mismatch at size={size}");
}
}
/// Deterministic coverage of the O_DIRECT writer's aligned-batch + tail
/// state machine on Linux, independent of whether the CI filesystem
/// supports O_DIRECT: the writer is driven over a plain file with a small
/// alignment and staging capacity so both a full-batch flush and the tail
/// split are exercised for every size class.
#[cfg(target_os = "linux")]
#[tokio::test]
async fn direct_writer_state_machine_round_trips_over_plain_file() {
use std::io::Read;
use tempfile::tempdir;
let dir = tempdir().expect("tempdir");
let align = 512usize;
let capacity = 4 * align; // forces multi-batch flushes for larger sizes
for &size in &[0usize, 1, 511, 512, 513, 1024, 2048, 2049, 5000] {
let path = dir.path().join(format!("s{size}"));
let file = std::fs::OpenOptions::new()
.create(true)
.write(true)
.truncate(true)
.read(true)
.open(&path)
.expect("open plain file");
let content: Vec<u8> = (0..size).map(|i| (i.wrapping_mul(7).wrapping_add(3)) as u8).collect();
let mut writer = DirectWriter::from_std_file_for_test(file, align, capacity);
let mut off = 0;
while off < content.len() {
let end = (off + 300).min(content.len());
writer.write_all(&content[off..end]).await.expect("write_all");
off = end;
}
writer.shutdown().await.expect("shutdown");
let mut got = Vec::new();
std::fs::File::open(&path)
.expect("reopen")
.read_to_end(&mut got)
.expect("read_to_end");
assert_eq!(got, content, "state-machine round-trip mismatch at size={size}");
}
}
/// Differential test for the LocalIoBackend refactor: every read shape
/// (pread via mmap_copy, pread via direct_read_copy, bounded stream, full
/// read) must return byte-identical data for the same (offset, length),
/// including ranges straddling page boundaries and the file tail.
#[cfg(unix)]
#[tokio::test]
async fn io_backend_read_shapes_return_identical_bytes() {
use tempfile::tempdir;
use tokio::io::AsyncReadExt;
const FILE_LEN: usize = 64 * 1024;
// Deterministic pseudo-random content (LCG) so failures are reproducible.
let mut state = 0x9e3779b97f4a7c15u64;
let content: Vec<u8> = (0..FILE_LEN)
.map(|_| {
state = state.wrapping_mul(6364136223846793005).wrapping_add(1442695040888963407);
(state >> 33) as u8
})
.collect();
let content = Bytes::from(content);
let root_dir = tempdir().expect("operation should succeed");
let endpoint = Endpoint::try_from(root_dir.path().to_string_lossy().as_ref()).expect("operation should succeed");
let disk = LocalDisk::new(&endpoint, false).await.expect("operation should succeed");
disk.make_volume("test-volume").await.expect("operation should succeed");
disk.write_all("test-volume", "blob.bin", content.clone())
.await
.expect("operation should succeed");
let page = mmap_page_size().expect("page size should be available") as usize;
let ranges = [
(0usize, FILE_LEN),
(1, 17),
(page - 1, 2),
(page, page),
(2 * page - 1, page + 2),
(FILE_LEN - 7, 7),
(0, 0),
];
for (offset, length) in ranges {
let expected = content.slice(offset..offset + length);
for method in [
RUSTFS_OBJECT_MMAP_READ_METHOD_MMAP_COPY,
RUSTFS_OBJECT_MMAP_READ_METHOD_DIRECT_READ_COPY,
] {
let got = temp_env::async_with_vars([(ENV_RUSTFS_OBJECT_MMAP_READ_METHOD, Some(method))], async {
disk.read_file_mmap_copy("test-volume", "blob.bin", offset, length)
.await
.expect("operation should succeed")
})
.await;
assert_eq!(got, expected, "pread_bytes({method}) mismatch at offset={offset} length={length}");
}
let mut stream = disk
.read_file_stream("test-volume", "blob.bin", offset, length)
.await
.expect("operation should succeed");
let mut streamed = vec![0u8; length];
stream.read_exact(&mut streamed).await.expect("operation should succeed");
assert_eq!(
Bytes::from(streamed),
expected,
"open_read_stream mismatch at offset={offset} length={length}"
);
}
let mut full = disk
.read_file("test-volume", "blob.bin")
.await
.expect("operation should succeed");
let mut all = Vec::new();
full.read_to_end(&mut all).await.expect("operation should succeed");
assert_eq!(Bytes::from(all), content, "open_full_read mismatch");
}
/// The O_DIRECT read path must return the same bytes as the buffered
/// path for unaligned shard sizes and ranges, and must silently fall
/// back (never error) on filesystems that reject O_DIRECT (e.g. tmpfs).
/// Both legs are covered regardless of which filesystem backs tempdir.
#[cfg(target_os = "linux")]
#[tokio::test]
async fn direct_io_read_matches_buffered_path_or_falls_back() {
use tempfile::tempdir;
// Unaligned on purpose: 3 blocks + 7 bytes.
const FILE_LEN: usize = 4096 * 3 + 7;
let mut state = 0x2545f4914f6cdd1du64;
let content: Vec<u8> = (0..FILE_LEN)
.map(|_| {
state = state.wrapping_mul(6364136223846793005).wrapping_add(1442695040888963407);
(state >> 33) as u8
})
.collect();
let content = Bytes::from(content);
let root_dir = tempdir().expect("operation should succeed");
let endpoint = Endpoint::try_from(root_dir.path().to_string_lossy().as_ref()).expect("operation should succeed");
let disk = LocalDisk::new(&endpoint, false).await.expect("operation should succeed");
disk.make_volume("test-volume").await.expect("operation should succeed");
disk.write_all("test-volume", "shard.bin", content.clone())
.await
.expect("operation should succeed");
let ranges = [
(0usize, FILE_LEN),
(0, 4096),
(4095, 4098),
(4096 * 2, 4096 + 7),
(FILE_LEN - 7, 7),
];
for (offset, length) in ranges {
let expected = content.slice(offset..offset + length);
// Threshold 1 forces every non-empty read through the O_DIRECT attempt.
let got = temp_env::async_with_vars(
[
(ENV_RUSTFS_OBJECT_DIRECT_IO_READ_ENABLE, Some("true")),
(ENV_RUSTFS_OBJECT_DIRECT_IO_READ_THRESHOLD, Some("1")),
],
async {
disk.read_file_mmap_copy("test-volume", "shard.bin", offset, length)
.await
.expect("O_DIRECT-eligible read must succeed (direct or fallback)")
},
)
.await;
assert_eq!(got, expected, "direct-io read mismatch at offset={offset} length={length}");
}
}
/// pread_direct_aligned must never leak alignment padding: the returned
/// buffer is exactly the requested logical range.
#[cfg(target_os = "linux")]
#[test]
fn pread_direct_aligned_exact_range_or_unsupported() {
use std::io::Write;
let dir = tempfile::tempdir().expect("operation should succeed");
let file_path = dir.path().join("blob.bin");
let content: Vec<u8> = (0..4096 * 2 + 13).map(|i| (i % 251) as u8).collect();
std::fs::File::create(&file_path)
.and_then(|mut f| f.write_all(&content))
.expect("operation should succeed");
let state = DirectIoReadState::new();
match pread_direct_aligned(&file_path, 4090, 100, &state) {
Ok(bytes) => {
assert_eq!(&bytes[..], &content[4090..4190], "padding must not leak");
}
Err(err) => {
assert!(
is_direct_io_unsupported(&err),
"only unsupported-filesystem errors are acceptable here: {err}"
);
}
}
}
/// P1.5 benchmark gate harness (backlog#893): O_DIRECT vs mmap-copy.
///
/// Ignored by default; run explicitly in release mode on a Linux box:
///
/// ```text
/// RUSTFS_BENCH_DIR=/data/rustfs/bench \
/// RUSTFS_OBJECT_DIRECT_IO_READ_ENABLE=true RUSTFS_OBJECT_DIRECT_IO_READ_THRESHOLD=1 \
/// cargo test --release -p rustfs-ecstore --lib direct_read_bench_gate -- --ignored --nocapture
/// ```
///
/// Baseline run: same command without the two DIRECT_IO vars. Knobs:
/// RUSTFS_BENCH_SHARD_MIB (8), RUSTFS_BENCH_FILE_COUNT (64),
/// RUSTFS_BENCH_READS (256). Cold cache is enforced with
/// fadvise(DONTNEED) over the dataset between rounds. Every read is
/// verified for length; the first four shards are verified byte-for-byte
/// before timing starts (correctness before performance).
#[cfg(target_os = "linux")]
#[tokio::test]
#[ignore = "benchmark harness, run explicitly in release mode"]
async fn direct_read_bench_gate() {
use std::time::Instant;
fn env_usize(name: &str, default: usize) -> usize {
std::env::var(name).ok().and_then(|v| v.parse().ok()).unwrap_or(default)
}
fn cpu_time_secs() -> f64 {
// SAFETY: getrusage with RUSAGE_SELF and a zeroed out-param.
#[allow(unsafe_code)]
unsafe {
let mut ru: libc::rusage = std::mem::zeroed();
if libc::getrusage(libc::RUSAGE_SELF, &mut ru) != 0 {
return f64::NAN;
}
let tv = |t: libc::timeval| t.tv_sec as f64 + t.tv_usec as f64 / 1e6;
tv(ru.ru_utime) + tv(ru.ru_stime)
}
}
fn drop_dataset_cache(paths: &[PathBuf]) {
use rustix::fs::{Advice, fadvise};
for p in paths {
if let Ok(f) = std::fs::File::open(p) {
let _ = fadvise(&f, 0, None, Advice::DontNeed);
}
}
}
fn percentile(sorted: &[f64], p: f64) -> f64 {
let idx = ((sorted.len() as f64 - 1.0) * p).round() as usize;
sorted[idx]
}
fn gen_content(len: usize, seed: u64) -> Bytes {
let mut state = seed;
let v: Vec<u8> = (0..len)
.map(|_| {
state = state.wrapping_mul(6364136223846793005).wrapping_add(1442695040888963407);
(state >> 33) as u8
})
.collect();
Bytes::from(v)
}
let bench_dir = std::env::var("RUSTFS_BENCH_DIR").expect("set RUSTFS_BENCH_DIR to a directory on the target disk");
let shard_mib = env_usize("RUSTFS_BENCH_SHARD_MIB", 8);
let file_count = env_usize("RUSTFS_BENCH_FILE_COUNT", 64);
let reads = env_usize("RUSTFS_BENCH_READS", 256);
let shard_len = shard_mib * 1024 * 1024;
const VOLUME: &str = "bench-volume";
std::fs::create_dir_all(&bench_dir).expect("create bench dir");
let endpoint = Endpoint::try_from(bench_dir.as_str()).expect("endpoint");
let disk = LocalDisk::new(&endpoint, false).await.expect("local disk");
let _ = disk.make_volume(VOLUME).await;
// Populate, skipping shards that already exist with the right size.
let mut paths = Vec::with_capacity(file_count);
for i in 0..file_count {
let name = format!("shard-{i:04}.bin");
let abs = disk.get_object_path(VOLUME, &name).expect("path");
let need_write = std::fs::metadata(&abs).map(|m| m.len() as usize != shard_len).unwrap_or(true);
if need_write {
disk.write_all(VOLUME, &name, gen_content(shard_len, 0x9e3779b9 + i as u64))
.await
.expect("populate shard");
}
paths.push(abs);
}
// Correctness gate before any timing.
for i in 0..4.min(file_count) {
let name = format!("shard-{i:04}.bin");
let got = disk
.read_file_mmap_copy(VOLUME, &name, 0, shard_len)
.await
.expect("verify read");
assert_eq!(got, gen_content(shard_len, 0x9e3779b9 + i as u64), "shard {i} content mismatch");
}
let mut idx_state = 0xdeadbeefu64;
let mut pick = |n: usize| {
idx_state = idx_state.wrapping_mul(6364136223846793005).wrapping_add(1);
((idx_state >> 33) as usize) % n
};
for _ in 0..8 {
let name = format!("shard-{:04}.bin", pick(file_count));
let _ = disk
.read_file_mmap_copy(VOLUME, &name, 0, shard_len)
.await
.expect("warmup read");
}
drop_dataset_cache(&paths);
// Concurrency models the EC GET shape: FuturesUnordered over shard
// reads. concurrency=1 keeps the original sequential behavior.
let concurrency = env_usize("RUSTFS_BENCH_CONCURRENCY", 1).max(1);
let disk = std::sync::Arc::new(disk);
let mut latencies_us = Vec::with_capacity(reads);
let cpu_before = cpu_time_secs();
let wall_start = Instant::now();
let mut done = 0usize;
while done < reads {
use futures::StreamExt;
let batch = concurrency.min(reads - done);
let mut tasks = futures::stream::FuturesUnordered::new();
for _ in 0..batch {
let name = format!("shard-{:04}.bin", pick(file_count));
let disk = disk.clone();
tasks.push(async move {
let t = Instant::now();
let bytes = disk
.read_file_mmap_copy(VOLUME, &name, 0, shard_len)
.await
.expect("bench read");
assert_eq!(bytes.len(), shard_len);
t.elapsed().as_secs_f64() * 1e6
});
}
while let Some(lat) = tasks.next().await {
latencies_us.push(lat);
}
done += batch;
if done.is_multiple_of(file_count) {
drop_dataset_cache(&paths);
}
}
let wall = wall_start.elapsed().as_secs_f64();
let cpu = cpu_time_secs() - cpu_before;
latencies_us.sort_by(|a, b| a.partial_cmp(b).expect("finite"));
let mean = latencies_us.iter().sum::<f64>() / latencies_us.len() as f64;
let direct_enabled = std::env::var(ENV_RUSTFS_OBJECT_DIRECT_IO_READ_ENABLE).unwrap_or_default();
println!(
"BENCH_RESULT {{\"direct_io_enabled\":\"{direct_enabled}\",\"concurrency\":{concurrency},\"shard_mib\":{shard_mib},\"file_count\":{file_count},\
\"reads\":{reads},\"p50_us\":{:.1},\"p95_us\":{:.1},\"p99_us\":{:.1},\"mean_us\":{:.1},\"wall_s\":{wall:.3},\
\"cpu_s\":{cpu:.3},\"throughput_mib_s\":{:.1}}}",
percentile(&latencies_us, 0.50),
percentile(&latencies_us, 0.95),
percentile(&latencies_us, 0.99),
mean,
(reads * shard_mib) as f64 / wall,
);
}
}