Compare commits

..

2 Commits

Author SHA1 Message Date
overtrue 0c3d26f208 chore(rustfs): adjudicate the remaining 36 bare dead_code allows
Finishes backlog#1823 step 10 for `rustfs/src`: the 36 bare allows still spread across 23 files, after #6254 took the four densest ones. Stripped first, then clippy asked which the compiler actually missed — 22 were inert.

Ten items behind the rest are deleted, in three recurring shapes:

- **No-argument shims.** `storage/options.rs`'s `get_content_sha256`, `skip_content_sha256_cksum` and `get_content_sha256_cksum` each forward to a `_with_query` variant carrying 8, 2 and 2 live callers; none of the shims had one. This is the same leftover as `auth.rs`'s two in #6254 — adding the `query` parameter left the old signatures behind.
- **A local shadow.** `admin/handlers/replication.rs`'s `is_local_host` was never called; its one apparent call site uses `rustfs_utils::net::is_local_host`.
- **A dead cluster.** `cache_clear` is reachable only from `invalidate_all_bucket_validation_cache`, which nothing calls.

Also removed: `head_prefix.rs::is_prefix_key`, `startup_iam.rs::reset_test_failure_counter` (no consumer, including under `rustfs/tests` and `crates/e2e_test`), and `init.rs::spawn_server`.

Four keep a reasoned allow. `admin/route_policy.rs::validate_admin_route_policy_specs` and `storage/ecfs_extend.rs::get_adaptive_buffer_size_with_profile` are exercised only by tests, which the lib target cannot see. `app/object_usecase.rs::io_strategy`, `storage/access.rs::region` and `storage/concurrency/manager.rs::priority_queue` are written and never read back.

`rustfs/src` now has no bare `#[allow(dead_code)]` left; the seven remaining carry a trailing `//` justification and are a separate question.

Refs backlog#1823
2026-08-19 13:27:39 +08:00
overtrue ed07f2dc03 chore(rustfs): adjudicate 37 bare dead_code allows in the four densest files
Continues backlog#1823 step 10 into the `rustfs` crate after #6161, #6162, #6173 and #6187 cleared the libraries. Each allow was stripped first and clippy asked which ones the compiler actually missed, so the verdicts rest on the diagnostic rather than on reading.

27 of the 37 were inert — including all eight in `admin/handlers/tier.rs` and seventeen of the nineteen in `storage/concurrency/io_schedule.rs`. Removing them changes no diagnostic.

Six items behind the remaining allows are deleted:

- `auth.rs`'s `determine_auth_type_and_version` and `is_request_presigned_signature_v4` are no-argument shims over their `_with_query` variants, which carry 2 and 5 live callers respectively. Neither shim had one.
- `io_schedule.rs`'s `lifetime_average_wait`, whose sibling accessors (`observation_count`, `average_wait`, `smoothed_load_level`) are all consumed.
- `console.rs`'s `version()`, `license()` and `doc()`. The live accessor is `version_info()`.

Two keep a reasoned allow. `io_schedule.rs`'s `original_priority` is written and never read back. And `console.rs`'s `config_handler`, with `Config::port` and `to_json()` which only it uses: that handler is covered by a test but no route registers it, so `/rustfs/console/api/v1/config` currently falls through to the SPA's static fallback. Deleting it would erase the only signal that the endpoint is meant to exist, so it stays annotated and the missing route is filed on the issue.

Refs backlog#1823
2026-08-19 12:44:33 +08:00
48 changed files with 1469 additions and 2794 deletions
+2 -9
View File
@@ -236,19 +236,12 @@ async fn audit_pipeline_reports_empty_runtime_snapshots() {
}
#[tokio::test]
async fn stopping_audit_replay_workers_is_a_no_op_when_there_are_none() {
async fn audit_runtime_facade_stops_empty_replay_workers() {
let registry = Arc::new(Mutex::new(AuditRegistry::new()));
let replay_workers = Arc::new(RwLock::new(rustfs_targets::ReplayWorkerManager::new()));
let facade = AuditRuntimeFacade::new(registry, Arc::clone(&replay_workers));
let facade = AuditRuntimeFacade::new(registry, replay_workers);
facade.stop_replay_workers().await;
// The stop path takes the manager's workers and hands them to the adapter,
// so an empty facade must leave it empty rather than wedge it, and a second
// call — which shutdown paths make — must stay harmless (rustfs/backlog#1836).
assert!(replay_workers.read().await.is_empty());
facade.stop_replay_workers().await;
assert!(replay_workers.read().await.is_empty());
}
#[tokio::test]
+9
View File
@@ -228,6 +228,15 @@ pub const DEFAULT_SCANNER_MAX_CONCURRENT_DISK_SCANS: usize = 4;
/// Default object interval for cooperative scanner yields.
pub const DEFAULT_SCANNER_YIELD_EVERY_N_OBJECTS: u64 = 128;
/// Compatibility flag kept for Patch 3 rollback windows.
///
/// Inline scanner heal execution has been removed in favor of heal-candidate enqueue.
/// When this flag is enabled, RustFS logs a warning and continues to use enqueue-based heal.
pub const ENV_SCANNER_INLINE_HEAL_ENABLE: &str = "RUSTFS_SCANNER_INLINE_HEAL_ENABLE";
/// Default inline scanner heal compatibility mode.
pub const DEFAULT_SCANNER_INLINE_HEAL_ENABLE: bool = false;
/// Scanner speed preset controlling throttling behavior.
///
/// Each preset defines three parameters:
-170
View File
@@ -168,24 +168,6 @@ async fn wait_for_version_expired(
}
}
async fn wait_for_key_versions_empty(client: &Client, bucket: &str, key: &str, deadline: StdDuration) -> TestResult {
let start = std::time::Instant::now();
loop {
let listing = client.list_object_versions().bucket(bucket).prefix(key).send().await?;
if listing.versions().is_empty() && listing.delete_markers().is_empty() {
return Ok(());
}
if start.elapsed() >= deadline {
return Err(format!(
"object {bucket}/{key} still had versions or delete markers after {}s: {listing:?}",
deadline.as_secs()
)
.into());
}
tokio::time::sleep(StdDuration::from_millis(500)).await;
}
}
/// Build a prefix-scoped `Days`-based expiration rule.
fn expiration_rule(id: &str, prefix: &str, days: i32) -> Result<LifecycleRule, Box<dyn std::error::Error + Send + Sync>> {
let rule = LifecycleRule::builder()
@@ -211,21 +193,6 @@ fn noncurrent_expiration_rule(
Ok(rule)
}
fn noncurrent_expiration_with_delete_marker_cleanup_rule(
id: &str,
prefix: &str,
days: i32,
) -> Result<LifecycleRule, Box<dyn std::error::Error + Send + Sync>> {
let rule = LifecycleRule::builder()
.id(id)
.filter(LifecycleRuleFilter::builder().prefix(prefix).build())
.expiration(LifecycleExpiration::builder().expired_object_delete_marker(true).build())
.noncurrent_version_expiration(NoncurrentVersionExpiration::builder().noncurrent_days(days).build())
.status(ExpirationStatus::Enabled)
.build()?;
Ok(rule)
}
async fn put_expiration_config(client: &Client, bucket: &str, rule: LifecycleRule) -> TestResult {
let lifecycle = BucketLifecycleConfiguration::builder().rules(rule).build()?;
client
@@ -445,143 +412,6 @@ async fn test_lifecycle_noncurrent_version_expiry_removes_only_old_version() ->
Ok(())
}
/// A combined `NoncurrentDays=1` and `ExpiredObjectDeleteMarker=true` rule
/// must remove a noncurrent data version and then its sole latest delete
/// marker, without expiring current-only objects. A second prefix with only
/// noncurrent expiry proves that marker cleanup comes from EODM.
#[tokio::test(flavor = "multi_thread", worker_threads = 2)]
async fn test_lifecycle_noncurrent_expiry_then_cleans_expired_delete_marker() -> TestResult {
let mut env = RustFSTestEnvironment::new().await?;
let mut extra_env = fast_lifecycle_env();
extra_env.push(("RUSTFS_ILM_DEBUG_DAY_SECS", "2"));
env.start_rustfs_server_with_env(vec![], &extra_env).await?;
let client = env.create_s3_client();
let bucket = "ilm-expired-delete-marker";
client.create_bucket().bucket(bucket).send().await?;
client
.put_bucket_versioning()
.bucket(bucket)
.versioning_configuration(
VersioningConfiguration::builder()
.status(BucketVersioningStatus::Enabled)
.build(),
)
.send()
.await?;
let cascade_key = "cascade/deleted.txt";
let cascade_put = client
.put_object()
.bucket(bucket)
.key(cascade_key)
.body(ByteStream::from_static(b"cascade payload"))
.send()
.await?;
let cascade_data_version = cascade_put
.version_id()
.map(str::to_string)
.expect("cascade PUT returns a version id");
let cascade_delete = client.delete_object().bucket(bucket).key(cascade_key).send().await?;
let cascade_marker_version = cascade_delete
.version_id()
.map(str::to_string)
.expect("cascade DELETE returns a marker version id");
assert_eq!(cascade_delete.delete_marker(), Some(true));
let survivor_key = "cascade/current-only.txt";
client
.put_object()
.bucket(bucket)
.key(survivor_key)
.body(ByteStream::from_static(b"current payload"))
.send()
.await?;
let survivor_before = client.get_object().bucket(bucket).key(survivor_key).send().await?;
assert_eq!(survivor_before.body.collect().await?.into_bytes().as_ref(), b"current payload");
let control_key = "nve-only/deleted.txt";
let control_put = client
.put_object()
.bucket(bucket)
.key(control_key)
.body(ByteStream::from_static(b"control payload"))
.send()
.await?;
let control_data_version = control_put
.version_id()
.map(str::to_string)
.expect("control PUT returns a version id");
let control_delete = client.delete_object().bucket(bucket).key(control_key).send().await?;
let control_marker_version = control_delete
.version_id()
.map(str::to_string)
.expect("control DELETE returns a marker version id");
assert_eq!(control_delete.delete_marker(), Some(true));
let cascade_before = client
.list_object_versions()
.bucket(bucket)
.prefix(cascade_key)
.send()
.await?;
assert!(
cascade_before
.versions()
.iter()
.any(|version| version.version_id() == Some(cascade_data_version.as_str())),
"cascade data version must exist before lifecycle is installed: {cascade_before:?}"
);
assert!(
cascade_before
.delete_markers()
.iter()
.any(|marker| { marker.version_id() == Some(cascade_marker_version.as_str()) && marker.is_latest() == Some(true) }),
"cascade latest delete marker must exist before lifecycle is installed: {cascade_before:?}"
);
let lifecycle = BucketLifecycleConfiguration::builder()
.rules(noncurrent_expiration_with_delete_marker_cleanup_rule(
"expire-and-clean-marker",
"cascade/",
1,
)?)
.rules(noncurrent_expiration_rule("expire-only", "nve-only/", 1)?)
.build()?;
client
.put_bucket_lifecycle_configuration()
.bucket(bucket)
.lifecycle_configuration(lifecycle)
.send()
.await?;
wait_for_key_versions_empty(&client, bucket, cascade_key, StdDuration::from_secs(90)).await?;
wait_for_version_expired(&client, bucket, control_key, &control_data_version, StdDuration::from_secs(90)).await?;
let survivor = client.get_object().bucket(bucket).key(survivor_key).send().await?;
assert_eq!(survivor.body.collect().await?.into_bytes().as_ref(), b"current payload");
let control_after = client
.list_object_versions()
.bucket(bucket)
.prefix(control_key)
.send()
.await?;
assert!(
control_after.versions().is_empty(),
"NVE-only control must remove its data version: {control_after:?}"
);
assert!(
control_after
.delete_markers()
.iter()
.any(|marker| { marker.version_id() == Some(control_marker_version.as_str()) && marker.is_latest() == Some(true) }),
"NVE-only control must preserve its latest delete marker: {control_after:?}"
);
Ok(())
}
/// `Days=0` expiration is invalid per S3 semantics (`Days` must be a positive
/// integer >= 1). A `PutBucketLifecycleConfiguration` carrying a zero-day rule
/// must be rejected with `InvalidArgument` (HTTP 400) - see crates/lifecycle
+6 -81
View File
@@ -41,10 +41,6 @@ use bytes::Bytes;
use futures::lock::Mutex;
use metrics::counter;
use rustfs_filemeta::{FileInfo, ObjectPartInfo, RawFileInfo};
use rustfs_io_metrics::internode_metrics::{
INTERNODE_STAGE_READ_VERSION_REQUEST_ENCODE, INTERNODE_STAGE_READ_VERSION_RESPONSE_DECODE,
INTERNODE_STAGE_READ_VERSION_RPC_ROUNDTRIP,
};
use rustfs_protos::ChannelClass;
use rustfs_protos::evict_failed_connection;
use rustfs_protos::proto_gen::node_service::RenamePartRequest;
@@ -68,7 +64,7 @@ use std::{
atomic::{AtomicBool, AtomicU32, Ordering},
},
task::{Context, Poll},
time::{Duration, Instant},
time::Duration,
};
use tokio::time;
use tokio::{
@@ -1794,16 +1790,6 @@ fn decode_msgpack_or_json<T: DeserializeOwned>(binary: &[u8], json: &str, value_
}
}
fn read_version_stage_timer(attribution_enabled: bool) -> Option<Instant> {
attribution_enabled.then(Instant::now)
}
fn record_read_version_stage(stage: &'static str, started_at: Option<Instant>) {
if let Some(started_at) = started_at {
crate::cluster::rpc::runtime_sources::record_remote_disk_grpc_read_version_stage(stage, started_at.elapsed());
}
}
/// Aggregate encoded size (bytes) of a `ReadMultiple` response, preferring the msgpack payloads
/// and falling back to the JSON compatibility strings. Used to size the RPC for the payload
/// histogram / large-payload alerting (grpc-optimization P0 instrumentation).
@@ -2719,11 +2705,8 @@ impl DiskAPI for RemoteDisk {
state = "started",
"Remote disk RPC started"
);
let read_version_attribution_enabled = rustfs_io_metrics::get_stage_metrics_enabled();
let encode_started = read_version_stage_timer(read_version_attribution_enabled);
let encoded_opts = compat_json(opts).and_then(|opts_str| encode_msgpack(opts).map(|opts_bin| (opts_str, opts_bin)));
record_read_version_stage(INTERNODE_STAGE_READ_VERSION_REQUEST_ENCODE, encode_started);
let (opts_str, opts_bin) = encoded_opts?;
let opts_str = compat_json(opts)?;
let opts_bin = encode_msgpack(opts)?;
// Idempotent version read: eligible for the bounded transient-network retry so a single
// reset-by-peer during the read-after-write window does not erode the metadata read
@@ -2739,14 +2722,6 @@ impl DiskAPI for RemoteDisk {
.get_client()
.await
.map_err(|err| Error::other(format!("can not get client, err: {err}")))?;
let request_payload_bytes = read_version_attribution_enabled.then(|| {
disk.len()
.saturating_add(volume.len())
.saturating_add(path.len())
.saturating_add(version_id.len())
.saturating_add(opts_str.len())
.saturating_add(opts_bin.len())
});
let request = Request::new(ReadVersionRequest {
disk,
volume: volume.to_string(),
@@ -2756,47 +2731,14 @@ impl DiskAPI for RemoteDisk {
opts_bin: opts_bin.into(),
});
crate::cluster::rpc::runtime_sources::record_remote_disk_grpc_read_version_request();
if let Some(request_payload_bytes) = request_payload_bytes {
crate::cluster::rpc::runtime_sources::record_remote_disk_grpc_read_version_sent_bytes(request_payload_bytes);
}
let rpc_started = read_version_stage_timer(read_version_attribution_enabled);
let response = match client.read_version(request).await {
Ok(response) => {
record_read_version_stage(INTERNODE_STAGE_READ_VERSION_RPC_ROUNDTRIP, rpc_started);
response.into_inner()
}
Err(err) => {
record_read_version_stage(INTERNODE_STAGE_READ_VERSION_RPC_ROUNDTRIP, rpc_started);
crate::cluster::rpc::runtime_sources::record_remote_disk_grpc_read_version_error();
return Err(err.into());
}
};
let response = client.read_version(request).await?.into_inner();
if !response.success {
crate::cluster::rpc::runtime_sources::record_remote_disk_grpc_read_version_error();
return Err(response.error.unwrap_or_default().into());
}
crate::cluster::rpc::runtime_sources::record_remote_disk_grpc_read_version_recv_bytes(
response.file_info.len().saturating_add(response.file_info_bin.len()),
);
let decode_started = read_version_stage_timer(read_version_attribution_enabled);
let file_info = match decode_msgpack_or_json::<FileInfo>(&response.file_info_bin, &response.file_info, "FileInfo")
.and_then(|file_info| {
validate_decoded_file_info(&file_info)?;
Ok(file_info)
}) {
Ok(file_info) => {
record_read_version_stage(INTERNODE_STAGE_READ_VERSION_RESPONSE_DECODE, decode_started);
file_info
}
Err(err) => {
record_read_version_stage(INTERNODE_STAGE_READ_VERSION_RESPONSE_DECODE, decode_started);
crate::cluster::rpc::runtime_sources::record_remote_disk_grpc_read_version_error();
return Err(err);
}
};
let file_info = decode_msgpack_or_json::<FileInfo>(&response.file_info_bin, &response.file_info, "FileInfo")?;
validate_decoded_file_info(&file_info)?;
Ok(file_info)
},
@@ -7989,17 +7931,12 @@ mod tests {
}
#[tokio::test]
#[serial]
async fn read_version_uses_the_metadata_timeout_on_a_stalled_peer() {
runtime_sources::ensure_test_rpc_secret();
let Some((base_addr, accept_task)) = spawn_stalled_grpc_peer().await else {
return;
};
let remote_disk = remote_disk_for_addr(&base_addr).await;
let metrics = rustfs_io_metrics::internode_metrics::global_internode_metrics();
let previous_stage_metrics = rustfs_io_metrics::get_stage_metrics_enabled();
metrics.reset_for_test();
rustfs_io_metrics::set_get_stage_metrics_enabled(true);
temp_env::async_with_vars(
[
@@ -8023,18 +7960,6 @@ mod tests {
)
.await;
rustfs_io_metrics::set_get_stage_metrics_enabled(previous_stage_metrics);
let snapshot = metrics.snapshot();
assert!(
snapshot.outgoing_requests_total >= 1,
"ReadVersion call site should record outgoing attempts when attribution is enabled"
);
assert!(
snapshot.sent_bytes_total > 0,
"ReadVersion call site should record request payload bytes when attribution is enabled"
);
metrics.reset_for_test();
remote_disk.cancel_token.cancel();
accept_task.abort();
}
@@ -14,11 +14,10 @@
use rustfs_io_metrics::internode_metrics::{
INTERNODE_MSGPACK_CODEC_JSON, INTERNODE_MSGPACK_CODEC_MSGPACK, INTERNODE_MSGPACK_DIRECTION_RESPONSE,
INTERNODE_OPERATION_GRPC_READ_ALL, INTERNODE_OPERATION_GRPC_READ_MULTIPLE, INTERNODE_OPERATION_GRPC_READ_VERSION,
INTERNODE_OPERATION_GRPC_WRITE_ALL, INTERNODE_OPERATION_PUT_FILE_STREAM, INTERNODE_OPERATION_READ_FILE_STREAM,
INTERNODE_TRANSPORT_BACKEND_GRPC, INTERNODE_TRANSPORT_BACKEND_TCP_HTTP, global_internode_metrics,
INTERNODE_OPERATION_GRPC_READ_ALL, INTERNODE_OPERATION_GRPC_READ_MULTIPLE, INTERNODE_OPERATION_GRPC_WRITE_ALL,
INTERNODE_OPERATION_PUT_FILE_STREAM, INTERNODE_OPERATION_READ_FILE_STREAM, INTERNODE_TRANSPORT_BACKEND_GRPC,
INTERNODE_TRANSPORT_BACKEND_TCP_HTTP, global_internode_metrics,
};
use std::time::Duration;
#[cfg(test)]
use rustfs_io_metrics::internode_metrics::InternodeMetricsSnapshot;
@@ -83,59 +82,6 @@ pub(crate) fn record_remote_disk_grpc_read_all_request() {
.record_outgoing_request_for_operation_and_backend(INTERNODE_OPERATION_GRPC_READ_ALL, INTERNODE_TRANSPORT_BACKEND_GRPC);
}
pub(crate) fn record_remote_disk_grpc_read_version_request() {
if !rustfs_io_metrics::get_stage_metrics_enabled() {
return;
}
global_internode_metrics().record_outgoing_request_for_operation_and_backend(
INTERNODE_OPERATION_GRPC_READ_VERSION,
INTERNODE_TRANSPORT_BACKEND_GRPC,
);
}
pub(crate) fn record_remote_disk_grpc_read_version_error() {
if !rustfs_io_metrics::get_stage_metrics_enabled() {
return;
}
global_internode_metrics()
.record_error_for_operation_and_backend(INTERNODE_OPERATION_GRPC_READ_VERSION, INTERNODE_TRANSPORT_BACKEND_GRPC);
}
pub(crate) fn record_remote_disk_grpc_read_version_sent_bytes(bytes: usize) {
if !rustfs_io_metrics::get_stage_metrics_enabled() {
return;
}
global_internode_metrics().record_sent_bytes_for_operation_and_backend(
INTERNODE_OPERATION_GRPC_READ_VERSION,
INTERNODE_TRANSPORT_BACKEND_GRPC,
bytes,
);
}
pub(crate) fn record_remote_disk_grpc_read_version_recv_bytes(bytes: usize) {
if !rustfs_io_metrics::get_stage_metrics_enabled() {
return;
}
global_internode_metrics().record_recv_bytes_for_operation_and_backend(
INTERNODE_OPERATION_GRPC_READ_VERSION,
INTERNODE_TRANSPORT_BACKEND_GRPC,
bytes,
);
record_grpc_payload_size(INTERNODE_OPERATION_GRPC_READ_VERSION, bytes);
}
pub(crate) fn record_remote_disk_grpc_read_version_stage(stage: &'static str, duration: Duration) {
if !rustfs_io_metrics::get_stage_metrics_enabled() {
return;
}
global_internode_metrics().record_stage_duration_for_operation_and_backend(
INTERNODE_OPERATION_GRPC_READ_VERSION,
INTERNODE_TRANSPORT_BACKEND_GRPC,
stage,
duration,
);
}
pub(crate) fn record_remote_disk_grpc_read_all_recv_bytes(bytes: usize) {
global_internode_metrics().record_recv_bytes_for_operation_and_backend(
INTERNODE_OPERATION_GRPC_READ_ALL,
-8
View File
@@ -1098,10 +1098,6 @@ pub(crate) async fn sync_dir_files_with_limiter(dir: impl AsRef<Path>, disk_perm
let files = run_file_sync_blocking(disk_permits.clone(), move || {
let files = regular_files(&scan_dir)?;
if files.len() < PARALLEL_FILE_SYNC_THRESHOLD {
rustfs_io_metrics::record_put_rename_fdatasync_batch(
rustfs_io_metrics::PUT_RENAME_FDATASYNC_BATCH_MODE_SERIAL,
files.len(),
);
sync_files(&files)?;
let fsync_started = rustfs_io_metrics::put_stage_timer();
let result = fsync_dir_std(scan_dir);
@@ -1119,10 +1115,6 @@ pub(crate) async fn sync_dir_files_with_limiter(dir: impl AsRef<Path>, disk_perm
let Some(files) = files else {
return Ok(());
};
rustfs_io_metrics::record_put_rename_fdatasync_batch(
rustfs_io_metrics::PUT_RENAME_FDATASYNC_BATCH_MODE_PARALLEL,
files.len(),
);
futures::stream::iter(files.into_iter().map(Ok::<_, io::Error>))
.try_for_each_concurrent(MAX_PARALLEL_FILE_SYNCS, |path| {
let disk_permits = disk_permits.clone();
@@ -3417,25 +3417,6 @@ impl SetDisks {
quorum_wait_started,
);
let (results, mut file_infos) = fanout_result.map_err(|_| DiskError::Unexpected)?;
if rustfs_io_metrics::put_stage_metrics_enabled() {
let mut fanout_success = 0;
let mut fanout_error = 0;
let mut fanout_panic = 0;
for result in &results {
match result {
Ok(Ok(_)) => fanout_success += 1,
Ok(Err(_)) => fanout_error += 1,
Err(_) => fanout_panic += 1,
}
}
rustfs_io_metrics::record_put_rename_quorum_wait_fanout(
results.len(),
write_quorum,
fanout_success,
fanout_error,
fanout_panic,
);
}
for (idx, result) in results.iter().enumerate() {
match result {
+35 -1
View File
@@ -759,7 +759,7 @@ impl HealChannelProcessor {
#[cfg(test)]
mod tests {
use super::super::DiskStore;
use super::super::{DiskStore, Endpoint};
use super::*;
use crate::heal::manager::HealConfig;
use crate::heal::storage::{HealObjectInfo, HealStorageAPI};
@@ -776,18 +776,45 @@ mod tests {
async fn get_object_meta(&self, _bucket: &str, _object: &str) -> crate::Result<Option<HealObjectInfo>> {
Ok(None)
}
async fn get_object_data(&self, _bucket: &str, _object: &str) -> crate::Result<Option<Vec<u8>>> {
Ok(None)
}
async fn put_object_data(&self, _bucket: &str, _object: &str, _data: &[u8]) -> crate::Result<()> {
Ok(())
}
async fn delete_object(&self, _bucket: &str, _object: &str) -> crate::Result<()> {
Ok(())
}
async fn verify_object_integrity(&self, _bucket: &str, _object: &str) -> crate::Result<bool> {
Ok(true)
}
async fn ec_decode_rebuild(&self, _bucket: &str, _object: &str) -> crate::Result<Vec<u8>> {
Ok(vec![])
}
async fn get_disk_status(&self, _endpoint: &Endpoint) -> crate::Result<crate::heal::storage::DiskStatus> {
Ok(crate::heal::storage::DiskStatus::Ok)
}
async fn format_disk(&self, _endpoint: &Endpoint) -> crate::Result<()> {
Ok(())
}
async fn get_bucket_info(&self, _bucket: &str) -> crate::Result<Option<crate::heal::storage_api::status::BucketInfo>> {
Ok(None)
}
async fn heal_bucket_metadata(&self, _bucket: &str) -> crate::Result<()> {
Ok(())
}
async fn list_buckets(&self) -> crate::Result<Vec<crate::heal::storage_api::status::BucketInfo>> {
Ok(vec![])
}
async fn object_exists(&self, _bucket: &str, _object: &str) -> crate::Result<bool> {
Ok(false)
}
async fn get_object_size(&self, _bucket: &str, _object: &str) -> crate::Result<Option<u64>> {
Ok(None)
}
async fn get_object_checksum(&self, _bucket: &str, _object: &str) -> crate::Result<Option<String>> {
Ok(None)
}
async fn heal_object(
&self,
_bucket: &str,
@@ -810,6 +837,13 @@ mod tests {
) -> crate::Result<(rustfs_madmin::heal_commands::HealResultItem, Option<crate::Error>)> {
Ok((rustfs_madmin::heal_commands::HealResultItem::default(), None))
}
async fn list_objects_for_heal(
&self,
_bucket: &str,
_prefix: &str,
) -> crate::Result<Vec<crate::heal::storage::HealListItem>> {
Ok(vec![])
}
async fn list_objects_for_heal_page(
&self,
_bucket: &str,
+31 -1
View File
@@ -1267,7 +1267,7 @@ mod resume_loop_tests {
CheckpointManager, RESUME_CHECKPOINT_FILE, ReplacementTargetIdentity, ResumeDeleteFailure, ResumeManager, ResumeUtils,
compose_key,
};
use crate::heal::storage::{HealLifecycleExpiryContext, HealListItem, HealObjectInfo, HealStorageAPI};
use crate::heal::storage::{DiskStatus, HealLifecycleExpiryContext, HealListItem, HealObjectInfo, HealStorageAPI};
use crate::heal::storage_api::status::BucketInfo;
use crate::heal::{
BUCKET_META_PREFIX, DiskOption, DiskStore, EcstoreError, Endpoint, HealDiskExt as _, RUSTFS_META_BUCKET, new_disk,
@@ -1448,15 +1448,36 @@ mod resume_loop_tests {
async fn get_object_meta(&self, _b: &str, _o: &str) -> Result<Option<HealObjectInfo>> {
Ok(None)
}
async fn get_object_data(&self, _b: &str, _o: &str) -> Result<Option<Vec<u8>>> {
Ok(None)
}
async fn put_object_data(&self, _b: &str, _o: &str, _d: &[u8]) -> Result<()> {
Ok(())
}
async fn delete_object(&self, _b: &str, _o: &str) -> Result<()> {
Ok(())
}
async fn verify_object_integrity(&self, _b: &str, _o: &str) -> Result<bool> {
Ok(true)
}
async fn ec_decode_rebuild(&self, _b: &str, _o: &str) -> Result<Vec<u8>> {
Ok(Vec::new())
}
async fn get_disk_status(&self, _e: &Endpoint) -> Result<DiskStatus> {
Ok(DiskStatus::Ok)
}
async fn format_disk(&self, _e: &Endpoint) -> Result<()> {
Ok(())
}
async fn get_bucket_info(&self, bucket: &str) -> Result<Option<BucketInfo>> {
Ok(Some(BucketInfo {
name: bucket.to_string(),
..Default::default()
}))
}
async fn heal_bucket_metadata(&self, _b: &str) -> Result<()> {
Ok(())
}
async fn list_buckets(&self) -> Result<Vec<BucketInfo>> {
Ok(Vec::new())
}
@@ -1464,6 +1485,12 @@ mod resume_loop_tests {
// Must never be consulted: the resume loop always goes through heal_object.
panic!("object_exists must not be called by the resume heal loop");
}
async fn get_object_size(&self, _b: &str, _o: &str) -> Result<Option<u64>> {
Ok(None)
}
async fn get_object_checksum(&self, _b: &str, _o: &str) -> Result<Option<String>> {
Ok(None)
}
async fn load_heal_lifecycle_expiry_context(&self, _bucket: &str) -> Result<Option<HealLifecycleExpiryContext>> {
Ok((!self.lifecycle_expired.lock().unwrap().is_empty()).then(HealLifecycleExpiryContext::test))
}
@@ -1529,6 +1556,9 @@ mod resume_loop_tests {
ReplacementCommitEvidence::Error(message) => Err(Error::other(message)),
}
}
async fn list_objects_for_heal(&self, _b: &str, _p: &str) -> Result<Vec<HealListItem>> {
Ok(Vec::new())
}
async fn list_objects_for_heal_page(
&self,
_bucket: &str,
+40
View File
@@ -3882,14 +3882,42 @@ mod tests {
Ok(None)
}
async fn get_object_data(&self, _bucket: &str, _object: &str) -> Result<Option<Vec<u8>>> {
Ok(None)
}
async fn put_object_data(&self, _bucket: &str, _object: &str, _data: &[u8]) -> Result<()> {
Ok(())
}
async fn delete_object(&self, _bucket: &str, _object: &str) -> Result<()> {
Ok(())
}
async fn verify_object_integrity(&self, _bucket: &str, _object: &str) -> Result<bool> {
Ok(true)
}
async fn ec_decode_rebuild(&self, _bucket: &str, _object: &str) -> Result<Vec<u8>> {
Ok(Vec::new())
}
async fn get_disk_status(&self, _endpoint: &Endpoint) -> Result<crate::heal::storage::DiskStatus> {
Ok(crate::heal::storage::DiskStatus::Ok)
}
async fn format_disk(&self, _endpoint: &Endpoint) -> Result<()> {
Ok(())
}
async fn get_bucket_info(&self, _bucket: &str) -> Result<Option<BucketInfo>> {
Ok(None)
}
async fn heal_bucket_metadata(&self, _bucket: &str) -> Result<()> {
Ok(())
}
async fn list_buckets(&self) -> Result<Vec<BucketInfo>> {
if let Some(hook) = manager_recovery_test_hook() {
*hook.listed.lock().expect("manager recovery listed lock should not poison") = true;
@@ -3901,6 +3929,14 @@ mod tests {
Ok(bucket == "retry-transition")
}
async fn get_object_size(&self, _bucket: &str, _object: &str) -> Result<Option<u64>> {
Ok(None)
}
async fn get_object_checksum(&self, _bucket: &str, _object: &str) -> Result<Option<String>> {
Ok(None)
}
async fn heal_object(
&self,
bucket: &str,
@@ -3962,6 +3998,10 @@ mod tests {
Ok((HealResultItem::default(), None))
}
async fn list_objects_for_heal(&self, _bucket: &str, _prefix: &str) -> Result<Vec<crate::heal::storage::HealListItem>> {
Ok(Vec::new())
}
async fn list_objects_for_heal_page(
&self,
_bucket: &str,
+542 -73
View File
@@ -27,7 +27,7 @@ use super::storage_api::storage::{
BucketInfo, BucketOperations, DiskSetSelector, HealOperations as _, ListOperations as _, ObjectIO as _,
ObjectOperations as _, StorageAdminApi,
};
use super::{DiskStore, ECStore, HealDiskExt as _, StorageError, resume::ReplacementTargetIdentity};
use super::{DiskStore, ECStore, Endpoint, HealDiskExt as _, StorageError, resume::ReplacementTargetIdentity};
pub use super::{HealObjectInfo, HealObjectOptions, HealPutObjReader};
#[derive(Debug, Clone, Copy, Default, PartialEq, Eq)]
@@ -65,6 +65,7 @@ const LOG_COMPONENT_HEAL: &str = "heal";
const LOG_SUBSYSTEM_STORAGE: &str = "storage";
const EVENT_HEAL_STORAGE_OBJECT_IO: &str = "heal_storage_object_io";
const EVENT_HEAL_STORAGE_OBJECT_READ_LIMIT: &str = "heal_storage_object_read_limit";
const EVENT_HEAL_STORAGE_OBJECT_VERIFY: &str = "heal_storage_object_verify";
const EVENT_HEAL_STORAGE_ADMIN_OP: &str = "heal_storage_admin_op";
const EVENT_HEAL_STORAGE_REPAIR_OP: &str = "heal_storage_repair_op";
@@ -311,23 +312,56 @@ pub struct HealListItem {
pub is_delete_marker: bool,
}
/// Disk status for heal operations
#[derive(Debug, Clone, PartialEq, Eq)]
pub enum DiskStatus {
/// Ok
Ok,
/// Offline
Offline,
/// Corrupt
Corrupt,
/// Missing
Missing,
/// Permission denied
PermissionDenied,
/// Faulty
Faulty,
/// Root mount
RootMount,
/// Unknown
Unknown,
/// Unformatted
Unformatted,
}
/// Heal storage layer interface
#[async_trait]
pub trait HealStorageAPI: Send + Sync {
/// Get object meta
///
/// Reserved for HS-01 MRF wiring (rustfs/backlog#1865): MRF intents
/// currently execute through `heal_object`; keep this entry point for the
/// metadata-corruption variant that must inspect metadata first.
async fn get_object_meta(&self, bucket: &str, object: &str) -> Result<Option<HealObjectInfo>>;
/// Get object data
async fn get_object_data(&self, bucket: &str, object: &str) -> Result<Option<Vec<u8>>>;
/// Put object data
async fn put_object_data(&self, bucket: &str, object: &str, data: &[u8]) -> Result<()>;
/// Delete object
async fn delete_object(&self, bucket: &str, object: &str) -> Result<()>;
/// Check object integrity
async fn verify_object_integrity(&self, bucket: &str, object: &str) -> Result<bool>;
/// EC decode rebuild
///
/// Reserved for HS-01 MRF wiring (rustfs/backlog#1865): urgent ECDecode
/// requests currently execute through `heal_object`; keep the explicit
/// rebuild-and-read path for the decode-failure fast variant.
async fn ec_decode_rebuild(&self, bucket: &str, object: &str) -> Result<Vec<u8>>;
/// Get disk status
async fn get_disk_status(&self, endpoint: &Endpoint) -> Result<DiskStatus>;
/// Format disk
async fn format_disk(&self, endpoint: &Endpoint) -> Result<()>;
/// Get bucket info
async fn get_bucket_info(&self, bucket: &str) -> Result<Option<BucketInfo>>;
@@ -353,12 +387,21 @@ pub trait HealStorageAPI: Send + Sync {
Ok(false)
}
/// Fix bucket metadata
async fn heal_bucket_metadata(&self, bucket: &str) -> Result<()>;
/// Get all buckets
async fn list_buckets(&self) -> Result<Vec<BucketInfo>>;
/// Check object exists
async fn object_exists(&self, bucket: &str, object: &str) -> Result<bool>;
/// Get object size
async fn get_object_size(&self, bucket: &str, object: &str) -> Result<Option<u64>>;
/// Get object checksum
async fn get_object_checksum(&self, bucket: &str, object: &str) -> Result<Option<String>>;
/// Heal object using ecstore
async fn heal_object(
&self,
@@ -410,6 +453,12 @@ pub trait HealStorageAPI: Send + Sync {
Ok(false)
}
/// List object versions for healing (returns all versions, may use significant memory for large buckets)
///
/// WARNING: This method loads all object versions into memory at once. For buckets with many
/// objects/versions, consider using `list_objects_for_heal_page` instead to process versions in pages.
async fn list_objects_for_heal(&self, bucket: &str, prefix: &str) -> Result<Vec<HealListItem>>;
/// List object versions for healing with pagination (returns one page and continuation token)
/// Returns (versions, next_continuation_token, is_truncated). The continuation token is an
/// opaque composite `(marker, version_marker)` value — see `encode_heal_token`/`decode_heal_token`.
@@ -478,11 +527,89 @@ impl ECStoreHealStorage {
pub fn new(ecstore: Arc<ECStore>) -> Self {
Self { ecstore }
}
}
fn is_transient_object_exists_message(message: &str) -> bool {
let message = message.to_ascii_lowercase();
[
"failed to acquire read lock",
"lock acquisition failed",
"lock acquisition timeout",
"quorum not reached",
"deadline has elapsed",
"timed out",
"network error",
"transport error",
"connection refused",
]
.iter()
.any(|pattern| message.contains(pattern))
}
fn is_transient_object_exists_error(err: &StorageError) -> bool {
if err.is_quorum_error() {
return true;
}
match err {
StorageError::Lock(lock_err) => lock_err.is_retryable() || is_transient_object_exists_message(&lock_err.to_string()),
StorageError::Io(io_err) => is_transient_object_exists_message(&io_err.to_string()),
StorageError::SlowDown | StorageError::OperationCanceled => true,
_ => false,
}
}
#[async_trait]
impl HealStorageAPI for ECStoreHealStorage {
async fn get_object_meta(&self, bucket: &str, object: &str) -> Result<Option<HealObjectInfo>> {
debug!(
target: "rustfs::heal::storage",
event = EVENT_HEAL_STORAGE_OBJECT_IO,
component = LOG_COMPONENT_HEAL,
subsystem = LOG_SUBSYSTEM_STORAGE,
operation = "get_object_meta",
bucket,
object,
"Heal storage request started"
);
match self.ecstore.get_object_info(bucket, object, &Default::default()).await {
Ok(info) => Ok(Some(info)),
Err(e) => {
// Map ObjectNotFound to None to align with Option return type
if matches!(e, StorageError::ObjectNotFound(_, _)) {
debug!(
target: "rustfs::heal::storage",
event = EVENT_HEAL_STORAGE_OBJECT_IO,
component = LOG_COMPONENT_HEAL,
subsystem = LOG_SUBSYSTEM_STORAGE,
operation = "get_object_meta",
bucket,
object,
result = "not_found",
"Heal storage object metadata missing"
);
Ok(None)
} else {
error!(
target: "rustfs::heal::storage",
event = EVENT_HEAL_STORAGE_OBJECT_IO,
component = LOG_COMPONENT_HEAL,
subsystem = LOG_SUBSYSTEM_STORAGE,
operation = "get_object_meta",
bucket,
object,
result = "failed",
error = %e,
"Heal storage request failed"
);
Err(Error::other(e))
}
}
}
}
/// Read back an object's bytes, capped to bound memory.
///
/// Private support for the reserved `ec_decode_rebuild` (HS-01); not part
/// of the storage trait surface.
async fn get_object_data(&self, bucket: &str, object: &str) -> Result<Option<Vec<u8>>> {
debug!(
target: "rustfs::heal::storage",
@@ -568,85 +695,196 @@ impl ECStoreHealStorage {
}
Ok(Some(buf))
}
}
fn is_transient_object_exists_message(message: &str) -> bool {
let message = message.to_ascii_lowercase();
[
"failed to acquire read lock",
"lock acquisition failed",
"lock acquisition timeout",
"quorum not reached",
"deadline has elapsed",
"timed out",
"network error",
"transport error",
"connection refused",
]
.iter()
.any(|pattern| message.contains(pattern))
}
fn is_transient_object_exists_error(err: &StorageError) -> bool {
if err.is_quorum_error() {
return true;
}
match err {
StorageError::Lock(lock_err) => lock_err.is_retryable() || is_transient_object_exists_message(&lock_err.to_string()),
StorageError::Io(io_err) => is_transient_object_exists_message(&io_err.to_string()),
StorageError::SlowDown | StorageError::OperationCanceled => true,
_ => false,
}
}
#[async_trait]
impl HealStorageAPI for ECStoreHealStorage {
async fn get_object_meta(&self, bucket: &str, object: &str) -> Result<Option<HealObjectInfo>> {
async fn put_object_data(&self, bucket: &str, object: &str, data: &[u8]) -> Result<()> {
debug!(
target: "rustfs::heal::storage",
event = EVENT_HEAL_STORAGE_OBJECT_IO,
component = LOG_COMPONENT_HEAL,
subsystem = LOG_SUBSYSTEM_STORAGE,
operation = "get_object_meta",
operation = "put_object_data",
bucket,
object,
bytes = data.len(),
"Heal storage request started"
);
let mut reader = HealPutObjReader::from_vec(data.to_vec());
match (*self.ecstore)
.put_object(bucket, object, &mut reader, &Default::default())
.await
{
Ok(_) => {
debug!(
target: "rustfs::heal::storage",
event = EVENT_HEAL_STORAGE_OBJECT_IO,
component = LOG_COMPONENT_HEAL,
subsystem = LOG_SUBSYSTEM_STORAGE,
operation = "put_object_data",
bucket,
object,
result = "ok",
"Heal storage object write completed"
);
Ok(())
}
Err(e) => {
error!(
target: "rustfs::heal::storage",
event = EVENT_HEAL_STORAGE_OBJECT_IO,
component = LOG_COMPONENT_HEAL,
subsystem = LOG_SUBSYSTEM_STORAGE,
operation = "put_object_data",
bucket,
object,
result = "failed",
error = %e,
"Heal storage request failed"
);
Err(Error::other(e))
}
}
}
async fn delete_object(&self, bucket: &str, object: &str) -> Result<()> {
debug!(
target: "rustfs::heal::storage",
event = EVENT_HEAL_STORAGE_OBJECT_IO,
component = LOG_COMPONENT_HEAL,
subsystem = LOG_SUBSYSTEM_STORAGE,
operation = "delete_object",
bucket,
object,
"Heal storage request started"
);
match self.ecstore.get_object_info(bucket, object, &Default::default()).await {
Ok(info) => Ok(Some(info)),
match self.ecstore.delete_object(bucket, object, Default::default()).await {
Ok(_) => {
debug!(
target: "rustfs::heal::storage",
event = EVENT_HEAL_STORAGE_OBJECT_IO,
component = LOG_COMPONENT_HEAL,
subsystem = LOG_SUBSYSTEM_STORAGE,
operation = "delete_object",
bucket,
object,
result = "ok",
"Heal storage object delete completed"
);
Ok(())
}
Err(e) => {
// Map ObjectNotFound to None to align with Option return type
if matches!(e, StorageError::ObjectNotFound(_, _)) {
debug!(
error!(
target: "rustfs::heal::storage",
event = EVENT_HEAL_STORAGE_OBJECT_IO,
component = LOG_COMPONENT_HEAL,
subsystem = LOG_SUBSYSTEM_STORAGE,
operation = "delete_object",
bucket,
object,
result = "failed",
error = %e,
"Heal storage request failed"
);
Err(Error::other(e))
}
}
}
async fn verify_object_integrity(&self, bucket: &str, object: &str) -> Result<bool> {
debug!(
target: "rustfs::heal::storage",
event = EVENT_HEAL_STORAGE_OBJECT_VERIFY,
component = LOG_COMPONENT_HEAL,
subsystem = LOG_SUBSYSTEM_STORAGE,
bucket,
object,
state = "started",
"Heal storage object verification started"
);
// Check object metadata first
match self.get_object_meta(bucket, object).await? {
Some(obj_info) => {
if obj_info.size < 0 {
warn!(
target: "rustfs::heal::storage",
event = EVENT_HEAL_STORAGE_OBJECT_IO,
event = EVENT_HEAL_STORAGE_OBJECT_VERIFY,
component = LOG_COMPONENT_HEAL,
subsystem = LOG_SUBSYSTEM_STORAGE,
operation = "get_object_meta",
bucket,
object,
result = "not_found",
"Heal storage object metadata missing"
state = "invalid_size",
"Heal storage object verification failed"
);
Ok(None)
} else {
error!(
target: "rustfs::heal::storage",
event = EVENT_HEAL_STORAGE_OBJECT_IO,
component = LOG_COMPONENT_HEAL,
subsystem = LOG_SUBSYSTEM_STORAGE,
operation = "get_object_meta",
bucket,
object,
result = "failed",
error = %e,
"Heal storage request failed"
);
Err(Error::other(e))
return Ok(false);
}
// Stream-read the object to a sink to avoid loading into memory
match (*self.ecstore)
.get_object_reader(bucket, object, None, Default::default(), &Default::default())
.await
{
Ok(reader) => {
let mut stream = reader.stream;
match tokio::io::copy(&mut stream, &mut tokio::io::sink()).await {
Ok(_) => {
debug!(
target: "rustfs::heal::storage",
event = EVENT_HEAL_STORAGE_OBJECT_VERIFY,
component = LOG_COMPONENT_HEAL,
subsystem = LOG_SUBSYSTEM_STORAGE,
bucket,
object,
state = "ok",
"Heal storage object verified"
);
Ok(true)
}
Err(e) => {
warn!(
target: "rustfs::heal::storage",
event = EVENT_HEAL_STORAGE_OBJECT_VERIFY,
component = LOG_COMPONENT_HEAL,
subsystem = LOG_SUBSYSTEM_STORAGE,
bucket,
object,
state = "stream_read_failed",
error = %e,
"Heal storage object verification failed"
);
Ok(false)
}
}
}
Err(e) => {
warn!(
target: "rustfs::heal::storage",
event = EVENT_HEAL_STORAGE_OBJECT_VERIFY,
component = LOG_COMPONENT_HEAL,
subsystem = LOG_SUBSYSTEM_STORAGE,
bucket,
object,
state = "reader_open_failed",
error = %e,
"Heal storage object verification failed"
);
Ok(false)
}
}
}
None => {
warn!(
target: "rustfs::heal::storage",
event = EVENT_HEAL_STORAGE_OBJECT_VERIFY,
component = LOG_COMPONENT_HEAL,
subsystem = LOG_SUBSYSTEM_STORAGE,
bucket,
object,
state = "metadata_missing",
"Heal storage object verification failed"
);
Ok(false)
}
}
}
@@ -738,6 +976,81 @@ impl HealStorageAPI for ECStoreHealStorage {
}
}
async fn get_disk_status(&self, endpoint: &Endpoint) -> Result<DiskStatus> {
debug!(
target: "rustfs::heal::storage",
event = EVENT_HEAL_STORAGE_ADMIN_OP,
component = LOG_COMPONENT_HEAL,
subsystem = LOG_SUBSYSTEM_STORAGE,
operation = "get_disk_status",
endpoint = ?endpoint,
state = "started",
"Heal storage admin operation started"
);
// TODO: implement disk status check using ecstore
// For now, return Ok status
debug!(
target: "rustfs::heal::storage",
event = EVENT_HEAL_STORAGE_ADMIN_OP,
component = LOG_COMPONENT_HEAL,
subsystem = LOG_SUBSYSTEM_STORAGE,
operation = "get_disk_status",
endpoint = ?endpoint,
result = "ok",
disk_status = "ok",
"Heal storage disk status resolved"
);
Ok(DiskStatus::Ok)
}
async fn format_disk(&self, endpoint: &Endpoint) -> Result<()> {
debug!(
target: "rustfs::heal::storage",
event = EVENT_HEAL_STORAGE_ADMIN_OP,
component = LOG_COMPONENT_HEAL,
subsystem = LOG_SUBSYSTEM_STORAGE,
operation = "format_disk",
endpoint = ?endpoint,
state = "started",
"Heal storage admin operation started"
);
// Use ecstore's heal_format
match self.heal_format(false).await {
Ok((_, error)) => {
if error.is_some() {
return Err(Error::other(format!("Format failed: {error:?}")));
}
debug!(
target: "rustfs::heal::storage",
event = EVENT_HEAL_STORAGE_ADMIN_OP,
component = LOG_COMPONENT_HEAL,
subsystem = LOG_SUBSYSTEM_STORAGE,
operation = "format_disk",
endpoint = ?endpoint,
result = "ok",
"Heal storage disk format completed"
);
Ok(())
}
Err(e) => {
error!(
target: "rustfs::heal::storage",
event = EVENT_HEAL_STORAGE_ADMIN_OP,
component = LOG_COMPONENT_HEAL,
subsystem = LOG_SUBSYSTEM_STORAGE,
operation = "format_disk",
endpoint = ?endpoint,
result = "failed",
error = %e,
"Heal storage admin operation failed"
);
Err(e)
}
}
}
async fn get_bucket_info(&self, bucket: &str) -> Result<Option<BucketInfo>> {
debug!(
target: "rustfs::heal::storage",
@@ -848,6 +1161,61 @@ impl HealStorageAPI for ECStoreHealStorage {
}
}
async fn heal_bucket_metadata(&self, bucket: &str) -> Result<()> {
debug!(
target: "rustfs::heal::storage",
event = EVENT_HEAL_STORAGE_REPAIR_OP,
component = LOG_COMPONENT_HEAL,
subsystem = LOG_SUBSYSTEM_STORAGE,
operation = "heal_bucket_metadata",
bucket,
state = "started",
"Heal storage repair started"
);
let heal_opts = HealOpts {
recursive: true,
dry_run: false,
remove: false,
recreate: false,
scan_mode: HealScanMode::Normal,
update_parity: false,
no_lock: false,
pool: None,
set: None,
};
match self.heal_bucket(bucket, &heal_opts).await {
Ok(_) => {
debug!(
target: "rustfs::heal::storage",
event = EVENT_HEAL_STORAGE_REPAIR_OP,
component = LOG_COMPONENT_HEAL,
subsystem = LOG_SUBSYSTEM_STORAGE,
operation = "heal_bucket_metadata",
bucket,
result = "ok",
"Heal storage bucket metadata repaired"
);
Ok(())
}
Err(e) => {
error!(
target: "rustfs::heal::storage",
event = EVENT_HEAL_STORAGE_REPAIR_OP,
component = LOG_COMPONENT_HEAL,
subsystem = LOG_SUBSYSTEM_STORAGE,
operation = "heal_bucket_metadata",
bucket,
result = "failed",
error = %e,
"Heal storage repair failed"
);
Err(e)
}
}
}
async fn list_buckets(&self) -> Result<Vec<BucketInfo>> {
debug!(
target: "rustfs::heal::storage",
@@ -947,6 +1315,48 @@ impl HealStorageAPI for ECStoreHealStorage {
}
}
async fn get_object_size(&self, bucket: &str, object: &str) -> Result<Option<u64>> {
debug!(
target: "rustfs::heal::storage",
event = EVENT_HEAL_STORAGE_OBJECT_IO,
component = LOG_COMPONENT_HEAL,
subsystem = LOG_SUBSYSTEM_STORAGE,
operation = "get_object_size",
bucket,
object,
"Heal storage request started"
);
match self.get_object_meta(bucket, object).await {
Ok(Some(obj_info)) => Ok(Some(obj_info.size as u64)),
Ok(None) => Ok(None),
Err(e) => Err(e),
}
}
async fn get_object_checksum(&self, bucket: &str, object: &str) -> Result<Option<String>> {
debug!(
target: "rustfs::heal::storage",
event = EVENT_HEAL_STORAGE_OBJECT_IO,
component = LOG_COMPONENT_HEAL,
subsystem = LOG_SUBSYSTEM_STORAGE,
operation = "get_object_checksum",
bucket,
object,
"Heal storage request started"
);
match self.get_object_meta(bucket, object).await {
Ok(Some(obj_info)) => {
// Convert checksum bytes to hex string
let checksum = obj_info.checksum.iter().map(|b| format!("{b:02x}")).collect::<String>();
Ok(Some(checksum))
}
Ok(None) => Ok(None),
Err(e) => Err(e),
}
}
async fn heal_object(
&self,
bucket: &str,
@@ -1137,6 +1547,65 @@ impl HealStorageAPI for ECStoreHealStorage {
.map_err(Error::Storage)
}
async fn list_objects_for_heal(&self, bucket: &str, prefix: &str) -> Result<Vec<HealListItem>> {
debug!(
target: "rustfs::heal::storage",
event = EVENT_HEAL_STORAGE_ADMIN_OP,
component = LOG_COMPONENT_HEAL,
subsystem = LOG_SUBSYSTEM_STORAGE,
operation = "list_objects_for_heal",
bucket,
prefix,
state = "started",
"Heal storage admin operation started"
);
warn!(
target: "rustfs::heal::storage",
event = EVENT_HEAL_STORAGE_ADMIN_OP,
component = LOG_COMPONENT_HEAL,
subsystem = LOG_SUBSYSTEM_STORAGE,
operation = "list_objects_for_heal",
bucket,
prefix,
state = "memory_heavy",
"Heal storage version listing loads all versions into memory (footprint is per-version, not per-object)"
);
let mut all_objects: Vec<HealListItem> = Vec::new();
let mut continuation_token: Option<String> = None;
loop {
let (page_objects, next_token, is_truncated) = self
.list_objects_for_heal_page(bucket, prefix, continuation_token.as_deref(), false)
.await?;
all_objects.extend(page_objects);
if !is_truncated {
break;
}
continuation_token = next_heal_listing_token(bucket, prefix, next_token, is_truncated)?;
if continuation_token.is_none() {
break;
}
}
debug!(
target: "rustfs::heal::storage",
event = EVENT_HEAL_STORAGE_ADMIN_OP,
component = LOG_COMPONENT_HEAL,
subsystem = LOG_SUBSYSTEM_STORAGE,
operation = "list_objects_for_heal",
bucket,
prefix,
object_count = all_objects.len(),
result = "ok",
"Heal storage object listing completed"
);
Ok(all_objects)
}
async fn list_objects_for_heal_page(
&self,
bucket: &str,
+45 -2
View File
@@ -2822,7 +2822,7 @@ impl std::fmt::Debug for HealTask {
mod tests {
use super::super::{DiskOption, DiskStore, Endpoint, HealDiskExt as _, new_disk};
use super::*;
use crate::heal::storage::{HealListItem, HealObjectInfo};
use crate::heal::storage::{DiskStatus, HealListItem, HealObjectInfo};
use rustfs_common::trace_bus::{TraceEvent, TraceFunc, TraceKind, TraceSubscription, TraceVal, subscribe_trace_events};
use rustfs_madmin::heal_commands::{HealDriveInfo, HealResultItem, Infos};
use std::collections::{HashMap, VecDeque};
@@ -3354,6 +3354,7 @@ mod tests {
object_exists_by_name: Mutex<HashMap<String, MockObjectExists>>,
heal_object_outcome: Mutex<Option<MockHealObjectOutcome>>,
heal_object_outcomes: Mutex<HashMap<String, VecDeque<MockHealObjectOutcome>>>,
deleted_objects: Mutex<Vec<String>>,
format_no_heal_required: Mutex<bool>,
global_format_calls: Mutex<u32>,
replacement_format_calls: Mutex<Vec<(usize, usize, Vec<String>)>>,
@@ -3546,10 +3547,35 @@ mod tests {
Ok(None)
}
async fn get_object_data(&self, _bucket: &str, _object: &str) -> Result<Option<Vec<u8>>> {
Ok(None)
}
async fn put_object_data(&self, _bucket: &str, _object: &str, _data: &[u8]) -> Result<()> {
Ok(())
}
async fn delete_object(&self, _bucket: &str, object: &str) -> Result<()> {
self.deleted_objects.lock().unwrap().push(object.to_string());
Ok(())
}
async fn verify_object_integrity(&self, _bucket: &str, _object: &str) -> Result<bool> {
Ok(true)
}
async fn ec_decode_rebuild(&self, _bucket: &str, _object: &str) -> Result<Vec<u8>> {
Ok(Vec::new())
}
async fn get_disk_status(&self, _endpoint: &Endpoint) -> Result<DiskStatus> {
Ok(DiskStatus::Ok)
}
async fn format_disk(&self, _endpoint: &Endpoint) -> Result<()> {
Ok(())
}
async fn get_bucket_info(&self, bucket: &str) -> Result<Option<BucketInfo>> {
Ok(Some(BucketInfo {
name: bucket.to_string(),
@@ -3564,6 +3590,10 @@ mod tests {
Ok(*self.usage_baseline.lock().unwrap())
}
async fn heal_bucket_metadata(&self, _bucket: &str) -> Result<()> {
Ok(())
}
async fn list_buckets(&self) -> Result<Vec<BucketInfo>> {
let buckets = self
.listed_buckets
@@ -3591,6 +3621,14 @@ mod tests {
Ok(self.object_exists.lock().unwrap().unwrap_or(true))
}
async fn get_object_size(&self, _bucket: &str, _object: &str) -> Result<Option<u64>> {
Ok(None)
}
async fn get_object_checksum(&self, _bucket: &str, _object: &str) -> Result<Option<String>> {
Ok(None)
}
async fn heal_object(
&self,
bucket: &str,
@@ -3726,6 +3764,10 @@ mod tests {
Ok(*self.replacement_targets_ready.lock().unwrap())
}
async fn list_objects_for_heal(&self, _bucket: &str, _prefix: &str) -> Result<Vec<HealListItem>> {
Ok(vec![heal_item("object-a"), heal_item("object-b")])
}
async fn list_objects_for_heal_page(
&self,
bucket: &str,
@@ -4816,7 +4858,7 @@ mod tests {
}
#[tokio::test]
async fn test_heal_failure_with_remove_corrupted_propagates_remove_flag() {
async fn test_heal_failure_with_remove_corrupted_does_not_delete_object() {
let storage = Arc::new(MockStorage {
object_exists: Mutex::new(Some(true)),
heal_object_outcome: Mutex::new(Some(MockHealObjectOutcome::OkWithOtherError(
@@ -4842,6 +4884,7 @@ mod tests {
let err = task.execute().await.expect_err("heal failure should still be reported");
assert!(matches!(err, Error::TaskExecutionFailed { .. }));
assert!(storage.deleted_objects.lock().unwrap().is_empty());
assert!(storage.object_heal_opts.lock().unwrap()[0].remove);
}
+42 -2
View File
@@ -352,8 +352,8 @@ pub(crate) fn set_heal_queue_length(count: usize) {
mod tests {
use super::{
Error, HEAL_RUNTIME_INIT_TEST_HOOK, HealRuntimeInitTestHook, get_heal_channel_processor, get_heal_manager,
heal::DiskStore, heal::manager::HealConfig, heal::storage::HealListItem, heal::storage::HealObjectInfo,
heal::storage::HealStorageAPI, init_heal_manager, run_owned_initialization,
heal::DiskStore, heal::Endpoint, heal::manager::HealConfig, heal::storage::DiskStatus, heal::storage::HealListItem,
heal::storage::HealObjectInfo, heal::storage::HealStorageAPI, init_heal_manager, run_owned_initialization,
};
use crate::heal::storage_api::status::BucketInfo;
use rustfs_common::heal_channel::HealOpts;
@@ -370,14 +370,42 @@ mod tests {
Ok(None)
}
async fn get_object_data(&self, _bucket: &str, _object: &str) -> Result<Option<Vec<u8>>, Error> {
Ok(None)
}
async fn put_object_data(&self, _bucket: &str, _object: &str, _data: &[u8]) -> Result<(), Error> {
Ok(())
}
async fn delete_object(&self, _bucket: &str, _object: &str) -> Result<(), Error> {
Ok(())
}
async fn verify_object_integrity(&self, _bucket: &str, _object: &str) -> Result<bool, Error> {
Ok(true)
}
async fn ec_decode_rebuild(&self, _bucket: &str, _object: &str) -> Result<Vec<u8>, Error> {
Ok(Vec::new())
}
async fn get_disk_status(&self, _endpoint: &Endpoint) -> Result<DiskStatus, Error> {
Ok(DiskStatus::Ok)
}
async fn format_disk(&self, _endpoint: &Endpoint) -> Result<(), Error> {
Ok(())
}
async fn get_bucket_info(&self, _bucket: &str) -> Result<Option<BucketInfo>, Error> {
Ok(None)
}
async fn heal_bucket_metadata(&self, _bucket: &str) -> Result<(), Error> {
Ok(())
}
async fn list_buckets(&self) -> Result<Vec<BucketInfo>, Error> {
Ok(Vec::new())
}
@@ -386,6 +414,14 @@ mod tests {
Ok(false)
}
async fn get_object_size(&self, _bucket: &str, _object: &str) -> Result<Option<u64>, Error> {
Ok(None)
}
async fn get_object_checksum(&self, _bucket: &str, _object: &str) -> Result<Option<String>, Error> {
Ok(None)
}
async fn heal_object(
&self,
_bucket: &str,
@@ -404,6 +440,10 @@ mod tests {
Ok((HealResultItem::default(), None))
}
async fn list_objects_for_heal(&self, _bucket: &str, _prefix: &str) -> Result<Vec<HealListItem>, Error> {
Ok(Vec::new())
}
async fn list_objects_for_heal_page(
&self,
_bucket: &str,
+82 -23
View File
@@ -117,16 +117,13 @@ fn test_format_set_disk_id_from_i32_valid() {
assert_eq!(result.unwrap(), "pool_0_set_1");
}
/// A wall-clock lower bound for "the timestamp was actually read from the
/// clock": 2020-01-01. `unwrap_or_default()` on a pre-epoch clock yields 0, and
/// the old versions of these tests bound the fields to `_` and so could not tell
/// that apart from a real reading (rustfs/backlog#1836).
const SANE_EPOCH_SECS: u64 = 1_577_836_800;
#[test]
fn test_resume_state_timestamp_handling() {
use rustfs_heal::heal::resume::ResumeState;
// Test that ResumeState creation doesn't panic even if system time is before epoch
// This is a theoretical test - in practice, system time should never be before epoch
// But we want to ensure unwrap_or_default handles edge cases
let state = ResumeState::new(
"test-task".to_string(),
"test-type".to_string(),
@@ -134,30 +131,22 @@ fn test_resume_state_timestamp_handling() {
vec!["bucket1".to_string()],
);
assert!(
state.start_time > SANE_EPOCH_SECS,
"start_time fell back to the default instead of reading the clock: {}",
state.start_time
);
assert!(
state.last_update >= state.start_time,
"last_update {} must not predate start_time {}",
state.last_update,
state.start_time
);
// Verify fields are initialized (u64 is always >= 0)
// The important thing is that unwrap_or_default prevents panic
let _ = state.start_time;
let _ = state.last_update;
}
#[test]
fn test_resume_checkpoint_timestamp_handling() {
use rustfs_heal::heal::resume::ResumeCheckpoint;
// Test that ResumeCheckpoint creation doesn't panic
let checkpoint = ResumeCheckpoint::new("test-task".to_string());
assert!(
checkpoint.checkpoint_time > SANE_EPOCH_SECS,
"checkpoint_time fell back to the default instead of reading the clock: {}",
checkpoint.checkpoint_time
);
// Verify field is initialized (u64 is always >= 0)
// The important thing is that unwrap_or_default prevents panic
let _ = checkpoint.checkpoint_time;
}
#[test]
@@ -184,18 +173,45 @@ fn test_heal_task_status_atomic_update() {
async fn get_object_meta(&self, _bucket: &str, _object: &str) -> rustfs_heal::Result<Option<HealObjectInfo>> {
Ok(None)
}
async fn get_object_data(&self, _bucket: &str, _object: &str) -> rustfs_heal::Result<Option<Vec<u8>>> {
Ok(None)
}
async fn put_object_data(&self, _bucket: &str, _object: &str, _data: &[u8]) -> rustfs_heal::Result<()> {
Ok(())
}
async fn delete_object(&self, _bucket: &str, _object: &str) -> rustfs_heal::Result<()> {
Ok(())
}
async fn verify_object_integrity(&self, _bucket: &str, _object: &str) -> rustfs_heal::Result<bool> {
Ok(true)
}
async fn ec_decode_rebuild(&self, _bucket: &str, _object: &str) -> rustfs_heal::Result<Vec<u8>> {
Ok(vec![])
}
async fn get_disk_status(&self, _endpoint: &Endpoint) -> rustfs_heal::Result<rustfs_heal::heal::storage::DiskStatus> {
Ok(rustfs_heal::heal::storage::DiskStatus::Ok)
}
async fn format_disk(&self, _endpoint: &Endpoint) -> rustfs_heal::Result<()> {
Ok(())
}
async fn get_bucket_info(&self, _bucket: &str) -> rustfs_heal::Result<Option<BucketInfo>> {
Ok(None)
}
async fn heal_bucket_metadata(&self, _bucket: &str) -> rustfs_heal::Result<()> {
Ok(())
}
async fn list_buckets(&self) -> rustfs_heal::Result<Vec<BucketInfo>> {
Ok(vec![])
}
async fn object_exists(&self, _bucket: &str, _object: &str) -> rustfs_heal::Result<bool> {
Ok(false)
}
async fn get_object_size(&self, _bucket: &str, _object: &str) -> rustfs_heal::Result<Option<u64>> {
Ok(None)
}
async fn get_object_checksum(&self, _bucket: &str, _object: &str) -> rustfs_heal::Result<Option<String>> {
Ok(None)
}
async fn heal_object(
&self,
_bucket: &str,
@@ -218,6 +234,9 @@ fn test_heal_task_status_atomic_update() {
) -> rustfs_heal::Result<(rustfs_madmin::heal_commands::HealResultItem, Option<rustfs_heal::Error>)> {
Ok((rustfs_madmin::heal_commands::HealResultItem::default(), None))
}
async fn list_objects_for_heal(&self, _bucket: &str, _prefix: &str) -> rustfs_heal::Result<Vec<HealListItem>> {
Ok(vec![])
}
async fn list_objects_for_heal_page(
&self,
_bucket: &str,
@@ -259,7 +278,7 @@ fn test_heal_task_status_atomic_update() {
#[tokio::test]
async fn test_heal_task_transient_object_exists_skip_avoids_recreate() {
use rustfs_heal::heal::storage::{HealListItem, HealObjectInfo, HealStorageAPI};
use rustfs_heal::heal::storage::{DiskStatus, HealListItem, HealObjectInfo, HealStorageAPI};
use rustfs_heal::heal::task::{HealOptions, HealPriority, HealRequest, HealTask, HealTaskStatus, HealType};
use std::sync::{
Arc,
@@ -277,14 +296,42 @@ async fn test_heal_task_transient_object_exists_skip_avoids_recreate() {
Ok(None)
}
async fn get_object_data(&self, _bucket: &str, _object: &str) -> rustfs_heal::Result<Option<Vec<u8>>> {
Ok(None)
}
async fn put_object_data(&self, _bucket: &str, _object: &str, _data: &[u8]) -> rustfs_heal::Result<()> {
Ok(())
}
async fn delete_object(&self, _bucket: &str, _object: &str) -> rustfs_heal::Result<()> {
Ok(())
}
async fn verify_object_integrity(&self, _bucket: &str, _object: &str) -> rustfs_heal::Result<bool> {
Ok(true)
}
async fn ec_decode_rebuild(&self, _bucket: &str, _object: &str) -> rustfs_heal::Result<Vec<u8>> {
Ok(Vec::new())
}
async fn get_disk_status(&self, _endpoint: &Endpoint) -> rustfs_heal::Result<DiskStatus> {
Ok(DiskStatus::Ok)
}
async fn format_disk(&self, _endpoint: &Endpoint) -> rustfs_heal::Result<()> {
Ok(())
}
async fn get_bucket_info(&self, _bucket: &str) -> rustfs_heal::Result<Option<BucketInfo>> {
Ok(None)
}
async fn heal_bucket_metadata(&self, _bucket: &str) -> rustfs_heal::Result<()> {
Ok(())
}
async fn list_buckets(&self) -> rustfs_heal::Result<Vec<BucketInfo>> {
Ok(Vec::new())
}
@@ -296,6 +343,14 @@ async fn test_heal_task_transient_object_exists_skip_avoids_recreate() {
))
}
async fn get_object_size(&self, _bucket: &str, _object: &str) -> rustfs_heal::Result<Option<u64>> {
Ok(None)
}
async fn get_object_checksum(&self, _bucket: &str, _object: &str) -> rustfs_heal::Result<Option<String>> {
Ok(None)
}
async fn heal_object(
&self,
_bucket: &str,
@@ -322,6 +377,10 @@ async fn test_heal_task_transient_object_exists_skip_avoids_recreate() {
Ok((rustfs_madmin::heal_commands::HealResultItem::default(), None))
}
async fn list_objects_for_heal(&self, _bucket: &str, _prefix: &str) -> rustfs_heal::Result<Vec<HealListItem>> {
Ok(Vec::new())
}
async fn list_objects_for_heal_page(
&self,
_bucket: &str,
+27 -120
View File
@@ -47,13 +47,6 @@ pub const INTERNODE_MSGPACK_DIRECTION_REQUEST: &str = "request";
pub const INTERNODE_MSGPACK_DIRECTION_RESPONSE: &str = "response";
pub const INTERNODE_MSGPACK_CODEC_MSGPACK: &str = "msgpack";
pub const INTERNODE_MSGPACK_CODEC_JSON: &str = "json";
pub const INTERNODE_STAGE_READ_VERSION_REQUEST_ENCODE: &str = "read_version_request_encode";
pub const INTERNODE_STAGE_READ_VERSION_REQUEST_DECODE: &str = "read_version_request_decode";
pub const INTERNODE_STAGE_READ_VERSION_DISK_READ: &str = "read_version_disk_read";
pub const INTERNODE_STAGE_READ_VERSION_RESPONSE_JSON_ENCODE: &str = "read_version_response_json_encode";
pub const INTERNODE_STAGE_READ_VERSION_RESPONSE_MSGPACK_ENCODE: &str = "read_version_response_msgpack_encode";
pub const INTERNODE_STAGE_READ_VERSION_RPC_ROUNDTRIP: &str = "read_version_rpc_roundtrip";
pub const INTERNODE_STAGE_READ_VERSION_RESPONSE_DECODE: &str = "read_version_response_decode";
const OPERATION_LABEL: &str = "operation";
const BACKEND_LABEL: &str = "backend";
@@ -74,7 +67,6 @@ const INTERNODE_OPERATION_REQUESTS_OUTGOING_TOTAL: &str = "rustfs_system_network
const INTERNODE_OPERATION_REQUESTS_INCOMING_TOTAL: &str = "rustfs_system_network_internode_operation_requests_incoming_total";
const INTERNODE_OPERATION_ERRORS_TOTAL: &str = "rustfs_system_network_internode_operation_errors_total";
const INTERNODE_OPERATION_DURATION_MS: &str = "rustfs_system_network_internode_operation_duration_ms";
const INTERNODE_OPERATION_STAGE_DURATION_MS: &str = "rustfs_system_network_internode_operation_stage_duration_ms";
const INTERNODE_OPERATION_CLASSIFIED_ERRORS_TOTAL: &str = "rustfs_system_network_internode_operation_classified_errors_total";
const INTERNODE_OPERATION_RETRIES_TOTAL: &str = "rustfs_system_network_internode_operation_retries_total";
const INTERNODE_OPERATION_RETRY_SUCCESSES_TOTAL: &str = "rustfs_system_network_internode_operation_retry_successes_total";
@@ -113,7 +105,6 @@ const SERVER_OPERATION_BACKEND_HTTP_VERSION_LABELS: &[&str] = &[SERVER_LABEL, OP
const SERVER_OPERATION_BACKEND_FAILURE_REASON_LABELS: &[&str] =
&[SERVER_LABEL, OPERATION_LABEL, BACKEND_LABEL, FAILURE_REASON_LABEL];
const SERVER_OPERATION_BACKEND_RPC_PATH_LABELS: &[&str] = &[SERVER_LABEL, OPERATION_LABEL, BACKEND_LABEL, RPC_PATH_LABEL];
const SERVER_OPERATION_BACKEND_STAGE_LABELS: &[&str] = &[SERVER_LABEL, OPERATION_LABEL, BACKEND_LABEL, STAGE_LABEL];
const SERVER_LABELS: &[&str] = &[SERVER_LABEL];
const SERVER_REASON_LABELS: &[&str] = &[SERVER_LABEL, REASON_LABEL];
const SERVER_QUORUM_FAILURE_LABELS: &[&str] = &[SERVER_LABEL, STAGE_LABEL, DOMINANT_ERROR_LABEL];
@@ -143,10 +134,6 @@ pub const INTERNODE_OPERATION_METRICS: &[InternodeOperationMetricDescriptor] = &
name: INTERNODE_OPERATION_DURATION_MS,
labels: SERVER_OPERATION_BACKEND_LABELS,
},
InternodeOperationMetricDescriptor {
name: INTERNODE_OPERATION_STAGE_DURATION_MS,
labels: SERVER_OPERATION_BACKEND_STAGE_LABELS,
},
InternodeOperationMetricDescriptor {
name: INTERNODE_OPERATION_CLASSIFIED_ERRORS_TOTAL,
labels: SERVER_OPERATION_BACKEND_CLASSIFICATION_LABELS,
@@ -407,24 +394,6 @@ impl InternodeMetrics {
.record(duration_ms);
}
pub fn record_stage_duration_for_operation_and_backend(
&self,
operation: &'static str,
backend: &'static str,
stage: &'static str,
duration: Duration,
) {
let duration_ms = duration.as_secs_f64() * 1000.0;
metrics::histogram!(
INTERNODE_OPERATION_STAGE_DURATION_MS,
SERVER_LABEL => current_server_label(),
OPERATION_LABEL => operation,
BACKEND_LABEL => backend,
STAGE_LABEL => stage
)
.record(duration_ms);
}
pub fn record_classified_error_for_operation_and_backend(
&self,
operation: &'static str,
@@ -1019,90 +988,42 @@ mod tests {
assert_eq!(snapshot.replay_cache_evictions_total, 3);
}
#[test]
fn operation_stage_duration_records_low_cardinality_stage_labels() {
let recorder = DebuggingRecorder::new();
let snapshotter = recorder.snapshotter();
let metrics = InternodeMetrics::default();
with_local_recorder(&recorder, || {
metrics.record_stage_duration_for_operation_and_backend(
INTERNODE_OPERATION_GRPC_READ_VERSION,
INTERNODE_TRANSPORT_BACKEND_GRPC,
INTERNODE_STAGE_READ_VERSION_RPC_ROUNDTRIP,
Duration::from_micros(125),
);
});
let entries: Vec<_> = snapshotter
.snapshot()
.into_vec()
.into_iter()
.filter(|(composite, _, _, _)| composite.key().name() == INTERNODE_OPERATION_STAGE_DURATION_MS)
.collect();
assert_eq!(entries.len(), 1);
let labels: HashMap<_, _> = entries[0]
.0
.key()
.labels()
.map(|label| (label.key().to_string(), label.value().to_string()))
.collect();
assert_eq!(
labels.get(OPERATION_LABEL).map(String::as_str),
Some(INTERNODE_OPERATION_GRPC_READ_VERSION)
);
assert_eq!(labels.get(BACKEND_LABEL).map(String::as_str), Some(INTERNODE_TRANSPORT_BACKEND_GRPC));
assert_eq!(
labels.get(STAGE_LABEL).map(String::as_str),
Some(INTERNODE_STAGE_READ_VERSION_RPC_ROUNDTRIP)
);
assert!(labels.get(SERVER_LABEL).is_some_and(|value| !value.is_empty()));
match &entries[0].3 {
DebugValue::Histogram(samples) => assert_eq!(samples.iter().map(|sample| sample.0).collect::<Vec<_>>(), vec![0.125]),
other => panic!("{INTERNODE_OPERATION_STAGE_DURATION_MS} must be a histogram, got {other:?}"),
}
}
#[test]
fn operation_metric_descriptors_include_backend_and_operation_labels() {
assert_eq!(INTERNODE_OPERATION_METRICS.len(), 22);
assert_eq!(INTERNODE_OPERATION_METRICS.len(), 21);
for metric in &INTERNODE_OPERATION_METRICS[..6] {
assert_eq!(metric.labels, &[SERVER_LABEL, OPERATION_LABEL, BACKEND_LABEL]);
}
assert_eq!(
INTERNODE_OPERATION_METRICS[6].labels,
&[SERVER_LABEL, OPERATION_LABEL, BACKEND_LABEL, STAGE_LABEL]
);
for metric in &INTERNODE_OPERATION_METRICS[7..10] {
for metric in &INTERNODE_OPERATION_METRICS[6..9] {
assert_eq!(metric.labels, &[SERVER_LABEL, OPERATION_LABEL, BACKEND_LABEL, CLASSIFICATION_LABEL]);
}
assert_eq!(
INTERNODE_OPERATION_METRICS[10].labels,
INTERNODE_OPERATION_METRICS[9].labels,
&[SERVER_LABEL, OPERATION_LABEL, BACKEND_LABEL, HTTP_VERSION_LABEL]
);
for metric in &INTERNODE_OPERATION_METRICS[11..13] {
for metric in &INTERNODE_OPERATION_METRICS[10..12] {
assert_eq!(metric.labels, &[SERVER_LABEL, OPERATION_LABEL, BACKEND_LABEL]);
}
assert_eq!(
INTERNODE_OPERATION_METRICS[13].labels,
INTERNODE_OPERATION_METRICS[12].labels,
&[SERVER_LABEL, OPERATION_LABEL, BACKEND_LABEL, FAILURE_REASON_LABEL]
);
assert_eq!(
INTERNODE_OPERATION_METRICS[13].labels,
&[SERVER_LABEL, OPERATION_LABEL, BACKEND_LABEL, RPC_PATH_LABEL]
);
assert_eq!(
INTERNODE_OPERATION_METRICS[14].labels,
&[SERVER_LABEL, OPERATION_LABEL, BACKEND_LABEL, RPC_PATH_LABEL]
);
assert_eq!(
INTERNODE_OPERATION_METRICS[15].labels,
&[SERVER_LABEL, OPERATION_LABEL, BACKEND_LABEL, RPC_PATH_LABEL]
);
for metric in &INTERNODE_OPERATION_METRICS[16..18] {
for metric in &INTERNODE_OPERATION_METRICS[15..17] {
assert_eq!(metric.labels, &[SERVER_LABEL]);
}
assert_eq!(INTERNODE_OPERATION_METRICS[18].labels, &[SERVER_LABEL, REASON_LABEL]);
assert_eq!(INTERNODE_OPERATION_METRICS[19].labels, &[SERVER_LABEL, STAGE_LABEL, DOMINANT_ERROR_LABEL]);
assert_eq!(INTERNODE_OPERATION_METRICS[17].labels, &[SERVER_LABEL, REASON_LABEL]);
assert_eq!(INTERNODE_OPERATION_METRICS[18].labels, &[SERVER_LABEL, STAGE_LABEL, DOMINANT_ERROR_LABEL]);
// Payload histogram + large-payload counter carry operation+backend labels.
assert_eq!(INTERNODE_OPERATION_METRICS[19].labels, &[SERVER_LABEL, OPERATION_LABEL, BACKEND_LABEL]);
assert_eq!(INTERNODE_OPERATION_METRICS[20].labels, &[SERVER_LABEL, OPERATION_LABEL, BACKEND_LABEL]);
assert_eq!(INTERNODE_OPERATION_METRICS[21].labels, &[SERVER_LABEL, OPERATION_LABEL, BACKEND_LABEL]);
}
#[test]
@@ -1133,66 +1054,62 @@ mod tests {
);
assert_eq!(
INTERNODE_OPERATION_METRICS[6].name,
"rustfs_system_network_internode_operation_stage_duration_ms"
);
assert_eq!(
INTERNODE_OPERATION_METRICS[7].name,
"rustfs_system_network_internode_operation_classified_errors_total"
);
assert_eq!(
INTERNODE_OPERATION_METRICS[8].name,
INTERNODE_OPERATION_METRICS[7].name,
"rustfs_system_network_internode_operation_retries_total"
);
assert_eq!(
INTERNODE_OPERATION_METRICS[9].name,
INTERNODE_OPERATION_METRICS[8].name,
"rustfs_system_network_internode_operation_retry_successes_total"
);
assert_eq!(
INTERNODE_OPERATION_METRICS[10].name,
INTERNODE_OPERATION_METRICS[9].name,
"rustfs_system_network_internode_operation_http_versions_total"
);
assert_eq!(
INTERNODE_OPERATION_METRICS[11].name,
INTERNODE_OPERATION_METRICS[10].name,
"rustfs_system_network_internode_operation_stall_timeouts_total"
);
assert_eq!(
INTERNODE_OPERATION_METRICS[12].name,
INTERNODE_OPERATION_METRICS[11].name,
"rustfs_system_network_internode_operation_write_shutdown_errors_total"
);
assert_eq!(
INTERNODE_OPERATION_METRICS[13].name,
INTERNODE_OPERATION_METRICS[12].name,
"rustfs_system_network_internode_rpc_auth_failures_total"
);
assert_eq!(
INTERNODE_OPERATION_METRICS[14].name,
INTERNODE_OPERATION_METRICS[13].name,
"rustfs_system_network_internode_replay_cache_overflow_by_operation_total"
);
assert_eq!(
INTERNODE_OPERATION_METRICS[15].name,
INTERNODE_OPERATION_METRICS[14].name,
"rustfs_system_network_internode_replay_cache_records_total"
);
assert_eq!(
INTERNODE_OPERATION_METRICS[16].name,
INTERNODE_OPERATION_METRICS[15].name,
"rustfs_system_network_internode_replay_cache_entries"
);
assert_eq!(
INTERNODE_OPERATION_METRICS[17].name,
INTERNODE_OPERATION_METRICS[16].name,
"rustfs_system_network_internode_replay_cache_capacity"
);
assert_eq!(
INTERNODE_OPERATION_METRICS[18].name,
INTERNODE_OPERATION_METRICS[17].name,
"rustfs_system_network_internode_replay_cache_evictions_total"
);
assert_eq!(
INTERNODE_OPERATION_METRICS[19].name,
INTERNODE_OPERATION_METRICS[18].name,
"rustfs_system_storage_erasure_write_quorum_failures_total"
);
assert_eq!(
INTERNODE_OPERATION_METRICS[20].name,
INTERNODE_OPERATION_METRICS[19].name,
"rustfs_system_network_internode_operation_payload_bytes"
);
assert_eq!(
INTERNODE_OPERATION_METRICS[21].name,
INTERNODE_OPERATION_METRICS[20].name,
"rustfs_system_network_internode_operation_large_payloads_total"
);
assert_eq!(INTERNODE_OPERATION_GRPC_READ_MULTIPLE, "grpc_read_multiple");
@@ -1212,16 +1129,6 @@ mod tests {
assert_eq!(INTERNODE_MSGPACK_DIRECTION_RESPONSE, "response");
assert_eq!(INTERNODE_MSGPACK_CODEC_MSGPACK, "msgpack");
assert_eq!(INTERNODE_MSGPACK_CODEC_JSON, "json");
assert_eq!(INTERNODE_STAGE_READ_VERSION_REQUEST_ENCODE, "read_version_request_encode");
assert_eq!(INTERNODE_STAGE_READ_VERSION_REQUEST_DECODE, "read_version_request_decode");
assert_eq!(INTERNODE_STAGE_READ_VERSION_DISK_READ, "read_version_disk_read");
assert_eq!(INTERNODE_STAGE_READ_VERSION_RESPONSE_JSON_ENCODE, "read_version_response_json_encode");
assert_eq!(
INTERNODE_STAGE_READ_VERSION_RESPONSE_MSGPACK_ENCODE,
"read_version_response_msgpack_encode"
);
assert_eq!(INTERNODE_STAGE_READ_VERSION_RPC_ROUNDTRIP, "read_version_rpc_roundtrip");
assert_eq!(INTERNODE_STAGE_READ_VERSION_RESPONSE_DECODE, "read_version_response_decode");
assert_eq!(
INTERNODE_SIGNATURE_V1_FALLBACK_TOTAL,
"rustfs_system_network_internode_signature_v1_fallback_total"
-110
View File
@@ -120,14 +120,6 @@ pub const PUT_STAGE_SET_DISK_RENAME_BACKUP_DIR_FSYNC: &str = "set_disk_rename_ba
pub const PUT_STAGE_SET_DISK_RENAME_ANCESTOR_DIR_FSYNC: &str = "set_disk_rename_ancestor_dir_fsync";
pub const PUT_STAGE_SET_DISK_RENAME_RENAME_SYSCALL: &str = "set_disk_rename_rename_syscall";
pub const PUT_RENAME_FDATASYNC_BATCH_MODE_SERIAL: &str = "serial";
pub const PUT_RENAME_FDATASYNC_BATCH_MODE_PARALLEL: &str = "parallel";
pub const PUT_RENAME_QUORUM_FANOUT_STATE_SCHEDULED: &str = "scheduled";
pub const PUT_RENAME_QUORUM_FANOUT_STATE_WRITE_QUORUM: &str = "write_quorum";
pub const PUT_RENAME_QUORUM_FANOUT_STATE_SUCCESS: &str = "success";
pub const PUT_RENAME_QUORUM_FANOUT_STATE_ERROR: &str = "error";
pub const PUT_RENAME_QUORUM_FANOUT_STATE_PANIC: &str = "panic";
#[inline(always)]
pub fn get_stage_metrics_enabled() -> bool {
GET_STAGE_METRICS_ENABLED.load(Ordering::Relaxed)
@@ -2050,44 +2042,6 @@ pub fn record_put_object_stage_duration_from(stage: &'static str, started_at: Op
}
}
#[inline(always)]
fn put_stage_count_value(value: usize) -> f64 {
match u32::try_from(value) {
Ok(value) => f64::from(value),
Err(_) => f64::from(u32::MAX),
}
}
#[inline(always)]
pub fn record_put_rename_fdatasync_batch(mode: &'static str, files: usize) {
if !put_stage_metrics_enabled() {
return;
}
histogram!("rustfs_s3_put_object_rename_fdatasync_batch_files", "mode" => mode).record(put_stage_count_value(files));
}
#[inline(always)]
pub fn record_put_rename_quorum_wait_fanout(
scheduled: usize,
write_quorum: usize,
success: usize,
error: usize,
panicked: usize,
) {
if !put_stage_metrics_enabled() {
return;
}
for (state, count) in [
(PUT_RENAME_QUORUM_FANOUT_STATE_SCHEDULED, scheduled),
(PUT_RENAME_QUORUM_FANOUT_STATE_WRITE_QUORUM, write_quorum),
(PUT_RENAME_QUORUM_FANOUT_STATE_SUCCESS, success),
(PUT_RENAME_QUORUM_FANOUT_STATE_ERROR, error),
(PUT_RENAME_QUORUM_FANOUT_STATE_PANIC, panicked),
] {
histogram!("rustfs_s3_put_object_rename_quorum_wait_fanout_disks", "state" => state).record(put_stage_count_value(count));
}
}
/// Record generic internal operation stage duration (non-PUT paths).
/// Use this for metacache walks, listing, lifecycle, and other background
/// operations that are NOT part of the PUT object hot path.
@@ -3168,70 +3122,6 @@ mod tests {
assert!(stages.iter().all(|stage| recorded.contains(*stage)));
}
#[test]
fn put_rename_code_level_metrics_are_static_and_gated() {
let _guard = METRICS_FLAG_LOCK.lock().unwrap_or_else(|e| e.into_inner());
let recorder = DebuggingRecorder::new();
let snapshotter = recorder.snapshotter();
metrics::with_local_recorder(&recorder, || {
set_put_stage_metrics_enabled(false);
record_put_rename_fdatasync_batch(PUT_RENAME_FDATASYNC_BATCH_MODE_SERIAL, 2);
record_put_rename_quorum_wait_fanout(4, 3, 3, 1, 0);
set_put_stage_metrics_enabled(true);
record_put_rename_fdatasync_batch(PUT_RENAME_FDATASYNC_BATCH_MODE_PARALLEL, 9);
record_put_rename_quorum_wait_fanout(4, 3, 3, 1, 0);
set_put_stage_metrics_enabled(false);
});
let rows = snapshotter.snapshot().into_vec();
assert_eq!(histogram_samples(&rows, "rustfs_s3_put_object_rename_fdatasync_batch_files"), vec![9.0]);
let batch_modes = rows
.iter()
.filter(|(composite, _, _, _)| {
composite.kind() == MetricKind::Histogram
&& composite.key().name() == "rustfs_s3_put_object_rename_fdatasync_batch_files"
})
.flat_map(|(composite, _, _, _)| {
composite
.key()
.labels()
.filter(|label| label.key() == "mode")
.map(|label| label.value().to_string())
.collect::<Vec<_>>()
})
.collect::<HashSet<_>>();
assert_eq!(batch_modes, HashSet::from([PUT_RENAME_FDATASYNC_BATCH_MODE_PARALLEL.to_string()]));
let quorum_samples = histogram_samples(&rows, "rustfs_s3_put_object_rename_quorum_wait_fanout_disks");
assert_eq!(quorum_samples, vec![0.0, 1.0, 3.0, 3.0, 4.0]);
let quorum_states = rows
.iter()
.filter(|(composite, _, _, _)| {
composite.kind() == MetricKind::Histogram
&& composite.key().name() == "rustfs_s3_put_object_rename_quorum_wait_fanout_disks"
})
.flat_map(|(composite, _, _, _)| {
composite
.key()
.labels()
.filter(|label| label.key() == "state")
.map(|label| label.value().to_string())
.collect::<Vec<_>>()
})
.collect::<HashSet<_>>();
assert_eq!(
quorum_states,
HashSet::from([
PUT_RENAME_QUORUM_FANOUT_STATE_SCHEDULED.to_string(),
PUT_RENAME_QUORUM_FANOUT_STATE_WRITE_QUORUM.to_string(),
PUT_RENAME_QUORUM_FANOUT_STATE_SUCCESS.to_string(),
PUT_RENAME_QUORUM_FANOUT_STATE_ERROR.to_string(),
PUT_RENAME_QUORUM_FANOUT_STATE_PANIC.to_string(),
])
);
}
#[test]
fn test_put_object_diagnostic_buckets() {
assert_eq!(put_object_size_bucket(0), "unknown");
+1 -20
View File
@@ -48,7 +48,6 @@ pub struct LocalClient {
struct LocalGuardEntry {
guard: FastLockGuard,
acquired_at: SystemTime,
last_refreshed: SystemTime,
expires_at: SystemTime,
deadline: Instant,
ttl: Duration,
@@ -61,7 +60,6 @@ impl LocalGuardEntry {
Self {
guard,
acquired_at,
last_refreshed: acquired_at,
expires_at: acquired_at.checked_add(ttl).unwrap_or(acquired_at),
deadline: monotonic_now.checked_add(ttl).unwrap_or(monotonic_now),
ttl,
@@ -76,7 +74,6 @@ impl LocalGuardEntry {
let now = SystemTime::now();
let monotonic_now = Instant::now();
self.expires_at = now.checked_add(self.ttl).unwrap_or(now);
self.last_refreshed = now;
self.deadline = monotonic_now.checked_add(self.ttl).unwrap_or(monotonic_now);
}
}
@@ -350,7 +347,7 @@ impl LockClient for LocalClient {
owner: entry.guard.owner().to_string(),
acquired_at: entry.acquired_at,
expires_at: entry.expires_at,
last_refreshed: entry.last_refreshed,
last_refreshed: SystemTime::now(),
metadata: LockMetadata::default(),
priority: LockPriority::Normal,
wait_start_time: None,
@@ -374,7 +371,6 @@ impl LockClient for LocalClient {
owner: entry.guard.owner().to_string(),
acquired_at: entry.acquired_at,
remaining_ttl: entry.deadline.saturating_duration_since(Instant::now()),
guard_id: (!entry.guard.is_disabled()).then(|| entry.guard.guard_id()),
}));
}
leases
@@ -488,12 +484,6 @@ mod tests {
.success
);
let initial = client.list_lock_leases().await.pop().expect("acquired lock should be listed");
let initial_status = client
.check_status(&lock_id)
.await
.expect("initial lock status should be readable")
.expect("newly acquired lock should remain held");
assert_eq!(initial_status.last_refreshed, initial_status.acquired_at);
tokio::time::advance(Duration::from_secs(20)).await;
let aging = client
@@ -502,13 +492,6 @@ mod tests {
.pop()
.expect("held lock should remain listed before refresh");
assert_eq!(aging.remaining_ttl, Duration::from_secs(10));
let aging_status = client
.check_status(&lock_id)
.await
.expect("aging lock status should be readable")
.expect("aging lock should remain held");
assert_eq!(aging_status.last_refreshed, initial_status.last_refreshed);
assert!(client.refresh(&lock_id).await.expect("refresh should return a result"));
let refreshed = client
@@ -523,9 +506,7 @@ mod tests {
.expect("refreshed lock should remain held");
assert_eq!(refreshed.acquired_at, initial.acquired_at);
assert_eq!(refreshed.guard_id, initial.guard_id);
assert_eq!(status.acquired_at, initial.acquired_at);
assert!(status.last_refreshed > initial_status.last_refreshed);
assert_eq!(refreshed.remaining_ttl, Duration::from_secs(30));
tokio::time::advance(Duration::from_secs(30)).await;
+6 -39
View File
@@ -100,7 +100,7 @@ impl FastObjectLockManager {
Ok(()) => {
let guard = FastLockGuard::new(request.key, request.mode, request.owner, shard.clone());
// Register guard to prevent premature cleanup
shard.register_guard_with_info(guard.guard_id(), guard.key(), guard.mode(), guard.owner());
shard.register_guard(guard.guard_id());
Ok(guard)
}
Err(err) => Err(err),
@@ -223,7 +223,7 @@ impl FastObjectLockManager {
if acquired {
let guard = FastLockGuard::new(key.clone(), mode, owner.clone(), shard.clone());
shard.register_guard_with_info(guard.guard_id(), guard.key(), guard.mode(), guard.owner());
shard.register_guard(guard.guard_id());
all_successful.push(key);
guards.push(guard);
}
@@ -252,7 +252,7 @@ impl FastObjectLockManager {
match shard.acquire_lock(request).await {
Ok(()) => {
let guard = FastLockGuard::new(request.key.clone(), request.mode, request.owner.clone(), shard.clone());
shard.register_guard_with_info(guard.guard_id(), guard.key(), guard.mode(), guard.owner());
shard.register_guard(guard.guard_id());
acquired_guards.push(guard);
}
Err(err) => {
@@ -310,15 +310,6 @@ impl FastObjectLockManager {
infos
}
/// Enumerate held locks with holder counts and stable holder identities.
pub fn list_locks_with_holder_generations(&self) -> Vec<(crate::fast_lock::types::ObjectLockInfo, u32, Option<Vec<u64>>)> {
let mut infos = Vec::new();
for shard in &self.shards {
infos.extend(shard.list_locks_with_holder_generations());
}
infos
}
/// Force-release every holder of the lock on `key`.
///
/// Returns the number of owners released (0 if the resource was not locked).
@@ -565,15 +556,15 @@ mod tests {
let write_key = ObjectKey::new("bucket", "write-object");
let read_key = ObjectKey::new("bucket", "read-object");
let write_guard = manager
let _write_guard = manager
.acquire_write_lock(write_key.clone(), "writer")
.await
.expect("write lock should acquire");
let read_guard = manager
let _read_guard = manager
.acquire_read_lock(read_key.clone(), "reader")
.await
.expect("read lock should acquire");
let second_read_guard = manager
let _second_read_guard = manager
.acquire_read_lock(read_key.clone(), "reader")
.await
.expect("second read lock should acquire");
@@ -602,30 +593,6 @@ mod tests {
.expect("write holder count listed");
assert_eq!(*write_holder_count, 1);
let generations = manager.list_locks_with_holder_generations();
let (_, _, read_generations) = generations
.iter()
.find(|(info, _, _)| info.key == read_key)
.expect("read holder generations listed");
let mut expected_read_generations = vec![read_guard.guard_id(), second_read_guard.guard_id()];
expected_read_generations.sort_unstable();
assert_eq!(read_generations.as_ref(), Some(&expected_read_generations));
let (_, _, write_generations) = generations
.iter()
.find(|(info, _, _)| info.key == write_key)
.expect("write holder generation listed");
assert_eq!(write_generations.as_ref(), Some(&vec![write_guard.guard_id()]));
drop(read_guard);
let remaining = manager.list_locks_with_holder_generations();
let (_, remaining_count, remaining_generations) = remaining
.iter()
.find(|(info, _, _)| info.key == read_key)
.expect("remaining read holder generation listed");
assert_eq!(*remaining_count, 1);
assert_eq!(remaining_generations.as_ref(), Some(&vec![second_read_guard.guard_id()]));
manager.shutdown().await;
}
+6 -74
View File
@@ -24,20 +24,7 @@ use crate::fast_lock::{
state::ObjectLockState,
types::{LockMode, LockResult, ObjectKey, ObjectLockRequest},
};
#[derive(Debug)]
struct ActiveGuardInfo {
key: ObjectKey,
mode: LockMode,
owner: Arc<str>,
}
#[derive(Debug, PartialEq, Eq, Hash)]
struct GuardHolderKey {
key: ObjectKey,
mode: LockMode,
owner: Arc<str>,
}
use std::collections::HashSet;
/// Lock shard to reduce global contention
#[derive(Debug)]
@@ -51,7 +38,7 @@ pub struct LockShard {
/// Shard ID for debugging
_shard_id: usize,
/// Active guard IDs to prevent cleanup of locks with live guards
active_guards: parking_lot::Mutex<HashMap<u64, Option<ActiveGuardInfo>>>,
active_guards: parking_lot::Mutex<HashSet<u64>>,
}
/// Cancellation-safe waiter counter ticket.
@@ -97,7 +84,7 @@ impl LockShard {
object_pool: ObjectStatePool::new(),
metrics: ShardMetrics::new(),
_shard_id: shard_id,
active_guards: parking_lot::Mutex::new(HashMap::new()),
active_guards: parking_lot::Mutex::new(HashSet::new()),
}
}
@@ -340,7 +327,7 @@ impl LockShard {
// First, try to remove the guard from active set
let guard_was_active = {
let mut guards = self.active_guards.lock();
guards.remove(&guard_id).is_some()
guards.remove(&guard_id)
};
// If guard was not active, this is a double-release attempt
@@ -388,19 +375,8 @@ impl LockShard {
/// Register a guard to prevent premature cleanup
pub fn register_guard(&self, guard_id: u64) {
self.active_guards.lock().insert(guard_id, None);
}
pub(crate) fn register_guard_with_info(&self, guard_id: u64, key: &ObjectKey, mode: LockMode, owner: &Arc<str>) {
let mut guards = self.active_guards.lock();
guards.insert(
guard_id,
Some(ActiveGuardInfo {
key: key.clone(),
mode,
owner: owner.clone(),
}),
);
guards.insert(guard_id);
}
/// Unregister a guard (called when guard is dropped)
@@ -420,7 +396,7 @@ impl LockShard {
#[cfg(test)]
pub fn is_guard_active(&self, guard_id: u64) -> bool {
let guards = self.active_guards.lock();
guards.contains_key(&guard_id)
guards.contains(&guard_id)
}
/// Calculate adaptive timeout based on current system load and request priority
@@ -626,50 +602,6 @@ impl LockShard {
infos
}
pub(crate) fn list_locks_with_holder_generations(
&self,
) -> Vec<(crate::fast_lock::types::ObjectLockInfo, u32, Option<Vec<u64>>)> {
// Snapshot lock state before guard registrations. Acquires register after
// mutating state, while releases unregister before mutating state, so a
// concurrent transition can only make the cohort mismatch and fall back.
let infos = self.list_locks_with_holder_counts();
let guards = self.active_guards.lock();
let mut guard_ids_by_holder: HashMap<GuardHolderKey, Vec<u64>> = HashMap::with_capacity(guards.len());
for (&guard_id, guard) in guards
.iter()
.filter_map(|(guard_id, guard)| guard.as_ref().map(|guard| (guard_id, guard)))
{
let key = GuardHolderKey {
key: guard.key.clone(),
mode: guard.mode,
owner: guard.owner.clone(),
};
guard_ids_by_holder
.entry(key)
.and_modify(|guard_ids| guard_ids.push(guard_id))
.or_insert_with(|| vec![guard_id]);
}
drop(guards);
for guard_ids in guard_ids_by_holder.values_mut() {
guard_ids.sort_unstable();
}
infos
.into_iter()
.map(|(info, holder_count)| {
let key = GuardHolderKey {
key: info.key.clone(),
mode: info.mode,
owner: info.owner.clone(),
};
let generation = guard_ids_by_holder
.remove(&key)
.filter(|guard_ids| u32::try_from(guard_ids.len()).ok() == Some(holder_count));
(info, holder_count, generation)
})
.collect()
}
/// Force-release every holder of a lock on `key`, regardless of owner.
///
/// Returns the number of owners that were released. Used by the admin
+1 -1
View File
@@ -257,7 +257,7 @@ impl std::fmt::Display for ObjectKey {
}
/// Lock type for object operations
#[derive(Debug, Clone, Copy, PartialEq, Eq, Hash, Serialize, Deserialize)]
#[derive(Debug, Clone, Copy, PartialEq, Eq, Serialize, Deserialize)]
pub enum LockMode {
/// Shared lock for read operations
Shared,
-2
View File
@@ -90,8 +90,6 @@ pub struct LockLeaseInfo {
pub owner: String,
/// Original acquisition time. Refreshes do not change this value.
pub acquired_at: SystemTime,
/// Opaque guard identity used to reject stale diagnostic snapshots.
pub guard_id: Option<u64>,
/// Remaining lease duration derived from the monotonic lease deadline.
pub remaining_ttl: Duration,
}
+45 -5
View File
@@ -1425,6 +1425,10 @@ fn maintenance_inspection_decision(generation: u64, current_generation: u64, att
}
}
fn single_disk_default_cycle_secs(_features: ScannerMaintenanceFeatures) -> Option<u64> {
None
}
fn single_disk_default_speed() -> ScannerSpeed {
ScannerSpeed::Default
}
@@ -1588,12 +1592,9 @@ async fn configure_scanner_defaults(
scanner_maintenance_generation(),
)
});
// Single-disk keeps the speed-preset-derived default cycle (60s at the
// `default` preset) instead of a special shorter cycle: no measured
// cold-start ILM latency basis for an override, and clean-idle backoff
// already stretches idle cadence. Decision record: backlog#1878 (HS-16).
let default_cycle_secs = single_disk_default_cycle_secs(features);
set_scanner_default_speed(single_disk_default_speed());
set_scanner_default_cycle_secs(None);
set_scanner_default_cycle_secs(default_cycle_secs);
info!(
target: "rustfs::scanner",
event = EVENT_SCANNER_RUNTIME_CONFIG,
@@ -1602,6 +1603,7 @@ async fn configure_scanner_defaults(
env_speed = ENV_SCANNER_SPEED,
env_cycle = ENV_SCANNER_CYCLE,
env_start_delay = ENV_SCANNER_START_DELAY_SECS,
?default_cycle_secs,
lifecycle_active = features.lifecycle,
replication_active = features.replication,
feature_inspection_failed = features.inspection_failed,
@@ -6948,6 +6950,11 @@ mod tests {
});
}
#[test]
fn test_single_disk_default_cycle_uses_speed_based_interval_without_maintenance_features() {
assert_eq!(single_disk_default_cycle_secs(ScannerMaintenanceFeatures::default()), None);
}
#[test]
fn test_single_disk_default_speed_uses_regular_scanner_default() {
assert_eq!(single_disk_default_speed(), ScannerSpeed::Default);
@@ -7408,6 +7415,39 @@ mod tests {
});
}
#[test]
fn test_single_disk_default_cycle_preserves_regular_cycle_for_lifecycle() {
assert_eq!(
single_disk_default_cycle_secs(ScannerMaintenanceFeatures {
lifecycle: true,
..Default::default()
}),
None
);
}
#[test]
fn test_single_disk_default_cycle_preserves_regular_cycle_for_replication() {
assert_eq!(
single_disk_default_cycle_secs(ScannerMaintenanceFeatures {
replication: true,
..Default::default()
}),
None
);
}
#[test]
fn test_single_disk_default_cycle_preserves_regular_cycle_on_inspection_failure() {
assert_eq!(
single_disk_default_cycle_secs(ScannerMaintenanceFeatures {
inspection_failed: true,
..Default::default()
}),
None
);
}
#[test]
#[serial]
fn test_cycle_interval_keeps_default_cycle_with_explicit_speed() {
+59
View File
@@ -71,6 +71,7 @@ const EVENT_SCANNER_METADATA_CORRUPT: &str = "scanner_metadata_corrupt";
const EVENT_SCANNER_LIFECYCLE_ACTION: &str = "scanner_lifecycle_action";
const EVENT_SCANNER_HEAL_ADMISSION: &str = "scanner_heal_admission";
const EVENT_SCANNER_ALERT_STATE: &str = "scanner_alert_state";
const EVENT_SCANNER_COMPAT_STATE: &str = "scanner_compat_state";
const DATA_USAGE_UPDATE_DIR_CYCLES: u32 = 16;
const DATA_SCANNER_COMPACT_LEAST_OBJECT: usize = 500;
@@ -91,6 +92,7 @@ const ENV_FAILED_OBJECTS_MAX: &str = "RUSTFS_DATA_USAGE_FAILED_OBJECTS_MAX";
const DEFAULT_FAILED_OBJECT_TTL_SECS: u32 = 86_400;
const DEFAULT_FAILED_OBJECTS_MAX: u32 = 10_000;
const DEFAULT_SCANNER_DEEP_VERIFY_COOLDOWN_SECS: u64 = 60;
const METRIC_SCANNER_INLINE_HEAL_TOTAL: &str = "rustfs_scanner_inline_heal_total";
const METRIC_SCANNER_EXCESS_OBJECT_VERSIONS_TOTAL: &str = "rustfs_scanner_excess_object_versions_total";
const METRIC_SCANNER_EXCESS_OBJECT_VERSION_SIZE_TOTAL: &str = "rustfs_scanner_excess_object_version_size_total";
const METRIC_SCANNER_EXCESS_FOLDERS_TOTAL: &str = "rustfs_scanner_excess_folders_total";
@@ -194,6 +196,8 @@ fn emit_scanner_alert_event(event_name: &str, bucket: &str, object: &str, size:
}
const MAX_PENDING_SCANNER_HEALS_PER_BUCKET: usize = 10_000;
static SCANNER_INLINE_HEAL_WARN_ONCE: Once = Once::new();
static SCANNER_INLINE_HEAL_METRICS_ONCE: Once = Once::new();
static SCANNER_ALERT_METRICS_ONCE: Once = Once::new();
#[cfg(test)]
@@ -247,6 +251,27 @@ fn effective_object_heal_scan_mode(heal_bitrot: bool, mod_time: Option<OffsetDat
}
}
fn scanner_inline_heal_enabled() -> bool {
scanner_inline_heal_enabled_from_value(std::env::var(rustfs_config::ENV_SCANNER_INLINE_HEAL_ENABLE).ok().as_deref())
}
fn scanner_inline_heal_enabled_from_value(value: Option<&str>) -> bool {
match value {
Some(value) => matches!(value.trim().to_ascii_lowercase().as_str(), "1" | "true" | "on" | "yes"),
None => rustfs_config::DEFAULT_SCANNER_INLINE_HEAL_ENABLE,
}
}
fn ensure_scanner_inline_heal_metric_registered() {
SCANNER_INLINE_HEAL_METRICS_ONCE.call_once(|| {
describe_counter!(
METRIC_SCANNER_INLINE_HEAL_TOTAL,
"Total number of inline heal operations executed directly by scanner."
);
counter!(METRIC_SCANNER_INLINE_HEAL_TOTAL).increment(0);
});
}
fn ensure_scanner_alert_metrics_registered() {
SCANNER_ALERT_METRICS_ONCE.call_once(|| {
describe_counter!(
@@ -480,6 +505,24 @@ fn should_alert_excessive_versions(remaining_versions: usize, cumulative_size: i
(too_many_versions, too_large_versions)
}
fn warn_inline_heal_compat_requested() {
if !scanner_inline_heal_enabled() {
return;
}
SCANNER_INLINE_HEAL_WARN_ONCE.call_once(|| {
warn!(
target: "rustfs::scanner::folder",
event = EVENT_SCANNER_COMPAT_STATE,
component = LOG_COMPONENT_SCANNER,
subsystem = LOG_SUBSYSTEM_HEAL,
env = rustfs_config::ENV_SCANNER_INLINE_HEAL_ENABLE,
state = "inline_heal_rollback_unsupported",
"Scanner inline-heal rollback is unsupported; using async heal admission"
);
});
}
fn non_negative_i64_to_u64(value: i64) -> u64 {
value.max(0) as u64
}
@@ -1239,6 +1282,7 @@ impl ScannerItem {
async fn heal_actions(&mut self, oi: &ObjectInfo, actual_size: i64, size_summary: &mut SizeSummary) -> i64 {
if self.heal_enabled {
warn_inline_heal_compat_requested();
self.enqueue_heal(oi).await;
}
@@ -3187,6 +3231,8 @@ pub async fn scan_data_folder(
) -> Result<DataUsageCache, ScannerError> {
use crate::data_usage_define::DATA_USAGE_ROOT;
ensure_scanner_inline_heal_metric_registered();
// Check that we're not trying to scan the root
if cache.info.name.is_empty() || cache.info.name == DATA_USAGE_ROOT {
return Err(ScannerError::Other("internal error: root scan attempted".to_string()));
@@ -4279,6 +4325,19 @@ mod tests {
assert!(!scanner.new_cache.info.failed_objects.contains_key("expired"));
}
#[test]
fn test_scanner_inline_heal_enabled_defaults_to_false() {
assert!(!scanner_inline_heal_enabled_from_value(None));
}
#[test]
fn test_scanner_inline_heal_enabled_reads_env_override() {
assert!(scanner_inline_heal_enabled_from_value(Some("true")));
assert!(scanner_inline_heal_enabled_from_value(Some("YES")));
assert!(scanner_inline_heal_enabled_from_value(Some("1")));
assert!(!scanner_inline_heal_enabled_from_value(Some("false")));
}
#[test]
fn test_build_object_heal_request_omits_nil_version_id() {
let request = build_object_heal_request(
@@ -1,109 +0,0 @@
# Heal/Scanner 配置与语义对照(MinIO parity 决策记录)
对应 backlog rustfs/backlog#1878(父 #1862,批 HS-14/HS-16/HS-18)。本页沉淀三项"决策 + 文档化"结论:scanner idle 节流语义对照与迁移警告(HS-14)、单机默认扫描周期决策(HS-16)、stale multipart 与 tmp/.trash 清理三段核对(HS-18),并顺带收录 bitrot_cycle 与 alert_excess_folders 两项已确认的默认值差异。所有 MinIO 侧结论均于 2026-08 按 minio/minio master 逐源码核对(引用文件为上游路径),不转述二手资料。
运行时旋钮的完整清单、状态端点与调参流程见 [Scanner Runtime Controls](scanner-runtime-controls.md)excess 告警阈值差异见 [Scanner Excess Alerts](scanner-excess-alerts_zh.md)heal 并发模型对照见 [Heal 并发安全说明](heal-concurrency-safety-notes-zh.md)。
## 1. HS-14scanner idle 节流语义对照
### RustFS 当前语义(三因子)
RustFS 的 scanner 步进节流由三个因子共同决定(crates/scanner/src/sleeper.rs):
1. **总闸 `scanner.idle_mode` / `RUSTFS_SCANNER_IDLE_MODE`(默认 `true`**`false` 时所有节流 sleep 全部跳过,scanner 全速推进;`true` 时按下面两因子计算 sleep。
2. **速度档**`scanner.speed` / `RUSTFS_SCANNER_SPEED`,默认 `default`):档位表与 MinIO 完全一致(见下表)。目录级 sleep = `1ms × factor`(上限 `max_wait`);对象级 sleep = `本对象处理耗时 × factor`,下限 1ms、上限 `max_wait`
3. **前台读退避下限**`current_foreground_read_activity()` 取并发 GetObject 请求数(rustfs/src/storage/concurrency/request_guard.rs 的 `GetObjectGuard`)与流式读计数(`ForegroundReadGuard`)的较大值,换算为 `10ms × 活跃读数`、封顶 250ms 的下限;该下限对目录级与对象级 sleep 都生效(`.max(foreground_sleep)`),且**可以超过速度档的 `max_wait`**(自身封顶 250ms)。速度档为 `fastest`factor=0)时预设 sleep 为 0,但只要 `idle_mode=true`,前台读下限仍然生效。
| 速度档 | sleep factor | 单次 sleep 上限 | 周期间隔 |
|---|---:|---:|---:|
| `fastest` | 0 | 0 | 1s |
| `fast` | 1× | 100ms | 1m |
| `default` | 2× | 1s | 1m |
| `slow` | 10× | 15s | 1m |
| `slowest` | 100× | 15s | 30m |
实际行为矩阵(RustFS):
| `idle_mode` | 速度档 | 前台并发读 = 0 | 前台并发读 > 0 |
|---|---|---|---|
| `false` | 任意 | 完全不休眠,全速 | 完全不休眠,全速(前台退避也被总闸关闭) |
| `true` | `fastest` | 预设 sleep = 0,等效全速 | 每步 sleep = 前台读下限(10ms×读数,封顶 250ms) |
| `true` | 其余档 | 每步 sleep = 预设值(1ms~15s 封顶) | 每步 sleep = max(预设值, 前台读下限) |
周期间隔的解析优先级为 env `RUSTFS_SCANNER_CYCLE` > 持久化 `scanner.cycle` > `scanner.start_delay` > 启动期默认覆盖(当前恒无)> 速度档派生(crates/scanner/src/runtime_config.rs)。另有 `scanner.yield_every_n_objects`(默认 128)的协作式让出,与节流 sleep 相互独立。
### MinIO 当前语义(master 逐源码核对)
MinIO 的对应开关是 `scanner:idle_speed` / `MINIO_SCANNER_IDLE_SPEED`internal/config/scanner/scanner.go):取值为空串或 `on`(默认)时 `IdleMode=0`,取值 `off``IdleMode=1`。启动/配置加载时一次性写入 `scannerIdleMode`cmd/config-current.go),扫描侧闭包 `weSleep = scannerIdleMode.Load() == 0`cmd/xl-storage-disk-id-check.go):**`on`(默认)= 目录级与对象级节流 sleep 始终插入(按速度档 factorminSleep 100µs);`off` = 两条节流路径完全不 sleep,全速扫描**。当前上游没有任何按 S3 请求/磁盘活动动态调整节流的逻辑——这是静态开关。
命名具有误导性,是历史残留:2024-01 之前 `weSleep` 由磁盘活动驱动("Entire queue is full, so we sleep",即有并发 S3/heal 活动才 sleep),minio/minio#18734commit 7705605b)把该活动门替换为上述静态配置(初版取值 `throttled`/`full`,后改为 `on`/`off`),上游残留注释 "default is throttled when idle"、"Sleep always or based on incoming S3 requests" 均是替换前的语义描述,与现行代码不符。
### 对照与迁移警告
| 维度 | RustFS | MinIOmaster |
|---|---|---|
| 开关名 | `scanner.idle_mode` / `RUSTFS_SCANNER_IDLE_MODE` | `scanner:idle_speed` / `MINIO_SCANNER_IDLE_SPEED` |
| 取值 | 布尔 `true`/`false` | `on`/`off` |
| 默认 | `true`(节流开启) | `on`(节流开启) |
| 开 = | 节流总闸开:速度档 sleep + 前台读下限 | 节流总闸开:速度档 sleep |
| 关 = | 完全不休眠(含前台读下限一并失效) | 完全不休眠 |
| 活动耦合 | 有:前台并发读抬高 sleep 下限(10ms×读数,封顶 250ms) | 无(2024-01 起为静态开关) |
| 速度档表 | 两边完全一致(上表) | 同左 |
迁移警告:
- **环境变量名不可照搬**RustFS 只读取 `RUSTFS_*` 前缀,不解析 `MINIO_SCANNER_*` 任何别名(crates/scanner、crates/utils 的 env 读取无别名链,测试还专门断言 `MINIO_SCANNER_SPEED`/`MINIO_SCANNER_CYCLE` 不泄漏生效)。照搬 `MINIO_SCANNER_IDLE_SPEED=off` 到 RustFS 会静默无效,必须改写成 `RUSTFS_SCANNER_IDLE_MODE=false`
- **取值词表不同**`on/off` vs `true/false`,不能原样复制。
- **方向澄清(修正父 issue 的预设)**:按当前上游源码,MinIO `idle_speed` 与 RustFS `idle_mode` 在"开=节流、关=全速"方向上是一致的,并非反向;父 issue 中"MinIO on=集群空闲才节流、off=始终按 delay 节流"的矩阵描述的是 2024-01 之前的活动耦合行为与反向解读,与 master 不符。真正需要写进迁移手册的差异是:MinIO 的 `idle_speed` 名称暗示"空闲时才慢"但实际是静态总闸;RustFS 的 `idle_mode=true` 在总闸之上还叠加了 MinIO 没有的前台读保护下限。
- **`false` 是大锤**RustFS `idle_mode=false` 会连前台读退避一起关闭,scanner 与前台读完全抢盘;仅在 benchmark 或可独占 IO 的窗口使用。
**决策(HS-14):保持现状。** RustFS 语义更直观(`idle_mode` = 节流总闸,`true` 即自适应限速),且比 MinIO 多一层前台读保护;不新增 `RUSTFS_SCANNER_IDLE_SPEED` 兼容别名(无社区强诉求不做,避免双入口漂移)。本节即对照表与迁移警告的正式落点。
## 2. bitrot_cycle 默认差异
| 项 | RustFS | MinIO |
|---|---|---|
| 键 | `heal.bitrot_cycle` / `RUSTFS_SCANNER_BITROT_CYCLE_SECS`scanner.bitrot_cycle 为兼容旧键) | `heal:bitrotscan` / `MINIO_HEAL_BITROTSCAN` |
| 默认 | 30 天(crates/config/src/constants/heal.rs 的 `DEFAULT_HEAL_BITROT_CYCLE_SECS`):按墙钟周期把扫描切深扫(deep bitrot | `off`internal/config/heal/heal.go 默认 `EnableOff`):不做周期性深扫,仅普通扫描 + 管理端手动深扫 |
| 对齐方式 | 迁移 MinIO 行为:`heal.bitrot_cycle=off``RUSTFS_SCANNER_BITROT_CYCLE_SECS=disabled` | 反向:`heal:bitrotscan=<秒>` |
RustFS 的 30 天默认是刻意的耐用性默认(周期性全量 bitrot 校验),代价是每 30 天一轮深扫 IO;单机场景另有清洁空闲退避封顶约 42 分钟的墙钟保护(见 scanner-runtime-controls.md)。这是行为差异而非缺陷,文档化即可。
## 3. alert_excess_folders 默认差异
RustFS 默认 65538(容纳 Proxmox Backup Server 每目录 65536 chunk 的布局),MinIO 默认 50000。差异原因、另两个 excess 阈值(versions=100 相同、version_size TiB vs TB)、事件名映射与冷却语义已完整记录在 [Scanner Excess Alerts](scanner-excess-alerts_zh.md),此处不重复。
## 4. HS-18stale multipart 与 tmp/.trash 清理三段核对
MinIO 把"清理已删除数据"拆成三段:stale upload 先 rename 进 `.minio.sys/tmp/.trash/<uuid>` 隔离(rename 快、原子);trash 由独立例程排空;tmp 下非 trash 的旧目录单独回收。逐段核对 RustFS:
| 段 | MinIO | RustFS | 判定 |
|---|---|---|---|
| stale multipart → 隔离 | `cleanupStaleUploadsOnDisk`cmd/erasure-multipart.go)逐盘列出 multipart 目录,按 uploadID 目录名里的 UnixNano 判龄,超过 `stale_uploads_expiry`(默认 24h)即 `renameAll``.minio.sys/tmp/.trash/<uuid>`,空 sha 目录、tmp 旧目录同法 | `cleanup_stale_multipart_uploads_in_set`crates/ecstore/src/bucket/lifecycle/bucket_lifecycle_ops.rs)发现候选后取 ns 写锁 + 重查(`lock_stale_multipart_cleanup`),`delete_all_with_quorum` 扇出逐盘递归删除,而 LocalDisk 的递归删除内部就是 `move_to_trash`crates/ecstore/src/disk/local.rs)把目录 rename 进 `.rustfs.sys/tmp/.trash/<uuid>` | 行为等价(都是先隔离后清理);RustFS 额外有写锁 + quorum 重查 + 锁丢失 fencecrates/ecstore/src/set_disk/ops/multipart.rs 的 `StaleMultipartCleanupGuard`),防并发 CompleteMultipartUpload 竞争,安全性强于 MinIO 的无锁 rename |
| trash 排空 | 每 `delete_cleanup_interval`(默认 5minternal/config/api/api.go)逐盘删 `.trash` 内条目,逐条以 `deleteCleanupSleeper`factor 5 / 25mscmd/globals.go)节流 | 每盘独立 `cleanup_deleted_objects_loop``DELETED_OBJECTS_CLEANUP_INTERVAL` = 5mcrates/ecstore/src/disk/local.rs),先排空 `.trash` 再回收 tmp 旧目录;排空为顺序 `remove_dir_all`/`remove_file`**无逐条 sleep 节流** | 基本等价;唯一差异是 RustFS 排空不节流,trash 积压大时单轮 IO 更突发(5m 周期天然限频),文档化,如实测出现清理风暴再补节流 |
| tmp 非 trash 旧目录 | 并在 `cleanupStaleUploadsOnDisk` 内:非 `.trash` 的 tmp 目录超过 `stale_uploads_expiry`24hrename 进 trash(随 6h 任务) | `cleanup_stale_tmp_objects`crates/ecstore/src/disk/local.rs)随 5m 循环执行:非 `.trash` 目录超过 `STALE_TMP_OBJECT_EXPIRY` = 24h 即 rename 进 trash;另有启动时 tmp → tmp-old 整体换名 + 后台删除的崩溃安全路径 | 行为等价(阈值同为 24h);RustFS 检查频率 5m vs MinIO 6h,回收更及时 |
周期与环境变量默认值对照(两边一致):
| 项 | RustFS | MinIO |
|---|---|---|
| stale upload 过期阈值 | `RUSTFS_API_STALE_UPLOADS_EXPIRY`,默认 24h | `MINIO_API_STALE_UPLOADS_EXPIRY`,默认 24h |
| stale multipart 清理周期 | `RUSTFS_API_STALE_UPLOADS_CLEANUP_INTERVAL`,默认 6h | `MINIO_API_STALE_UPLOADS_CLEANUP_INTERVAL`,默认 6h |
| trash 排空周期 | 5m(常量,暂无开关) | `MINIO_API_DELETE_CLEANUP_INTERVAL`,默认 5m |
关于 rustfs/src/delete_tail_activity.rs:它**不覆盖三段中的任何一段**。该模块是 delete 尾部活动的进程内指标计数(inflight gauge + 耗时 histogram),供 allocator 回收压力判断(rustfs/src/allocator_reclaim.rs)使用;生产代码目前只在对象复用路径使用 `Replication`/`Notify` 两个 stage 计数,`Tail`/`Cleanup` 枚举值暂无调用点。
崩溃残留窗口结论:
- trash 内部残留(排空中途崩溃):`.trash/<uuid>` 是自包含目录,下一轮 5m tick 重扫 `.trash` 自然收敛,与 MinIO 相同。
- 跨盘扇出中途崩溃(部分盘已 rename 进 trash、其余未动):若剩余盘数仍满足写 quorum,下一轮 6h 任务重新发现候选并重删,自然收敛;若已清理盘数超过 parity(剩余低于写 quorum),`check_multipart_upload_path_exists``FileNotFound` 不在 `OBJECT_OP_IGNORED_ERRS`crates/ecstore/src/disk/error_reduce.rs)而判 quorum 失败,候选被跳过,残留 uploadID 目录不会被该任务收敛(不可见于 S3 API,仅占盘空间)。该窗口极窄(逐盘 rename 为毫秒级,需恰在扇出中途且已过 parity 盘时进程死亡)。MinIO 同场景会收敛(逐盘独立处理、无 quorum 闸门)。**分级:有崩溃残留窗口(极窄)→ 登记后续修复**;修复需为清理守卫提供把"已不存在"计为达成终态的专用 quorum 变体(不能改共享的 `check_multipart_upload_path_exists` 语义,它同时服务 CompleteMultipartUpload),超出本批"几行小修"边界,不在本 PR 扩 scope。
## 5. HS-16:单机(ErasureSD)默认扫描周期决策
启动期曾有预留钩子 `single_disk_default_cycle_secs`,可按维护特征(lifecycle/replication/巡检失败)为单机覆盖默认周期,但从未接线、恒返回 `None`,已删除(本批 PR)。决策:**单机默认周期保持速度档派生(`default` 档 = 60s),不做特殊覆盖**。理由:其一,无任何实测依据表明单机冷启动 ILM 延迟需要更短周期,凭空缩短只会放大空闲扫描频次;其二,单机已有清洁空闲退避(连续干净周期间隔翻倍,默认 bitrot 窗口下封顶约 42 分钟,见 scanner-runtime-controls.md),空闲时的周期压力已被消化;其三,若确有诉求,用户可用 `RUSTFS_SCANNER_CYCLE` / `scanner.cycle` 显式配置,无需内置特殊路径。需要更激进短周期的场景应先拿实测数据再议。
## 6. 决策摘要
- HS-14:保持 `RUSTFS_SCANNER_IDLE_MODE` 现语义(true=节流总闸+前台读下限,false=全速),文档化对照表与迁移警告,不做兼容别名。
- HS-16:删除恒 `None` 的单机默认周期钩子,单机周期保持速度档派生 + 清洁空闲退避。
- HS-18:三段清理行为等价(trash 排空无逐条节流、tmp 回收频率 5m vs 6h 两处小差异文档化);跨盘扇出的极窄崩溃残留窗口登记后续;周期默认值 24h/6h/5m 与 MinIO 对齐。
@@ -1,504 +0,0 @@
# RustFS heal & scanner vs MinIO — comprehensive parity analysis (v2, 2026-08-16)
> English | [中文版](rustfs-heal-scanner-vs-minio-comprehensive-analysis-2026-08-16_zh.md)
- Date: 2026-08-16 (based on that day's `main` code; audit HEAD ≈ `a118d7e4f`)
- Scope: `crates/heal` (src 19,560 lines + tests 2,274 lines), `crates/scanner` (src ~26,000 lines + tests), `crates/data-usage`, the heal/heal_walk/bitrot_self_verify and config parts of `crates/ecstore`, `crates/common/src/heal_channel.rs`, `crates/madmin` (heal/scanner wire types), `rustfs/src` (startup wiring, admin handlers, cluster RPC)
- Parity baseline: minio/minio master (HEAD `7aac2a2c5b`; the repo has entered maintenance mode with master frozen, i.e. its final state)
- Method: four parallel audit tracks (heal crate / scanner crate / ecstore integration layer / MinIO source study), with key conclusions verified by hand one by one (points marked "verified first-hand" below were checked against the source directly)
- This document supersedes `docs/rustfs-heal-scanner-vs-minio-parity-assessment.md` (2026-06-15, v1). Since v1 there have been more than 80 heal/scanner commits (the full automatic drive-replacement healing chain, the resume state machine, making usage convergence authoritative, cluster-level heal coordination, ILM restore semantics, etc.), so v1's feature inventory and gap judgments are comprehensively outdated; v1 conclusions such as "bloom filter missing" were verified this round to be **misjudgments** (see §5.4).
---
## 0. Conclusion summary
1. **Overall verdict: the core functional chains of heal and scanner are complete.** Object-level heal (quorum arbitration + ETag fallback + bitrot Deep verification + dangling handling), erasure set deep scans (per-set disk-walk union enumeration), per-version resumable scans (schema'd persistence layer + CAS atomic publish + crash-window backfill), automatic drive-replacement healing (readiness validation + identity fencing + durable intent + completion proof), the scanner cycle loop (leader lock + persisted leader-epoch fence), data usage statistics (bucket-level/cluster-level, primary + backup + observed snapshots, epoch/cycle anti-rollback), the full ILM action set (expiry/transition/noncurrent/free-version/delete-marker cleanup), and the admin Start/Query/Cancel protocol (clientToken semantics aligned with madmin) — all of these are implemented and carry regression tests. There are **no empty implementations / early-return stubs** inside the two crates; every exceptional path has logs + metrics + error semantics.
2. **The main gaps concentrate on "entry points and the observability surface", not on the repair algorithms themselves**: the MRF/ECDecode/Metadata task executors are implemented but have no production trigger entry (`HealEvent` is entirely unwired); `CheckAbandonedParts` is `NotImplemented` at all three ecstore layers; the heal/scanner trace channels are missing; scanner excess S3 events are missing; madmin client methods are missing (only wire types exist); heal byte-level progress/ETA is not implemented.
3. **Important corrections to the v1 understanding**: the bloom filter has been **removed** from current MinIO master (`.bloomcycle.bin` stores only a cycle count), so RustFS's current state matches MinIO; the MinIO scanner is likewise a **cluster-level leader singleton**, and RustFS's leader.lock model is the same shape as MinIO's; RustFS's ETag majority-fallback arbitration is already implemented (`crates/ecstore/src/set_disk/ops/heal.rs:525-567,679`, verified first-hand) — the arbitration gap v1 worried about does not exist.
4. **RustFS exceeds MinIO in several places**: the remote_scanner RPC protocol (remote peers scan locally instead of the leader reading remote drives across the network), the persisted leader-epoch CAS fence, cycle budgets and per-set/per-disk concurrency gates, the pending-heal ledger, the durable replacement intent + completion proof state machine, foreground pressure gating (mainline throttle), and the cluster heal control coordinator + envelope replay protection.
5. Gap severity tally: 8 P1 items (behavioral/operational alignment gaps), 9 P2 items (completeness), 3 P3 items (cleanup/low risk), and 7 items of "not pursuing parity by design". Full list in §6.
---
## 1. Architecture overview
### 1.1 RustFS's three-layer architecture
RustFS splits the heal/scanner functionality that MinIO keeps inside the `cmd/` monolith into three layers plus two standalone crates:
| Layer | Location | Responsibilities |
|---|---|---|
| Primitives layer | `crates/ecstore/src/set_disk/ops/heal.rs` (~3,240 lines), `ops/heal_walk.rs`, `ops/bitrot_self_verify.rs`; upper wrappers `store/heal.rs`, `store/heal_walk.rs`, `core/sets.rs` | Object/bucket/format/replacement-drive format repair, disk-walk union enumeration, write-path bitrot self-verification; the `rustfs_storage_api::HealOperations` contract is implemented by `SetDisks`/`Sets`/`ECStore` (`crates/storage-api/src/object.rs:503-519`) |
| heal runtime | `crates/heal` | Process-level HealManager (priority queue/scheduler/auto disk scanner/resumable resume), HealChannelProcessor (consumes the global heal channel), drive-replacement recovery state machine |
| scanner runtime | `crates/scanner` | Data usage scanning, ILM evaluation and enqueueing, heal candidate production, replication usage statistics, remote scanner RPC |
| Shared protocol | `crates/common/src/heal_channel.rs` (~776 lines) | Start/Query/Cancel command channel, `HealOpts`/`HealScanMode`/`HealRequestSource`/`HealAdmission*` shared types, `HealResultItem` (madmin) |
| Shared data | `crates/data-usage` | `DataUsageEntry/Info`, histograms, `hash_path`; produced by the scanner, consumed by ecstore/admin |
Startup chain (wiring verified first-hand):
1. `rustfs/src/startup_services.rs:93``init_background_service_runtime(store)`.
2. `rustfs/src/startup_background.rs:41-81`: create the global heal service cancel token; read `RUSTFS_SCANNER_ENABLED` (alias `RUSTFS_ENABLE_SCANNER`, default true) and `RUSTFS_HEAL_ENABLED` (alias `RUSTFS_ENABLE_HEAL`, default true); **the heal manager is initialized whenever either heal or scanner is enabled** (heal candidates produced by the scanner need a consumer; with both off, the heal channel is not initialized and `send_heal_request` reports "Heal channel not initialized").
3. `crates/heal/src/lib.rs:142-216`: atomic initialization inside an owned task (a caller cancel cannot leave a half-initialized manager behind, `lib.rs:123-131`; `GLOBAL_HEAL_RUNTIME_INIT` mutex single-flight) → `HealManager::start()``rustfs_common::heal_channel::init_heal_channels()` → spawn `HealChannelProcessor::start_with_receipts`.
4. `crates/heal/src/heal/manager.rs:1301-1356` `HealManager::start`: `start_scheduler()` (`manager.rs:2394-2461`, interval default 10s + `Notify` event-driven wakeup) → `process_unclean_shutdown()` (`manager.rs:1362-1695`) → when `enable_auto_heal` (default true), `start_auto_disk_scanner()` (`manager.rs:2464-2999`).
5. After the server is ready, `rustfs/src/startup_lifecycle.rs:150-152`: when `enable_scanner`, `init_data_scanner(token, store)` (`crates/scanner/src/scanner.rs:1293-1372`).
6. Graceful shutdown: `rustfs/src/startup_shutdown.rs:308` `shutdown_ahm_services()` (cancel token); `:414` `clear_unclean_shutdown_markers()`.
### 1.2 MinIO's corresponding structure (final master state)
| MinIO file | Responsibilities |
|---|---|
| `cmd/admin-heal-ops.go` | Manual admin heal sequence (healSequence, clientToken/forceStart/forceStop) |
| `cmd/global-heal.go` | Resident background heal queue (newBgHealSequence, token fixed `0000-…`, never ends) + `healErasureSet` (full-object heal per set) |
| `cmd/background-heal-ops.go` | healRoutine worker pool (`_MINIO_HEAL_WORKERS`, default GOMAXPROCS/2) consuming healTask |
| `cmd/mrf.go` | MRF (Most Recent Fail) queue (capacity 100,000), persisted at process exit to `.minio.sys/buckets/.heal/mrf/list.bin` with startup replay |
| `cmd/background-newdisks-heal-ops.go` | Automatic resync for new/replaced drives (monitorLocalDisksAndHeal 10s polling + healFreshDisk + healingTracker) |
| `cmd/erasure-healing.go` / `erasure-healing-common.go` | Object-level heal core (~800 lines), listAndHeal |
| `cmd/data-scanner.go` | Scanner loop (globalLeaderLock cluster singleton) + folderScanner + applyActions |
| `cmd/erasure.go` (nsScanner) / `erasure-server-pool.go` | NSScanner three-layer structure |
| `cmd/bucket-lifecycle.go` | ILM executor (expiry/transition worker pools) |
| `cmd/xl-storage.go` | DiskInfo.Healing, CheckParts/VerifyFile, CleanAbandonedData, RenameData healing branch |
| `cmd/prepare-storage.go` | waitForFormatErasure new-drive startup handshake |
### 1.3 Architecture-level differences (design trade-offs, not defects)
1. **heal queue model**: MinIO funnels every heal (scanner sampling/MRF/admin/new-disk resync) into a single channel + a fixed worker pool (new-disk resync additionally has a per-drive worker pool); RustFS is a multi-policy scheduler built from a priority heap + dedup-merge + capacity-tiered dropping + per-set bulkhead + foreground pressure gating (`manager.rs:3003-3420`). RustFS is more expressive, at the cost of an observability question around "duplicate requests being merged" (already pointed out in v1; the current `HealAdmissionReceipt` canonical task_id + alias mechanism answers it, `manager.rs:1759-1846`).
2. **scanner remote-drive access**: the MinIO leader transparently reads and writes remote-node drives through the disk abstraction layer; the RustFS leader pushes scan execution down to the remote peer to run locally via the remote_scanner RPC (`crates/scanner/src/remote_scanner.rs`), with only results and progress heartbeats sent back. Both are cluster single-leader. RustFS's approach saves the leader↔remote metadata read amplification, at the cost of maintaining a separate RPC protocol (HMAC per-frame authentication, session replay cache, fence re-validation, `remote_scanner.rs:52-61,405-496,1024-1065`).
3. **heal state persistence**: MinIO uses a single file `.healing.bin` (msgp healingTracker, reset whenever the diskID mismatches); RustFS uses a schema'd multi-file layout (resume/checkpoint/intent/seal/proof, each CAS-published, `resume.rs:38-61`), with the crash window explicitly backfilled (`erasure_healer.rs:389-402`, `resume.rs:1027-1057`).
4. **write-path self-protection**: MinIO relies on background heal to converge after writes; RustFS, after the commit rename in PutObject/CompleteMultipartUpload, actively checks `convergence.needs_heal()` and immediately enqueues an object heal (`set_disk/ops/object.rs:2291-2306`, `ops/multipart.rs:2574-2589`), and additionally has read repair (`io_primitives.rs:1040-1160`).
---
## 2. Heal implemented-feature panorama
### 2.1 Task types (`HealType`, `crates/heal/src/heal/task.rs:85-111`)
| Type | Semantics | Executor | Production trigger |
|---|---|---|---|
| `Cluster` | all buckets healed in turn (structure + optional recursive objects), in-batch retry ≤3 | `heal_cluster` task.rs:1420-1490 | channel: empty bucket means Cluster (channel.rs:576-577) |
| `Object{bucket,object,version_id}` | single object/version; when absent, rebuild per `recreate_missing` or error out | `heal_object` task.rs:855-1146 | admin, scanner, read-repair, write-path convergence, add_partial |
| `Bucket{bucket}` | bucket metadata/structure; `recursive` additionally walks all object versions | `heal_bucket` task.rs:1284-1418 + `heal_bucket_objects` task.rs:1508-1698 | admin (POST /v3/heal/{bucket}), scanner `build_bucket_heal_request` |
| `Prefix{bucket,prefix}` | recursive by prefix | `heal_prefix` task.rs:1492-1506 | channel: `recursive && prefix` non-empty (channel.rs:578-585) |
| `ErasureSet{buckets,set_disk_id}` | format repair + healing marker + per-bucket preprocessing + resumable per-version deep scan | `heal_erasure_set` task.rs:2158-2642 | admin (pool/set params), auto disk scanner, unclean shutdown, renew_disk, durable replacement recovery |
| `Metadata{bucket,object}` | metadata only (Deep, does not rebuild data) | `heal_metadata` task.rs:1700-1859 | **no production trigger** (§6 HS-01) |
| `MRF{meta_path}` | failure-path-driven Deep repair (recursive+update_parity) | `heal_mrf` task.rs:1861-1992 | **no production trigger** (only `HealEvent` can generate it, unwired) |
| `ECDecode{bucket,object,version_id}` | EC decode rebuild (Deep+recreate+update_parity), Urgent priority | `heal_ec_decode` task.rs:1994-2156 | **no production trigger** (only `HealEvent` can generate it, unwired) |
Priorities `Low/Normal/High/Urgent` (task.rs:168-179); state machine `Pending/Running/Retrying/Completed/Failed/Cancelled/Timeout` (task.rs:225-241).
### 2.2 Trigger-path panorama (beyond admin)
| Channel | source | Priority | Evidence |
|---|---|---|---|
| Scanner periodic sampling (1/1024, `RUSTFS_HEAL_OBJECT_SELECT_PROB`) | Scanner | Low | `scanner_folder.rs:2117-2136`, `:1150`; `remove_corrupted=HEAL_DELETE_DANGLING(true)`, `recreate_missing=false` (`common/heal_channel.rs:24`, `scanner_folder.rs:510-511`) |
| Scanner metadata corruption (get_size failure classified HealMetadata) | Scanner | High | `scanner_folder.rs:2147-2208`, `:1244-1260` |
| Scanner abandoned children (present in cache, absent on disk, list_path_raw quorum verification) | Scanner | High (bucket-level + object-level) | `scanner_folder.rs:2528-2792` |
| Scanner pending-heal ledger retry (persisted after rejection by a full heal channel, ≤128 per bucket per round, 10k cap) | Scanner | original priority | `scanner_folder.rs:1721-1763`, `:99-100` |
| auto disk scanner (unformatted drive confirmed via replacement_readiness / `runtime_state=="returning"` drive / durable-intent re-entry) | AutoHeal | Low | `manager.rs:2464-2999` |
| unclean shutdown recovery (startup reads the `unclean-shutdown` marker → ErasureSet heal for all local sets) | AutoHeal | Low | `manager.rs:1362-1695` |
| write-path convergence (after PutObject/CompleteMultipartUpload, `convergence.needs_heal()`) | Internal | Normal | `set_disk/ops/object.rs:2291-2306`, `ops/multipart.rs:2574-2589` |
| partial-object heal (add_partial) | Internal | Normal | `set_disk/ops/object.rs:5808-5825` |
| stale data-directory cleanup leftover enqueue | Internal | Normal | `set_disk/core/io_primitives.rs:3880-3907` |
| read repair (metadata_read_error / missing_shards / decode_error, TTL dedup cache) | ReadRepair | Low | `set_disk/read.rs:407,995,1079``submit_read_repair_heal` (`io_primitives.rs:1105-1160`), `recreate_missing=true` |
| drive reconnect hits UnformattedDisk → send_heal_disk | AutoHeal | Normal | `set_disk/ops/locking.rs:339-347` |
| Admin API (incl. cluster coordinator routing) | Admin | High | `rustfs/src/admin/handlers/heal.rs:174-212`, `:771-930` |
| cluster RPC heal (peer invocation) | — | — | `rustfs/src/storage/rpc/node_service/heal.rs`, `ecstore/src/cluster/rpc/peer_s3_client.rs:296,1209` |
Note: MinIO's MRF channel (read-path immediate delivery on missing/corrupt parts + queue persistence + shutdown replay, `cmd/mrf.go`, `erasure-object.go:395-410,800-812`) is **partially replaced** in RustFS by read-repair + write-path convergence; the three executors `HealType::MRF`/`ECDecode`/`Metadata` have no production entry (see §6 HS-01 for details).
### 2.3 Object-level heal semantics (ecstore `set_disk/ops/heal.rs`)
Flow (`heal_object_with_explicit_version_regen` from :426):
1. Take the object write lock (unless `no_lock`); an `object` ending with `/` goes through object-directory heal (`heal_object_dir_locked` :1587-1717: dangling determination + `remove` deletion + missing-volume rebuild).
2. `read_all_fileinfo` reads xl.meta from all disks; all-not-found is treated as already deleted and returns.
3. **quorum arbitration + ETag fallback** (verified first-hand): `list_online_disks` treats the mod-time quorum as authoritative; when quorum fails it falls back to ETag majority arbitration (`:525-567` `filter_by_etag`/`quorum_etag`); `pick_valid_fileinfo` picks the canonical metadata; the cannotHeal determination for "number of bad-meta disks > parity" is waived when the ETag agrees across all disks (`:679`). Matches MinIO's dual arbitration in `filterDisksByETag`.
4. `disks_with_all_parts` (:562-572) validates parts per `scan_mode`: **Normal only stats (CheckParts semantics), Deep does full bitrot verification (VerifyFile semantics)**; when a Normal scan detects `FileCorrupt` it automatically escalates to Deep and retries once (`:2022-2031`, same shape as MinIO erasure-healing.go:1101-1106); a no-parity object (EC:0) with a bitrot failure is judged unrecoverable (`:700-726`).
5. `should_heal_object_on_disk` (:606-650) classifies each disk as missing/corrupt/offline/outdated → rebuild: per-part bitrot reader/writer (using per-part checksum + algorithm), write into a temporary volume then rename to commit (`HEAL_RENAME_INCOMPLETE` retry semantics :24); dangling-deletion safety check `dangling_delete_safety` (:1488); **orphan data-directory reclamation `reclaim_orphan_data_dirs_best_effort` (:1428)** — this part covers the main scenarios of MinIO's `CleanAbandonedData` (but there is no standalone `CheckAbandonedParts` API, see §6 HS-02).
6. Versioned objects: enumerate "every version" (`storage.rs:1494-1530`); the delete-marker path is decided by `latest_meta.deleted` (`storage.rs:262-277` comment); regression tests `tests/heal_b5_versioned_regression_test.rs:282,334`.
7. Explicit-version rebuild `try_regenerate_explicit_version_meta` (:1318); cleanup of local leftovers of transitioned objects.
8. The write path additionally has shard-level bitrot self-verification `verify_written_bitrot_shards` (`ops/bitrot_self_verify.rs:45-129`, HighwayHash256S, verifying freshly written shards right before the final rename, serving the EC:0 no-parity case) — **note this is not background bitrot patrol**; background patrol is carried by scanner bitrot_cycle-driven Deep heal.
heal-crate-side wrapper (`task.rs:855-1146`): existence check (transient errors become `TransientSkip` to avoid false failures :551-569); scanner synthetic-directory normalization (:1148-1180); `recreate_missing` rebuild (:1183-1282); data-usage-cache object-lock timeout exemption (:571-653); not-found → treated_as_deleted success (:1012-1029); results `HealResultItem` keep at most 1024 entries + truncated flag (:50,845-852).
Recursive walk (`heal_bucket_objects` task.rs:1508-1698): paginated enumeration of all versions including delete markers, transient-error exponential-backoff retry ≤3 (2^n + jitter :620-627), failure-sample log truncation ≤5 entries, aggregated `BatchHealFailure`.
### 2.4 erasure set heal and resumable scans
`heal_erasure_set` (task.rs:2158-2642) runs in four phases (4-step progress tracking):
1. **Replacement intent and recovery-drive selection** (AutoHeal only + non-empty heal_endpoints): reuse the drive holding the durable intent / exclude the target endpoints and pick surviving drives; already-completed generations get an idempotent CleanupPending wrap-up.
2. **Format repair**: `heal_replacement_format(dry_run, pool, set, targets)` (`storage.rs:1372-1384`, trait default fail-closed); per-target-drive results must all be ok (`erasure_healer.rs:97-102`) + identity-fence re-check (task.rs:2410-2420).
3. **healing marker**: write an owner CAS marker `{set_disk_id}:{task_id}` to the target drive (`mod.rs:80-229`, CAS + rollback + unique concurrent owner), which makes `DiskInfo.healing` true (assignment chain verified first-hand `set_disk/mod.rs:4988`).
4. **Per-bucket preprocessing + resumable deep scan**: `ErasureSetHealer::heal_erasure_set` (`erasure_healer.rs:242-278`).
`ErasureSetHealer` scan details (benchmarked against MinIO `healErasureSet`; the `heal_walk.rs:15-23` module comment explicitly cites MinIO `global-heal.go`'s listPathRaw + objQuorum=1 + mergeXLV2Versions):
- **Enumerator choice (backlog#920)**: Deep or AutoHeal → per-set **disk-walk union enumeration** `list_versions_for_heal_page_disk_walk` ("exists on any drive" means sub-quorum reconstructible; `storage.rs:1559-1644`, page bounds 1,000 objects/10,000 versions, `dw1:` cursor); ordinary requests go through read-quorum `list_object_versions`.
- **Resume cursor**: the authoritative cursor is an opaque continuation token (`v1:` = marker JSON, `dw1:` = disk-walk key; the two namespaces are mutually exclusive against misreads, `storage.rs:81-260`); after each completed page, persist the cursor first, then clear the dedup set (`erasure_healer.rs:922-927`).
- **In-page concurrency**: FuturesUnordered + Semaphore, default `RUSTFS_HEAL_PAGE_OBJECT_CONCURRENCY=8`, Deep/AutoHeal forces 1 (`erasure_healer.rs:105-142`).
- **per-version dedup**: `compose_key` length-prefix injection encoding (`resume.rs:281-288`).
- **Error classification**: truly absent (FileNotFound etc.) → Absent (counted as success); infrastructure-transient (quorum/DiskNotFound/SlowDown etc.) → Transient (counted as skipped); everything else Failed (`erasure_healer.rs:148-182`; the comment cites backlog#856/#799 B7: offline drives must not be recorded healed/absent).
- **Loop protection**: abort when an empty page is truncated or the page-tail version identity does not advance (:933-949).
- **Completion determination**: if any of failed/skipped/failed_buckets is >0, do not mark complete; `schedule_retry()` resets both the resume and checkpoint layers (:561-626; backlog#855/B6/#1033: a skip round must not be marked complete).
- **Replacement-drive commit proof**: physical read-back on the target endpoints `replacement_targets_have_version` (`ops/heal.rs:340-412`); unconfirmed → transient skip.
### 2.5 Automatic drive-replacement healing (replacement recovery)
- **Identification** (`replacement_readiness.rs:25-73`): `replacement_mount_lease_root()` exists, canonicalize succeeds, is a mount point, the physical device id is non-empty, disjoint from the root device, and shares no physical device with sibling drives (Linux uses /proc/self/mountinfo mount-id+dev+ino). The non-root mount check has a regression test (`manager.rs:3549`).
- **State machine** (`resume.rs:63-73`): `Intent → Rebuilding → (write proof) Verified → CleanupPending → cleanup`; `Abandoned` is a terminal state; state transitions write the persistence layer first, then mutate (`save_state_strict`).
- **Persistence** (`resume.rs:38-61`, schema ResumeState=5/Checkpoint=5/proof=1): `{task_id}_ahm_resume_state.json`, `_ahm_checkpoint.json`, and intent/seal/completion_proof under the `buckets/ahm-replacement/` namespace; torn write + no seal is recognizable and rebuilt atomically (:1316-1338); CAS publish, refuses to overwrite a concurrently valid proof (:1512-1585).
- **Recovery**: both unclean shutdown and the periodic scan recover unfinished/pending-cleanup replacement generations from surviving drives (`manager.rs:1435-1640,2663-2815`); multi-generation conflict / validation failure → freeze that set (`replacement_recovery_blocked_sets`, `manager.rs:69-87,2782-2815`).
- **External snapshot**: `current_replacement_recovery_snapshot` (`lib.rs:262-333`) merges local surviving-drive records; conflict → Unknown / non-definitive; admin `GET /v4/heal/replacement-recovery`.
### 2.6 Scheduler (manager.rs)
- Priority heap + FIFO within the same priority (:148-191,330-347); dedup key per type (:469-506); enqueue three-state dedup active→queued→retrying (:1759-1785); duplicates default to Merged and return the canonical task_id (`HealAdmissionReceipt`, :1821-1846) + client token alias (:1219-1246).
- Capacity: when the queue is full, best-effort sources (Scanner/AutoHeal/ReadRepair) or low-priority items get Dropped(QueueFull); Admin/Internal may evict queued lower-priority items (`push_displacing_lower_priority` :353-396); 80%/95% tiered pressure handling (:885-909).
- Concurrency: global `max_concurrent_heals` (default 4) + per-set bulkhead `max_concurrent_per_set` (default 1) (:3040-3073,3434-3447).
- Foreground pressure gating, mainline throttle: delay best-effort tasks when foreground read/write permit utilization is ≥80% (:919-1009,2999-3020).
- Timeout: task-level aggregate timeout (default 300s), remaining budget preserved across retries (task.rs:444-451, PR #6101).
- Recoverable retry: `is_recoverable_heal()` (error.rs:83-136) ≤3 attempts, 2^n backoff capped at 30s; retries hold ownership inside a standalone backoff task (:3235-3382).
- Completion states are retained for 10 minutes for querying (:42).
### 2.7 Admin API and cluster coordination
- Routes (`rustfs/src/admin/handlers/heal.rs:174-212`): `POST /rustfs/admin/v3/heal/`, `/heal/{bucket}`, `/heal/{bucket}/{prefix}` (the same POST distinguishes start/query/cancel by the query `clientToken/forceStart/forceStop`, aligned with mc admin heal semantics); `POST /v3/background-heal/status`; `GET /v4/heal/replacement-recovery`. Permission `HealAdminAction` (route_policy.rs:334-341).
- Cluster coordination (heal.rs:771-930 + `node_service.rs:514-606`): `heal_topology_fingerprint` + deterministic-by-topology coordinator-node selection + coordinator epoch; envelope validation + SHA256 digest replay protection; when the coordinator is not local, go through peer gRPC `heal_control`; `probe_heal_control` capability probe (rolling-upgrade scenario).
- Request: the body is `HealOpts` (`recursive/dryRun/remove/recreate/scanMode(0/1/2)/updateParity/nolock/pool/set`, serde camelCase, fields aligned with madmin.HealOpts); a root heal start requires `recursive=true` or a `pool+set` pair; body cap 1MB.
- Response: `HealStartSuccess{clientToken, clientAddress, startTime}`; `HealTaskStatus{summary, detail, startTime, settings, items, truncated, progress}` (summary ∈ running/finished/stopped/notFound); `BackgroundHealStatus` (bitrot start time/cycle/current mode + `disabled/uninitialized/idle/active/degraded` states — an unreachable peer is explicitly degraded rather than impersonating idle, issue #5850) + `healOperations` as a priority×source matrix + cluster progress.
- `HealResultItem`/`HealDriveInfo`/`HealItemType`/DriveState enums are JSON-compatible with madmin (`crates/madmin/src/heal_commands.rs:19-65`).
- A status payload over 8MiB is truncated by halving (channel.rs:37,73-104); path-token validation (wrong token rejected; an empty path matches Cluster only).
### 2.8 heal metrics and logs
Metrics: `rustfs_heal_admission_total{source,result,reason,context}`, `rustfs_heal_task_start_total`, `rustfs_heal_task_running{type,set}`, `rustfs_heal_queue_delay_seconds`, `rustfs_heal_scheduler_skip_total`, `rustfs_heal_mainline_throttle_total`, `rustfs_heal_page_concurrency_current{set}`, `rustfs_heal_candidate_enqueue/merge/drop/priority_reject_total`, `rustfs_heal_read_repair_dedup_total{reason}`, etc. All logs are structured event style (PR #5720); per-object logs are demoted to prevent storms (`demote_to_debug_when!`, #5716/#5719/#5727).
---
## 3. Scanner implemented-feature panorama
### 3.1 Loop, leader, immediate triggering
- **Cluster single leader**: distributed ns write lock `leader.lock` (`scanner.rs:3156-3207`, timeout default 5s) + **persisted leader-epoch CAS fence**: the leader writes (cycle, leader_epoch) encoded as `RSCYC001` into `.bloomcycle.bin` using an ETag precondition (`scanner.rs:118,1850-1861,2177-2334`); usage snapshots additionally carry an epoch fence (:2087-2153). Lock lost → cancel the current cycle, converging within 30s (:108-111,2623-2642).
- One round executes immediately after the lock is acquired; cycle = `RUSTFS_SCANNER_CYCLE` > config cycle > start_delay > deployment default > speed tier (±10% jitter, floor 1s).
- **clean-idle exponential backoff**: consecutive fully-clean idle intervals double (capped at 24h; bitrot-cycle compression cap; disabled when a bucket has active lifecycle/replication rules, :383-456,1382-1512).
- **superseded/deferred backoff**: exponential backoff from 5s capped at 30min (:105-106,3432-3438); maintenance probing failures get an independent backoff (:459-505).
- **Immediate wakeup**: ① dirty-usage fast path — write-path put/delete/multipart/bucket operations call `record_dirty_usage_bucket` (`scanner_io.rs:222-235`; call sites include `rustfs/src/app/object_usecase.rs:6221`), bump the generation and Notify-wake the leader; dirty buckets are queued first (`scanner_io.rs:462-488`); ② maintenance-config changes (lifecycle/replication settings call `record_scanner_maintenance_change`); ③ runtime-config hot updates generation+Notify; ④ cluster activity snapshot changes.
- **Cluster coordination**: `probe_scanner_activity` gathers this node's and peers' `ScannerNodeActivity` (instance_id/namespace_generation/maintenance_generation/protocol_version/topology_digest/data_movement_active/dirty usage); the topology digest covers pools/sets/drives URLs; a mismatched protocol version refuses to share the cache lock (`scanner.rs:970-1068`); **cycles are deferred during data movement (rebalance/decommission)** (`scanner_io.rs:2226-2374`); at cycle end, per-peer RPC confirms the dirty-usage ack (`scanner.rs:2925-2952`).
### 3.2 Traversal model
- The main traversal is a **full directory walk** (tokio::fs::read_dir recursion, `scanner_folder.rs:1915-2234`), not via metacache; metacache/`list_path_raw` is used only for the abandoned-children cross-drive verification (:2528-2792).
- Three-level concurrency: leader → per-set (semaphore default 4) → per-disk bucket scans (default 4) → single-drive recursion; a cache lock per bucket per set `.scanner-cycle.lock.pool-N.set-M` (losing the lock cancels that bucket's scan; lock contention re-queues); single-scan admission per drive (local drives also go through the semaphore, `scanner_io.rs:3246-3274`).
- Bucket ordering: after shuffle, re-ordered as dirty → uncached → cached (`scanner_io.rs:2947-2949,462-488`); entries within a directory sorted by name + resume-hint rotation (`scanner_folder.rs:333-359`).
- **Resumable scanning**: `DataUsageScanCheckpoint{version,resume_after,reason}` persisted in the cache info (`data_usage_define.rs:68,293-307`); written on budget exhaustion/cancel; resumption has Used/Stale/NoHint metrics; the resume unit is a directory (no cross-cycle object-level pagination).
- Erasure semantics: finding `xl.meta` marks an object boundary with no descent; at most 64 UUID data-dir candidate entries probed; data without metadata → record failed + high-priority heal; symlink directories ignored / cycles skipped.
- Cooperative yielding: `yield_now` every N objects (default 128).
### 3.3 Large-bucket skip strategy (benchmarked against MinIO compaction)
1. Cache-currency reuse: if the bucket and scan plan are unchanged (name/source/snapshot_complete/plan digest/next_cycle/leader_epoch/cache_key_format all match), the whole bucket is skipped (`scanner_io.rs:1062-1109`).
2. compacted-directory 16-cycle rotation window: rescan only when `hash mod (next_cycle, 16)` hits, otherwise copy from the old cache (`scanner_folder.rs:74,2429-2442`).
3. compaction thresholds: children <500 or pure-object leaves compress into a single entry; subfolders ≥2500 (root 10000) pre-compressed; children ≥10000 reduced (:75-78,2314-2340,2846-2887).
4. failed-object TTL skip: 86400s / at most 10,000 entries (:88-91,1354-1381).
Compared with MinIO master: MinIO's skip strategy is likewise hash-mod-16 cycles + a compaction threshold tree (500/10000/2500), and the **bloom filter has been removed from master**. RustFS's constants and structure share the same origin as MinIO's current state (MinIO does not adopt cross-drive dirty-generation prioritization; RustFS additionally has two more skip layers — plan digest and cache-currency validation).
### 3.4 data usage statistics
- Dimensions: per-directory entry (size/objects/versions/delete_markers/size histogram/version histogram/replication stats/failed_objects/per-tier stats/children/compacted, `data-usage/src/data_usage.rs:661-679`); per-object SizeSummary (incl. per-ARN replication-target stats and tier stats; tier classification: fully transitioned counts toward its tier, otherwise by storage class; free versions not counted); bucket-level `BucketUsageInfo`; cluster-level `DataUsageInfo` (incl. scanner_cycle/scanner_epoch fence + usage_snapshot_complete).
- Storage: per bucket per set `{bucket}/.usage-cache.bin` (primary + `.bkp` backup + CAS retry); the authoritative cluster snapshot `buckets/data-usage/data-usage.json` (`.bkp` synced every 10 cycles, legacy path compatible); stale snapshots rejected on write (triple epoch/cycle/last_update determination); observation snapshots superseded by a race are stored separately as `data-usage-observed.json`.
- Consumption: `replace_bucket_usage_memory_from_info` refreshes bucket-usage memory + two-level cache invalidation (`scanner.rs:4142-4152`) → bucket stats/quota/admin account_info/system; the write path overlays memory in real time; at startup, reading the snapshot detects a cold cache and skips startup delay.
- Incomplete multipart uploads are not counted (consistent with MinIO, which also does not scan the multipart bucket).
### 3.5 ILM integration
- Per object `ScannerItem::apply_actions` (`scanner_folder.rs:747-1032`): `Evaluator::new(lifecycle).with_lock_retention(...).with_replication_config(...).eval()` batch evaluation.
- Implemented actions (the full IlmAction set, `common/src/metrics.rs:34-45`): expiry deletes (Delete/DeleteRestored/DeleteRestoredVersion), all-versions deletes (DeleteAllVersions/DelMarkerDeleteAllVersions, stop further versions after handling), transition (Transition/TransitionVersion, tier list read at runtime), noncurrent batches (DeleteVersionAction → `enqueue_by_newer_noncurrent`), free-version cleanup (`enqueue_free_version`), object-lock retention constraints. **A one-to-one mapping onto MinIO's 9 ILM actions.**
- Execution model: the scanner is the "discover and enqueue" role (the expiry/transition queues live in ecstore `bucket_lifecycle_ops.rs`); actions are consumed by worker pools — the same shape as MinIO's globalExpiryState/globalTransitionState.
- AbortIncompleteMultipartUpload is not executed inside scanner/ILM (MinIO likewise: `internal/bucket/lifecycle/rule.go` has a FIXME, and it is actually carried by the `erasureSets.cleanupStaleUploads` global routine); in RustFS it is an independent ecstore background task `init_background_stale_multipart_upload_cleanup` (`bucket_lifecycle_ops.rs:3289-3320`) + on-demand at bucket deletion.
- Integration-test coverage: transition+restore, free-version, noncurrent, delete-marker, 0-day, background-scan expiry (`scanner/tests/lifecycle_integration_test.rs:1071-2095`).
### 3.6 heal candidate production (scanner side)
- Sampling: `hash mod_alt(next_cycle/prob_div, 1024/prob_div)`; when rescanning via the compacted branch, prob_div=16 gives an equivalent ×16 probability (the same compensation as MinIO, `scanner_folder.rs:125-127,2117-2122`).
- deep/normal: cycle-level `get_cycle_scan_mode` (bitrot_cycle default 30d, `scanner.rs:1626-1657`) → object-level with `HealScanMode::Deep`; fresh objects (modified within 60s) are demoted to Normal (:146-155); state persisted in `.background-heal.json` (`BackgroundHealInfo{bitrot_start_time,bitrot_start_cycle,current_scan_mode}`, same path and structure as MinIO).
- The scanner only enqueues, never executes inline (inline heal was removed; the compat flag only warns, `scanner_folder.rs:411-427`); `HealScanMode::Deep` is just a marker — the bitrot-verification read happens at the heal consumer (the ecstore Deep path).
- Metadata corruption → high-priority heal (`classify_get_size_failure` → HealMetadata); abandoned children → list_path_raw quorum verification + bucket-level/object-level high-priority heal; healing drives get sticky skipping (`should_heal` :1628-1648).
- pending-heal ledger: candidates rejected by a full heal channel are persisted into the cache info and retried next round.
- Replication heal: `queue_replication_heal` → the replication queue (going through the replication channel, not the heal channel); per-ARN replication usage statistics.
### 3.7 remote_scanner RPC protocol (RustFS-specific)
Requests ≤16KB msgpack (version/request_id/server_epoch/session_id/session_sequence/bucket/next_cycle/leader_epoch/scan_plan_digest/skip_healing/scan_mode/budget); frames ≤2MB, HMAC-SHA256 per-frame authentication (domain `rustfs-ns-scanner-frame-v3`); progress heartbeats 1s (250ms in budget mode); phase announcements Scanning→Persisting; RPC lifetime cap 24h, disconnect grace 2min; anti-replay session+sequence cache (capacity 65536); the server validates leader-fence and persisted-cycle consistency + fence re-validation every 5s; results Complete/Partial/NamespaceNotFound/CycleAhead; remote drives without v4-protocol support fall back to the leader scanning locally (`remote_scanner.rs` whole file; `scanner_io.rs:2750-2812`).
### 3.8 Rate limiting / budgets / hot updates / observability
- DynamicSleeper proportional backoff (speed tiers fastest/fast/default/slow/slowest, same five-tier parameters as MinIO); idle_mode master switch; an extra backoff capped at 250ms per request (10ms base) driven by foreground S3 read traffic.
- Cycle budget ScannerCycleBudget: max_duration/max_objects/max_directories (default 0 = unlimited); partial cycles still advance the cycle count.
- runtime_config with three-layer sources (env > config > default) and per-field source markers (Env/Config/ScannerCompatConfig/Default); admin `PUT /v3/config` hot update → generation+Notify takes effect immediately; `GET /v3/scanner/status` returns enabled/freshness(fresh/stale/unknown)/metrics/cycle_schedule/runtime_config; `GET /v3/ilm/expiry/status` returns expiry queue/workers/missed/blocked.
- Metrics: leader lock; cycle complete/partial/deferred/superseded; versions scanned; per-source (Usage/Lifecycle/BucketReplication/SiteReplication/Heal/Bitrot/Alerts) checked/executed/queued/missed; checkpoint set/used/stale; current path (per-disk+bucket in real time); cache save series; concurrency series; alerts (excess versions/version size/folders).
---
## 4. Item-by-item parity versus MinIO
### 4.1 heal trigger-channel comparison
| MinIO channel | RustFS counterpart | Status |
|---|---|---|
| A. Manual admin heal (healSequence, clientToken/forceStart/forceStop) | heal channel Start/Query/Cancel + cluster coordinator + envelope replay protection | ✅ equivalent and enhanced (cluster routing); sequence-semantics differences in §6 HS-06 |
| B. Resident background heal queue (newBgHealSequence + healRoutine worker pool) | HealManager resident scheduler + priority queue + bulkhead | ✅ equivalent and enhanced |
| C. Automatic new/replaced-drive resync (monitorLocalDisksAndHeal 10s + healFreshDisk + healingTracker + waitForFormatErasure handshake) | auto disk scanner (10s) + replacement_readiness + durable intent/proof state machine + heal_replacement_format | ✅ equivalent and enhanced (identity fence + completion proof; MinIO's tracker is stronger on external visibility, see §6 HS-07) |
| D. MRF (100k queue + persisted list.bin + shutdown replay + read-path corrupt delivery) | read-repair (Low + TTL dedup) + write-path convergence heal carry it partially; the `HealType::MRF` executor has no production entry | ⚠️ partially equivalent (§6 HS-01) |
| E. Scanner sampled heal (1/1024 + compacted ×16 compensation) + abandoned children | the same sampling + ×16 compensation + abandoned children + pending-heal ledger | ✅ equivalent and enhanced (the ledger) |
| F. Read-path inline trigger → MRF (GetObject part missing/corrupt, metadata rebuild missingBlocks>0) | read repair (three entries: missing_shards/decode_error/metadata_read_error) | ✅ equivalent (enqueued into the heal queue rather than the MRF queue) |
### 4.2 Object-level heal semantics comparison
| Feature | MinIO | RustFS | Status |
|---|---|---|---|
| mod-time quorum arbitration | listOnlineDisks | same | ✅ |
| ETag majority fallback (clock drift) | filterDisksByETag | `filter_by_etag`/`quorum_etag` (heal.rs:525-567) | ✅ verified first-hand |
| cannotHeal ETag waiver | waived on all-consistent ETag retry | heal.rs:679 | ✅ |
| Normal=CheckParts (stat) / Deep=VerifyFile (bitrot) | yes | `disks_with_all_parts` by scan_mode (ops/heal.rs:562-572,978-1024) | ✅ |
| Normal detecting corrupt auto-escalates to one Deep retry | erasure-healing.go:1101-1106 | ops/heal.rs:2022-2031 | ✅ |
| dangling determination (not-found > parity) + deletion auditing | isObjectDangling/deleteIfDangling | `dangling_delete_safety` (:1488) + scanner HEAL_DELETE_DANGLING | ✅ (audit-tags details differ) |
| Orphan data-dir/inline cleanup (CleanAbandonedData) | CheckAbandonedParts (invoked explicitly on scanner sampling + admin Remove) | in-heal-path `reclaim_orphan_data_dirs_best_effort` (:1428); standalone API NotImplemented at all three layers | ⚠️ partially equivalent (§6 HS-02) |
| Versioned/delete-marker heal | HealObject versionID; nullVersionID special case | per-version enumeration + delete-marker latest heal (B5 regression) | ✅ |
| Object-level healing metadata marker (x-minio-healing, RenameData skips version cleanup) | yes | no object-level marker; relies on drive-level healing.bin + NSLock + rename semantics | ⚠️ evaluation item (§6 HS-12) |
| Distribution/Index consistency, three lines of defense | yes (manual modification rejected) | target-drive format results all-ok check + identity fence | ✅ (different granularity) |
| no-parity (EC:0) objects | bitrot treated as unrecoverable | judged unrecoverable (:700-726) + write self-verification | ✅ enhanced (write-path self-verification) |
| three-layer distribution inconsistency refuses heal | yes | heal_walk normalization + page-bound defense | ✅ (different implementation approach) |
| multipart orphan reconciliation | carried by CheckAbandonedParts | explicitly NotImplemented (carried by lifecycle cleanup) | ⚠️ §6 HS-02 |
| suspended/decommissioned pool handling | skipped via IsSuspended | deferral semantics (store/heal.rs:192-207, PR #5876) | ✅ |
| heal mutually exclusive with concurrent deletes | NSLock + healing marker | NSLock + write lock | ✅ |
### 4.3 new-drive resync comparison
| MinIO | RustFS | Status |
|---|---|---|
| waitForFormatErasure handshake waiting indefinitely on four classes of recoverable errors | startup drive resolution + renew_disk reconnect path | ✅ (different model: RustFS does not block at startup waiting for format) |
| HealFormat NSLock + errNoHealRequired + refFormat-mismatch rejection | `heal_format`/`heal_replacement_format` fail-closed + target-slot restriction (PR #1787 semantics) | ✅ enhanced |
| per (pool,set) distributed lock preventing concurrent resync | set-level queue dedup + bulkhead (manager.rs:2854-2889) | ✅ |
| brand-new-cluster detection (drives-to-heal == total drives does not trigger) | replacement_readiness (independent mount point / physical-device validation, non-root) | ✅ enhanced |
| healingTracker (.healing.bin: Bytes/Items counters, QueuedBuckets/HealedBuckets, Resume snapshot, RetryAttempts ≤4, HealID linkage, diskID-change reset) | resume/checkpoint schema'd persistence + durable intent/proof (per-task files, CAS) | ✅ equivalent and enhanced (crash-window backfill); but **external snapshot visibility** is weaker than MinIO's (§6 HS-07) |
| skip versions written after heal start (ModTime > Started) | no such filter | ⚠️ §6 HS-13 |
| skip ILM-expired versions (filterLifecycle) | no such filter | ⚠️ §6 HS-13 |
| worker count max(GOMAXPROCS,NR)/4 floor 4, heal:drive_workers override | in-page concurrency 8 (Deep/AutoHeal forced to 1) + per-set bulkhead | ✅ (different parameter model) |
| waitForLowHTTPReq yield per entry | mainline throttle (foreground-utilization gating) | ✅ enhanced |
| heal scope includes the two pseudo-buckets `.minio.sys/config` and `.minio.sys/buckets`; newest bucket first | ErasureSet task pre-processes per bucket (meta-bucket semantics carried by heal_bucket) | ✅ (no "newest first" ordering) |
| whole-failure retry ≤4 (resetHealing + errRetryHealing) | schedule_retry resets both layers + recoverable retry ≤3 | ✅ |
### 4.4 scanner comparison
| MinIO | RustFS | Status |
|---|---|---|
| cluster single leader (globalLeaderLock) | leader.lock + persisted leader-epoch CAS fence | ✅ enhanced (epoch fence against split-brain; MinIO has no persisted epoch) |
| `.bloomcycle.bin` stores only the cycle (bloom removed) | same path stores cycle+leader_epoch (RSCYC001) | ✅ aligned (v1 misjudgment corrected) |
| folderScanner hash-mod-16 + compaction (500/10000/2500) | same constants + plan digest + cache-currency validation + dirty-first | ✅ enhanced |
| ≤GOMAXPROCS parallel scans per drive; healing drives excluded | per-set/per-disk semaphores + sticky skip of healing drives | ✅ |
| scannerSleeper (factor 2/max 1s, speed tiers hot-swapped) | DynamicSleeper same + idle_mode + foreground-read backoff | ✅ enhanced |
| idle semantics: `scanner:idle_speed=on` (throttle only in idle windows, full speed when busy) | `RUSTFS_SCANNER_IDLE_MODE=true` (master switch for rate limiting) | ⚠️ opposite semantic direction, §6 HS-14 |
| applyActions order (heal→ILM→replication→alerts) | apply_actions same order (heal candidates→ILM→replication heal→alerts) | ✅ |
| ILM 9 actions + batch evaluation + DeletePrefixObject optimization | same 9 actions + batch evaluation + expiry queue | ✅ (whether DeleteAllVersions has the single-call optimization was not checked line by line) |
| abandoned children (listPathRaw minDisks=N/2 detects under-written drives) | list_path_raw + quorum verification + high-priority heal | ✅ |
| incomplete multipart independent routine (6h interval/24h expiry, rename into .trash) | ecstore independent background task (configurable interval/expiry) | ✅ (trash two-stage cleanup detail differences, §6 HS-18) |
| usage dimensions (size/objects/versions/DM/histograms/replication/tier/bucket level) | full coverage + cluster snapshot with triple anti-rollback | ✅ enhanced |
| prefix-level usage (loadPrefixUsageFromBackend, consumed by console) | the cache holds the directory tree but flattens only to bucket level | ❌ §6 HS-08 |
| excess events s3:ObjectManyVersions/LargeVersions/PrefixManyFolders + auditing | metrics alert_excess_* only (defaults 100/1TiB/65538 vs MinIO 100/1TB/50000) | ⚠️ §6 HS-04/HS-17 |
| scanner metrics v3 (bucket_scans/directories/objects/versions/last_activity) | full rustfs_scanner_* suite + freshness | ✅ (different naming scheme) |
| TraceScanner / realtime metrics (mc admin scanner status/trace) | no trace channel; /v3/scanner/status has its own structure | ⚠️ §6 HS-03 |
### 4.5 admin/CLI/API surface comparison
| MinIO | RustFS | Status |
|---|---|---|
| `POST /minio/admin/v3/heal/...` start/status/cancel | `POST /rustfs/admin/v3/heal/...` same three states | ✅ (different path prefix is expected) |
| `HealStartSuccess`/`HealTaskStatus`/`HealResultItem`/DriveState | same-named fields JSON-compatible | ✅ |
| `POST /v3/background-heal/status` (BgHealState aggregate) | same path + degraded semantics + operations matrix | ✅ enhanced (no MRF per-endpoint sub-state, because there is no MRF) |
| `GET /v3/healthinfo` per-drive `HealInfo *HealingDisk` | no equivalent healthinfo heal field (replacement-recovery v4 covers part of it) | ⚠️ §6 HS-07 |
| madmin client HealStart/HealStatus/BackgroundHealStatus/ScannerStatus methods | wire types only, no client methods | ❌ §6 HS-05 |
| mc admin heal --pool/--set, --scan-mode, --force-start/stop | HealOpts full field support (pool/set/scanMode/forceStart/forceStop) | ✅ (server-side ready; missing the mc-side entry, HS-05) |
| ErrHealAlreadyRunning / ErrHealOverlappingPaths typed errors | dedup-merge + eviction semantics; no typed overlap rejection | ⚠️ §6 HS-06 |
| result backpressure (maxUnconsumedItems=1000, 10s keep-alive streaming, 24h unconsumed abort) | snapshot-style query (1024 entries + 8MiB truncation + 10min retention) | ⚠️ §6 HS-06 |
| `mc support inspect`/healing-bin offline dump | none (inspect.rs exists but the healing dump is unconfirmed) | ⚠️ P3 |
### 4.6 observability surface comparison
| Dimension | MinIO | RustFS | Status |
|---|---|---|---|
| heal metrics | minio_heal_objects_total/heal_total/errors_total/time_last_activity + v3 drive_health 2=healing | full rustfs_heal_* suite (admission/queue delay/running/throttle/page concurrency) | ✅ (RustFS lacks an equivalent of the single drive_health=healing gauge; DiskInfo.healing is already assigned) |
| scanner metrics | v3 6 + realtime 18 items | full rustfs_scanner_* suite + per-source dimensions | ✅ |
| ILM metrics | v3 5 (expiry/transition pending/active/missed + versions_scanned) | ilm expiry status API + scanner per-source | ✅ (different metrics and API shape) |
| trace | TraceHealing/TraceScanner channels | none | ❌ §6 HS-03 |
| auditing | HealObject events, dangling-deletion audit, scanner:manyversions etc. | structured logs (event style) + metrics; no audit-log events | ⚠️ §6 HS-04 |
| progress | healingTracker Bytes/Items/QueuedBuckets/current object + usage-cache total baseline | HealProgress{scanned/healed/failed/bytes/current_object/percentage}; bytes_processed annotated as 0, estimated_completion_time always None | ⚠️ §6 HS-07 |
### 4.7 configuration surface comparison (defaults)
| MinIO | RustFS | Notes |
|---|---|---|
| `heal:bitrotscan` (default off; on=every cycle; Nm=N×30×24h) | `heal.bitrot_cycle` / `RUSTFS_SCANNER_BITROT_CYCLE_SECS` (default 30d=2592000s; 0/on=Deep every cycle, off=disabled) | ✅ same semantics (RustFS default 30d, MinIO default off — **different defaults**, RustFS more aggressive) |
| `heal:max_io=100`/`max_sleep=250ms` (waitForLowIO) | mainline throttle thresholds 80%/80%, max_sleep 250ms | ✅ same shape (different threshold model) |
| `heal:drive_workers` (default -1 auto) | in-page concurrency 8 + per-set 1 | ✅ same shape |
| `_MINIO_HEAL_WORKERS` (GOMAXPROCS/2) | `RUSTFS_HEAL_MAX_CONCURRENT_HEALS=4` + `_MAX_CONCURRENT_PER_SET=1` | ✅ |
| `_MINIO_AUTO_DRIVE_HEALING` (on) | `RUSTFS_HEAL_AUTO_HEAL_ENABLE=true` | ✅ |
| `_MINIO_SCANNER` (on) | `RUSTFS_SCANNER_ENABLED=true` | ✅ |
| `scanner:speed` five tiers (default=2x/1s/1m) | same five tiers, same names, same parameters | ✅ |
| `scanner:idle_speed` (on) | `RUSTFS_SCANNER_IDLE_MODE` (true) | ⚠️ semantic direction (HS-14) |
| `scanner:alert_excess_versions=100` | 100 | ✅ |
| `scanner:alert_excess_folders=50000` | 65538 (compatible with the PBS layout) | ⚠️ HS-17 |
| `ilm:expiration_workers=100`/`transition_workers=100` | ecstore expiry/transition worker pools (keys under the ilm subsystem) | ✅ (defaults not checked item by item) |
| `api:stale_upload_cleanup_interval=6h`/`expiry=24h` | ecstore background task, configurable via env | ✅ (defaults not checked item by item) |
| — (none) | `RUSTFS_HEAL_QUEUE_SIZE=10000`, `_TASK_TIMEOUT_SECS=300`, `_INTERVAL_SECS=10`, `_LOW_PRIORITY_MERGE/DROP`, `_PAGE_*`, `_SET_BULKHEAD`, `_MAINLINE_*`, `RUSTFS_SCANNER_CYCLE_MAX_*` budgets, `_MAX_CONCURRENT_SET/DISK_SCANS=4`, `_YIELD_EVERY_N_OBJECTS=128`, etc. | RustFS-specific (finer-grained) |
### 4.8 Where RustFS exceeds MinIO
1. remote_scanner RPC (scan execution pushed down to the remote peer locally, with HMAC authentication/replay cache/fence re-validation/disconnect grace).
2. Persisted leader-epoch CAS fence + usage-snapshot epoch/cycle anti-rollback (MinIO has only the lock, no persisted epoch).
3. Cycle budgets (max_duration/objects/directories) + partial-cycle advancement semantics.
4. per-set/per-disk scan concurrency gates + a cache lock per bucket per set.
5. pending-heal ledger (heal candidates are not lost when the heal channel is full).
6. Drive-replacement durable intent + completion proof state machine + identity fence (MinIO's healingTracker has no proof).
7. mainline throttle foreground pressure gating (driven by permit utilization).
8. Cluster heal control coordinator + envelope replay protection + explicit degraded fallback.
9. Write-path shard bitrot self-verification (the EC:0 case).
10. dirty-usage fast-path wakeup (immediate write-path notification + dirty buckets first).
11. heal runtime observability matrix (priority×source operations snapshot).
12. workload admission integration (the heal scheduler reads the foreground pressure snapshot).
---
## 5. Gap and improvement list
Severity definitions: P1 = behavioral/operational alignment gap (affects production operations or toolchain compatibility); P2 = completeness (the feature exists but is missing a corner); P3 = cleanup/low risk. Each item includes current-state evidence, MinIO behavior, impact, recommendation, and acceptance.
### P1 (8 items)
**HS-01 The MRF/ECDecode/Metadata heal task types have no production trigger; HealEvent unwired**
- Current state: the `HealType::MRF/ECDecode/Metadata` executors are complete (task.rs:1700-2156) but have no production trigger anywhere in the repo; `HealEvent`/`HealEventHandler` (event.rs:50-367) has zero references outside the crate (verified first-hand by grep); channel conversion produces only Cluster/Object/Bucket/Prefix/ErasureSet (channel.rs:566-601).
- MinIO: mrf.go has a standalone MRF queue (capacity 100k, drop-and-count when full), msgp persistence to `.heal/mrf/list.bin` at process exit + startup replay, 1s delay for enqueues <1s (waiting for network recovery), healSleeper rate limiting; on the read path, GetObject part missing/corrupt, metadata rebuild missingBlocks>0, partial Put success, DeleteObject, multipart, and the peer client add up to 7+ delivery points.
- Impact: RustFS's read-repair + write-path convergence covers the main scenarios, but lacks: ① an event-driven Urgent ECDecode rebuild entry (on ecstore decode failure there is currently only Low read-repair); ② a metadata-only heal entry (the scanner's HealMetadata classification exists but goes through ordinary object heal); ③ MRF queue persistence (unconsumed repair intents are lost on restart — partially mitigated by the scanner's pending-heal ledger).
- Recommendation: a pick-one-of-three decision — (a) wire HealEvent (emit events at ecstore decode-failure/metadata-corruption points) + implement a persistent retry ledger; (b) delete the MRF/ECDecode/Metadata dead code and keep only a documentation note; (c) keep the executors and demote HealEvent to an internal API. (a) is recommended, but first quantify whether read-repair already meets the response-time requirements for decode-failure scenarios.
- Acceptance: an e2e decode-failure → Urgent heal-request chain; replay of pending repair intents after restart; HealEvent ring-buffer metrics.
**HS-02 CheckAbandonedParts NotImplemented at all three layers (missing standalone abandoned-data reconciliation entry)**
- Current state: `set_disk/ops/heal.rs:2052-2056`, `core/sets.rs:1144-1148`, `store/heal.rs:258-266` explicitly return `Err(NotImplemented)` at all three layers (verified first-hand); the comment reads "intentionally retained above the set layer until there is a concrete caller".
- MinIO: `CheckAbandonedParts` → per-drive `CleanAbandonedData`: read xl.meta → list UUID data-dirs + inline entries → diff against getDataDirs → delete surplus data-dirs/inline entries and rewrite xl.meta; invoked explicitly on scanner-sampled heals and admin heal Remove.
- Impact: RustFS's in-heal-path `reclaim_orphan_data_dirs_best_effort` (:1428) covers "reclaim orphan directories while healing", but ① there is no standalone trigger point (MinIO can also clean abandoned data before an object reaches the heal threshold); ② orphan inline-data entry cleanup is unconfirmed; ③ multipart orphan reconciliation is explicitly out of scope (a design decision, carried by lifecycle).
- Recommendation: evaluate promoting `reclaim_orphan_data_dirs_best_effort` to a fixed step of heal_object (if it is not already) + implement a real HealOperations::check_abandoned_parts (calling the same reclamation logic), or explicitly document "carried by lifecycle" and close the API surface.
- Acceptance: construct data-dir/inline orphans → cleaned after scanner sampling/admin heal; the three-layer API returns success or an explicitly documented NotSupported.
**HS-03 heal/scanner trace channels missing**
- Current state: zero hits for TraceHealing/TraceScanner (verified first-hand by grepping the whole repo).
- MinIO: `madmin.TraceHealing` (mc admin trace --healing, FuncName=heal.Bucket/heal.Object/heal.CheckAbandonedParts, with dry/remove/mode/version-id/disks/bytes), `TraceScanner` (mc admin scanner trace, supports --filter-size/--response-duration).
- Impact: no way to observe in real time the latency and parameters of individual heal/scanner actions; troubleshooting can rely only on aggregated metrics and logs.
- Recommendation: instrument heal-channel execution and scanner folder/item handling, and hook them into the existing admin trace subscription surface (reuse the rustfs trace infrastructure if it exists; otherwise extend it per madmin TraceType).
- Acceptance: an mc-equivalent tool can subscribe to the heal/scanner trace stream.
**HS-04 Scanner excess S3 events and auditing missing**
- Current state: only `rustfs_scanner_excess_*_total` metrics (versions 100 / version size 1TiB / folders 65538).
- MinIO: emits `s3:ObjectManyVersions` (>100 versions), `s3:ObjectLargeVersions` (cumulative >1TB), `s3:PrefixManyFolders` (>50000 subdirectories) events (UserAgent: Scanner) + scanner:manyversions/largeversions/manyprefixes auditing.
- Impact: users relying on event subscriptions for capacity governance (console/external auditing) receive no alerts.
- Recommendation: hook the scanner_folder alert points into notify event publishing (reusing the lifecycle event-channel semantics).
- Acceptance: after configuring bucket notifications, an over-threshold object triggers an event.
**HS-05 madmin client methods missing**
- Current state: `crates/madmin/src/heal_commands.rs` has only wire types (HealDriveInfo/Infos/HealResultItem); no HealStart/HealStatus/BackgroundHealStatus/ScannerStatus client methods.
- MinIO: madmin-go provides the full client; mc admin heal/scanner/status/trace are all built on it.
- Impact: admin tools like mc cannot directly drive the RustFS heal/scanner admin surface; automated operations must hand-write HTTP.
- Recommendation: add the client following the madmin-go interface shape (the server side is ready; this is pure client work).
- Acceptance: complete the start→query→cancel flow with the madmin client.
**HS-06 admin heal sequence semantics differ from MinIO**
- Current state: duplicate/overlapping requests are dedup-merged (returning the canonical task_id) or evicted; no ErrHealAlreadyRunning/ErrHealOverlappingPaths typed errors (verified first-hand: manager.rs:1309's already_running is an idempotent-startup guard, not an admin semantic); results are snapshot-style queries (1024 entries/8MiB truncation/10min retention), not MinIO's streaming increments (clientToken pulls increments + maxUnconsumedItems=1000 backpressure + 10s keep-alive + 24h unconsumed abort).
- Impact: mc admin heal's interaction model (long connection pulling increments) behaves against RustFS as multiple snapshot polls; automation scripts cannot easily distinguish "merged" from "newly started".
- Recommendation: ① incremental semantics: channel query supports item increments since the last clientToken (or a cursor); ② overlapping requests return a typed error code (or an explicit merged_into field in the receipt — the existing alias mechanism already provides the base); ③ verify forceStart's stop-old-then-start-new semantics.
- Acceptance: an madmin-compatible client polling in the MinIO style can retrieve the full item set.
**HS-07 healing progress and drive-level healing state insufficiently visible externally**
- Current state: byte-recovery progress `progress.bytes_processed = 0 // set to 0 for now` (erasure_healer.rs:967); `HealProgress::estimated_completion_time` is always None and `HealStatistics::add_healed_objects` is never written (progress.rs:38,135-139 zero calls); healthinfo has no per-drive HealInfo equivalent (MinIO HealingDisk: BytesDone/Failed/Skipped, ObjectsTotal baseline, QueuedBuckets/HealedBuckets, Resume snapshot, current object); v3 metrics lack an equivalent of the single drive_health=2 (healing) gauge.
- Impact: during a drive rebuild (potentially hours to days) operations cannot answer "where are we / how much is left / when will it finish".
- Recommendation: ① accumulate bytes in erasure set heal (heal_object already yields the object size); ② read the object-total baseline from usage-cache (the same approach as MinIO); ③ expose a per-drive healing snapshot in admin healthinfo/background status (DiskInfo.healing already exists; add the aggregated exposure); ④ derive the ETA from baseline + rate.
- Acceptance: during a drive rebuild, admin shows byte progress and ETA; an mc info-equivalent output shows the Healing flag.
**HS-08 prefix-level usage not exposed**
- Current state: the DataUsageCache holds the directory-tree entries (organized by hash_path), but `dui()` flattens only to the bucket name (data_usage_define.rs:858-915).
- MinIO: `loadPrefixUsageFromBackend` (30s cache) aggregates prefix usage from each set's `.usage-cache.bin`, consumed by console bucket-prefix statistics.
- Impact: console/front ends cannot show prefix-level usage; there is no API to locate "which prefix is using the space" in a large bucket.
- Recommendation: implement a prefix-flattening query API (the data is already in the cache; this is pure aggregation and exposure work).
- Acceptance: a ListBuckets/PrefixUsage API returns statistics matching the prefix filter.
### P2 (9 items)
**HS-09 get_disk_status always returns Ok (the only TODO)**: `crates/heal/src/heal/storage.rs:930-943` (verified first-hand). Currently no production caller (low risk). Recommendation: delete the method or wire it to the real ecstore disk status (the DiskStatus enum is already defined).
**HS-10 About 1/3 of HealStorageAPI methods are dead code**: get_object_meta/get_object_data/put_object_data/delete_object/verify_object_integrity/ec_decode_rebuild/get_disk_status/format_disk/heal_bucket_metadata/get_object_size/get_object_checksum/list_objects_for_heal (the non-paginated version, with its own memory_heavy warning) all have 0 callers. Recommendation: clean up or wire them together with the HS-01 decision (dead interfaces mislead future maintainers into thinking a call path exists).
**HS-11 bitrot self-test missing**: MinIO at startup runs bitrotSelfTest over known vectors for the four algorithms and exits Fatal on failure (guarding against silent data corruption). RustFS has no equivalent (verified first-hand by grep). Recommendation: at startup, run known-vector self-tests for HighwayHash256S and the other algorithms in use (low cost, high value).
**HS-12 object-level healing metadata marker evaluation**: during heal, MinIO tags objects with `x-minio-healing:true`, and RenameData uses it to skip version cleanup/legacy purge (missing it lets heal and concurrent deletes destroy each other). RustFS has no object-level marker (verified first-hand by grep; object.rs has no healing branch) and relies on NSLock + rename semantics. Recommendation: audit whether the RustFS rename-commit path has a "heal commit racing concurrent delete/version cleanup" window; if not, document the difference, and if so, add a marker-equivalent mechanism.
**HS-13 erasure set heal lacks "skip newly written / ILM-expired versions" filters**: MinIO resync skips versions with ModTime>tracker.Started (so heal does not chase the tail of new writes) and ILM-expired versions (so work is not wasted). RustFS's erasure_healer does not implement such filters (per-version dedup exists; time/ILM filters do not). Impact: a long tail on rebuild completion (the completion decision for a continuously written bucket is pushed out by new versions) and wasted heal work. Recommendation: add a started_at time filter at the disk-walk enumeration point + an evaluator pre-check.
**HS-14 scanner idle semantics point the opposite way from MinIO**: MinIO `scanner:idle_speed=on` (default) means "throttle only when the cluster is idle, full speed when busy"; RustFS `RUSTFS_SCANNER_IDLE_MODE=true` (default) is a master switch for rate limiting (false = never sleep at all). The default behaviors may end up similar (both throttle), but the parameter semantics are not interchangeable; migration docs must state this explicitly; if mc config compatibility is the goal, a rename/re-semantization is needed. Recommendation: document the difference first, then evaluate aligning the semantics.
**HS-15 alert_excess_folders default differs**: RustFS 65538 (compatible with the PBS/Proxmox layout, scanner_folder.rs:79) vs MinIO 50000. The behavioral difference is that the trigger threshold differs out of the box. Recommendation: document it (keeping 65538 has local rationale).
**HS-16 single-node default-cycle hook not enabled**: `single_disk_default_cycle_secs(_features) -> None` is always empty (scanner.rs:1428-1430); single-node deployments get no dedicated default-cycle override. Recommendation: after deciding the single-node default-cycle policy, enable or delete the hook.
**HS-17 DeleteAllVersions batch-optimization check**: MinIO uses the single DeletePrefix+DeletePrefixObject call instead of per-version fan-out. Whether RustFS's expiry-queue path has the same optimization was not verified line by line (integration tests cover behavioral correctness). Recommendation: check the `apply_expiry_rule` all-versions delete path; if there is no prefix single-call optimization, evaluate adding it.
### P3 (3 items)
**HS-18 trash/temp-directory two-stage cleanup detail check**: MinIO cleans `.minio.sys/tmp/.trash` (delete_cleanup_interval default 5m + deleteCleanupSleeper) and stale uploads are renamed into trash in two stages. RustFS has delete_tail_activity.rs and the stale multipart task; whether the two-stage semantics are fully aligned was not verified line by line. Recommendation: align or document.
**HS-19 root-heal direct path is dead code**: `should_handle_root_heal_directly` is always false (admin/handlers/heal.rs:1200-1202, locked by a test); the store.heal_format direct branch is unreachable. Recommendation: delete the dead branch or restore the direct path as a fallback for cluster-coordination failure.
**HS-20 compat flags and dead metrics cleanup**: `RUSTFS_SCANNER_INLINE_HEAL_ENABLE` (enabling only warns) + the dead `rustfs_scanner_inline_heal_total` metric + the scanner-domain code in `rustfs_common::metrics` awaiting layering migration (backlog #1843 already filed). Recommendation: clean up along with the layering migration.
### Not pursuing parity by design (7 items, recorded to prevent later misreading as gaps)
1. **bloom filter**: removed from MinIO master; RustFS reuses `.bloomcycle.bin` as the cycle/epoch fence, consistent with MinIO's current state.
2. **scanner cluster single leader**: both sides agree; RustFS additionally has the epoch fence.
3. **heal emits no S3 bucket notification**: both sides agree (heal results go through admin status).
4. **incomplete multipart not executed inside scanner/ILM**: both sides agree (independent background routine).
5. **inline heal removal**: a deliberate RustFS choice (the scanner only enqueues); MinIO's applyHealing inline path is not a parity target.
6. **heal-sequence resident keep-alive (10s blank write-back)**: RustFS's snapshot-query model differs; handling incremental semantics per HS-06 is enough — do not copy the streaming keep-alive.
7. **`.trash`/`tmp-old` path-name compatibility**: RustFS's layout constants are independent; no literal alignment with MinIO paths.
---
## 6. Configuration defaults master table (RustFS)
heal (env prefix `RUSTFS_HEAL_`, `crates/config/src/constants/heal.rs`, consumed at `manager.rs:724-800`):
| Setting | Default | Hot update |
|---|---|---|
| AUTO_HEAL_ENABLE | true | no |
| QUEUE_SIZE | 10000 | no |
| INTERVAL_SECS | 10 | no (fixed at startup) |
| TASK_TIMEOUT_SECS | 300 | no |
| MAX_CONCURRENT_HEALS | 4 | no |
| MAX_CONCURRENT_PER_SET | 1 (≤min(global, value)) | no |
| LOW_PRIORITY_MERGE_ENABLE | true | no |
| LOW_PRIORITY_DROP_WHEN_FULL | true | no |
| PAGE_OBJECT_CONCURRENCY | 8 (Deep/AutoHeal forced to 1) | no |
| EVENT_DRIVEN_SCHEDULER_ENABLE | true | no |
| SET_BULKHEAD_ENABLE | true | no |
| PAGE_PARALLEL_ENABLE | true | no |
| MAINLINE_THROTTLE_ENABLE | true | no |
| MAINLINE_READ/WRITE_UTILIZATION_HIGH_PERCENT | 80/80 | no |
| MAINLINE_MAX_SLEEP_MS | 250 | no |
| (master switch) RUSTFS_HEAL_ENABLED | true | no |
| admin subsystem heal.bitrot_cycle | 30d | yes (via scanner runtime config) |
scanner (admin subsystem `scanner`, `crates/config/src/constants/scanner.rs` + `ecstore/src/config/scanner.rs` + `runtime_config.rs:527-673`):
| Key | env | Default |
|---|---|---|
| speed | RUSTFS_SCANNER_SPEED | default (2x/1s/60s) |
| delay / max_wait / cycle / start_delay | RUSTFS_SCANNER_* | derived/empty |
| cycle_max_duration/objects/directories | …_MAX_* | 0 (unlimited) |
| bitrot_cycle | …_BITROT_CYCLE_SECS | 2592000 (30d; 0/on=every cycle, off=disabled) |
@@ -1,568 +0,0 @@
# RustFS heal / scanner 全量功能分析与 MinIO 对标(v2)
> English version: [rustfs-heal-scanner-vs-minio-comprehensive-analysis-2026-08-16.md](rustfs-heal-scanner-vs-minio-comprehensive-analysis-2026-08-16.md)
- 日期:2026-08-16(基于 main 分支当日代码,审计时 HEAD ≈ `a118d7e4f`
- 范围:`crates/heal`src 19,560 行 + tests 2,274 行)、`crates/scanner`src 约 26,000 行 + tests)、`crates/data-usage``crates/ecstore` 中 heal/heal_walk/bitrot_self_verify 与 config、`crates/common/src/heal_channel.rs``crates/madmin`heal/scanner wire 类型)、`rustfs/src`startup wiring、admin handlers、集群 RPC
- 对标基线:minio/minio masterHEAD `7aac2a2c5b`,仓库已进入维护模式,master 冻结,即最终态)
- 方法:四路并行审计(heal crate / scanner crate / ecstore 集成层 / MinIO 源码研究),关键结论逐条人工抽验(文内标注"已亲验"处为一手验证)
- 本文档取代 `docs/rustfs-heal-scanner-vs-minio-parity-assessment.md`2026-06-15 v1)。v1 之后 heal/scanner 相关提交超过 80 个(换盘自动修复全链路、resume 状态机、usage 收敛权威化、集群级 heal 协调、ILM restore 语义等),v1 的功能清单与差距判断已全面过时;v1 中"bloom filter 缺失"等结论经本次核实为**误判**(详见 §5.4)。
---
## 0. 结论摘要
1. **总体判断:heal 与 scanner 的核心功能链路已经完整**。对象级 healquorum 仲裁 + ETag 兜底 + bitrot Deep 校验 + dangling 处理)、erasure set 深扫(per-set disk-walk 并集枚举)、按版本断点续扫(schema 化持久层 + CAS 原子发布 + 崩溃窗口补齐)、换盘自动修复(readiness 校验 + 身份围栏 + durable intent + completion proof)、scanner 周期循环(leader lock + 持久化 leader-epoch 围栏)、data usage 统计(桶级/集群级、主+备+观测快照、epoch/cycle 防回退)、ILM 全动作(expiry/transition/noncurrent/free-version/delete-marker 清理)、admin Start/Query/Cancel 协议(clientToken 语义对齐 madmin)——以上均有实现且带回归测试。两个 crate 内**没有空实现/早退桩**,异常路径全部有日志 + 指标 + 错误语义。
2. **主要缺口集中在"入口与观测面",而不是修复算法本身**MRF/ECDecode/Metadata 三类任务执行体已实现但无生产触发入口(`HealEvent` 完全未接线);`CheckAbandonedParts` 在 ecstore 三层全部 `NotImplemented`heal/scanner trace 通道缺失;scanner 超限 S3 事件缺失;madmin 客户端方法缺失(只有 wire 类型);heal 字节级进度/ETA 未实现。
3. **与 v1 认知的重要修正**bloom filter 在 MinIO 当前 master **已删除**`.bloomcycle.bin` 只存 cycle 计数),RustFS 现状与 MinIO 一致;MinIO scanner 同样是**集群级 leader 单例**RustFS 的 leader.lock 模型与 MinIO 同型;RustFS 的 ETag 多数派兜底仲裁已实现(`crates/ecstore/src/set_disk/ops/heal.rs:525-567,679`,已亲验),v1 担心的仲裁缺口不存在。
4. **RustFS 在多处超出 MinIO**remote_scanner RPC 协议(远端 peer 本地扫描而非 leader 跨网读远盘)、持久化 leader-epoch CAS 围栏、周期预算与 per-set/per-disk 并发闸、pending-heal 账本、durable replacement intent + completion proof 状态机、前台压力门控(mainline throttle)、集群 heal control coordinator + envelope 重放防护。
5. 差距分级统计:P1(行为/运维对齐缺口)8 项,P2(完善性)9 项,P3(清理/低风险)3 项,"按设计不追平"7 项。完整清单见 §6。
---
## 1. 架构总览
### 1.1 RustFS 三层架构
RustFS 把 MinIO 在 `cmd/` 内单体的 heal/scanner 拆成三层 + 两个独立 crate:
| 层 | 位置 | 职责 |
|---|---|---|
| 原语层 | `crates/ecstore/src/set_disk/ops/heal.rs`~3,240 行)、`ops/heal_walk.rs``ops/bitrot_self_verify.rs`;上层封装 `store/heal.rs``store/heal_walk.rs``core/sets.rs` | 对象/桶/format/替换盘格式修复、disk-walk 并集枚举、写入路径 bitrot 自校验;由 `SetDisks`/`Sets`/`ECStore` 实现 `rustfs_storage_api::HealOperations` 契约(`crates/storage-api/src/object.rs:503-519` |
| heal 运行时 | `crates/heal` | 进程级 HealManager(优先级队列/调度器/auto disk scanner/断点续传 resume)、HealChannelProcessor(消费全局 heal channel)、换盘替换恢复状态机 |
| scanner 运行时 | `crates/scanner` | 数据使用扫描、ILM 评估与入队、heal 候选生产、复制用量统计、remote scanner RPC |
| 共享协议 | `crates/common/src/heal_channel.rs`~776 行) | Start/Query/Cancel 命令通道、`HealOpts`/`HealScanMode`/`HealRequestSource`/`HealAdmission*` 共享类型、`HealResultItem`madmin |
| 共享数据 | `crates/data-usage` | `DataUsageEntry/Info`、直方图、`hash_path`scanner 产生、ecstore/admin 消费 |
启动链路(已亲验 wiring):
1. `rustfs/src/startup_services.rs:93``init_background_service_runtime(store)`
2. `rustfs/src/startup_background.rs:41-81`:创建全局 heal 服务取消令牌;读 `RUSTFS_SCANNER_ENABLED`(别名 `RUSTFS_ENABLE_SCANNER`,默认 true)与 `RUSTFS_HEAL_ENABLED`(别名 `RUSTFS_ENABLE_HEAL`,默认 true);**只要 heal 或 scanner 任一开启就初始化 heal manager**scanner 产生的 heal 候选需要消费端;两者都关时 heal channel 不初始化,`send_heal_request` 报 "Heal channel not initialized")。
3. `crates/heal/src/lib.rs:142-216`owned task 内原子初始化(caller 取消不会遗留半初始化 manager,`lib.rs:123-131``GLOBAL_HEAL_RUNTIME_INIT` 互斥单飞)→ `HealManager::start()``rustfs_common::heal_channel::init_heal_channels()` → spawn `HealChannelProcessor::start_with_receipts`
4. `crates/heal/src/heal/manager.rs:1301-1356` `HealManager::start``start_scheduler()``manager.rs:2394-2461`interval 默认 10s + `Notify` 事件驱动唤醒)→ `process_unclean_shutdown()``manager.rs:1362-1695`)→ `enable_auto_heal`(默认 true)时 `start_auto_disk_scanner()``manager.rs:2464-2999`)。
5. server ready 后 `rustfs/src/startup_lifecycle.rs:150-152``enable_scanner``init_data_scanner(token, store)``crates/scanner/src/scanner.rs:1293-1372`)。
6. 优雅停机:`rustfs/src/startup_shutdown.rs:308` `shutdown_ahm_services()`(取消令牌);`:414` `clear_unclean_shutdown_markers()`
### 1.2 MinIO 对应结构(master 最终态)
| MinIO 文件 | 职责 |
|---|---|
| `cmd/admin-heal-ops.go` | 手动 admin heal 序列(healSequence、clientToken/forceStart/forceStop |
| `cmd/global-heal.go` | 常驻后台 heal 队列(newBgHealSequencetoken 固定 `0000-…`,永不结束)+ `healErasureSet`(逐 set 全量对象 heal |
| `cmd/background-heal-ops.go` | healRoutine worker 池(`_MINIO_HEAL_WORKERS`,默认 GOMAXPROCS/2)消费 healTask |
| `cmd/mrf.go` | MRFMost Recent Fail)队列(容量 100,000),进程退出时持久化 `.minio.sys/buckets/.heal/mrf/list.bin` 并启动回放 |
| `cmd/background-newdisks-heal-ops.go` | 新盘/换盘自动 resyncmonitorLocalDisksAndHeal 10s 轮询 + healFreshDisk + healingTracker |
| `cmd/erasure-healing.go` / `erasure-healing-common.go` | 对象级 heal 核心(~800 行)、listAndHeal |
| `cmd/data-scanner.go` | scanner 循环(globalLeaderLock 集群单例)+ folderScanner + applyActions |
| `cmd/erasure.go`nsScanner/ `erasure-server-pool.go` | NSScanner 三层结构 |
| `cmd/bucket-lifecycle.go` | ILM 执行器(expiry/transition worker 池) |
| `cmd/xl-storage.go` | DiskInfo.Healing、CheckParts/VerifyFile、CleanAbandonedData、RenameData healing 分支 |
| `cmd/prepare-storage.go` | waitForFormatErasure 新盘启动握手 |
### 1.3 架构级差异(设计取舍,非缺陷)
1. **heal 队列模型**MinIO 所有 healscanner 抽样/MRF/admin/新盘 resync)汇入单 channel + 固定 worker 池(新盘 resync 另有 per-drive worker 池);RustFS 是优先级堆 + 去重合并 + 容量分级丢弃 + per-set bulkhead + 前台压力门控的多策略调度器(`manager.rs:3003-3420`)。RustFS 表达力更强,代价是"重复请求被合并"的可观测性问题(v1 已指出,现有 `HealAdmissionReceipt` canonical task_id + alias 机制回应了它,`manager.rs:1759-1846`)。
2. **scanner 远端盘访问**MinIO leader 通过磁盘抽象层透明读写远端节点磁盘;RustFS leader 通过 remote_scanner RPC 把扫描执行下放到远端 peer 本地进行(`crates/scanner/src/remote_scanner.rs`),只回传结果与进度心跳。两者都是集群单 leader。RustFS 方案省 leader↔远端的元数据读放大,代价是需要维护独立 RPC 协议(HMAC 逐帧认证、会话重放缓存、fence 复验,`remote_scanner.rs:52-61,405-496,1024-1065`)。
3. **heal 状态持久化**MinIO 用单文件 `.healing.bin`msgp healingTrackerdiskID 不匹配即重置);RustFS 用 schema 化多文件(resume/checkpoint/intent/seal/proof 各自 CAS 发布,`resume.rs:38-61`),崩溃窗口显式补齐(`erasure_healer.rs:389-402``resume.rs:1027-1057`)。
4. **写路径自保护**MinIO 写入后靠后台 heal 收敛;RustFS 在 PutObject/CompleteMultipartUpload 提交 rename 后主动检查 `convergence.needs_heal()` 并立即入队对象 heal`set_disk/ops/object.rs:2291-2306``ops/multipart.rs:2574-2589`),另有读修复 read repair`io_primitives.rs:1040-1160`)。
---
## 2. Heal 已实现功能全景
### 2.1 任务类型(`HealType``crates/heal/src/heal/task.rs:85-111`
| 类型 | 语义 | 执行体 | 生产触发方 |
|---|---|---|---|
| `Cluster` | 所有 bucket 依次 heal(结构 + 可选递归对象),批内重试 ≤3 | `heal_cluster` task.rs:1420-1490 | channelbucket 为空即 Clusterchannel.rs:576-577 |
| `Object{bucket,object,version_id}` | 单对象/版本;不存在时按 `recreate_missing` 重建或报错 | `heal_object` task.rs:855-1146 | admin、scanner、read-repair、写路径收敛、add_partial |
| `Bucket{bucket}` | 桶元数据/结构;`recursive` 再遍历全部对象版本 | `heal_bucket` task.rs:1284-1418 + `heal_bucket_objects` task.rs:1508-1698 | adminPOST /v3/heal/{bucket})、scanner `build_bucket_heal_request` |
| `Prefix{bucket,prefix}` | 按前缀递归 | `heal_prefix` task.rs:1492-1506 | channel`recursive && prefix` 非空(channel.rs:578-585 |
| `ErasureSet{buckets,set_disk_id}` | format 修复 + healing 标记 + 逐桶预处理 + 可恢复逐版本深扫 | `heal_erasure_set` task.rs:2158-2642 | adminpool/set 参数)、auto disk scanner、unclean shutdown、renew_disk、durable replacement 恢复 |
| `Metadata{bucket,object}` | 仅元数据(Deep、不重建数据) | `heal_metadata` task.rs:1700-1859 | **无生产触发方**(§6 HS-01 |
| `MRF{meta_path}` | 失败路径驱动的 Deep 修复(recursive+update_parity | `heal_mrf` task.rs:1861-1992 | **无生产触发方**(仅 `HealEvent` 可生成,未接线) |
| `ECDecode{bucket,object,version_id}` | EC 解码重建(Deep+recreate+update_parity),Urgent 优先级 | `heal_ec_decode` task.rs:1994-2156 | **无生产触发方**(仅 `HealEvent` 可生成,未接线) |
优先级 `Low/Normal/High/Urgent`task.rs:168-179);状态机 `Pending/Running/Retrying/Completed/Failed/Cancelled/Timeout`task.rs:225-241)。
### 2.2 触发路径全景(admin 之外)
| 通道 | source | 优先级 | 证据 |
|---|---|---|---|
| Scanner 周期抽样(1/1024`RUSTFS_HEAL_OBJECT_SELECT_PROB` | Scanner | Low | `scanner_folder.rs:2117-2136``:1150``remove_corrupted=HEAL_DELETE_DANGLING(true)``recreate_missing=false``common/heal_channel.rs:24``scanner_folder.rs:510-511` |
| Scanner 元数据损坏(get_size 失败分类 HealMetadata | Scanner | High | `scanner_folder.rs:2147-2208``:1244-1260` |
| Scanner abandoned children(缓存有、盘上无,list_path_raw quorum 核查) | Scanner | High(桶级+对象级) | `scanner_folder.rs:2528-2792` |
| Scanner pending-heal 账本重试(heal 通道满被拒后持久化,每桶每轮 ≤128 条、上限 10k) | Scanner | 原优先级 | `scanner_folder.rs:1721-1763``:99-100` |
| auto disk scannerunformatted 盘经 replacement_readiness 确认 / `runtime_state=="returning"` 盘 / durable intent 重入) | AutoHeal | Low | `manager.rs:2464-2999` |
| unclean shutdown 恢复(启动读 `unclean-shutdown` 标记 → 全部本地 set ErasureSet heal | AutoHeal | Low | `manager.rs:1362-1695` |
| 写路径收敛(PutObject/CompleteMultipartUpload 后 `convergence.needs_heal()` | Internal | Normal | `set_disk/ops/object.rs:2291-2306``ops/multipart.rs:2574-2589` |
| 部分对象 healadd_partial | Internal | Normal | `set_disk/ops/object.rs:5808-5825` |
| 旧数据目录清理残留 enqueue | Internal | Normal | `set_disk/core/io_primitives.rs:3880-3907` |
| 读修复(metadata_read_error / missing_shards / decode_errorTTL 去重缓存) | ReadRepair | Low | `set_disk/read.rs:407,995,1079``submit_read_repair_heal``io_primitives.rs:1105-1160`),`recreate_missing=true` |
| 盘重连遇 UnformattedDisk → send_heal_disk | AutoHeal | Normal | `set_disk/ops/locking.rs:339-347` |
| Admin API(含集群 coordinator 路由) | Admin | High | `rustfs/src/admin/handlers/heal.rs:174-212``:771-930` |
| 集群 RPC healpeer 调用) | — | — | `rustfs/src/storage/rpc/node_service/heal.rs``ecstore/src/cluster/rpc/peer_s3_client.rs:296,1209` |
注意:MinIO 的 MRF 通道(读路径检出 part 缺失/损坏即时投递 + 队列持久化 + shutdown 回放,`cmd/mrf.go``erasure-object.go:395-410,800-812`)在 RustFS 由 read-repair + 写路径收敛**部分替代**;`HealType::MRF`/`ECDecode`/`Metadata` 三个执行体没有生产入口(详见 §6 HS-01)。
### 2.3 对象级 heal 语义(ecstore `set_disk/ops/heal.rs`
流程(`heal_object_with_explicit_version_regen` :426 起):
1. 取对象写锁(除非 `no_lock`);`object``/` 结尾走对象目录 heal`heal_object_dir_locked` :1587-1717dangling 判定 + `remove` 删除 + 缺 volume 重建)。
2. `read_all_fileinfo` 全盘读 xl.meta,全部 not-found 视为已删除返回。
3. **quorum 仲裁 + ETag 兜底**(已亲验):`list_online_disks` 以 mod-time quorum 为准;quorum 失效时回退 ETag 多数派仲裁(`:525-567` `filter_by_etag`/`quorum_etag`);`pick_valid_fileinfo` 选 canonical 元数据;"meta 坏盘数 > parity" 的 cannotHeal 判定在 ETag 全盘一致时豁免(`:679`)。与 MinIO `filterDisksByETag` 双仲裁一致。
4. `disks_with_all_parts`:562-572)按 `scan_mode` 校验 part**Normal 仅 statCheckParts 语义),Deep 做全量 bitrot 校验(VerifyFile 语义)**Normal 扫描检出 `FileCorrupt` 自动升级 Deep 重试一次(`:2022-2031`,与 MinIO erasure-healing.go:1101-1106 同型);无 parity 对象(EC:0)bitrot 失败判不可恢复(`:700-726`)。
5. `should_heal_object_on_disk`:606-650)逐盘分类 missing/corrupt/offline/outdated → 重建:per-part bitrot reader/writer(用 per-part checksum + 算法)、写临时卷后 rename 提交(`HEAL_RENAME_INCOMPLETE` 重试语义 :24);dangling 删除安全检查 `dangling_delete_safety`:1488);**孤儿数据目录回收 `reclaim_orphan_data_dirs_best_effort`(:1428**——这部分覆盖了 MinIO `CleanAbandonedData` 的主场景(但无独立 `CheckAbandonedParts` API,见 §6 HS-02)。
6. 版本化对象:枚举"每个版本"(`storage.rs:1494-1530`);delete-marker 路径由 `latest_meta.deleted` 决定(`storage.rs:262-277` 注释);回归测试 `tests/heal_b5_versioned_regression_test.rs:282,334`
7. 显式版本重建 `try_regenerate_explicit_version_meta`:1318);transitioned 对象本地残留清理。
8. 写入路径另有 shard 级 bitrot 自校验 `verify_written_bitrot_shards``ops/bitrot_self_verify.rs:45-129`HighwayHash256S,最终 rename 前校验刚写出的 shard,服务 EC:0 无 parity 场景)——**注意这不是后台 bitrot 巡检**;后台巡检由 scanner bitrot_cycle 驱动 Deep heal 承担。
heal crate 侧包装(`task.rs:855-1146`):存在性检查(瞬时错误转 `TransientSkip` 不误判失败 :551-569);scanner 合成目录规范化(:1148-1180);`recreate_missing` 重建(:1183-1282);data-usage-cache 对象锁超时豁免(:571-653);not-found → treated_as_deleted 成功(:1012-1029);结果 `HealResultItem` 保留至多 1024 条 + truncated 标志(:50,845-852)。
递归遍历(`heal_bucket_objects` task.rs:1508-1698):分页枚举全部版本含 delete marker、瞬时错误指数退避重试 ≤32^n + 抖动 :620-627)、失败样本日志截断 ≤5 条、聚合 `BatchHealFailure`
### 2.4 erasure set heal 与断点续扫
`heal_erasure_set`task.rs:2158-2642)四阶段(4 步进度跟踪):
1. **替换意图与恢复盘选择**(仅 AutoHeal + heal_endpoints 非空):复用 durable intent 所在盘 / 排除目标端点选幸存盘;已完成代(CleanupPending)幂等收尾。
2. **格式修复**`heal_replacement_format(dry_run, pool, set, targets)``storage.rs:1372-1384`trait 默认实现 fail-closed);逐目标盘结果必须全 ok(`erasure_healer.rs:97-102`+ 身份围栏复核(task.rs:2410-2420)。
3. **healing 标记**:对目标盘写 owner CAS 标记 `{set_disk_id}:{task_id}``mod.rs:80-229`CAS + 回滚 + 并发唯一 owner),使 `DiskInfo.healing` 为真(已亲验赋值链 `set_disk/mod.rs:4988`)。
4. **逐桶预处理 + 可恢复深扫**`ErasureSetHealer::heal_erasure_set``erasure_healer.rs:242-278`)。
`ErasureSetHealer` 扫描细节(对标 MinIO `healErasureSet``heal_walk.rs:15-23` 模块注释明确引用 MinIO `global-heal.go` 的 listPathRaw + objQuorum=1 + mergeXLV2Versions):
- **枚举器选择(backlog#920**Deep 或 AutoHeal → per-set **disk-walk 并集枚举** `list_versions_for_heal_page_disk_walk`"任意盘上存在"即 sub-quorum 可重建;`storage.rs:1559-1644`,页界 1000 对象/10,000 版本,`dw1:` cursor);普通请求走 read-quorum `list_object_versions`
- **续扫游标**:权威 cursor 为 opaque continuation token`v1:`=marker JSON、`dw1:`=disk-walk key,两命名空间互斥防误读,`storage.rs:81-260`);每完成一页先持久化 cursor 再清 dedup 集合(`erasure_healer.rs:922-927`)。
- **页内并发**FuturesUnordered + Semaphore,默认 `RUSTFS_HEAL_PAGE_OBJECT_CONCURRENCY=8`Deep/AutoHeal 强制 1`erasure_healer.rs:105-142`)。
- **per-version dedup**`compose_key` 长度前缀注入编码(`resume.rs:281-288`)。
- **错误分类**:真缺席(FileNotFound 等)→ Absent(计成功);基础设施瞬时(quorum/DiskNotFound/SlowDown 等)→ Transient(计 skipped);其余 Failed`erasure_healer.rs:148-182`,注释引 backlog#856/#799 B7:离线盘不得记 healed/absent)。
- **防死循环**:空页 truncated 或页尾版本身份不前进即中止(:933-949)。
- **完成判定**failed/skipped/failed_buckets 任一 >0 不标记完成,`schedule_retry()` 复位 resume+checkpoint 两层(:561-626backlog#855/B6/#1033skip 轮不得标记完成)。
- **替换盘提交证据**:目标端点物理回读 `replacement_targets_have_version``ops/heal.rs:340-412`),未确认 → transient skip。
### 2.5 换盘自动修复(replacement recovery
- **识别**`replacement_readiness.rs:25-73`):`replacement_mount_lease_root()` 存在、canonicalize 成功、是挂载点、物理设备 id 非空、与根设备不相交、不与兄弟盘共享物理设备(Linux 用 /proc/self/mountinfo mount-id+dev+ino)。非 root 挂载检查有回归测试(`manager.rs:3549`)。
- **状态机**`resume.rs:63-73`):`Intent → Rebuilding →(写 proofVerified → CleanupPending → 清理``Abandoned` 终态;跨状态迁移先写持久层再变更(`save_state_strict`)。
- **持久化**`resume.rs:38-61`schema ResumeState=5/Checkpoint=5/proof=1):`{task_id}_ahm_resume_state.json``_ahm_checkpoint.json``buckets/ahm-replacement/` 命名空间下 intent/seal/completion_prooftorn write + 无 seal 可识别并原子重建(:1316-1338);CAS 发布、拒绝覆盖并发有效 proof:1512-1585)。
- **恢复**unclean shutdown 与周期扫描都从幸存盘恢复未完成/待清理替换代(`manager.rs:1435-1640,2663-2815`);多代冲突/校验失败 → 冻结该 set(`replacement_recovery_blocked_sets``manager.rs:69-87,2782-2815`)。
- **对外快照**`current_replacement_recovery_snapshot``lib.rs:262-333`)合并本地幸存盘记录,冲突 → Unknown/非 definitiveadmin `GET /v4/heal/replacement-recovery`
### 2.6 调度器(manager.rs
- 优先级堆 + 同优先级 FIFO:148-191,330-347);dedup key 按类型(:469-506);入队三态查重 active→queued→retrying:1759-1785);重复默认 Merged 并返回 canonical task_id`HealAdmissionReceipt`:1821-1846+ client token alias:1219-1246)。
- 容量:队列满时 best-effort 来源(Scanner/AutoHeal/ReadRepair)或低优先级被 Dropped(QueueFull)Admin/Internal 可驱逐低优先级排队项(`push_displacing_lower_priority` :353-396);80%/95% 压力分级(:885-909)。
- 并发:全局 `max_concurrent_heals`(默认 4+ per-set bulkhead `max_concurrent_per_set`(默认 1)(:3040-3073,3434-3447)。
- 前台压力门控 mainline throttle:前台读/写 permit 利用率 ≥80% 时延迟 best-effort 任务(:919-1009,2999-3020)。
- 超时:任务级聚合超时(默认 300s),跨重试保留剩余预算(task.rs:444-451PR #6101)。
- 可恢复重试:`is_recoverable_heal()`error.rs:83-136)≤3 次、2^n 退避封顶 30sretry 在独立 backoff task 中持有所有权(:3235-3382)。
- 完成态保留 10 分钟供查询(:42)。
### 2.7 Admin API 与集群协调
- 路由(`rustfs/src/admin/handlers/heal.rs:174-212`):`POST /rustfs/admin/v3/heal/``/heal/{bucket}``/heal/{bucket}/{prefix}`(同一 POST 按 query `clientToken/forceStart/forceStop` 区分 start/query/cancel,与 mc admin heal 语义对齐);`POST /v3/background-heal/status``GET /v4/heal/replacement-recovery`。权限 `HealAdminAction`route_policy.rs:334-341)。
- 集群协调(heal.rs:771-930 + `node_service.rs:514-606`):`heal_topology_fingerprint` + 按拓扑确定性选 coordinator 节点 + coordinator epochenvelope 校验 + SHA256 digest 重放缓防重放;coordinator 非本机走 peer gRPC `heal_control``probe_heal_control` 能力探测(滚动升级场景)。
- 请求:body 为 `HealOpts``recursive/dryRun/remove/recreate/scanMode(0/1/2)/updateParity/nolock/pool/set`serde camelCase,与 madmin.HealOpts 字段对齐);根 heal start 需 `recursive=true``pool+set` 成对;body 上限 1MB。
- 响应:`HealStartSuccess{clientToken, clientAddress, startTime}``HealTaskStatus{summary, detail, startTime, settings, items, truncated, progress}`summary ∈ running/finished/stopped/notFound);`BackgroundHealStatus`(bitrot 起始时间/周期/当前模式 + `disabled/uninitialized/idle/active/degraded` 状态——peer 不可达显式 degraded 不冒充 idleissue #5850 + `healOperations` 按优先级×来源矩阵 + 集群进度)。
- `HealResultItem`/`HealDriveInfo`/`HealItemType`/DriveState 枚举与 madmin JSON 兼容(`crates/madmin/src/heal_commands.rs:19-65`)。
- 状态 payload 超 8MiB 对折截断(channel.rs:37,73-104);path-token 校验(错误 token 拒绝,空 path 仅匹配 Cluster)。
### 2.8 heal 指标与日志
指标:`rustfs_heal_admission_total{source,result,reason,context}``rustfs_heal_task_start_total``rustfs_heal_task_running{type,set}``rustfs_heal_queue_delay_seconds``rustfs_heal_scheduler_skip_total``rustfs_heal_mainline_throttle_total``rustfs_heal_page_concurrency_current{set}``rustfs_heal_candidate_enqueue/merge/drop/priority_reject_total``rustfs_heal_read_repair_dedup_total{reason}` 等。日志全部结构化 event stylePR #5720);per-object 日志降级防风暴(`demote_to_debug_when!`#5716/#5719/#5727)。
---
## 3. Scanner 已实现功能全景
### 3.1 循环、leader、立即触发
- **集群单 leader**:分布式 ns 写锁 `leader.lock``scanner.rs:3156-3207`,超时默认 5s+ **持久化 leader-epoch CAS 围栏**leader 用 ETag 前置条件向 `.bloomcycle.bin``RSCYC001` 编码的 (cycle, leader_epoch)`scanner.rs:118,1850-1861,2177-2334`);usage 快照再打 epoch fence:2087-2153)。锁丢失 → 取消当前周期,30s 收敛(:108-111,2623-2642)。
- 抢锁后立即执行一轮;周期 = `RUSTFS_SCANNER_CYCLE` > config cycle > start_delay > 部署默认 > 速度档位(±10% 抖动、下限 1s)。
- **clean-idle 指数退避**:连续完整无脏周期间隔 ×2(封顶 24h;bitrot 周期压缩上限;桶有 lifecycle/replication 活动规则禁用,:383-456,1382-1512)。
- **superseded/deferred 退避**:5s 起指数退避封顶 30min:105-106,3432-3438);维护探测失败独立退避(:459-505)。
- **立即唤醒**:① dirty-usage 快路径——写路径 put/delete/multipart/bucket 操作调用 `record_dirty_usage_bucket``scanner_io.rs:222-235`;调用点 `rustfs/src/app/object_usecase.rs:6221` 等),自增 generation 并 Notify 唤醒 leader,脏桶优先排队(`scanner_io.rs:462-488`);② 维护配置变更(lifecycle/replication 设置时 `record_scanner_maintenance_change`);③ 运行时配置热更 generation+Notify;④ 集群活动快照变化。
- **集群协调**`probe_scanner_activity` 汇集本机+peer 的 `ScannerNodeActivity`instance_id/namespace_generation/maintenance_generation/protocol_version/topology_digest/data_movement_active/dirty usage),拓扑摘要覆盖 pools/sets/drives URL,协议版本不齐拒绝共享缓存锁(`scanner.rs:970-1068`);**数据迁移(rebalance/decommission)期间推迟周期**`scanner_io.rs:2226-2374`);周期结束逐 peer RPC 确认 dirty-usage ack`scanner.rs:2925-2952`)。
### 3.2 遍历模型
- 主遍历是**全量目录 walk**tokio::fs::read_dir 递归,`scanner_folder.rs:1915-2234`),不走 metacachemetacache/`list_path_raw` 仅用于 abandoned children 跨盘核查(:2528-2792)。
- 三级并发:leader → per-set(信号量默认 4)→ per-disk 桶扫描(默认 4)→ 单盘递归;每桶每 set 缓存锁 `.scanner-cycle.lock.pool-N.set-M`(锁丢失取消该桶扫描,锁竞争重排队);每盘单扫描准入(本地盘也走信号量,`scanner_io.rs:3246-3274`)。
- 桶顺序:shuffle 后按 dirty → 未缓存 → 已缓存重排(`scanner_io.rs:2947-2949,462-488`);目录内按名字排序 + resume 提示旋转(`scanner_folder.rs:333-359`)。
- **断点续扫**`DataUsageScanCheckpoint{version,resume_after,reason}` 持久于缓存 info`data_usage_define.rs:68,293-307`);预算耗尽/取消写入,恢复有 Used/Stale/NoHint 指标;续扫单位是目录(无跨周期对象级分页)。
- erasure 语义:发现 `xl.meta` 即对象边界不下钻;UUID data-dir 候选最多探测 64 entry;有数据无元数据 → 记 failed + 高优 healsymlink 目录忽略/环跳过。
- 协作让出:每 N 对象(默认 128)`yield_now`
### 3.3 大桶跳过策略(对标 MinIO compaction
1. 缓存当前性复用:桶与扫描计划未变(name/source/snapshot_complete/plan digest/next_cycle/leader_epoch/cache_key_format 全匹配)整桶跳过(`scanner_io.rs:1062-1109`)。
2. compacted 目录 16 周期轮换窗口:`hash mod (next_cycle, 16)` 命中才重扫,否则从旧缓存拷贝(`scanner_folder.rs:74,2429-2442`)。
3. compaction 阈值:子项 <500 或纯对象叶子压缩为单 entry;子文件夹 ≥2500(根 10000)预压缩;children ≥10000 归约(:75-78,2314-2340,2846-2887)。
4. 失败对象 TTL 跳过:86400s/最多 10000 条(:88-91,1354-1381)。
与 MinIO master 对比:MinIO 的跳过策略同样是 hash-mod-16 周期 + compaction 阈值树(500/10000/2500),**bloom filter 已从 master 删除**。RustFS 的常量与结构与 MinIO 现状同源(MinIO 未采用跨盘 dirty-generation 优先,RustFS 额外多两层跳过——plan digest 与缓存当前性校验)。
### 3.4 data usage 统计
- 维度:每目录 entrysize/objects/versions/delete_markers/大小直方图/版本直方图/复制统计/failed_objects/per-tier stats/children/compacted`data-usage/src/data_usage.rs:661-679`);每对象 SizeSummary(含 per-ARN 复制目标统计、tier 统计,tier 分类:transitioned 完成记入其 tier 否则按 storage classfree version 不计);桶级 `BucketUsageInfo`;集群级 `DataUsageInfo`(含 scanner_cycle/scanner_epoch 围栏 + usage_snapshot_complete)。
- 存储:每桶每 set `{bucket}/.usage-cache.bin`(主 + `.bkp` 备份 + CAS 重试);权威集群快照 `buckets/data-usage/data-usage.json`(每 10 周期同步 `.bkp`,legacy 路径兼容);陈旧快照拒绝写入(epoch/cycle/last_update 三重判定);被竞争 superseded 的观测快照另存 `data-usage-observed.json`
- 消费:`replace_bucket_usage_memory_from_info` 刷新桶用量内存 + 两层缓存失效(`scanner.rs:4142-4152`)→ bucket stats/quota/admin account_info/system;写路径内存实时叠加 overlay;启动读快照判断冷缓存跳过启动延迟。
- 未完成 multipart 不参与统计(与 MinIO 一致,MinIO 也不扫 multipart 桶)。
### 3.5 ILM 集成
- 每对象 `ScannerItem::apply_actions``scanner_folder.rs:747-1032`):`Evaluator::new(lifecycle).with_lock_retention(...).with_replication_config(...).eval()` 批量评估。
- 已实现动作(IlmAction 全集,`common/src/metrics.rs:34-45`):expiry 删除(Delete/DeleteRestored/DeleteRestoredVersion)、全版本删除(DeleteAllVersions/DelMarkerDeleteAllVersions,处理后停止后续版本)、transitionTransition/TransitionVersiontier 列表运行时读取)、noncurrent 批量(DeleteVersionAction → `enqueue_by_newer_noncurrent`)、free-version 清理(`enqueue_free_version`)、object-lock retention 约束。**与 MinIO 的 9 个 ILM 动作一一对应**。
- 执行模型:scanner 是"发现与入队"角色(expiry 队列/transition 队列在 ecstore `bucket_lifecycle_ops.rs`),动作由 worker 池消费——与 MinIO globalExpiryState/globalTransitionState 同型。
- AbortIncompleteMultipartUpload 不在 scanner/ILM 内执行(MinIO 同样不在:`internal/bucket/lifecycle/rule.go` 有 FIXME,实际由 `erasureSets.cleanupStaleUploads` 全局例程承担);RustFS 由 ecstore 独立后台任务 `init_background_stale_multipart_upload_cleanup``bucket_lifecycle_ops.rs:3289-3320`+ 桶删除时 on-demand。
- 集成测试覆盖:transition+restore、free-version、noncurrent、delete-marker、0-day、后台扫描过期(`scanner/tests/lifecycle_integration_test.rs:1071-2095`)。
### 3.6 heal 候选生产(scanner 侧)
- 抽样:`hash mod_alt(next_cycle/prob_div, 1024/prob_div)`,进入 compacted 分支重扫时 prob_div=16 等效概率 ×16(与 MinIO 同款补偿,`scanner_folder.rs:125-127,2117-2122`)。
- deep/normal:周期级 `get_cycle_scan_mode`bitrot_cycle 默认 30d`scanner.rs:1626-1657`)→ 对象级带 `HealScanMode::Deep`;新鲜对象(60s 内修改)降级 Normal:146-155);状态持久 `.background-heal.json``BackgroundHealInfo{bitrot_start_time,bitrot_start_cycle,current_scan_mode}`,与 MinIO 同路径同结构)。
- scanner 只入队不内联执行(内联 heal 已移除,兼容旗标仅告警,`scanner_folder.rs:411-427`);`HealScanMode::Deep` 只是标记,bitrot 校验读发生在 heal 消费端(ecstore Deep 路径)。
- 元数据损坏 → 高优 heal`classify_get_size_failure` → HealMetadata);abandoned children → list_path_raw quorum 核查 + 桶级/对象级高优 heal;healing 盘粘性跳过(`should_heal` :1628-1648)。
- pending-heal 账本:heal 通道满被拒持久化到缓存 info,下轮重试。
- 复制 heal`queue_replication_heal` → replication 队列(走 replication 通道而非 heal channel);per-ARN 复制用量统计。
### 3.7 remote_scanner RPC 协议(RustFS 特有)
请求 ≤16KB msgpackversion/request_id/server_epoch/session_id/session_sequence/bucket/next_cycle/leader_epoch/scan_plan_digest/skip_healing/scan_mode/budget);帧 ≤2MB、HMAC-SHA256 逐帧认证(域 `rustfs-ns-scanner-frame-v3`);进度心跳 1s(预算模式 250ms);阶段播报 Scanning→PersistingRPC 生命周期上限 24h、断连宽限 2min;防重放 session+sequence 缓存(容量 65536);服务端校验 leader fence 与持久化 cycle 一致 + 每 5s fence 复验;结果 Complete/Partial/NamespaceNotFound/CycleAhead;不支持 v4 协议的远端盘回退 leader 本地扫描(`remote_scanner.rs` 全文件;`scanner_io.rs:2750-2812`)。
### 3.8 限速/预算/热更/观测
- DynamicSleeper 比例退避(速度档 fastest/fast/default/slow/slowest,同 MinIO 五档参数);idle_mode 总闸;前台 S3 读流量每请求 10ms 封顶 250ms 额外退避。
- 周期预算 ScannerCycleBudgetmax_duration/max_objects/max_directories(默认 0=不限),partial 周期仍推进 cycle 计数。
- runtime_config 三层来源(env > config > default)逐字段来源标记(Env/Config/ScannerCompatConfig/Default),admin `PUT /v3/config` 热更 → generation+Notify 即时生效;`GET /v3/scanner/status` 返回 enabled/freshness(fresh/stale/unknown)/metrics/cycle_schedule/runtime_config`GET /v3/ilm/expiry/status` 返回 expiry 队列/worker/missed/blocked。
- 指标:leader lock、周期 complete/partial/deferred/superseded、versions scanned、per-sourceUsage/Lifecycle/BucketReplication/SiteReplication/Heal/Bitrot/Alertschecked/executed/queued/missed、checkpoint set/used/stale、当前路径(per-disk+bucket 实时)、缓存 save 系列、并发系列、告警(excess versions/version size/folders)。
---
## 4. 与 MinIO 逐项对标
### 4.1 heal 触发通道对照
| MinIO 通道 | RustFS 对应 | 状态 |
|---|---|---|
| A. 手动 admin healhealSequenceclientToken/forceStart/forceStop | heal channel Start/Query/Cancel + 集群 coordinator + envelope 重放防护 | ✅ 等价且增强(集群路由);序列语义差异见 §6 HS-06 |
| B. 常驻后台 heal 队列(newBgHealSequence + healRoutine worker 池) | HealManager 常驻调度器 + 优先级队列 + bulkhead | ✅ 等价且增强 |
| C. 新盘/换盘自动 resyncmonitorLocalDisksAndHeal 10s + healFreshDisk + healingTracker + waitForFormatErasure 握手) | auto disk scanner10s+ replacement_readiness + durable intent/proof 状态机 + heal_replacement_format | ✅ 等价且增强(identity fence + completion proofMinIO 的 tracker 面向对外可见性更强,见 §6 HS-07) |
| D. MRF(队列 100k + 持久化 list.bin + shutdown 回放 + 读路径 corrupt 投递) | read-repairLow+TTL 去重)+ 写路径 convergence heal 部分承担;`HealType::MRF` 执行体无生产入口 | ⚠️ 部分等价(§6 HS-01) |
| E. Scanner 抽样 heal1/1024 + compacted ×16 补偿)+ abandoned children | 同款抽样 + ×16 补偿 + abandoned children + pending-heal 账本 | ✅ 等价且增强(账本) |
| F. 读路径内联触发 → MRFGetObject part 缺失/损坏、元数据重建 missingBlocks>0 | read repairmissing_shards/decode_error/metadata_read_error 三入口) | ✅ 等价(入 heal 队列而非 MRF 队列) |
### 4.2 对象级 heal 语义对照
| 特性 | MinIO | RustFS | 状态 |
|---|---|---|---|
| mod-time quorum 仲裁 | listOnlineDisks | 同 | ✅ |
| ETag 多数派兜底(时钟漂移) | filterDisksByETag | `filter_by_etag`/`quorum_etag`heal.rs:525-567 | ✅ 已亲验 |
| cannotHeal 的 ETag 豁免 | ETag 全一致豁免重试 | heal.rs:679 | ✅ |
| Normal=CheckPartsstat/ Deep=VerifyFilebitrot | 是 | `disks_with_all_parts` 按 scan_modeops/heal.rs:562-572,978-1024 | ✅ |
| Normal 检出 corrupt 自动升 Deep 重试一次 | erasure-healing.go:1101-1106 | ops/heal.rs:2022-2031 | ✅ |
| dangling 判定(not-found > parity+ 删除审计 | isObjectDangling/deleteIfDangling | `dangling_delete_safety`:1488+ scanner HEAL_DELETE_DANGLING | ✅(审计 tags 细节有差异) |
| 孤儿 data-dir/inline 清理(CleanAbandonedData | CheckAbandonedPartsscanner 抽中 + admin Remove 时显式调用) | heal 路径内 `reclaim_orphan_data_dirs_best_effort`:1428);独立 API 三层 NotImplemented | ⚠️ 部分等价(§6 HS-02 |
| 版本化/delete-marker heal | HealObject versionIDnullVersionID 特判 | 逐版本枚举 + delete-marker latest healB5 回归) | ✅ |
| 对象级 healing 元数据标记(x-minio-healingRenameData 跳过版本清理) | 有 | 无对象级标记;依赖盘级 healing.bin + NSLock + rename 语义 | ⚠️ 评估项(§6 HS-12) |
| Distribution/Index 一致性三处防线 | 有(manual modification 拒绝) | 目标盘格式结果全 ok 校验 + 身份围栏 | ✅(粒度不同) |
| 无 parityEC:0)对象 | bitrot 不可恢复处理 | 判不可恢复(:700-726)+ 写入自校验 | ✅ 增强(写路径自校验) |
| 三层分布不一致拒绝 heal | 有 | heal_walk 归一化 + 页界防御 | ✅(实现方式不同) |
| multipart 孤儿对账 | CheckAbandonedParts 承担 | 显式 NotImplemented(由 lifecycle 清理承担) | ⚠️ §6 HS-02 |
| suspended/decommissioned pool 处理 | IsSuspended 跳过 | deferral 语义(store/heal.rs:192-207PR #5876 | ✅ |
| heal 与并发删除互斥 | NSLock + healing 标记 | NSLock + 写锁 | ✅ |
### 4.3 新盘 resync 对照
| MinIO | RustFS | 状态 |
|---|---|---|
| waitForFormatErasure 四类可恢复错误无限等待握手 | startup 盘解析 + renew_disk 重连路径 | ✅(模型不同:RustFS 不在启动时阻塞等待 format) |
| HealFormat NSLock + errNoHealRequired + refFormat 不一致拒绝 | `heal_format`/`heal_replacement_format` fail-closed + 目标槽位限定(PR #1787 语义) | ✅ 增强 |
| per (pool,set) 分布式锁防并发 resync | set 级队列去重 + bulkheadmanager.rs:2854-2889 | ✅ |
| 全新集群检测(待 heal 盘数==总盘数不触发) | replacement_readiness(独立挂载点/物理设备校验,非 root) | ✅ 增强 |
| healingTracker.healing.binBytes/Items 计数、QueuedBuckets/HealedBuckets、Resume 快照、RetryAttempts ≤4、HealID 联动、diskID 变更重置) | resume/checkpoint schema 化持久层 + durable intent/proofper-task 文件,CAS) | ✅ 等价且增强(崩溃窗口补齐);但**对外快照可见性**弱于 MinIO(§6 HS-07 |
| 跳过 heal 开始后新写入版本(ModTime > Started | 无同款过滤 | ⚠️ §6 HS-13 |
| 跳过 ILM 已过期版本(filterLifecycle | 无同款过滤 | ⚠️ §6 HS-13 |
| worker 数 max(GOMAXPROCS,NR)/4 下限 4heal:drive_workers 覆盖 | 页内并发 8Deep/AutoHeal 强制 1+ per-set bulkhead | ✅(参数模型不同) |
| 每 entry waitForLowHTTPReq 让路 | mainline throttle(前台利用率门控) | ✅ 增强 |
| heal 范围含 `.minio.sys/config``.minio.sys/buckets` 两个伪桶;最新桶优先 | ErasureSet 任务逐 bucket 预处理(含 meta bucket 语义由 heal_bucket 承担) | ✅(顺序无"最新优先" |
| 失败整体重试 ≤4 次(resetHealing + errRetryHealing | schedule_retry 复位双层 + 可恢复重试 ≤3 | ✅ |
### 4.4 scanner 对照
| MinIO | RustFS | 状态 |
|---|---|---|
| 集群单 leaderglobalLeaderLock | leader.lock + 持久化 leader-epoch CAS 围栏 | ✅ 增强(epoch 围栏防脑裂,MinIO 无持久化 epoch |
| `.bloomcycle.bin` 只存 cyclebloom 已删除) | 同路径存 cycle+leader_epochRSCYC001 | ✅ 对齐(v1 误判已修正) |
| folderScanner hash-mod-16 + compaction500/10000/2500 | 同款常量 + plan digest + 缓存当前性校验 + dirty 优先 | ✅ 增强 |
| 每盘扫描并行 ≤GOMAXPROCShealing 盘排除 | per-set/per-disk 信号量 + healing 盘粘性跳过 | ✅ |
| scannerSleeperfactor 2/max 1sspeed 档热更) | DynamicSleeper 同款 + idle_mode + 前台读退避 | ✅ 增强 |
| idle 语义:`scanner:idle_speed=on`(空闲时段才节流,忙时全速) | `RUSTFS_SCANNER_IDLE_MODE=true`(启用限速总闸) | ⚠️ 语义方向相反,§6 HS-14 |
| applyActions 顺序(heal→ILM→复制→告警) | apply_actions 同序(heal 候选→ILM→复制 heal→告警) | ✅ |
| ILM 9 动作 + 批量评估 + DeletePrefixObject 优化 | 同 9 动作 + 批量评估 + expiry 队列 | ✅(DeleteAllVersions 是否单调用优化未逐行核) |
| abandoned childrenlistPathRaw minDisks=N/2 发现漏写盘) | list_path_raw + quorum 核查 + 高优 heal | ✅ |
| incomplete multipart 独立例程(6h 间隔/24h 过期,rename 进 .trash | ecstore 独立后台任务(可配间隔/过期) | ✅(trash 二段清理细节差异,§6 HS-18) |
| usage 维度(size/objects/versions/DM/直方图/复制/tier/bucket 级) | 全覆盖 + 集群快照三重防回退 | ✅ 增强 |
| prefix 级 usageloadPrefixUsageFromBackendconsole 消费) | 缓存内有目录树但仅 flatten 桶级 | ❌ §6 HS-08 |
| 超限事件 s3:ObjectManyVersions/LargeVersions/PrefixManyFolders + 审计 | 仅指标 alert_excess_*(默认 100/1TiB/65538 vs MinIO 100/1TB/50000 | ⚠️ §6 HS-04/HS-17 |
| scanner 指标 v3bucket_scans/directories/objects/versions/last_activity | rustfs_scanner_* 全套 + freshness | ✅(命名体系不同) |
| TraceScanner / realtime metricsmc admin scanner status/trace | 无 trace 通道;/v3/scanner/status 自有结构 | ⚠️ §6 HS-03 |
### 4.5 admin/CLI/API 面对照
| MinIO | RustFS | 状态 |
|---|---|---|
| `POST /minio/admin/v3/heal/...` start/status/cancel | `POST /rustfs/admin/v3/heal/...` 同三态 | ✅(路径前缀不同属预期) |
| `HealStartSuccess`/`HealTaskStatus`/`HealResultItem`/DriveState | 同名字段 JSON 兼容 | ✅ |
| `POST /v3/background-heal/status`BgHealState 聚合) | 同路径 + degraded 语义 + operations 矩阵 | ✅ 增强(MRF per-endpoint 子状态无,因无 MRF |
| `GET /v3/healthinfo` 每 drive `HealInfo *HealingDisk` | 无同款 healthinfo heal 字段(replacement-recovery v4 承担部分) | ⚠️ §6 HS-07 |
| madmin 客户端 HealStart/HealStatus/BackgroundHealStatus/ScannerStatus 方法 | 仅 wire 类型,无客户端方法 | ❌ §6 HS-05 |
| mc admin heal --pool/--set、--scan-mode、--force-start/stop | HealOpts 全字段支持(pool/set/scanMode/forceStart/forceStop | ✅(服务端就绪;缺 mc 侧入口,HS-05) |
| ErrHealAlreadyRunning / ErrHealOverlappingPaths 类型化错误 | 去重合并 + 驱逐语义;无类型化重叠拒绝 | ⚠️ §6 HS-06 |
| 结果 backpressuremaxUnconsumedItems=1000、10s 保活流式、24h 未消费 abort) | 快照式查询(1024 条 + 8MiB 截断 + 10min 保留) | ⚠️ §6 HS-06 |
| `mc support inspect`/healing-bin 离线 dump | 无(inspect.rs 存在但 healing dump 未确认) | ⚠️ P3 |
### 4.6 观测面对照
| 维度 | MinIO | RustFS | 状态 |
|---|---|---|---|
| heal 指标 | minio_heal_objects_total/heal_total/errors_total/time_last_activity + v3 drive_health 2=healing | rustfs_heal_* 全套(admission/queue delay/running/throttle/page concurrency | ✅(RustFS 缺 drive_health=healing 单一 gauge 等价物;DiskInfo.healing 已赋值) |
| scanner 指标 | v3 6 个 + realtime 18 项 | rustfs_scanner_* 全套 + per-source 维度 | ✅ |
| ILM 指标 | v3 5 个(expiry/transition pending/active/missed + versions_scanned | ilm expiry status API + scanner per-source | ✅(指标与 API 形态不同) |
| trace | TraceHealing/TraceScanner 两通道 | 无 | ❌ §6 HS-03 |
| 审计 | HealObject 事件、dangling 删除审计、scanner:manyversions 等 | 结构化日志(event style+ 指标;无 audit log 事件 | ⚠️ §6 HS-04 |
| 进度 | healingTracker Bytes/Items/QueuedBuckets/当前对象 + usage-cache 总量基线 | HealProgress{scanned/healed/failed/bytes/current_object/percentage}bytes_processed 注释为 0、estimated_completion_time 恒 None | ⚠️ §6 HS-07 |
### 4.7 配置面对照(默认值)
| MinIO | RustFS | 备注 |
|---|---|---|
| `heal:bitrotscan`(默认 offon=每轮;Nm=N×30×24h | `heal.bitrot_cycle` / `RUSTFS_SCANNER_BITROT_CYCLE_SECS`(默认 30d=2592000s0/on=每轮 Deepoff=禁用) | ✅ 同语义(RustFS 默认 30dMinIO 默认 off——**默认值不同**RustFS 更激进) |
| `heal:max_io=100`/`max_sleep=250ms`waitForLowIO | mainline throttle 阈值 80%/80%、max_sleep 250ms | ✅ 同型(阈值模型不同) |
| `heal:drive_workers`(默认 -1 自动) | 页内并发 8 + per-set 1 | ✅ 同型 |
| `_MINIO_HEAL_WORKERS`GOMAXPROCS/2 | `RUSTFS_HEAL_MAX_CONCURRENT_HEALS=4` + `_MAX_CONCURRENT_PER_SET=1` | ✅ |
| `_MINIO_AUTO_DRIVE_HEALING`on | `RUSTFS_HEAL_AUTO_HEAL_ENABLE=true` | ✅ |
| `_MINIO_SCANNER`on | `RUSTFS_SCANNER_ENABLED=true` | ✅ |
| `scanner:speed` 五档(default=2x/1s/1m | 同五档同名同参数 | ✅ |
| `scanner:idle_speed`on | `RUSTFS_SCANNER_IDLE_MODE`(true | ⚠️ 语义方向(HS-14 |
| `scanner:alert_excess_versions=100` | 100 | ✅ |
| `scanner:alert_excess_folders=50000` | 65538(兼容 PBS 布局) | ⚠️ HS-17 |
| `ilm:expiration_workers=100`/`transition_workers=100` | ecstore expiry/transition worker 池(键见 ilm 子系统) | ✅(默认值未逐项核对) |
| `api:stale_upload_cleanup_interval=6h`/`expiry=24h` | ecstore 后台任务 env 可配 | ✅(默认值未逐项核对) |
| —(无) | `RUSTFS_HEAL_QUEUE_SIZE=10000``_TASK_TIMEOUT_SECS=300``_INTERVAL_SECS=10``_LOW_PRIORITY_MERGE/DROP``_PAGE_*``_SET_BULKHEAD``_MAINLINE_*``RUSTFS_SCANNER_CYCLE_MAX_*` 预算、`_MAX_CONCURRENT_SET/DISK_SCANS=4``_YIELD_EVERY_N_OBJECTS=128` 等 | RustFS 特有(更细粒度) |
### 4.8 RustFS 超出 MinIO 的部分
1. remote_scanner RPC(扫描执行下放远端 peer 本地,含 HMAC 认证/重放缓存/fence 复验/断连宽限)。
2. 持久化 leader-epoch CAS 围栏 + usage 快照 epoch/cycle 防回退(MinIO 仅锁,无持久 epoch)。
3. 周期预算(max_duration/objects/directories+ partial 周期推进语义。
4. per-set/per-disk 扫描并发闸 + 每桶每 set 缓存锁。
5. pending-heal 账本(heal 通道满不丢候选)。
6. 换盘 durable intent + completion proof 状态机 + 身份围栏(MinIO healingTracker 无 proof)。
7. mainline throttle 前台压力门控(permit 利用率驱动)。
8. 集群 heal control coordinator + envelope 重放防护 + degraded 显式降级。
9. 写路径 shard bitrot 自校验(EC:0 场景)。
10. dirty-usage 快路径唤醒(写路径即时通知 + 脏桶优先)。
11. heal 运行时可观测矩阵(优先级×来源 operations snapshot)。
12. workload admission 联动(heal 调度器读前台压力快照)。
---
## 5. 差距与改进清单
分级定义:P1=行为/运维对齐缺口(影响生产运维或工具链兼容);P2=完善性(功能在但缺一角);P3=清理/低风险。每项含现状证据、MinIO 行为、影响、建议、验收方式。
### P18 项)
**HS-01 MRF/ECDecode/Metadata 三类 heal 任务无生产触发入口,HealEvent 未接线**
- 现状:`HealType::MRF/ECDecode/Metadata` 执行体完整(task.rs:1700-2156)但全仓库无生产触发方;`HealEvent`/`HealEventHandler`event.rs:50-367crate 外零引用(已亲验 grep);channel 转换只产生 Cluster/Object/Bucket/Prefix/ErasureSetchannel.rs:566-601)。
- MinIOmrf.go 独立 MRF 队列(容量 100k,满丢弃计数)、进程退出 msgp 持久化 `.heal/mrf/list.bin` + 启动回放、入队 <1s 延迟 1s(等网络恢复)、healSleeper 限速;读路径 GetObject part 缺失/损坏、元数据重建 missingBlocks>0、Put 部分成功、DeleteObject、multipart、peer client 共 7+ 投递点。
- 影响:RustFS 的 read-repair + 写路径收敛覆盖了主场景,但缺少:① 事件驱动的 Urgent ECDecode 重建入口(ecstore 解码失败时目前仅 Low read-repair);② Metadata-only heal 入口(scanner HealMetadata 分类存在但走普通对象 heal);③ MRF 队列持久化(重启丢未消费修复意图——scanner pending-heal 账本部分缓解)。
- 建议:三选一决策——(a) 接线 HealEvent(在 ecstore 解码失败/metadata 损坏点发事件)+ 实现持久化重试账本;(b) 删除 MRF/ECDecode/Metadata 死代码只保留文档说明;(c) 保留执行体、把 HealEvent 降级为内部 API。推荐 (a) 但需先量化 read-repair 是否已覆盖解码失败场景的响应时间要求。
- 验收:解码失败 → Urgent heal 请求链路 e2e;重启后 pending 修复意图回放;HealEvent 环形缓冲指标。
**HS-02 CheckAbandonedParts 三层 NotImplementedabandoned data 独立对账入口缺失)**
- 现状:`set_disk/ops/heal.rs:2052-2056``core/sets.rs:1144-1148``store/heal.rs:258-266` 三层显式 `Err(NotImplemented)`(已亲验),注释"intentionally retained above the set layer until there is a concrete caller"。
- MinIO`CheckAbandonedParts` → 每盘 `CleanAbandonedData`:读 xl.meta → 列 UUID data-dir + inline entries → 与 getDataDirs 差集 → 删多余 data-dir/inline 并重写 xl.meta;由 scanner 抽中 heal 与 admin heal Remove 时显式调用。
- 影响:RustFS heal 路径内 `reclaim_orphan_data_dirs_best_effort`:1428)覆盖"heal 时回收孤儿目录",但 ① 无独立触发点(MinIO 在对象未到 heal 阈值时也能清 abandoned data);② inline data 孤儿条目清理未确认;③ multipart 孤儿对账明确不做(设计决定,由 lifecycle 承担)。
- 建议:评估把 `reclaim_orphan_data_dirs_best_effort` 提升为 heal_object 固定步骤(若尚非)+ 实现 HealOperations::check_abandoned_parts 真实现(调用同一回收逻辑),或明确文档化"由 lifecycle 承担"并关闭 API 面。
- 验收:构造 data-dir/inline 孤儿 → scanner 抽样/admin heal 后被清理;三层 API 返回成功或显式 NotSupported 文档化。
**HS-03 heal/scanner trace 通道缺失**
- 现状:TraceHealing/TraceScanner 零命中(已亲验 grep 全仓库)。
- MinIO`madmin.TraceHealing`mc admin trace --healingFuncName=heal.Bucket/heal.Object/heal.CheckAbandonedParts,带 dry/remove/mode/version-id/disks/bytes)、`TraceScanner`mc admin scanner trace,支持 --filter-size/--response-duration)。
- 影响:无法实时观测单个 heal/scanner 动作的耗时与参数;排障只能靠指标聚合与日志。
- 建议:在 heal channel 执行与 scanner folder/item 处理埋点,接入现有 admin trace 订阅面(若 rustfs 已有 trace 基建则复用,无则按 madmin TraceType 扩展)。
- 验收:mc 等价工具能订阅 heal/scanner trace 流。
**HS-04 scanner 超限 S3 事件与审计缺失**
- 现状:仅 `rustfs_scanner_excess_*_total` 指标(versions 100/version size 1TiB/folders 65538)。
- MinIO:发 `s3:ObjectManyVersions`>100 版本)、`s3:ObjectLargeVersions`(累计 >1TB)、`s3:PrefixManyFolders`>50000 子目录)事件(UserAgent: Scanner+ scanner:manyversions/largeversions/manyprefixes 审计。
- 影响:依赖事件订阅做容量治理的用户(console/外部审计)收不到告警。
- 建议:scanner_folder 告警点接入 notify 事件发布(复用 lifecycle 事件通道语义)。
- 验收:配置桶通知后超限对象触发事件。
**HS-05 madmin 客户端方法缺失**
- 现状:`crates/madmin/src/heal_commands.rs` 只有 wire 类型(HealDriveInfo/Infos/HealResultItem);无 HealStart/HealStatus/BackgroundHealStatus/ScannerStatus 客户端方法。
- MinIOmadmin-go 提供完整客户端;mc admin heal/scanner/status/trace 都建立在上面。
- 影响:mc 等管理工具无法直接对接 RustFS heal/scanner 管理面;自动化运维只能手写 HTTP。
- 建议:按 madmin-go 接口形状补客户端(服务端已就绪,纯客户端工作)。
- 验收:用 madmin 客户端完成 start→query→cancel 全流程。
**HS-06 admin heal 序列语义与 MinIO 差异**
- 现状:重复/重叠请求被去重合并(返回 canonical task_id)或驱逐;无 ErrHealAlreadyRunning/ErrHealOverlappingPaths 类型化错误(已亲验:manager.rs:1309 的 already_running 是幂等启动保护,非 admin 语义);结果为快照式查询(1024 条/8MiB 截断/10min 保留),非 MinIO 的流式增量(clientToken 拉增量 + maxUnconsumedItems=1000 backpressure + 10s 保活 + 24h 未消费 abort)。
- 影响:mc admin heal 的交互模型(长连接拉增量)对 RustFS 表现为多次快照轮询;自动化脚本难以区分"已合并"与"新启动"。
- 建议:① 增量语义:channel query 支持自上次 clientToken 起的 items 增量(或 cursor);② 重叠请求返回类型化错误码(或 receipt 中显式 merged_into 字段——现有 alias 机制已有基础);③ forceStart 先停旧再启新语义核对。
- 验收:madmin 兼容客户端按 MinIO 模式轮询能取得全量 items。
**HS-07 healing 进度与盘级 healing 状态对外可见性不足**
- 现状:bytes 恢复进度 `progress.bytes_processed = 0 // set to 0 for now`erasure_healer.rs:967);`HealProgress::estimated_completion_time` 恒 None、`HealStatistics::add_healed_objects` 未写入(progress.rs:38,135-139 零调用);healthinfo 无每盘 HealInfo 等价(MinIO HealingDiskBytesDone/Failed/Skipped、ObjectsTotal 基线、QueuedBuckets/HealedBuckets、Resume 快照、当前 object);v3 指标无 drive_health=2(healing) 单一 gauge 等价。
- 影响:换盘重建(可能数小时~天)期间运维无法回答"进行到哪/还剩多少/预计何时完成"。
- 建议:① erasure set heal 统计 bytesheal_object 返回对象大小已可得);② 从 usage-cache 读对象总量基线(MinIO 同款做法);③ admin healthinfo/背景状态暴露每盘 healing 快照(DiskInfo.healing 已有,补聚合暴露);④ ETA 由基线+速率推导。
- 验收:换盘重建中 admin 可见 bytes 进度与 ETAmc info 等价输出 Healing 标志。
**HS-08 prefix 级 usage 未暴露**
- 现状:DataUsageCache 内目录树 entry 存在(hash_path 组织),但 `dui()` 只 flatten 到桶名(data_usage_define.rs:858-915)。
- MinIO`loadPrefixUsageFromBackend`30s cache)从每 set `.usage-cache.bin` 聚合 prefix usageconsole 桶前缀统计消费。
- 影响:console/前端无法展示前缀级用量;大桶定位"哪个前缀占空间"无 API。
- 建议:实现 flatten 前缀查询 API(数据已在缓存内,纯聚合与暴露工作)。
- 验收:ListBuckets/PrefixUsage API 返回与前缀过滤匹配的统计。
### P29 项)
**HS-09 get_disk_status 恒返回 Ok(唯一 TODO**`crates/heal/src/heal/storage.rs:930-943`(已亲验)。当前无生产调用方(低风险)。建议:删除该方法或接 ecstore disk 状态真实现(DiskStatus 枚举已定义)。
**HS-10 HealStorageAPI 约 1/3 方法为死代码**get_object_meta/get_object_data/put_object_data/delete_object/verify_object_integrity/ec_decode_rebuild/get_disk_status/format_disk/heal_bucket_metadata/get_object_size/get_object_checksum/list_objects_for_heal(非分页版,自带 memory_heavy 警告)均 0 调用方。建议:随 HS-01 决策一并清理或接线(死接口误导后续维护者以为存在调用路径)。
**HS-11 bitrot 自检缺失**MinIO 启动时 bitrotSelfTest 对四算法已知向量自检失败即 Fatal(防静默数据损坏)。RustFS 无等价(已亲验 grep)。建议:启动时对 HighwayHash256S 等在用算法做已知向量自检(低成本高价值)。
**HS-12 对象级 healing 元数据标记评估**MinIO heal 期间对象打 `x-minio-healing:true`RenameData 据此跳过版本清理/legacy purge(漏掉会导致 heal 与并发删除互毁)。RustFS 无对象级标记(已亲验 grep object.rs 无 healing 分支),依赖 NSLock + rename 语义。建议:审计 RustFS rename 提交路径是否存在"heal 提交与并发 delete/version 清理竞争"窗口;若无则文档化差异,若有则补标记等价机制。
**HS-13 erasure set heal 无"跳过新写入/ILM 已过期版本"过滤**MinIO resync 跳过 ModTime>tracker.Started 的版本(避免 heal 追新写入尾巴)与 ILM 已过期版本(避免白做)。RustFS erasure_healer 未实现同款过滤(按版本 dedup 有,时间/ILM 过滤无)。影响:重建尾部长尾(持续写入的桶 heal 完成判定被新版本推迟)与无效 heal 工作量。建议:disk-walk 枚举处加 started_at 时间过滤 + evaluator 预检。
**HS-14 scanner idle 语义方向与 MinIO 相反**MinIO `scanner:idle_speed=on`(默认)= 集群空闲时才节流、忙时全速;RustFS `RUSTFS_SCANNER_IDLE_MODE=true`(默认)= 限速总闸(false=完全不休眠)。两者默认行为可能相近(都限速)但参数语义不可互换,迁移文档需显式说明;若追求 mc config 兼容需重命名/重语义。建议:先文档化差异,评估是否对齐语义。
**HS-15 alert_excess_folders 默认值差异**RustFS 65538(兼容 PBS/Proxmox 布局,scanner_folder.rs:79vs MinIO 50000。行为差异默认即触发阈值不同。建议:文档化(保留 65538 有本地理由)。
**HS-16 单机默认周期钩子未启用**:`single_disk_default_cycle_secs(_features) -> None` 恒空(scanner.rs:1428-1430),单机部署无专属默认周期覆盖。建议:决定单机默认周期策略后启用或删除钩子。
**HS-17 DeleteAllVersions 批量优化核对**MinIO 用 DeletePrefix+DeletePrefixObject 单调用代替逐版本 fan-out。RustFS expiry 队列路径是否同款优化未逐行核实(集成测试覆盖行为正确性)。建议:核对 `apply_expiry_rule` 全版本删除路径,若无前缀单调用优化则评估补齐。
### P33 项)
**HS-18 trash/临时目录二段清理细节核对**:MinIO `.minio.sys/tmp/.trash` 清理(delete_cleanup_interval 默认 5m + deleteCleanupSleeper)与 stale uploads rename-into-trash 二段式。RustFS 有 delete_tail_activity.rs 与 stale multipart 任务,二段语义是否完整对齐未逐行核实。建议:对照补齐或文档化。
**HS-19 root heal 直连死路径清理**`should_handle_root_heal_directly` 恒 falseadmin/handlers/heal.rs:1200-1202,测试锁定),store.heal_format 直连分支不可达。建议:删除死分支或恢复直连路径作为集群协调失败的降级。
**HS-20 兼容旗标与死指标清理**:`RUSTFS_SCANNER_INLINE_HEAL_ENABLE`(开启仅告警)+ `rustfs_scanner_inline_heal_total` 死指标 + `rustfs_common::metrics` 中 scanner 域代码分层迁移(backlog #1843 已登记)。建议:随分层迁移一并清理。
### 按设计不追平(7 项,记录以防后续误判为缺口)
1. **bloom filter**MinIO master 已删除;RustFS `.bloomcycle.bin` 复用为 cycle/epoch 围栏与 MinIO 现状一致。
2. **scanner 集群单 leader**:双方一致;RustFS 额外有 epoch 围栏。
3. **heal 不发 S3 bucket notification**:双方一致(heal 结果走 admin status)。
4. **incomplete multipart 不在 scanner/ILM 内执行**:双方一致(独立后台例程)。
5. **内联 heal 移除**RustFS 有意为之(scanner 只入队),MinIO 的 applyHealing 内联路径不做对标。
6. **heal 序列常驻保活(10s 空白回写)**:RustFS 快照式查询模型不同,按 HS-06 处理增量语义即可,不复制流式保活。
7. **`.trash`/`tmp-old` 路径名兼容**:RustFS 布局常量独立,不逐字对齐 MinIO 路径。
---
## 6. 配置默认值总表(RustFS
healenv 前缀 `RUSTFS_HEAL_``crates/config/src/constants/heal.rs`,消费于 `manager.rs:724-800`):
| 配置 | 默认 | 热更新 |
|---|---|---|
| AUTO_HEAL_ENABLE | true | 否 |
| QUEUE_SIZE | 10000 | 否 |
| INTERVAL_SECS | 10 | 否(启动时固定) |
| TASK_TIMEOUT_SECS | 300 | 否 |
| MAX_CONCURRENT_HEALS | 4 | 否 |
| MAX_CONCURRENT_PER_SET | 1(≤min(全局,值) | 否 |
| LOW_PRIORITY_MERGE_ENABLE | true | 否 |
| LOW_PRIORITY_DROP_WHEN_FULL | true | 否 |
| PAGE_OBJECT_CONCURRENCY | 8Deep/AutoHeal 强制 1 | 否 |
| EVENT_DRIVEN_SCHEDULER_ENABLE | true | 否 |
| SET_BULKHEAD_ENABLE | true | 否 |
| PAGE_PARALLEL_ENABLE | true | 否 |
| MAINLINE_THROTTLE_ENABLE | true | 否 |
| MAINLINE_READ/WRITE_UTILIZATION_HIGH_PERCENT | 80/80 | 否 |
| MAINLINE_MAX_SLEEP_MS | 250 | 否 |
| (总开关)RUSTFS_HEAL_ENABLED | true | 否 |
| admin 子系统 heal.bitrot_cycle | 30d | 是(经 scanner runtime config |
scanneradmin 子系统 `scanner``crates/config/src/constants/scanner.rs` + `ecstore/src/config/scanner.rs` + `runtime_config.rs:527-673`):
| 键 | env | 默认 |
|---|---|---|
| speed | RUSTFS_SCANNER_SPEED | default2x/1s/60s |
| delay / max_wait / cycle / start_delay | RUSTFS_SCANNER_* | 派生/空 |
| cycle_max_duration/objects/directories | …_MAX_* | 0(不限) |
| bitrot_cycle | …_BITROT_CYCLE_SECS | 259200030d0/on=每轮,off=禁用) |
| idle_mode | …_IDLE_MODE | true |
| cache_save_timeout | …_CACHE_SAVE_TIMEOUT_SECS | 30s |
| max_concurrent_set_scans / disk_scans | …_MAX_CONCURRENT_* | 4/4 |
| yield_every_n_objects | …_YIELD_EVERY_N_OBJECTS | 128 |
| alert_excess_versions / version_size / folders | …_ALERT_* | 100 / 1TiB / 65538 |
scanner 内部 env`RUSTFS_DATA_USAGE_UPDATE_DIR_CYCLES=16``RUSTFS_HEAL_OBJECT_SELECT_PROB=1024``RUSTFS_SCANNER_DEEP_VERIFY_COOLDOWN_SECS=60``RUSTFS_DATA_USAGE_FAILED_OBJECT_TTL_SECS=86400`/`_MAX=10000``RUSTFS_LOCK_ACQUIRE_TIMEOUT=5s``RUSTFS_SCANNER_ENABLED=true``RUSTFS_SCANNER_INLINE_HEAL_ENABLE=false`(兼容告警)。
全部 17 个 scanner 键支持 env > config 双通道 + admin PUT 热更(generation+Notify 即时生效);heal 运行时参数目前仅 env(无 admin 热更入口,`Arc<RwLock<HealConfig>>` 结构已预留)。
---
## 7. 相关 backlog / 历史索引
- 换盘自动修复系列(已闭环):backlog #1786(冗余假绿算法)、#1787(目标槽位限定)、#1789resume 与 healing marker 绑定 replacement 实例)、#1791(黑白盒验收矩阵)。
- #801 DiskInfo.healing 从未赋值(已修复闭环,现 `set_disk/mod.rs:4988` 有赋值链)。
- #1651 Scanner 指标节点/source/bucket-drive 维度(OPEN,本分析 §3.8/§4.6 相关)。
- #1843 crates/common 83% scanner/heal 域代码分层迁移(OPEN,含 HS-20)。
- 代码注释引用的历史缺陷(现已有防护与回归测试):#856/#799 B7(离线盘误记 healed)、#855/B6/#1033skip 不得标记完成)、#920sub-quorum 并集枚举)、#856 B5(按版本续扫)、#5173bitrot trailing bytes)、#5029(回归节点 stale 版本合并)。
- v1 对标文档:`docs/rustfs-heal-scanner-vs-minio-parity-assessment.md`(本文取代)、落地手册 `docs/rustfs-heal-scanner-vs-minio-improvement-playbook.md`(部分条目已被后续实现超越)。
- 换盘深度分析:`docs/new-disk-replacement-and-healing-deep-analysis-zh.md``docs/node-disk-identity-and-healing-analysis-zh.md`
## 8. 审计方法与局限
- 四路并行审计(heal crate 逐文件、scanner crate 逐文件、ecstore 集成层 wiring、MinIO master 源码研究)+ 主会话对关键"缺失"结论逐条亲验(get_disk_status TODO、HealEvent 零外部引用、.bloomcycle.bin 无 bloom 实现、check_abandoned_parts 三层 NotImplemented、ETag 兜底已实现、trace 通道零命中、already_running 语义)。
- 未逐行核实的点(已在文中标注"未确认/未逐行核"):DeleteAllVersions 前缀单调用优化(HS-17)、trash 二段清理细节(HS-18)、ilm worker 默认值对照、stale multipart 默认值对照、mc CLI flag 逐字拼写(MinIO 侧)。其中 HS-17 与 HS-18 已于 2026-08-19 完成逐行核实,结论见 §9.2/§9.3。
- MinIO 侧引用以其 master `7aac2a2c5b` 为准;RustFS 侧行号以 2026-08-16 工作区为准,后续演进请以符号名检索为准。
## 9. 落地结果(2026-08-19 更新)
本审计衍生的 14 个子 issuebacklog #1865~#1878)已全部闭环。本节为差距清单 HS-01~HS-20 的最终处置记录,也是下一轮对标重审的增量基线。
### 9.1 已落地(PR 均已合并 main)
- HS-01 MRF 接线 + 持久化修复账本(#1865PR #6189):决策选 (a)。common MRF channelbounded 8192、try_send 永不阻塞)+ heal mrf_queue100k 条 / 8MiB 双限环形)+ `buckets/.heal/mrf/journal.bin` CRC 持久化回放(torn tail 截断、回放后删除)+ 三投递点(read decode_error→Urgent ECDecode、scanner 元数据损坏→High Metadata、add_partial→Normal+ `RUSTFS_HEAL_MRF_ENABLE` 一键回退。
- HS-02 abandoned parts/data-dir 对账(#1866PR #6179):接通 abandoned 检查入口,保留 dry-run / reclaim 计数。
- HS-03 heal/scanner trace 通道(#1867PR #6179):进程内 trace bus + `/v3/trace` admin 流式订阅 + heal task / abandoned-parts / scanner folder / ILM / heal-candidate trace producer。
- HS-04 scanner 超限 S3 事件(#1868PR #6176):`s3:Scanner:ManyVersions/LargeVersions/BigPrefix` 三事件 + 24h 边沿冷却;HS-15 阈值差异文档化(`docs/operations/scanner-excess-alerts.md`)。
- HS-05 madmin 客户端一期(#1869PR #6166):SigV4 admin 客户端 heal/scanner 方法;增量消费方法待 follow-up(协议已由 HS-06 并入)。
- HS-06 admin heal 增量语义与类型化重叠(#1870PR #6206):`sinceSeq/nextSeq/minSeq` 增量游标(wire additive、缺省=全量快照)+ `RUSTFS_HEAL_OVERLAP_POLICY`(默认 merge 不变;minio_error 下 AlreadyRunning/OverlappingPaths 类型化拒绝)+ forceStart 先停旧再启新。
- HS-07 healing 进度可见性(#1871PR #6179):data-usage 总量基线 + baseline/current/healed 计数。
- HS-08 prefix usage#1872PR #6171):`GET /v3/usage/{bucket}`
- HS-11 bitrot 启动自检(#1873PR #6165)。
- HS-13 heal 跳过过滤(#1875PR #6179):过滤命中版本不再计为失败。
- HS-16 单机周期钩子(#1878PR #6250):删恒 None 钩子,决策记录见 `docs/operations/heal-scanner-parity-notes-zh.md`
- HS-09/10/19/20 死代码清理批(#1877PR #6256):净 911 行零行为变更;`get_disk_status` TODO(全仓库唯一产品 TODO)清零;HS-01 联动的 `ec_decode_rebuild`/`get_object_meta` 保留并加 Reserved 注释(MRF 当前经 `heal_object` 执行)。
### 9.2 核对后确认"已实现 / 非缺口"(审计期误判修正,累计四例)
- bloom filter(§0 已修正):MinIO master 已删除,双方现状一致。
- ETag 兜底仲裁(§0 已修正):RustFS 已有实现(`set_disk/ops/heal.rs`)。
- HS-17#18762026-08-19 逐行核实后关闭):DeleteAllVersions 前缀单调用优化 RustFS 已完整实现——`apply_expiry_on_non_transitioned_objects``delete_all()` 两 action 设 `delete_prefix + delete_prefix_object` 后单次 `delete_object``bucket_lifecycle_ops.rs:5047-5056`),SetDisks 分支一次写锁 + 一次全版本 quorum 读 + 内联逐版本 object-lock 检查(`set_disk/ops/object.rs:5566-5612`),与 MinIO `expire.go``applyExpiryOnNonTransitionedObjects` 逐行对齐。§8 原列"未逐行核实"的本项已有结论:现状即优化路径,无需实现。
- HS-14#1878PR #6250 附带核对):MinIO"idle=空闲才节流"是 2024-01 minio/minio#18734 之前的行为(`scannerIdleMode` 现为静态配置,`idle_speed=on` 默认即始终按速度档节流,"idle"命名是历史残留);RustFS `RUSTFS_SCANNER_IDLE_MODE` 与 MinIO 当前语义方向一致,且另有 MinIO 没有的前台读退避下限。真实迁移陷阱(变量须 `RUSTFS_` 前缀、`on/off` vs `true/false` 词表、`false` 连前台保护一起关)已文档化于 `docs/operations/heal-scanner-parity-notes-zh.md`
### 9.3 审计型结论(无需改代码)
- HS-12#1874PR #6183):不存在 MinIO 用 `x-minio-healing` 防御的那类竞争——所有同 (bucket, object) 提交面在同一把对象级 ns 写锁互斥,heal 锁 guard 覆盖 rename 提交全程;交付 2 个并发不变量回归测试 + `docs/operations/heal-concurrency-safety-notes-zh.md` 交点矩阵。
- HS-18#18782026-08-19 逐行核实):trash/tmp 三段清理全对齐——stale multipart 隔离-清理等价且更安全(`delete_all_with_quorum` 逐盘递归删即 `move_to_trash` rename 进 `.rustfs.sys/tmp/.trash`,另有锁 + fence)、trash 排空基本等价(无逐条 sleeper 节流,5m 周期天然限频)、tmp 非 trash 24h 回收等价(RustFS 5m 比 MinIO 6h 更及时);周期默认 24h/6h/5m 三项全对齐。§8 原列"未逐行核实"的本项已有结论。
### 9.4 移交 follow-up(汇总于 backlog#1862 评论区)
HS-01 bitrot GET→MRF 全链路 e2e、kill -9 journal 回放 e2e、队列满压测 RSS(≤ 预算+10%);HS-05/06 madmin 增量消费方法 + wire 单一来源化 + embedded e2e + 多轮轮询 soakHS-08 多盘 scanner 周期 e2eHS-04 超限审计条目;HS-18 低于 quorum 的 stale-multipart 崩溃残留窗口(扇出中途崩溃且已清盘数 > parity 时 FileNotFound 不在忽略集导致不自然收敛,修复需专用 quorum 变体)。
下一轮重审建议:跟随 heal/scanner 下一个大特性落地后触发,以本节为增量基线。
+9 -32
View File
@@ -13,7 +13,7 @@
// limitations under the License.
use crate::admin::{
auth::authorize_admin_request,
auth::validate_admin_request,
handlers::audit_runtime_config::{load_server_config_from_store, update_audit_config_and_reload},
handlers::target_descriptor::{
AdminTargetSpec, EndpointKey, RuntimeHealthStatus, TargetEndpointSource, admin_target_spec_from_builtin,
@@ -23,8 +23,9 @@ use crate::admin::{
},
router::{AdminOperation, Operation, S3Router},
};
use crate::auth::{check_key_valid, get_session_token};
use crate::server::{
ADMIN_PREFIX, is_audit_module_enabled, refresh_audit_module_enabled, refresh_persisted_module_switches_from_store,
ADMIN_PREFIX, RemoteAddr, is_audit_module_enabled, refresh_audit_module_enabled, refresh_persisted_module_switches_from_store,
};
use http::StatusCode;
use hyper::Method;
@@ -212,14 +213,14 @@ fn audit_target_specs() -> &'static [AdminTargetSpec] {
&AUDIT_TARGET_SPECS
}
/// The pre-check keeps these endpoints' historical missing-credentials message;
/// the shared gate reports "get cred failed".
async fn authorize_audit_admin_request(req: &S3Request<Body>, action: AdminAction) -> S3Result<()> {
if req.credentials.is_none() {
let Some(input_cred) = &req.credentials else {
return Err(s3_error!(InvalidRequest, "credentials not found"));
}
authorize_admin_request(req, vec![Action::AdminAction(action)]).await?;
Ok(())
};
let (cred, owner) =
check_key_valid(get_session_token(&req.uri, &req.headers).unwrap_or_default(), &input_cred.access_key).await?;
let remote_addr = req.extensions.get::<Option<RemoteAddr>>().and_then(|opt| opt.map(|a| a.0));
validate_admin_request(&req.headers, &cred, owner, false, vec![Action::AdminAction(action)], remote_addr).await
}
fn audit_target_mutation_block_reason(config: &Config, target_type: &str, target_name: &str) -> S3Result<Option<String>> {
@@ -823,30 +824,6 @@ mod tests {
});
}
/// These endpoints authorize through the shared admin gate, which reports
/// "get cred failed" for a credential-less request. The pre-check keeps the
/// message they have always returned (rustfs/backlog#1829).
#[tokio::test]
async fn audit_target_gate_keeps_its_missing_credentials_message() {
let req = S3Request {
input: Body::from(String::new()),
method: Method::PUT,
uri: http::Uri::from_static("/rustfs/admin/v3/audit/target"),
headers: http::HeaderMap::new(),
extensions: http::Extensions::new(),
credentials: None,
region: None,
service: None,
trailing_headers: None,
};
let err = authorize_audit_admin_request(&req, AdminAction::SetBucketTargetAction)
.await
.expect_err("a request without credentials must be rejected");
assert_eq!(err.code(), &s3s::S3ErrorCode::InvalidRequest);
assert_eq!(err.message(), Some("credentials not found"));
}
#[test]
fn audit_target_handlers_require_admin_authorization_contract() {
let src = include_str!("audit.rs");
+17 -32
View File
@@ -14,7 +14,7 @@
use crate::admin::storage_api::cluster::{CapabilityState, CapabilityStatus, ObservabilitySnapshot, TopologySnapshot};
use crate::admin::{
auth::authorize_admin_request,
auth::validate_admin_request,
router::{AdminOperation, Operation, S3Router},
runtime_sources::default_admin_usecase,
storage_api::cluster::{
@@ -24,10 +24,11 @@ use crate::admin::{
},
system,
};
use crate::auth::{check_key_valid, get_session_token};
use crate::cluster_snapshot::{
ClusterReadOnlySnapshot, ClusterRuntimeReadinessState, ClusterRuntimeStatusSnapshot, cluster_has_actionable_pressure,
};
use crate::server::{ADMIN_PREFIX, ReadinessDegradedReason};
use crate::server::{ADMIN_PREFIX, ReadinessDegradedReason, RemoteAddr};
use http::{HeaderMap, HeaderValue, StatusCode};
use hyper::Method;
use matchit::Params;
@@ -65,15 +66,23 @@ pub(crate) struct ClusterSnapshotDiscoveryResponse {
pub components: Option<ClusterComponentStatusView>,
}
/// The pre-check keeps this endpoint's historical missing-credentials message;
/// the shared gate reports "get cred failed".
async fn authorize_cluster_snapshot_request(req: &S3Request<Body>) -> S3Result<()> {
if req.credentials.is_none() {
let Some(input_cred) = &req.credentials else {
return Err(s3_error!(InvalidRequest, "authentication required"));
}
};
authorize_admin_request(req, vec![Action::AdminAction(AdminAction::ServerInfoAdminAction)]).await?;
Ok(())
let (cred, owner) =
check_key_valid(get_session_token(&req.uri, &req.headers).unwrap_or_default(), &input_cred.access_key).await?;
validate_admin_request(
&req.headers,
&cred,
owner,
false,
vec![Action::AdminAction(AdminAction::ServerInfoAdminAction)],
req.extensions.get::<Option<RemoteAddr>>().and_then(|opt| opt.map(|a| a.0)),
)
.await
}
fn build_json_response(
@@ -944,30 +953,6 @@ mod tests {
);
}
/// This endpoint authorizes through the shared admin gate, which reports
/// "get cred failed" for a credential-less request. The pre-check keeps the
/// message it has always returned (rustfs/backlog#1829).
#[tokio::test]
async fn cluster_snapshot_gate_keeps_its_missing_credentials_message() {
let req = s3s::S3Request {
input: s3s::Body::from(String::new()),
method: http::Method::GET,
uri: http::Uri::from_static("/rustfs/admin/v4/cluster/snapshot"),
headers: http::HeaderMap::new(),
extensions: http::Extensions::new(),
credentials: None,
region: None,
service: None,
trailing_headers: None,
};
let err = super::authorize_cluster_snapshot_request(&req)
.await
.expect_err("a request without credentials must be rejected");
assert_eq!(err.code(), &s3s::S3ErrorCode::InvalidRequest);
assert_eq!(err.message(), Some("authentication required"));
}
#[test]
fn cluster_snapshot_response_serializes_none_snapshot() {
let value = serde_json::to_value(ClusterSnapshotResponse { snapshot: None }).expect("serialize response");
+23 -165
View File
@@ -23,10 +23,11 @@
//! backing infrastructure (in-process log ring buffer, cross-node object
//! speedtest harness).
use crate::admin::auth::authorize_admin_request;
use crate::admin::auth::validate_admin_request;
use crate::admin::router::{AdminOperation, Operation, S3Router};
use crate::admin::storage_api::access::spawn_traced;
use crate::server::ADMIN_PREFIX;
use crate::auth::{check_key_valid, get_session_token};
use crate::server::{ADMIN_PREFIX, RemoteAddr};
use crate::storage::storage_api::get_global_lock_clients;
use bytes::Bytes;
use futures::{Stream, StreamExt, future::join_all};
@@ -41,7 +42,6 @@ use s3s::{Body, S3Error, S3ErrorCode, S3Request, S3Response, S3Result, StdError,
use serde::{Deserialize, Serialize};
use std::collections::HashMap;
use std::pin::Pin;
use std::sync::Arc;
use std::task::{Context, Poll};
use std::time::{Duration, SystemTime};
use tokio::sync::{Semaphore, SemaphorePermit, mpsc};
@@ -132,15 +132,16 @@ pub fn register_diagnostics_route(r: &mut S3Router<AdminOperation>) -> std::io::
// Shared auth helper
// ---------------------------------------------------------------------------
/// The pre-check keeps these endpoints' historical `AccessDenied` missing-credentials
/// response; the shared gate reports `InvalidRequest` "get cred failed".
async fn authorize(req: &S3Request<Body>, action: AdminAction) -> S3Result<()> {
if req.credentials.is_none() {
let Some(input_cred) = req.credentials.as_ref() else {
return Err(s3_error!(AccessDenied, "Signature is required"));
}
};
authorize_admin_request(req, vec![Action::AdminAction(action)]).await?;
Ok(())
let (cred, owner) =
check_key_valid(get_session_token(&req.uri, &req.headers).unwrap_or_default(), &input_cred.access_key).await?;
let remote_addr = req.extensions.get::<Option<RemoteAddr>>().and_then(|opt| opt.map(|a| a.0));
validate_admin_request(&req.headers, &cred, owner, false, vec![Action::AdminAction(action)], remote_addr).await
}
fn json_response<T: Serialize>(status: StatusCode, value: &T) -> S3Result<S3Response<(StatusCode, Body)>> {
@@ -247,14 +248,13 @@ struct LeaseHolderState {
acquired_at: SystemTime,
ttl_secs: u64,
holder_count: u32,
guard_ids: Option<Vec<u64>>,
}
fn build_top_locks_response(
limit: usize,
now: SystemTime,
lease_infos: Vec<LockLeaseInfo>,
fast_infos: Vec<(rustfs_lock::ObjectLockInfo, u32, Option<Vec<u64>>)>,
fast_infos: Vec<(rustfs_lock::ObjectLockInfo, u32)>,
) -> TopLocksResponse {
let mut lease_holders = HashMap::with_capacity(lease_infos.len());
@@ -277,27 +277,17 @@ fn build_top_locks_response(
}
state.ttl_secs = state.ttl_secs.max(ttl_secs);
state.holder_count = state.holder_count.saturating_add(1);
match (state.guard_ids.as_mut(), info.guard_id) {
(Some(guard_ids), Some(guard_id)) => guard_ids.push(guard_id),
_ => state.guard_ids = None,
}
})
.or_insert(LeaseHolderState {
acquired_at: info.acquired_at,
ttl_secs,
holder_count: 1,
guard_ids: info.guard_id.map(|guard_id| vec![guard_id]),
});
}
for state in lease_holders.values_mut() {
if let Some(guard_ids) = &mut state.guard_ids {
guard_ids.sort_unstable();
}
}
let mut infos: Vec<_> = fast_infos
.into_iter()
.map(|(info, holder_count, guard_ids)| {
.map(|(info, holder_count)| {
let mode = match info.mode {
LockMode::Shared => TopLockMode::Read,
LockMode::Exclusive => TopLockMode::Write,
@@ -308,10 +298,11 @@ fn build_top_locks_response(
owner: info.owner.to_string(),
};
let priority = lock_priority_label(info.priority);
// Match the complete holder cohort so replacements cannot reuse stale lease data.
// Shared-owner timestamps do not roll back when a newer sibling releases, so only their count is stable.
let state = match lease_holders.remove(&key) {
Some(lease)
if lease.holder_count == holder_count && lease.guard_ids.is_some() && lease.guard_ids == guard_ids =>
if lease.holder_count == holder_count
&& (mode == TopLockMode::Read || info.acquired_at <= lease.acquired_at) =>
{
TopLockState {
acquired_at: lease.acquired_at,
@@ -358,11 +349,8 @@ fn build_top_locks_response(
}
}
async fn collect_top_locks_with_clients(
limit: usize,
manager: Arc<rustfs_lock::GlobalLockManager>,
clients: Vec<Arc<dyn rustfs_lock::client::LockClient>>,
) -> TopLocksResponse {
async fn collect_top_locks(limit: usize) -> TopLocksResponse {
let manager = get_global_lock_manager();
let Some(fast) = manager.as_fast_lock_manager() else {
return TopLocksResponse {
total: 0,
@@ -374,29 +362,21 @@ async fn collect_top_locks_with_clients(
};
};
let lease_infos = if clients.is_empty() {
Vec::new()
} else {
join_all(clients.iter().map(|client| client.list_lock_leases()))
let lease_infos = if let Some(clients) = get_global_lock_clients() {
join_all(clients.values().map(|client| client.list_lock_leases()))
.await
.into_iter()
.flatten()
.collect()
} else {
Vec::new()
};
// Capture holders last so released or replaced lease guards fail the merge checks.
let fast_infos = fast.list_locks_with_holder_generations();
let fast_infos = fast.list_locks_with_holder_counts();
build_top_locks_response(limit, SystemTime::now(), lease_infos, fast_infos)
}
async fn collect_top_locks(limit: usize) -> TopLocksResponse {
let manager = get_global_lock_manager();
let clients = get_global_lock_clients()
.map(|clients| clients.values().cloned().collect())
.unwrap_or_default();
collect_top_locks_with_clients(limit, manager, clients).await
}
fn parse_top_locks_limit(uri: &Uri) -> usize {
query_value(uri, "count")
.and_then(|v| v.parse::<usize>().ok())
@@ -1076,22 +1056,6 @@ mod tests {
}
}
/// These endpoints authorize through the shared admin gate, which rejects a
/// credential-less request with `InvalidRequest` "get cred failed". The
/// pre-check keeps the `AccessDenied` response they have always returned
/// (rustfs/backlog#1829).
#[tokio::test]
async fn diagnostics_gate_keeps_its_missing_credentials_response() {
let err = authorize(
&build_request(Method::GET, "/rustfs/admin/v3/top/locks"),
AdminAction::ServerInfoAdminAction,
)
.await
.expect_err("a request without credentials must be rejected");
assert_eq!(err.code(), &S3ErrorCode::AccessDenied);
assert_eq!(err.message(), Some("Signature is required"));
}
#[tokio::test]
async fn top_locks_handler_rejects_missing_credentials() {
let err = TopLocksHandler {}
@@ -1265,7 +1229,6 @@ mod tests {
let mixed_resource = ObjectKey::new("bucket", "mixed-object");
let replaced_resource = ObjectKey::new("bucket", "replaced-object");
let remaining_shared_resource = ObjectKey::new("bucket", "remaining-shared-object");
let opaque_resource = ObjectKey::new("bucket", "opaque-object");
let response = build_top_locks_response(
TOP_LOCKS_DEFAULT_LIMIT,
@@ -1276,7 +1239,6 @@ mod tests {
lock_type: LockType::Shared,
owner: "owner-a".to_string(),
acquired_at: now - Duration::from_secs(50),
guard_id: Some(11),
remaining_ttl: Duration::from_secs(5),
},
LockLeaseInfo {
@@ -1284,7 +1246,6 @@ mod tests {
lock_type: LockType::Shared,
owner: "owner-a".to_string(),
acquired_at: now - Duration::from_secs(40),
guard_id: Some(18),
remaining_ttl: Duration::from_secs(20),
},
LockLeaseInfo {
@@ -1292,7 +1253,6 @@ mod tests {
lock_type: LockType::Shared,
owner: "owner-c".to_string(),
acquired_at: now - Duration::from_secs(30),
guard_id: Some(13),
remaining_ttl: Duration::from_secs(25),
},
LockLeaseInfo {
@@ -1300,7 +1260,6 @@ mod tests {
lock_type: LockType::Exclusive,
owner: "owner-d".to_string(),
acquired_at: now - Duration::from_secs(15),
guard_id: Some(12),
remaining_ttl: Duration::from_secs(18),
},
LockLeaseInfo {
@@ -1308,7 +1267,6 @@ mod tests {
lock_type: LockType::Exclusive,
owner: "owner-e".to_string(),
acquired_at: now - Duration::from_secs(30),
guard_id: Some(14),
remaining_ttl: Duration::from_secs(25),
},
LockLeaseInfo {
@@ -1316,17 +1274,8 @@ mod tests {
lock_type: LockType::Shared,
owner: "owner-f".to_string(),
acquired_at: now - Duration::from_secs(30),
guard_id: Some(16),
remaining_ttl: Duration::from_secs(22),
},
LockLeaseInfo {
resource: opaque_resource.clone(),
lock_type: LockType::Exclusive,
owner: "owner-g".to_string(),
acquired_at: now - Duration::from_secs(30),
guard_id: None,
remaining_ttl: Duration::from_secs(30),
},
],
vec![
(
@@ -1339,7 +1288,6 @@ mod tests {
priority: rustfs_lock::fast_lock::LockPriority::Normal,
},
1,
Some(vec![15]),
),
(
rustfs_lock::ObjectLockInfo {
@@ -1351,7 +1299,6 @@ mod tests {
priority: rustfs_lock::fast_lock::LockPriority::Normal,
},
1,
Some(vec![16]),
),
(
rustfs_lock::ObjectLockInfo {
@@ -1363,7 +1310,6 @@ mod tests {
priority: rustfs_lock::fast_lock::LockPriority::Normal,
},
1,
Some(vec![12]),
),
(
rustfs_lock::ObjectLockInfo {
@@ -1375,7 +1321,6 @@ mod tests {
priority: rustfs_lock::fast_lock::LockPriority::Normal,
},
2,
Some(vec![11, 18]),
),
(
rustfs_lock::ObjectLockInfo {
@@ -1387,7 +1332,6 @@ mod tests {
priority: rustfs_lock::fast_lock::LockPriority::Normal,
},
1,
Some(vec![17]),
),
(
rustfs_lock::ObjectLockInfo {
@@ -1399,24 +1343,11 @@ mod tests {
priority: rustfs_lock::fast_lock::LockPriority::Normal,
},
2,
None,
),
(
rustfs_lock::ObjectLockInfo {
key: opaque_resource,
mode: LockMode::Exclusive,
owner: "owner-g".into(),
acquired_at: now - Duration::from_secs(4),
expires_at: now + Duration::from_secs(6),
priority: rustfs_lock::fast_lock::LockPriority::Normal,
},
1,
None,
),
],
);
assert_eq!(response.total, 7);
assert_eq!(response.total, 6);
let leased = response
.locks
.iter()
@@ -1461,47 +1392,6 @@ mod tests {
.find(|entry| entry.object == "remaining-shared-object")
.expect("an older surviving shared lease should remain lease-backed");
assert_eq!(remaining_shared.ttl_secs, 22);
let opaque = response
.locks
.iter()
.find(|entry| entry.object == "opaque-object")
.expect("generation-less holder should remain visible");
assert_eq!(opaque.ttl_secs, 6);
}
#[test]
fn top_locks_rejects_replaced_shared_generation() {
let now = SystemTime::UNIX_EPOCH + Duration::from_secs(100);
let resource = ObjectKey::new("bucket", "replaced-shared-object");
let response = build_top_locks_response(
TOP_LOCKS_DEFAULT_LIMIT,
now,
vec![LockLeaseInfo {
resource: resource.clone(),
lock_type: LockType::Shared,
owner: "owner-a".to_string(),
acquired_at: now - Duration::from_secs(30),
guard_id: Some(1),
remaining_ttl: Duration::from_secs(20),
}],
vec![(
rustfs_lock::ObjectLockInfo {
key: resource,
mode: LockMode::Shared,
owner: "owner-a".into(),
acquired_at: now - Duration::from_secs(2),
expires_at: now + Duration::from_secs(4),
priority: rustfs_lock::fast_lock::LockPriority::Normal,
},
1,
Some(vec![2]),
)],
);
let entry = response.locks.first().expect("replacement remains visible");
assert_eq!(entry.ttl_secs, 4);
assert_eq!(entry.elapsed_secs, 2);
}
#[tokio::test]
@@ -1541,36 +1431,4 @@ mod tests {
drop(guard);
}
#[tokio::test(start_paused = true)]
async fn collect_top_locks_uses_refreshed_local_lease() {
use rustfs_lock::{FastObjectLockManager, GlobalLockManager, LocalClient, LockClient, LockRequest};
let manager = Arc::new(GlobalLockManager::Enabled(Arc::new(FastObjectLockManager::new())));
let client = Arc::new(LocalClient::with_manager(manager.clone()));
let request = LockRequest::new(ObjectKey::new("diag-bucket", "renewed-object"), LockType::Exclusive, "diag-owner")
.with_ttl(Duration::from_secs(30));
let lock_id = request.lock_id.clone();
assert!(
client
.acquire_lock(&request)
.await
.expect("local lock acquisition should succeed")
.success
);
tokio::time::advance(Duration::from_secs(20)).await;
assert!(client.refresh(&lock_id).await.expect("local lease refresh should succeed"));
let clients: Vec<Arc<dyn rustfs_lock::client::LockClient>> = vec![client.clone()];
let response = collect_top_locks_with_clients(TOP_LOCKS_DEFAULT_LIMIT, manager, clients).await;
let entry = response
.locks
.iter()
.find(|entry| entry.bucket == "diag-bucket" && entry.object == "renewed-object")
.expect("refreshed local lock should be listed");
assert!(entry.ttl_secs >= 29, "collector must use the refreshed lease deadline");
assert!(client.release(&lock_id).await.expect("local lock release should succeed"));
}
}
+10 -32
View File
@@ -13,7 +13,7 @@
// limitations under the License.
use crate::admin::{
auth::authorize_admin_request,
auth::validate_admin_request,
handlers::notify_runtime_access::{get_notification_system, load_notification_config_snapshot},
handlers::supervise_admin_mutation,
handlers::target_descriptor::{
@@ -26,8 +26,10 @@ use crate::admin::{
runtime_sources::{AppContext, app_context_from_req},
service::config::{preflight_dynamic_config_reload_for_context, signal_dynamic_config_reload_checked_for_context},
};
use crate::auth::{check_key_valid, get_session_token};
use crate::server::{
ADMIN_PREFIX, is_notify_module_enabled, refresh_notify_module_enabled, refresh_persisted_module_switches_from_store,
ADMIN_PREFIX, RemoteAddr, is_notify_module_enabled, refresh_notify_module_enabled,
refresh_persisted_module_switches_from_store,
};
use http::StatusCode;
use hyper::Method;
@@ -262,14 +264,14 @@ fn notification_target_specs() -> &'static [AdminTargetSpec] {
// --- Helper Functions ---
/// The pre-check keeps these endpoints' historical missing-credentials message;
/// the shared gate reports "get cred failed".
async fn authorize_notification_admin_request(req: &S3Request<Body>, action: AdminAction) -> S3Result<()> {
if req.credentials.is_none() {
let Some(input_cred) = &req.credentials else {
return Err(s3_error!(InvalidRequest, "credentials not found"));
}
authorize_admin_request(req, vec![Action::AdminAction(action)]).await?;
Ok(())
};
let (cred, owner) =
check_key_valid(get_session_token(&req.uri, &req.headers).unwrap_or_default(), &input_cred.access_key).await?;
let remote_addr = req.extensions.get::<Option<RemoteAddr>>().and_then(|opt| opt.map(|a| a.0));
validate_admin_request(&req.headers, &cred, owner, false, vec![Action::AdminAction(action)], remote_addr).await
}
fn target_mutation_block_reason(config: &Config, target_type: &str, target_name: &str) -> S3Result<Option<String>> {
@@ -985,30 +987,6 @@ mod tests {
);
}
/// These endpoints authorize through the shared admin gate, which reports
/// "get cred failed" for a credential-less request. The pre-check keeps the
/// message they have always returned (rustfs/backlog#1829).
#[tokio::test]
async fn notification_target_gate_keeps_its_missing_credentials_message() {
let req = S3Request {
input: Body::from(String::new()),
method: Method::PUT,
uri: http::Uri::from_static("/rustfs/admin/v3/notification/target"),
headers: http::HeaderMap::new(),
extensions: http::Extensions::new(),
credentials: None,
region: None,
service: None,
trailing_headers: None,
};
let err = authorize_notification_admin_request(&req, AdminAction::SetBucketTargetAction)
.await
.expect_err("a request without credentials must be rejected");
assert_eq!(err.code(), &s3s::S3ErrorCode::InvalidRequest);
assert_eq!(err.message(), Some("credentials not found"));
}
#[test]
fn notification_target_handlers_require_admin_authorization_contract() {
let src = include_str!("event.rs");
+31 -44
View File
@@ -14,7 +14,7 @@
use crate::admin::storage_api::cluster::CapabilityStatus;
use crate::admin::{
auth::authorize_admin_request,
auth::validate_admin_request,
handlers::{cluster_snapshot, plugins_instances, system},
plugin_contract::{
PluginContractDomain, PluginInstanceDiagnosticCode, PluginInstanceDiagnosticCount, PluginInstanceEntry,
@@ -22,7 +22,8 @@ use crate::admin::{
},
router::{AdminOperation, Operation, S3Router},
};
use crate::server::ADMIN_PREFIX;
use crate::auth::{check_key_valid, get_session_token};
use crate::server::{ADMIN_PREFIX, RemoteAddr};
use http::{HeaderMap, HeaderValue, StatusCode};
use hyper::Method;
use matchit::Params;
@@ -182,26 +183,42 @@ fn map_extension_instance(instance: PluginInstanceEntry) -> ExtensionInstanceEnt
}
}
/// The pre-check keeps this endpoint's historical missing-credentials message;
/// the shared gate reports "get cred failed".
async fn authorize_extension_catalog_request(req: &S3Request<Body>) -> S3Result<()> {
if req.credentials.is_none() {
let Some(input_cred) = &req.credentials else {
return Err(s3_error!(InvalidRequest, "authentication required"));
}
};
authorize_admin_request(req, vec![Action::AdminAction(AdminAction::ServerInfoAdminAction)]).await?;
Ok(())
let (cred, owner) =
check_key_valid(get_session_token(&req.uri, &req.headers).unwrap_or_default(), &input_cred.access_key).await?;
validate_admin_request(
&req.headers,
&cred,
owner,
false,
vec![Action::AdminAction(AdminAction::ServerInfoAdminAction)],
req.extensions.get::<Option<RemoteAddr>>().and_then(|opt| opt.map(|a| a.0)),
)
.await
}
/// The pre-check keeps this endpoint's historical missing-credentials message;
/// the shared gate reports "get cred failed".
async fn authorize_extension_instance_request(req: &S3Request<Body>) -> S3Result<()> {
if req.credentials.is_none() {
let Some(input_cred) = &req.credentials else {
return Err(s3_error!(InvalidRequest, "authentication required"));
}
};
authorize_admin_request(req, vec![Action::AdminAction(AdminAction::GetBucketTargetAction)]).await?;
Ok(())
let (cred, owner) =
check_key_valid(get_session_token(&req.uri, &req.headers).unwrap_or_default(), &input_cred.access_key).await?;
validate_admin_request(
&req.headers,
&cred,
owner,
false,
vec![Action::AdminAction(AdminAction::GetBucketTargetAction)],
req.extensions.get::<Option<RemoteAddr>>().and_then(|opt| opt.map(|a| a.0)),
)
.await
}
fn build_json_response(
@@ -303,36 +320,6 @@ mod tests {
);
}
/// Both extension gates authorize through the shared admin gate, which reports
/// "get cred failed" for a credential-less request. The pre-check keeps the
/// message these endpoints have always returned (rustfs/backlog#1829).
#[tokio::test]
async fn extension_gates_keep_their_missing_credentials_message() {
let credential_less_request = || s3s::S3Request {
input: s3s::Body::from(String::new()),
method: http::Method::GET,
uri: http::Uri::from_static("/rustfs/admin/v4/extensions/catalog"),
headers: http::HeaderMap::new(),
extensions: http::Extensions::new(),
credentials: None,
region: None,
service: None,
trailing_headers: None,
};
for err in [
super::authorize_extension_catalog_request(&credential_less_request())
.await
.expect_err("a request without credentials must be rejected"),
super::authorize_extension_instance_request(&credential_less_request())
.await
.expect_err("a request without credentials must be rejected"),
] {
assert_eq!(err.code(), &s3s::S3ErrorCode::InvalidRequest);
assert_eq!(err.message(), Some("authentication required"));
}
}
#[test]
fn builtin_ops_schemas_register_cleanly_in_runtime_registries() {
let mut diagnostics_registry = rustfs_targets::OpsDiagnosticsRegistry::new();
+124 -57
View File
@@ -14,9 +14,10 @@
use crate::admin::auth::{authenticate_request, validate_admin_request};
use crate::admin::router::{AdminOperation, Operation, S3Router};
use crate::admin::runtime_sources::app_context_from_req;
use crate::admin::runtime_sources::{app_context_from_req, object_store_from_extensions};
use crate::admin::storage_api::bucket::is_reserved_or_invalid_bucket;
use crate::admin::storage_api::bucket::utils::is_valid_object_prefix;
use crate::admin::storage_api::contract::heal::HealOperations as _;
use crate::server::ADMIN_PREFIX;
use crate::server::RemoteAddr;
use crate::storage::rpc::node_service::heal::{
@@ -1218,6 +1219,41 @@ fn validate_heal_request_mode(hip: &HealInitParams) -> S3Result<()> {
Ok(())
}
fn should_handle_root_heal_directly(_hip: &HealInitParams) -> bool {
false
}
fn map_root_heal_status(heal_err: Option<crate::admin::storage_api::error::Error>) -> S3Result<()> {
match heal_err {
None => Ok(()),
Some(crate::admin::storage_api::error::StorageError::NoHealRequired) => {
info!(
event = EVENT_ADMIN_RESPONSE_EMITTED,
component = LOG_COMPONENT_ADMIN_API,
subsystem = LOG_SUBSYSTEM_HEAL_ADMIN,
operation = "root_heal",
result = "success",
state = "no_heal_required",
"admin response emitted"
);
Ok(())
}
Some(err) => {
warn!(
event = EVENT_ADMIN_REQUEST_FAILED,
component = LOG_COMPONENT_ADMIN_API,
subsystem = LOG_SUBSYSTEM_HEAL_ADMIN,
operation = "root_heal",
result = "failed",
reason = "root_heal_failed",
error = %err,
"admin request failed"
);
Err(s3_error!(InternalError, "root heal failed: {err}"))
}
}
}
fn json_response(status: StatusCode, body: Vec<u8>) -> S3Response<(StatusCode, Body)> {
let mut headers = HeaderMap::new();
headers.insert(CONTENT_TYPE, HeaderValue::from_static("application/json"));
@@ -1322,6 +1358,50 @@ impl Operation for HealHandler {
}
};
let hip = extract_heal_init_params(&bytes, &req.uri, params)?;
// The heal channel currently models bucket/object work. Root heal reuses the
// existing format-heal path directly so `/v3/heal/` is accepted intentionally.
if should_handle_root_heal_directly(&hip) {
let Some(store) = object_store_from_extensions(&req.extensions) else {
warn!(
event = EVENT_ADMIN_REQUEST_FAILED,
component = LOG_COMPONENT_ADMIN_API,
subsystem = LOG_SUBSYSTEM_HEAL_ADMIN,
operation = "root_heal",
result = "failed",
reason = "server_not_initialized",
"admin request failed"
);
return Err(s3_error!(InternalError, "server not initialized"));
};
let (_, heal_err) = store.heal_format(hip.hs.dry_run).await.map_err(|e| {
warn!(
event = EVENT_ADMIN_REQUEST_FAILED,
component = LOG_COMPONENT_ADMIN_API,
subsystem = LOG_SUBSYSTEM_HEAL_ADMIN,
operation = "root_heal",
result = "failed",
reason = "heal_format_failed",
error = %e,
"admin request failed"
);
s3_error!(InternalError, "root heal failed: {e}")
})?;
map_root_heal_status(heal_err)?;
let body = encode_heal_start_success("root-heal".to_string(), client_address)?;
info!(
event = EVENT_ADMIN_RESPONSE_EMITTED,
component = LOG_COMPONENT_ADMIN_API,
subsystem = LOG_SUBSYSTEM_HEAL_ADMIN,
operation = "root_heal",
result = "success",
state = "started",
"admin response emitted"
);
return Ok(json_response(StatusCode::OK, body));
}
validate_heal_request_mode(&hip)?;
let response_operation = if hip.force_stop {
"cancel_heal"
@@ -1534,9 +1614,11 @@ mod tests {
build_replacement_recovery_status_response, encode_background_heal_status, encode_heal_control_path,
encode_heal_start_success, encode_heal_task_status, execute_after_heal_control_capability, heal_channel_response_items,
heal_channel_response_progress, heal_channel_response_summary, heal_control_response_id, json_response,
map_heal_response, merge_peer_heal_statuses, peer_topology_complete, query_peer_heal_status,
query_peer_replacement_recovery_status, reject_heal_admission, validate_heal_request_mode, validate_heal_target,
map_heal_response, map_root_heal_status, merge_peer_heal_statuses, peer_topology_complete, query_peer_heal_status,
query_peer_replacement_recovery_status, reject_heal_admission, should_handle_root_heal_directly,
validate_heal_request_mode, validate_heal_target,
};
use crate::admin::storage_api::error::StorageError;
use crate::storage::rpc::node_service::heal::{
NodeHealProgress, NodeHealStatusSnapshot, NodeReplacementRecoveryStatusSnapshot, encode_node_replacement_recovery_status,
};
@@ -2004,63 +2086,48 @@ mod tests {
}
#[test]
fn test_root_heal_shapes_route_through_cluster_coordination() {
// Root heal has no direct local store path: every start shape is either
// rejected by validate_heal_request_mode or submitted to the cluster
// heal channel as an Admin-sourced request (see HealHandler::call).
let accepted_root_starts = [
HealInitParams {
hs: HealOpts {
recursive: true,
..Default::default()
},
..Default::default()
},
HealInitParams {
force_start: true,
hs: HealOpts {
recursive: true,
..Default::default()
},
..Default::default()
},
HealInitParams {
hs: HealOpts {
pool: Some(1),
set: Some(2),
..Default::default()
},
..Default::default()
},
];
for hip in accepted_root_starts {
validate_heal_request_mode(&hip).expect("accepted root heal start must reach cluster coordination");
let request = build_heal_channel_request(&hip);
assert_eq!(request.bucket, "", "root heal must stay cluster-scoped");
assert_eq!(request.source, HealRequestSource::Admin);
assert!(!request.id.is_empty(), "cluster heal requests carry a dedup id");
}
fn test_should_handle_root_heal_directly_is_disabled_for_root_start_modes() {
assert!(!should_handle_root_heal_directly(&HealInitParams::default()));
assert!(!should_handle_root_heal_directly(&HealInitParams {
force_start: true,
..Default::default()
}));
}
// Shapes that cannot start a tracked heal (plain start, bare force_start
// without recursive, bare pool) are rejected instead of falling back to
// a direct local path.
for hip in [
HealInitParams::default(),
HealInitParams {
force_start: true,
#[test]
fn test_should_handle_root_heal_directly_skips_query_cancel_and_bucket_targets() {
assert!(!should_handle_root_heal_directly(&HealInitParams {
client_token: "heal-token".to_string(),
..Default::default()
}));
assert!(!should_handle_root_heal_directly(&HealInitParams {
force_stop: true,
..Default::default()
}));
assert!(!should_handle_root_heal_directly(&HealInitParams {
bucket: "bucket".to_string(),
..Default::default()
}));
assert!(!should_handle_root_heal_directly(&HealInitParams {
hs: HealOpts {
pool: Some(1),
set: Some(2),
..Default::default()
},
HealInitParams {
hs: HealOpts {
pool: Some(1),
..Default::default()
},
..Default::default()
},
] {
let err = validate_heal_request_mode(&hip).expect_err("unscoped root heal start must be rejected");
assert_eq!(err.code(), &S3ErrorCode::InvalidRequest);
}
..Default::default()
}));
}
#[test]
fn test_map_root_heal_status_allows_no_heal_required() {
map_root_heal_status(Some(StorageError::NoHealRequired)).expect("NoHealRequired should stay non-fatal");
}
#[test]
fn test_map_root_heal_status_rejects_fatal_errors() {
let err = map_root_heal_status(Some(StorageError::Unexpected)).expect_err("fatal status must fail");
assert_eq!(err.code(), &S3ErrorCode::InternalError);
assert!(err.to_string().contains("root heal failed: Unexpected error"));
}
#[test]
+18 -7
View File
@@ -18,10 +18,12 @@
//! keeping the response format explicitly NDJSON. It is not a Prometheus text
//! exposition endpoint.
use crate::admin::auth::authorize_admin_request;
use crate::admin::auth::validate_admin_request;
use crate::admin::router::Operation;
use crate::admin::storage_api::access::spawn_traced;
use crate::admin::storage_api::metrics::{CollectMetricsOpts, MetricType, collect_local_metrics};
use crate::auth::{check_key_valid, get_session_token};
use crate::server::RemoteAddr;
use bytes::Bytes;
use futures::{Stream, StreamExt};
use http::{HeaderMap, HeaderValue, Uri};
@@ -180,15 +182,24 @@ impl ByteStream for MetricsStream {}
pub struct MetricsHandler {}
/// The pre-check keeps this endpoint's historical `AccessDenied` missing-credentials
/// response; the shared gate reports `InvalidRequest` "get cred failed".
async fn authorize_metrics_request(req: &S3Request<Body>) -> S3Result<()> {
if req.credentials.is_none() {
let Some(input_cred) = req.credentials.as_ref() else {
return Err(s3_error!(AccessDenied, "Signature is required"));
}
};
authorize_admin_request(req, vec![Action::AdminAction(AdminAction::GetMetricsAction)]).await?;
Ok(())
let (cred, owner) =
check_key_valid(get_session_token(&req.uri, &req.headers).unwrap_or_default(), &input_cred.access_key).await?;
let remote_addr = req.extensions.get::<Option<RemoteAddr>>().and_then(|opt| opt.map(|a| a.0));
validate_admin_request(
&req.headers,
&cred,
owner,
false,
vec![Action::AdminAction(AdminAction::GetMetricsAction)],
remote_addr,
)
.await
}
#[async_trait::async_trait]
+17 -32
View File
@@ -17,13 +17,14 @@ use crate::admin::service::config::{
preflight_dynamic_config_reload_for_context, signal_dynamic_config_reload_checked_for_context,
};
use crate::admin::{
auth::authorize_admin_request,
auth::validate_admin_request,
handlers::supervise_admin_mutation,
router::{AdminOperation, Operation, S3Router},
};
use crate::auth::{check_key_valid, get_session_token};
use crate::server::{
ADMIN_PREFIX, MODULE_SWITCHES_SIGNAL_SUBSYSTEM, ModuleSwitchSnapshot, ModuleSwitchSource, PersistedModuleSwitches,
apply_audit_module_switch_for_context, current_module_switch_snapshot, mark_event_notifier_reconciled,
RemoteAddr, apply_audit_module_switch_for_context, current_module_switch_snapshot, mark_event_notifier_reconciled,
mark_event_notifier_unreconciled, refresh_audit_module_enabled, refresh_notify_module_enabled,
refresh_persisted_module_switches_from, refresh_persisted_module_switches_from_store, save_persisted_module_switches_to,
validate_module_switch_update,
@@ -113,15 +114,23 @@ fn build_response<T: Serialize>(
Ok(S3Response::with_headers((status, Body::from(data)), header))
}
/// The pre-check keeps these endpoints' historical missing-credentials message;
/// the shared gate reports "get cred failed".
async fn authorize_module_switch_request(req: &S3Request<Body>, action: AdminAction) -> S3Result<()> {
if req.credentials.is_none() {
let Some(input_cred) = &req.credentials else {
return Err(s3_error!(InvalidRequest, "authentication required"));
}
};
authorize_admin_request(req, vec![Action::AdminAction(action)]).await?;
Ok(())
let (cred, owner) =
check_key_valid(get_session_token(&req.uri, &req.headers).unwrap_or_default(), &input_cred.access_key).await?;
validate_admin_request(
&req.headers,
&cred,
owner,
false,
vec![Action::AdminAction(action)],
req.extensions.get::<Option<RemoteAddr>>().and_then(|opt| opt.map(|a| a.0)),
)
.await
}
async fn refresh_module_switch_snapshot() -> S3Result<ModuleSwitchSnapshot> {
@@ -260,30 +269,6 @@ impl Operation for UpdateModuleSwitchesHandler {
mod tests {
use super::{ModuleSwitchDiscovery, ModuleSwitchSource, ModuleSwitchesResponse};
/// These endpoints authorize through the shared admin gate, which reports
/// "get cred failed" for a credential-less request. The pre-check keeps the
/// message they have always returned (rustfs/backlog#1829).
#[tokio::test]
async fn module_switch_gate_keeps_its_missing_credentials_message() {
let req = s3s::S3Request {
input: s3s::Body::from(String::new()),
method: http::Method::GET,
uri: http::Uri::from_static("/rustfs/admin/v3/module-switches"),
headers: http::HeaderMap::new(),
extensions: http::Extensions::new(),
credentials: None,
region: None,
service: None,
trailing_headers: None,
};
let err = super::authorize_module_switch_request(&req, rustfs_policy::policy::action::AdminAction::ServerInfoAdminAction)
.await
.expect_err("a request without credentials must be rejected");
assert_eq!(err.code(), &s3s::S3ErrorCode::InvalidRequest);
assert_eq!(err.message(), Some("authentication required"));
}
#[test]
fn module_switch_handlers_require_admin_authorization_contract() {
let src = include_str!("module_switch.rs");
+12 -32
View File
@@ -21,11 +21,12 @@
//! that bucket, and with `bucket`+`object` it flushes that one identity — the
//! only remediation for a poisoned entry short of a node restart.
use crate::admin::auth::authorize_admin_request;
use crate::admin::auth::validate_admin_request;
use crate::admin::router::{AdminOperation, Operation, S3Router};
use crate::admin::runtime_sources::current_object_data_cache;
use crate::app::object_data_cache::ObjectDataCacheAdapter;
use crate::server::ADMIN_PREFIX;
use crate::auth::{check_key_valid, get_session_token};
use crate::server::{ADMIN_PREFIX, RemoteAddr};
use http::{HeaderMap, HeaderValue};
use hyper::{Method, StatusCode};
use matchit::Params;
@@ -75,14 +76,17 @@ pub fn register_object_data_cache_route(r: &mut S3Router<AdminOperation>) -> std
Ok(())
}
/// The pre-check keeps these endpoints' historical missing-credentials message;
/// the shared gate reports "get cred failed".
async fn authorize(req: &S3Request<Body>, action: AdminAction) -> S3Result<()> {
if req.credentials.is_none() {
let Some(input_cred) = req.credentials.as_ref() else {
return Err(s3_error!(InvalidRequest, "missing credentials"));
}
authorize_admin_request(req, vec![Action::AdminAction(action)]).await?;
Ok(())
};
let (cred, owner) =
check_key_valid(get_session_token(&req.uri, &req.headers).unwrap_or_default(), &input_cred.access_key).await?;
let remote_addr = req
.extensions
.get::<Option<RemoteAddr>>()
.and_then(|opt| opt.map(|addr| addr.0));
validate_admin_request(&req.headers, &cred, owner, false, vec![Action::AdminAction(action)], remote_addr).await
}
fn json_response<T: Serialize>(body: &T) -> S3Result<S3Response<(StatusCode, Body)>> {
@@ -204,30 +208,6 @@ mod tests {
assert_eq!(invalidation_outcome(&ObjectDataCacheInvalidationResult::NoOp), ("noop", 0));
}
/// These endpoints authorize through the shared admin gate, which reports
/// "get cred failed" for a credential-less request. The pre-check keeps the
/// message they have always returned (rustfs/backlog#1829).
#[tokio::test]
async fn authorize_keeps_its_missing_credentials_message() {
let req = S3Request {
input: Body::from(String::new()),
method: Method::GET,
uri: "/rustfs/admin/v3/object-data-cache/stats".parse().expect("uri should parse"),
headers: HeaderMap::new(),
extensions: http::Extensions::new(),
credentials: None,
region: None,
service: None,
trailing_headers: None,
};
let err = authorize(&req, AdminAction::ServerInfoAdminAction)
.await
.expect_err("a request without credentials must be rejected");
assert_eq!(err.code(), &S3ErrorCode::InvalidRequest);
assert_eq!(err.message(), Some("missing credentials"));
}
#[test]
fn stats_handler_requires_server_info_action() {
// Guard the auth contract: the stats endpoint is a read, the flush
+17 -32
View File
@@ -13,7 +13,7 @@
// limitations under the License.
use crate::admin::{
auth::authorize_admin_request,
auth::validate_admin_request,
plugin_contract::{
PluginCatalogAdminDiscovery, PluginCatalogDomainEntry, PluginCatalogEntry, PluginCatalogResponse, PluginContractDomain,
PluginContractEntrypointKind, PluginContractPackaging, PluginDistributionContract, PluginRuntimeContract,
@@ -21,7 +21,8 @@ use crate::admin::{
router::{AdminOperation, Operation, S3Router},
runtime_sources::default_admin_usecase,
};
use crate::server::ADMIN_PREFIX;
use crate::auth::{check_key_valid, get_session_token};
use crate::server::{ADMIN_PREFIX, RemoteAddr};
use http::{HeaderMap, HeaderValue, StatusCode};
use hyper::Method;
use matchit::Params;
@@ -113,15 +114,23 @@ fn merge_catalog_descriptor(plugins: &mut HashMap<&'static str, PluginCatalogEnt
}
}
/// The pre-check keeps this endpoint's historical missing-credentials message;
/// the shared gate reports "get cred failed".
async fn authorize_plugin_catalog_request(req: &S3Request<Body>) -> S3Result<()> {
if req.credentials.is_none() {
let Some(input_cred) = &req.credentials else {
return Err(s3_error!(InvalidRequest, "authentication required"));
}
};
authorize_admin_request(req, vec![Action::AdminAction(AdminAction::ServerInfoAdminAction)]).await?;
Ok(())
let (cred, owner) =
check_key_valid(get_session_token(&req.uri, &req.headers).unwrap_or_default(), &input_cred.access_key).await?;
validate_admin_request(
&req.headers,
&cred,
owner,
false,
vec![Action::AdminAction(AdminAction::ServerInfoAdminAction)],
req.extensions.get::<Option<RemoteAddr>>().and_then(|opt| opt.map(|a| a.0)),
)
.await
}
fn build_json_response(
@@ -166,30 +175,6 @@ mod tests {
);
}
/// This endpoint authorizes through the shared admin gate, which reports
/// "get cred failed" for a credential-less request. The pre-check keeps the
/// message it has always returned (rustfs/backlog#1829).
#[tokio::test]
async fn plugin_catalog_gate_keeps_its_missing_credentials_message() {
let req = s3s::S3Request {
input: s3s::Body::from(String::new()),
method: http::Method::GET,
uri: http::Uri::from_static("/rustfs/admin/v4/plugins/catalog"),
headers: http::HeaderMap::new(),
extensions: http::Extensions::new(),
credentials: None,
region: None,
service: None,
trailing_headers: None,
};
let err = super::authorize_plugin_catalog_request(&req)
.await
.expect_err("a request without credentials must be rejected");
assert_eq!(err.code(), &s3s::S3ErrorCode::InvalidRequest);
assert_eq!(err.message(), Some("authentication required"));
}
#[test]
fn plugin_catalog_contains_representative_builtin_targets() {
let response = build_catalog_response();
+32 -45
View File
@@ -13,7 +13,7 @@
// limitations under the License.
use crate::admin::{
auth::authorize_admin_request,
auth::validate_admin_request,
handlers::audit_runtime_config::{load_server_config_from_store, remove_audit_target_config, set_audit_target_config},
handlers::notify_runtime_access::{
load_notification_config_snapshot, remove_notification_target_config, set_notification_target_config,
@@ -29,9 +29,10 @@ use crate::admin::{
},
router::{AdminOperation, Operation, S3Router},
};
use crate::auth::{check_key_valid, get_session_token};
use crate::server::{
ADMIN_PREFIX, is_audit_module_enabled, is_notify_module_enabled, refresh_audit_module_enabled, refresh_notify_module_enabled,
refresh_persisted_module_switches_from_store,
ADMIN_PREFIX, RemoteAddr, is_audit_module_enabled, is_notify_module_enabled, refresh_audit_module_enabled,
refresh_notify_module_enabled, refresh_persisted_module_switches_from_store,
};
use hyper::{Method, StatusCode};
use matchit::Params;
@@ -562,26 +563,42 @@ fn plugin_instance_matches_query(instance: &PluginInstanceEntry, query: &str) ->
.any(|field| field.to_ascii_lowercase().contains(&query))
}
/// The pre-check keeps this endpoint's historical missing-credentials message;
/// the shared gate reports "get cred failed".
async fn authorize_plugin_instance_request(req: &S3Request<Body>) -> S3Result<()> {
if req.credentials.is_none() {
let Some(input_cred) = &req.credentials else {
return Err(s3_error!(InvalidRequest, "authentication required"));
}
};
authorize_admin_request(req, vec![Action::AdminAction(AdminAction::GetBucketTargetAction)]).await?;
Ok(())
let (cred, owner) =
check_key_valid(get_session_token(&req.uri, &req.headers).unwrap_or_default(), &input_cred.access_key).await?;
validate_admin_request(
&req.headers,
&cred,
owner,
false,
vec![Action::AdminAction(AdminAction::GetBucketTargetAction)],
req.extensions.get::<Option<RemoteAddr>>().and_then(|opt| opt.map(|a| a.0)),
)
.await
}
/// The pre-check keeps this endpoint's historical missing-credentials message;
/// the shared gate reports "get cred failed".
async fn authorize_plugin_instance_write_request(req: &S3Request<Body>) -> S3Result<()> {
if req.credentials.is_none() {
let Some(input_cred) = &req.credentials else {
return Err(s3_error!(InvalidRequest, "authentication required"));
}
};
authorize_admin_request(req, vec![Action::AdminAction(AdminAction::SetBucketTargetAction)]).await?;
Ok(())
let (cred, owner) =
check_key_valid(get_session_token(&req.uri, &req.headers).unwrap_or_default(), &input_cred.access_key).await?;
validate_admin_request(
&req.headers,
&cred,
owner,
false,
vec![Action::AdminAction(AdminAction::SetBucketTargetAction)],
req.extensions.get::<Option<RemoteAddr>>().and_then(|opt| opt.map(|a| a.0)),
)
.await
}
fn plugin_instance_mutation_block_reason(
@@ -925,36 +942,6 @@ mod tests {
);
}
/// Both instance gates authorize through the shared admin gate, which reports
/// "get cred failed" for a credential-less request. The pre-check keeps the
/// message these endpoints have always returned (rustfs/backlog#1829).
#[tokio::test]
async fn plugin_instance_gates_keep_their_missing_credentials_message() {
let credential_less_request = || S3Request {
input: Body::from(String::new()),
method: Method::GET,
uri: Uri::from_static("/rustfs/admin/v4/plugins/instances"),
headers: HeaderMap::new(),
extensions: Extensions::new(),
credentials: None,
region: None,
service: None,
trailing_headers: None,
};
for err in [
super::authorize_plugin_instance_request(&credential_less_request())
.await
.expect_err("a request without credentials must be rejected"),
super::authorize_plugin_instance_write_request(&credential_less_request())
.await
.expect_err("a request without credentials must be rejected"),
] {
assert_eq!(err.code(), &s3s::S3ErrorCode::InvalidRequest);
assert_eq!(err.message(), Some("authentication required"));
}
}
#[test]
fn configured_instance_without_runtime_appears_offline() {
let config = Config(HashMap::from([(
+18 -7
View File
@@ -12,22 +12,33 @@
// See the License for the specific language governing permissions and
// limitations under the License.
use crate::admin::{auth::authorize_admin_request, router::Operation};
use crate::admin::{auth::validate_admin_request, router::Operation};
use crate::auth::{check_key_valid, get_session_token};
use crate::server::RemoteAddr;
use http::StatusCode;
use matchit::Params;
use rustfs_policy::policy::action::{Action, AdminAction};
use s3s::{Body, S3Request, S3Response, S3Result, s3_error};
use tracing::info;
/// The pre-check keeps these endpoints' historical `AccessDenied` missing-credentials
/// response; the shared gate reports `InvalidRequest` "get cred failed".
pub(super) async fn authorize_profile_request(req: &S3Request<Body>) -> S3Result<()> {
if req.credentials.is_none() {
let Some(input_cred) = req.credentials.as_ref() else {
return Err(s3_error!(AccessDenied, "Signature is required"));
}
};
authorize_admin_request(req, vec![Action::AdminAction(AdminAction::ProfilingAdminAction)]).await?;
Ok(())
let (cred, owner) =
check_key_valid(get_session_token(&req.uri, &req.headers).unwrap_or_default(), &input_cred.access_key).await?;
let remote_addr = req.extensions.get::<Option<RemoteAddr>>().and_then(|opt| opt.map(|a| a.0));
validate_admin_request(
&req.headers,
&cred,
owner,
false,
vec![Action::AdminAction(AdminAction::ProfilingAdminAction)],
remote_addr,
)
.await
}
pub(super) fn profile_not_implemented_response(message: String) -> S3Response<(StatusCode, Body)> {
+10 -26
View File
@@ -13,10 +13,11 @@
// limitations under the License.
use super::profile::{authorize_profile_request, profile_not_implemented_response};
use crate::admin::auth::authorize_admin_request;
use crate::admin::auth::validate_admin_request;
use crate::admin::router::{AdminOperation, Operation, S3Router};
use crate::admin::storage_api::access::spawn_traced;
use crate::server::ADMIN_PREFIX;
use crate::auth::{check_key_valid, get_session_token};
use crate::server::{ADMIN_PREFIX, RemoteAddr};
use bytes::Bytes;
use futures::{Stream, StreamExt};
use http::{HeaderMap, HeaderValue};
@@ -88,14 +89,14 @@ pub fn register_profiling_route(r: &mut S3Router<AdminOperation>) -> std::io::Re
}
/// Authorize a request against a single admin action (profiling or trace).
/// The pre-check keeps these endpoints' historical `AccessDenied` missing-credentials
/// response; the shared gate reports `InvalidRequest` "get cred failed".
async fn authorize_action(req: &S3Request<Body>, action: AdminAction) -> S3Result<()> {
if req.credentials.is_none() {
let Some(input_cred) = req.credentials.as_ref() else {
return Err(s3_error!(AccessDenied, "Signature is required"));
}
authorize_admin_request(req, vec![Action::AdminAction(action)]).await?;
Ok(())
};
let (cred, owner) =
check_key_valid(get_session_token(&req.uri, &req.headers).unwrap_or_default(), &input_cred.access_key).await?;
let remote_addr = req.extensions.get::<Option<RemoteAddr>>().and_then(|opt| opt.map(|a| a.0));
validate_admin_request(&req.headers, &cred, owner, false, vec![Action::AdminAction(action)], remote_addr).await
}
pub struct ProfileHandler {}
@@ -529,7 +530,7 @@ fn trace_value_string(value: &TraceVal) -> String {
mod tests {
use super::{
ProfileControlHandler, ProfileHandler, ProfileStatusHandler, ProfilingDownloadHandler, ProfilingStartHandler,
TraceHandler, TraceKindFilter, TraceStreamFilter, TraceWireRecord, authorize_action,
TraceHandler, TraceKindFilter, TraceStreamFilter, TraceWireRecord,
};
use crate::admin::router::Operation;
use http::{Extensions, HeaderMap, Uri};
@@ -538,7 +539,6 @@ mod tests {
use rustfs_common::trace_bus::{TraceEvent, TraceFunc, TraceKind};
use rustfs_madmin::service_commands::ServiceTraceOpts;
use rustfs_madmin::trace::TraceType;
use rustfs_policy::policy::action::AdminAction;
use s3s::{Body, S3ErrorCode, S3Request, S3Result};
use std::time::{Duration, UNIX_EPOCH};
@@ -563,22 +563,6 @@ mod tests {
TraceStreamFilter::from_request(&uri, &opts)
}
/// The profiling/trace endpoints authorize through the shared admin gate, which
/// rejects a credential-less request with `InvalidRequest` "get cred failed". The
/// pre-check keeps the `AccessDenied` response they have always returned
/// (rustfs/backlog#1829).
#[tokio::test]
async fn profile_admin_gate_keeps_its_missing_credentials_response() {
let err = authorize_action(
&build_profile_request("/rustfs/admin/v3/profiling/start"),
AdminAction::ProfilingAdminAction,
)
.await
.expect_err("a request without credentials must be rejected");
assert_eq!(err.code(), &S3ErrorCode::AccessDenied);
assert_eq!(err.message(), Some("Signature is required"));
}
#[tokio::test]
async fn profile_handler_rejects_missing_credentials() {
let result = ProfileHandler {}
+23 -31
View File
@@ -12,11 +12,12 @@
// See the License for the specific language governing permissions and
// limitations under the License.
use crate::admin::auth::authorize_admin_request;
use crate::admin::auth::validate_admin_request;
use crate::admin::router::{AdminOperation, Operation, S3Router};
use crate::admin::runtime_sources::current_scanner_metrics_report;
use crate::auth::{check_key_valid, get_session_token};
use crate::module_switches::{ENV_SCANNER_ENABLED, scanner_enabled_from_env};
use crate::server::ADMIN_PREFIX;
use crate::server::{ADMIN_PREFIX, RemoteAddr};
use chrono::Utc;
use http::{HeaderMap, HeaderValue};
use hyper::{Method, StatusCode};
@@ -153,14 +154,29 @@ pub fn register_scanner_route(r: &mut S3Router<AdminOperation>) -> std::io::Resu
Ok(())
}
/// The pre-check keeps these endpoints' historical missing-credentials message;
/// the shared gate reports "get cred failed".
async fn validate_scanner_status_request(req: &S3Request<Body>) -> S3Result<Credentials> {
if req.credentials.is_none() {
let Some(input_cred) = req.credentials.as_ref() else {
return Err(s3_error!(InvalidRequest, "missing credentials"));
}
};
authorize_admin_request(req, vec![Action::AdminAction(AdminAction::ServerInfoAdminAction)]).await
let (cred, owner) =
check_key_valid(get_session_token(&req.uri, &req.headers).unwrap_or_default(), &input_cred.access_key).await?;
let remote_addr = req
.extensions
.get::<Option<RemoteAddr>>()
.and_then(|opt| opt.map(|addr| addr.0));
validate_admin_request(
&req.headers,
&cred,
owner,
false,
vec![Action::AdminAction(AdminAction::ServerInfoAdminAction)],
remote_addr,
)
.await?;
Ok(cred)
}
fn json_response(body: Vec<u8>) -> S3Result<S3Response<(StatusCode, Body)>> {
@@ -213,30 +229,6 @@ impl Operation for IlmExpiryStatusHandler {
mod tests {
use super::*;
/// These endpoints authorize through the shared admin gate, which reports
/// "get cred failed" for a credential-less request. The pre-check keeps the
/// message they have always returned (rustfs/backlog#1829).
#[tokio::test]
async fn scanner_status_gate_keeps_its_missing_credentials_message() {
let req = S3Request {
input: Body::from(String::new()),
method: Method::GET,
uri: http::Uri::from_static("/rustfs/admin/v3/scanner/status"),
headers: HeaderMap::new(),
extensions: http::Extensions::new(),
credentials: None,
region: None,
service: None,
trailing_headers: None,
};
let err = validate_scanner_status_request(&req)
.await
.expect_err("a request without credentials must be rejected");
assert_eq!(err.code(), &S3ErrorCode::InvalidRequest);
assert_eq!(err.message(), Some("missing credentials"));
}
#[test]
fn scanner_disabled_reason_reports_startup_env_key() {
assert_eq!(scanner_disabled_reason(true), None);
+13 -34
View File
@@ -19,10 +19,11 @@
//! usage caches, with a one-level sub-prefix breakdown — the data console
//! buckets view MinIO serves from `loadPrefixUsageFromBackend`.
use crate::admin::auth::authorize_admin_request;
use crate::admin::auth::validate_admin_request;
use crate::admin::handlers::system::data_usage_info_gate_actions;
use crate::admin::router::{AdminOperation, Operation, S3Router};
use crate::server::ADMIN_PREFIX;
use crate::auth::{check_key_valid, get_session_token};
use crate::server::{ADMIN_PREFIX, RemoteAddr};
use http::{HeaderMap, HeaderValue, StatusCode};
use hyper::Method;
use matchit::Params;
@@ -69,10 +70,15 @@ fn parse_usage_prefix_query(query: Option<&str>) -> S3Result<(String, usize)> {
#[async_trait::async_trait]
impl Operation for BucketPrefixUsageHandler {
async fn call(&self, req: S3Request<Body>, params: Params<'_, '_>) -> S3Result<S3Response<(StatusCode, Body)>> {
// The shared gate reports the same `InvalidRequest` "get cred failed" this
// handler has always returned for a credential-less request, so it needs no
// message-preserving pre-check.
authorize_admin_request(&req, data_usage_info_gate_actions()).await?;
let Some(input_cred) = req.credentials else {
return Err(s3_error!(InvalidRequest, "get cred failed"));
};
let (cred, owner) =
check_key_valid(get_session_token(&req.uri, &req.headers).unwrap_or_default(), &input_cred.access_key).await?;
let remote_addr = req.extensions.get::<Option<RemoteAddr>>().and_then(|opt| opt.map(|a| a.0));
validate_admin_request(&req.headers, &cred, owner, false, data_usage_info_gate_actions(), remote_addr).await?;
let bucket = params.get("bucket").unwrap_or_default().to_string();
if bucket.is_empty() {
@@ -98,40 +104,13 @@ impl Operation for BucketPrefixUsageHandler {
#[cfg(test)]
mod tests {
use super::{BucketPrefixUsageHandler, DEFAULT_MAX_ENTRIES, MAX_ENTRIES_LIMIT, parse_usage_prefix_query};
use crate::admin::router::Operation;
use super::{DEFAULT_MAX_ENTRIES, MAX_ENTRIES_LIMIT, parse_usage_prefix_query};
use s3s::S3Error;
fn query(raw: &str) -> Result<(String, usize), S3Error> {
parse_usage_prefix_query(Some(raw))
}
/// This endpoint authorizes through the shared admin gate, whose
/// credential-less rejection is the same `InvalidRequest` "get cred failed"
/// the handler returned inline before (rustfs/backlog#1829), so no
/// message-preserving pre-check is needed here.
#[tokio::test]
async fn prefix_usage_handler_keeps_its_missing_credentials_message() {
let req = s3s::S3Request {
input: s3s::Body::from(String::new()),
method: http::Method::GET,
uri: http::Uri::from_static("/rustfs/admin/v3/usage/bucket"),
headers: http::HeaderMap::new(),
extensions: http::Extensions::new(),
credentials: None,
region: None,
service: None,
trailing_headers: None,
};
let err = BucketPrefixUsageHandler {}
.call(req, matchit::Params::new())
.await
.expect_err("a request without credentials must be rejected");
assert_eq!(err.code(), &s3s::S3ErrorCode::InvalidRequest);
assert_eq!(err.message(), Some("get cred failed"));
}
#[test]
fn defaults_apply_when_no_query_is_given() {
assert_eq!(parse_usage_prefix_query(None).unwrap(), (String::new(), DEFAULT_MAX_ENTRIES));
+4
View File
@@ -909,6 +909,10 @@ pub(crate) mod contract {
};
}
pub(crate) mod heal {
pub(crate) use super::super::storage_contracts::HealOperations;
}
pub(crate) mod list {
pub(crate) use super::super::storage_contracts::ListOperations;
}
+24 -5
View File
@@ -97,7 +97,7 @@ use super::storage_api::object_usecase::set_disk::{
};
use super::storage_api::object_usecase::sse::{
DecryptionRequest, EncryptionRequest, SSEType, SseKmsPrincipal, apply_bucket_default_lock_retention,
authorize_sse_kms_object_read, bucket_default_write_sse, build_ssec_read_headers, encryption_material_to_metadata,
authorize_sse_kms_object_read, build_ssec_read_headers, encryption_material_to_metadata,
extract_server_side_encryption_from_headers, extract_ssec_params_from_headers, extract_ssekms_context_from_headers,
get_buffer_size_opt_in, load_bucket_object_lock_config_state, map_get_object_reader_error, sse_decryption, sse_encryption,
validate_bucket_object_lock_enabled_state,
@@ -169,8 +169,8 @@ use s3s::dto::{
ObjectLockLegalHoldStatus, ObjectLockMode, ObjectLockRetention, ObjectLockRetentionMode, ObjectPart, PutObjectInput,
PutObjectOutput, Range, RequestCharged, RestoreObjectInput, RestoreObjectOutput, RestoreStatus, SSECustomerAlgorithm,
SSECustomerKeyMD5, SSEKMSKeyId, SelectObjectContentInput, SelectObjectContentOutput, ServerSideEncryption,
ServerSideEncryptionConfiguration, StorageClass, StreamingBlob, TaggingDirective, TaggingHeader, Timestamp, TimestampFormat,
WebsiteRedirectLocation,
ServerSideEncryptionByDefault, ServerSideEncryptionConfiguration, StorageClass, StreamingBlob, TaggingDirective,
TaggingHeader, Timestamp, TimestampFormat, WebsiteRedirectLocation,
};
use s3s::header::{X_AMZ_RESTORE, X_AMZ_RESTORE_OUTPUT_PATH};
use s3s::stream::{ByteStream, DynByteStream, RemainingLength};
@@ -2676,6 +2676,25 @@ fn has_put_sse_request_headers(headers: &HeaderMap) -> bool {
|| headers.get(AMZ_SERVER_SIDE_ENCRYPTION_KMS_ID).is_some()
}
/// Managed SSE resolved from a bucket default encryption rule on the copy path.
///
/// Unknown algorithms fall back to AES256, the same total mapping as the PUT and
/// extract paths and the storage-layer resolver (`prepare_sse_configuration`), which
/// `sse_encryption` re-runs when it mints the destination DEK. Resolving `None` here
/// instead lets a same-name copy under a malformed bucket default pass the
/// `copy_changes_encryption` guard and take the metadata-only shortcut while the
/// storage layer still encrypts: fresh DEK metadata is committed beside the untouched
/// plaintext blocks and the object becomes unreadable. Reachable only via corrupt or
/// hand-edited bucket metadata — PutBucketEncryption rejects unknown algorithms
/// (backlog#1826).
fn bucket_default_write_sse(sse: &ServerSideEncryptionByDefault) -> ServerSideEncryption {
match sse.sse_algorithm.as_str() {
"AES256" => ServerSideEncryption::from_static(ServerSideEncryption::AES256),
"aws:kms" => ServerSideEncryption::from_static(ServerSideEncryption::AWS_KMS),
_ => ServerSideEncryption::from_static(ServerSideEncryption::AES256),
}
}
/// Resolve the effective server-side encryption for a write against the bucket's
/// default encryption configuration.
///
@@ -10096,8 +10115,8 @@ mod tests {
DefaultRetention, Delete, DeleteMarkerReplication, DeleteMarkerReplicationStatus, DeleteReplication,
DeleteReplicationStatus, Destination, ExistingObjectReplication, ExistingObjectReplicationStatus, ObjectIdentifier,
ObjectLockConfiguration, ObjectLockEnabled, ObjectLockRule, ReplicaModifications, ReplicaModificationsStatus,
ReplicationConfiguration, ReplicationRule, ReplicationRuleStatus, RestoreRequest, ServerSideEncryptionByDefault,
ServerSideEncryptionConfiguration, ServerSideEncryptionRule, SourceSelectionCriteria,
ReplicationConfiguration, ReplicationRule, ReplicationRuleStatus, RestoreRequest, ServerSideEncryptionConfiguration,
ServerSideEncryptionRule, SourceSelectionCriteria,
};
use std::pin::Pin;
use std::sync::Arc;
+2 -3
View File
@@ -1032,9 +1032,8 @@ pub(crate) mod sse {
validate_bucket_object_lock_enabled_state,
};
pub(crate) use crate::storage::storage_api::sse_consumer::{
EncryptionKeyKind, SSEType, bucket_default_write_sse, build_ssec_read_headers, encryption_material_to_metadata,
extract_ssec_params_from_headers, extract_ssekms_context_from_headers, map_get_object_reader_error,
mark_encrypted_multipart_metadata,
EncryptionKeyKind, SSEType, build_ssec_read_headers, encryption_material_to_metadata, extract_ssec_params_from_headers,
extract_ssekms_context_from_headers, map_get_object_reader_error, mark_encrypted_multipart_metadata,
};
}
+45 -1
View File
@@ -2206,7 +2206,7 @@ mod tests {
previous_scanner_activity_response, remove_heal_control_replay, scanner_activity_response, stop_rebalance_response,
};
use crate::storage::rpc::node_service::heal::heal_topology_fingerprint;
use crate::storage::storage_api::rpc_consumer::node_service::{DiskError, HealBucketInfo};
use crate::storage::storage_api::rpc_consumer::node_service::{DiskError, HealBucketInfo, HealEndpoint};
use crate::storage::storage_api::set_tonic_canonical_body_digest;
use crate::storage::storage_api::{
Endpoint,
@@ -2334,14 +2334,42 @@ mod tests {
Ok(None)
}
async fn get_object_data(&self, _bucket: &str, _object: &str) -> rustfs_heal::Result<Option<Vec<u8>>> {
Ok(None)
}
async fn put_object_data(&self, _bucket: &str, _object: &str, _data: &[u8]) -> rustfs_heal::Result<()> {
Ok(())
}
async fn delete_object(&self, _bucket: &str, _object: &str) -> rustfs_heal::Result<()> {
Ok(())
}
async fn verify_object_integrity(&self, _bucket: &str, _object: &str) -> rustfs_heal::Result<bool> {
Ok(true)
}
async fn ec_decode_rebuild(&self, _bucket: &str, _object: &str) -> rustfs_heal::Result<Vec<u8>> {
Ok(Vec::new())
}
async fn get_disk_status(&self, _endpoint: &HealEndpoint) -> rustfs_heal::Result<rustfs_heal::heal::storage::DiskStatus> {
Ok(rustfs_heal::heal::storage::DiskStatus::Ok)
}
async fn format_disk(&self, _endpoint: &HealEndpoint) -> rustfs_heal::Result<()> {
Ok(())
}
async fn get_bucket_info(&self, _bucket: &str) -> rustfs_heal::Result<Option<HealBucketInfo>> {
Ok(None)
}
async fn heal_bucket_metadata(&self, _bucket: &str) -> rustfs_heal::Result<()> {
Ok(())
}
async fn list_buckets(&self) -> rustfs_heal::Result<Vec<HealBucketInfo>> {
Ok(Vec::new())
}
@@ -2350,6 +2378,14 @@ mod tests {
Ok(false)
}
async fn get_object_size(&self, _bucket: &str, _object: &str) -> rustfs_heal::Result<Option<u64>> {
Ok(None)
}
async fn get_object_checksum(&self, _bucket: &str, _object: &str) -> rustfs_heal::Result<Option<String>> {
Ok(None)
}
async fn heal_object(
&self,
_bucket: &str,
@@ -2375,6 +2411,14 @@ mod tests {
Ok((rustfs_madmin::heal_commands::HealResultItem::default(), None))
}
async fn list_objects_for_heal(
&self,
_bucket: &str,
_prefix: &str,
) -> rustfs_heal::Result<Vec<rustfs_heal::heal::storage::HealListItem>> {
Ok(Vec::new())
}
async fn list_objects_for_heal_page(
&self,
_bucket: &str,
+27 -156
View File
@@ -23,15 +23,12 @@ use bytes::Bytes;
use rustfs_filemeta::FileInfo;
use rustfs_io_metrics::internode_metrics::{
INTERNODE_MSGPACK_CODEC_JSON, INTERNODE_MSGPACK_CODEC_MSGPACK, INTERNODE_MSGPACK_DIRECTION_REQUEST,
INTERNODE_OPERATION_GRPC_READ_ALL, INTERNODE_OPERATION_GRPC_READ_VERSION, INTERNODE_OPERATION_GRPC_WRITE_ALL,
INTERNODE_STAGE_READ_VERSION_DISK_READ, INTERNODE_STAGE_READ_VERSION_REQUEST_DECODE,
INTERNODE_STAGE_READ_VERSION_RESPONSE_JSON_ENCODE, INTERNODE_STAGE_READ_VERSION_RESPONSE_MSGPACK_ENCODE,
INTERNODE_TRANSPORT_BACKEND_GRPC, global_internode_metrics,
INTERNODE_OPERATION_GRPC_READ_ALL, INTERNODE_OPERATION_GRPC_WRITE_ALL, INTERNODE_TRANSPORT_BACKEND_GRPC,
global_internode_metrics,
};
use rustfs_protos::proto_gen::node_service::*;
use serde::de::DeserializeOwned;
use std::io::Cursor;
use std::time::Instant;
use tonic::{Request, Response, Status};
use tracing::debug;
@@ -204,21 +201,6 @@ fn encode_read_multiple_response_payloads(
Ok((read_multiple_resps_json, read_multiple_resps_bin))
}
fn internode_stage_timer(attribution_enabled: bool) -> Option<Instant> {
attribution_enabled.then(Instant::now)
}
fn record_read_version_stage(stage: &'static str, started_at: Option<Instant>) {
if let Some(started_at) = started_at {
global_internode_metrics().record_stage_duration_for_operation_and_backend(
INTERNODE_OPERATION_GRPC_READ_VERSION,
INTERNODE_TRANSPORT_BACKEND_GRPC,
stage,
started_at.elapsed(),
);
}
}
fn encode_batch_read_version_response_payloads(
batch_read_version_resps: &[BatchReadVersionResp],
request_decoded_from_msgpack: bool,
@@ -703,42 +685,11 @@ impl NodeService {
request: Request<ReadVersionRequest>,
) -> Result<Response<ReadVersionResponse>, Status> {
let request = request.into_inner();
let metrics = global_internode_metrics();
let read_version_attribution_enabled = rustfs_io_metrics::get_stage_metrics_enabled();
if read_version_attribution_enabled {
metrics.record_incoming_request_for_operation_and_backend(
INTERNODE_OPERATION_GRPC_READ_VERSION,
INTERNODE_TRANSPORT_BACKEND_GRPC,
);
metrics.record_recv_bytes_for_operation_and_backend(
INTERNODE_OPERATION_GRPC_READ_VERSION,
INTERNODE_TRANSPORT_BACKEND_GRPC,
request
.disk
.len()
.saturating_add(request.volume.len())
.saturating_add(request.path.len())
.saturating_add(request.version_id.len())
.saturating_add(request.opts.len())
.saturating_add(request.opts_bin.len()),
);
}
if let Some(disk) = self.find_disk(&request.disk).await {
let request_had_msgpack_payload = !request.opts_bin.is_empty();
let decode_started = internode_stage_timer(read_version_attribution_enabled);
let opts = match decode_msgpack_or_json::<ReadOptions>(&request.opts_bin, &request.opts, "ReadOptions") {
Ok(options) => {
record_read_version_stage(INTERNODE_STAGE_READ_VERSION_REQUEST_DECODE, decode_started);
options
}
Ok(options) => options,
Err(err) => {
record_read_version_stage(INTERNODE_STAGE_READ_VERSION_REQUEST_DECODE, decode_started);
if read_version_attribution_enabled {
metrics.record_error_for_operation_and_backend(
INTERNODE_OPERATION_GRPC_READ_VERSION,
INTERNODE_TRANSPORT_BACKEND_GRPC,
);
}
return Ok(Response::new(ReadVersionResponse {
success: false,
file_info: String::new(),
@@ -747,88 +698,42 @@ impl NodeService {
}));
}
};
let disk_read_started = internode_stage_timer(read_version_attribution_enabled);
match disk
.read_version("", &request.volume, &request.path, &request.version_id, &opts)
.await
{
Ok(file_info) => {
record_read_version_stage(INTERNODE_STAGE_READ_VERSION_DISK_READ, disk_read_started);
let json_encode_started = internode_stage_timer(read_version_attribution_enabled);
let file_info_json = compat_response_json(&file_info, request_had_msgpack_payload);
record_read_version_stage(INTERNODE_STAGE_READ_VERSION_RESPONSE_JSON_ENCODE, json_encode_started);
let msgpack_encode_started = internode_stage_timer(read_version_attribution_enabled);
let file_info_bin = encode_file_info_msgpack(&file_info);
record_read_version_stage(INTERNODE_STAGE_READ_VERSION_RESPONSE_MSGPACK_ENCODE, msgpack_encode_started);
match (file_info_json, file_info_bin) {
(Ok(file_info), Ok(file_info_bin)) => {
if read_version_attribution_enabled {
metrics.record_sent_bytes_for_operation_and_backend(
INTERNODE_OPERATION_GRPC_READ_VERSION,
INTERNODE_TRANSPORT_BACKEND_GRPC,
file_info.len().saturating_add(file_info_bin.len()),
);
}
Ok(Response::new(ReadVersionResponse {
success: true,
file_info,
file_info_bin: file_info_bin.into(),
error: None,
}))
}
(Err(err), _) => {
if read_version_attribution_enabled {
metrics.record_error_for_operation_and_backend(
INTERNODE_OPERATION_GRPC_READ_VERSION,
INTERNODE_TRANSPORT_BACKEND_GRPC,
);
}
Ok(Response::new(ReadVersionResponse {
success: false,
file_info: String::new(),
file_info_bin: Vec::new().into(),
error: Some(DiskError::other(format!("encode data failed: {err}")).into()),
}))
}
(_, Err(err)) => {
if read_version_attribution_enabled {
metrics.record_error_for_operation_and_backend(
INTERNODE_OPERATION_GRPC_READ_VERSION,
INTERNODE_TRANSPORT_BACKEND_GRPC,
);
}
Ok(Response::new(ReadVersionResponse {
success: false,
file_info: String::new(),
file_info_bin: Vec::new().into(),
error: Some(DiskError::other(format!("encode data failed: {err}")).into()),
}))
}
(Ok(file_info), Ok(file_info_bin)) => Ok(Response::new(ReadVersionResponse {
success: true,
file_info,
file_info_bin: file_info_bin.into(),
error: None,
})),
(Err(err), _) => Ok(Response::new(ReadVersionResponse {
success: false,
file_info: String::new(),
file_info_bin: Vec::new().into(),
error: Some(DiskError::other(format!("encode data failed: {err}")).into()),
})),
(_, Err(err)) => Ok(Response::new(ReadVersionResponse {
success: false,
file_info: String::new(),
file_info_bin: Vec::new().into(),
error: Some(DiskError::other(format!("encode data failed: {err}")).into()),
})),
}
}
Err(err) => {
record_read_version_stage(INTERNODE_STAGE_READ_VERSION_DISK_READ, disk_read_started);
if read_version_attribution_enabled {
metrics.record_error_for_operation_and_backend(
INTERNODE_OPERATION_GRPC_READ_VERSION,
INTERNODE_TRANSPORT_BACKEND_GRPC,
);
}
Ok(Response::new(ReadVersionResponse {
success: false,
file_info: String::new(),
file_info_bin: Vec::new().into(),
error: Some(err.into()),
}))
}
Err(err) => Ok(Response::new(ReadVersionResponse {
success: false,
file_info: String::new(),
file_info_bin: Vec::new().into(),
error: Some(err.into()),
})),
}
} else {
if read_version_attribution_enabled {
metrics.record_error_for_operation_and_backend(
INTERNODE_OPERATION_GRPC_READ_VERSION,
INTERNODE_TRANSPORT_BACKEND_GRPC,
);
}
Ok(Response::new(ReadVersionResponse {
success: false,
file_info: String::new(),
@@ -1615,16 +1520,12 @@ mod tests {
encode_batch_read_version_response_payloads, encode_file_info_msgpack, encode_msgpack, encode_msgpack_named,
encode_read_multiple_response_payloads, encode_rename_data_response_payloads,
};
use crate::storage::rpc::node_service::make_server;
use crate::storage::storage_api::ReadMultipleResp;
use crate::storage::storage_api::RenameDataResp;
use crate::storage::storage_api::rpc_consumer::node_service::BatchReadVersionResp;
use rustfs_filemeta::FileInfo;
use rustfs_io_metrics::internode_metrics::global_internode_metrics;
use rustfs_protos::proto_gen::node_service::ReadVersionRequest;
use serde::{Deserialize, Serialize};
use serial_test::serial;
use tonic::Request;
#[derive(Debug, PartialEq, Eq, Serialize, Deserialize)]
struct SamplePayload {
@@ -1632,36 +1533,6 @@ mod tests {
count: u32,
}
#[tokio::test]
#[serial]
async fn handle_read_version_records_attribution_for_missing_disk() {
let metrics = global_internode_metrics();
let previous_stage_metrics = rustfs_io_metrics::get_stage_metrics_enabled();
metrics.reset_for_test();
rustfs_io_metrics::set_get_stage_metrics_enabled(true);
let response = make_server()
.handle_read_version(Request::new(ReadVersionRequest {
disk: "missing-disk".to_string(),
volume: "bucket".to_string(),
path: "object".to_string(),
version_id: String::new(),
opts: String::new(),
opts_bin: Vec::new().into(),
}))
.await
.expect("ReadVersion handler should return a response")
.into_inner();
rustfs_io_metrics::set_get_stage_metrics_enabled(previous_stage_metrics);
let snapshot = metrics.snapshot();
assert!(!response.success);
assert_eq!(snapshot.incoming_requests_total, 1);
assert_eq!(snapshot.errors_total, 1);
assert!(snapshot.recv_bytes_total > 0);
metrics.reset_for_test();
}
#[test]
fn decode_msgpack_or_json_prefers_binary_payload() {
let payload = SamplePayload {
+6 -20
View File
@@ -164,7 +164,7 @@ use rustfs_utils::http::headers::{
AMZ_SERVER_SIDE_ENCRYPTION_CUSTOMER_KEY_MD5, AMZ_SERVER_SIDE_ENCRYPTION_KMS_CONTEXT,
};
use rustfs_utils::path::path_join_buf;
use s3s::dto::{SSECustomerAlgorithm, SSECustomerKey, SSECustomerKeyMD5, SSEKMSKeyId, ServerSideEncryptionByDefault};
use s3s::dto::{SSECustomerAlgorithm, SSECustomerKey, SSECustomerKeyMD5, SSEKMSKeyId};
use std::borrow::Cow;
// ============================================================================
@@ -203,24 +203,6 @@ pub struct SseConfiguration {
/// Effective KMS key ID (after considering bucket defaults)
pub effective_kms_key_id: Option<SSEKMSKeyId>,
}
/// Managed SSE resolved from a bucket default encryption rule on a write path.
///
/// The single mapping shared by every writer: this resolver, and the PUT, COPY
/// and extract paths in `app::object_usecase`, which reach it through
/// `resolve_bucket_default_sse`. Unknown algorithms fall back to AES256 rather
/// than to `None`. Resolving `None` instead lets a same-name copy under a
/// malformed bucket default pass the `copy_changes_encryption` guard and take
/// the metadata-only shortcut while this layer still encrypts: fresh DEK
/// metadata is committed beside the untouched plaintext blocks and the object
/// becomes unreadable. Reachable only via corrupt or hand-edited bucket
/// metadata — PutBucketEncryption rejects unknown algorithms (backlog#1826).
pub(crate) fn bucket_default_write_sse(sse: &ServerSideEncryptionByDefault) -> ServerSideEncryption {
match sse.sse_algorithm.as_str() {
"AES256" => ServerSideEncryption::from_static(ServerSideEncryption::AES256),
"aws:kms" => ServerSideEncryption::from_static(ServerSideEncryption::AWS_KMS),
_ => ServerSideEncryption::from_static(ServerSideEncryption::AES256),
}
}
/// Prepare SSE configuration by resolving request parameters with bucket defaults
///
@@ -284,7 +266,11 @@ async fn prepare_sse_configuration(
has_kms_key_id = sse.kms_master_key_id.is_some(),
"Bucket SSE default resolved"
);
bucket_default_write_sse(sse)
match sse.sse_algorithm.as_str() {
"AES256" => ServerSideEncryption::from_static(ServerSideEncryption::AES256),
"aws:kms" => ServerSideEncryption::from_static(ServerSideEncryption::AWS_KMS),
_ => ServerSideEncryption::from_static(ServerSideEncryption::AES256), // fallback
}
})
})
});
+5 -3
View File
@@ -252,6 +252,8 @@ pub(crate) mod rpc_consumer {
};
pub(crate) type StorageResult<T> = super::super::Result<T>;
#[cfg(test)]
pub(crate) type HealEndpoint = super::super::ecstore_disk::endpoint::Endpoint;
#[cfg(test)]
pub(crate) type HealBucketInfo = super::super::contract::bucket::BucketInfo;
@@ -344,9 +346,9 @@ pub(crate) mod s3_api_consumer {
pub(crate) mod sse_consumer {
pub(crate) use super::super::sse::{
EncryptionKeyKind, SSEType, bucket_default_write_sse, build_ssec_read_headers, encryption_material_to_metadata,
extract_ssec_params_from_headers, extract_ssekms_context_from_headers, log_sse_kms_key_policy_mode,
map_get_object_reader_error, mark_encrypted_multipart_metadata,
EncryptionKeyKind, SSEType, build_ssec_read_headers, encryption_material_to_metadata, extract_ssec_params_from_headers,
extract_ssekms_context_from_headers, log_sse_kms_key_policy_mode, map_get_object_reader_error,
mark_encrypted_multipart_metadata,
};
pub(crate) use super::{
DecryptionRequest, EncryptionRequest, PrepareEncryptionRequest, SseKmsPrincipal, apply_bucket_default_lock_retention,