feat(heal): add progress and trace observability (#6179)

* feat(heal): track erasure set progress baseline

Record erasure-set heal byte progress from per-object results and seed progress totals from complete usage-cache snapshots when available.

Keep usage-cache failures observational so heal execution continues without a baseline.

Co-Authored-By: heihutu <heihutu@gmail.com>

* feat(heal): skip filtered erasure set versions

Skip erasure-set versions written after the durable heal start time, and queue lifecycle-expired versions for expiry before skipping them.

Track new-version and ILM-expired skips separately so progress can explain completed baseline work without treating these skips as retry-blocking failures.

Co-Authored-By: heihutu <heihutu@gmail.com>

* feat(heal): wire abandoned data-dir cleanup check

Connect check_abandoned_parts through ECStore, pool, and set layers so heal can invoke the existing orphan data-dir reclaim path instead of returning NotImplemented.

Add dry-run support to the reclaim scan and cover dry-run plus scoped set behavior with regression tests.

Co-Authored-By: heihutu <heihutu@gmail.com>

* feat(obs): add heal scanner trace bus

Introduce an in-process broadcast trace bus with typed heal and scanner events, lazy event construction, and bounded lagged-subscriber behavior.

Cover zero-subscriber publishing, subscription delivery, drop accounting, and lagged receivers with focused common-crate tests.

Co-Authored-By: heihutu <heihutu@gmail.com>

* feat(obs): stream heal trace events from admin API

Wire the admin trace endpoint to the common trace bus for heal/scanner events, including kind, regex, and threshold filtering.

Co-Authored-By: heihutu <heihutu@gmail.com>

* feat(obs): emit heal trace events

Publish heal task lifecycle and abandoned-parts cleanup events through the common trace bus so the admin trace stream has live heal diagnostics.

Co-Authored-By: heihutu <heihutu@gmail.com>

* feat(obs): emit scanner trace events

Publish scanner folder, lifecycle action, and heal-candidate events through the common trace bus for live admin scanner diagnostics.

Co-Authored-By: heihutu <heihutu@gmail.com>

* fix(heal): route data usage loader through storage api

Keep ECStore data-usage facade access behind the heal storage_api boundary so architecture migration guards can validate the heal progress path.

Co-Authored-By: heihutu <heihutu@gmail.com>

* perf(heal): avoid lifecycle snapshots on ordinary heal pages

Only request lifecycle object snapshots when the heal pass has lifecycle expiry context. This keeps ordinary listing and disk-walk pages from cloning FileInfo/ObjectInfo payloads while preserving the skip path that queues expired versions.

Co-Authored-By: heihutu <heihutu@gmail.com>

* test(heal): update bug-fix mocks for lifecycle snapshots

Carry the lifecycle snapshot opt-in argument through the remaining heal bug-fix test mocks so all-targets clippy covers the updated storage trait.

Co-Authored-By: heihutu <heihutu@gmail.com>

* test(rustfs): sync heal storage mock signature

Update the rustfs storage RPC test mock for the lifecycle snapshot opt-in argument and cover it with rustfs all-targets clippy.

Co-Authored-By: heihutu <heihutu@gmail.com>

* test(e2e): allocate smoke ports across nextest processes

Serialize E2E port selection with a small /tmp allocator so nextest workers do not reuse the same just-released ephemeral port before RustFS binds it.

Co-Authored-By: heihutu <heihutu@gmail.com>

---------

Co-authored-by: heihutu <heihutu@gmail.com>
This commit is contained in:
houseme
2026-08-18 08:29:29 +08:00
committed by GitHub
parent 7cb91a0190
commit 360bceafce
30 changed files with 2383 additions and 112 deletions
+1
View File
@@ -767,6 +767,7 @@ mod tests {
_bucket: &str,
_prefix: &str,
_continuation_token: Option<&str>,
_include_lifecycle_object_info: bool,
) -> crate::Result<(Vec<crate::heal::storage::HealListItem>, Option<String>, bool)> {
Ok((vec![], None, false))
}
+317 -27
View File
@@ -23,13 +23,14 @@ use crate::heal::{
};
use crate::{Error, Result};
use futures::{StreamExt, stream::FuturesUnordered};
use metrics::gauge;
use metrics::{counter, gauge};
use rustfs_common::heal_channel::{HealOpts, HealRequestSource, HealScanMode};
use rustfs_madmin::heal_commands::HealResultItem;
use std::sync::{
Arc,
atomic::{AtomicUsize, Ordering},
};
use std::time::{Duration, UNIX_EPOCH};
use tokio::sync::{RwLock, Semaphore};
use tracing::{debug, error, warn};
@@ -47,6 +48,21 @@ enum HealObjectOutcome {
Failed,
}
fn result_object_size_u64(result: &HealResultItem) -> u64 {
u64::try_from(result.object_size).unwrap_or(u64::MAX)
}
const NEW_VERSION_SKIP_GRACE_SECS: u64 = 60;
const NANOS_PER_SECOND: i128 = 1_000_000_000;
fn should_skip_new_version(mod_time_unix_nanos: Option<i128>, started_at_secs: u64) -> bool {
let Some(mod_time_unix_nanos) = mod_time_unix_nanos else {
return false;
};
let cutoff_secs = started_at_secs.saturating_add(NEW_VERSION_SKIP_GRACE_SECS);
mod_time_unix_nanos > i128::from(cutoff_secs).saturating_mul(NANOS_PER_SECOND)
}
struct PageConcurrencyGuard {
in_flight: Arc<AtomicUsize>,
set_label: String,
@@ -492,6 +508,7 @@ impl ErasureSetHealer {
&mut skipped_objects,
resume_manager,
checkpoint_manager,
state.start_time,
)
.await;
@@ -658,6 +675,7 @@ impl ErasureSetHealer {
skipped_objects: &mut u64,
resume_manager: &ResumeManager,
checkpoint_manager: &CheckpointManager,
started_at_secs: u64,
) -> Result<()> {
debug!(
target: "rustfs::heal::erasure_healer",
@@ -710,6 +728,7 @@ impl ErasureSetHealer {
// The end-of-pass summary reports the full failed/skipped counts.
let mut transient_skip_samples_logged = 0_u64;
let mut failure_samples_logged = 0_u64;
let mut bytes_processed = self.progress.read().await.bytes_processed;
// backlog#920: select the per-erasure-set DISK-WALK union enumerator when
// the scan is Deep OR the request came from AutoHeal — these are the paths
@@ -718,17 +737,25 @@ impl ErasureSetHealer {
// which stays the default.
let use_disk_walk =
matches!(self.heal_opts.scan_mode, HealScanMode::Deep) || matches!(self.source, HealRequestSource::AutoHeal);
let lifecycle_expiry_context = self.storage.load_heal_lifecycle_expiry_context(bucket).await?;
let include_lifecycle_object_info = lifecycle_expiry_context.is_some();
loop {
self.verify_replacement_identity_fence("page scan").await?;
// Get one page of object versions
let (objects, next_token, is_truncated) = if use_disk_walk {
self.storage
.list_versions_for_heal_page_disk_walk(set_disk_id, bucket, "", continuation_token.as_deref())
.list_versions_for_heal_page_disk_walk(
set_disk_id,
bucket,
"",
continuation_token.as_deref(),
include_lifecycle_object_info,
)
.await?
} else {
self.storage
.list_objects_for_heal_page(bucket, "", continuation_token.as_deref())
.list_objects_for_heal_page(bucket, "", continuation_token.as_deref(), include_lifecycle_object_info)
.await?
};
let page_is_empty = objects.is_empty();
@@ -736,6 +763,7 @@ impl ErasureSetHealer {
let page_resume_index = *current_object_index;
let semaphore = Arc::new(Semaphore::new(page_concurrency_limit));
let mut page_tasks = FuturesUnordered::new();
let mut completed_in_page = 0usize;
// Capture the last version identity of this page for the anti-loop guard.
let page_last = objects.last().map(|item| (item.name.clone(), item.version_id.clone()));
@@ -751,6 +779,75 @@ impl ErasureSetHealer {
continue;
}
if should_skip_new_version(item.mod_time_unix_nanos, started_at_secs) {
checkpoint_manager.add_processed_object(key).await?;
*processed_objects = processed_objects.saturating_add(1);
completed_in_page = completed_in_page.saturating_add(1);
counter!("rustfs_heal_skipped_new_versions_total").increment(1);
{
let mut progress = self.progress.write().await;
progress.record_skipped_new_version();
progress.set_current_object(Some(format!("skipped_new: {bucket}/{}", item.name)));
progress.update_progress(*processed_objects, *successful_objects, *failed_objects, bytes_processed);
}
debug!(
target: "rustfs::heal::erasure_healer",
event = EVENT_HEAL_ERASURE_OBJECT_STATE,
component = LOG_COMPONENT_HEAL,
subsystem = LOG_SUBSYSTEM_ERASURE_HEALER,
set_disk_id,
bucket,
object = %item.name,
version_id = ?item.version_id,
state = "skipped_new_version",
"Erasure set object version skipped because it was written after heal started"
);
if completed_in_page.is_multiple_of(100) {
checkpoint_manager.update_position(bucket_index, page_resume_index).await?;
}
continue;
}
if let Some(context) = lifecycle_expiry_context.as_ref()
&& self
.storage
.enqueue_heal_lifecycle_expiry(
context,
bucket,
&item.name,
item.version_id.as_deref(),
item.lifecycle_object_info.as_ref(),
)
.await?
{
checkpoint_manager.add_processed_object(key).await?;
*processed_objects = processed_objects.saturating_add(1);
completed_in_page = completed_in_page.saturating_add(1);
counter!("rustfs_heal_skipped_ilm_expired_total").increment(1);
{
let mut progress = self.progress.write().await;
progress.record_skipped_ilm_expired();
progress.set_current_object(Some(format!("skipped_ilm: {bucket}/{}", item.name)));
progress.update_progress(*processed_objects, *successful_objects, *failed_objects, bytes_processed);
}
debug!(
target: "rustfs::heal::erasure_healer",
event = EVENT_HEAL_ERASURE_OBJECT_STATE,
component = LOG_COMPONENT_HEAL,
subsystem = LOG_SUBSYSTEM_ERASURE_HEALER,
set_disk_id,
bucket,
object = %item.name,
version_id = ?item.version_id,
state = "skipped_ilm_expired",
"Erasure set object version skipped because lifecycle expiry was queued"
);
if completed_in_page.is_multiple_of(100) {
checkpoint_manager.update_position(bucket_index, page_resume_index).await?;
}
continue;
}
resume_manager
.set_current_item(Some(bucket.to_string()), Some(item.name.clone()))
.await?;
@@ -777,7 +874,7 @@ impl ErasureSetHealer {
let _permit = match permit {
Ok(permit) => permit,
Err(err) => return (dedup_key, object_name, version_id, Err(err)),
Err(err) => return (dedup_key, object_name, version_id, (0, Err(err))),
};
let _in_flight_guard = PageConcurrencyGuard::new(in_flight, set_label);
@@ -788,7 +885,7 @@ impl ErasureSetHealer {
// recorded as skipped-ok rather than failed. The delete-marker
// vs data path is chosen internally in ops/heal.rs.
let result = if cancel_token.is_cancelled() {
Err(Error::TaskCancelled)
(0, Err(Error::TaskCancelled))
} else {
match storage
.heal_object(&bucket_name, &object_name, version_id.as_deref(), &heal_opts)
@@ -797,8 +894,9 @@ impl ErasureSetHealer {
Ok((result, None))
if target_outcomes_complete(&result, &target_endpoints) =>
{
let object_size = result_object_size_u64(&result);
if !replacement_commit_evidence_required {
Ok(true)
(object_size, Ok(true))
} else {
match storage
.replacement_targets_have_version(
@@ -810,27 +908,42 @@ impl ErasureSetHealer {
)
.await
{
Ok(true) => Ok(true),
Ok(false) => Err(Error::transient_skip(format!(
Ok(true) => (object_size, Ok(true)),
Ok(false) => (object_size, Err(Error::transient_skip(format!(
"Skipped heal for {bucket_name}/{object_name} because replacement target readback did not confirm the committed version"
))),
Err(err) => Err(Error::transient_skip(format!(
)))),
Err(err) => (object_size, Err(Error::transient_skip(format!(
"Skipped heal for {bucket_name}/{object_name} because replacement target readback failed: {err}"
))),
)))),
}
}
}
Ok((_result, None)) if !target_endpoints.is_empty() => Err(Error::transient_skip(format!(
"Skipped heal for {bucket_name}/{object_name} because a replacement target was not committed"
))),
Ok((_result, None)) => Ok(true),
Ok((_, Some(err))) if is_missing_object_dir_heal_result(&object_name, &err) => Ok(false),
Ok((_, Some(err))) | Err(err) => match Self::classify_heal_object_error(&err) {
HealObjectOutcome::Absent => Ok(false),
HealObjectOutcome::Transient => Err(Error::transient_skip(format!(
"Skipped heal for {bucket_name}/{object_name} due to transient error: {err}"
},
Ok((result, None)) if !target_endpoints.is_empty() => (
result_object_size_u64(&result),
Err(Error::transient_skip(format!(
"Skipped heal for {bucket_name}/{object_name} because a replacement target was not committed"
))),
HealObjectOutcome::Failed => Err(err),
),
Ok((result, None)) => (result_object_size_u64(&result), Ok(true)),
Ok((result, Some(err))) if is_missing_object_dir_heal_result(&object_name, &err) => {
(result_object_size_u64(&result), Ok(false))
}
Ok((result, Some(err))) => {
let object_size = result_object_size_u64(&result);
match Self::classify_heal_object_error(&err) {
HealObjectOutcome::Absent => (object_size, Ok(false)),
HealObjectOutcome::Transient => (object_size, Err(Error::transient_skip(format!(
"Skipped heal for {bucket_name}/{object_name} due to transient error: {err}"
)))),
HealObjectOutcome::Failed => (object_size, Err(err)),
}
}
Err(err) => match Self::classify_heal_object_error(&err) {
HealObjectOutcome::Absent => (0, Ok(false)),
HealObjectOutcome::Transient => (0, Err(Error::transient_skip(format!(
"Skipped heal for {bucket_name}/{object_name} due to transient error: {err}"
)))),
HealObjectOutcome::Failed => (0, Err(err)),
},
}
};
@@ -839,11 +952,12 @@ impl ErasureSetHealer {
});
}
let mut completed_in_page = 0usize;
while let Some((key, object, version_id, result)) = page_tasks.next().await {
let (object_size, result) = result;
match result {
Ok(true) => {
*successful_objects += 1;
bytes_processed = bytes_processed.saturating_add(object_size);
checkpoint_manager.add_processed_object(key).await?;
debug!(
target: "rustfs::heal::erasure_healer",
@@ -861,6 +975,7 @@ impl ErasureSetHealer {
Ok(false) => {
checkpoint_manager.add_processed_object(key).await?;
*successful_objects += 1;
bytes_processed = bytes_processed.saturating_add(object_size);
debug!(
target: "rustfs::heal::erasure_healer",
event = EVENT_HEAL_ERASURE_OBJECT_STATE,
@@ -877,6 +992,7 @@ impl ErasureSetHealer {
Err(err @ Error::TaskCancelled) | Err(err @ Error::TaskTimeout) => return Err(err),
Err(Error::TransientSkip { message }) => {
*skipped_objects += 1;
bytes_processed = bytes_processed.saturating_add(object_size);
checkpoint_manager.add_skipped_object(key).await?;
demote_to_debug_when!(!take_failure_log_sample(&mut transient_skip_samples_logged), warn, target: "rustfs::heal::erasure_healer", {
event = EVENT_HEAL_ERASURE_OBJECT_STATE,
@@ -893,6 +1009,7 @@ impl ErasureSetHealer {
}
Err(err) => {
*failed_objects += 1;
bytes_processed = bytes_processed.saturating_add(object_size);
checkpoint_manager.add_failed_object(key).await?;
demote_to_debug_when!(!take_failure_log_sample(&mut failure_samples_logged), warn, target: "rustfs::heal::erasure_healer", {
event = EVENT_HEAL_ERASURE_OBJECT_STATE,
@@ -911,6 +1028,11 @@ impl ErasureSetHealer {
*processed_objects += 1;
completed_in_page += 1;
{
let mut progress = self.progress.write().await;
progress.set_current_object(Some(format!("{bucket}/{object}")));
progress.update_progress(*processed_objects, *successful_objects, *failed_objects, bytes_processed);
}
if completed_in_page.is_multiple_of(100) {
checkpoint_manager.update_position(bucket_index, page_resume_index).await?;
@@ -964,7 +1086,9 @@ impl ErasureSetHealer {
progress.objects_scanned = state.total_objects;
progress.objects_healed = state.successful_objects;
progress.objects_failed = state.failed_objects;
progress.bytes_processed = 0; // set to 0 for now, can be extended later
progress.bytes_processed = 0; // Resume state tracks object counts, not byte counters.
progress.start_time = UNIX_EPOCH.checked_add(Duration::from_secs(state.start_time));
progress.last_update_time = UNIX_EPOCH.checked_add(Duration::from_secs(state.last_update));
progress.set_current_object(state.current_object.clone());
}
}
@@ -1135,13 +1259,15 @@ mod resume_loop_tests {
//! that emits programmable multi-version pages. These exercise the real loop
//! logic (cursor seeding, per-version dedup, anti-loop guard, absence
//! handling) — not merely a mock's own output.
use super::{ErasureSetHealer, target_outcomes_complete};
use super::{
ErasureSetHealer, NANOS_PER_SECOND, NEW_VERSION_SKIP_GRACE_SECS, should_skip_new_version, target_outcomes_complete,
};
use crate::heal::progress::HealProgress;
use crate::heal::resume::{
CheckpointManager, RESUME_CHECKPOINT_FILE, ReplacementTargetIdentity, ResumeDeleteFailure, ResumeManager, ResumeUtils,
compose_key,
};
use crate::heal::storage::{DiskStatus, HealListItem, HealObjectInfo, HealStorageAPI};
use crate::heal::storage::{DiskStatus, HealLifecycleExpiryContext, HealListItem, HealObjectInfo, HealStorageAPI};
use crate::heal::storage_api::status::BucketInfo;
use crate::heal::{
BUCKET_META_PREFIX, DiskOption, DiskStore, EcstoreError, Endpoint, HealDiskExt as _, RUSTFS_META_BUCKET, new_disk,
@@ -1149,7 +1275,7 @@ mod resume_loop_tests {
use crate::{Error, Result};
use rustfs_common::heal_channel::{HealOpts, HealRequestSource};
use rustfs_madmin::heal_commands::{HealDriveInfo, HealResultItem, Infos};
use std::collections::{HashMap, VecDeque};
use std::collections::{HashMap, HashSet, VecDeque};
use std::sync::atomic::{AtomicBool, Ordering};
use std::sync::{Arc, Mutex};
use tempfile::TempDir;
@@ -1160,10 +1286,37 @@ mod resume_loop_tests {
HealListItem {
name: name.to_string(),
version_id: version.map(str::to_string),
mod_time_unix_nanos: None,
lifecycle_object_info: None,
is_delete_marker: delete_marker,
}
}
fn item_with_mod_time(name: &str, version: Option<&str>, mod_time_secs: u64) -> HealListItem {
HealListItem {
name: name.to_string(),
version_id: version.map(str::to_string),
mod_time_unix_nanos: Some(i128::from(mod_time_secs).saturating_mul(NANOS_PER_SECOND)),
lifecycle_object_info: None,
is_delete_marker: false,
}
}
#[test]
fn new_version_filter_respects_grace_boundary() {
let started_at = 1_700_000_000;
assert!(!should_skip_new_version(None, started_at));
assert!(!should_skip_new_version(
Some(i128::from(started_at + NEW_VERSION_SKIP_GRACE_SECS).saturating_mul(NANOS_PER_SECOND)),
started_at,
));
assert!(should_skip_new_version(
Some(i128::from(started_at + NEW_VERSION_SKIP_GRACE_SECS + 1).saturating_mul(NANOS_PER_SECOND)),
started_at,
));
}
#[test]
fn target_outcomes_require_each_requested_endpoint_once_and_ok() {
let result = HealResultItem {
@@ -1246,8 +1399,10 @@ mod resume_loop_tests {
/// Target-specific physical readback evidence per `compose_key`; the
/// fake models a healthy backend unless a test explicitly revokes it.
replacement_commit_evidence: Mutex<HashMap<String, ReplacementCommitEvidence>>,
lifecycle_expired: Mutex<HashSet<String>>,
/// every heal_object call recorded as (name, version_id)
heal_calls: Mutex<Vec<(String, Option<String>)>>,
list_include_lifecycle_object_info: Mutex<Vec<bool>>,
replacement_target_identity_sequences: Mutex<VecDeque<Vec<ReplacementTargetIdentity>>>,
fail_listing: AtomicBool,
}
@@ -1274,9 +1429,15 @@ mod resume_loop_tests {
.unwrap()
.insert(compose_key(name, version), ReplacementCommitEvidence::Error(message.to_string()));
}
fn set_lifecycle_expired(&self, name: &str, version: Option<&str>) {
self.lifecycle_expired.lock().unwrap().insert(compose_key(name, version));
}
fn calls(&self) -> Vec<(String, Option<String>)> {
self.heal_calls.lock().unwrap().clone()
}
fn list_include_lifecycle_object_info_calls(&self) -> Vec<bool> {
self.list_include_lifecycle_object_info.lock().unwrap().clone()
}
fn fail_listing(&self) {
self.fail_listing.store(true, Ordering::SeqCst);
}
@@ -1330,6 +1491,23 @@ mod resume_loop_tests {
async fn get_object_checksum(&self, _b: &str, _o: &str) -> Result<Option<String>> {
Ok(None)
}
async fn load_heal_lifecycle_expiry_context(&self, _bucket: &str) -> Result<Option<HealLifecycleExpiryContext>> {
Ok((!self.lifecycle_expired.lock().unwrap().is_empty()).then(HealLifecycleExpiryContext::test))
}
async fn enqueue_heal_lifecycle_expiry(
&self,
_context: &HealLifecycleExpiryContext,
_bucket: &str,
object: &str,
version_id: Option<&str>,
_object_info: Option<&HealObjectInfo>,
) -> Result<bool> {
Ok(self
.lifecycle_expired
.lock()
.unwrap()
.contains(&compose_key(object, version_id)))
}
async fn heal_object(
&self,
_bucket: &str,
@@ -1386,7 +1564,12 @@ mod resume_loop_tests {
_bucket: &str,
_prefix: &str,
continuation_token: Option<&str>,
include_lifecycle_object_info: bool,
) -> Result<(Vec<HealListItem>, Option<String>, bool)> {
self.list_include_lifecycle_object_info
.lock()
.unwrap()
.push(include_lifecycle_object_info);
if self.fail_listing.load(Ordering::SeqCst) {
return Err(Error::other("injected listing failure"));
}
@@ -1476,6 +1659,7 @@ mod resume_loop_tests {
/// Drive one bucket heal pass; returns (processed, successful, failed, skipped, result).
async fn run(env: &Env) -> (u64, u64, u64, u64, Result<()>) {
let state = env.resume.get_state().await;
let mut current_object_index = 0usize;
let mut processed = 0u64;
let mut successful = 0u64;
@@ -1494,6 +1678,7 @@ mod resume_loop_tests {
&mut skipped,
&env.resume,
&env.checkpoint,
state.start_time,
)
.await;
(processed, successful, failed, skipped, result)
@@ -1559,6 +1744,7 @@ mod resume_loop_tests {
let mut successful = 0;
let mut failed = 0;
let mut skipped = 0;
let started_at = env.resume.get_state().await.start_time;
let error = healer
.heal_bucket_with_resume(
@@ -1572,6 +1758,7 @@ mod resume_loop_tests {
&mut skipped,
&env.resume,
&env.checkpoint,
started_at,
)
.await
.expect_err("a remounted target must not begin a new page scan");
@@ -1641,6 +1828,109 @@ mod resume_loop_tests {
assert_eq!(skipped, 0);
}
#[tokio::test]
async fn erasure_set_progress_accumulates_healed_object_bytes() {
let env = make_env().await;
env.storage.set_page(
None,
Page {
items: vec![item("first", Some("v1"), false), item("second", Some("v2"), false)],
next: None,
truncated: false,
},
);
env.storage.set_result(
"first",
Some("v1"),
HealResultItem {
object_size: 1024,
..Default::default()
},
);
env.storage.set_result(
"second",
Some("v2"),
HealResultItem {
object_size: 2048,
..Default::default()
},
);
let (processed, successful, failed, skipped, result) = run(&env).await;
result.expect("page heal should succeed");
assert_eq!(processed, 2);
assert_eq!(successful, 2);
assert_eq!(failed, 0);
assert_eq!(skipped, 0);
let progress = env.healer.progress.read().await;
assert_eq!(progress.objects_scanned, 2);
assert_eq!(progress.objects_healed, 2);
assert_eq!(progress.objects_failed, 0);
assert_eq!(progress.bytes_processed, 3072);
assert!(matches!(progress.current_object.as_deref(), Some("b/first" | "b/second")));
}
#[tokio::test]
async fn erasure_set_skips_versions_written_after_heal_started() {
let env = make_env().await;
let started_at = env.resume.get_state().await.start_time;
env.storage.set_page(
None,
Page {
items: vec![
item_with_mod_time("old", Some("v1"), started_at + NEW_VERSION_SKIP_GRACE_SECS),
item_with_mod_time("new", Some("v2"), started_at + NEW_VERSION_SKIP_GRACE_SECS + 1),
],
next: None,
truncated: false,
},
);
let (processed, successful, failed, skipped, result) = run(&env).await;
result.expect("page heal should succeed");
assert_eq!(processed, 2);
assert_eq!(successful, 1);
assert_eq!(failed, 0);
assert_eq!(skipped, 0);
assert_eq!(env.storage.calls(), vec![("old".to_string(), Some("v1".to_string()))]);
let progress = env.healer.progress.read().await;
assert_eq!(progress.skipped_new_versions, 1);
assert_eq!(progress.objects_scanned, 2);
assert_eq!(progress.objects_healed, 1);
assert_eq!(progress.objects_failed, 0);
}
#[tokio::test]
async fn erasure_set_skips_versions_queued_for_lifecycle_expiry() {
let env = make_env().await;
env.storage.set_page(
None,
Page {
items: vec![item("expired", Some("v1"), false), item("kept", Some("v2"), false)],
next: None,
truncated: false,
},
);
env.storage.set_lifecycle_expired("expired", Some("v1"));
let (processed, successful, failed, skipped, result) = run(&env).await;
result.expect("page heal should succeed");
assert_eq!(processed, 2);
assert_eq!(successful, 1);
assert_eq!(failed, 0);
assert_eq!(skipped, 0);
assert_eq!(env.storage.calls(), vec![("kept".to_string(), Some("v2".to_string()))]);
assert_eq!(env.storage.list_include_lifecycle_object_info_calls(), vec![true]);
let progress = env.healer.progress.read().await;
assert_eq!(progress.skipped_ilm_expired, 1);
assert_eq!(progress.objects_scanned, 2);
assert_eq!(progress.objects_healed, 1);
assert_eq!(progress.objects_failed, 0);
}
#[tokio::test]
async fn bucket_listing_failure_does_not_mark_set_completed() {
let env = make_env().await;
+30
View File
@@ -2385,8 +2385,27 @@ impl HealManager {
snapshot.objects_scanned = snapshot.objects_scanned.saturating_add(progress.objects_scanned);
snapshot.objects_healed = snapshot.objects_healed.saturating_add(progress.objects_healed);
snapshot.objects_failed = snapshot.objects_failed.saturating_add(progress.objects_failed);
snapshot.skipped_new_versions = snapshot.skipped_new_versions.saturating_add(progress.skipped_new_versions);
snapshot.skipped_ilm_expired = snapshot.skipped_ilm_expired.saturating_add(progress.skipped_ilm_expired);
snapshot.objects_total_count = snapshot.objects_total_count.saturating_add(progress.objects_total_count);
snapshot.objects_total_size = snapshot.objects_total_size.saturating_add(progress.objects_total_size);
snapshot.bytes_processed = snapshot.bytes_processed.saturating_add(progress.bytes_processed);
snapshot.start_time = match (snapshot.start_time, progress.start_time) {
(Some(current), Some(next)) => Some(current.min(next)),
(None, next) => next,
(current, None) => current,
};
snapshot.last_update_time = match (snapshot.last_update_time, progress.last_update_time) {
(Some(current), Some(next)) => Some(current.max(next)),
(None, next) => next,
(current, None) => current,
};
if progress.current_object.is_some() {
snapshot.current_object = progress.current_object;
}
}
snapshot.refresh_progress_percentage();
snapshot.refresh_estimated_completion_time();
Some(snapshot)
}
@@ -3208,6 +3227,7 @@ impl HealManager {
} else {
completed_task.get_status().await
};
let completed_progress = completed_task.get_progress().await;
let completed_status_entry = CompletedHealStatus {
heal_type: completed_task.heal_type.clone(),
status: completed_status.clone(),
@@ -3223,6 +3243,7 @@ impl HealManager {
match completed_status {
HealTaskStatus::Completed => {
stats.update_task_completion(true);
stats.add_healed_objects(completed_progress.objects_healed, completed_progress.bytes_processed);
}
HealTaskStatus::Retrying { .. } => {}
_ => {
@@ -3749,6 +3770,7 @@ mod tests {
_bucket: &str,
_prefix: &str,
_continuation_token: Option<&str>,
_include_lifecycle_object_info: bool,
) -> Result<(Vec<crate::heal::storage::HealListItem>, Option<String>, bool)> {
Ok((Vec::new(), None, false))
}
@@ -5396,6 +5418,8 @@ mod tests {
));
{
let mut progress = first.progress.write().await;
progress.start_time = Some(SystemTime::now() - Duration::from_secs(20));
progress.set_total_baseline(12, 8192);
progress.update_progress(7, 3, 1, 4096);
}
@@ -5405,6 +5429,8 @@ mod tests {
));
{
let mut progress = second.progress.write().await;
progress.start_time = Some(SystemTime::now() - Duration::from_secs(10));
progress.set_total_baseline(8, 4096);
progress.update_progress(11, 5, 2, 2048);
}
@@ -5419,7 +5445,11 @@ mod tests {
assert_eq!(progress.objects_scanned, 18);
assert_eq!(progress.objects_healed, 8);
assert_eq!(progress.objects_failed, 3);
assert_eq!(progress.objects_total_count, 20);
assert_eq!(progress.objects_total_size, 12288);
assert_eq!(progress.bytes_processed, 6144);
assert!((progress.progress_percentage - 50.0).abs() < 0.001);
assert!(progress.estimated_completion_time.is_some());
}
#[tokio::test]
+160 -6
View File
@@ -13,7 +13,7 @@
// limitations under the License.
use serde::{Deserialize, Serialize};
use std::time::SystemTime;
use std::time::{Duration, SystemTime};
#[derive(Debug, Default, Clone, Serialize, Deserialize)]
#[serde(rename_all = "camelCase")]
@@ -24,6 +24,14 @@ pub struct HealProgress {
pub objects_healed: u64,
/// Objects failed
pub objects_failed: u64,
/// Versions skipped because they were written after this heal started
pub skipped_new_versions: u64,
/// Versions skipped because lifecycle already selected them for expiry
pub skipped_ilm_expired: u64,
/// Baseline object count from the latest complete usage snapshot
pub objects_total_count: u64,
/// Baseline object bytes from the latest complete usage snapshot
pub objects_total_size: u64,
/// Bytes processed
pub bytes_processed: u64,
/// Current object
@@ -54,10 +62,56 @@ impl HealProgress {
self.bytes_processed = bytes;
self.last_update_time = Some(SystemTime::now());
// calculate progress percentage
let total = scanned + healed + failed;
self.refresh_progress_percentage();
self.refresh_estimated_completion_time();
}
pub fn set_total_baseline(&mut self, objects_total_count: u64, objects_total_size: u64) {
self.objects_total_count = objects_total_count;
self.objects_total_size = objects_total_size;
self.last_update_time = Some(SystemTime::now());
self.refresh_progress_percentage();
self.refresh_estimated_completion_time();
}
pub fn record_skipped_new_version(&mut self) {
self.skipped_new_versions = self.skipped_new_versions.saturating_add(1);
self.last_update_time = Some(SystemTime::now());
self.refresh_progress_percentage();
self.refresh_estimated_completion_time();
}
pub fn record_skipped_ilm_expired(&mut self) {
self.skipped_ilm_expired = self.skipped_ilm_expired.saturating_add(1);
self.last_update_time = Some(SystemTime::now());
self.refresh_progress_percentage();
self.refresh_estimated_completion_time();
}
fn completed_for_baseline(&self) -> u64 {
self.objects_healed
.saturating_add(self.objects_failed)
.saturating_add(self.skipped_new_versions)
.saturating_add(self.skipped_ilm_expired)
}
pub(crate) fn refresh_progress_percentage(&mut self) {
if self.objects_total_size > 0 {
self.progress_percentage = ((self.bytes_processed as f64 / self.objects_total_size as f64) * 100.0).min(100.0);
return;
}
if self.objects_total_count > 0 {
let completed = self.completed_for_baseline();
self.progress_percentage = ((completed as f64 / self.objects_total_count as f64) * 100.0).min(100.0);
return;
}
let total = self
.objects_scanned
.saturating_add(self.objects_healed)
.saturating_add(self.objects_failed);
if total > 0 {
self.progress_percentage = (healed as f64 / total as f64) * 100.0;
self.progress_percentage = (self.objects_healed as f64 / total as f64) * 100.0;
}
}
@@ -66,9 +120,36 @@ impl HealProgress {
self.last_update_time = Some(SystemTime::now());
}
pub fn refresh_estimated_completion_time(&mut self) {
let Some(start_time) = self.start_time else {
self.estimated_completion_time = None;
return;
};
if self.is_completed() || !(0.0..100.0).contains(&self.progress_percentage) || self.bytes_processed == 0 {
self.estimated_completion_time = None;
return;
}
let elapsed = match SystemTime::now().duration_since(start_time) {
Ok(elapsed) if !elapsed.is_zero() => elapsed,
_ => {
self.estimated_completion_time = None;
return;
}
};
let estimated_total_secs = elapsed.as_secs_f64() * 100.0 / self.progress_percentage;
self.estimated_completion_time = start_time.checked_add(Duration::from_secs_f64(estimated_total_secs));
}
pub fn is_completed(&self) -> bool {
self.progress_percentage >= 100.0
|| self.objects_scanned > 0 && self.objects_healed + self.objects_failed >= self.objects_scanned
if self.progress_percentage >= 100.0 {
return true;
}
if self.objects_total_count > 0 || self.objects_total_size > 0 {
return false;
}
self.objects_scanned > 0 && self.objects_healed.saturating_add(self.objects_failed) >= self.objects_scanned
}
pub fn get_success_rate(&self) -> f64 {
@@ -158,6 +239,10 @@ mod tests {
assert_eq!(progress.objects_scanned, 0);
assert_eq!(progress.objects_healed, 0);
assert_eq!(progress.objects_failed, 0);
assert_eq!(progress.skipped_new_versions, 0);
assert_eq!(progress.skipped_ilm_expired, 0);
assert_eq!(progress.objects_total_count, 0);
assert_eq!(progress.objects_total_size, 0);
assert_eq!(progress.bytes_processed, 0);
assert_eq!(progress.progress_percentage, 0.0);
assert!(progress.start_time.is_some());
@@ -181,6 +266,73 @@ mod tests {
assert!(progress.last_update_time.is_some());
}
#[test]
fn test_heal_progress_estimates_completion_time_from_progress() {
let mut progress = HealProgress::new();
progress.start_time = Some(SystemTime::now() - Duration::from_secs(10));
progress.update_progress(100, 25, 0, 4096);
let eta = progress
.estimated_completion_time
.expect("partial byte progress should estimate completion");
assert!(eta > SystemTime::now());
}
#[test]
fn test_heal_progress_uses_byte_baseline_for_percentage() {
let mut progress = HealProgress::new();
progress.set_total_baseline(10, 8192);
progress.update_progress(100, 25, 0, 4096);
assert!((progress.progress_percentage - 50.0).abs() < 0.001);
}
#[test]
fn test_heal_progress_uses_object_baseline_when_bytes_unknown() {
let mut progress = HealProgress::new();
progress.set_total_baseline(10, 0);
progress.update_progress(100, 3, 2, 0);
assert!((progress.progress_percentage - 50.0).abs() < 0.001);
}
#[test]
fn test_heal_progress_counts_skipped_versions_for_object_baseline() {
let mut progress = HealProgress::new();
progress.set_total_baseline(10, 0);
progress.update_progress(100, 3, 2, 0);
progress.record_skipped_new_version();
assert_eq!(progress.skipped_new_versions, 1);
assert!((progress.progress_percentage - 60.0).abs() < 0.001);
}
#[test]
fn test_heal_progress_does_not_estimate_completion_without_bytes() {
let mut progress = HealProgress::new();
progress.start_time = Some(SystemTime::now() - Duration::from_secs(10));
progress.update_progress(100, 25, 0, 0);
assert!(progress.estimated_completion_time.is_none());
}
#[test]
fn test_heal_progress_with_baseline_is_not_completed_by_processed_count() {
let mut progress = HealProgress::new();
progress.start_time = Some(SystemTime::now() - Duration::from_secs(10));
progress.set_total_baseline(10, 8192);
progress.update_progress(1, 1, 0, 1024);
assert!(!progress.is_completed());
assert!(progress.estimated_completion_time.is_some());
}
#[test]
fn test_heal_progress_update_progress_zero_total() {
let mut progress = HealProgress::new();
@@ -251,6 +403,8 @@ mod tests {
assert_eq!(json["objectsScanned"], 10);
assert_eq!(json["objectsHealed"], 8);
assert_eq!(json["objectsFailed"], 2);
assert_eq!(json["skippedNewVersions"], 0);
assert_eq!(json["skippedIlmExpired"], 0);
assert_eq!(json["bytesProcessed"], 1024);
assert_eq!(json["currentObject"], "test-bucket/test-object");
assert!(json["progressPercentage"].is_number());
+169 -7
View File
@@ -22,6 +22,7 @@ use serde::{Deserialize, Serialize};
use std::sync::Arc;
use tracing::{debug, error, warn};
use super::storage_api::owner::{EcstoreHealLifecycleExpiryContext, ecstore_load_admin_data_usage_from_backend_cached};
use super::storage_api::storage::{
BucketInfo, BucketOperations, DiskSetSelector, HealOperations as _, ListOperations as _, ObjectIO as _,
ObjectOperations as _, StorageAdminApi,
@@ -29,6 +30,37 @@ use super::storage_api::storage::{
use super::{DiskStore, ECStore, Endpoint, HealDiskExt as _, StorageError, resume::ReplacementTargetIdentity};
pub use super::{HealObjectInfo, HealObjectOptions, HealPutObjReader};
#[derive(Debug, Clone, Copy, Default, PartialEq, Eq)]
pub struct HealBucketUsageBaseline {
pub objects_count: u64,
pub bytes: u64,
}
pub struct HealLifecycleExpiryContext {
inner: HealLifecycleExpiryContextInner,
}
enum HealLifecycleExpiryContextInner {
Ecstore(EcstoreHealLifecycleExpiryContext),
#[allow(dead_code)]
Test,
}
impl HealLifecycleExpiryContext {
fn ecstore(inner: EcstoreHealLifecycleExpiryContext) -> Self {
Self {
inner: HealLifecycleExpiryContextInner::Ecstore(inner),
}
}
#[cfg(test)]
pub(crate) fn test() -> Self {
Self {
inner: HealLifecycleExpiryContextInner::Test,
}
}
}
const LOG_COMPONENT_HEAL: &str = "heal";
const LOG_SUBSYSTEM_STORAGE: &str = "storage";
const EVENT_HEAL_STORAGE_OBJECT_IO: &str = "heal_storage_object_io";
@@ -272,6 +304,10 @@ pub struct HealListItem {
pub name: String,
/// normalized version id (`None` when the version is nil/absent)
pub version_id: Option<String>,
/// version modification time as Unix nanoseconds
pub mod_time_unix_nanos: Option<i128>,
/// object snapshot for lifecycle evaluation
pub lifecycle_object_info: Option<HealObjectInfo>,
/// whether this version is a delete marker (observability only)
pub is_delete_marker: bool,
}
@@ -329,6 +365,28 @@ pub trait HealStorageAPI: Send + Sync {
/// Get bucket info
async fn get_bucket_info(&self, bucket: &str) -> Result<Option<BucketInfo>>;
/// Aggregate usage-cache baselines for the requested buckets.
async fn erasure_set_usage_baseline(&self, _buckets: &[String]) -> Result<Option<HealBucketUsageBaseline>> {
Ok(None)
}
/// Load per-bucket lifecycle expiry context for heal skips.
async fn load_heal_lifecycle_expiry_context(&self, _bucket: &str) -> Result<Option<HealLifecycleExpiryContext>> {
Ok(None)
}
/// Queue lifecycle expiry for a version that heal can skip.
async fn enqueue_heal_lifecycle_expiry(
&self,
_context: &HealLifecycleExpiryContext,
_bucket: &str,
_object: &str,
_version_id: Option<&str>,
_object_info: Option<&HealObjectInfo>,
) -> Result<bool> {
Ok(false)
}
/// Fix bucket metadata
async fn heal_bucket_metadata(&self, bucket: &str) -> Result<()>;
@@ -409,6 +467,7 @@ pub trait HealStorageAPI: Send + Sync {
bucket: &str,
prefix: &str,
continuation_token: Option<&str>,
include_lifecycle_object_info: bool,
) -> Result<(Vec<HealListItem>, Option<String>, bool)>;
/// List versions for healing via a per-erasure-set DISK-WALK union enumerator
@@ -427,8 +486,10 @@ pub trait HealStorageAPI: Send + Sync {
bucket: &str,
prefix: &str,
continuation_token: Option<&str>,
include_lifecycle_object_info: bool,
) -> Result<(Vec<HealListItem>, Option<String>, bool)> {
self.list_objects_for_heal_page(bucket, prefix, continuation_token).await
self.list_objects_for_heal_page(bucket, prefix, continuation_token, include_lifecycle_object_info)
.await
}
/// Get disk for resume functionality.
@@ -1021,6 +1082,85 @@ impl HealStorageAPI for ECStoreHealStorage {
}
}
async fn erasure_set_usage_baseline(&self, buckets: &[String]) -> Result<Option<HealBucketUsageBaseline>> {
if buckets.is_empty() {
return Ok(None);
}
let info = match ecstore_load_admin_data_usage_from_backend_cached(self.ecstore.clone()).await {
Ok(info) if info.is_complete_bucket_usage_snapshot() => info,
Ok(_) | Err(_) => return Ok(None),
};
let mut baseline = HealBucketUsageBaseline::default();
for bucket in buckets {
if let Some(usage) = info.buckets_usage.get(bucket) {
baseline.objects_count = baseline.objects_count.saturating_add(usage.objects_count);
baseline.bytes = baseline.bytes.saturating_add(usage.size);
}
}
Ok(Some(baseline))
}
async fn load_heal_lifecycle_expiry_context(&self, bucket: &str) -> Result<Option<HealLifecycleExpiryContext>> {
match self.ecstore.load_heal_lifecycle_expiry_context(bucket).await {
Ok(Some(context)) => Ok(Some(HealLifecycleExpiryContext::ecstore(context))),
Ok(None) => Ok(None),
Err(err) => {
debug!(
target: "rustfs::heal::storage",
event = EVENT_HEAL_STORAGE_ADMIN_OP,
component = LOG_COMPONENT_HEAL,
subsystem = LOG_SUBSYSTEM_STORAGE,
operation = "load_heal_lifecycle_expiry_context",
bucket,
result = "failed",
error = %err,
"Heal storage lifecycle expiry context load failed"
);
Ok(None)
}
}
}
async fn enqueue_heal_lifecycle_expiry(
&self,
context: &HealLifecycleExpiryContext,
bucket: &str,
object: &str,
version_id: Option<&str>,
object_info: Option<&HealObjectInfo>,
) -> Result<bool> {
let context = match &context.inner {
HealLifecycleExpiryContextInner::Ecstore(context) => context,
HealLifecycleExpiryContextInner::Test => return Ok(false),
};
match self
.ecstore
.enqueue_heal_lifecycle_expiry(context, bucket, object, version_id, object_info)
.await
{
Ok(queued) => Ok(queued),
Err(err) => {
debug!(
target: "rustfs::heal::storage",
event = EVENT_HEAL_STORAGE_ADMIN_OP,
component = LOG_COMPONENT_HEAL,
subsystem = LOG_SUBSYSTEM_STORAGE,
operation = "enqueue_heal_lifecycle_expiry",
bucket,
object,
version_id = ?version_id,
result = "failed",
error = %err,
"Heal storage lifecycle expiry check failed"
);
Ok(false)
}
}
}
async fn heal_bucket_metadata(&self, bucket: &str) -> Result<()> {
debug!(
target: "rustfs::heal::storage",
@@ -1436,7 +1576,7 @@ impl HealStorageAPI for ECStoreHealStorage {
loop {
let (page_objects, next_token, is_truncated) = self
.list_objects_for_heal_page(bucket, prefix, continuation_token.as_deref())
.list_objects_for_heal_page(bucket, prefix, continuation_token.as_deref(), false)
.await?;
all_objects.extend(page_objects);
@@ -1471,6 +1611,7 @@ impl HealStorageAPI for ECStoreHealStorage {
bucket: &str,
prefix: &str,
continuation_token: Option<&str>,
include_lifecycle_object_info: bool,
) -> Result<(Vec<HealListItem>, Option<String>, bool)> {
debug!(
target: "rustfs::heal::storage",
@@ -1522,10 +1663,19 @@ impl HealStorageAPI for ECStoreHealStorage {
let page_objects: Vec<HealListItem> = list_info
.objects
.into_iter()
.map(|obj| HealListItem {
name: obj.name,
version_id: obj.version_id.filter(|u| !u.is_nil()).map(|u| u.to_string()),
is_delete_marker: obj.delete_marker,
.map(|mut obj| {
obj.version_id = obj.version_id.filter(|u| !u.is_nil());
let version_id = obj.version_id.map(|u| u.to_string());
let mod_time_unix_nanos = obj.mod_time.map(|mod_time| mod_time.unix_timestamp_nanos());
let is_delete_marker = obj.delete_marker;
let lifecycle_object_info = include_lifecycle_object_info.then(|| obj.clone());
HealListItem {
name: obj.name,
version_id,
mod_time_unix_nanos,
lifecycle_object_info,
is_delete_marker,
}
})
.collect();
let page_count = page_objects.len();
@@ -1562,6 +1712,7 @@ impl HealStorageAPI for ECStoreHealStorage {
bucket: &str,
prefix: &str,
continuation_token: Option<&str>,
include_lifecycle_object_info: bool,
) -> Result<(Vec<HealListItem>, Option<String>, bool)> {
// Per-page bounds for the disk-walk union enumerator. Objects are atomic
// (never split across pages), so version_budget only bounds how many
@@ -1590,7 +1741,16 @@ impl HealStorageAPI for ECStoreHealStorage {
let (versions, next_forward, is_truncated) = self
.ecstore
.heal_walk_versions_page(pool_idx, set_idx, bucket, prefix, forward_to.as_deref(), BATCH_OBJECTS, VERSION_BUDGET)
.heal_walk_versions_page(
pool_idx,
set_idx,
bucket,
prefix,
forward_to.as_deref(),
BATCH_OBJECTS,
VERSION_BUDGET,
include_lifecycle_object_info,
)
.await
.map_err(|e| {
error!(
@@ -1614,6 +1774,8 @@ impl HealStorageAPI for ECStoreHealStorage {
.map(|v| HealListItem {
name: v.name,
version_id: v.version_id,
mod_time_unix_nanos: v.mod_time_unix_nanos,
lifecycle_object_info: v.lifecycle_object_info,
is_delete_marker: v.is_delete_marker,
})
.collect();
+9 -4
View File
@@ -12,7 +12,10 @@
// See the License for the specific language governing permissions and
// limitations under the License.
pub(crate) use rustfs_ecstore::api::data_usage::DATA_USAGE_CACHE_NAME as ECSTORE_DATA_USAGE_CACHE_NAME;
pub(crate) use rustfs_ecstore::api::data_usage::{
DATA_USAGE_CACHE_NAME as ECSTORE_DATA_USAGE_CACHE_NAME,
load_admin_data_usage_from_backend_cached as ecstore_load_admin_data_usage_from_backend_cached,
};
pub(crate) use rustfs_ecstore::api::disk::endpoint::Endpoint as EcstoreEndpoint;
pub(crate) use rustfs_ecstore::api::disk::error::{DiskError as EcstoreDiskError, Result as EcstoreDiskResult};
pub(crate) use rustfs_ecstore::api::disk::{
@@ -25,7 +28,9 @@ pub(crate) use rustfs_ecstore::api::disk::{
pub(crate) use rustfs_ecstore::api::disk::{DiskOption as EcstoreDiskOption, new_disk as ecstore_new_disk};
pub(crate) use rustfs_ecstore::api::error::{Error as EcstoreErrorType, StorageError as EcstoreStorageError};
pub(crate) use rustfs_ecstore::api::runtime::local_disk_map_read as ecstore_local_disk_map_read;
pub(crate) use rustfs_ecstore::api::storage::ECStore as EcstoreStore;
pub(crate) use rustfs_ecstore::api::storage::{
ECStore as EcstoreStore, HealLifecycleExpiryContext as EcstoreHealLifecycleExpiryContext,
};
use rustfs_storage_api as storage_contracts;
pub(crate) mod owner {
@@ -34,8 +39,8 @@ pub(crate) mod owner {
pub(crate) use super::{
ECSTORE_BUCKET_META_PREFIX, ECSTORE_DATA_USAGE_CACHE_NAME, ECSTORE_HEALING_MARKER_PATH, ECSTORE_RUSTFS_META_BUCKET,
EcstoreConditionalFileUpdate, EcstoreDeleteOptions, EcstoreDiskAPI, EcstoreDiskBytes, EcstoreDiskError,
EcstoreDiskResult, EcstoreDiskStore, EcstoreEndpoint, EcstoreErrorType, EcstoreStorageError, EcstoreStore,
ecstore_local_disk_map_read,
EcstoreDiskResult, EcstoreDiskStore, EcstoreEndpoint, EcstoreErrorType, EcstoreHealLifecycleExpiryContext,
EcstoreStorageError, EcstoreStore, ecstore_load_admin_data_usage_from_backend_cached, ecstore_local_disk_map_read,
};
#[cfg(test)]
+235 -3
View File
@@ -19,11 +19,12 @@ use crate::heal::{
resume::{
CheckpointManager, ReplacementPhase, ReplacementTargetIdentity, ResumeManager, replacement_target_identities_match,
},
storage::{HealStorageAPI, next_heal_listing_token},
storage::{HealBucketUsageBaseline, HealStorageAPI, next_heal_listing_token},
};
use crate::{Error, Result};
use metrics::{counter, histogram};
use rustfs_common::heal_channel::{HealOpts, HealRequestSource, HealScanMode};
use rustfs_common::trace_bus::{TraceEvent, TraceFunc, TraceKind, trace_emit};
use rustfs_madmin::heal_commands::HealResultItem;
use rustfs_utils::path::SLASH_SEPARATOR;
use serde::{Deserialize, Serialize};
@@ -178,6 +179,17 @@ pub enum HealPriority {
Urgent = 3,
}
impl HealPriority {
fn as_str(self) -> &'static str {
match self {
Self::Low => "low",
Self::Normal => "normal",
Self::High => "high",
Self::Urgent => "urgent",
}
}
}
/// Heal options
#[derive(Debug, Clone, Serialize, Deserialize)]
pub struct HealOptions {
@@ -498,6 +510,61 @@ impl HealTask {
}
}
fn emit_trace_task_state(&self, state: &'static str, duration: Duration, error: Option<&Error>) {
trace_emit(|| {
let mut event = TraceEvent::new(TraceKind::Heal, TraceFunc::HealTask)
.with_duration(duration)
.with_attr("task_id", self.id.as_str())
.with_attr("heal_type", self.heal_type.log_kind())
.with_attr("state", state)
.with_attr("source", self.source.as_str())
.with_attr("priority", self.priority.as_str())
.with_attr("retry_attempts", u64::from(self.retry_attempts))
.with_attr("dry_run", self.options.dry_run);
event = match &self.heal_type {
HealType::Cluster => event,
HealType::Object {
bucket,
object,
version_id,
} => {
let event = event.with_bucket(bucket.as_str()).with_object(object.as_str());
match version_id {
Some(version_id) => event.with_attr("version_id", version_id.as_str()),
None => event,
}
}
HealType::Bucket { bucket } => event.with_bucket(bucket.as_str()),
HealType::Prefix { bucket, prefix } => event.with_bucket(bucket.as_str()).with_object(prefix.as_str()),
HealType::ErasureSet { buckets, set_disk_id } => {
let bucket_count = u64::try_from(buckets.len()).unwrap_or(u64::MAX);
event
.with_attr("set_disk_id", set_disk_id.as_str())
.with_attr("bucket_count", bucket_count)
}
HealType::Metadata { bucket, object } => event.with_bucket(bucket.as_str()).with_object(object.as_str()),
HealType::ECDecode {
bucket,
object,
version_id,
} => {
let event = event.with_bucket(bucket.as_str()).with_object(object.as_str());
match version_id {
Some(version_id) => event.with_attr("version_id", version_id.as_str()),
None => event,
}
}
HealType::MRF { meta_path } => event.with_object(meta_path.as_str()),
};
match error {
Some(error) => event.with_attr("error", error.to_string()),
None => event,
}
});
}
async fn remaining_timeout(&self) -> Result<Option<Duration>> {
if let Some(total) = self.options.timeout {
let start_instant = { *self.task_start_instant.read().await };
@@ -717,6 +784,7 @@ impl HealTask {
queue_delay = ?queue_delay,
"Heal task started"
});
self.emit_trace_task_state("started", Duration::ZERO, None);
let result = match &self.heal_type {
HealType::Cluster => self.heal_cluster().await,
@@ -805,6 +873,14 @@ impl HealTask {
}
}
let terminal_state = match &result {
Ok(_) => "completed",
Err(Error::TaskCancelled) => "cancelled",
Err(Error::TaskTimeout) => "timed_out",
Err(_) => "failed",
};
self.emit_trace_task_state(terminal_state, start_instant.elapsed(), result.as_ref().err());
result
}
@@ -1535,7 +1611,7 @@ impl HealTask {
let (objects, next_token, is_truncated) = self
.await_with_control(
self.storage
.list_objects_for_heal_page(bucket, prefix, continuation_token.as_deref()),
.list_objects_for_heal_page(bucket, prefix, continuation_token.as_deref(), false),
)
.await?;
@@ -1697,6 +1773,23 @@ impl HealTask {
Ok(())
}
async fn apply_erasure_set_usage_baseline(&self, buckets: &[String]) -> Result<()> {
let baseline = match self
.await_with_control(self.storage.erasure_set_usage_baseline(buckets))
.await
{
Ok(Some(baseline)) => baseline,
Ok(None) => return Ok(()),
Err(err @ Error::TaskCancelled) | Err(err @ Error::TaskTimeout) => return Err(err),
Err(_) => return Ok(()),
};
let HealBucketUsageBaseline { objects_count, bytes } = baseline;
let mut progress = self.progress.write().await;
progress.set_total_baseline(objects_count, bytes);
Ok(())
}
async fn heal_metadata(&self, bucket: &str, object: &str) -> Result<()> {
debug!(
target: "rustfs::heal::task",
@@ -2298,6 +2391,8 @@ impl HealTask {
None
};
self.apply_erasure_set_usage_baseline(&buckets).await?;
let healing_marker = format!("{set_disk_id}:{}", self.id);
if let Some((disk, resume_manager, _)) = replacement_resume.as_ref() {
let state = resume_manager.get_state().await;
@@ -2602,7 +2697,8 @@ impl HealTask {
{
let mut progress = self.progress.write().await;
progress.update_progress(4, 4, 0, 0);
let bytes_processed = progress.bytes_processed;
progress.update_progress(4, 4, 0, bytes_processed);
}
match result {
@@ -2658,6 +2754,7 @@ mod tests {
use super::super::{DiskOption, DiskStore, Endpoint, HealDiskExt as _, new_disk};
use super::*;
use crate::heal::storage::{DiskStatus, HealListItem, HealObjectInfo};
use rustfs_common::trace_bus::{TraceEvent, TraceFunc, TraceKind, TraceSubscription, TraceVal, subscribe_trace_events};
use rustfs_madmin::heal_commands::{HealDriveInfo, HealResultItem, Infos};
use std::collections::{HashMap, VecDeque};
use std::sync::Mutex;
@@ -3203,6 +3300,8 @@ mod tests {
block_heal_object: Mutex<bool>,
resume_disk: Mutex<Option<DiskStore>>,
replacement_resume_disk: Mutex<Option<DiskStore>>,
usage_baseline: Mutex<Option<HealBucketUsageBaseline>>,
usage_baseline_error: Mutex<bool>,
}
#[test]
@@ -3265,11 +3364,69 @@ mod tests {
assert_eq!(samples_logged, MAX_BUCKET_FAILURE_LOG_SAMPLES);
}
#[tokio::test]
async fn execute_emits_heal_trace_task_state() {
let mut trace = subscribe_trace_events();
let storage = Arc::new(MockStorage::default());
let task = HealTask::from_request(
HealRequest::object("bucket-a".to_string(), "object-a".to_string(), Some("version-a".to_string())),
storage,
);
task.execute().await.expect("mock object heal should complete");
let started = recv_trace_task_state(&mut trace, &task.id, "started").await;
assert_eq!(started.kind, TraceKind::Heal);
assert_eq!(started.func, TraceFunc::HealTask);
assert_eq!(started.bucket.as_deref(), Some("bucket-a"));
assert_eq!(started.object.as_deref(), Some("object-a"));
assert_eq!(trace_attr_string(&started, "heal_type").as_deref(), Some("object"));
assert_eq!(trace_attr_string(&started, "source").as_deref(), Some("internal"));
assert_eq!(trace_attr_string(&started, "version_id").as_deref(), Some("version-a"));
let completed = recv_trace_task_state(&mut trace, &task.id, "completed").await;
assert_eq!(completed.kind, TraceKind::Heal);
assert_eq!(completed.func, TraceFunc::HealTask);
assert_eq!(trace_attr_string(&completed, "state").as_deref(), Some("completed"));
}
async fn recv_trace_task_state(trace: &mut TraceSubscription, task_id: &str, state: &str) -> TraceEvent {
for _ in 0..32 {
let event = tokio::time::timeout(Duration::from_secs(1), trace.recv())
.await
.expect("trace event should arrive")
.expect("trace bus should stay open");
if trace_attr_string(&event, "task_id").as_deref() == Some(task_id)
&& trace_attr_string(&event, "state").as_deref() == Some(state)
{
return (*event).clone();
}
}
panic!("expected trace state {state} for task {task_id}");
}
fn trace_attr_string(event: &TraceEvent, key: &str) -> Option<String> {
event.attrs.iter().find_map(|attr| {
if attr.key != key {
return None;
}
Some(match &attr.value {
TraceVal::Bool(value) => value.to_string(),
TraceVal::U64(value) => value.to_string(),
TraceVal::I64(value) => value.to_string(),
TraceVal::Str(value) => value.to_string(),
})
})
}
/// Build a latest, non-delete-marker heal list item with no version id.
fn heal_item(name: &str) -> HealListItem {
HealListItem {
name: name.to_string(),
version_id: None,
mod_time_unix_nanos: None,
lifecycle_object_info: None,
is_delete_marker: false,
}
}
@@ -3357,6 +3514,13 @@ mod tests {
}))
}
async fn erasure_set_usage_baseline(&self, _buckets: &[String]) -> Result<Option<HealBucketUsageBaseline>> {
if *self.usage_baseline_error.lock().unwrap() {
return Err(Error::Other("usage baseline unavailable".to_string()));
}
Ok(*self.usage_baseline.lock().unwrap())
}
async fn heal_bucket_metadata(&self, _bucket: &str) -> Result<()> {
Ok(())
}
@@ -3540,6 +3704,7 @@ mod tests {
bucket: &str,
prefix: &str,
continuation_token: Option<&str>,
_include_lifecycle_object_info: bool,
) -> Result<(Vec<HealListItem>, Option<String>, bool)> {
self.listed_prefixes.lock().unwrap().push(prefix.to_string());
if *self.truncate_without_token.lock().unwrap() {
@@ -4654,6 +4819,73 @@ mod tests {
assert!(storage.object_heal_opts.lock().unwrap().is_empty());
}
#[tokio::test]
async fn erasure_set_heal_applies_usage_baseline_to_progress() {
let temp = TempDir::new().expect("temporary directory should be created");
let disk = make_resume_disk(&temp).await;
let storage = Arc::new(MockStorage {
resume_disk: Mutex::new(Some(disk)),
usage_baseline: Mutex::new(Some(HealBucketUsageBaseline {
objects_count: 10,
bytes: 8,
})),
..Default::default()
});
let request = HealRequest::new(
HealType::ErasureSet {
buckets: vec!["bucket-a".to_string()],
set_disk_id: "pool_0_set_0".to_string(),
},
HealOptions {
timeout: None,
..Default::default()
},
HealPriority::Normal,
);
let task = HealTask::from_request(request, storage);
task.heal_erasure_set(vec!["bucket-a".to_string()], "pool_0_set_0".to_string())
.await
.expect("erasure set heal should complete");
let progress = task.get_progress().await;
assert_eq!(progress.objects_total_count, 10);
assert_eq!(progress.objects_total_size, 8);
assert_eq!(progress.bytes_processed, 2);
assert!((progress.progress_percentage - 25.0).abs() < 0.001);
}
#[tokio::test]
async fn erasure_set_heal_ignores_usage_baseline_errors() {
let temp = TempDir::new().expect("temporary directory should be created");
let disk = make_resume_disk(&temp).await;
let storage = Arc::new(MockStorage {
resume_disk: Mutex::new(Some(disk)),
usage_baseline_error: Mutex::new(true),
..Default::default()
});
let request = HealRequest::new(
HealType::ErasureSet {
buckets: vec!["bucket-a".to_string()],
set_disk_id: "pool_0_set_0".to_string(),
},
HealOptions {
timeout: None,
..Default::default()
},
HealPriority::Normal,
);
let task = HealTask::from_request(request, storage);
task.heal_erasure_set(vec!["bucket-a".to_string()], "pool_0_set_0".to_string())
.await
.expect("usage baseline failures should not fail erasure set heal");
let progress = task.get_progress().await;
assert_eq!(progress.objects_total_count, 0);
assert_eq!(progress.objects_total_size, 0);
}
#[tokio::test]
async fn resumable_erasure_set_execution_is_cancelled_while_object_heal_is_pending() {
let temp = TempDir::new().expect("temporary directory should be created");
+1
View File
@@ -445,6 +445,7 @@ mod tests {
_bucket: &str,
_prefix: &str,
_continuation_token: Option<&str>,
_include_lifecycle_object_info: bool,
) -> Result<(Vec<HealListItem>, Option<String>, bool), Error> {
Ok((Vec::new(), None, false))
}
@@ -176,7 +176,7 @@ async fn enumerate_all_versions(heal_storage: &Arc<ECStoreHealStorage>, bucket:
let mut token: Option<String> = None;
loop {
let (page, next, truncated) = heal_storage
.list_objects_for_heal_page(bucket, "", token.as_deref())
.list_objects_for_heal_page(bucket, "", token.as_deref(), false)
.await
.expect("list_objects_for_heal_page failed");
items.extend(page);
@@ -166,7 +166,7 @@ async fn enumerate_b5(heal_storage: &Arc<ECStoreHealStorage>, bucket: &str) -> V
let mut token: Option<String> = None;
loop {
let (page, next, truncated) = heal_storage
.list_objects_for_heal_page(bucket, "", token.as_deref())
.list_objects_for_heal_page(bucket, "", token.as_deref(), false)
.await
.expect("b5 list page failed");
items.extend(page);
@@ -187,7 +187,7 @@ async fn enumerate_disk_walk(heal_storage: &Arc<ECStoreHealStorage>, bucket: &st
let mut token: Option<String> = None;
loop {
let (page, next, truncated) = heal_storage
.list_versions_for_heal_page_disk_walk(SET_DISK_ID, bucket, "", token.as_deref())
.list_versions_for_heal_page_disk_walk(SET_DISK_ID, bucket, "", token.as_deref(), false)
.await
.expect("disk-walk list page failed");
items.extend(page);
@@ -418,7 +418,7 @@ mod serial_tests {
let mut pages = 0usize;
loop {
let (versions, next_forward, truncated) = ecstore
.heal_walk_versions_page(0, 0, bucket, "", forward.as_deref(), 2, 100_000)
.heal_walk_versions_page(0, 0, bucket, "", forward.as_deref(), 2, 100_000, false)
.await
.expect("heal_walk_versions_page failed");
pages += 1;
+2
View File
@@ -242,6 +242,7 @@ fn test_heal_task_status_atomic_update() {
_bucket: &str,
_prefix: &str,
_continuation_token: Option<&str>,
_include_lifecycle_object_info: bool,
) -> rustfs_heal::Result<(Vec<HealListItem>, Option<String>, bool)> {
Ok((vec![], None, false))
}
@@ -385,6 +386,7 @@ async fn test_heal_task_transient_object_exists_skip_avoids_recreate() {
_bucket: &str,
_prefix: &str,
_continuation_token: Option<&str>,
_include_lifecycle_object_info: bool,
) -> rustfs_heal::Result<(Vec<HealListItem>, Option<String>, bool)> {
Ok((Vec::new(), None, false))
}