feat(heal): add progress and trace observability (#6179)

* feat(heal): track erasure set progress baseline

Record erasure-set heal byte progress from per-object results and seed progress totals from complete usage-cache snapshots when available.

Keep usage-cache failures observational so heal execution continues without a baseline.

Co-Authored-By: heihutu <heihutu@gmail.com>

* feat(heal): skip filtered erasure set versions

Skip erasure-set versions written after the durable heal start time, and queue lifecycle-expired versions for expiry before skipping them.

Track new-version and ILM-expired skips separately so progress can explain completed baseline work without treating these skips as retry-blocking failures.

Co-Authored-By: heihutu <heihutu@gmail.com>

* feat(heal): wire abandoned data-dir cleanup check

Connect check_abandoned_parts through ECStore, pool, and set layers so heal can invoke the existing orphan data-dir reclaim path instead of returning NotImplemented.

Add dry-run support to the reclaim scan and cover dry-run plus scoped set behavior with regression tests.

Co-Authored-By: heihutu <heihutu@gmail.com>

* feat(obs): add heal scanner trace bus

Introduce an in-process broadcast trace bus with typed heal and scanner events, lazy event construction, and bounded lagged-subscriber behavior.

Cover zero-subscriber publishing, subscription delivery, drop accounting, and lagged receivers with focused common-crate tests.

Co-Authored-By: heihutu <heihutu@gmail.com>

* feat(obs): stream heal trace events from admin API

Wire the admin trace endpoint to the common trace bus for heal/scanner events, including kind, regex, and threshold filtering.

Co-Authored-By: heihutu <heihutu@gmail.com>

* feat(obs): emit heal trace events

Publish heal task lifecycle and abandoned-parts cleanup events through the common trace bus so the admin trace stream has live heal diagnostics.

Co-Authored-By: heihutu <heihutu@gmail.com>

* feat(obs): emit scanner trace events

Publish scanner folder, lifecycle action, and heal-candidate events through the common trace bus for live admin scanner diagnostics.

Co-Authored-By: heihutu <heihutu@gmail.com>

* fix(heal): route data usage loader through storage api

Keep ECStore data-usage facade access behind the heal storage_api boundary so architecture migration guards can validate the heal progress path.

Co-Authored-By: heihutu <heihutu@gmail.com>

* perf(heal): avoid lifecycle snapshots on ordinary heal pages

Only request lifecycle object snapshots when the heal pass has lifecycle expiry context. This keeps ordinary listing and disk-walk pages from cloning FileInfo/ObjectInfo payloads while preserving the skip path that queues expired versions.

Co-Authored-By: heihutu <heihutu@gmail.com>

* test(heal): update bug-fix mocks for lifecycle snapshots

Carry the lifecycle snapshot opt-in argument through the remaining heal bug-fix test mocks so all-targets clippy covers the updated storage trait.

Co-Authored-By: heihutu <heihutu@gmail.com>

* test(rustfs): sync heal storage mock signature

Update the rustfs storage RPC test mock for the lifecycle snapshot opt-in argument and cover it with rustfs all-targets clippy.

Co-Authored-By: heihutu <heihutu@gmail.com>

* test(e2e): allocate smoke ports across nextest processes

Serialize E2E port selection with a small /tmp allocator so nextest workers do not reuse the same just-released ephemeral port before RustFS binds it.

Co-Authored-By: heihutu <heihutu@gmail.com>

---------

Co-authored-by: heihutu <heihutu@gmail.com>
This commit is contained in:
houseme
2026-08-18 08:29:29 +08:00
committed by GitHub
parent 7cb91a0190
commit 360bceafce
30 changed files with 2383 additions and 112 deletions
+56 -5
View File
@@ -16,6 +16,7 @@ use super::super::*;
use crate::disk::disk_store::DiskStoreRenameDataExt;
use crate::io_support::bitrot::object_mmap_read_enabled;
use crate::storage_api_contracts::namespace::NamespaceLocking as _;
use rustfs_common::trace_bus::{TraceEvent, TraceFunc, TraceKind, trace_emit};
use tracing::trace;
const LOG_COMPONENT_ECSTORE: &str = "ecstore";
@@ -2057,11 +2058,61 @@ impl crate::storage_api_contracts::heal::HealOperations for SetDisks {
Err(Error::DiskNotFound)
}
#[tracing::instrument(skip(self))]
async fn check_abandoned_parts(&self, _bucket: &str, _object: &str, _opts: &HealOpts) -> Result<()> {
// Multipart orphan reconciliation is intentionally retained above the set layer
// until there is a concrete caller and a stable lower-level contract to implement.
Err(StorageError::NotImplemented)
#[tracing::instrument(level = "debug", skip(self, opts), fields(bucket = %bucket, object = %object, dry_run = opts.dry_run))]
async fn check_abandoned_parts(&self, bucket: &str, object: &str, opts: &HealOpts) -> Result<()> {
let started_at = std::time::Instant::now();
let _write_lock_guard = if !opts.no_lock {
let ns_lock = self.new_ns_lock(bucket, object).await?;
Some(
ns_lock
.get_write_lock(get_lock_acquire_timeout())
.await
.map_err(|e| self.map_namespace_lock_error(bucket, object, "write", e))?,
)
} else {
None
};
let removed = if opts.dry_run {
self.dry_run_reclaim_orphan_data_dirs(bucket, object).await?
} else {
self.reclaim_orphan_data_dirs(bucket, object).await?
};
let state = if opts.dry_run && removed > 0 {
"dry_run_matched"
} else if removed > 0 {
"reclaimed"
} else {
"checked"
};
let data_dirs = u64::try_from(removed).unwrap_or(u64::MAX);
trace_emit(|| {
TraceEvent::new(TraceKind::Heal, TraceFunc::HealCheckAbandonedParts)
.with_bucket(bucket)
.with_object(object)
.with_duration(started_at.elapsed())
.with_attr("state", state)
.with_attr("dry_run", opts.dry_run)
.with_attr("data_dirs", data_dirs)
});
if removed > 0 {
trace!(
event = "heal_abandoned_parts",
component = LOG_COMPONENT_ECSTORE,
subsystem = LOG_SUBSYSTEM_HEAL,
state = if opts.dry_run { "dry_run_matched" } else { "reclaimed" },
result = "ok",
bucket,
object,
dry_run = opts.dry_run,
data_dirs = removed,
"Heal abandoned parts checked object data directories"
);
}
Ok(())
}
}
+46 -4
View File
@@ -23,6 +23,7 @@
//! per-version `SetDisks::heal_object`.
use super::super::*;
use crate::object_api::ObjectInfo;
use std::collections::HashSet;
use std::sync::Mutex;
use std::sync::atomic::{AtomicBool, AtomicUsize, Ordering};
@@ -39,12 +40,16 @@ const BACKGROUND_WALKDIR_STALL_TIMEOUT: Duration = Duration::from_secs(60);
/// it must not gate healing logic — the delete-marker vs data path is chosen
/// inside `ops/heal.rs` from the resolved latest metadata. `version_id` is
/// normalized (nil/absent UUID => `None`).
#[derive(Debug, Clone, PartialEq, Eq)]
#[derive(Debug, Clone)]
pub struct HealWalkVersion {
/// object key
pub name: String,
/// normalized version id (`None` when the version is nil/absent)
pub version_id: Option<String>,
/// version modification time as Unix nanoseconds
pub mod_time_unix_nanos: Option<i128>,
/// object snapshot for lifecycle evaluation
pub lifecycle_object_info: Option<ObjectInfo>,
/// whether this version is a delete marker (observability only)
pub is_delete_marker: bool,
}
@@ -63,6 +68,7 @@ struct HealWalkCollector {
bucket: String,
batch_objects: usize,
version_budget: usize,
include_lifecycle_object_info: bool,
objects: Mutex<Vec<HealWalkObject>>,
decode_error: Mutex<Option<DiskError>>,
version_total: AtomicUsize,
@@ -116,10 +122,25 @@ impl HealWalkCollector {
let mut versions = Vec::with_capacity(fiv.versions.len() + fiv.free_versions.len());
for fi in fiv.versions.iter().chain(fiv.free_versions.iter()) {
let version_uuid = fi.version_id.filter(|version_id| !version_id.is_nil());
let lifecycle_object_info = if self.include_lifecycle_object_info {
let mut lifecycle_fi = fi.clone();
lifecycle_fi.version_id = version_uuid;
Some(ObjectInfo::from_file_info(
&lifecycle_fi,
&self.bucket,
&entry.name,
version_uuid.is_some(),
))
} else {
None
};
versions.push(HealWalkVersion {
name: entry.name.clone(),
// Normalize: nil/absent version id => None.
version_id: fi.version_id.filter(|u| !u.is_nil()).map(|u| u.to_string()),
version_id: version_uuid.map(|u| u.to_string()),
mod_time_unix_nanos: fi.mod_time.map(|mod_time| mod_time.unix_timestamp_nanos()),
lifecycle_object_info,
is_delete_marker: fi.deleted,
});
}
@@ -173,11 +194,26 @@ impl HealWalkCollector {
}
};
for fi in fiv.versions.iter().chain(fiv.free_versions.iter()) {
let vid = fi.version_id.filter(|u| !u.is_nil()).map(|u| u.to_string());
let version_uuid = fi.version_id.filter(|version_id| !version_id.is_nil());
let vid = version_uuid.map(|u| u.to_string());
if seen.insert(vid.clone()) {
let lifecycle_object_info = if self.include_lifecycle_object_info {
let mut lifecycle_fi = fi.clone();
lifecycle_fi.version_id = version_uuid;
Some(ObjectInfo::from_file_info(
&lifecycle_fi,
&self.bucket,
&entry.name,
version_uuid.is_some(),
))
} else {
None
};
versions.push(HealWalkVersion {
name: entry.name.clone(),
version_id: vid,
mod_time_unix_nanos: fi.mod_time.map(|mod_time| mod_time.unix_timestamp_nanos()),
lifecycle_object_info,
is_delete_marker: fi.deleted,
});
}
@@ -255,6 +291,7 @@ impl SetDisks {
forward_to: Option<&str>,
batch_objects: usize,
version_budget: usize,
include_lifecycle_object_info: bool,
) -> disk::error::Result<(Vec<HealWalkVersion>, Option<String>, bool)> {
assert!(batch_objects >= 2, "heal_walk_versions_page requires batch_objects >= 2");
@@ -264,6 +301,7 @@ impl SetDisks {
bucket: bucket.to_string(),
batch_objects,
version_budget: version_budget.max(1),
include_lifecycle_object_info,
objects: Mutex::new(Vec::new()),
decode_error: Mutex::new(None),
version_total: AtomicUsize::new(0),
@@ -347,6 +385,7 @@ mod tests {
bucket: "bucket".to_string(),
batch_objects: 2,
version_budget: 2,
include_lifecycle_object_info: false,
objects: Mutex::new(Vec::new()),
decode_error: Mutex::new(None),
version_total: AtomicUsize::new(0),
@@ -388,6 +427,8 @@ mod tests {
HealWalkVersion {
name: name.to_string(),
version_id: Some(id.to_string()),
mod_time_unix_nanos: None,
lifecycle_object_info: None,
is_delete_marker: dm,
}
}
@@ -491,6 +532,7 @@ mod tests {
bucket: "bucket".to_string(),
batch_objects: 1000,
version_budget: 10_000,
include_lifecycle_object_info: false,
objects: Mutex::new(Vec::new()),
version_total: AtomicUsize::new(0),
decode_error: Mutex::new(None),
@@ -567,7 +609,7 @@ mod tests {
.expect("corrupt test metadata should be written");
let error = set_disks
.heal_walk_versions_page(bucket, "", None, 2, 2)
.heal_walk_versions_page(bucket, "", None, 2, 2, false)
.await
.expect_err("semantic metadata corruption must fail the heal disk walk");