mirror of
https://github.com/rustfs/rustfs.git
synced 2026-08-19 19:16:17 +00:00
feat(heal): add progress and trace observability (#6179)
* feat(heal): track erasure set progress baseline Record erasure-set heal byte progress from per-object results and seed progress totals from complete usage-cache snapshots when available. Keep usage-cache failures observational so heal execution continues without a baseline. Co-Authored-By: heihutu <heihutu@gmail.com> * feat(heal): skip filtered erasure set versions Skip erasure-set versions written after the durable heal start time, and queue lifecycle-expired versions for expiry before skipping them. Track new-version and ILM-expired skips separately so progress can explain completed baseline work without treating these skips as retry-blocking failures. Co-Authored-By: heihutu <heihutu@gmail.com> * feat(heal): wire abandoned data-dir cleanup check Connect check_abandoned_parts through ECStore, pool, and set layers so heal can invoke the existing orphan data-dir reclaim path instead of returning NotImplemented. Add dry-run support to the reclaim scan and cover dry-run plus scoped set behavior with regression tests. Co-Authored-By: heihutu <heihutu@gmail.com> * feat(obs): add heal scanner trace bus Introduce an in-process broadcast trace bus with typed heal and scanner events, lazy event construction, and bounded lagged-subscriber behavior. Cover zero-subscriber publishing, subscription delivery, drop accounting, and lagged receivers with focused common-crate tests. Co-Authored-By: heihutu <heihutu@gmail.com> * feat(obs): stream heal trace events from admin API Wire the admin trace endpoint to the common trace bus for heal/scanner events, including kind, regex, and threshold filtering. Co-Authored-By: heihutu <heihutu@gmail.com> * feat(obs): emit heal trace events Publish heal task lifecycle and abandoned-parts cleanup events through the common trace bus so the admin trace stream has live heal diagnostics. Co-Authored-By: heihutu <heihutu@gmail.com> * feat(obs): emit scanner trace events Publish scanner folder, lifecycle action, and heal-candidate events through the common trace bus for live admin scanner diagnostics. Co-Authored-By: heihutu <heihutu@gmail.com> * fix(heal): route data usage loader through storage api Keep ECStore data-usage facade access behind the heal storage_api boundary so architecture migration guards can validate the heal progress path. Co-Authored-By: heihutu <heihutu@gmail.com> * perf(heal): avoid lifecycle snapshots on ordinary heal pages Only request lifecycle object snapshots when the heal pass has lifecycle expiry context. This keeps ordinary listing and disk-walk pages from cloning FileInfo/ObjectInfo payloads while preserving the skip path that queues expired versions. Co-Authored-By: heihutu <heihutu@gmail.com> * test(heal): update bug-fix mocks for lifecycle snapshots Carry the lifecycle snapshot opt-in argument through the remaining heal bug-fix test mocks so all-targets clippy covers the updated storage trait. Co-Authored-By: heihutu <heihutu@gmail.com> * test(rustfs): sync heal storage mock signature Update the rustfs storage RPC test mock for the lifecycle snapshot opt-in argument and cover it with rustfs all-targets clippy. Co-Authored-By: heihutu <heihutu@gmail.com> * test(e2e): allocate smoke ports across nextest processes Serialize E2E port selection with a small /tmp allocator so nextest workers do not reuse the same just-released ephemeral port before RustFS binds it. Co-Authored-By: heihutu <heihutu@gmail.com> --------- Co-authored-by: heihutu <heihutu@gmail.com>
This commit is contained in:
@@ -6998,6 +6998,100 @@ mod tests {
|
||||
assert!(object_dir.join(STORAGE_FORMAT_FILE).exists(), "metadata must be preserved");
|
||||
}
|
||||
|
||||
async fn recv_abandoned_parts_trace(
|
||||
trace: &mut rustfs_common::trace_bus::TraceSubscription,
|
||||
bucket: &str,
|
||||
object: &str,
|
||||
state: &str,
|
||||
) -> rustfs_common::trace_bus::TraceEvent {
|
||||
for _ in 0..32 {
|
||||
let event = tokio::time::timeout(std::time::Duration::from_secs(1), trace.recv())
|
||||
.await
|
||||
.expect("abandoned-parts trace event should arrive")
|
||||
.expect("trace bus should stay open");
|
||||
if event.kind == rustfs_common::trace_bus::TraceKind::Heal
|
||||
&& event.func == rustfs_common::trace_bus::TraceFunc::HealCheckAbandonedParts
|
||||
&& event.bucket.as_deref() == Some(bucket)
|
||||
&& event.object.as_deref() == Some(object)
|
||||
&& trace_attr_string(&event, "state").as_deref() == Some(state)
|
||||
{
|
||||
return (*event).clone();
|
||||
}
|
||||
}
|
||||
|
||||
panic!("expected abandoned-parts trace state {state} for {bucket}/{object}");
|
||||
}
|
||||
|
||||
fn trace_attr_string(event: &rustfs_common::trace_bus::TraceEvent, key: &str) -> Option<String> {
|
||||
event.attrs.iter().find_map(|attr| {
|
||||
if attr.key != key {
|
||||
return None;
|
||||
}
|
||||
Some(match &attr.value {
|
||||
rustfs_common::trace_bus::TraceVal::Bool(value) => value.to_string(),
|
||||
rustfs_common::trace_bus::TraceVal::U64(value) => value.to_string(),
|
||||
rustfs_common::trace_bus::TraceVal::I64(value) => value.to_string(),
|
||||
rustfs_common::trace_bus::TraceVal::Str(value) => value.to_string(),
|
||||
})
|
||||
})
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
async fn check_abandoned_parts_dry_run_counts_without_deleting() {
|
||||
let mut trace = rustfs_common::trace_bus::subscribe_trace_events();
|
||||
let (dir, disk) = make_single_local_disk().await;
|
||||
let live = Uuid::new_v4();
|
||||
let orphan = Uuid::new_v4();
|
||||
|
||||
let object_dir = dir.path().join("bucket").join("obj");
|
||||
write_object_meta_with_data_dirs(&object_dir, "bucket", "obj", &[live]).await;
|
||||
fs::create_dir_all(object_dir.join(live.to_string()))
|
||||
.await
|
||||
.expect("live data dir should be created");
|
||||
fs::create_dir_all(object_dir.join(orphan.to_string()))
|
||||
.await
|
||||
.expect("orphan data dir should be created");
|
||||
|
||||
let set = make_set_disks_with(vec![Some(disk)]).await;
|
||||
set.check_abandoned_parts(
|
||||
"bucket",
|
||||
"obj",
|
||||
&HealOpts {
|
||||
dry_run: true,
|
||||
no_lock: true,
|
||||
..Default::default()
|
||||
},
|
||||
)
|
||||
.await
|
||||
.expect("dry-run abandoned-parts check should succeed");
|
||||
let dry_run_trace = recv_abandoned_parts_trace(&mut trace, "bucket", "obj", "dry_run_matched").await;
|
||||
assert_eq!(trace_attr_string(&dry_run_trace, "dry_run").as_deref(), Some("true"));
|
||||
assert_eq!(trace_attr_string(&dry_run_trace, "data_dirs").as_deref(), Some("1"));
|
||||
|
||||
assert!(object_dir.join(live.to_string()).exists(), "referenced data dir must be preserved");
|
||||
assert!(object_dir.join(orphan.to_string()).exists(), "dry-run must not remove orphaned data dir");
|
||||
|
||||
set.check_abandoned_parts(
|
||||
"bucket",
|
||||
"obj",
|
||||
&HealOpts {
|
||||
no_lock: true,
|
||||
..Default::default()
|
||||
},
|
||||
)
|
||||
.await
|
||||
.expect("abandoned-parts check should reclaim stale data dir");
|
||||
let reclaim_trace = recv_abandoned_parts_trace(&mut trace, "bucket", "obj", "reclaimed").await;
|
||||
assert_eq!(trace_attr_string(&reclaim_trace, "dry_run").as_deref(), Some("false"));
|
||||
assert_eq!(trace_attr_string(&reclaim_trace, "data_dirs").as_deref(), Some("1"));
|
||||
|
||||
assert!(
|
||||
object_dir.join(live.to_string()).exists(),
|
||||
"referenced data dir must remain after reclaim"
|
||||
);
|
||||
assert!(!object_dir.join(orphan.to_string()).exists(), "orphaned data dir must be removed");
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
async fn reclaim_orphan_data_dirs_recovers_deferred_cleanup_after_restart() {
|
||||
let (dir, disk) = make_single_local_disk().await;
|
||||
@@ -12233,11 +12327,18 @@ mod tests {
|
||||
.expect_err("unsupported copy_object_part should return a typed error");
|
||||
assert!(matches!(copy_part_err, StorageError::NotImplemented));
|
||||
|
||||
let abandoned_err = set_disks
|
||||
.check_abandoned_parts("bucket", "object", &HealOpts::default())
|
||||
set_disks
|
||||
.check_abandoned_parts(
|
||||
"bucket",
|
||||
"object",
|
||||
&HealOpts {
|
||||
dry_run: true,
|
||||
no_lock: true,
|
||||
..Default::default()
|
||||
},
|
||||
)
|
||||
.await
|
||||
.expect_err("abandoned-parts check should stay in the upper reconciliation layer");
|
||||
assert!(matches!(abandoned_err, StorageError::NotImplemented));
|
||||
.expect("abandoned-parts check should be callable on empty disk sets");
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
|
||||
Reference in New Issue
Block a user