mirror of
https://github.com/rustfs/rustfs.git
synced 2026-08-21 20:06:37 +00:00
feat(obs): add bounded metrics dimensions (#5645)
* feat(obs): add drive topology detail metrics Expose additive drive info, topology, state, and per-drive API metrics while preserving the existing drive metric label sets. Backlog: rustfs/backlog#1655 Co-Authored-By: heihutu <heihutu@gmail.com> * fix(obs): preserve suspect drive runtime state Keep suspect as a bounded drive runtime state and avoid all-zero runtime_state samples for that storage health state. Co-Authored-By: heihutu <heihutu@gmail.com> * fix(obs): skip unknown drive inode samples Avoid exporting zero inode gauges for missing or stale drive snapshots and ignore zero-count API latency buckets. Co-Authored-By: heihutu <heihutu@gmail.com> * feat(obs): add scanner source work detail metrics Expose additive scanner source and cycle work metrics with bounded server/source/state labels while leaving the existing aggregate scanner metrics unchanged. Co-Authored-By: heihutu <heihutu@gmail.com> * feat(obs): add ilm action detail metrics Expose additive ILM action/state task metrics with a server label while preserving the existing aggregate ILM series. Co-Authored-By: heihutu <heihutu@gmail.com> * feat(obs): add delivery target server metrics Expose additive audit and notification delivery target metrics with server labels and extend removed-target tombstones for the server-aware series. Co-Authored-By: heihutu <heihutu@gmail.com> * feat(obs): add replication target flow metrics Expose additive bucket replication target sent and failed-flow metrics while preserving existing bucket aggregates and target backlog series. Co-Authored-By: heihutu <heihutu@gmail.com> * feat(obs): add request server metrics Expose additive API request metrics with server labels while preserving the existing request and traffic metric label sets. Co-Authored-By: heihutu <heihutu@gmail.com> * style(obs): apply rustfmt to metrics changes Apply rustfmt output to the metrics dimension changes without altering behavior. Co-Authored-By: heihutu <heihutu@gmail.com> * style(obs): reuse audit target label constant Use the exported audit target_id label constant for legacy audit target metrics. Co-Authored-By: heihutu <heihutu@gmail.com> * feat(obs): populate drive disk metrics Co-Authored-By: heihutu <heihutu@gmail.com> * feat(obs): add scanner bucket drive result metrics Co-Authored-By: heihutu <heihutu@gmail.com> * feat(obs): add replication proxy server metrics Co-Authored-By: heihutu <heihutu@gmail.com> * fix(obs): address metric liveness review Use checked division for drive API latency aggregation and keep recovered drive, scanner current-cycle, replication flow, audit target, and notification target series from retaining stale values. Co-Authored-By: heihutu <heihutu@gmail.com> * fix(obs): address metric dimension review Co-Authored-By: heihutu <heihutu@gmail.com> * fix(obs): address additional metric review Co-Authored-By: heihutu <heihutu@gmail.com> * fix(obs): count drive calls at start Co-Authored-By: heihutu <heihutu@gmail.com> * fix(obs): address metrics dimension review Co-Authored-By: heihutu <heihutu@gmail.com> * fix(metrics): address dimension review gaps Co-Authored-By: heihutu <heihutu@gmail.com> * fix(metrics): address scanner review follow-ups Co-Authored-By: heihutu <heihutu@gmail.com> * fix(metrics): address runtime review follow-ups Co-Authored-By: heihutu <heihutu@gmail.com> * fix(metrics): reduce disk metric contention Co-Authored-By: heihutu <heihutu@gmail.com> * fix(metrics): address runtime review follow-ups Co-Authored-By: heihutu <heihutu@gmail.com> * fix(metrics): retire stale dimension series Co-Authored-By: heihutu <heihutu@gmail.com> --------- Co-authored-by: heihutu <heihutu@gmail.com>
This commit is contained in:
@@ -322,7 +322,16 @@ impl SetDisks {
|
||||
pub async fn renew_disk(&self, ep: &Endpoint) {
|
||||
debug!("renew_disk: start {:?}", ep);
|
||||
|
||||
let (new_disk, fm) = match Self::connect_endpoint(ep).await {
|
||||
let previous_health = {
|
||||
let disks = self.disks.read().await;
|
||||
disks
|
||||
.iter()
|
||||
.filter_map(|disk| disk.as_ref())
|
||||
.find(|disk| disk.endpoint() == *ep)
|
||||
.and_then(|disk| disk.local_health_tracker_epoch_for_reconnect())
|
||||
};
|
||||
|
||||
let (new_disk, fm) = match Self::connect_endpoint(ep, previous_health).await {
|
||||
Ok(res) => res,
|
||||
Err(e) => {
|
||||
warn!("renew_disk: connect_endpoint err {:?}", &e);
|
||||
@@ -400,13 +409,17 @@ impl SetDisks {
|
||||
Err(Error::other("DriveID: not found"))
|
||||
}
|
||||
|
||||
pub(in crate::set_disk) async fn connect_endpoint(ep: &Endpoint) -> disk::error::Result<(DiskStore, FormatV3)> {
|
||||
let disk = new_disk(
|
||||
pub(in crate::set_disk) async fn connect_endpoint(
|
||||
ep: &Endpoint,
|
||||
reconnect: Option<disk::disk_store::ReconnectDiskHealthState>,
|
||||
) -> disk::error::Result<(DiskStore, FormatV3)> {
|
||||
let disk = crate::disk::new_disk_with_health_tracker(
|
||||
ep,
|
||||
&DiskOption {
|
||||
cleanup: false,
|
||||
health_check: true,
|
||||
},
|
||||
reconnect,
|
||||
)
|
||||
.await?;
|
||||
|
||||
@@ -714,6 +727,25 @@ mod tests {
|
||||
renewed_disk.health_check_enabled_for_test(),
|
||||
"renewed disks must keep health monitoring enabled so later faulty marks can recover"
|
||||
);
|
||||
renewed_disk
|
||||
.disk_info(&DiskInfoOptions::default())
|
||||
.await
|
||||
.expect("renewed disk_info should record a drive API metric");
|
||||
renewed_disk.force_runtime_state_for_test(disk::health_state::RuntimeDriveHealthState::Offline);
|
||||
|
||||
set_disks.renew_disk(&endpoints[0]).await;
|
||||
|
||||
let disks = set_disks.get_disks_internal().await;
|
||||
let renewed_again = disks[0]
|
||||
.as_ref()
|
||||
.expect("second renew_disk should keep the recovered disk attached");
|
||||
assert_eq!(
|
||||
renewed_again
|
||||
.metrics_snapshot()
|
||||
.and_then(|metrics| metrics.api_calls.get("disk_info").copied()),
|
||||
Some(1),
|
||||
"disk reconnect must preserve the local drive metrics tracker epoch"
|
||||
);
|
||||
|
||||
drop(temp_dirs);
|
||||
}
|
||||
|
||||
Reference in New Issue
Block a user