feat(obs): add bounded metrics dimensions (#5645)

* feat(obs): add drive topology detail metrics

Expose additive drive info, topology, state, and per-drive API metrics while preserving the existing drive metric label sets.

Backlog: rustfs/backlog#1655

Co-Authored-By: heihutu <heihutu@gmail.com>

* fix(obs): preserve suspect drive runtime state

Keep suspect as a bounded drive runtime state and avoid all-zero runtime_state samples for that storage health state.

Co-Authored-By: heihutu <heihutu@gmail.com>

* fix(obs): skip unknown drive inode samples

Avoid exporting zero inode gauges for missing or stale drive snapshots and ignore zero-count API latency buckets.

Co-Authored-By: heihutu <heihutu@gmail.com>

* feat(obs): add scanner source work detail metrics

Expose additive scanner source and cycle work metrics with bounded server/source/state labels while leaving the existing aggregate scanner metrics unchanged.

Co-Authored-By: heihutu <heihutu@gmail.com>

* feat(obs): add ilm action detail metrics

Expose additive ILM action/state task metrics with a server label while preserving the existing aggregate ILM series.

Co-Authored-By: heihutu <heihutu@gmail.com>

* feat(obs): add delivery target server metrics

Expose additive audit and notification delivery target metrics with server labels and extend removed-target tombstones for the server-aware series.

Co-Authored-By: heihutu <heihutu@gmail.com>

* feat(obs): add replication target flow metrics

Expose additive bucket replication target sent and failed-flow metrics while preserving existing bucket aggregates and target backlog series.

Co-Authored-By: heihutu <heihutu@gmail.com>

* feat(obs): add request server metrics

Expose additive API request metrics with server labels while preserving the existing request and traffic metric label sets.

Co-Authored-By: heihutu <heihutu@gmail.com>

* style(obs): apply rustfmt to metrics changes

Apply rustfmt output to the metrics dimension changes without altering behavior.

Co-Authored-By: heihutu <heihutu@gmail.com>

* style(obs): reuse audit target label constant

Use the exported audit target_id label constant for legacy audit target metrics.

Co-Authored-By: heihutu <heihutu@gmail.com>

* feat(obs): populate drive disk metrics

Co-Authored-By: heihutu <heihutu@gmail.com>

* feat(obs): add scanner bucket drive result metrics

Co-Authored-By: heihutu <heihutu@gmail.com>

* feat(obs): add replication proxy server metrics

Co-Authored-By: heihutu <heihutu@gmail.com>

* fix(obs): address metric liveness review

Use checked division for drive API latency aggregation and keep recovered drive, scanner current-cycle, replication flow, audit target, and notification target series from retaining stale values.

Co-Authored-By: heihutu <heihutu@gmail.com>

* fix(obs): address metric dimension review

Co-Authored-By: heihutu <heihutu@gmail.com>

* fix(obs): address additional metric review

Co-Authored-By: heihutu <heihutu@gmail.com>

* fix(obs): count drive calls at start

Co-Authored-By: heihutu <heihutu@gmail.com>

* fix(obs): address metrics dimension review

Co-Authored-By: heihutu <heihutu@gmail.com>

* fix(metrics): address dimension review gaps

Co-Authored-By: heihutu <heihutu@gmail.com>

* fix(metrics): address scanner review follow-ups

Co-Authored-By: heihutu <heihutu@gmail.com>

* fix(metrics): address runtime review follow-ups

Co-Authored-By: heihutu <heihutu@gmail.com>

* fix(metrics): reduce disk metric contention

Co-Authored-By: heihutu <heihutu@gmail.com>

* fix(metrics): address runtime review follow-ups

Co-Authored-By: heihutu <heihutu@gmail.com>

* fix(metrics): retire stale dimension series

Co-Authored-By: heihutu <heihutu@gmail.com>

---------

Co-authored-by: heihutu <heihutu@gmail.com>
This commit is contained in:
houseme
2026-08-03 09:03:34 +08:00
committed by GitHub
parent 988cd8adbb
commit 035ce5d784
36 changed files with 6282 additions and 666 deletions
+45 -8
View File
@@ -4359,7 +4359,13 @@ async fn get_disks_info(disks: &[Option<DiskStore>], eps: &[Endpoint]) -> Vec<ru
let offline_duration_seconds = disk.offline_duration_secs();
let capacity_snapshot = disk.last_capacity_snapshot();
if runtime_state.should_probe_for_admin() || runtime_state == disk::health_state::RuntimeDriveHealthState::Suspect {
match disk.disk_info(&DiskInfoOptions::default()).await {
match disk
.disk_info(&DiskInfoOptions {
metrics: true,
..Default::default()
})
.await
{
Ok(res) => {
disk.record_capacity_probe(res.total, res.used, res.free);
ret.push(rustfs_madmin::Disk {
@@ -4390,6 +4396,7 @@ async fn get_disks_info(disks: &[Option<DiskStore>], eps: &[Endpoint]) -> Vec<ru
utilization: utilization_percent(res.total, res.used),
used_inodes: res.used_inodes,
free_inodes: res.free_inodes,
metrics: Some(res.metrics),
..Default::default()
});
}
@@ -4397,12 +4404,14 @@ async fn get_disks_info(disks: &[Option<DiskStore>], eps: &[Endpoint]) -> Vec<ru
let mut disk_info = rustfs_madmin::Disk {
state: err.to_string(),
endpoint: eps[i].to_string(),
drive_path: eps[i].get_file_path(),
local: eps[i].is_local,
pool_index: eps[i].pool_idx,
set_index: eps[i].set_idx,
disk_index: eps[i].disk_idx,
runtime_state: Some(runtime_state.as_str().to_string()),
offline_duration_seconds,
metrics: disk.metrics_snapshot(),
..Default::default()
};
if let Some((total, used, free, _)) = capacity_snapshot {
@@ -4421,16 +4430,15 @@ async fn get_disks_info(disks: &[Option<DiskStore>], eps: &[Endpoint]) -> Vec<ru
}
}
} else {
ret.push(build_runtime_snapshot_disk(
&eps[i],
runtime_state,
offline_duration_seconds,
capacity_snapshot,
));
let mut disk_info =
build_runtime_snapshot_disk(&eps[i], runtime_state, offline_duration_seconds, capacity_snapshot);
disk_info.metrics = disk.metrics_snapshot();
ret.push(disk_info);
}
} else {
ret.push(rustfs_madmin::Disk {
endpoint: eps[i].to_string(),
drive_path: eps[i].get_file_path(),
local: eps[i].is_local,
pool_index: eps[i].pool_idx,
set_index: eps[i].set_idx,
@@ -4456,6 +4464,7 @@ fn build_runtime_snapshot_disk(
) -> rustfs_madmin::Disk {
let mut disk = rustfs_madmin::Disk {
endpoint: endpoint.to_string(),
drive_path: endpoint.get_file_path(),
local: endpoint.is_local,
pool_index: endpoint.pool_idx,
set_index: endpoint.set_idx,
@@ -7709,14 +7718,42 @@ mod tests {
assert_eq!(info[0].state, "ok");
assert_eq!(info[0].runtime_state.as_deref(), Some("online"));
assert!(!info[0].drive_path.is_empty(), "online disk should keep immediate disk_info probe");
assert!(
info[0]
.metrics
.as_ref()
.and_then(|metrics| metrics.api_calls.get("disk_info"))
.copied()
.unwrap_or_default()
> 0,
"online disk should expose disk_info operation metrics"
);
assert_eq!(info[1].state, "ok");
assert_eq!(info[1].runtime_state.as_deref(), Some("suspect"));
assert!(!info[1].drive_path.is_empty(), "suspect disk should still probe for fresher disk info");
assert!(
info[1]
.metrics
.as_ref()
.and_then(|metrics| metrics.last_minute.get("disk_info"))
.map(|action| action.count)
.unwrap_or_default()
> 0,
"suspect disk should expose last-minute disk_info latency"
);
assert_eq!(info[2].state, "offline");
assert_eq!(info[2].runtime_state.as_deref(), Some("offline"));
assert!(info[2].drive_path.is_empty(), "offline disk should use runtime snapshot fallback");
assert_eq!(
info[2].drive_path,
endpoints[2].get_file_path(),
"offline disk should keep stable endpoint path"
);
assert!(
info[2].metrics.is_some(),
"offline runtime fallback should preserve disk metrics snapshot"
);
}
#[tokio::test]
+35 -3
View File
@@ -322,7 +322,16 @@ impl SetDisks {
pub async fn renew_disk(&self, ep: &Endpoint) {
debug!("renew_disk: start {:?}", ep);
let (new_disk, fm) = match Self::connect_endpoint(ep).await {
let previous_health = {
let disks = self.disks.read().await;
disks
.iter()
.filter_map(|disk| disk.as_ref())
.find(|disk| disk.endpoint() == *ep)
.and_then(|disk| disk.local_health_tracker_epoch_for_reconnect())
};
let (new_disk, fm) = match Self::connect_endpoint(ep, previous_health).await {
Ok(res) => res,
Err(e) => {
warn!("renew_disk: connect_endpoint err {:?}", &e);
@@ -400,13 +409,17 @@ impl SetDisks {
Err(Error::other("DriveID: not found"))
}
pub(in crate::set_disk) async fn connect_endpoint(ep: &Endpoint) -> disk::error::Result<(DiskStore, FormatV3)> {
let disk = new_disk(
pub(in crate::set_disk) async fn connect_endpoint(
ep: &Endpoint,
reconnect: Option<disk::disk_store::ReconnectDiskHealthState>,
) -> disk::error::Result<(DiskStore, FormatV3)> {
let disk = crate::disk::new_disk_with_health_tracker(
ep,
&DiskOption {
cleanup: false,
health_check: true,
},
reconnect,
)
.await?;
@@ -714,6 +727,25 @@ mod tests {
renewed_disk.health_check_enabled_for_test(),
"renewed disks must keep health monitoring enabled so later faulty marks can recover"
);
renewed_disk
.disk_info(&DiskInfoOptions::default())
.await
.expect("renewed disk_info should record a drive API metric");
renewed_disk.force_runtime_state_for_test(disk::health_state::RuntimeDriveHealthState::Offline);
set_disks.renew_disk(&endpoints[0]).await;
let disks = set_disks.get_disks_internal().await;
let renewed_again = disks[0]
.as_ref()
.expect("second renew_disk should keep the recovered disk attached");
assert_eq!(
renewed_again
.metrics_snapshot()
.and_then(|metrics| metrics.api_calls.get("disk_info").copied()),
Some(1),
"disk reconnect must preserve the local drive metrics tracker epoch"
);
drop(temp_dirs);
}