Files
rustfs/crates/obs/src/metrics/collectors/notification_target.rs
T
houseme 035ce5d784 feat(obs): add bounded metrics dimensions (#5645)
* feat(obs): add drive topology detail metrics

Expose additive drive info, topology, state, and per-drive API metrics while preserving the existing drive metric label sets.

Backlog: rustfs/backlog#1655

Co-Authored-By: heihutu <heihutu@gmail.com>

* fix(obs): preserve suspect drive runtime state

Keep suspect as a bounded drive runtime state and avoid all-zero runtime_state samples for that storage health state.

Co-Authored-By: heihutu <heihutu@gmail.com>

* fix(obs): skip unknown drive inode samples

Avoid exporting zero inode gauges for missing or stale drive snapshots and ignore zero-count API latency buckets.

Co-Authored-By: heihutu <heihutu@gmail.com>

* feat(obs): add scanner source work detail metrics

Expose additive scanner source and cycle work metrics with bounded server/source/state labels while leaving the existing aggregate scanner metrics unchanged.

Co-Authored-By: heihutu <heihutu@gmail.com>

* feat(obs): add ilm action detail metrics

Expose additive ILM action/state task metrics with a server label while preserving the existing aggregate ILM series.

Co-Authored-By: heihutu <heihutu@gmail.com>

* feat(obs): add delivery target server metrics

Expose additive audit and notification delivery target metrics with server labels and extend removed-target tombstones for the server-aware series.

Co-Authored-By: heihutu <heihutu@gmail.com>

* feat(obs): add replication target flow metrics

Expose additive bucket replication target sent and failed-flow metrics while preserving existing bucket aggregates and target backlog series.

Co-Authored-By: heihutu <heihutu@gmail.com>

* feat(obs): add request server metrics

Expose additive API request metrics with server labels while preserving the existing request and traffic metric label sets.

Co-Authored-By: heihutu <heihutu@gmail.com>

* style(obs): apply rustfmt to metrics changes

Apply rustfmt output to the metrics dimension changes without altering behavior.

Co-Authored-By: heihutu <heihutu@gmail.com>

* style(obs): reuse audit target label constant

Use the exported audit target_id label constant for legacy audit target metrics.

Co-Authored-By: heihutu <heihutu@gmail.com>

* feat(obs): populate drive disk metrics

Co-Authored-By: heihutu <heihutu@gmail.com>

* feat(obs): add scanner bucket drive result metrics

Co-Authored-By: heihutu <heihutu@gmail.com>

* feat(obs): add replication proxy server metrics

Co-Authored-By: heihutu <heihutu@gmail.com>

* fix(obs): address metric liveness review

Use checked division for drive API latency aggregation and keep recovered drive, scanner current-cycle, replication flow, audit target, and notification target series from retaining stale values.

Co-Authored-By: heihutu <heihutu@gmail.com>

* fix(obs): address metric dimension review

Co-Authored-By: heihutu <heihutu@gmail.com>

* fix(obs): address additional metric review

Co-Authored-By: heihutu <heihutu@gmail.com>

* fix(obs): count drive calls at start

Co-Authored-By: heihutu <heihutu@gmail.com>

* fix(obs): address metrics dimension review

Co-Authored-By: heihutu <heihutu@gmail.com>

* fix(metrics): address dimension review gaps

Co-Authored-By: heihutu <heihutu@gmail.com>

* fix(metrics): address scanner review follow-ups

Co-Authored-By: heihutu <heihutu@gmail.com>

* fix(metrics): address runtime review follow-ups

Co-Authored-By: heihutu <heihutu@gmail.com>

* fix(metrics): reduce disk metric contention

Co-Authored-By: heihutu <heihutu@gmail.com>

* fix(metrics): address runtime review follow-ups

Co-Authored-By: heihutu <heihutu@gmail.com>

* fix(metrics): retire stale dimension series

Co-Authored-By: heihutu <heihutu@gmail.com>

---------

Co-authored-by: heihutu <heihutu@gmail.com>
2026-08-03 09:03:34 +08:00

209 lines
8.1 KiB
Rust

// Copyright 2024 RustFS Team
//
// Licensed under the Apache License, Version 2.0 (the "License");
// you may not use this file except in compliance with the License.
// You may obtain a copy of the License at
//
// http://www.apache.org/licenses/LICENSE-2.0
//
// Unless required by applicable law or agreed to in writing, software
// distributed under the License is distributed on an "AS IS" BASIS,
// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
// See the License for the specific language governing permissions and
// limitations under the License.
#![allow(dead_code)]
use crate::metrics::report::PrometheusMetric;
use crate::metrics::schema::notification_target::{
NOTIFICATION_TARGET_FAILED_MESSAGES_BY_SERVER_MD, NOTIFICATION_TARGET_FAILED_MESSAGES_MD,
NOTIFICATION_TARGET_FAILED_STORE_LENGTH_BY_SERVER_MD, NOTIFICATION_TARGET_FAILED_STORE_LENGTH_MD,
NOTIFICATION_TARGET_QUEUE_LENGTH_BY_SERVER_MD, NOTIFICATION_TARGET_QUEUE_LENGTH_MD,
NOTIFICATION_TARGET_TOTAL_MESSAGES_BY_SERVER_MD, NOTIFICATION_TARGET_TOTAL_MESSAGES_MD, SERVER, TARGET_ID, TARGET_TYPE,
};
use std::borrow::Cow;
#[derive(Debug, Clone, Default)]
pub struct NotificationTargetStats {
pub failed_messages: u64,
pub failed_store_length: u64,
pub queue_length: u64,
pub target_id: String,
pub target_type: String,
pub total_messages: u64,
}
#[derive(Debug, Clone, Default)]
pub(crate) struct NotificationTargetRuntimeStats {
pub(crate) server: String,
pub(crate) target: NotificationTargetStats,
}
pub fn collect_notification_target_metrics(stats: &[NotificationTargetStats]) -> Vec<PrometheusMetric> {
if stats.is_empty() {
return Vec::new();
}
let mut metrics = Vec::with_capacity(stats.len() * 8);
for stat in stats {
let target_id: Cow<'static, str> = Cow::Owned(stat.target_id.clone());
let target_type: Cow<'static, str> = Cow::Owned(stat.target_type.clone());
metrics.push(
PrometheusMetric::from_descriptor(&NOTIFICATION_TARGET_FAILED_MESSAGES_MD, stat.failed_messages as f64)
.with_label(TARGET_ID, target_id.clone())
.with_label(TARGET_TYPE, target_type.clone()),
);
metrics.push(
PrometheusMetric::from_descriptor(&NOTIFICATION_TARGET_FAILED_STORE_LENGTH_MD, stat.failed_store_length as f64)
.with_label(TARGET_ID, target_id.clone())
.with_label(TARGET_TYPE, target_type.clone()),
);
metrics.push(
PrometheusMetric::from_descriptor(&NOTIFICATION_TARGET_QUEUE_LENGTH_MD, stat.queue_length as f64)
.with_label(TARGET_ID, target_id.clone())
.with_label(TARGET_TYPE, target_type.clone()),
);
metrics.push(
PrometheusMetric::from_descriptor(&NOTIFICATION_TARGET_TOTAL_MESSAGES_MD, stat.total_messages as f64)
.with_label(TARGET_ID, target_id)
.with_label(TARGET_TYPE, target_type),
);
}
metrics
}
pub(crate) fn collect_notification_target_runtime_metrics(stats: &[NotificationTargetRuntimeStats]) -> Vec<PrometheusMetric> {
if stats.is_empty() {
return Vec::new();
}
let legacy_stats = stats.iter().map(|stat| stat.target.clone()).collect::<Vec<_>>();
let mut metrics = collect_notification_target_metrics(&legacy_stats);
metrics.reserve(stats.len() * 4);
for stat in stats {
let server: Cow<'static, str> = Cow::Owned(stat.server.clone());
let target_id: Cow<'static, str> = Cow::Owned(stat.target.target_id.clone());
let target_type: Cow<'static, str> = Cow::Owned(stat.target.target_type.clone());
metrics.push(
PrometheusMetric::from_descriptor(
&NOTIFICATION_TARGET_FAILED_MESSAGES_BY_SERVER_MD,
stat.target.failed_messages as f64,
)
.with_label(SERVER, server.clone())
.with_label(TARGET_ID, target_id.clone())
.with_label(TARGET_TYPE, target_type.clone()),
);
metrics.push(
PrometheusMetric::from_descriptor(
&NOTIFICATION_TARGET_FAILED_STORE_LENGTH_BY_SERVER_MD,
stat.target.failed_store_length as f64,
)
.with_label(SERVER, server.clone())
.with_label(TARGET_ID, target_id.clone())
.with_label(TARGET_TYPE, target_type.clone()),
);
metrics.push(
PrometheusMetric::from_descriptor(&NOTIFICATION_TARGET_QUEUE_LENGTH_BY_SERVER_MD, stat.target.queue_length as f64)
.with_label(SERVER, server.clone())
.with_label(TARGET_ID, target_id.clone())
.with_label(TARGET_TYPE, target_type.clone()),
);
metrics.push(
PrometheusMetric::from_descriptor(
&NOTIFICATION_TARGET_TOTAL_MESSAGES_BY_SERVER_MD,
stat.target.total_messages as f64,
)
.with_label(SERVER, server)
.with_label(TARGET_ID, target_id)
.with_label(TARGET_TYPE, target_type),
);
}
metrics
}
#[cfg(test)]
mod tests {
use super::*;
use crate::metrics::schema::MetricType;
#[test]
fn test_collect_notification_target_metrics() {
let stats = [NotificationTargetStats {
failed_messages: 2,
failed_store_length: 3,
queue_length: 4,
target_id: "primary:webhook".to_string(),
target_type: "webhook".to_string(),
total_messages: 42,
}];
let metrics = collect_notification_target_runtime_metrics(&[NotificationTargetRuntimeStats {
server: "node1:9000".to_string(),
target: stats[0].clone(),
}]);
assert_eq!(metrics.len(), 8);
assert!(metrics.iter().any(|metric| {
metric.value == 3.0
&& metric.name == NOTIFICATION_TARGET_FAILED_STORE_LENGTH_MD.get_full_metric_name()
&& metric
.labels
.iter()
.any(|(key, value)| *key == TARGET_ID && value == "primary:webhook")
}));
assert!(metrics.iter().any(|metric| {
metric.value == 42.0
&& metric
.labels
.iter()
.any(|(key, value)| *key == TARGET_ID && value == "primary:webhook")
&& metric
.labels
.iter()
.any(|(key, value)| *key == TARGET_TYPE && value == "webhook")
}));
assert!(metrics.iter().any(|metric| {
metric.value == 4.0
&& metric.name == NOTIFICATION_TARGET_QUEUE_LENGTH_BY_SERVER_MD.get_full_metric_name()
&& metric
.labels
.iter()
.any(|(key, value)| *key == SERVER && value == "node1:9000")
&& metric
.labels
.iter()
.any(|(key, value)| *key == TARGET_ID && value == "primary:webhook")
&& metric
.labels
.iter()
.any(|(key, value)| *key == TARGET_TYPE && value == "webhook")
}));
}
#[test]
fn notification_target_totals_are_exported_as_gauges() {
assert_eq!(NOTIFICATION_TARGET_FAILED_MESSAGES_MD.metric_type, MetricType::Gauge);
assert_eq!(NOTIFICATION_TARGET_FAILED_STORE_LENGTH_MD.metric_type, MetricType::Gauge);
assert_eq!(NOTIFICATION_TARGET_QUEUE_LENGTH_MD.metric_type, MetricType::Gauge);
assert_eq!(NOTIFICATION_TARGET_TOTAL_MESSAGES_MD.metric_type, MetricType::Gauge);
}
#[test]
fn notification_target_stats_struct_literal_keeps_legacy_fields() {
let stats = vec![NotificationTargetStats {
failed_messages: 2,
failed_store_length: 3,
queue_length: 4,
target_id: "primary:webhook".to_string(),
target_type: "webhook".to_string(),
total_messages: 42,
}];
assert_eq!(collect_notification_target_metrics(&stats).len(), 4);
}
}