mirror of
https://github.com/rustfs/rustfs.git
synced 2026-08-29 08:27:06 +00:00
035ce5d784
* feat(obs): add drive topology detail metrics Expose additive drive info, topology, state, and per-drive API metrics while preserving the existing drive metric label sets. Backlog: rustfs/backlog#1655 Co-Authored-By: heihutu <heihutu@gmail.com> * fix(obs): preserve suspect drive runtime state Keep suspect as a bounded drive runtime state and avoid all-zero runtime_state samples for that storage health state. Co-Authored-By: heihutu <heihutu@gmail.com> * fix(obs): skip unknown drive inode samples Avoid exporting zero inode gauges for missing or stale drive snapshots and ignore zero-count API latency buckets. Co-Authored-By: heihutu <heihutu@gmail.com> * feat(obs): add scanner source work detail metrics Expose additive scanner source and cycle work metrics with bounded server/source/state labels while leaving the existing aggregate scanner metrics unchanged. Co-Authored-By: heihutu <heihutu@gmail.com> * feat(obs): add ilm action detail metrics Expose additive ILM action/state task metrics with a server label while preserving the existing aggregate ILM series. Co-Authored-By: heihutu <heihutu@gmail.com> * feat(obs): add delivery target server metrics Expose additive audit and notification delivery target metrics with server labels and extend removed-target tombstones for the server-aware series. Co-Authored-By: heihutu <heihutu@gmail.com> * feat(obs): add replication target flow metrics Expose additive bucket replication target sent and failed-flow metrics while preserving existing bucket aggregates and target backlog series. Co-Authored-By: heihutu <heihutu@gmail.com> * feat(obs): add request server metrics Expose additive API request metrics with server labels while preserving the existing request and traffic metric label sets. Co-Authored-By: heihutu <heihutu@gmail.com> * style(obs): apply rustfmt to metrics changes Apply rustfmt output to the metrics dimension changes without altering behavior. Co-Authored-By: heihutu <heihutu@gmail.com> * style(obs): reuse audit target label constant Use the exported audit target_id label constant for legacy audit target metrics. Co-Authored-By: heihutu <heihutu@gmail.com> * feat(obs): populate drive disk metrics Co-Authored-By: heihutu <heihutu@gmail.com> * feat(obs): add scanner bucket drive result metrics Co-Authored-By: heihutu <heihutu@gmail.com> * feat(obs): add replication proxy server metrics Co-Authored-By: heihutu <heihutu@gmail.com> * fix(obs): address metric liveness review Use checked division for drive API latency aggregation and keep recovered drive, scanner current-cycle, replication flow, audit target, and notification target series from retaining stale values. Co-Authored-By: heihutu <heihutu@gmail.com> * fix(obs): address metric dimension review Co-Authored-By: heihutu <heihutu@gmail.com> * fix(obs): address additional metric review Co-Authored-By: heihutu <heihutu@gmail.com> * fix(obs): count drive calls at start Co-Authored-By: heihutu <heihutu@gmail.com> * fix(obs): address metrics dimension review Co-Authored-By: heihutu <heihutu@gmail.com> * fix(metrics): address dimension review gaps Co-Authored-By: heihutu <heihutu@gmail.com> * fix(metrics): address scanner review follow-ups Co-Authored-By: heihutu <heihutu@gmail.com> * fix(metrics): address runtime review follow-ups Co-Authored-By: heihutu <heihutu@gmail.com> * fix(metrics): reduce disk metric contention Co-Authored-By: heihutu <heihutu@gmail.com> * fix(metrics): address runtime review follow-ups Co-Authored-By: heihutu <heihutu@gmail.com> * fix(metrics): retire stale dimension series Co-Authored-By: heihutu <heihutu@gmail.com> --------- Co-authored-by: heihutu <heihutu@gmail.com>
209 lines
8.1 KiB
Rust
209 lines
8.1 KiB
Rust
// Copyright 2024 RustFS Team
|
|
//
|
|
// Licensed under the Apache License, Version 2.0 (the "License");
|
|
// you may not use this file except in compliance with the License.
|
|
// You may obtain a copy of the License at
|
|
//
|
|
// http://www.apache.org/licenses/LICENSE-2.0
|
|
//
|
|
// Unless required by applicable law or agreed to in writing, software
|
|
// distributed under the License is distributed on an "AS IS" BASIS,
|
|
// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
|
// See the License for the specific language governing permissions and
|
|
// limitations under the License.
|
|
|
|
#![allow(dead_code)]
|
|
|
|
use crate::metrics::report::PrometheusMetric;
|
|
use crate::metrics::schema::notification_target::{
|
|
NOTIFICATION_TARGET_FAILED_MESSAGES_BY_SERVER_MD, NOTIFICATION_TARGET_FAILED_MESSAGES_MD,
|
|
NOTIFICATION_TARGET_FAILED_STORE_LENGTH_BY_SERVER_MD, NOTIFICATION_TARGET_FAILED_STORE_LENGTH_MD,
|
|
NOTIFICATION_TARGET_QUEUE_LENGTH_BY_SERVER_MD, NOTIFICATION_TARGET_QUEUE_LENGTH_MD,
|
|
NOTIFICATION_TARGET_TOTAL_MESSAGES_BY_SERVER_MD, NOTIFICATION_TARGET_TOTAL_MESSAGES_MD, SERVER, TARGET_ID, TARGET_TYPE,
|
|
};
|
|
use std::borrow::Cow;
|
|
|
|
#[derive(Debug, Clone, Default)]
|
|
pub struct NotificationTargetStats {
|
|
pub failed_messages: u64,
|
|
pub failed_store_length: u64,
|
|
pub queue_length: u64,
|
|
pub target_id: String,
|
|
pub target_type: String,
|
|
pub total_messages: u64,
|
|
}
|
|
|
|
#[derive(Debug, Clone, Default)]
|
|
pub(crate) struct NotificationTargetRuntimeStats {
|
|
pub(crate) server: String,
|
|
pub(crate) target: NotificationTargetStats,
|
|
}
|
|
|
|
pub fn collect_notification_target_metrics(stats: &[NotificationTargetStats]) -> Vec<PrometheusMetric> {
|
|
if stats.is_empty() {
|
|
return Vec::new();
|
|
}
|
|
|
|
let mut metrics = Vec::with_capacity(stats.len() * 8);
|
|
for stat in stats {
|
|
let target_id: Cow<'static, str> = Cow::Owned(stat.target_id.clone());
|
|
let target_type: Cow<'static, str> = Cow::Owned(stat.target_type.clone());
|
|
|
|
metrics.push(
|
|
PrometheusMetric::from_descriptor(&NOTIFICATION_TARGET_FAILED_MESSAGES_MD, stat.failed_messages as f64)
|
|
.with_label(TARGET_ID, target_id.clone())
|
|
.with_label(TARGET_TYPE, target_type.clone()),
|
|
);
|
|
metrics.push(
|
|
PrometheusMetric::from_descriptor(&NOTIFICATION_TARGET_FAILED_STORE_LENGTH_MD, stat.failed_store_length as f64)
|
|
.with_label(TARGET_ID, target_id.clone())
|
|
.with_label(TARGET_TYPE, target_type.clone()),
|
|
);
|
|
metrics.push(
|
|
PrometheusMetric::from_descriptor(&NOTIFICATION_TARGET_QUEUE_LENGTH_MD, stat.queue_length as f64)
|
|
.with_label(TARGET_ID, target_id.clone())
|
|
.with_label(TARGET_TYPE, target_type.clone()),
|
|
);
|
|
metrics.push(
|
|
PrometheusMetric::from_descriptor(&NOTIFICATION_TARGET_TOTAL_MESSAGES_MD, stat.total_messages as f64)
|
|
.with_label(TARGET_ID, target_id)
|
|
.with_label(TARGET_TYPE, target_type),
|
|
);
|
|
}
|
|
|
|
metrics
|
|
}
|
|
|
|
pub(crate) fn collect_notification_target_runtime_metrics(stats: &[NotificationTargetRuntimeStats]) -> Vec<PrometheusMetric> {
|
|
if stats.is_empty() {
|
|
return Vec::new();
|
|
}
|
|
|
|
let legacy_stats = stats.iter().map(|stat| stat.target.clone()).collect::<Vec<_>>();
|
|
let mut metrics = collect_notification_target_metrics(&legacy_stats);
|
|
metrics.reserve(stats.len() * 4);
|
|
for stat in stats {
|
|
let server: Cow<'static, str> = Cow::Owned(stat.server.clone());
|
|
let target_id: Cow<'static, str> = Cow::Owned(stat.target.target_id.clone());
|
|
let target_type: Cow<'static, str> = Cow::Owned(stat.target.target_type.clone());
|
|
|
|
metrics.push(
|
|
PrometheusMetric::from_descriptor(
|
|
&NOTIFICATION_TARGET_FAILED_MESSAGES_BY_SERVER_MD,
|
|
stat.target.failed_messages as f64,
|
|
)
|
|
.with_label(SERVER, server.clone())
|
|
.with_label(TARGET_ID, target_id.clone())
|
|
.with_label(TARGET_TYPE, target_type.clone()),
|
|
);
|
|
metrics.push(
|
|
PrometheusMetric::from_descriptor(
|
|
&NOTIFICATION_TARGET_FAILED_STORE_LENGTH_BY_SERVER_MD,
|
|
stat.target.failed_store_length as f64,
|
|
)
|
|
.with_label(SERVER, server.clone())
|
|
.with_label(TARGET_ID, target_id.clone())
|
|
.with_label(TARGET_TYPE, target_type.clone()),
|
|
);
|
|
metrics.push(
|
|
PrometheusMetric::from_descriptor(&NOTIFICATION_TARGET_QUEUE_LENGTH_BY_SERVER_MD, stat.target.queue_length as f64)
|
|
.with_label(SERVER, server.clone())
|
|
.with_label(TARGET_ID, target_id.clone())
|
|
.with_label(TARGET_TYPE, target_type.clone()),
|
|
);
|
|
metrics.push(
|
|
PrometheusMetric::from_descriptor(
|
|
&NOTIFICATION_TARGET_TOTAL_MESSAGES_BY_SERVER_MD,
|
|
stat.target.total_messages as f64,
|
|
)
|
|
.with_label(SERVER, server)
|
|
.with_label(TARGET_ID, target_id)
|
|
.with_label(TARGET_TYPE, target_type),
|
|
);
|
|
}
|
|
|
|
metrics
|
|
}
|
|
|
|
#[cfg(test)]
|
|
mod tests {
|
|
use super::*;
|
|
use crate::metrics::schema::MetricType;
|
|
|
|
#[test]
|
|
fn test_collect_notification_target_metrics() {
|
|
let stats = [NotificationTargetStats {
|
|
failed_messages: 2,
|
|
failed_store_length: 3,
|
|
queue_length: 4,
|
|
target_id: "primary:webhook".to_string(),
|
|
target_type: "webhook".to_string(),
|
|
total_messages: 42,
|
|
}];
|
|
|
|
let metrics = collect_notification_target_runtime_metrics(&[NotificationTargetRuntimeStats {
|
|
server: "node1:9000".to_string(),
|
|
target: stats[0].clone(),
|
|
}]);
|
|
|
|
assert_eq!(metrics.len(), 8);
|
|
assert!(metrics.iter().any(|metric| {
|
|
metric.value == 3.0
|
|
&& metric.name == NOTIFICATION_TARGET_FAILED_STORE_LENGTH_MD.get_full_metric_name()
|
|
&& metric
|
|
.labels
|
|
.iter()
|
|
.any(|(key, value)| *key == TARGET_ID && value == "primary:webhook")
|
|
}));
|
|
assert!(metrics.iter().any(|metric| {
|
|
metric.value == 42.0
|
|
&& metric
|
|
.labels
|
|
.iter()
|
|
.any(|(key, value)| *key == TARGET_ID && value == "primary:webhook")
|
|
&& metric
|
|
.labels
|
|
.iter()
|
|
.any(|(key, value)| *key == TARGET_TYPE && value == "webhook")
|
|
}));
|
|
assert!(metrics.iter().any(|metric| {
|
|
metric.value == 4.0
|
|
&& metric.name == NOTIFICATION_TARGET_QUEUE_LENGTH_BY_SERVER_MD.get_full_metric_name()
|
|
&& metric
|
|
.labels
|
|
.iter()
|
|
.any(|(key, value)| *key == SERVER && value == "node1:9000")
|
|
&& metric
|
|
.labels
|
|
.iter()
|
|
.any(|(key, value)| *key == TARGET_ID && value == "primary:webhook")
|
|
&& metric
|
|
.labels
|
|
.iter()
|
|
.any(|(key, value)| *key == TARGET_TYPE && value == "webhook")
|
|
}));
|
|
}
|
|
|
|
#[test]
|
|
fn notification_target_totals_are_exported_as_gauges() {
|
|
assert_eq!(NOTIFICATION_TARGET_FAILED_MESSAGES_MD.metric_type, MetricType::Gauge);
|
|
assert_eq!(NOTIFICATION_TARGET_FAILED_STORE_LENGTH_MD.metric_type, MetricType::Gauge);
|
|
assert_eq!(NOTIFICATION_TARGET_QUEUE_LENGTH_MD.metric_type, MetricType::Gauge);
|
|
assert_eq!(NOTIFICATION_TARGET_TOTAL_MESSAGES_MD.metric_type, MetricType::Gauge);
|
|
}
|
|
|
|
#[test]
|
|
fn notification_target_stats_struct_literal_keeps_legacy_fields() {
|
|
let stats = vec![NotificationTargetStats {
|
|
failed_messages: 2,
|
|
failed_store_length: 3,
|
|
queue_length: 4,
|
|
target_id: "primary:webhook".to_string(),
|
|
target_type: "webhook".to_string(),
|
|
total_messages: 42,
|
|
}];
|
|
|
|
assert_eq!(collect_notification_target_metrics(&stats).len(), 4);
|
|
}
|
|
}
|