mirror of
https://github.com/rustfs/rustfs.git
synced 2026-08-16 01:48:21 +00:00
035ce5d784
* feat(obs): add drive topology detail metrics Expose additive drive info, topology, state, and per-drive API metrics while preserving the existing drive metric label sets. Backlog: rustfs/backlog#1655 Co-Authored-By: heihutu <heihutu@gmail.com> * fix(obs): preserve suspect drive runtime state Keep suspect as a bounded drive runtime state and avoid all-zero runtime_state samples for that storage health state. Co-Authored-By: heihutu <heihutu@gmail.com> * fix(obs): skip unknown drive inode samples Avoid exporting zero inode gauges for missing or stale drive snapshots and ignore zero-count API latency buckets. Co-Authored-By: heihutu <heihutu@gmail.com> * feat(obs): add scanner source work detail metrics Expose additive scanner source and cycle work metrics with bounded server/source/state labels while leaving the existing aggregate scanner metrics unchanged. Co-Authored-By: heihutu <heihutu@gmail.com> * feat(obs): add ilm action detail metrics Expose additive ILM action/state task metrics with a server label while preserving the existing aggregate ILM series. Co-Authored-By: heihutu <heihutu@gmail.com> * feat(obs): add delivery target server metrics Expose additive audit and notification delivery target metrics with server labels and extend removed-target tombstones for the server-aware series. Co-Authored-By: heihutu <heihutu@gmail.com> * feat(obs): add replication target flow metrics Expose additive bucket replication target sent and failed-flow metrics while preserving existing bucket aggregates and target backlog series. Co-Authored-By: heihutu <heihutu@gmail.com> * feat(obs): add request server metrics Expose additive API request metrics with server labels while preserving the existing request and traffic metric label sets. Co-Authored-By: heihutu <heihutu@gmail.com> * style(obs): apply rustfmt to metrics changes Apply rustfmt output to the metrics dimension changes without altering behavior. Co-Authored-By: heihutu <heihutu@gmail.com> * style(obs): reuse audit target label constant Use the exported audit target_id label constant for legacy audit target metrics. Co-Authored-By: heihutu <heihutu@gmail.com> * feat(obs): populate drive disk metrics Co-Authored-By: heihutu <heihutu@gmail.com> * feat(obs): add scanner bucket drive result metrics Co-Authored-By: heihutu <heihutu@gmail.com> * feat(obs): add replication proxy server metrics Co-Authored-By: heihutu <heihutu@gmail.com> * fix(obs): address metric liveness review Use checked division for drive API latency aggregation and keep recovered drive, scanner current-cycle, replication flow, audit target, and notification target series from retaining stale values. Co-Authored-By: heihutu <heihutu@gmail.com> * fix(obs): address metric dimension review Co-Authored-By: heihutu <heihutu@gmail.com> * fix(obs): address additional metric review Co-Authored-By: heihutu <heihutu@gmail.com> * fix(obs): count drive calls at start Co-Authored-By: heihutu <heihutu@gmail.com> * fix(obs): address metrics dimension review Co-Authored-By: heihutu <heihutu@gmail.com> * fix(metrics): address dimension review gaps Co-Authored-By: heihutu <heihutu@gmail.com> * fix(metrics): address scanner review follow-ups Co-Authored-By: heihutu <heihutu@gmail.com> * fix(metrics): address runtime review follow-ups Co-Authored-By: heihutu <heihutu@gmail.com> * fix(metrics): reduce disk metric contention Co-Authored-By: heihutu <heihutu@gmail.com> * fix(metrics): address runtime review follow-ups Co-Authored-By: heihutu <heihutu@gmail.com> * fix(metrics): retire stale dimension series Co-Authored-By: heihutu <heihutu@gmail.com> --------- Co-authored-by: heihutu <heihutu@gmail.com>
255 lines
9.0 KiB
Rust
255 lines
9.0 KiB
Rust
// Copyright 2024 RustFS Team
|
|
//
|
|
// Licensed under the Apache License, Version 2.0 (the "License");
|
|
// you may not use this file except in compliance with the License.
|
|
// You may obtain a copy of the License at
|
|
//
|
|
// http://www.apache.org/licenses/LICENSE-2.0
|
|
//
|
|
// Unless required by applicable law or agreed to in writing, software
|
|
// distributed under the License is distributed on an "AS IS" BASIS,
|
|
// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
|
// See the License for the specific language governing permissions and
|
|
// limitations under the License.
|
|
|
|
#![allow(dead_code)]
|
|
|
|
use crate::{MetricDescriptor, MetricName, new_gauge_md, subsystems};
|
|
use std::sync::LazyLock;
|
|
|
|
pub const SERVER_LABEL: &str = "server";
|
|
|
|
pub static REPLICATION_AVERAGE_ACTIVE_WORKERS_MD: LazyLock<MetricDescriptor> = LazyLock::new(|| {
|
|
new_gauge_md(
|
|
MetricName::ReplicationAverageActiveWorkers,
|
|
"Average number of active replication workers",
|
|
&[],
|
|
subsystems::REPLICATION,
|
|
)
|
|
});
|
|
|
|
pub static REPLICATION_AVERAGE_ACTIVE_WORKERS_BY_SERVER_MD: LazyLock<MetricDescriptor> = LazyLock::new(|| {
|
|
new_gauge_md(
|
|
MetricName::Custom("average_active_workers_by_server".to_string()),
|
|
"Average number of active replication workers by server",
|
|
&[SERVER_LABEL],
|
|
subsystems::REPLICATION,
|
|
)
|
|
});
|
|
|
|
pub static REPLICATION_AVERAGE_QUEUED_BYTES_MD: LazyLock<MetricDescriptor> = LazyLock::new(|| {
|
|
new_gauge_md(
|
|
MetricName::ReplicationAverageQueuedBytes,
|
|
"Average number of bytes queued for replication since server start",
|
|
&[],
|
|
subsystems::REPLICATION,
|
|
)
|
|
});
|
|
|
|
pub static REPLICATION_AVERAGE_QUEUED_BYTES_BY_SERVER_MD: LazyLock<MetricDescriptor> = LazyLock::new(|| {
|
|
new_gauge_md(
|
|
MetricName::Custom("average_queued_bytes_by_server".to_string()),
|
|
"Average number of bytes queued for replication since server start by server",
|
|
&[SERVER_LABEL],
|
|
subsystems::REPLICATION,
|
|
)
|
|
});
|
|
|
|
pub static REPLICATION_AVERAGE_QUEUED_COUNT_MD: LazyLock<MetricDescriptor> = LazyLock::new(|| {
|
|
new_gauge_md(
|
|
MetricName::ReplicationAverageQueuedCount,
|
|
"Average number of objects queued for replication since server start",
|
|
&[],
|
|
subsystems::REPLICATION,
|
|
)
|
|
});
|
|
|
|
pub static REPLICATION_AVERAGE_QUEUED_COUNT_BY_SERVER_MD: LazyLock<MetricDescriptor> = LazyLock::new(|| {
|
|
new_gauge_md(
|
|
MetricName::Custom("average_queued_count_by_server".to_string()),
|
|
"Average number of objects queued for replication since server start by server",
|
|
&[SERVER_LABEL],
|
|
subsystems::REPLICATION,
|
|
)
|
|
});
|
|
|
|
pub static REPLICATION_AVERAGE_DATA_TRANSFER_RATE_MD: LazyLock<MetricDescriptor> = LazyLock::new(|| {
|
|
new_gauge_md(
|
|
MetricName::ReplicationAverageDataTransferRate,
|
|
"Average replication data transfer rate in bytes/sec",
|
|
&[],
|
|
subsystems::REPLICATION,
|
|
)
|
|
});
|
|
|
|
pub static REPLICATION_AVERAGE_DATA_TRANSFER_RATE_BY_SERVER_MD: LazyLock<MetricDescriptor> = LazyLock::new(|| {
|
|
new_gauge_md(
|
|
MetricName::Custom("average_data_transfer_rate_by_server".to_string()),
|
|
"Average replication data transfer rate in bytes/sec by server",
|
|
&[SERVER_LABEL],
|
|
subsystems::REPLICATION,
|
|
)
|
|
});
|
|
|
|
pub static REPLICATION_CURRENT_ACTIVE_WORKERS_MD: LazyLock<MetricDescriptor> = LazyLock::new(|| {
|
|
new_gauge_md(
|
|
MetricName::ReplicationCurrentActiveWorkers,
|
|
"Total number of active replication workers",
|
|
&[],
|
|
subsystems::REPLICATION,
|
|
)
|
|
});
|
|
|
|
pub static REPLICATION_CURRENT_ACTIVE_WORKERS_BY_SERVER_MD: LazyLock<MetricDescriptor> = LazyLock::new(|| {
|
|
new_gauge_md(
|
|
MetricName::Custom("current_active_workers_by_server".to_string()),
|
|
"Total number of active replication workers by server",
|
|
&[SERVER_LABEL],
|
|
subsystems::REPLICATION,
|
|
)
|
|
});
|
|
|
|
pub static REPLICATION_CURRENT_DATA_TRANSFER_RATE_MD: LazyLock<MetricDescriptor> = LazyLock::new(|| {
|
|
new_gauge_md(
|
|
MetricName::ReplicationCurrentDataTransferRate,
|
|
"Current replication data transfer rate in bytes/sec",
|
|
&[],
|
|
subsystems::REPLICATION,
|
|
)
|
|
});
|
|
|
|
pub static REPLICATION_CURRENT_DATA_TRANSFER_RATE_BY_SERVER_MD: LazyLock<MetricDescriptor> = LazyLock::new(|| {
|
|
new_gauge_md(
|
|
MetricName::Custom("current_data_transfer_rate_by_server".to_string()),
|
|
"Current replication data transfer rate in bytes/sec by server",
|
|
&[SERVER_LABEL],
|
|
subsystems::REPLICATION,
|
|
)
|
|
});
|
|
|
|
pub static REPLICATION_LAST_MINUTE_QUEUED_BYTES_MD: LazyLock<MetricDescriptor> = LazyLock::new(|| {
|
|
new_gauge_md(
|
|
MetricName::ReplicationLastMinuteQueuedBytes,
|
|
"Number of bytes queued for replication in the last full minute",
|
|
&[],
|
|
subsystems::REPLICATION,
|
|
)
|
|
});
|
|
|
|
pub static REPLICATION_LAST_MINUTE_QUEUED_BYTES_BY_SERVER_MD: LazyLock<MetricDescriptor> = LazyLock::new(|| {
|
|
new_gauge_md(
|
|
MetricName::Custom("last_minute_queued_bytes_by_server".to_string()),
|
|
"Number of bytes queued for replication in the last full minute by server",
|
|
&[SERVER_LABEL],
|
|
subsystems::REPLICATION,
|
|
)
|
|
});
|
|
|
|
pub static REPLICATION_LAST_MINUTE_QUEUED_COUNT_MD: LazyLock<MetricDescriptor> = LazyLock::new(|| {
|
|
new_gauge_md(
|
|
MetricName::ReplicationLastMinuteQueuedCount,
|
|
"Number of objects queued for replication in the last full minute",
|
|
&[],
|
|
subsystems::REPLICATION,
|
|
)
|
|
});
|
|
|
|
pub static REPLICATION_LAST_MINUTE_QUEUED_COUNT_BY_SERVER_MD: LazyLock<MetricDescriptor> = LazyLock::new(|| {
|
|
new_gauge_md(
|
|
MetricName::Custom("last_minute_queued_count_by_server".to_string()),
|
|
"Number of objects queued for replication in the last full minute by server",
|
|
&[SERVER_LABEL],
|
|
subsystems::REPLICATION,
|
|
)
|
|
});
|
|
|
|
pub static REPLICATION_MAX_ACTIVE_WORKERS_MD: LazyLock<MetricDescriptor> = LazyLock::new(|| {
|
|
new_gauge_md(
|
|
MetricName::ReplicationMaxActiveWorkers,
|
|
"Maximum number of active replication workers seen since server start",
|
|
&[],
|
|
subsystems::REPLICATION,
|
|
)
|
|
});
|
|
|
|
pub static REPLICATION_MAX_ACTIVE_WORKERS_BY_SERVER_MD: LazyLock<MetricDescriptor> = LazyLock::new(|| {
|
|
new_gauge_md(
|
|
MetricName::Custom("max_active_workers_by_server".to_string()),
|
|
"Maximum number of active replication workers seen since server start by server",
|
|
&[SERVER_LABEL],
|
|
subsystems::REPLICATION,
|
|
)
|
|
});
|
|
|
|
pub static REPLICATION_MAX_QUEUED_BYTES_MD: LazyLock<MetricDescriptor> = LazyLock::new(|| {
|
|
new_gauge_md(
|
|
MetricName::ReplicationMaxQueuedBytes,
|
|
"Maximum number of bytes queued for replication since server start",
|
|
&[],
|
|
subsystems::REPLICATION,
|
|
)
|
|
});
|
|
|
|
pub static REPLICATION_MAX_QUEUED_BYTES_BY_SERVER_MD: LazyLock<MetricDescriptor> = LazyLock::new(|| {
|
|
new_gauge_md(
|
|
MetricName::Custom("max_queued_bytes_by_server".to_string()),
|
|
"Maximum number of bytes queued for replication since server start by server",
|
|
&[SERVER_LABEL],
|
|
subsystems::REPLICATION,
|
|
)
|
|
});
|
|
|
|
pub static REPLICATION_MAX_QUEUED_COUNT_MD: LazyLock<MetricDescriptor> = LazyLock::new(|| {
|
|
new_gauge_md(
|
|
MetricName::ReplicationMaxQueuedCount,
|
|
"Maximum number of objects queued for replication since server start",
|
|
&[],
|
|
subsystems::REPLICATION,
|
|
)
|
|
});
|
|
|
|
pub static REPLICATION_MAX_QUEUED_COUNT_BY_SERVER_MD: LazyLock<MetricDescriptor> = LazyLock::new(|| {
|
|
new_gauge_md(
|
|
MetricName::Custom("max_queued_count_by_server".to_string()),
|
|
"Maximum number of objects queued for replication since server start by server",
|
|
&[SERVER_LABEL],
|
|
subsystems::REPLICATION,
|
|
)
|
|
});
|
|
|
|
pub static REPLICATION_MAX_DATA_TRANSFER_RATE_MD: LazyLock<MetricDescriptor> = LazyLock::new(|| {
|
|
new_gauge_md(
|
|
MetricName::ReplicationMaxDataTransferRate,
|
|
"Maximum replication data transfer rate in bytes/sec seen since server start",
|
|
&[],
|
|
subsystems::REPLICATION,
|
|
)
|
|
});
|
|
|
|
pub static REPLICATION_MAX_DATA_TRANSFER_RATE_BY_SERVER_MD: LazyLock<MetricDescriptor> = LazyLock::new(|| {
|
|
new_gauge_md(
|
|
MetricName::Custom("max_data_transfer_rate_by_server".to_string()),
|
|
"Maximum replication data transfer rate in bytes/sec seen since server start by server",
|
|
&[SERVER_LABEL],
|
|
subsystems::REPLICATION,
|
|
)
|
|
});
|
|
|
|
pub static REPLICATION_RECENT_BACKLOG_COUNT_MD: LazyLock<MetricDescriptor> = LazyLock::new(|| {
|
|
new_gauge_md(
|
|
MetricName::ReplicationRecentBacklogCount,
|
|
"Legacy replication backlog indicator: failed target objects plus objects currently queued on this node",
|
|
&[],
|
|
subsystems::REPLICATION,
|
|
)
|
|
});
|
|
|
|
pub static REPLICATION_RECENT_BACKLOG_COUNT_BY_SERVER_MD: LazyLock<MetricDescriptor> = LazyLock::new(|| {
|
|
new_gauge_md(
|
|
MetricName::Custom("recent_backlog_count_by_server".to_string()),
|
|
"Objects currently in replication backlog by server",
|
|
&[SERVER_LABEL],
|
|
subsystems::REPLICATION,
|
|
)
|
|
});
|