mirror of
https://github.com/rustfs/rustfs.git
synced 2026-08-06 13:27:43 +00:00
506ded59ad
fix(metrics): normalize peer-health key so servers_offline_total counts each peer once rustfs_cluster_servers_offline_total is the count of distinct keys in the addr-keyed CLUSTER_PEER_HEALTH map. The data path keys a peer by endpoint.grid_host() (scheme://host:port) while the lock path keys it by url::Url::to_string() (scheme://host:port/, trailing slash), so one physical peer becomes two health entries and a single downed node reports 2 (2N for N). Normalize the address (trim trailing slashes) at the map boundary in record_peer_reachable/record_peer_unreachable/cluster_peer_is_offline/ cluster_peer_should_bypass so both forms collapse to one canonical key. Pure observability fix; peer selection and quorum are unchanged. Fixes rustfs/backlog#1043 Co-authored-by: heihutu <heihutu@gmail.com>
849 lines
37 KiB
Rust
849 lines
37 KiB
Rust
// Copyright 2024 RustFS Team
|
|
//
|
|
// Licensed under the Apache License, Version 2.0 (the "License");
|
|
// you may not use this file except in compliance with the License.
|
|
// You may obtain a copy of the License at
|
|
//
|
|
// http://www.apache.org/licenses/LICENSE-2.0
|
|
//
|
|
// Unless required by applicable law or agreed to in writing, software
|
|
// distributed under the License is distributed on an "AS IS" BASIS,
|
|
// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
|
// See the License for the specific language governing permissions and
|
|
// limitations under the License.
|
|
|
|
use metrics::{counter, gauge};
|
|
use std::collections::HashMap;
|
|
use std::sync::{
|
|
Arc, LazyLock, Mutex,
|
|
atomic::{AtomicU64, Ordering},
|
|
};
|
|
use std::time::{Duration, Instant, SystemTime, UNIX_EPOCH};
|
|
|
|
pub const INTERNODE_OPERATION_READ_FILE_STREAM: &str = "read_file_stream";
|
|
pub const INTERNODE_OPERATION_PUT_FILE_STREAM: &str = "put_file_stream";
|
|
pub const INTERNODE_OPERATION_WALK_DIR: &str = "walk_dir";
|
|
pub const INTERNODE_OPERATION_GRPC_READ_ALL: &str = "grpc_read_all";
|
|
pub const INTERNODE_OPERATION_GRPC_WRITE_ALL: &str = "grpc_write_all";
|
|
pub const INTERNODE_OPERATION_GRPC_READ_MULTIPLE: &str = "grpc_read_multiple";
|
|
pub const INTERNODE_TRANSPORT_BACKEND_TCP_HTTP: &str = "tcp-http";
|
|
pub const INTERNODE_TRANSPORT_BACKEND_GRPC: &str = "grpc";
|
|
pub const INTERNODE_TRANSPORT_BACKEND_UNKNOWN: &str = "unknown";
|
|
|
|
/// Direction of a msgpack/JSON codec decode, for the JSON-fallback counter: a server decoding a
|
|
/// peer's request vs a client decoding a peer's response (grpc-optimization P2).
|
|
pub const INTERNODE_MSGPACK_DIRECTION_REQUEST: &str = "request";
|
|
pub const INTERNODE_MSGPACK_DIRECTION_RESPONSE: &str = "response";
|
|
|
|
const OPERATION_LABEL: &str = "operation";
|
|
const BACKEND_LABEL: &str = "backend";
|
|
const CLASSIFICATION_LABEL: &str = "classification";
|
|
const STAGE_LABEL: &str = "stage";
|
|
const DOMINANT_ERROR_LABEL: &str = "dominant_error";
|
|
const HTTP_VERSION_LABEL: &str = "http_version";
|
|
const DIRECTION_LABEL: &str = "direction";
|
|
const MESSAGE_LABEL: &str = "message";
|
|
const INTERNODE_OPERATION_SENT_BYTES_TOTAL: &str = "rustfs_system_network_internode_operation_sent_bytes_total";
|
|
const INTERNODE_OPERATION_RECV_BYTES_TOTAL: &str = "rustfs_system_network_internode_operation_recv_bytes_total";
|
|
const INTERNODE_OPERATION_REQUESTS_OUTGOING_TOTAL: &str = "rustfs_system_network_internode_operation_requests_outgoing_total";
|
|
const INTERNODE_OPERATION_REQUESTS_INCOMING_TOTAL: &str = "rustfs_system_network_internode_operation_requests_incoming_total";
|
|
const INTERNODE_OPERATION_ERRORS_TOTAL: &str = "rustfs_system_network_internode_operation_errors_total";
|
|
const INTERNODE_OPERATION_DURATION_MS: &str = "rustfs_system_network_internode_operation_duration_ms";
|
|
const INTERNODE_OPERATION_CLASSIFIED_ERRORS_TOTAL: &str = "rustfs_system_network_internode_operation_classified_errors_total";
|
|
const INTERNODE_OPERATION_RETRIES_TOTAL: &str = "rustfs_system_network_internode_operation_retries_total";
|
|
const INTERNODE_OPERATION_RETRY_SUCCESSES_TOTAL: &str = "rustfs_system_network_internode_operation_retry_successes_total";
|
|
const INTERNODE_OPERATION_HTTP_VERSIONS_TOTAL: &str = "rustfs_system_network_internode_operation_http_versions_total";
|
|
const INTERNODE_OPERATION_STALL_TIMEOUTS_TOTAL: &str = "rustfs_system_network_internode_operation_stall_timeouts_total";
|
|
const INTERNODE_OPERATION_WRITE_SHUTDOWN_ERRORS_TOTAL: &str =
|
|
"rustfs_system_network_internode_operation_write_shutdown_errors_total";
|
|
const INTERNODE_OPERATION_PAYLOAD_BYTES: &str = "rustfs_system_network_internode_operation_payload_bytes";
|
|
const INTERNODE_OPERATION_LARGE_PAYLOADS_TOTAL: &str = "rustfs_system_network_internode_operation_large_payloads_total";
|
|
const INTERNODE_MSGPACK_JSON_FALLBACK_TOTAL: &str = "rustfs_system_network_internode_msgpack_json_fallback_total";
|
|
const ERASURE_WRITE_QUORUM_FAILURES_TOTAL: &str = "rustfs_system_storage_erasure_write_quorum_failures_total";
|
|
|
|
#[derive(Debug, Clone, Copy, PartialEq, Eq)]
|
|
pub struct InternodeOperationMetricDescriptor {
|
|
pub name: &'static str,
|
|
pub labels: &'static [&'static str],
|
|
}
|
|
|
|
const OPERATION_BACKEND_LABELS: &[&str] = &[OPERATION_LABEL, BACKEND_LABEL];
|
|
const OPERATION_BACKEND_CLASSIFICATION_LABELS: &[&str] = &[OPERATION_LABEL, BACKEND_LABEL, CLASSIFICATION_LABEL];
|
|
const OPERATION_BACKEND_HTTP_VERSION_LABELS: &[&str] = &[OPERATION_LABEL, BACKEND_LABEL, HTTP_VERSION_LABEL];
|
|
const QUORUM_FAILURE_LABELS: &[&str] = &[STAGE_LABEL, DOMINANT_ERROR_LABEL];
|
|
|
|
pub const INTERNODE_OPERATION_METRICS: &[InternodeOperationMetricDescriptor] = &[
|
|
InternodeOperationMetricDescriptor {
|
|
name: INTERNODE_OPERATION_SENT_BYTES_TOTAL,
|
|
labels: OPERATION_BACKEND_LABELS,
|
|
},
|
|
InternodeOperationMetricDescriptor {
|
|
name: INTERNODE_OPERATION_RECV_BYTES_TOTAL,
|
|
labels: OPERATION_BACKEND_LABELS,
|
|
},
|
|
InternodeOperationMetricDescriptor {
|
|
name: INTERNODE_OPERATION_REQUESTS_OUTGOING_TOTAL,
|
|
labels: OPERATION_BACKEND_LABELS,
|
|
},
|
|
InternodeOperationMetricDescriptor {
|
|
name: INTERNODE_OPERATION_REQUESTS_INCOMING_TOTAL,
|
|
labels: OPERATION_BACKEND_LABELS,
|
|
},
|
|
InternodeOperationMetricDescriptor {
|
|
name: INTERNODE_OPERATION_ERRORS_TOTAL,
|
|
labels: OPERATION_BACKEND_LABELS,
|
|
},
|
|
InternodeOperationMetricDescriptor {
|
|
name: INTERNODE_OPERATION_DURATION_MS,
|
|
labels: OPERATION_BACKEND_LABELS,
|
|
},
|
|
InternodeOperationMetricDescriptor {
|
|
name: INTERNODE_OPERATION_CLASSIFIED_ERRORS_TOTAL,
|
|
labels: OPERATION_BACKEND_CLASSIFICATION_LABELS,
|
|
},
|
|
InternodeOperationMetricDescriptor {
|
|
name: INTERNODE_OPERATION_RETRIES_TOTAL,
|
|
labels: OPERATION_BACKEND_CLASSIFICATION_LABELS,
|
|
},
|
|
InternodeOperationMetricDescriptor {
|
|
name: INTERNODE_OPERATION_RETRY_SUCCESSES_TOTAL,
|
|
labels: OPERATION_BACKEND_CLASSIFICATION_LABELS,
|
|
},
|
|
InternodeOperationMetricDescriptor {
|
|
name: INTERNODE_OPERATION_HTTP_VERSIONS_TOTAL,
|
|
labels: OPERATION_BACKEND_HTTP_VERSION_LABELS,
|
|
},
|
|
InternodeOperationMetricDescriptor {
|
|
name: INTERNODE_OPERATION_STALL_TIMEOUTS_TOTAL,
|
|
labels: OPERATION_BACKEND_LABELS,
|
|
},
|
|
InternodeOperationMetricDescriptor {
|
|
name: INTERNODE_OPERATION_WRITE_SHUTDOWN_ERRORS_TOTAL,
|
|
labels: OPERATION_BACKEND_LABELS,
|
|
},
|
|
InternodeOperationMetricDescriptor {
|
|
name: ERASURE_WRITE_QUORUM_FAILURES_TOTAL,
|
|
labels: QUORUM_FAILURE_LABELS,
|
|
},
|
|
InternodeOperationMetricDescriptor {
|
|
name: INTERNODE_OPERATION_PAYLOAD_BYTES,
|
|
labels: OPERATION_BACKEND_LABELS,
|
|
},
|
|
InternodeOperationMetricDescriptor {
|
|
name: INTERNODE_OPERATION_LARGE_PAYLOADS_TOTAL,
|
|
labels: OPERATION_BACKEND_LABELS,
|
|
},
|
|
];
|
|
|
|
#[derive(Debug, Clone, Copy, Default, PartialEq, Eq)]
|
|
pub struct InternodeMetricsSnapshot {
|
|
pub sent_bytes_total: u64,
|
|
pub recv_bytes_total: u64,
|
|
pub outgoing_requests_total: u64,
|
|
pub incoming_requests_total: u64,
|
|
pub errors_total: u64,
|
|
pub dial_errors_total: u64,
|
|
pub dial_avg_time_nanos: u64,
|
|
pub last_dial_unix_millis: u64,
|
|
pub operation_http_versions_total: u64,
|
|
pub operation_stall_timeouts_total: u64,
|
|
pub operation_write_shutdown_errors_total: u64,
|
|
}
|
|
|
|
#[derive(Debug, Default)]
|
|
pub struct InternodeMetrics {
|
|
sent_bytes_total: AtomicU64,
|
|
recv_bytes_total: AtomicU64,
|
|
outgoing_requests_total: AtomicU64,
|
|
incoming_requests_total: AtomicU64,
|
|
errors_total: AtomicU64,
|
|
dial_errors_total: AtomicU64,
|
|
dial_total_time_nanos: AtomicU64,
|
|
dial_samples_total: AtomicU64,
|
|
last_dial_unix_millis: AtomicU64,
|
|
operation_http_versions_total: AtomicU64,
|
|
operation_stall_timeouts_total: AtomicU64,
|
|
operation_write_shutdown_errors_total: AtomicU64,
|
|
}
|
|
|
|
impl InternodeMetrics {
|
|
pub fn record_sent_bytes(&self, bytes: usize) {
|
|
let bytes = bytes as u64;
|
|
if bytes == 0 {
|
|
return;
|
|
}
|
|
self.sent_bytes_total.fetch_add(bytes, Ordering::Relaxed);
|
|
counter!("rustfs_system_network_internode_sent_bytes_total").increment(bytes);
|
|
}
|
|
|
|
pub fn record_sent_bytes_for_operation(&self, operation: &'static str, bytes: usize) {
|
|
self.record_sent_bytes_for_operation_and_backend(operation, INTERNODE_TRANSPORT_BACKEND_UNKNOWN, bytes);
|
|
}
|
|
|
|
pub fn record_sent_bytes_for_operation_and_backend(&self, operation: &'static str, backend: &'static str, bytes: usize) {
|
|
self.record_sent_bytes(bytes);
|
|
|
|
let bytes = bytes as u64;
|
|
if bytes == 0 {
|
|
return;
|
|
}
|
|
counter!(INTERNODE_OPERATION_SENT_BYTES_TOTAL, OPERATION_LABEL => operation, BACKEND_LABEL => backend).increment(bytes);
|
|
}
|
|
|
|
pub fn record_recv_bytes(&self, bytes: usize) {
|
|
let bytes = bytes as u64;
|
|
if bytes == 0 {
|
|
return;
|
|
}
|
|
self.recv_bytes_total.fetch_add(bytes, Ordering::Relaxed);
|
|
counter!("rustfs_system_network_internode_recv_bytes_total").increment(bytes);
|
|
}
|
|
|
|
pub fn record_recv_bytes_for_operation(&self, operation: &'static str, bytes: usize) {
|
|
self.record_recv_bytes_for_operation_and_backend(operation, INTERNODE_TRANSPORT_BACKEND_UNKNOWN, bytes);
|
|
}
|
|
|
|
pub fn record_recv_bytes_for_operation_and_backend(&self, operation: &'static str, backend: &'static str, bytes: usize) {
|
|
self.record_recv_bytes(bytes);
|
|
|
|
let bytes = bytes as u64;
|
|
if bytes == 0 {
|
|
return;
|
|
}
|
|
counter!(INTERNODE_OPERATION_RECV_BYTES_TOTAL, OPERATION_LABEL => operation, BACKEND_LABEL => backend).increment(bytes);
|
|
}
|
|
|
|
pub fn record_outgoing_request(&self) {
|
|
self.outgoing_requests_total.fetch_add(1, Ordering::Relaxed);
|
|
counter!("rustfs_system_network_internode_requests_outgoing_total").increment(1);
|
|
}
|
|
|
|
pub fn record_outgoing_request_for_operation(&self, operation: &'static str) {
|
|
self.record_outgoing_request_for_operation_and_backend(operation, INTERNODE_TRANSPORT_BACKEND_UNKNOWN);
|
|
}
|
|
|
|
pub fn record_outgoing_request_for_operation_and_backend(&self, operation: &'static str, backend: &'static str) {
|
|
self.record_outgoing_request();
|
|
counter!(INTERNODE_OPERATION_REQUESTS_OUTGOING_TOTAL, OPERATION_LABEL => operation, BACKEND_LABEL => backend)
|
|
.increment(1);
|
|
}
|
|
|
|
pub fn record_incoming_request(&self) {
|
|
self.incoming_requests_total.fetch_add(1, Ordering::Relaxed);
|
|
counter!("rustfs_system_network_internode_requests_incoming_total").increment(1);
|
|
}
|
|
|
|
pub fn record_incoming_request_for_operation(&self, operation: &'static str) {
|
|
self.record_incoming_request_for_operation_and_backend(operation, INTERNODE_TRANSPORT_BACKEND_UNKNOWN);
|
|
}
|
|
|
|
pub fn record_incoming_request_for_operation_and_backend(&self, operation: &'static str, backend: &'static str) {
|
|
self.record_incoming_request();
|
|
counter!(INTERNODE_OPERATION_REQUESTS_INCOMING_TOTAL, OPERATION_LABEL => operation, BACKEND_LABEL => backend)
|
|
.increment(1);
|
|
}
|
|
|
|
pub fn record_error(&self) {
|
|
self.errors_total.fetch_add(1, Ordering::Relaxed);
|
|
counter!("rustfs_system_network_internode_errors_total").increment(1);
|
|
}
|
|
|
|
pub fn record_error_for_operation(&self, operation: &'static str) {
|
|
self.record_error_for_operation_and_backend(operation, INTERNODE_TRANSPORT_BACKEND_UNKNOWN);
|
|
}
|
|
|
|
pub fn record_error_for_operation_and_backend(&self, operation: &'static str, backend: &'static str) {
|
|
self.record_error();
|
|
counter!(INTERNODE_OPERATION_ERRORS_TOTAL, OPERATION_LABEL => operation, BACKEND_LABEL => backend).increment(1);
|
|
}
|
|
|
|
pub fn record_duration_for_operation_and_backend(&self, operation: &'static str, backend: &'static str, duration: Duration) {
|
|
let duration_ms = duration.as_secs_f64() * 1000.0;
|
|
metrics::histogram!(INTERNODE_OPERATION_DURATION_MS, OPERATION_LABEL => operation, BACKEND_LABEL => backend)
|
|
.record(duration_ms);
|
|
}
|
|
|
|
pub fn record_classified_error_for_operation_and_backend(
|
|
&self,
|
|
operation: &'static str,
|
|
backend: &'static str,
|
|
classification: &'static str,
|
|
) {
|
|
counter!(
|
|
INTERNODE_OPERATION_CLASSIFIED_ERRORS_TOTAL,
|
|
OPERATION_LABEL => operation,
|
|
BACKEND_LABEL => backend,
|
|
CLASSIFICATION_LABEL => classification
|
|
)
|
|
.increment(1);
|
|
}
|
|
|
|
pub fn record_retry_for_operation_and_backend(
|
|
&self,
|
|
operation: &'static str,
|
|
backend: &'static str,
|
|
classification: &'static str,
|
|
) {
|
|
counter!(
|
|
INTERNODE_OPERATION_RETRIES_TOTAL,
|
|
OPERATION_LABEL => operation,
|
|
BACKEND_LABEL => backend,
|
|
CLASSIFICATION_LABEL => classification
|
|
)
|
|
.increment(1);
|
|
}
|
|
|
|
pub fn record_retry_success_for_operation_and_backend(
|
|
&self,
|
|
operation: &'static str,
|
|
backend: &'static str,
|
|
classification: &'static str,
|
|
) {
|
|
counter!(
|
|
INTERNODE_OPERATION_RETRY_SUCCESSES_TOTAL,
|
|
OPERATION_LABEL => operation,
|
|
BACKEND_LABEL => backend,
|
|
CLASSIFICATION_LABEL => classification
|
|
)
|
|
.increment(1);
|
|
}
|
|
|
|
pub fn record_http_version_for_operation_and_backend(
|
|
&self,
|
|
operation: &'static str,
|
|
backend: &'static str,
|
|
http_version: &'static str,
|
|
) {
|
|
self.operation_http_versions_total.fetch_add(1, Ordering::Relaxed);
|
|
counter!(
|
|
INTERNODE_OPERATION_HTTP_VERSIONS_TOTAL,
|
|
OPERATION_LABEL => operation,
|
|
BACKEND_LABEL => backend,
|
|
HTTP_VERSION_LABEL => http_version
|
|
)
|
|
.increment(1);
|
|
}
|
|
|
|
pub fn record_stall_timeout_for_operation_and_backend(&self, operation: &'static str, backend: &'static str) {
|
|
self.operation_stall_timeouts_total.fetch_add(1, Ordering::Relaxed);
|
|
counter!(INTERNODE_OPERATION_STALL_TIMEOUTS_TOTAL, OPERATION_LABEL => operation, BACKEND_LABEL => backend).increment(1);
|
|
}
|
|
|
|
pub fn record_write_shutdown_error_for_operation_and_backend(&self, operation: &'static str, backend: &'static str) {
|
|
self.operation_write_shutdown_errors_total.fetch_add(1, Ordering::Relaxed);
|
|
counter!(INTERNODE_OPERATION_WRITE_SHUTDOWN_ERRORS_TOTAL, OPERATION_LABEL => operation, BACKEND_LABEL => backend)
|
|
.increment(1);
|
|
}
|
|
|
|
/// Record the payload size (bytes) of a completed internode operation into a histogram
|
|
/// keyed by operation+backend. Used to size which unary `bytes`-carrying RPCs
|
|
/// (`ReadAll`/`ReadMultiple`/`WriteAll`) would benefit from being moved off the shared
|
|
/// control-plane channel (see docs/grpc-optimization P1).
|
|
pub fn record_operation_payload_bytes(&self, operation: &'static str, backend: &'static str, bytes: usize) {
|
|
metrics::histogram!(INTERNODE_OPERATION_PAYLOAD_BYTES, OPERATION_LABEL => operation, BACKEND_LABEL => backend)
|
|
.record(bytes as f64);
|
|
}
|
|
|
|
/// Increment the large-payload counter for an operation+backend whose payload exceeded the
|
|
/// caller-configured warning threshold. Feeds alerting on large unary RPCs that contend with
|
|
/// latency-sensitive control-plane traffic on the shared connection.
|
|
pub fn record_large_operation_payload(&self, operation: &'static str, backend: &'static str) {
|
|
counter!(INTERNODE_OPERATION_LARGE_PAYLOADS_TOTAL, OPERATION_LABEL => operation, BACKEND_LABEL => backend).increment(1);
|
|
}
|
|
|
|
/// Count a decode that fell back to the JSON compatibility field because the msgpack `_bin`
|
|
/// payload was absent. Both internode RPC directions dual-encode msgpack + JSON today; this
|
|
/// counter must read zero across a release window before the redundant JSON fields can be
|
|
/// dropped (grpc-optimization P2). `direction` is [`INTERNODE_MSGPACK_DIRECTION_REQUEST`] or
|
|
/// [`INTERNODE_MSGPACK_DIRECTION_RESPONSE`]; `message` is the low-cardinality value name.
|
|
pub fn record_msgpack_json_fallback(&self, direction: &'static str, message: &'static str) {
|
|
counter!(INTERNODE_MSGPACK_JSON_FALLBACK_TOTAL, DIRECTION_LABEL => direction, MESSAGE_LABEL => message).increment(1);
|
|
}
|
|
|
|
pub fn record_erasure_write_quorum_failure(&self, stage: &'static str, dominant_error: &'static str) {
|
|
counter!(
|
|
ERASURE_WRITE_QUORUM_FAILURES_TOTAL,
|
|
STAGE_LABEL => stage,
|
|
DOMINANT_ERROR_LABEL => dominant_error
|
|
)
|
|
.increment(1);
|
|
}
|
|
|
|
pub fn record_dial_result(&self, duration: Duration, success: bool) {
|
|
let elapsed_nanos = duration.as_nanos().min(u128::from(u64::MAX)) as u64;
|
|
self.dial_total_time_nanos.fetch_add(elapsed_nanos, Ordering::Relaxed);
|
|
let samples = self.dial_samples_total.fetch_add(1, Ordering::Relaxed) + 1;
|
|
let total = self.dial_total_time_nanos.load(Ordering::Relaxed);
|
|
gauge!("rustfs_system_network_internode_dial_avg_time_nanos").set(total as f64 / samples as f64);
|
|
|
|
if !success {
|
|
self.dial_errors_total.fetch_add(1, Ordering::Relaxed);
|
|
counter!("rustfs_system_network_internode_dial_errors_total").increment(1);
|
|
}
|
|
|
|
let now_ms = SystemTime::now()
|
|
.duration_since(UNIX_EPOCH)
|
|
.unwrap_or_default()
|
|
.as_millis()
|
|
.min(u128::from(u64::MAX)) as u64;
|
|
self.last_dial_unix_millis.store(now_ms, Ordering::Relaxed);
|
|
}
|
|
|
|
pub fn snapshot(&self) -> InternodeMetricsSnapshot {
|
|
let dial_samples_total = self.dial_samples_total.load(Ordering::Relaxed);
|
|
let dial_total_time_nanos = self.dial_total_time_nanos.load(Ordering::Relaxed);
|
|
let dial_avg_time_nanos = dial_total_time_nanos.checked_div(dial_samples_total).unwrap_or(0);
|
|
|
|
InternodeMetricsSnapshot {
|
|
sent_bytes_total: self.sent_bytes_total.load(Ordering::Relaxed),
|
|
recv_bytes_total: self.recv_bytes_total.load(Ordering::Relaxed),
|
|
outgoing_requests_total: self.outgoing_requests_total.load(Ordering::Relaxed),
|
|
incoming_requests_total: self.incoming_requests_total.load(Ordering::Relaxed),
|
|
errors_total: self.errors_total.load(Ordering::Relaxed),
|
|
dial_errors_total: self.dial_errors_total.load(Ordering::Relaxed),
|
|
dial_avg_time_nanos,
|
|
last_dial_unix_millis: self.last_dial_unix_millis.load(Ordering::Relaxed),
|
|
operation_http_versions_total: self.operation_http_versions_total.load(Ordering::Relaxed),
|
|
operation_stall_timeouts_total: self.operation_stall_timeouts_total.load(Ordering::Relaxed),
|
|
operation_write_shutdown_errors_total: self.operation_write_shutdown_errors_total.load(Ordering::Relaxed),
|
|
}
|
|
}
|
|
|
|
#[doc(hidden)]
|
|
pub fn reset_for_test(&self) {
|
|
self.sent_bytes_total.store(0, Ordering::Relaxed);
|
|
self.recv_bytes_total.store(0, Ordering::Relaxed);
|
|
self.outgoing_requests_total.store(0, Ordering::Relaxed);
|
|
self.incoming_requests_total.store(0, Ordering::Relaxed);
|
|
self.errors_total.store(0, Ordering::Relaxed);
|
|
self.dial_errors_total.store(0, Ordering::Relaxed);
|
|
self.dial_total_time_nanos.store(0, Ordering::Relaxed);
|
|
self.dial_samples_total.store(0, Ordering::Relaxed);
|
|
self.last_dial_unix_millis.store(0, Ordering::Relaxed);
|
|
self.operation_http_versions_total.store(0, Ordering::Relaxed);
|
|
self.operation_stall_timeouts_total.store(0, Ordering::Relaxed);
|
|
self.operation_write_shutdown_errors_total.store(0, Ordering::Relaxed);
|
|
}
|
|
}
|
|
|
|
pub fn global_internode_metrics() -> &'static Arc<InternodeMetrics> {
|
|
static GLOBAL_INTERNODE_METRICS: LazyLock<Arc<InternodeMetrics>> = LazyLock::new(|| Arc::new(InternodeMetrics::default()));
|
|
&GLOBAL_INTERNODE_METRICS
|
|
}
|
|
|
|
// ── Cluster peer online/offline health (grpc-optimization P3) ──
|
|
// Tracks reachability of each internode peer and exposes the count of offline peers as a gauge,
|
|
// for parity with MinIO's `minio_cluster_servers_offline_total`. This is pure observability: it
|
|
// does not change peer selection or quorum. A peer flips offline after a configured number of
|
|
// consecutive failures and back online on the next successful dial.
|
|
|
|
/// Gauge: number of internode peers currently considered offline.
|
|
const CLUSTER_SERVERS_OFFLINE_TOTAL: &str = "rustfs_cluster_servers_offline_total";
|
|
|
|
#[derive(Debug)]
|
|
struct PeerHealthState {
|
|
online: bool,
|
|
consecutive_failures: u32,
|
|
/// Last time a request was let through to re-probe an offline peer (grpc-optimization P3
|
|
/// offline bypass). `None` means "not yet re-probed since going offline".
|
|
last_reprobe: Option<Instant>,
|
|
}
|
|
|
|
impl Default for PeerHealthState {
|
|
fn default() -> Self {
|
|
// A newly observed peer is assumed online until it accrues failures.
|
|
Self {
|
|
online: true,
|
|
consecutive_failures: 0,
|
|
last_reprobe: None,
|
|
}
|
|
}
|
|
}
|
|
|
|
static CLUSTER_PEER_HEALTH: LazyLock<Mutex<HashMap<String, PeerHealthState>>> = LazyLock::new(|| Mutex::new(HashMap::new()));
|
|
|
|
fn publish_offline_gauge(peers: &HashMap<String, PeerHealthState>) {
|
|
let offline = peers.values().filter(|peer| !peer.online).count();
|
|
gauge!(CLUSTER_SERVERS_OFFLINE_TOTAL).set(offline as f64);
|
|
}
|
|
|
|
/// Canonicalize a peer address before using it as the `CLUSTER_PEER_HEALTH` key.
|
|
///
|
|
/// The same physical peer is referenced by different subsystems in slightly different string forms
|
|
/// — the data path keys by `endpoint.grid_host()` (`scheme://host:port`, no trailing slash) while
|
|
/// the lock path keys by `url::Url::to_string()` (`scheme://host:port/`, trailing slash). Without
|
|
/// normalization each form becomes its own health entry, so one downed node counts as 2 in the
|
|
/// `rustfs_cluster_servers_offline_total` gauge (and 2N for N nodes). Trimming trailing slashes
|
|
/// collapses them to a single canonical key.
|
|
fn normalize_peer_key(addr: &str) -> &str {
|
|
addr.trim_end_matches('/')
|
|
}
|
|
|
|
/// Record that a cluster peer is reachable: mark it online and reset its consecutive-failure
|
|
/// counter. Called on a successful dial to `addr`.
|
|
pub fn record_peer_reachable(addr: &str) {
|
|
// Recover from a poisoned lock so peer-health tracking and the offline gauge never stall permanently.
|
|
let mut peers = CLUSTER_PEER_HEALTH.lock().unwrap_or_else(|poisoned| poisoned.into_inner());
|
|
let entry = peers.entry(normalize_peer_key(addr).to_string()).or_default();
|
|
entry.online = true;
|
|
entry.consecutive_failures = 0;
|
|
entry.last_reprobe = None;
|
|
publish_offline_gauge(&peers);
|
|
}
|
|
|
|
/// Record a failed interaction with a cluster peer (dial failure or RPC-triggered eviction). After
|
|
/// `failure_threshold` (>= 1) consecutive failures the peer flips offline.
|
|
pub fn record_peer_unreachable(addr: &str, failure_threshold: u32) {
|
|
// Recover from a poisoned lock so peer-health tracking and the offline gauge never stall permanently.
|
|
let mut peers = CLUSTER_PEER_HEALTH.lock().unwrap_or_else(|poisoned| poisoned.into_inner());
|
|
let entry = peers.entry(normalize_peer_key(addr).to_string()).or_default();
|
|
entry.consecutive_failures = entry.consecutive_failures.saturating_add(1);
|
|
if entry.consecutive_failures >= failure_threshold.max(1) {
|
|
entry.online = false;
|
|
}
|
|
publish_offline_gauge(&peers);
|
|
}
|
|
|
|
/// Whether a cluster peer is currently considered offline (known and marked offline).
|
|
pub fn cluster_peer_is_offline(addr: &str) -> bool {
|
|
let peers = CLUSTER_PEER_HEALTH.lock().unwrap_or_else(|poisoned| poisoned.into_inner());
|
|
peers.get(normalize_peer_key(addr)).map(|peer| !peer.online).unwrap_or(false)
|
|
}
|
|
|
|
/// Decide whether to fast-fail (bypass) an offline peer instead of attempting to reach it
|
|
/// (grpc-optimization P3 offline bypass). Returns `true` to bypass.
|
|
///
|
|
/// Self-healing: for an offline peer this returns `true` most of the time, but lets one request
|
|
/// through every `reprobe_interval` (returning `false` and recording the re-probe time) so the peer
|
|
/// can recover via a normal dial even if no background monitor is running. Online peers are never
|
|
/// bypassed.
|
|
pub fn cluster_peer_should_bypass(addr: &str, reprobe_interval: Duration) -> bool {
|
|
let mut peers = CLUSTER_PEER_HEALTH.lock().unwrap_or_else(|poisoned| poisoned.into_inner());
|
|
let Some(entry) = peers.get_mut(normalize_peer_key(addr)) else {
|
|
return false;
|
|
};
|
|
if entry.online {
|
|
return false;
|
|
}
|
|
let now = Instant::now();
|
|
let due = match entry.last_reprobe {
|
|
None => true,
|
|
Some(last) => now.duration_since(last) >= reprobe_interval,
|
|
};
|
|
if due {
|
|
// Let this request through to re-probe; do not bypass.
|
|
entry.last_reprobe = Some(now);
|
|
false
|
|
} else {
|
|
true
|
|
}
|
|
}
|
|
|
|
#[cfg(test)]
|
|
fn cluster_peer_online(addr: &str) -> Option<bool> {
|
|
CLUSTER_PEER_HEALTH
|
|
.lock()
|
|
.ok()?
|
|
.get(normalize_peer_key(addr))
|
|
.map(|peer| peer.online)
|
|
}
|
|
|
|
#[cfg(test)]
|
|
fn cluster_peer_health_keys() -> Vec<String> {
|
|
CLUSTER_PEER_HEALTH
|
|
.lock()
|
|
.map(|peers| peers.keys().cloned().collect())
|
|
.unwrap_or_default()
|
|
}
|
|
|
|
#[cfg(test)]
|
|
mod tests {
|
|
use super::*;
|
|
|
|
#[test]
|
|
fn snapshot_reports_recorded_values() {
|
|
let metrics = global_internode_metrics();
|
|
metrics.reset_for_test();
|
|
|
|
metrics.record_sent_bytes(64);
|
|
metrics.record_recv_bytes(32);
|
|
metrics.record_outgoing_request();
|
|
metrics.record_incoming_request();
|
|
metrics.record_error();
|
|
metrics.record_dial_result(Duration::from_millis(9), true);
|
|
metrics.record_dial_result(Duration::from_millis(3), false);
|
|
|
|
let snapshot = metrics.snapshot();
|
|
assert_eq!(snapshot.sent_bytes_total, 64);
|
|
assert_eq!(snapshot.recv_bytes_total, 32);
|
|
assert_eq!(snapshot.outgoing_requests_total, 1);
|
|
assert_eq!(snapshot.incoming_requests_total, 1);
|
|
assert_eq!(snapshot.errors_total, 1);
|
|
assert_eq!(snapshot.dial_errors_total, 1);
|
|
assert_eq!(snapshot.dial_avg_time_nanos, 6_000_000);
|
|
assert!(snapshot.last_dial_unix_millis > 0);
|
|
|
|
metrics.reset_for_test();
|
|
}
|
|
|
|
#[test]
|
|
fn operation_metrics_also_update_aggregate_snapshot() {
|
|
let metrics = InternodeMetrics::default();
|
|
|
|
metrics.record_sent_bytes_for_operation_and_backend(
|
|
INTERNODE_OPERATION_READ_FILE_STREAM,
|
|
INTERNODE_TRANSPORT_BACKEND_TCP_HTTP,
|
|
128,
|
|
);
|
|
metrics.record_recv_bytes_for_operation_and_backend(
|
|
INTERNODE_OPERATION_PUT_FILE_STREAM,
|
|
INTERNODE_TRANSPORT_BACKEND_TCP_HTTP,
|
|
256,
|
|
);
|
|
metrics.record_outgoing_request_for_operation_and_backend(
|
|
INTERNODE_OPERATION_GRPC_WRITE_ALL,
|
|
INTERNODE_TRANSPORT_BACKEND_GRPC,
|
|
);
|
|
metrics.record_incoming_request_for_operation_and_backend(
|
|
INTERNODE_OPERATION_GRPC_READ_ALL,
|
|
INTERNODE_TRANSPORT_BACKEND_GRPC,
|
|
);
|
|
metrics.record_error_for_operation_and_backend(INTERNODE_OPERATION_WALK_DIR, INTERNODE_TRANSPORT_BACKEND_TCP_HTTP);
|
|
|
|
let snapshot = metrics.snapshot();
|
|
assert_eq!(snapshot.sent_bytes_total, 128);
|
|
assert_eq!(snapshot.recv_bytes_total, 256);
|
|
assert_eq!(snapshot.outgoing_requests_total, 1);
|
|
assert_eq!(snapshot.incoming_requests_total, 1);
|
|
assert_eq!(snapshot.errors_total, 1);
|
|
}
|
|
|
|
#[test]
|
|
fn operation_metric_descriptors_include_backend_and_operation_labels() {
|
|
assert_eq!(INTERNODE_OPERATION_METRICS.len(), 15);
|
|
for metric in &INTERNODE_OPERATION_METRICS[..6] {
|
|
assert_eq!(metric.labels, &[OPERATION_LABEL, BACKEND_LABEL]);
|
|
}
|
|
for metric in &INTERNODE_OPERATION_METRICS[6..9] {
|
|
assert_eq!(metric.labels, &[OPERATION_LABEL, BACKEND_LABEL, CLASSIFICATION_LABEL]);
|
|
}
|
|
assert_eq!(
|
|
INTERNODE_OPERATION_METRICS[9].labels,
|
|
&[OPERATION_LABEL, BACKEND_LABEL, HTTP_VERSION_LABEL]
|
|
);
|
|
for metric in &INTERNODE_OPERATION_METRICS[10..12] {
|
|
assert_eq!(metric.labels, &[OPERATION_LABEL, BACKEND_LABEL]);
|
|
}
|
|
assert_eq!(INTERNODE_OPERATION_METRICS[12].labels, &[STAGE_LABEL, DOMINANT_ERROR_LABEL]);
|
|
// Payload histogram + large-payload counter carry operation+backend labels.
|
|
assert_eq!(INTERNODE_OPERATION_METRICS[13].labels, &[OPERATION_LABEL, BACKEND_LABEL]);
|
|
assert_eq!(INTERNODE_OPERATION_METRICS[14].labels, &[OPERATION_LABEL, BACKEND_LABEL]);
|
|
}
|
|
|
|
#[test]
|
|
fn operation_metric_names_and_low_cardinality_values_are_stable() {
|
|
assert_eq!(INTERNODE_OPERATION_READ_FILE_STREAM, "read_file_stream");
|
|
assert_eq!(INTERNODE_OPERATION_PUT_FILE_STREAM, "put_file_stream");
|
|
assert_eq!(INTERNODE_OPERATION_WALK_DIR, "walk_dir");
|
|
assert_eq!(INTERNODE_OPERATION_GRPC_READ_ALL, "grpc_read_all");
|
|
assert_eq!(INTERNODE_OPERATION_GRPC_WRITE_ALL, "grpc_write_all");
|
|
|
|
assert_eq!(INTERNODE_TRANSPORT_BACKEND_TCP_HTTP, "tcp-http");
|
|
assert_eq!(INTERNODE_TRANSPORT_BACKEND_GRPC, "grpc");
|
|
assert_eq!(INTERNODE_TRANSPORT_BACKEND_UNKNOWN, "unknown");
|
|
|
|
assert_eq!(
|
|
INTERNODE_OPERATION_METRICS[5].name,
|
|
"rustfs_system_network_internode_operation_duration_ms"
|
|
);
|
|
assert_eq!(
|
|
INTERNODE_OPERATION_METRICS[6].name,
|
|
"rustfs_system_network_internode_operation_classified_errors_total"
|
|
);
|
|
assert_eq!(
|
|
INTERNODE_OPERATION_METRICS[7].name,
|
|
"rustfs_system_network_internode_operation_retries_total"
|
|
);
|
|
assert_eq!(
|
|
INTERNODE_OPERATION_METRICS[8].name,
|
|
"rustfs_system_network_internode_operation_retry_successes_total"
|
|
);
|
|
assert_eq!(
|
|
INTERNODE_OPERATION_METRICS[9].name,
|
|
"rustfs_system_network_internode_operation_http_versions_total"
|
|
);
|
|
assert_eq!(
|
|
INTERNODE_OPERATION_METRICS[10].name,
|
|
"rustfs_system_network_internode_operation_stall_timeouts_total"
|
|
);
|
|
assert_eq!(
|
|
INTERNODE_OPERATION_METRICS[11].name,
|
|
"rustfs_system_network_internode_operation_write_shutdown_errors_total"
|
|
);
|
|
assert_eq!(
|
|
INTERNODE_OPERATION_METRICS[12].name,
|
|
"rustfs_system_storage_erasure_write_quorum_failures_total"
|
|
);
|
|
assert_eq!(
|
|
INTERNODE_OPERATION_METRICS[13].name,
|
|
"rustfs_system_network_internode_operation_payload_bytes"
|
|
);
|
|
assert_eq!(
|
|
INTERNODE_OPERATION_METRICS[14].name,
|
|
"rustfs_system_network_internode_operation_large_payloads_total"
|
|
);
|
|
assert_eq!(INTERNODE_OPERATION_GRPC_READ_MULTIPLE, "grpc_read_multiple");
|
|
assert_eq!(
|
|
INTERNODE_MSGPACK_JSON_FALLBACK_TOTAL,
|
|
"rustfs_system_network_internode_msgpack_json_fallback_total"
|
|
);
|
|
assert_eq!(INTERNODE_MSGPACK_DIRECTION_REQUEST, "request");
|
|
assert_eq!(INTERNODE_MSGPACK_DIRECTION_RESPONSE, "response");
|
|
}
|
|
|
|
#[test]
|
|
fn msgpack_json_fallback_counter_records_without_panicking() {
|
|
// Smoke test: the counter accepts both directions and a static message label.
|
|
let metrics = InternodeMetrics::default();
|
|
metrics.record_msgpack_json_fallback(INTERNODE_MSGPACK_DIRECTION_REQUEST, "FileInfo");
|
|
metrics.record_msgpack_json_fallback(INTERNODE_MSGPACK_DIRECTION_RESPONSE, "RawFileInfo");
|
|
}
|
|
|
|
#[test]
|
|
fn cluster_peer_flips_offline_after_threshold_and_back_online() {
|
|
// Unique addr keeps this independent of the process-global registry / other tests.
|
|
let addr = "http://cluster-peer-health-unit-test:9000";
|
|
assert_eq!(cluster_peer_online(addr), None);
|
|
|
|
record_peer_unreachable(addr, 3);
|
|
record_peer_unreachable(addr, 3);
|
|
assert_eq!(cluster_peer_online(addr), Some(true), "still online below threshold");
|
|
|
|
record_peer_unreachable(addr, 3);
|
|
assert_eq!(cluster_peer_online(addr), Some(false), "offline at threshold");
|
|
|
|
record_peer_reachable(addr);
|
|
assert_eq!(cluster_peer_online(addr), Some(true), "back online after a reachable dial");
|
|
}
|
|
|
|
#[test]
|
|
fn cluster_peer_threshold_is_clamped_to_at_least_one() {
|
|
let addr = "http://cluster-peer-health-clamp-test:9000";
|
|
// A zero threshold must not mean "never offline"; one failure suffices.
|
|
record_peer_unreachable(addr, 0);
|
|
assert_eq!(cluster_peer_online(addr), Some(false));
|
|
record_peer_reachable(addr);
|
|
assert_eq!(cluster_peer_online(addr), Some(true));
|
|
}
|
|
|
|
#[test]
|
|
fn peer_health_key_is_normalized_across_address_forms() {
|
|
// Same physical peer, two address forms the codebase actually uses: the data path's
|
|
// grid_host() (no trailing slash) and the lock path's url::Url::to_string() (trailing slash).
|
|
// They MUST collapse to one health entry, else one downed node counts as 2 in the gauge.
|
|
let bare = "http://cluster-peer-health-normalize-test:9000";
|
|
let slashed = "http://cluster-peer-health-normalize-test:9000/";
|
|
|
|
record_peer_unreachable(bare, 1);
|
|
record_peer_unreachable(slashed, 1);
|
|
|
|
let keys: Vec<String> = cluster_peer_health_keys()
|
|
.into_iter()
|
|
.filter(|k| k.contains("cluster-peer-health-normalize-test"))
|
|
.collect();
|
|
assert_eq!(keys.len(), 1, "two address forms of one peer must be a single health entry, got {keys:?}");
|
|
|
|
// Either form observes the peer offline, and a reachable dial via one form clears the other.
|
|
assert!(cluster_peer_is_offline(bare));
|
|
assert!(cluster_peer_is_offline(slashed));
|
|
record_peer_reachable(slashed);
|
|
assert!(!cluster_peer_is_offline(bare), "reachable via one form must mark the shared entry online");
|
|
}
|
|
|
|
#[test]
|
|
fn cluster_servers_offline_total_name_is_stable() {
|
|
assert_eq!(CLUSTER_SERVERS_OFFLINE_TOTAL, "rustfs_cluster_servers_offline_total");
|
|
}
|
|
|
|
#[test]
|
|
fn cluster_peer_should_bypass_is_self_healing() {
|
|
let addr = "http://cluster-peer-bypass-selfheal-test:9000";
|
|
let interval = Duration::from_secs(30);
|
|
|
|
// Online peer: never bypassed.
|
|
record_peer_reachable(addr);
|
|
assert!(!cluster_peer_should_bypass(addr, interval));
|
|
|
|
// Take it offline (threshold 1 for the test).
|
|
record_peer_unreachable(addr, 1);
|
|
assert!(cluster_peer_is_offline(addr));
|
|
|
|
// First decision after going offline is a re-probe (not bypassed)...
|
|
assert!(!cluster_peer_should_bypass(addr, interval));
|
|
// ...and subsequent ones within the interval are bypassed.
|
|
assert!(cluster_peer_should_bypass(addr, interval));
|
|
assert!(cluster_peer_should_bypass(addr, interval));
|
|
|
|
// A zero interval always allows a re-probe (never strands the peer).
|
|
assert!(!cluster_peer_should_bypass(addr, Duration::ZERO));
|
|
|
|
// Recovery clears bypass entirely.
|
|
record_peer_reachable(addr);
|
|
assert!(!cluster_peer_should_bypass(addr, interval));
|
|
}
|
|
|
|
#[test]
|
|
fn cluster_peer_should_bypass_ignores_unknown_and_online_peers() {
|
|
let addr = "http://cluster-peer-bypass-unknown-test:9000";
|
|
// Unknown peer: not bypassed.
|
|
assert!(!cluster_peer_should_bypass(addr, Duration::from_secs(5)));
|
|
// Known-online peer: not bypassed.
|
|
record_peer_reachable(addr);
|
|
assert!(!cluster_peer_should_bypass(addr, Duration::from_secs(5)));
|
|
}
|
|
|
|
#[test]
|
|
fn classified_and_retry_metrics_update_counters() {
|
|
let metrics = InternodeMetrics::default();
|
|
|
|
metrics.record_classified_error_for_operation_and_backend(
|
|
INTERNODE_OPERATION_PUT_FILE_STREAM,
|
|
INTERNODE_TRANSPORT_BACKEND_TCP_HTTP,
|
|
"connection_reset",
|
|
);
|
|
metrics.record_retry_for_operation_and_backend(
|
|
INTERNODE_OPERATION_PUT_FILE_STREAM,
|
|
INTERNODE_TRANSPORT_BACKEND_TCP_HTTP,
|
|
"connection_reset",
|
|
);
|
|
metrics.record_retry_success_for_operation_and_backend(
|
|
INTERNODE_OPERATION_PUT_FILE_STREAM,
|
|
INTERNODE_TRANSPORT_BACKEND_TCP_HTTP,
|
|
"connection_reset",
|
|
);
|
|
metrics.record_http_version_for_operation_and_backend(
|
|
INTERNODE_OPERATION_PUT_FILE_STREAM,
|
|
INTERNODE_TRANSPORT_BACKEND_TCP_HTTP,
|
|
"http/1.1",
|
|
);
|
|
metrics.record_stall_timeout_for_operation_and_backend(
|
|
INTERNODE_OPERATION_READ_FILE_STREAM,
|
|
INTERNODE_TRANSPORT_BACKEND_TCP_HTTP,
|
|
);
|
|
metrics.record_write_shutdown_error_for_operation_and_backend(
|
|
INTERNODE_OPERATION_PUT_FILE_STREAM,
|
|
INTERNODE_TRANSPORT_BACKEND_TCP_HTTP,
|
|
);
|
|
metrics.record_erasure_write_quorum_failure("write", "connection_reset");
|
|
|
|
let snapshot = metrics.snapshot();
|
|
assert_eq!(snapshot.sent_bytes_total, 0);
|
|
assert_eq!(snapshot.recv_bytes_total, 0);
|
|
assert_eq!(snapshot.outgoing_requests_total, 0);
|
|
assert_eq!(snapshot.incoming_requests_total, 0);
|
|
assert_eq!(snapshot.operation_http_versions_total, 1);
|
|
assert_eq!(snapshot.operation_stall_timeouts_total, 1);
|
|
assert_eq!(snapshot.operation_write_shutdown_errors_total, 1);
|
|
}
|
|
}
|