refactor(logging): standardize protocol and observability events (#3419)

* refactor(logging): standardize object capacity events

* refactor(logging): standardize protocol server events

* refactor(logging): standardize swift protocol events

* refactor(logging): standardize observability events

* refactor(logging): move masking helper and extend guardrails
This commit is contained in:
houseme
2026-06-14 07:14:45 +08:00
committed by GitHub
parent 22460243bf
commit efa89a98ed
48 changed files with 2976 additions and 434 deletions
@@ -47,6 +47,10 @@ use thiserror::Error;
use tracing::warn;
const LOG_COMPONENT_OBS: &str = "obs";
const LOG_SUBSYSTEM_GPU_METRICS: &str = "gpu_metrics";
const EVENT_GPU_METRICS_STATE: &str = "gpu_metrics_state";
/// GPU statistics.
///
/// Contains GPU memory usage metrics for the monitored process.
@@ -138,7 +142,14 @@ impl GpuCollector {
}
}
} else {
warn!("Could not get GPU stats, recording 0 for GPU memory usage");
warn!(
event = EVENT_GPU_METRICS_STATE,
component = LOG_COMPONENT_OBS,
subsystem = LOG_SUBSYSTEM_GPU_METRICS,
result = "process_stats_unavailable",
fallback_memory_usage = 0,
"gpu metrics state changed"
);
}
} else {
return Err(GpuError::DeviceError("No GPU device found".to_string()));
+19 -15
View File
@@ -94,6 +94,10 @@ use tokio::time::Instant;
use tokio_util::sync::CancellationToken;
use tracing::warn;
const LOG_COMPONENT_OBS: &str = "obs";
const LOG_SUBSYSTEM_METRICS_RUNTIME: &str = "metrics_runtime";
const EVENT_METRICS_RUNTIME_STATE: &str = "metrics_runtime_state";
/// Default interval for system monitoring metrics (15 seconds)
const DEFAULT_SYSTEM_METRICS_INTERVAL: Duration = Duration::from_secs(15);
/// Environment variable for system monitoring interval
@@ -497,7 +501,7 @@ pub fn init_metrics_runtime(token: CancellationToken) {
report_metrics(&metrics);
}
_ = token_clone.cancelled() => {
warn!("Metrics collection for cluster stats cancelled.");
warn!(event = EVENT_METRICS_RUNTIME_STATE, component = LOG_COMPONENT_OBS, subsystem = LOG_SUBSYSTEM_METRICS_RUNTIME, collector = "cluster_stats", state = "cancelled", "metrics runtime state changed");
return;
}
}
@@ -537,7 +541,7 @@ pub fn init_metrics_runtime(token: CancellationToken) {
}
}
_ = token_clone.cancelled() => {
warn!("Metrics collection for supplementary cluster stats cancelled.");
warn!(event = EVENT_METRICS_RUNTIME_STATE, component = LOG_COMPONENT_OBS, subsystem = LOG_SUBSYSTEM_METRICS_RUNTIME, collector = "supplementary_cluster_stats", state = "cancelled", "metrics runtime state changed");
return;
}
}
@@ -556,7 +560,7 @@ pub fn init_metrics_runtime(token: CancellationToken) {
report_metrics(&metrics);
}
_ = token_clone.cancelled() => {
warn!("Metrics collection for bucket stats cancelled.");
warn!(event = EVENT_METRICS_RUNTIME_STATE, component = LOG_COMPONENT_OBS, subsystem = LOG_SUBSYSTEM_METRICS_RUNTIME, collector = "bucket_stats", state = "cancelled", "metrics runtime state changed");
return;
}
}
@@ -577,7 +581,7 @@ pub fn init_metrics_runtime(token: CancellationToken) {
report_metrics(&metrics);
}
_ = token_clone.cancelled() => {
warn!("Metrics collection for node/disk stats cancelled.");
warn!(event = EVENT_METRICS_RUNTIME_STATE, component = LOG_COMPONENT_OBS, subsystem = LOG_SUBSYSTEM_METRICS_RUNTIME, collector = "node_disk_stats", state = "cancelled", "metrics runtime state changed");
return;
}
}
@@ -601,7 +605,7 @@ pub fn init_metrics_runtime(token: CancellationToken) {
let current_live_keys = repl_bw_live_keys(&stats);
if !monitor_available {
warn!("Bucket monitor unavailable; skip replication bandwidth key-state transition this cycle.");
warn!(event = EVENT_METRICS_RUNTIME_STATE, component = LOG_COMPONENT_OBS, subsystem = LOG_SUBSYSTEM_METRICS_RUNTIME, collector = "bucket_replication_bandwidth", result = "bucket_monitor_unavailable", "metrics runtime state changed");
}
update_repl_bw_zero_tombstones(
monitor_available,
@@ -626,7 +630,7 @@ pub fn init_metrics_runtime(token: CancellationToken) {
expire_repl_bw_zero_tombstones(monitor_available, &mut zero_tombstones);
}
_ = token_clone.cancelled() => {
warn!("Metrics collection for bucket replication bandwidth stats cancelled.");
warn!(event = EVENT_METRICS_RUNTIME_STATE, component = LOG_COMPONENT_OBS, subsystem = LOG_SUBSYSTEM_METRICS_RUNTIME, collector = "bucket_replication_bandwidth", state = "cancelled", "metrics runtime state changed");
return;
}
}
@@ -653,7 +657,7 @@ pub fn init_metrics_runtime(token: CancellationToken) {
report_metrics(&metrics);
}
_ = token_clone.cancelled() => {
warn!("Metrics collection for audit target stats cancelled.");
warn!(event = EVENT_METRICS_RUNTIME_STATE, component = LOG_COMPONENT_OBS, subsystem = LOG_SUBSYSTEM_METRICS_RUNTIME, collector = "audit_target_stats", state = "cancelled", "metrics runtime state changed");
return;
}
}
@@ -689,7 +693,7 @@ pub fn init_metrics_runtime(token: CancellationToken) {
report_metrics(&metrics);
}
_ = token_clone.cancelled() => {
warn!("Metrics collection for notification stats cancelled.");
warn!(event = EVENT_METRICS_RUNTIME_STATE, component = LOG_COMPONENT_OBS, subsystem = LOG_SUBSYSTEM_METRICS_RUNTIME, collector = "notification_stats", state = "cancelled", "metrics runtime state changed");
return;
}
}
@@ -718,7 +722,7 @@ pub fn init_metrics_runtime(token: CancellationToken) {
}
}
_ = token_clone.cancelled() => {
warn!("Metrics collection for background workflow stats cancelled.");
warn!(event = EVENT_METRICS_RUNTIME_STATE, component = LOG_COMPONENT_OBS, subsystem = LOG_SUBSYSTEM_METRICS_RUNTIME, collector = "background_workflow_stats", state = "cancelled", "metrics runtime state changed");
return;
}
}
@@ -742,7 +746,7 @@ pub fn init_metrics_runtime(token: CancellationToken) {
let current_pid = match sysinfo::get_current_pid() {
Ok(pid) => Some(pid),
Err(e) => {
warn!("Failed to get current PID for system monitoring: {}", e);
warn!(event = EVENT_METRICS_RUNTIME_STATE, component = LOG_COMPONENT_OBS, subsystem = LOG_SUBSYSTEM_METRICS_RUNTIME, collector = "system_monitoring", result = "current_pid_unavailable", error = %e, "metrics runtime state changed");
None
}
};
@@ -776,11 +780,11 @@ pub fn init_metrics_runtime(token: CancellationToken) {
metrics.extend(collect_gpu_metrics(&gpu_stats, &labels));
}
Err(e) => {
warn!("GPU metrics collection failed: {}", e);
warn!(event = EVENT_METRICS_RUNTIME_STATE, component = LOG_COMPONENT_OBS, subsystem = LOG_SUBSYSTEM_METRICS_RUNTIME, collector = "gpu_metrics", result = "collect_failed", error = %e, "metrics runtime state changed");
}
},
Err(e) => {
warn!("GPU collector initialization failed: {}", e);
warn!(event = EVENT_METRICS_RUNTIME_STATE, component = LOG_COMPONENT_OBS, subsystem = LOG_SUBSYSTEM_METRICS_RUNTIME, collector = "gpu_metrics", result = "collector_init_failed", error = %e, "metrics runtime state changed");
}
}
}
@@ -790,7 +794,7 @@ pub fn init_metrics_runtime(token: CancellationToken) {
}
}
_ = token_clone.cancelled() => {
warn!("Process metrics collection cancelled.");
warn!(event = EVENT_METRICS_RUNTIME_STATE, component = LOG_COMPONENT_OBS, subsystem = LOG_SUBSYSTEM_METRICS_RUNTIME, collector = "process_metrics", state = "cancelled", "metrics runtime state changed");
return;
}
}
@@ -812,7 +816,7 @@ pub fn init_metrics_runtime(token: CancellationToken) {
}
}
_ = token_clone.cancelled() => {
warn!("Metrics collection for internode network stats cancelled.");
warn!(event = EVENT_METRICS_RUNTIME_STATE, component = LOG_COMPONENT_OBS, subsystem = LOG_SUBSYSTEM_METRICS_RUNTIME, collector = "internode_network_stats", state = "cancelled", "metrics runtime state changed");
return;
}
}
@@ -865,7 +869,7 @@ fn current_process_metric_labels() -> Vec<(&'static str, Cow<'static, str>)> {
}
fn fallback_process_metric_labels(err: ProcessAttributeError) -> Vec<(&'static str, Cow<'static, str>)> {
warn!("Failed to collect process attributes for metrics labels: {}", err);
warn!(event = EVENT_METRICS_RUNTIME_STATE, component = LOG_COMPONENT_OBS, subsystem = LOG_SUBSYSTEM_METRICS_RUNTIME, collector = "process_metric_labels", result = "collect_failed", error = %err, "metrics runtime state changed");
vec![
("process_pid", Cow::Owned(std::process::id().to_string())),
("process_executable_name", Cow::Borrowed("unknown")),
+9 -8
View File
@@ -46,6 +46,10 @@ use std::time::Duration;
use sysinfo::{Networks, System};
use tracing::{instrument, warn};
const LOG_COMPONENT_OBS: &str = "obs";
const LOG_SUBSYSTEM_METRICS_COLLECTOR: &str = "metrics_collector";
const EVENT_METRICS_COLLECTOR_STATE: &str = "metrics_collector_state";
fn current_scanner_cycle_age_seconds(
current_cycle: u64,
current_started: chrono::DateTime<Utc>,
@@ -178,7 +182,7 @@ pub async fn collect_cluster_and_health_stats() -> (ClusterStats, ClusterHealthS
let (buckets_count, objects_count) = match load_data_usage_from_backend(store.clone()).await {
Ok(data_usage) => (data_usage.buckets_count, data_usage.objects_total_count),
Err(e) => {
warn!("Failed to load data usage from backend: {}", e);
warn!(event = EVENT_METRICS_COLLECTOR_STATE, component = LOG_COMPONENT_OBS, subsystem = LOG_SUBSYSTEM_METRICS_COLLECTOR, collector = "cluster_stats", result = "data_usage_load_failed", error = %e, "metrics collector state changed");
// Fall back to bucket list for buckets_count, objects_count stays 0.
let buckets = store
.list_bucket(&BucketOptions {
@@ -187,7 +191,7 @@ pub async fn collect_cluster_and_health_stats() -> (ClusterStats, ClusterHealthS
})
.await
.unwrap_or_else(|err| {
warn!("Failed to list buckets for cluster metrics: {}", err);
warn!(event = EVENT_METRICS_COLLECTOR_STATE, component = LOG_COMPONENT_OBS, subsystem = LOG_SUBSYSTEM_METRICS_COLLECTOR, collector = "cluster_stats", result = "bucket_list_failed", error = %err, "metrics collector state changed");
Vec::new()
});
(buckets.len() as u64, 0)
@@ -246,7 +250,7 @@ pub async fn collect_bucket_stats() -> Vec<BucketStats> {
let data_usage = match load_data_usage_from_backend(store.clone()).await {
Ok(info) => Some(info),
Err(e) => {
warn!("Failed to load data usage for bucket metrics: {}", e);
warn!(event = EVENT_METRICS_COLLECTOR_STATE, component = LOG_COMPONENT_OBS, subsystem = LOG_SUBSYSTEM_METRICS_COLLECTOR, collector = "bucket_stats", result = "data_usage_load_failed", error = %e, "metrics collector state changed");
None
}
};
@@ -261,7 +265,7 @@ pub async fn collect_bucket_stats() -> Vec<BucketStats> {
{
Ok(buckets) => buckets,
Err(e) => {
warn!("Failed to list buckets for bucket metrics: {}", e);
warn!(event = EVENT_METRICS_COLLECTOR_STATE, component = LOG_COMPONENT_OBS, subsystem = LOG_SUBSYSTEM_METRICS_COLLECTOR, collector = "bucket_stats", result = "bucket_list_failed", error = %e, "metrics collector state changed");
return Vec::new();
}
};
@@ -310,10 +314,7 @@ pub fn collect_bucket_replication_bandwidth_stats() -> Vec<BucketReplicationBand
.map(|(opts, details)| {
let target_arn = opts.replication_arn;
let limit_bytes_per_sec = u64::try_from(details.limit_bytes_per_sec).unwrap_or_else(|_| {
warn!(
"Invalid bandwidth limit value for target {:?}: {}",
target_arn, details.limit_bytes_per_sec
);
warn!(event = EVENT_METRICS_COLLECTOR_STATE, component = LOG_COMPONENT_OBS, subsystem = LOG_SUBSYSTEM_METRICS_COLLECTOR, collector = "bucket_replication_bandwidth", result = "invalid_limit_value", target_arn = ?target_arn, limit_value = details.limit_bytes_per_sec, "metrics collector state changed");
0
});