mirror of
https://github.com/rustfs/rustfs.git
synced 2026-08-29 16:37:07 +00:00
fix(connect): use persisted inventory for offline collectors (#6560)
This commit is contained in:
@@ -232,6 +232,30 @@ impl InventorySnapshot {
|
|||||||
Ok(hex_simd::encode_to_string(digest.finalize(), hex_simd::AsciiCase::Lower))
|
Ok(hex_simd::encode_to_string(digest.finalize(), hex_simd::AsciiCase::Lower))
|
||||||
}
|
}
|
||||||
|
|
||||||
|
pub(crate) fn rustfs_version(&self) -> &str {
|
||||||
|
&self.rustfs_version
|
||||||
|
}
|
||||||
|
|
||||||
|
pub(crate) fn node_count(&self) -> u16 {
|
||||||
|
self.node_count
|
||||||
|
}
|
||||||
|
|
||||||
|
pub(crate) fn drive_count(&self) -> u32 {
|
||||||
|
self.drive_count
|
||||||
|
}
|
||||||
|
|
||||||
|
pub(crate) fn capacity_total_bytes(&self) -> u64 {
|
||||||
|
self.capacity_total_bytes
|
||||||
|
}
|
||||||
|
|
||||||
|
pub(crate) fn capacity_used_bytes(&self) -> u64 {
|
||||||
|
self.capacity_used_bytes
|
||||||
|
}
|
||||||
|
|
||||||
|
pub(crate) fn coarse_flags(&self) -> &[InventoryFlag] {
|
||||||
|
&self.coarse_flags
|
||||||
|
}
|
||||||
|
|
||||||
fn validate(&self) -> Result<(), InventoryError> {
|
fn validate(&self) -> Result<(), InventoryError> {
|
||||||
if !valid_version(&self.rustfs_version) {
|
if !valid_version(&self.rustfs_version) {
|
||||||
return Err(InventoryError::RustfsVersion);
|
return Err(InventoryError::RustfsVersion);
|
||||||
|
|||||||
@@ -15,24 +15,23 @@
|
|||||||
//! Fixed Q07 L0/L1 collectors for an operator-triggered offline diagnostic.
|
//! Fixed Q07 L0/L1 collectors for an operator-triggered offline diagnostic.
|
||||||
|
|
||||||
use std::collections::BTreeSet;
|
use std::collections::BTreeSet;
|
||||||
|
use std::path::Path;
|
||||||
use std::sync::{Arc, LazyLock};
|
use std::sync::{Arc, LazyLock};
|
||||||
use std::time::Duration;
|
use std::time::Duration;
|
||||||
|
|
||||||
use rustfs_madmin::{ITEM_OFFLINE, StorageInfo};
|
|
||||||
use serde::Serialize;
|
use serde::Serialize;
|
||||||
use serde_json::{Value, json};
|
use serde_json::{Value, json};
|
||||||
use sysinfo::{Disks, Networks, RefreshKind, System};
|
use sysinfo::{Disks, Networks, RefreshKind, System};
|
||||||
use thiserror::Error;
|
use thiserror::Error;
|
||||||
use tokio::sync::Semaphore;
|
use tokio::sync::Semaphore;
|
||||||
use tokio_util::sync::CancellationToken;
|
use tokio_util::sync::CancellationToken;
|
||||||
use url::Url;
|
|
||||||
|
|
||||||
|
use super::super::inventory::{InventoryError, InventorySnapshot, InventoryStateStore};
|
||||||
use super::manifest_entry::ManifestEntry;
|
use super::manifest_entry::ManifestEntry;
|
||||||
use super::redaction::RedactionError;
|
use super::redaction::RedactionError;
|
||||||
|
|
||||||
const COLLECT_TIMEOUT: Duration = Duration::from_secs(2);
|
const COLLECT_TIMEOUT: Duration = Duration::from_secs(2);
|
||||||
const MAX_ENTRY_BYTES: usize = 16 * 1024;
|
const MAX_ENTRY_BYTES: usize = 16 * 1024;
|
||||||
const MAX_DRIVES: usize = 4_096;
|
|
||||||
static SYSTEM_SCAN_PERMIT: LazyLock<Arc<Semaphore>> = LazyLock::new(|| Arc::new(Semaphore::new(1)));
|
static SYSTEM_SCAN_PERMIT: LazyLock<Arc<Semaphore>> = LazyLock::new(|| Arc::new(Semaphore::new(1)));
|
||||||
|
|
||||||
#[derive(Clone, Copy, Debug, PartialEq, Eq, Serialize)]
|
#[derive(Clone, Copy, Debug, PartialEq, Eq, Serialize)]
|
||||||
@@ -119,19 +118,14 @@ impl OfflineCollector {
|
|||||||
self.field_id().split_once('.').expect("collector field ids are frozen").1
|
self.field_id().split_once('.').expect("collector field ids are frozen").1
|
||||||
}
|
}
|
||||||
|
|
||||||
fn value(self, storage: &StorageSnapshot, system: &SystemSnapshot) -> Value {
|
fn value(self, inventory: &InventorySnapshot, system: &SystemSnapshot) -> Value {
|
||||||
match self {
|
match self {
|
||||||
Self::RustfsVersion => json!(env!("CARGO_PKG_VERSION")),
|
Self::RustfsVersion => json!(inventory.rustfs_version()),
|
||||||
Self::NodeCount => json!(storage.node_count),
|
Self::NodeCount => json!(inventory.node_count()),
|
||||||
Self::DriveCount => json!(storage.drive_count),
|
Self::DriveCount => json!(inventory.drive_count()),
|
||||||
Self::CapacityUsedBytes => json!(storage.capacity_used_bytes),
|
Self::CapacityUsedBytes => json!(inventory.capacity_used_bytes()),
|
||||||
Self::CapacityTotalBytes => json!(storage.capacity_total_bytes),
|
Self::CapacityTotalBytes => json!(inventory.capacity_total_bytes()),
|
||||||
Self::CoarseHealthFlags => json!({
|
Self::CoarseHealthFlags => json!(inventory.coarse_flags()),
|
||||||
"degraded": storage.degraded,
|
|
||||||
"healing": storage.healing,
|
|
||||||
"offlineDrives": storage.offline_drives,
|
|
||||||
"scanning": storage.scanning,
|
|
||||||
}),
|
|
||||||
Self::OsSummary => json!(system.os_summary),
|
Self::OsSummary => json!(system.os_summary),
|
||||||
Self::KernelSummary => json!(system.kernel_summary),
|
Self::KernelSummary => json!(system.kernel_summary),
|
||||||
Self::CpuSummary => json!({ "architecture": system.architecture, "cores": system.cores }),
|
Self::CpuSummary => json!({ "architecture": system.architecture, "cores": system.cores }),
|
||||||
@@ -156,68 +150,22 @@ pub enum CollectorError {
|
|||||||
TimedOut,
|
TimedOut,
|
||||||
#[error("offline diagnostic collector task failed")]
|
#[error("offline diagnostic collector task failed")]
|
||||||
TaskFailed,
|
TaskFailed,
|
||||||
#[error("offline diagnostic storage topology exceeds its 4096 drive budget")]
|
|
||||||
StorageTopologyTooLarge,
|
|
||||||
#[error("offline diagnostic storage topology contains an invalid endpoint")]
|
|
||||||
InvalidStorageEndpoint,
|
|
||||||
#[error("offline diagnostic field {field_id} exceeds its {limit} byte entry budget")]
|
#[error("offline diagnostic field {field_id} exceeds its {limit} byte entry budget")]
|
||||||
EntryTooLarge { field_id: &'static str, limit: usize },
|
EntryTooLarge { field_id: &'static str, limit: usize },
|
||||||
#[error("offline diagnostic entry is not representable as JSON")]
|
#[error("offline diagnostic entry is not representable as JSON")]
|
||||||
NotRepresentable,
|
NotRepresentable,
|
||||||
#[error(transparent)]
|
#[error(transparent)]
|
||||||
|
Inventory(#[from] InventoryError),
|
||||||
|
#[error(transparent)]
|
||||||
Redaction(#[from] RedactionError),
|
Redaction(#[from] RedactionError),
|
||||||
}
|
}
|
||||||
|
|
||||||
#[derive(Debug)]
|
/// The bounded entries plus the capture time of their persisted L0 source.
|
||||||
struct StorageSnapshot {
|
#[derive(Debug, PartialEq)]
|
||||||
node_count: usize,
|
pub struct OfflineDiagnostics {
|
||||||
drive_count: usize,
|
pub entries: Vec<ManifestEntry>,
|
||||||
capacity_used_bytes: u64,
|
pub inventory_captured_at: String,
|
||||||
capacity_total_bytes: u64,
|
pub inventory_age: Duration,
|
||||||
offline_drives: usize,
|
|
||||||
degraded: bool,
|
|
||||||
healing: bool,
|
|
||||||
scanning: bool,
|
|
||||||
}
|
|
||||||
|
|
||||||
impl TryFrom<&StorageInfo> for StorageSnapshot {
|
|
||||||
type Error = CollectorError;
|
|
||||||
|
|
||||||
fn try_from(info: &StorageInfo) -> Result<Self, Self::Error> {
|
|
||||||
if info.disks.len() > MAX_DRIVES {
|
|
||||||
return Err(CollectorError::StorageTopologyTooLarge);
|
|
||||||
}
|
|
||||||
let node_count = info
|
|
||||||
.disks
|
|
||||||
.iter()
|
|
||||||
.map(|disk| {
|
|
||||||
let endpoint = Url::parse(&disk.endpoint).map_err(|_| CollectorError::InvalidStorageEndpoint)?;
|
|
||||||
let host = endpoint.host_str().ok_or(CollectorError::InvalidStorageEndpoint)?;
|
|
||||||
Ok((host.to_owned(), endpoint.port_or_known_default()))
|
|
||||||
})
|
|
||||||
.collect::<Result<BTreeSet<_>, CollectorError>>()?
|
|
||||||
.len();
|
|
||||||
let offline_drives = info.disks.iter().filter(|disk| disk.state == ITEM_OFFLINE).count();
|
|
||||||
Ok(Self {
|
|
||||||
node_count,
|
|
||||||
drive_count: info.disks.len(),
|
|
||||||
capacity_used_bytes: info
|
|
||||||
.disks
|
|
||||||
.iter()
|
|
||||||
.fold(0_u64, |total, disk| total.saturating_add(disk.used_space)),
|
|
||||||
capacity_total_bytes: info
|
|
||||||
.disks
|
|
||||||
.iter()
|
|
||||||
.fold(0_u64, |total, disk| total.saturating_add(disk.total_space)),
|
|
||||||
offline_drives,
|
|
||||||
degraded: info
|
|
||||||
.disks
|
|
||||||
.iter()
|
|
||||||
.any(|disk| !matches!(disk.state.as_str(), "ok" | "unformatted" | rustfs_madmin::ITEM_ONLINE)),
|
|
||||||
healing: info.disks.iter().any(|disk| disk.healing),
|
|
||||||
scanning: info.disks.iter().any(|disk| disk.scanning),
|
|
||||||
})
|
|
||||||
}
|
|
||||||
}
|
}
|
||||||
|
|
||||||
#[derive(Debug)]
|
#[derive(Debug)]
|
||||||
@@ -293,23 +241,33 @@ async fn collect_system_snapshot(cancel: &CancellationToken) -> Result<SystemSna
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
/// Collect all and only the Q07 offline L0/L1 fields, with one independently
|
/// Collect all and only the Q07 offline L0/L1 fields after acquiring the
|
||||||
/// bounded and redacted manifest entry per field.
|
/// stopped-runtime inventory lock.
|
||||||
pub async fn collect_offline_diagnostics(
|
pub async fn collect_offline_diagnostics(
|
||||||
storage_info: &StorageInfo,
|
state_root: &Path,
|
||||||
cancel: &CancellationToken,
|
cancel: &CancellationToken,
|
||||||
) -> Result<Vec<ManifestEntry>, CollectorError> {
|
) -> Result<OfflineDiagnostics, CollectorError> {
|
||||||
if cancel.is_cancelled() {
|
if cancel.is_cancelled() {
|
||||||
return Err(CollectorError::Cancelled);
|
return Err(CollectorError::Cancelled);
|
||||||
}
|
}
|
||||||
let storage = StorageSnapshot::try_from(storage_info)?;
|
let store = InventoryStateStore::from_state_root(state_root)?;
|
||||||
|
let _lock = store.try_runtime_lock()?;
|
||||||
|
let persisted = store.read_latest(chrono::Utc::now())?;
|
||||||
let system = collect_system_snapshot(cancel).await?;
|
let system = collect_system_snapshot(cancel).await?;
|
||||||
|
|
||||||
let mut entries = Vec::with_capacity(COLLECTORS.len());
|
let mut entries = Vec::with_capacity(COLLECTORS.len());
|
||||||
for collector in COLLECTORS {
|
for collector in COLLECTORS {
|
||||||
entries.push(ManifestEntry::from_value(collector, collector.value(&storage, &system), cancel)?);
|
entries.push(ManifestEntry::from_value(
|
||||||
|
collector,
|
||||||
|
collector.value(&persisted.snapshot, &system),
|
||||||
|
cancel,
|
||||||
|
)?);
|
||||||
}
|
}
|
||||||
Ok(entries)
|
Ok(OfflineDiagnostics {
|
||||||
|
entries,
|
||||||
|
inventory_captured_at: persisted.captured_at,
|
||||||
|
inventory_age: persisted.age,
|
||||||
|
})
|
||||||
}
|
}
|
||||||
|
|
||||||
#[cfg(test)]
|
#[cfg(test)]
|
||||||
@@ -346,8 +304,6 @@ mod test_support {
|
|||||||
mod tests {
|
mod tests {
|
||||||
use std::sync::atomic::Ordering;
|
use std::sync::atomic::Ordering;
|
||||||
|
|
||||||
use rustfs_madmin::StorageInfo;
|
|
||||||
|
|
||||||
use super::*;
|
use super::*;
|
||||||
|
|
||||||
async fn wait_for_active(expected: usize) {
|
async fn wait_for_active(expected: usize) {
|
||||||
@@ -366,16 +322,13 @@ mod tests {
|
|||||||
test_support::DELAY_MILLIS.store((COLLECT_TIMEOUT + Duration::from_millis(200)).as_millis() as u64, Ordering::SeqCst);
|
test_support::DELAY_MILLIS.store((COLLECT_TIMEOUT + Duration::from_millis(200)).as_millis() as u64, Ordering::SeqCst);
|
||||||
|
|
||||||
let first_cancel = CancellationToken::new();
|
let first_cancel = CancellationToken::new();
|
||||||
assert!(matches!(
|
assert!(matches!(collect_system_snapshot(&first_cancel).await, Err(CollectorError::TimedOut)));
|
||||||
collect_offline_diagnostics(&StorageInfo::default(), &first_cancel).await,
|
|
||||||
Err(CollectorError::TimedOut)
|
|
||||||
));
|
|
||||||
assert_eq!(test_support::ACTIVE.load(Ordering::SeqCst), 1, "timed-out blocking scan remains active");
|
assert_eq!(test_support::ACTIVE.load(Ordering::SeqCst), 1, "timed-out blocking scan remains active");
|
||||||
|
|
||||||
let second_cancel = CancellationToken::new();
|
let second_cancel = CancellationToken::new();
|
||||||
let second = tokio::spawn({
|
let second = tokio::spawn({
|
||||||
let second_cancel = second_cancel.clone();
|
let second_cancel = second_cancel.clone();
|
||||||
async move { collect_offline_diagnostics(&StorageInfo::default(), &second_cancel).await }
|
async move { collect_system_snapshot(&second_cancel).await }
|
||||||
});
|
});
|
||||||
tokio::time::sleep(Duration::from_millis(50)).await;
|
tokio::time::sleep(Duration::from_millis(50)).await;
|
||||||
assert_eq!(
|
assert_eq!(
|
||||||
@@ -392,7 +345,7 @@ mod tests {
|
|||||||
let third_cancel = CancellationToken::new();
|
let third_cancel = CancellationToken::new();
|
||||||
let third = tokio::spawn({
|
let third = tokio::spawn({
|
||||||
let third_cancel = third_cancel.clone();
|
let third_cancel = third_cancel.clone();
|
||||||
async move { collect_offline_diagnostics(&StorageInfo::default(), &third_cancel).await }
|
async move { collect_system_snapshot(&third_cancel).await }
|
||||||
});
|
});
|
||||||
wait_for_active(1).await;
|
wait_for_active(1).await;
|
||||||
third_cancel.cancel();
|
third_cancel.cancel();
|
||||||
@@ -401,7 +354,7 @@ mod tests {
|
|||||||
let fourth_cancel = CancellationToken::new();
|
let fourth_cancel = CancellationToken::new();
|
||||||
let fourth = tokio::spawn({
|
let fourth = tokio::spawn({
|
||||||
let fourth_cancel = fourth_cancel.clone();
|
let fourth_cancel = fourth_cancel.clone();
|
||||||
async move { collect_offline_diagnostics(&StorageInfo::default(), &fourth_cancel).await }
|
async move { collect_system_snapshot(&fourth_cancel).await }
|
||||||
});
|
});
|
||||||
tokio::time::sleep(Duration::from_millis(50)).await;
|
tokio::time::sleep(Duration::from_millis(50)).await;
|
||||||
assert_eq!(
|
assert_eq!(
|
||||||
|
|||||||
@@ -34,7 +34,7 @@ pub mod key_store;
|
|||||||
pub mod manifest_entry;
|
pub mod manifest_entry;
|
||||||
pub mod redaction;
|
pub mod redaction;
|
||||||
|
|
||||||
pub use collectors::{CollectorError, OfflineCollector, collect_offline_diagnostics};
|
pub use collectors::{CollectorError, OfflineCollector, OfflineDiagnostics, collect_offline_diagnostics};
|
||||||
pub use enrollment::{EnrollmentError, OfflineEnrollment, VerifiedChallenge};
|
pub use enrollment::{EnrollmentError, OfflineEnrollment, VerifiedChallenge};
|
||||||
pub use key_store::OfflineKeyStore;
|
pub use key_store::OfflineKeyStore;
|
||||||
pub use manifest_entry::ManifestEntry;
|
pub use manifest_entry::ManifestEntry;
|
||||||
|
|||||||
@@ -13,6 +13,7 @@
|
|||||||
// limitations under the License.
|
// limitations under the License.
|
||||||
|
|
||||||
use std::future::Future;
|
use std::future::Future;
|
||||||
|
use std::sync::Arc;
|
||||||
use std::time::Duration;
|
use std::time::Duration;
|
||||||
|
|
||||||
use chrono::Utc;
|
use chrono::Utc;
|
||||||
@@ -57,6 +58,7 @@ pub struct InventoryRuntime {
|
|||||||
shutdown: CancellationToken,
|
shutdown: CancellationToken,
|
||||||
status: watch::Receiver<InventoryStatus>,
|
status: watch::Receiver<InventoryStatus>,
|
||||||
task: Option<JoinHandle<()>>,
|
task: Option<JoinHandle<()>>,
|
||||||
|
_lock: Arc<std::fs::File>,
|
||||||
}
|
}
|
||||||
|
|
||||||
impl InventoryRuntime {
|
impl InventoryRuntime {
|
||||||
@@ -190,7 +192,7 @@ where
|
|||||||
let retry_schedule = config.schedule;
|
let retry_schedule = config.schedule;
|
||||||
let state_root = config.state_root().ok_or(InventoryError::StatePath)?;
|
let state_root = config.state_root().ok_or(InventoryError::StatePath)?;
|
||||||
let store = InventoryStateStore::from_state_root(state_root)?;
|
let store = InventoryStateStore::from_state_root(state_root)?;
|
||||||
let lock = store.try_runtime_lock()?;
|
let lock = Arc::new(store.try_runtime_lock()?);
|
||||||
let sender = if config.transport_enabled() {
|
let sender = if config.transport_enabled() {
|
||||||
Some(InventorySender::new(config)?)
|
Some(InventorySender::new(config)?)
|
||||||
} else {
|
} else {
|
||||||
@@ -199,8 +201,9 @@ where
|
|||||||
let shutdown = parent_shutdown.child_token();
|
let shutdown = parent_shutdown.child_token();
|
||||||
let task_shutdown = shutdown.clone();
|
let task_shutdown = shutdown.clone();
|
||||||
let (status_tx, status_rx) = watch::channel(InventoryStatus::Starting);
|
let (status_tx, status_rx) = watch::channel(InventoryStatus::Starting);
|
||||||
|
let task_lock = lock.clone();
|
||||||
let task = tokio::spawn(async move {
|
let task = tokio::spawn(async move {
|
||||||
let _lock = lock;
|
let _lock = task_lock;
|
||||||
let mut backoff = retry_schedule.initial_backoff;
|
let mut backoff = retry_schedule.initial_backoff;
|
||||||
loop {
|
loop {
|
||||||
if task_shutdown.is_cancelled() {
|
if task_shutdown.is_cancelled() {
|
||||||
@@ -331,6 +334,7 @@ where
|
|||||||
shutdown,
|
shutdown,
|
||||||
status: status_rx,
|
status: status_rx,
|
||||||
task: Some(task),
|
task: Some(task),
|
||||||
|
_lock: lock,
|
||||||
}))
|
}))
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -474,6 +478,7 @@ mod tests {
|
|||||||
task_inventory_shutdown.cancelled().await;
|
task_inventory_shutdown.cancelled().await;
|
||||||
let _ = inventory_stopped.send(());
|
let _ = inventory_stopped.send(());
|
||||||
});
|
});
|
||||||
|
let inventory_lock = Arc::new(tempfile::tempfile().expect("inventory runtime lock"));
|
||||||
let heartbeat = HeartbeatRuntime {
|
let heartbeat = HeartbeatRuntime {
|
||||||
shutdown: heartbeat_shutdown,
|
shutdown: heartbeat_shutdown,
|
||||||
status: heartbeat_status,
|
status: heartbeat_status,
|
||||||
@@ -483,6 +488,7 @@ mod tests {
|
|||||||
shutdown: inventory_shutdown,
|
shutdown: inventory_shutdown,
|
||||||
status: inventory_status,
|
status: inventory_status,
|
||||||
task: Some(inventory_task),
|
task: Some(inventory_task),
|
||||||
|
_lock: inventory_lock,
|
||||||
};
|
};
|
||||||
|
|
||||||
let shutdown = tokio::spawn(shutdown_connect_runtimes(Some(heartbeat), Some(inventory)));
|
let shutdown = tokio::spawn(shutdown_connect_runtimes(Some(heartbeat), Some(inventory)));
|
||||||
@@ -496,4 +502,38 @@ mod tests {
|
|||||||
.expect("runtime shutdown")
|
.expect("runtime shutdown")
|
||||||
.expect("shutdown task");
|
.expect("shutdown task");
|
||||||
}
|
}
|
||||||
|
|
||||||
|
#[cfg(target_os = "linux")]
|
||||||
|
#[tokio::test]
|
||||||
|
async fn dropping_inventory_handle_keeps_the_lock_until_its_task_exits() {
|
||||||
|
use std::os::unix::fs::PermissionsExt as _;
|
||||||
|
|
||||||
|
let temp = tempfile::tempdir_in(env!("CARGO_MANIFEST_DIR")).expect("safe temporary directory");
|
||||||
|
let state = temp.path().join("state");
|
||||||
|
std::fs::create_dir(&state).expect("state root");
|
||||||
|
std::fs::set_permissions(&state, std::fs::Permissions::from_mode(0o700)).expect("private state root");
|
||||||
|
let store = InventoryStateStore::from_state_root(&state).expect("inventory store");
|
||||||
|
let lock = Arc::new(store.try_runtime_lock().expect("runtime lock"));
|
||||||
|
let task_lock = lock.clone();
|
||||||
|
let (release, released) = tokio::sync::oneshot::channel();
|
||||||
|
let (finished, task_finished) = tokio::sync::oneshot::channel();
|
||||||
|
let task = tokio::spawn(async move {
|
||||||
|
let _ = released.await;
|
||||||
|
drop(task_lock);
|
||||||
|
let _ = finished.send(());
|
||||||
|
});
|
||||||
|
let (_, status) = watch::channel(InventoryStatus::Starting);
|
||||||
|
let runtime = InventoryRuntime {
|
||||||
|
shutdown: CancellationToken::new(),
|
||||||
|
status,
|
||||||
|
task: Some(task),
|
||||||
|
_lock: lock,
|
||||||
|
};
|
||||||
|
|
||||||
|
drop(runtime);
|
||||||
|
assert!(matches!(store.try_runtime_lock(), Err(InventoryError::AlreadyRunning)));
|
||||||
|
release.send(()).expect("release inventory task");
|
||||||
|
task_finished.await.expect("inventory task finished");
|
||||||
|
store.try_runtime_lock().expect("lock after task exit");
|
||||||
|
}
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -16,11 +16,27 @@
|
|||||||
|
|
||||||
use std::fs;
|
use std::fs;
|
||||||
use std::path::PathBuf;
|
use std::path::PathBuf;
|
||||||
|
#[cfg(target_os = "linux")]
|
||||||
|
use std::sync::{
|
||||||
|
Arc,
|
||||||
|
atomic::{AtomicUsize, Ordering},
|
||||||
|
};
|
||||||
|
#[cfg(target_os = "linux")]
|
||||||
|
use std::time::Duration;
|
||||||
|
|
||||||
use rustfs::connect::offline::{CollectorError, RedactionSource, collect_offline_diagnostics, redact_json};
|
#[cfg(target_os = "linux")]
|
||||||
use rustfs_madmin::{Disk, ITEM_OFFLINE, StorageInfo};
|
use rustfs::connect::offline::{CollectorError, collect_offline_diagnostics};
|
||||||
|
use rustfs::connect::offline::{RedactionSource, redact_json};
|
||||||
|
#[cfg(target_os = "linux")]
|
||||||
|
use rustfs::connect::{
|
||||||
|
CredentialStore, HeartbeatConfig, IdentityStore, InventoryError, InventoryFlag, InventorySchedule, InventorySnapshot,
|
||||||
|
InventoryStatus, spawn_inventory_runtime,
|
||||||
|
};
|
||||||
use serde_json::{Map, Value, json};
|
use serde_json::{Map, Value, json};
|
||||||
use sha2::{Digest as _, Sha256};
|
use sha2::{Digest as _, Sha256};
|
||||||
|
#[cfg(target_os = "linux")]
|
||||||
|
use tokio::sync::watch;
|
||||||
|
#[cfg(target_os = "linux")]
|
||||||
use tokio_util::sync::CancellationToken;
|
use tokio_util::sync::CancellationToken;
|
||||||
|
|
||||||
fn fixture_dir() -> PathBuf {
|
fn fixture_dir() -> PathBuf {
|
||||||
@@ -164,127 +180,175 @@ fn connect_offline_collectors_reject_oversize_raw_input_before_parsing() {
|
|||||||
);
|
);
|
||||||
}
|
}
|
||||||
|
|
||||||
|
#[cfg(target_os = "linux")]
|
||||||
|
async fn wait_for_inventory(status: &mut watch::Receiver<InventoryStatus>) {
|
||||||
|
tokio::time::timeout(Duration::from_secs(3), async {
|
||||||
|
loop {
|
||||||
|
if matches!(status.borrow_and_update().clone(), InventoryStatus::Unchanged { .. }) {
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
status.changed().await.expect("inventory status channel");
|
||||||
|
}
|
||||||
|
})
|
||||||
|
.await
|
||||||
|
.expect("inventory persistence timeout");
|
||||||
|
}
|
||||||
|
|
||||||
|
#[cfg(target_os = "linux")]
|
||||||
|
async fn wait_for_inventory_failure(status: &mut watch::Receiver<InventoryStatus>) {
|
||||||
|
tokio::time::timeout(Duration::from_secs(3), async {
|
||||||
|
loop {
|
||||||
|
if matches!(status.borrow_and_update().clone(), InventoryStatus::Failed { .. }) {
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
status.changed().await.expect("inventory status channel");
|
||||||
|
}
|
||||||
|
})
|
||||||
|
.await
|
||||||
|
.expect("inventory failure timeout");
|
||||||
|
}
|
||||||
|
|
||||||
|
#[cfg(target_os = "linux")]
|
||||||
|
fn private_directory(path: &std::path::Path) {
|
||||||
|
use std::os::unix::fs::PermissionsExt as _;
|
||||||
|
|
||||||
|
fs::set_permissions(path, fs::Permissions::from_mode(0o700)).expect("private directory permissions");
|
||||||
|
}
|
||||||
|
|
||||||
|
#[cfg(target_os = "linux")]
|
||||||
#[tokio::test]
|
#[tokio::test]
|
||||||
async fn connect_offline_collectors_emit_only_fixed_redacted_entries_and_honor_cancellation() {
|
async fn connect_offline_collectors_read_only_the_stopped_runtime_inventory() {
|
||||||
let storage = StorageInfo {
|
let temp = tempfile::tempdir_in(env!("CARGO_MANIFEST_DIR")).expect("safe temporary directory");
|
||||||
disks: vec![
|
let state = temp.path().join("state");
|
||||||
Disk {
|
fs::create_dir(&state).expect("state root");
|
||||||
endpoint: "https://node-a.private.example:9000/data-a".to_owned(),
|
private_directory(&state);
|
||||||
drive_path: "/secret/customer/path-a".to_owned(),
|
let config = HeartbeatConfig::new(
|
||||||
uuid: "private-drive-a".to_owned(),
|
"",
|
||||||
state: "ok".to_owned(),
|
Vec::new(),
|
||||||
total_space: 1_000,
|
IdentityStore::new(state.join("identity")),
|
||||||
used_space: 400,
|
CredentialStore::new(state.join("credential")),
|
||||||
..Disk::default()
|
state.join("heartbeat/state.json"),
|
||||||
},
|
);
|
||||||
Disk {
|
let shutdown = CancellationToken::new();
|
||||||
endpoint: "https://node-a.private.example:9000/data-b".to_owned(),
|
let runtime = spawn_inventory_runtime(
|
||||||
drive_path: "/secret/customer/path-b".to_owned(),
|
Some(config),
|
||||||
uuid: "private-drive-b".to_owned(),
|
InventorySchedule {
|
||||||
state: "unformatted".to_owned(),
|
cadence: Duration::from_secs(60),
|
||||||
total_space: 2_000,
|
jitter: Duration::ZERO,
|
||||||
used_space: 500,
|
},
|
||||||
..Disk::default()
|
&shutdown,
|
||||||
},
|
|| {
|
||||||
Disk {
|
std::future::ready(InventorySnapshot::new(
|
||||||
endpoint: "https://node-b.private.example:9000/data-c".to_owned(),
|
"1.4.2",
|
||||||
drive_path: "/secret/customer/path-c".to_owned(),
|
None,
|
||||||
uuid: "private-drive-c".to_owned(),
|
2,
|
||||||
state: ITEM_OFFLINE.to_owned(),
|
3,
|
||||||
total_space: 3_000,
|
6_000,
|
||||||
used_space: 600,
|
1_500,
|
||||||
healing: true,
|
[InventoryFlag::ClusterDegraded, InventoryFlag::DriveOffline],
|
||||||
..Disk::default()
|
))
|
||||||
},
|
},
|
||||||
],
|
)
|
||||||
..StorageInfo::default()
|
.expect("state-only inventory")
|
||||||
};
|
.expect("configured inventory");
|
||||||
let cancel = CancellationToken::new();
|
let mut status = runtime.status();
|
||||||
let entries = collect_offline_diagnostics(&storage, &cancel)
|
wait_for_inventory(&mut status).await;
|
||||||
|
|
||||||
|
assert!(matches!(
|
||||||
|
collect_offline_diagnostics(&state, &CancellationToken::new()).await,
|
||||||
|
Err(CollectorError::Inventory(InventoryError::AlreadyRunning))
|
||||||
|
));
|
||||||
|
runtime.shutdown().await;
|
||||||
|
|
||||||
|
let diagnostics = collect_offline_diagnostics(&state, &CancellationToken::new())
|
||||||
.await
|
.await
|
||||||
.expect("collect fixed offline entries");
|
.expect("collect fixed offline entries");
|
||||||
assert_eq!(entries.len(), 12);
|
assert_eq!(diagnostics.entries.len(), 12);
|
||||||
let encoded = serde_json::to_string(&entries).expect("manifest entries serialize");
|
assert!(diagnostics.inventory_captured_at.ends_with('Z'));
|
||||||
for forbidden in [
|
assert!(diagnostics.inventory_age < Duration::from_secs(10));
|
||||||
"node-a.private.example",
|
assert!(diagnostics.entries.iter().all(|entry| entry.field_id.starts_with("offline.")));
|
||||||
"node-b.private.example",
|
assert!(
|
||||||
"/secret/customer/path-a",
|
diagnostics
|
||||||
"/secret/customer/path-b",
|
.entries
|
||||||
"/secret/customer/path-c",
|
.iter()
|
||||||
"private-drive-a",
|
.all(|entry| entry.canonical_json.len() <= 16 * 1024)
|
||||||
"private-drive-b",
|
);
|
||||||
"private-drive-c",
|
|
||||||
] {
|
|
||||||
assert!(!encoded.contains(forbidden), "private storage metadata must not leave the collector");
|
|
||||||
}
|
|
||||||
assert!(entries.iter().all(|entry| entry.field_id.starts_with("offline.")));
|
|
||||||
assert!(entries.iter().all(|entry| entry.canonical_json.len() <= 16 * 1024));
|
|
||||||
let canonical = |field_id| {
|
let canonical = |field_id| {
|
||||||
entries
|
diagnostics
|
||||||
|
.entries
|
||||||
.iter()
|
.iter()
|
||||||
.find(|entry| entry.field_id == field_id)
|
.find(|entry| entry.field_id == field_id)
|
||||||
.unwrap_or_else(|| panic!("missing {field_id}"))
|
.unwrap_or_else(|| panic!("missing {field_id}"))
|
||||||
.canonical_json
|
.canonical_json
|
||||||
.as_str()
|
.as_str()
|
||||||
};
|
};
|
||||||
|
assert_eq!(canonical("offline.rustfsVersion"), r#"{"rustfsVersion":"1.4.2"}"#);
|
||||||
assert_eq!(canonical("offline.nodeCount"), r#"{"nodeCount":2}"#);
|
assert_eq!(canonical("offline.nodeCount"), r#"{"nodeCount":2}"#);
|
||||||
assert_eq!(canonical("offline.driveCount"), r#"{"driveCount":3}"#);
|
assert_eq!(canonical("offline.driveCount"), r#"{"driveCount":3}"#);
|
||||||
assert_eq!(canonical("offline.capacityUsedBytes"), r#"{"capacityUsedBytes":1500}"#);
|
assert_eq!(canonical("offline.capacityUsedBytes"), r#"{"capacityUsedBytes":1500}"#);
|
||||||
assert_eq!(canonical("offline.capacityTotalBytes"), r#"{"capacityTotalBytes":6000}"#);
|
assert_eq!(canonical("offline.capacityTotalBytes"), r#"{"capacityTotalBytes":6000}"#);
|
||||||
assert_eq!(
|
assert_eq!(
|
||||||
canonical("offline.coarseHealthFlags"),
|
canonical("offline.coarseHealthFlags"),
|
||||||
r#"{"coarseHealthFlags":{"degraded":true,"healing":true,"offlineDrives":1,"scanning":false}}"#
|
r#"{"coarseHealthFlags":["cluster.degraded","drive.offline"]}"#
|
||||||
);
|
);
|
||||||
|
|
||||||
let healthy = StorageInfo {
|
let cancelled = CancellationToken::new();
|
||||||
disks: ["ok", "unformatted", "online"]
|
cancelled.cancel();
|
||||||
.into_iter()
|
|
||||||
.enumerate()
|
|
||||||
.map(|(index, state)| Disk {
|
|
||||||
endpoint: format!("https://healthy.example:9000/data-{index}"),
|
|
||||||
state: state.to_owned(),
|
|
||||||
..Disk::default()
|
|
||||||
})
|
|
||||||
.collect(),
|
|
||||||
..StorageInfo::default()
|
|
||||||
};
|
|
||||||
let healthy_entries = collect_offline_diagnostics(&healthy, &CancellationToken::new())
|
|
||||||
.await
|
|
||||||
.expect("collect healthy storage summary");
|
|
||||||
assert_eq!(
|
|
||||||
healthy_entries
|
|
||||||
.iter()
|
|
||||||
.find(|entry| entry.field_id == "offline.coarseHealthFlags")
|
|
||||||
.expect("healthy coarse health entry")
|
|
||||||
.canonical_json,
|
|
||||||
r#"{"coarseHealthFlags":{"degraded":false,"healing":false,"offlineDrives":0,"scanning":false}}"#
|
|
||||||
);
|
|
||||||
|
|
||||||
cancel.cancel();
|
|
||||||
assert!(matches!(
|
assert!(matches!(
|
||||||
collect_offline_diagnostics(&storage, &cancel).await,
|
collect_offline_diagnostics(&state, &cancelled).await,
|
||||||
Err(CollectorError::Cancelled)
|
Err(CollectorError::Cancelled)
|
||||||
));
|
));
|
||||||
|
|
||||||
let oversized = StorageInfo {
|
fs::write(state.join("inventory/latest.json"), b"{}").expect("corrupt latest inventory");
|
||||||
disks: vec![Disk::default(); 4_097],
|
|
||||||
..StorageInfo::default()
|
|
||||||
};
|
|
||||||
let active = CancellationToken::new();
|
|
||||||
assert!(matches!(
|
assert!(matches!(
|
||||||
collect_offline_diagnostics(&oversized, &active).await,
|
collect_offline_diagnostics(&state, &CancellationToken::new()).await,
|
||||||
Err(CollectorError::StorageTopologyTooLarge)
|
Err(CollectorError::Inventory(InventoryError::EnvelopeInvalid))
|
||||||
));
|
|
||||||
|
|
||||||
let invalid_endpoint = StorageInfo {
|
|
||||||
disks: vec![Disk {
|
|
||||||
endpoint: "not-an-endpoint".to_owned(),
|
|
||||||
..Disk::default()
|
|
||||||
}],
|
|
||||||
..StorageInfo::default()
|
|
||||||
};
|
|
||||||
assert!(matches!(
|
|
||||||
collect_offline_diagnostics(&invalid_endpoint, &active).await,
|
|
||||||
Err(CollectorError::InvalidStorageEndpoint)
|
|
||||||
));
|
));
|
||||||
}
|
}
|
||||||
|
|
||||||
|
#[cfg(target_os = "linux")]
|
||||||
|
#[tokio::test]
|
||||||
|
async fn connect_offline_collectors_reject_a_live_runtime_after_its_task_fails() {
|
||||||
|
let temp = tempfile::tempdir_in(env!("CARGO_MANIFEST_DIR")).expect("safe temporary directory");
|
||||||
|
let state = temp.path().join("state");
|
||||||
|
fs::create_dir(&state).expect("state root");
|
||||||
|
private_directory(&state);
|
||||||
|
let config = HeartbeatConfig::new(
|
||||||
|
"",
|
||||||
|
Vec::new(),
|
||||||
|
IdentityStore::new(state.join("identity")),
|
||||||
|
CredentialStore::new(state.join("credential")),
|
||||||
|
state.join("heartbeat/state.json"),
|
||||||
|
);
|
||||||
|
let attempts = Arc::new(AtomicUsize::new(0));
|
||||||
|
let sample_attempts = attempts.clone();
|
||||||
|
let runtime = spawn_inventory_runtime(
|
||||||
|
Some(config),
|
||||||
|
InventorySchedule {
|
||||||
|
cadence: Duration::from_millis(1),
|
||||||
|
jitter: Duration::ZERO,
|
||||||
|
},
|
||||||
|
&CancellationToken::new(),
|
||||||
|
move || {
|
||||||
|
let attempt = sample_attempts.fetch_add(1, Ordering::SeqCst);
|
||||||
|
std::future::ready(if attempt == 0 {
|
||||||
|
InventorySnapshot::new("1.4.2", None, 1, 1, 100, 50, [])
|
||||||
|
} else {
|
||||||
|
Err(InventoryError::Capacity)
|
||||||
|
})
|
||||||
|
},
|
||||||
|
)
|
||||||
|
.expect("state-only inventory")
|
||||||
|
.expect("configured inventory");
|
||||||
|
let mut status = runtime.status();
|
||||||
|
wait_for_inventory_failure(&mut status).await;
|
||||||
|
|
||||||
|
assert!(matches!(
|
||||||
|
collect_offline_diagnostics(&state, &CancellationToken::new()).await,
|
||||||
|
Err(CollectorError::Inventory(InventoryError::AlreadyRunning))
|
||||||
|
));
|
||||||
|
runtime.shutdown().await;
|
||||||
|
collect_offline_diagnostics(&state, &CancellationToken::new())
|
||||||
|
.await
|
||||||
|
.expect("collector after runtime shutdown");
|
||||||
|
}
|
||||||
|
|||||||
Reference in New Issue
Block a user