mirror of
https://github.com/rustfs/rustfs.git
synced 2026-09-04 03:05:39 +00:00
fix(admin): answer background-heal/status partially when peers are unreachable (#5862)
This commit is contained in:
@@ -431,4 +431,104 @@ mod tests {
|
|||||||
)
|
)
|
||||||
.into())
|
.into())
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/// Issue #5850: `background-heal/status` must answer while a peer is down.
|
||||||
|
///
|
||||||
|
/// Exercises the production path in `read_cluster_heal_status` end to end,
|
||||||
|
/// which the unit tests around `merge_peer_heal_statuses` cannot: with one
|
||||||
|
/// node stopped, the endpoint must return 200 with
|
||||||
|
/// `clusterStatusComplete: false` and an explicit `degraded` (or, when
|
||||||
|
/// heal work is known active, `active`) state — never the previous
|
||||||
|
/// cluster-wide 500 — and must return to a complete, non-degraded answer
|
||||||
|
/// once the node rejoins. Reverting either all-or-nothing gate (the
|
||||||
|
/// topology early-return or the merge hard-fail) turns the down-window
|
||||||
|
/// response into a 500 and fails this test.
|
||||||
|
#[tokio::test]
|
||||||
|
#[serial]
|
||||||
|
async fn test_background_heal_status_degrades_while_peer_down_and_recovers_after_rejoin()
|
||||||
|
-> Result<(), Box<dyn Error + Send + Sync>> {
|
||||||
|
init_logging();
|
||||||
|
info!("Issue #5850: background-heal/status must degrade, not 500, while a peer is down");
|
||||||
|
|
||||||
|
let mut cluster = RustFSTestClusterEnvironment::new(4).await?;
|
||||||
|
cluster.set_env("RUSTFS_UNSAFE_BYPASS_DISK_CHECK", "true");
|
||||||
|
cluster.start().await?;
|
||||||
|
|
||||||
|
let status_url = format!("{}/rustfs/admin/v3/background-heal/status", cluster.nodes[0].url);
|
||||||
|
// Owned copies: the closure must not borrow `cluster`, which
|
||||||
|
// stop_node/start_node need mutably between polls.
|
||||||
|
let access_key = cluster.access_key.clone();
|
||||||
|
let secret_key = cluster.secret_key.clone();
|
||||||
|
let fetch_status = || async {
|
||||||
|
let body = signed_admin_post(&status_url, None, &access_key, &secret_key).await?;
|
||||||
|
let json: serde_json::Value =
|
||||||
|
serde_json::from_str(&body).map_err(|err| format!("heal status response is not JSON ({err}): {body}"))?;
|
||||||
|
Ok::<serde_json::Value, Box<dyn Error + Send + Sync>>(json)
|
||||||
|
};
|
||||||
|
|
||||||
|
// Healthy cluster: the answer must be definitive. Poll briefly — the
|
||||||
|
// peer grid may still be settling right after start().
|
||||||
|
let mut healthy = fetch_status().await?;
|
||||||
|
for _ in 0..30 {
|
||||||
|
if healthy["clusterStatusComplete"] == serde_json::Value::Bool(true) {
|
||||||
|
break;
|
||||||
|
}
|
||||||
|
sleep(Duration::from_secs(1)).await;
|
||||||
|
healthy = fetch_status().await?;
|
||||||
|
}
|
||||||
|
assert_eq!(
|
||||||
|
healthy["clusterStatusComplete"],
|
||||||
|
serde_json::Value::Bool(true),
|
||||||
|
"healthy cluster should report a complete heal status: {healthy}"
|
||||||
|
);
|
||||||
|
|
||||||
|
cluster.stop_node(1)?;
|
||||||
|
|
||||||
|
// While the peer is down every response must stay 200 (signed_admin_post
|
||||||
|
// fails on any non-2xx, so the old 500 fails the test immediately) and
|
||||||
|
// must degrade to an explicitly-partial answer. The peer query timeout
|
||||||
|
// is 5 s, so a couple of polls are enough for the dead peer to surface.
|
||||||
|
let mut degraded = serde_json::Value::Null;
|
||||||
|
for _ in 0..30 {
|
||||||
|
degraded = fetch_status().await?;
|
||||||
|
if degraded["clusterStatusComplete"] == serde_json::Value::Bool(false) {
|
||||||
|
break;
|
||||||
|
}
|
||||||
|
sleep(Duration::from_secs(1)).await;
|
||||||
|
}
|
||||||
|
assert_eq!(
|
||||||
|
degraded["clusterStatusComplete"],
|
||||||
|
serde_json::Value::Bool(false),
|
||||||
|
"heal status must mark itself partial while a peer is down: {degraded}"
|
||||||
|
);
|
||||||
|
let state = degraded["state"].as_str().unwrap_or_default();
|
||||||
|
assert!(
|
||||||
|
state == "degraded" || state == "active",
|
||||||
|
"a partial answer must be labeled degraded (or active for known work), got {state:?}: {degraded}"
|
||||||
|
);
|
||||||
|
|
||||||
|
cluster.start_node(1).await?;
|
||||||
|
|
||||||
|
// After the rejoin the endpoint must return to a definitive answer.
|
||||||
|
let mut recovered = serde_json::Value::Null;
|
||||||
|
for _ in 0..60 {
|
||||||
|
recovered = fetch_status().await?;
|
||||||
|
if recovered["clusterStatusComplete"] == serde_json::Value::Bool(true) {
|
||||||
|
break;
|
||||||
|
}
|
||||||
|
sleep(Duration::from_secs(1)).await;
|
||||||
|
}
|
||||||
|
assert_eq!(
|
||||||
|
recovered["clusterStatusComplete"],
|
||||||
|
serde_json::Value::Bool(true),
|
||||||
|
"heal status should be complete again after the node rejoined: {recovered}"
|
||||||
|
);
|
||||||
|
assert_ne!(
|
||||||
|
recovered["state"].as_str().unwrap_or_default(),
|
||||||
|
"degraded",
|
||||||
|
"a complete answer must not be labeled degraded: {recovered}"
|
||||||
|
);
|
||||||
|
|
||||||
|
Ok(())
|
||||||
|
}
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -231,6 +231,11 @@ enum HealRuntimeState {
|
|||||||
Uninitialized,
|
Uninitialized,
|
||||||
Idle,
|
Idle,
|
||||||
Active,
|
Active,
|
||||||
|
/// One or more nodes were unreachable and the reachable ones report no
|
||||||
|
/// active work, so no definitive cluster verdict exists. Distinct from
|
||||||
|
/// `Idle` on purpose: an unreachable peer might be healing, and a partial
|
||||||
|
/// answer must not read as "nothing is running".
|
||||||
|
Degraded,
|
||||||
}
|
}
|
||||||
|
|
||||||
#[cfg(test)]
|
#[cfg(test)]
|
||||||
@@ -386,6 +391,7 @@ fn merge_peer_heal_statuses(
|
|||||||
mut snapshots: Vec<NodeHealStatusSnapshot>,
|
mut snapshots: Vec<NodeHealStatusSnapshot>,
|
||||||
peer_statuses: Vec<Result<Option<NodeHealStatusSnapshot>, String>>,
|
peer_statuses: Vec<Result<Option<NodeHealStatusSnapshot>, String>>,
|
||||||
expected_nodes: usize,
|
expected_nodes: usize,
|
||||||
|
topology_complete: bool,
|
||||||
) -> S3Result<ClusterHealStatusSnapshot> {
|
) -> S3Result<ClusterHealStatusSnapshot> {
|
||||||
for peer_status in peer_statuses {
|
for peer_status in peer_statuses {
|
||||||
match peer_status {
|
match peer_status {
|
||||||
@@ -407,20 +413,36 @@ fn merge_peer_heal_statuses(
|
|||||||
component = LOG_COMPONENT_ADMIN_API,
|
component = LOG_COMPONENT_ADMIN_API,
|
||||||
subsystem = LOG_SUBSYSTEM_HEAL_ADMIN,
|
subsystem = LOG_SUBSYSTEM_HEAL_ADMIN,
|
||||||
operation = "background_heal_status",
|
operation = "background_heal_status",
|
||||||
result = "failed",
|
result = "degraded",
|
||||||
reason = "peer_status_unavailable",
|
reason = "peer_status_unavailable",
|
||||||
error = %err,
|
error = %err,
|
||||||
"cluster heal peer status unavailable"
|
"cluster heal peer status unavailable"
|
||||||
);
|
);
|
||||||
return Err(cluster_heal_status_unavailable("peer_status_unavailable"));
|
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
let complete = snapshots.len() == expected_nodes;
|
if snapshots.is_empty() {
|
||||||
|
return Err(cluster_heal_status_unavailable("no_node_status_available"));
|
||||||
|
}
|
||||||
|
// Completeness requires every condition, not just the snapshot count: the
|
||||||
|
// peer queries iterate `peer_clients` while the topology check also
|
||||||
|
// examines `all_peer_clients`, and the two are populated independently —
|
||||||
|
// so during a reconfiguration the count can equal `expected_nodes` while
|
||||||
|
// the topology is known-incomplete. Counting alone would report a
|
||||||
|
// definitive answer precisely when the membership itself is in doubt.
|
||||||
|
let complete = topology_complete && snapshots.len() == expected_nodes;
|
||||||
let mut status = aggregate_cluster_heal_status(snapshots);
|
let mut status = aggregate_cluster_heal_status(snapshots);
|
||||||
status.complete = complete;
|
status.complete = complete;
|
||||||
|
// A partial answer must never be mistakable for a definitive verdict: an
|
||||||
|
// unreachable peer might be mid-heal, so reporting the reachable nodes'
|
||||||
|
// "idle" (or disabled/uninitialized) as the cluster state would falsely
|
||||||
|
// reassure. Known-active work is safe to report as-is — it cannot
|
||||||
|
// reassure. Everything else degrades to an explicit Degraded state so the
|
||||||
|
// caller still gets the reachable nodes' data (queue depths, progress,
|
||||||
|
// bitrot cycle) with an honest label, instead of the previous cluster-wide
|
||||||
|
// 500 that carried no information at all (issue #5850).
|
||||||
if !complete && status.state != HealRuntimeState::Active {
|
if !complete && status.state != HealRuntimeState::Active {
|
||||||
return Err(cluster_heal_status_unavailable("peer_status_unsupported_without_known_active_work"));
|
status.state = HealRuntimeState::Degraded;
|
||||||
}
|
}
|
||||||
Ok(status)
|
Ok(status)
|
||||||
}
|
}
|
||||||
@@ -455,7 +477,12 @@ async fn read_cluster_heal_status(
|
|||||||
let Some(notification_system) = notification_system else {
|
let Some(notification_system) = notification_system else {
|
||||||
return Err(cluster_heal_status_unavailable("notification_system_unavailable"));
|
return Err(cluster_heal_status_unavailable("notification_system_unavailable"));
|
||||||
};
|
};
|
||||||
if !peer_topology_complete(
|
// An incomplete peer topology (a down member's client slot, a rolling
|
||||||
|
// upgrade) previously failed the whole endpoint here, before any peer was
|
||||||
|
// queried. Proceed instead: the per-peer query below reports unavailable
|
||||||
|
// slots as errors, and merge_peer_heal_statuses folds them into an
|
||||||
|
// explicit partial answer (issue #5850).
|
||||||
|
let topology_complete = peer_topology_complete(
|
||||||
expected_nodes,
|
expected_nodes,
|
||||||
notification_system.peer_clients.len(),
|
notification_system.peer_clients.len(),
|
||||||
notification_system
|
notification_system
|
||||||
@@ -469,8 +496,17 @@ async fn read_cluster_heal_status(
|
|||||||
.iter()
|
.iter()
|
||||||
.filter(|client| client.is_some())
|
.filter(|client| client.is_some())
|
||||||
.count(),
|
.count(),
|
||||||
) {
|
);
|
||||||
return Err(cluster_heal_status_unavailable("peer_topology_incomplete"));
|
if !topology_complete {
|
||||||
|
warn!(
|
||||||
|
event = EVENT_ADMIN_REQUEST_FAILED,
|
||||||
|
component = LOG_COMPONENT_ADMIN_API,
|
||||||
|
subsystem = LOG_SUBSYSTEM_HEAL_ADMIN,
|
||||||
|
operation = "background_heal_status",
|
||||||
|
result = "degraded",
|
||||||
|
reason = "peer_topology_incomplete",
|
||||||
|
"cluster heal status will be partial"
|
||||||
|
);
|
||||||
}
|
}
|
||||||
|
|
||||||
let peer_statuses = join_all(notification_system.peer_clients.iter().map(|client| async move {
|
let peer_statuses = join_all(notification_system.peer_clients.iter().map(|client| async move {
|
||||||
@@ -482,7 +518,7 @@ async fn read_cluster_heal_status(
|
|||||||
}))
|
}))
|
||||||
.await;
|
.await;
|
||||||
|
|
||||||
merge_peer_heal_statuses(snapshots, peer_statuses, expected_nodes)
|
merge_peer_heal_statuses(snapshots, peer_statuses, expected_nodes, topology_complete)
|
||||||
}
|
}
|
||||||
|
|
||||||
fn cluster_heal_control_unavailable(reason: &str) -> s3s::S3Error {
|
fn cluster_heal_control_unavailable(reason: &str) -> s3s::S3Error {
|
||||||
@@ -1936,7 +1972,7 @@ mod tests {
|
|||||||
}
|
}
|
||||||
|
|
||||||
#[test]
|
#[test]
|
||||||
fn test_peer_status_merge_fails_closed_but_degrades_for_older_peers() {
|
fn test_peer_status_merge_degrades_explicitly_and_never_claims_idle() {
|
||||||
let local = || {
|
let local = || {
|
||||||
NodeHealStatusSnapshot::for_test(
|
NodeHealStatusSnapshot::for_test(
|
||||||
true,
|
true,
|
||||||
@@ -1946,11 +1982,19 @@ mod tests {
|
|||||||
None,
|
None,
|
||||||
)
|
)
|
||||||
};
|
};
|
||||||
let error = merge_peer_heal_statuses(vec![local()], vec![Err("peer timeout".to_string())], 2)
|
// An unreachable peer must not fail the whole endpoint (issue #5850) —
|
||||||
.expect_err("peer failure must fail closed");
|
// but the safety property of the previous fail-closed behaviour is
|
||||||
assert_eq!(error.message(), Some("cluster heal status unavailable"));
|
// preserved: the partial answer is labelled Degraded, never Idle, so
|
||||||
|
// unknown peer work cannot be mistaken for "nothing is running".
|
||||||
|
let partial = merge_peer_heal_statuses(vec![local()], vec![Err("peer timeout".to_string())], 2, true)
|
||||||
|
.expect("an unreachable peer degrades the answer instead of destroying it");
|
||||||
|
assert!(!partial.complete);
|
||||||
|
assert_eq!(partial.state, HealRuntimeState::Degraded);
|
||||||
|
|
||||||
merge_peer_heal_statuses(vec![local()], vec![Ok(None)], 2).expect_err("unknown peer work must not be reported as idle");
|
let older_peer = merge_peer_heal_statuses(vec![local()], vec![Ok(None)], 2, true)
|
||||||
|
.expect("an older peer degrades the answer instead of destroying it");
|
||||||
|
assert!(!older_peer.complete);
|
||||||
|
assert_eq!(older_peer.state, HealRuntimeState::Degraded);
|
||||||
|
|
||||||
let known_active = NodeHealStatusSnapshot::for_test(
|
let known_active = NodeHealStatusSnapshot::for_test(
|
||||||
true,
|
true,
|
||||||
@@ -1962,9 +2006,56 @@ mod tests {
|
|||||||
},
|
},
|
||||||
None,
|
None,
|
||||||
);
|
);
|
||||||
let partial = merge_peer_heal_statuses(vec![known_active], vec![Ok(None)], 2)
|
let partial_active = merge_peer_heal_statuses(vec![known_active], vec![Ok(None)], 2, true)
|
||||||
.expect("known active work may be reported as an explicit partial status");
|
.expect("known active work may be reported as an explicit partial status");
|
||||||
assert!(!partial.complete);
|
assert!(!partial_active.complete);
|
||||||
|
assert_eq!(partial_active.state, HealRuntimeState::Active);
|
||||||
|
|
||||||
|
merge_peer_heal_statuses(Vec::new(), vec![Err("peer timeout".to_string())], 2, true)
|
||||||
|
.expect_err("no snapshot at all still fails closed");
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn test_incomplete_topology_forces_partial_even_with_full_snapshot_count() {
|
||||||
|
// peer_clients and all_peer_clients are populated independently, so
|
||||||
|
// during a reconfiguration every queried peer can answer — snapshot
|
||||||
|
// count == expected_nodes — while the topology check knows the
|
||||||
|
// membership itself is in doubt. Completeness must AND in the
|
||||||
|
// topology verdict or the answer looks definitive exactly when it
|
||||||
|
// is not.
|
||||||
|
let snapshot = || {
|
||||||
|
NodeHealStatusSnapshot::for_test(
|
||||||
|
true,
|
||||||
|
true,
|
||||||
|
BackgroundHealInfo::default(),
|
||||||
|
rustfs_heal::HealOperationsSnapshot::default(),
|
||||||
|
None,
|
||||||
|
)
|
||||||
|
};
|
||||||
|
let full_count_incomplete_topology = merge_peer_heal_statuses(vec![snapshot()], vec![Ok(Some(snapshot()))], 2, false)
|
||||||
|
.expect("incomplete topology degrades the answer instead of destroying it");
|
||||||
|
assert!(!full_count_incomplete_topology.complete);
|
||||||
|
assert_eq!(full_count_incomplete_topology.state, HealRuntimeState::Degraded);
|
||||||
|
|
||||||
|
let full_count_complete_topology = merge_peer_heal_statuses(vec![snapshot()], vec![Ok(Some(snapshot()))], 2, true)
|
||||||
|
.expect("complete topology and full count is a definitive answer");
|
||||||
|
assert!(full_count_complete_topology.complete);
|
||||||
|
assert_eq!(full_count_complete_topology.state, HealRuntimeState::Idle);
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn test_degraded_state_serializes_distinctly_from_idle() {
|
||||||
|
let encoded = encode_background_heal_status(
|
||||||
|
&BackgroundHealInfo::default(),
|
||||||
|
HealRuntimeState::Degraded,
|
||||||
|
rustfs_heal::HealOperationsSnapshot::default(),
|
||||||
|
None,
|
||||||
|
false,
|
||||||
|
)
|
||||||
|
.expect("degraded status must serialize");
|
||||||
|
let json: serde_json::Value = serde_json::from_slice(&encoded).expect("valid json");
|
||||||
|
assert_eq!(json["state"], "degraded");
|
||||||
|
assert_eq!(json["clusterStatusComplete"], false);
|
||||||
}
|
}
|
||||||
|
|
||||||
#[tokio::test]
|
#[tokio::test]
|
||||||
|
|||||||
Reference in New Issue
Block a user