fix: admit cold reads and readiness at read quorum (#8156)

EC 2+2 still meets read quorum with two of four nodes up. A survivor
that had not cached a bucket returned 503 on GET, HEAD, and List, and
/health/ready left the Service once write quorum was lost. Reads and
Service membership now follow read quorum and shared locks. Writes and
/minio/health/cluster still require write quorum and exclusive locks.

Signed-off-by: loverustfs <155562731+loverustfs@users.noreply.github.com>
Co-authored-by: Hauser <housemecn@gmail.com>
This commit is contained in:
RustFS
2026-09-27 21:58:55 +08:00
committed by GitHub
parent 4f61b007da
commit de484b93e0
6 changed files with 297 additions and 69 deletions
@@ -462,7 +462,10 @@ async fn assert_node_readiness_tracks_quorum(cluster: &mut RustFSTestClusterEnvi
}
let write_ready = survivors >= 3;
let read_quorum = survivors >= 2;
let expected_status = if write_ready { 200 } else { 503 };
// `/health/ready` keeps a node that can still serve reads in the Service.
let node_ready = read_quorum;
let node_status = if node_ready { 200 } else { 503 };
let cluster_status = if write_ready { 200 } else { 503 };
for (idx, client) in clients.iter().enumerate().take(survivors) {
let url = &cluster.nodes[idx].url;
let deadline = Instant::now() + Duration::from_secs(30);
@@ -472,27 +475,27 @@ async fn assert_node_readiness_tracks_quorum(cluster: &mut RustFSTestClusterEnvi
let response = http.get(format!("{url}/health/ready")).send().await?;
let status = response.status().as_u16();
let payload: serde_json::Value = response.json().await?;
if status == expected_status
&& payload["ready"] == write_ready
&& payload["details"]["storage"]["ready"] == write_ready
if status == node_status
&& payload["ready"] == node_ready
&& payload["details"]["storage"]["ready"] == node_ready
&& payload["details"]["storage"]["readQuorum"] == read_quorum
&& payload["details"]["storage"]["writeQuorum"] == write_ready
&& payload["details"]["poolMetadata"]["ready"] == true
&& payload["details"]["iam"]["ready"] == true
&& payload["details"]["lock"]["ready"] == write_ready
&& payload["details"]["lock"]["ready"] == read_quorum
{
break payload;
}
assert!(Instant::now() < deadline, "node {idx}, survivors={survivors}: HTTP {status}, {payload}");
tokio::time::sleep(Duration::from_millis(200)).await;
};
assert_eq!(payload["details"]["storage"]["readinessScope"], "write_quorum_and_pool_metadata");
assert_eq!(payload["details"]["storage"]["readinessScope"], "read_quorum");
assert_eq!(payload["details"]["storage"]["source"], "local_runtime");
assert_eq!(
payload["details"]["storage"]["status"],
if write_ready { "connected" } else { "disconnected" }
if node_ready { "connected" } else { "disconnected" }
);
if !write_ready {
if !node_ready {
assert!(
payload["degradedReasons"]
.as_array()
@@ -504,10 +507,10 @@ async fn assert_node_readiness_tracks_quorum(cluster: &mut RustFSTestClusterEnvi
for path in ["/health/ready", "/minio/health/ready"] {
let head = http.head(format!("{url}{path}")).send().await?;
assert_eq!(head.status().as_u16(), expected_status, "HEAD {path}, survivors={survivors}");
assert_eq!(head.status().as_u16(), node_status, "HEAD {path}, survivors={survivors}");
assert!(head.bytes().await?.is_empty());
let response = http.get(format!("{url}{path}")).send().await?;
assert_eq!(response.status().as_u16(), expected_status);
assert_eq!(response.status().as_u16(), node_status);
let body: serde_json::Value = response.json().await?;
assert_eq!(body["details"]["storage"], payload["details"]["storage"]);
assert_eq!(body["details"]["poolMetadata"], payload["details"]["poolMetadata"]);
@@ -526,7 +529,7 @@ async fn assert_node_readiness_tracks_quorum(cluster: &mut RustFSTestClusterEnvi
let response = http.get(format!("{url}{path}")).send().await?;
let status = response.status().as_u16();
let body: serde_json::Value = response.json().await?;
if status == expected_status
if status == cluster_status
&& body["details"]["storage"]["ready"] == storage_ready
&& body["details"]["lock"]["ready"] == write_ready
{
@@ -559,11 +562,12 @@ async fn assert_node_readiness_tracks_quorum(cluster: &mut RustFSTestClusterEnvi
let get = client.get_object().bucket(BUCKET).key(seed_key).send().await;
let get_status = match get {
Ok(object) => {
assert!(read_quorum, "GET must fail once read quorum is lost");
assert_eq!(object.body.collect().await?.into_bytes().as_ref(), seed_body);
200
}
Err(error) => {
assert!(!write_ready, "GET must succeed on a ready cluster: {error:?}");
assert!(!read_quorum, "GET must succeed while read quorum holds: {error:?}");
let status = error
.raw_response()
.expect("GET should have an HTTP response")
@@ -576,11 +580,12 @@ async fn assert_node_readiness_tracks_quorum(cluster: &mut RustFSTestClusterEnvi
let list = client.list_objects_v2().bucket(BUCKET).send().await;
let list_status = match list {
Ok(result) => {
assert!(read_quorum, "listing must fail once read quorum is lost");
assert!(result.contents().iter().any(|object| object.key() == Some(seed_key)));
200
}
Err(error) => {
assert!(!write_ready, "listing must succeed on a ready cluster: {error:?}");
assert!(!read_quorum, "listing must succeed while read quorum holds: {error:?}");
let status = error
.raw_response()
.expect("LIST should have an HTTP response")
+93
View File
@@ -2430,6 +2430,99 @@ mod tests {
restore_set_disks(&store, set, offline).await;
}
/// EC 2+2 with two disks offline still has object read quorum and has lost
/// write quorum. A node that has not cached the bucket must still serve
/// HEAD bucket, GET, HEAD object, and List; a write must fail.
#[tokio::test]
#[serial]
async fn cold_bucket_reads_succeed_below_write_quorum() {
let (_temp_dir, store) = setup_bucket_quorum_test_env(&[4], Some(2)).await;
metadata_sys::init_bucket_metadata_sys(store.clone(), Vec::new()).await;
let bucket = format!("cold-read-quorum-{}", Uuid::new_v4().simple());
let object = "seed-object";
let body = b"cold bucket reads must survive lost write quorum".to_vec();
store
.make_bucket(&bucket, &MakeBucketOptions::default())
.await
.expect("healthy namespace should accept bucket creation");
store
.put_object(&bucket, object, &mut PutObjReader::from_vec(body.clone()), &ObjectOptions::default())
.await
.expect("healthy erasure set should accept the seed object");
metadata_sys::init_bucket_metadata_sys(store.clone(), Vec::new()).await;
assert!(
metadata_sys::get_in(&store.ctx, &bucket).await.is_err(),
"the replacement metadata system must start without this bucket"
);
let set = &store.pools[0].disk_set[0];
let lock = set
.new_ns_lock(&bucket, object)
.await
.expect("seed namespace lock should resolve");
drop(
lock.get_write_lock(Duration::from_secs(30))
.await
.expect("seed physical fanout must finish before taking disks offline"),
);
let offline = take_set_disks_offline(&store, set, &[0, 1]).await;
let info = store
.get_bucket_info(&bucket, &BucketOptions::default())
.await
.expect("HEAD bucket must succeed at read quorum before the bucket is cached");
assert_eq!(info.name, bucket);
let mut reader = store
.get_object_reader(&bucket, object, None, Default::default(), &ObjectOptions::default())
.await
.expect("GET must succeed for a bucket that was not loaded at startup");
let mut restored = Vec::new();
reader
.stream
.read_to_end(&mut restored)
.await
.expect("quorum read should reconstruct the body");
assert_eq!(restored, body);
drop(reader);
let object_info = store
.get_object_info(&bucket, object, &ObjectOptions::default())
.await
.expect("HEAD object must succeed at read quorum");
assert_eq!(object_info.name, object);
assert_eq!(object_info.size, i64::try_from(body.len()).expect("body length fits i64"));
let listed = store
.clone()
.list_objects_v2(&bucket, "", None, None, 100, false, None, false)
.await
.expect("List must succeed at read quorum");
assert!(
listed.objects.iter().any(|item| item.name == object),
"list should include the seed object, got {:?}",
listed.objects.iter().map(|item| item.name.clone()).collect::<Vec<_>>()
);
assert!(
metadata_sys::get_in(&store.ctx, &bucket).await.is_ok(),
"the cold load should publish authoritative metadata"
);
let write_error = store
.put_object(&bucket, "rejected-object", &mut PutObjReader::from_vec(body), &ObjectOptions::default())
.await
.expect_err("writes must still fail without write quorum");
assert!(
matches!(
write_error,
StorageError::InsufficientWriteQuorum(_, _) | StorageError::ErasureWriteQuorum
),
"writes must fail with a write-quorum error, got {write_error}"
);
restore_set_disks(&store, set, offline).await;
}
#[tokio::test]
#[serial]
async fn lazy_metadata_load_rejects_below_read_quorum() {
+22 -19
View File
@@ -39,20 +39,21 @@ an unknown or unsupported peer-health snapshot degrades readiness with
- Liveness reports process availability and must not depend on storage, IAM,
lock quorum, or peer health.
- Node readiness reports the node's observed storage write quorum, local pool metadata write gate, IAM, and lock quorum. Runtime storage diagnostics do not change the existing startup `FullReady` publication gate or S3 request admission.
- A blocked pool metadata writer degrades node and cluster-write readiness with
`pool_meta_write_blocked`. Metadata save-gate inspection is bounded to 100 ms;
contention reports `pool_metadata_check_timeout` without installing a block.
Node readiness keeps the last confirmed writable observation for the normal
readiness-cache TTL during a transient inspection timeout, while preserving a
confirmed block and failing closed when no fresh observation exists.
- Node readiness (`/health/ready`, `/minio/health/ready`) reports whether this process can serve reads: storage read quorum, IAM, and shared-lock quorum. It is the Kubernetes Service membership signal. Losing write quorum, the pool-metadata writer, or exclusive-lock quorum does not remove a node that can still read. The public health handler still returns 503 until startup publishes `FullReady`, so a process that has never opened S3 admission does not join the Service. Runtime storage diagnostics do not change that startup gate.
- A blocked pool metadata writer degrades cluster-write readiness with
`pool_meta_write_blocked`. It does not withdraw `/health/ready`, and the node
probe omits that reason from `degradedReasons`. Metadata save-gate inspection
is bounded to 100 ms; contention reports `pool_metadata_check_timeout` on the
cluster-write probe without installing a block. The node probe keeps the last
confirmed writable observation for `details.poolMetadata` during a transient
inspection timeout, while preserving a confirmed block and failing that
detail closed when no fresh observation exists.
- The authenticated cluster snapshot extends its existing node-local metadata
gate inspection with safe reason, failure phase, and original block time. It
distinguishes timeout from a block and changes no admission or recovery
decision. Runtime readiness and gate status are separate bounded observations.
- Cluster write readiness requires write quorum and the runtime dependency
readiness used by `FullReady`.
- Cluster read readiness uses the storage read-quorum path and cluster-health timeout behavior. Its lock dependency still uses the per-set majority/write-lock health check. Actual shared namespace locks require `ceil(lock_clients / 2)`, so the cluster read probe is conservative: a four-client set can still admit some reads with two clients while the probe returns 503. This diagnostic change does not lower that probe's lock threshold.
- Cluster write readiness (`/minio/health/cluster`) requires storage write quorum, the pool-metadata write gate, and exclusive-lock quorum. Use it when a caller must know that new writes can commit. It is not the Service membership probe.
- Cluster read readiness uses the storage read-quorum path and cluster-health timeout behavior. Its lock dependency still uses the per-set majority/write-lock health check. Actual shared namespace locks require `n - n/2`, so the cluster read probe is conservative: a four-client set can still admit reads with two clients while that probe returns 503. Node `/health/ready` uses the shared-lock threshold instead, because that is what object reads acquire.
- `HEAD` health probes keep header/status semantics and do not require response
bodies.
@@ -62,22 +63,24 @@ The existing `details.storage.ready` boolean and `connected` / `disconnected` st
| Probe | `readinessScope` | `source` |
| --- | --- | --- |
| `/health/ready`, `/minio/health/ready` | `write_quorum_and_pool_metadata` | `local_runtime` |
| `/health/ready`, `/minio/health/ready` | `read_quorum` | `local_runtime` |
| `/minio/health/cluster` | `write_quorum_and_pool_metadata` | `storage_inventory` |
| `/minio/health/cluster/read` | `read_quorum` | `storage_inventory` |
Node readiness additionally reports `details.storage.readQuorum`, `details.storage.writeQuorum`, and `details.poolMetadata.ready`. The metadata component's status is `writable` or `unavailable`; existing typed degradation reasons distinguish a write block from an inspection timeout. A healthy metadata writer alone no longer makes the storage component ready.
Node readiness additionally reports `details.storage.readQuorum`, `details.storage.writeQuorum`, and `details.poolMetadata.ready`. `details.storage.ready` follows read quorum. `details.lock.ready` on this probe follows shared-lock quorum; on `/minio/health/cluster` it follows exclusive-lock quorum. The metadata component's status is `writable` or `unavailable`. A blocked metadata writer is visible there and on the cluster-write probe; it does not by itself make `/health/ready` return 503.
Node storage quorum uses configured drives per set, all configured pools/sets, and their Standard storage-class data/parity layout. Missing, duplicate, unreachable, or unhealthy disk observations cannot supply extra quorum votes. The read quorum is the data-drive count; the write quorum is that count plus one when data and parity counts are equal. These are observations of available storage slots, not guarantees that a particular object's metadata, shards, or required locks are available.
For a healthy IAM and metadata writer in a four-node, one-drive-per-node EC 2+2 set:
For a healthy IAM and metadata writer in a four-node, one-drive-per-node EC 2+2 set, after startup has published `FullReady`:
| Surviving nodes | Storage read quorum | Storage write quorum | Pool metadata ready | Node HTTP / top-level ready |
| --- | --- | --- | --- | --- |
| 4 or 3 | true | true | true | 200 / true |
| 2 | true | false | true | 503 / false |
| 1 | false | false | true | 503 / false |
| All restored | true | true | true | 200 / true |
| Surviving nodes | Storage read quorum | Storage write quorum | Shared locks | Exclusive locks | Node `/health/ready` | `/minio/health/cluster` |
| --- | --- | --- | --- | --- | --- | --- |
| 4 or 3 | true | true | true | true | 200 | 200 |
| 2 | true | false | true | false | 200 | 503 |
| 1 | false | false | false | false | 503 | 503 |
| All restored | true | true | true | true | 200 | 200 |
With two of four nodes up, GET, HEAD, and List of a bucket that this process has not yet cached still succeed. PUT and other mutations fail because write quorum and exclusive locks are gone.
The node path reads local disk-handle health and reuses the same reachable-host observation as its lock dependency, including the existing `RUSTFS_HEALTH_READINESS_CACHE_TTL_MS` cache. Only `Online` drives count; a reachable host with a `Returning` drive does not yet prove data I/O has recovered. It does not call cluster `storage_info`, local `disk_info`, or add disk-info RPCs. The entire storage inventory snapshot has a separate 100 ms wait budget; expiry reports `storage_readiness_check_timeout` and fails closed. Pool metadata inspection retains its own 100 ms budget; a timeout is counted by `rustfs_pool_metadata_check_timeouts_total` and uses the last confirmed node-local gate state only within the same cache TTL. These observations are not an atomic cluster snapshot and do not bypass the existing lock-probe timing or cache policy.
+2 -2
View File
@@ -56,8 +56,8 @@ When the whole cluster (or several nodes) went down and nodes are brought back o
| `startup_finalization` | Last startup steps are being published. |
2. Logs say what the node waits for. The IAM recovery loop retries with backoff and logs `event="iam_bootstrap_retry_failed"` with an actionable `hint` field (for example, "storage read quorum not met yet; waiting for enough cluster nodes/disks to come online"). After repeated failures the level escalates from WARN to ERROR; this still does not kill the process.
3. Recovery is automatic. Once storage read quorum is available, pending nodes can finish IAM bootstrap on the next retry. `/health/ready` returns `200` when storage write quorum, the metadata write gate, IAM, and lock readiness are satisfied. Restarting pending nodes does not speed this up.
4. Check readiness detail while waiting. `/health/ready` (and `/minio/health/ready`) separate `storage` / `poolMetadata` / `iam` / `lock` readiness. `storage.ready` summarizes write quorum plus the metadata write gate; `storage.readQuorum` and `storage.writeQuorum` show the separate quorum observations. A healthy `poolMetadata.ready` does not imply storage quorum. `degradedReasons` lists machine-readable causes such as `storage_quorum_unavailable`, `storage_and_lock_unavailable`, or `pool_metadata_check_timeout`. See the [storage detail contract](../architecture/readiness-matrix.md#storage-detail-contract) for probe scopes and sampling limits:
3. Recovery is automatic. Once storage read quorum is available, pending nodes can finish IAM bootstrap on the next retry. `/health/ready` returns `200` when startup has published `FullReady` and the node can still serve reads (storage read quorum, IAM, and shared-lock quorum). Write quorum is reported separately and is required by `/minio/health/cluster`, not by Service membership. Restarting pending nodes does not speed this up.
4. Check readiness detail while waiting. `/health/ready` (and `/minio/health/ready`) separate `storage` / `poolMetadata` / `iam` / `lock` readiness. `storage.ready` follows read quorum (`readinessScope` is `read_quorum`); `storage.readQuorum` and `storage.writeQuorum` show the two observations. `poolMetadata.ready` is the metadata writer and does not by itself withdraw `/health/ready`. `degradedReasons` on this probe names read-path causes such as `storage_quorum_unavailable` and `storage_and_lock_unavailable`. `pool_meta_write_blocked` and `pool_metadata_check_timeout` stay on `/minio/health/cluster`. See the [storage detail contract](../architecture/readiness-matrix.md#storage-detail-contract) for probe scopes and sampling limits:
```bash
curl -s http://<node>:9000/health/ready | jq
+12 -11
View File
@@ -402,7 +402,9 @@ pub(crate) fn build_health_payload(ctx: HealthPayloadContext<'_>) -> Value {
if ctx.include_dependency_details {
payload["details"] = build_component_details(ctx.storage_ready, ctx.iam_ready, ctx.lock_quorum_ready, ctx.kms_ready);
payload["details"]["storage"]["readinessScope"] = json!(match ctx.probe {
HealthProbe::ClusterRead => "read_quorum",
// Node readiness is the Service membership probe: it follows read
// quorum. Cluster write keeps the write-quorum scope.
HealthProbe::ClusterRead | HealthProbe::Readiness => "read_quorum",
_ => "write_quorum_and_pool_metadata",
});
payload["details"]["storage"]["source"] = json!(match ctx.probe {
@@ -461,13 +463,16 @@ mod tests {
with_var(rustfs_config::ENV_HEALTH_MINIMAL_RESPONSE_ENABLE, Some("false"), || {
for (read_quorum, write_quorum, metadata_ready, lock_ready) in [
(true, true, true, true),
(true, false, true, true),
(true, false, false, true),
(true, false, true, false),
(false, false, true, false),
(true, true, false, true),
(true, true, true, true),
] {
let mut report = ready_report();
report.readiness.storage_ready = write_quorum && metadata_ready;
// The node collector stores read quorum here. Write quorum and the
// pool-metadata writer remain detail fields and do not set HTTP status.
report.readiness.storage_ready = read_quorum;
report.readiness.lock_quorum_ready = lock_ready;
report.storage_details = Some(crate::shared_types::StorageReadinessDetails {
read_quorum_ready: read_quorum,
@@ -475,7 +480,7 @@ mod tests {
pool_metadata_write_ready: metadata_ready,
});
let parts = build_health_response_parts(Method::GET, HealthProbe::Readiness, Some(&report), "rustfs", None, None);
let expected_ready = write_quorum && metadata_ready && lock_ready;
let expected_ready = read_quorum && lock_ready;
assert_eq!(
parts.status_code,
if expected_ready {
@@ -486,20 +491,16 @@ mod tests {
);
let payload = parts.payload.expect("GET readiness body");
assert_eq!(payload["ready"], expected_ready);
assert_eq!(payload["details"]["storage"]["ready"], write_quorum && metadata_ready);
assert_eq!(payload["details"]["storage"]["ready"], read_quorum);
assert_eq!(
payload["details"]["storage"]["status"],
if write_quorum && metadata_ready {
"connected"
} else {
"disconnected"
}
if read_quorum { "connected" } else { "disconnected" }
);
assert_eq!(payload["details"]["storage"]["readQuorum"], read_quorum);
assert_eq!(payload["details"]["storage"]["writeQuorum"], write_quorum);
assert_eq!(payload["details"]["poolMetadata"]["ready"], metadata_ready);
assert_eq!(payload["details"]["storage"]["source"], "local_runtime");
assert_eq!(payload["details"]["storage"]["readinessScope"], "write_quorum_and_pool_metadata");
assert_eq!(payload["details"]["storage"]["readinessScope"], "read_quorum");
assert!(
build_health_response_parts(Method::HEAD, HealthProbe::Readiness, Some(&report), "rustfs", None, None)
.payload
+150 -24
View File
@@ -375,10 +375,15 @@ enum ClusterHealthProbeKind {
#[derive(Debug, Clone, Copy, Default, PartialEq, Eq)]
pub struct LockQuorumStatus {
/// Exclusive-lock quorum. Cluster write probes and startup publication use this.
pub ready: bool,
/// Shared-lock quorum. Node `/health/ready` uses this so a readable survivor
/// stays in the Service after exclusive locks are no longer obtainable.
pub read_ready: bool,
pub connected_clients: usize,
pub total_clients: usize,
pub required_quorum: usize,
pub required_read_quorum: usize,
}
const DISK_STATE_OK: &str = "ok";
@@ -635,6 +640,20 @@ fn pool_read_quorum(info: &StorageInfo, pool_idx: usize, set_drive_count: usize)
pool_erasure_layout(info, pool_idx, set_drive_count).map(|(data_drives, _)| data_drives)
}
/// Exclusive namespace locks use a strict majority of lock clients.
fn exclusive_lock_quorum(total_clients: usize) -> usize {
if total_clients > 1 { (total_clients / 2) + 1 } else { 1 }
}
/// Shared namespace locks use the same client threshold as `DistributedLock::read_quorum`.
fn shared_lock_quorum(total_clients: usize) -> usize {
if total_clients > 1 {
total_clients - (total_clients / 2)
} else {
1
}
}
fn configured_readiness_topology(info: &StorageInfo) -> Option<(&[usize], &[usize])> {
if info.backend.total_sets.is_empty()
|| info.backend.total_sets.len() != info.backend.drives_per_set.len()
@@ -841,6 +860,27 @@ fn dependency_readiness_report_from_write_status(
}
}
/// `/health/ready` reports whether this process can serve reads. Write quorum,
/// the pool-metadata writer, and exclusive locks stay on `/minio/health/cluster`.
fn apply_node_service_readiness(
mut storage: StorageWriteReadinessStatus,
mut details: StorageReadinessDetails,
lock_read_ready: bool,
iam_ready: bool,
peer_health_ready: bool,
) -> (DependencyReadiness, StorageWriteReadinessStatus, StorageReadinessDetails) {
details.pool_metadata_write_ready = storage.ready;
storage.ready = details.read_quorum_ready;
storage.pool_metadata_reason = None;
let readiness = DependencyReadiness {
storage_ready: storage.ready,
iam_ready,
lock_quorum_ready: lock_read_ready,
peer_health_ready,
};
(readiness, storage, details)
}
pub async fn collect_dependency_readiness_report() -> DependencyReadinessReport {
let iam_ready_raw = runtime_sources::current_iam_ready();
let storage = if let Some(cached) = load_cached_storage_readiness().await {
@@ -879,7 +919,6 @@ pub async fn collect_node_readiness_report() -> DependencyReadinessReport {
if let Some(store) = runtime_sources::current_object_store_handle() {
let pool_metadata_status = store.pool_meta_write_status().await;
storage = apply_node_pool_metadata_timeout_policy(pool_metadata_write_readiness(pool_metadata_status), Instant::now());
details.pool_metadata_write_ready = storage.ready;
match node_storage_snapshot(store.as_ref(), &lock_observation.online_hosts).await {
Ok(info) => {
details.read_quorum_ready = storage_read_ready_from_runtime_state(&info);
@@ -889,13 +928,14 @@ pub async fn collect_node_readiness_report() -> DependencyReadinessReport {
Err(_) => {}
}
}
storage.ready &= details.write_quorum_ready;
let readiness = DependencyReadiness {
storage_ready: storage.ready,
iam_ready: runtime_sources::current_iam_ready(),
lock_quorum_ready: lock_observation.status.ready,
peer_health_ready: collect_peer_health_readiness(),
};
// The HTTP layer still returns 503 until startup publishes `FullReady`.
let (readiness, storage, details) = apply_node_service_readiness(
storage,
details,
lock_observation.status.read_ready,
runtime_sources::current_iam_ready(),
collect_peer_health_readiness(),
);
let mut report = dependency_readiness_report_from_write_status(readiness, storage);
report.storage_details = Some(details);
if storage_check_timed_out {
@@ -1101,17 +1141,16 @@ fn set_lock_quorum_status(online_hosts: &HashSet<String>, set_endpoints: &[Endpo
}
let connected_clients = total_clients.iter().filter(|host| online_hosts.contains(*host)).count();
let required_quorum = if total_clients_len > 1 {
(total_clients_len / 2) + 1
} else {
1
};
let required_quorum = exclusive_lock_quorum(total_clients_len);
let required_read_quorum = shared_lock_quorum(total_clients_len);
LockQuorumStatus {
ready: connected_clients >= required_quorum,
read_ready: connected_clients >= required_read_quorum,
connected_clients,
total_clients: total_clients_len,
required_quorum,
required_read_quorum,
}
}
@@ -1119,6 +1158,10 @@ fn aggregate_lock_quorum_status(pool_endpoints: &EndpointServerPools, online_hos
let mut connected_clients = 0usize;
let mut total_clients = 0usize;
let mut required_quorum = 0usize;
let mut required_read_quorum = 0usize;
let mut write_ready = true;
let mut read_ready = true;
let mut saw_set = false;
for pool in pool_endpoints.as_ref() {
for set_idx in 0..pool.set_count {
@@ -1135,29 +1178,26 @@ fn aggregate_lock_quorum_status(pool_endpoints: &EndpointServerPools, online_hos
return LockQuorumStatus::default();
}
saw_set = true;
connected_clients += status.connected_clients;
total_clients += status.total_clients;
required_quorum += status.required_quorum;
if !status.ready {
return LockQuorumStatus {
ready: false,
connected_clients,
total_clients,
required_quorum,
};
}
required_read_quorum += status.required_read_quorum;
write_ready &= status.ready;
read_ready &= status.read_ready;
}
}
if total_clients == 0 {
if !saw_set {
LockQuorumStatus::default()
} else {
LockQuorumStatus {
ready: true,
ready: write_ready,
read_ready,
connected_clients,
total_clients,
required_quorum,
required_read_quorum,
}
}
}
@@ -1171,9 +1211,11 @@ async fn collect_lock_quorum_observation_uncached() -> LockQuorumObservation {
return LockQuorumObservation {
status: LockQuorumStatus {
ready: true,
read_ready: true,
connected_clients: 1,
total_clients: 1,
required_quorum: 1,
required_read_quorum: 1,
},
online_hosts: HashSet::new(),
};
@@ -2254,9 +2296,11 @@ mod tests {
);
assert!(status.ready);
assert!(status.read_ready);
assert_eq!(status.connected_clients, 4);
assert_eq!(status.total_clients, 4);
assert_eq!(status.required_quorum, 4);
assert_eq!(status.required_read_quorum, 2);
}
#[test]
@@ -2304,6 +2348,86 @@ mod tests {
aggregate_lock_quorum_status(&pools, &["node1:9000"].into_iter().map(str::to_string).collect::<HashSet<_>>());
assert!(!status.ready);
assert!(!status.read_ready);
}
#[test]
fn node_service_readiness_follows_read_quorum_not_write_or_pool_metadata() {
let blocked = StorageWriteReadinessStatus {
ready: false,
pool_metadata_reason: Some(ReadinessDegradedReason::PoolMetaWriteBlocked),
};
let (readiness, storage, details) = apply_node_service_readiness(
blocked,
StorageReadinessDetails {
read_quorum_ready: true,
write_quorum_ready: false,
pool_metadata_write_ready: true,
},
true,
true,
true,
);
assert!(readiness.storage_ready);
assert!(readiness.lock_quorum_ready);
assert!(!details.write_quorum_ready);
assert!(!details.pool_metadata_write_ready);
assert!(storage.pool_metadata_reason.is_none());
let report = dependency_readiness_report_from_write_status(readiness, storage);
assert!(
report.degraded_reasons.is_empty(),
"a readable node must stay ready when write quorum and the metadata writer are lost, got {:?}",
report.degraded_reasons
);
let (readiness, storage, _) = apply_node_service_readiness(
StorageWriteReadinessStatus {
ready: true,
pool_metadata_reason: None,
},
StorageReadinessDetails {
read_quorum_ready: false,
write_quorum_ready: false,
pool_metadata_write_ready: false,
},
false,
true,
true,
);
assert_eq!(
dependency_readiness_report_from_write_status(readiness, storage).degraded_reasons,
vec![ReadinessDegradedReason::StorageAndLockUnavailable]
);
}
#[test]
fn shared_lock_quorum_stays_ready_when_two_of_four_clients_remain() {
let endpoints = (0..4)
.map(|disk_idx| Endpoint {
url: url::Url::parse(&format!("http://node{disk_idx}:9000/data")).expect("valid test endpoint"),
is_local: disk_idx == 0,
pool_idx: 0,
set_idx: 0,
disk_idx,
})
.collect::<Vec<_>>();
let online = ["node0:9000", "node1:9000"]
.into_iter()
.map(str::to_string)
.collect::<HashSet<_>>();
let status = set_lock_quorum_status(&online, &endpoints);
assert_eq!(status.total_clients, 4);
assert_eq!(status.connected_clients, 2);
assert_eq!(status.required_quorum, 3);
assert_eq!(status.required_read_quorum, 2);
assert!(!status.ready, "exclusive locks need three of four clients");
assert!(status.read_ready, "shared locks remain available at two of four clients");
let one = ["node0:9000"].into_iter().map(str::to_string).collect::<HashSet<_>>();
let below_read = set_lock_quorum_status(&one, &endpoints);
assert!(!below_read.ready);
assert!(!below_read.read_ready);
}
#[test]
@@ -2628,9 +2752,11 @@ mod tests {
let observation = LockQuorumObservation {
status: LockQuorumStatus {
ready: true,
read_ready: true,
connected_clients: 2,
total_clients: 3,
required_quorum: 2,
required_read_quorum: 2,
},
online_hosts: HashSet::from(["node-a:9000".to_owned(), "node-b:9000".to_owned()]),
};