diff --git a/crates/e2e_test/src/namespace_lock_quorum_test.rs b/crates/e2e_test/src/namespace_lock_quorum_test.rs index c9c196f23..14e3fe49e 100644 --- a/crates/e2e_test/src/namespace_lock_quorum_test.rs +++ b/crates/e2e_test/src/namespace_lock_quorum_test.rs @@ -462,7 +462,10 @@ async fn assert_node_readiness_tracks_quorum(cluster: &mut RustFSTestClusterEnvi } let write_ready = survivors >= 3; let read_quorum = survivors >= 2; - let expected_status = if write_ready { 200 } else { 503 }; + // `/health/ready` keeps a node that can still serve reads in the Service. + let node_ready = read_quorum; + let node_status = if node_ready { 200 } else { 503 }; + let cluster_status = if write_ready { 200 } else { 503 }; for (idx, client) in clients.iter().enumerate().take(survivors) { let url = &cluster.nodes[idx].url; let deadline = Instant::now() + Duration::from_secs(30); @@ -472,27 +475,27 @@ async fn assert_node_readiness_tracks_quorum(cluster: &mut RustFSTestClusterEnvi let response = http.get(format!("{url}/health/ready")).send().await?; let status = response.status().as_u16(); let payload: serde_json::Value = response.json().await?; - if status == expected_status - && payload["ready"] == write_ready - && payload["details"]["storage"]["ready"] == write_ready + if status == node_status + && payload["ready"] == node_ready + && payload["details"]["storage"]["ready"] == node_ready && payload["details"]["storage"]["readQuorum"] == read_quorum && payload["details"]["storage"]["writeQuorum"] == write_ready && payload["details"]["poolMetadata"]["ready"] == true && payload["details"]["iam"]["ready"] == true - && payload["details"]["lock"]["ready"] == write_ready + && payload["details"]["lock"]["ready"] == read_quorum { break payload; } assert!(Instant::now() < deadline, "node {idx}, survivors={survivors}: HTTP {status}, {payload}"); tokio::time::sleep(Duration::from_millis(200)).await; }; - assert_eq!(payload["details"]["storage"]["readinessScope"], "write_quorum_and_pool_metadata"); + assert_eq!(payload["details"]["storage"]["readinessScope"], "read_quorum"); assert_eq!(payload["details"]["storage"]["source"], "local_runtime"); assert_eq!( payload["details"]["storage"]["status"], - if write_ready { "connected" } else { "disconnected" } + if node_ready { "connected" } else { "disconnected" } ); - if !write_ready { + if !node_ready { assert!( payload["degradedReasons"] .as_array() @@ -504,10 +507,10 @@ async fn assert_node_readiness_tracks_quorum(cluster: &mut RustFSTestClusterEnvi for path in ["/health/ready", "/minio/health/ready"] { let head = http.head(format!("{url}{path}")).send().await?; - assert_eq!(head.status().as_u16(), expected_status, "HEAD {path}, survivors={survivors}"); + assert_eq!(head.status().as_u16(), node_status, "HEAD {path}, survivors={survivors}"); assert!(head.bytes().await?.is_empty()); let response = http.get(format!("{url}{path}")).send().await?; - assert_eq!(response.status().as_u16(), expected_status); + assert_eq!(response.status().as_u16(), node_status); let body: serde_json::Value = response.json().await?; assert_eq!(body["details"]["storage"], payload["details"]["storage"]); assert_eq!(body["details"]["poolMetadata"], payload["details"]["poolMetadata"]); @@ -526,7 +529,7 @@ async fn assert_node_readiness_tracks_quorum(cluster: &mut RustFSTestClusterEnvi let response = http.get(format!("{url}{path}")).send().await?; let status = response.status().as_u16(); let body: serde_json::Value = response.json().await?; - if status == expected_status + if status == cluster_status && body["details"]["storage"]["ready"] == storage_ready && body["details"]["lock"]["ready"] == write_ready { @@ -559,11 +562,12 @@ async fn assert_node_readiness_tracks_quorum(cluster: &mut RustFSTestClusterEnvi let get = client.get_object().bucket(BUCKET).key(seed_key).send().await; let get_status = match get { Ok(object) => { + assert!(read_quorum, "GET must fail once read quorum is lost"); assert_eq!(object.body.collect().await?.into_bytes().as_ref(), seed_body); 200 } Err(error) => { - assert!(!write_ready, "GET must succeed on a ready cluster: {error:?}"); + assert!(!read_quorum, "GET must succeed while read quorum holds: {error:?}"); let status = error .raw_response() .expect("GET should have an HTTP response") @@ -576,11 +580,12 @@ async fn assert_node_readiness_tracks_quorum(cluster: &mut RustFSTestClusterEnvi let list = client.list_objects_v2().bucket(BUCKET).send().await; let list_status = match list { Ok(result) => { + assert!(read_quorum, "listing must fail once read quorum is lost"); assert!(result.contents().iter().any(|object| object.key() == Some(seed_key))); 200 } Err(error) => { - assert!(!write_ready, "listing must succeed on a ready cluster: {error:?}"); + assert!(!read_quorum, "listing must succeed while read quorum holds: {error:?}"); let status = error .raw_response() .expect("LIST should have an HTTP response") diff --git a/crates/ecstore/src/store/bucket.rs b/crates/ecstore/src/store/bucket.rs index fdd539f4a..b4b961f9b 100644 --- a/crates/ecstore/src/store/bucket.rs +++ b/crates/ecstore/src/store/bucket.rs @@ -2430,6 +2430,99 @@ mod tests { restore_set_disks(&store, set, offline).await; } + /// EC 2+2 with two disks offline still has object read quorum and has lost + /// write quorum. A node that has not cached the bucket must still serve + /// HEAD bucket, GET, HEAD object, and List; a write must fail. + #[tokio::test] + #[serial] + async fn cold_bucket_reads_succeed_below_write_quorum() { + let (_temp_dir, store) = setup_bucket_quorum_test_env(&[4], Some(2)).await; + metadata_sys::init_bucket_metadata_sys(store.clone(), Vec::new()).await; + let bucket = format!("cold-read-quorum-{}", Uuid::new_v4().simple()); + let object = "seed-object"; + let body = b"cold bucket reads must survive lost write quorum".to_vec(); + store + .make_bucket(&bucket, &MakeBucketOptions::default()) + .await + .expect("healthy namespace should accept bucket creation"); + store + .put_object(&bucket, object, &mut PutObjReader::from_vec(body.clone()), &ObjectOptions::default()) + .await + .expect("healthy erasure set should accept the seed object"); + + metadata_sys::init_bucket_metadata_sys(store.clone(), Vec::new()).await; + assert!( + metadata_sys::get_in(&store.ctx, &bucket).await.is_err(), + "the replacement metadata system must start without this bucket" + ); + let set = &store.pools[0].disk_set[0]; + let lock = set + .new_ns_lock(&bucket, object) + .await + .expect("seed namespace lock should resolve"); + drop( + lock.get_write_lock(Duration::from_secs(30)) + .await + .expect("seed physical fanout must finish before taking disks offline"), + ); + let offline = take_set_disks_offline(&store, set, &[0, 1]).await; + + let info = store + .get_bucket_info(&bucket, &BucketOptions::default()) + .await + .expect("HEAD bucket must succeed at read quorum before the bucket is cached"); + assert_eq!(info.name, bucket); + + let mut reader = store + .get_object_reader(&bucket, object, None, Default::default(), &ObjectOptions::default()) + .await + .expect("GET must succeed for a bucket that was not loaded at startup"); + let mut restored = Vec::new(); + reader + .stream + .read_to_end(&mut restored) + .await + .expect("quorum read should reconstruct the body"); + assert_eq!(restored, body); + drop(reader); + + let object_info = store + .get_object_info(&bucket, object, &ObjectOptions::default()) + .await + .expect("HEAD object must succeed at read quorum"); + assert_eq!(object_info.name, object); + assert_eq!(object_info.size, i64::try_from(body.len()).expect("body length fits i64")); + + let listed = store + .clone() + .list_objects_v2(&bucket, "", None, None, 100, false, None, false) + .await + .expect("List must succeed at read quorum"); + assert!( + listed.objects.iter().any(|item| item.name == object), + "list should include the seed object, got {:?}", + listed.objects.iter().map(|item| item.name.clone()).collect::>() + ); + assert!( + metadata_sys::get_in(&store.ctx, &bucket).await.is_ok(), + "the cold load should publish authoritative metadata" + ); + + let write_error = store + .put_object(&bucket, "rejected-object", &mut PutObjReader::from_vec(body), &ObjectOptions::default()) + .await + .expect_err("writes must still fail without write quorum"); + assert!( + matches!( + write_error, + StorageError::InsufficientWriteQuorum(_, _) | StorageError::ErasureWriteQuorum + ), + "writes must fail with a write-quorum error, got {write_error}" + ); + + restore_set_disks(&store, set, offline).await; + } + #[tokio::test] #[serial] async fn lazy_metadata_load_rejects_below_read_quorum() { diff --git a/docs/architecture/readiness-matrix.md b/docs/architecture/readiness-matrix.md index b050be8c0..c034c3024 100644 --- a/docs/architecture/readiness-matrix.md +++ b/docs/architecture/readiness-matrix.md @@ -39,20 +39,21 @@ an unknown or unsupported peer-health snapshot degrades readiness with - Liveness reports process availability and must not depend on storage, IAM, lock quorum, or peer health. -- Node readiness reports the node's observed storage write quorum, local pool metadata write gate, IAM, and lock quorum. Runtime storage diagnostics do not change the existing startup `FullReady` publication gate or S3 request admission. -- A blocked pool metadata writer degrades node and cluster-write readiness with - `pool_meta_write_blocked`. Metadata save-gate inspection is bounded to 100 ms; - contention reports `pool_metadata_check_timeout` without installing a block. - Node readiness keeps the last confirmed writable observation for the normal - readiness-cache TTL during a transient inspection timeout, while preserving a - confirmed block and failing closed when no fresh observation exists. +- Node readiness (`/health/ready`, `/minio/health/ready`) reports whether this process can serve reads: storage read quorum, IAM, and shared-lock quorum. It is the Kubernetes Service membership signal. Losing write quorum, the pool-metadata writer, or exclusive-lock quorum does not remove a node that can still read. The public health handler still returns 503 until startup publishes `FullReady`, so a process that has never opened S3 admission does not join the Service. Runtime storage diagnostics do not change that startup gate. +- A blocked pool metadata writer degrades cluster-write readiness with + `pool_meta_write_blocked`. It does not withdraw `/health/ready`, and the node + probe omits that reason from `degradedReasons`. Metadata save-gate inspection + is bounded to 100 ms; contention reports `pool_metadata_check_timeout` on the + cluster-write probe without installing a block. The node probe keeps the last + confirmed writable observation for `details.poolMetadata` during a transient + inspection timeout, while preserving a confirmed block and failing that + detail closed when no fresh observation exists. - The authenticated cluster snapshot extends its existing node-local metadata gate inspection with safe reason, failure phase, and original block time. It distinguishes timeout from a block and changes no admission or recovery decision. Runtime readiness and gate status are separate bounded observations. -- Cluster write readiness requires write quorum and the runtime dependency - readiness used by `FullReady`. -- Cluster read readiness uses the storage read-quorum path and cluster-health timeout behavior. Its lock dependency still uses the per-set majority/write-lock health check. Actual shared namespace locks require `ceil(lock_clients / 2)`, so the cluster read probe is conservative: a four-client set can still admit some reads with two clients while the probe returns 503. This diagnostic change does not lower that probe's lock threshold. +- Cluster write readiness (`/minio/health/cluster`) requires storage write quorum, the pool-metadata write gate, and exclusive-lock quorum. Use it when a caller must know that new writes can commit. It is not the Service membership probe. +- Cluster read readiness uses the storage read-quorum path and cluster-health timeout behavior. Its lock dependency still uses the per-set majority/write-lock health check. Actual shared namespace locks require `n - n/2`, so the cluster read probe is conservative: a four-client set can still admit reads with two clients while that probe returns 503. Node `/health/ready` uses the shared-lock threshold instead, because that is what object reads acquire. - `HEAD` health probes keep header/status semantics and do not require response bodies. @@ -62,22 +63,24 @@ The existing `details.storage.ready` boolean and `connected` / `disconnected` st | Probe | `readinessScope` | `source` | | --- | --- | --- | -| `/health/ready`, `/minio/health/ready` | `write_quorum_and_pool_metadata` | `local_runtime` | +| `/health/ready`, `/minio/health/ready` | `read_quorum` | `local_runtime` | | `/minio/health/cluster` | `write_quorum_and_pool_metadata` | `storage_inventory` | | `/minio/health/cluster/read` | `read_quorum` | `storage_inventory` | -Node readiness additionally reports `details.storage.readQuorum`, `details.storage.writeQuorum`, and `details.poolMetadata.ready`. The metadata component's status is `writable` or `unavailable`; existing typed degradation reasons distinguish a write block from an inspection timeout. A healthy metadata writer alone no longer makes the storage component ready. +Node readiness additionally reports `details.storage.readQuorum`, `details.storage.writeQuorum`, and `details.poolMetadata.ready`. `details.storage.ready` follows read quorum. `details.lock.ready` on this probe follows shared-lock quorum; on `/minio/health/cluster` it follows exclusive-lock quorum. The metadata component's status is `writable` or `unavailable`. A blocked metadata writer is visible there and on the cluster-write probe; it does not by itself make `/health/ready` return 503. Node storage quorum uses configured drives per set, all configured pools/sets, and their Standard storage-class data/parity layout. Missing, duplicate, unreachable, or unhealthy disk observations cannot supply extra quorum votes. The read quorum is the data-drive count; the write quorum is that count plus one when data and parity counts are equal. These are observations of available storage slots, not guarantees that a particular object's metadata, shards, or required locks are available. -For a healthy IAM and metadata writer in a four-node, one-drive-per-node EC 2+2 set: +For a healthy IAM and metadata writer in a four-node, one-drive-per-node EC 2+2 set, after startup has published `FullReady`: -| Surviving nodes | Storage read quorum | Storage write quorum | Pool metadata ready | Node HTTP / top-level ready | -| --- | --- | --- | --- | --- | -| 4 or 3 | true | true | true | 200 / true | -| 2 | true | false | true | 503 / false | -| 1 | false | false | true | 503 / false | -| All restored | true | true | true | 200 / true | +| Surviving nodes | Storage read quorum | Storage write quorum | Shared locks | Exclusive locks | Node `/health/ready` | `/minio/health/cluster` | +| --- | --- | --- | --- | --- | --- | --- | +| 4 or 3 | true | true | true | true | 200 | 200 | +| 2 | true | false | true | false | 200 | 503 | +| 1 | false | false | false | false | 503 | 503 | +| All restored | true | true | true | true | 200 | 200 | + +With two of four nodes up, GET, HEAD, and List of a bucket that this process has not yet cached still succeed. PUT and other mutations fail because write quorum and exclusive locks are gone. The node path reads local disk-handle health and reuses the same reachable-host observation as its lock dependency, including the existing `RUSTFS_HEALTH_READINESS_CACHE_TTL_MS` cache. Only `Online` drives count; a reachable host with a `Returning` drive does not yet prove data I/O has recovered. It does not call cluster `storage_info`, local `disk_info`, or add disk-info RPCs. The entire storage inventory snapshot has a separate 100 ms wait budget; expiry reports `storage_readiness_check_timeout` and fails closed. Pool metadata inspection retains its own 100 ms budget; a timeout is counted by `rustfs_pool_metadata_check_timeouts_total` and uses the last confirmed node-local gate state only within the same cache TTL. These observations are not an atomic cluster snapshot and do not bypass the existing lock-probe timing or cache policy. diff --git a/docs/operations/rolling-restart.md b/docs/operations/rolling-restart.md index 6dbd4297a..a3df35e01 100644 --- a/docs/operations/rolling-restart.md +++ b/docs/operations/rolling-restart.md @@ -56,8 +56,8 @@ When the whole cluster (or several nodes) went down and nodes are brought back o | `startup_finalization` | Last startup steps are being published. | 2. Logs say what the node waits for. The IAM recovery loop retries with backoff and logs `event="iam_bootstrap_retry_failed"` with an actionable `hint` field (for example, "storage read quorum not met yet; waiting for enough cluster nodes/disks to come online"). After repeated failures the level escalates from WARN to ERROR; this still does not kill the process. -3. Recovery is automatic. Once storage read quorum is available, pending nodes can finish IAM bootstrap on the next retry. `/health/ready` returns `200` when storage write quorum, the metadata write gate, IAM, and lock readiness are satisfied. Restarting pending nodes does not speed this up. -4. Check readiness detail while waiting. `/health/ready` (and `/minio/health/ready`) separate `storage` / `poolMetadata` / `iam` / `lock` readiness. `storage.ready` summarizes write quorum plus the metadata write gate; `storage.readQuorum` and `storage.writeQuorum` show the separate quorum observations. A healthy `poolMetadata.ready` does not imply storage quorum. `degradedReasons` lists machine-readable causes such as `storage_quorum_unavailable`, `storage_and_lock_unavailable`, or `pool_metadata_check_timeout`. See the [storage detail contract](../architecture/readiness-matrix.md#storage-detail-contract) for probe scopes and sampling limits: +3. Recovery is automatic. Once storage read quorum is available, pending nodes can finish IAM bootstrap on the next retry. `/health/ready` returns `200` when startup has published `FullReady` and the node can still serve reads (storage read quorum, IAM, and shared-lock quorum). Write quorum is reported separately and is required by `/minio/health/cluster`, not by Service membership. Restarting pending nodes does not speed this up. +4. Check readiness detail while waiting. `/health/ready` (and `/minio/health/ready`) separate `storage` / `poolMetadata` / `iam` / `lock` readiness. `storage.ready` follows read quorum (`readinessScope` is `read_quorum`); `storage.readQuorum` and `storage.writeQuorum` show the two observations. `poolMetadata.ready` is the metadata writer and does not by itself withdraw `/health/ready`. `degradedReasons` on this probe names read-path causes such as `storage_quorum_unavailable` and `storage_and_lock_unavailable`. `pool_meta_write_blocked` and `pool_metadata_check_timeout` stay on `/minio/health/cluster`. See the [storage detail contract](../architecture/readiness-matrix.md#storage-detail-contract) for probe scopes and sampling limits: ```bash curl -s http://:9000/health/ready | jq diff --git a/rustfs/src/server/health.rs b/rustfs/src/server/health.rs index b035a3f9a..e4103d5ff 100644 --- a/rustfs/src/server/health.rs +++ b/rustfs/src/server/health.rs @@ -402,7 +402,9 @@ pub(crate) fn build_health_payload(ctx: HealthPayloadContext<'_>) -> Value { if ctx.include_dependency_details { payload["details"] = build_component_details(ctx.storage_ready, ctx.iam_ready, ctx.lock_quorum_ready, ctx.kms_ready); payload["details"]["storage"]["readinessScope"] = json!(match ctx.probe { - HealthProbe::ClusterRead => "read_quorum", + // Node readiness is the Service membership probe: it follows read + // quorum. Cluster write keeps the write-quorum scope. + HealthProbe::ClusterRead | HealthProbe::Readiness => "read_quorum", _ => "write_quorum_and_pool_metadata", }); payload["details"]["storage"]["source"] = json!(match ctx.probe { @@ -461,13 +463,16 @@ mod tests { with_var(rustfs_config::ENV_HEALTH_MINIMAL_RESPONSE_ENABLE, Some("false"), || { for (read_quorum, write_quorum, metadata_ready, lock_ready) in [ (true, true, true, true), + (true, false, true, true), + (true, false, false, true), (true, false, true, false), (false, false, true, false), (true, true, false, true), - (true, true, true, true), ] { let mut report = ready_report(); - report.readiness.storage_ready = write_quorum && metadata_ready; + // The node collector stores read quorum here. Write quorum and the + // pool-metadata writer remain detail fields and do not set HTTP status. + report.readiness.storage_ready = read_quorum; report.readiness.lock_quorum_ready = lock_ready; report.storage_details = Some(crate::shared_types::StorageReadinessDetails { read_quorum_ready: read_quorum, @@ -475,7 +480,7 @@ mod tests { pool_metadata_write_ready: metadata_ready, }); let parts = build_health_response_parts(Method::GET, HealthProbe::Readiness, Some(&report), "rustfs", None, None); - let expected_ready = write_quorum && metadata_ready && lock_ready; + let expected_ready = read_quorum && lock_ready; assert_eq!( parts.status_code, if expected_ready { @@ -486,20 +491,16 @@ mod tests { ); let payload = parts.payload.expect("GET readiness body"); assert_eq!(payload["ready"], expected_ready); - assert_eq!(payload["details"]["storage"]["ready"], write_quorum && metadata_ready); + assert_eq!(payload["details"]["storage"]["ready"], read_quorum); assert_eq!( payload["details"]["storage"]["status"], - if write_quorum && metadata_ready { - "connected" - } else { - "disconnected" - } + if read_quorum { "connected" } else { "disconnected" } ); assert_eq!(payload["details"]["storage"]["readQuorum"], read_quorum); assert_eq!(payload["details"]["storage"]["writeQuorum"], write_quorum); assert_eq!(payload["details"]["poolMetadata"]["ready"], metadata_ready); assert_eq!(payload["details"]["storage"]["source"], "local_runtime"); - assert_eq!(payload["details"]["storage"]["readinessScope"], "write_quorum_and_pool_metadata"); + assert_eq!(payload["details"]["storage"]["readinessScope"], "read_quorum"); assert!( build_health_response_parts(Method::HEAD, HealthProbe::Readiness, Some(&report), "rustfs", None, None) .payload diff --git a/rustfs/src/server/readiness.rs b/rustfs/src/server/readiness.rs index d00401679..bdb455aff 100644 --- a/rustfs/src/server/readiness.rs +++ b/rustfs/src/server/readiness.rs @@ -375,10 +375,15 @@ enum ClusterHealthProbeKind { #[derive(Debug, Clone, Copy, Default, PartialEq, Eq)] pub struct LockQuorumStatus { + /// Exclusive-lock quorum. Cluster write probes and startup publication use this. pub ready: bool, + /// Shared-lock quorum. Node `/health/ready` uses this so a readable survivor + /// stays in the Service after exclusive locks are no longer obtainable. + pub read_ready: bool, pub connected_clients: usize, pub total_clients: usize, pub required_quorum: usize, + pub required_read_quorum: usize, } const DISK_STATE_OK: &str = "ok"; @@ -635,6 +640,20 @@ fn pool_read_quorum(info: &StorageInfo, pool_idx: usize, set_drive_count: usize) pool_erasure_layout(info, pool_idx, set_drive_count).map(|(data_drives, _)| data_drives) } +/// Exclusive namespace locks use a strict majority of lock clients. +fn exclusive_lock_quorum(total_clients: usize) -> usize { + if total_clients > 1 { (total_clients / 2) + 1 } else { 1 } +} + +/// Shared namespace locks use the same client threshold as `DistributedLock::read_quorum`. +fn shared_lock_quorum(total_clients: usize) -> usize { + if total_clients > 1 { + total_clients - (total_clients / 2) + } else { + 1 + } +} + fn configured_readiness_topology(info: &StorageInfo) -> Option<(&[usize], &[usize])> { if info.backend.total_sets.is_empty() || info.backend.total_sets.len() != info.backend.drives_per_set.len() @@ -841,6 +860,27 @@ fn dependency_readiness_report_from_write_status( } } +/// `/health/ready` reports whether this process can serve reads. Write quorum, +/// the pool-metadata writer, and exclusive locks stay on `/minio/health/cluster`. +fn apply_node_service_readiness( + mut storage: StorageWriteReadinessStatus, + mut details: StorageReadinessDetails, + lock_read_ready: bool, + iam_ready: bool, + peer_health_ready: bool, +) -> (DependencyReadiness, StorageWriteReadinessStatus, StorageReadinessDetails) { + details.pool_metadata_write_ready = storage.ready; + storage.ready = details.read_quorum_ready; + storage.pool_metadata_reason = None; + let readiness = DependencyReadiness { + storage_ready: storage.ready, + iam_ready, + lock_quorum_ready: lock_read_ready, + peer_health_ready, + }; + (readiness, storage, details) +} + pub async fn collect_dependency_readiness_report() -> DependencyReadinessReport { let iam_ready_raw = runtime_sources::current_iam_ready(); let storage = if let Some(cached) = load_cached_storage_readiness().await { @@ -879,7 +919,6 @@ pub async fn collect_node_readiness_report() -> DependencyReadinessReport { if let Some(store) = runtime_sources::current_object_store_handle() { let pool_metadata_status = store.pool_meta_write_status().await; storage = apply_node_pool_metadata_timeout_policy(pool_metadata_write_readiness(pool_metadata_status), Instant::now()); - details.pool_metadata_write_ready = storage.ready; match node_storage_snapshot(store.as_ref(), &lock_observation.online_hosts).await { Ok(info) => { details.read_quorum_ready = storage_read_ready_from_runtime_state(&info); @@ -889,13 +928,14 @@ pub async fn collect_node_readiness_report() -> DependencyReadinessReport { Err(_) => {} } } - storage.ready &= details.write_quorum_ready; - let readiness = DependencyReadiness { - storage_ready: storage.ready, - iam_ready: runtime_sources::current_iam_ready(), - lock_quorum_ready: lock_observation.status.ready, - peer_health_ready: collect_peer_health_readiness(), - }; + // The HTTP layer still returns 503 until startup publishes `FullReady`. + let (readiness, storage, details) = apply_node_service_readiness( + storage, + details, + lock_observation.status.read_ready, + runtime_sources::current_iam_ready(), + collect_peer_health_readiness(), + ); let mut report = dependency_readiness_report_from_write_status(readiness, storage); report.storage_details = Some(details); if storage_check_timed_out { @@ -1101,17 +1141,16 @@ fn set_lock_quorum_status(online_hosts: &HashSet, set_endpoints: &[Endpo } let connected_clients = total_clients.iter().filter(|host| online_hosts.contains(*host)).count(); - let required_quorum = if total_clients_len > 1 { - (total_clients_len / 2) + 1 - } else { - 1 - }; + let required_quorum = exclusive_lock_quorum(total_clients_len); + let required_read_quorum = shared_lock_quorum(total_clients_len); LockQuorumStatus { ready: connected_clients >= required_quorum, + read_ready: connected_clients >= required_read_quorum, connected_clients, total_clients: total_clients_len, required_quorum, + required_read_quorum, } } @@ -1119,6 +1158,10 @@ fn aggregate_lock_quorum_status(pool_endpoints: &EndpointServerPools, online_hos let mut connected_clients = 0usize; let mut total_clients = 0usize; let mut required_quorum = 0usize; + let mut required_read_quorum = 0usize; + let mut write_ready = true; + let mut read_ready = true; + let mut saw_set = false; for pool in pool_endpoints.as_ref() { for set_idx in 0..pool.set_count { @@ -1135,29 +1178,26 @@ fn aggregate_lock_quorum_status(pool_endpoints: &EndpointServerPools, online_hos return LockQuorumStatus::default(); } + saw_set = true; connected_clients += status.connected_clients; total_clients += status.total_clients; required_quorum += status.required_quorum; - - if !status.ready { - return LockQuorumStatus { - ready: false, - connected_clients, - total_clients, - required_quorum, - }; - } + required_read_quorum += status.required_read_quorum; + write_ready &= status.ready; + read_ready &= status.read_ready; } } - if total_clients == 0 { + if !saw_set { LockQuorumStatus::default() } else { LockQuorumStatus { - ready: true, + ready: write_ready, + read_ready, connected_clients, total_clients, required_quorum, + required_read_quorum, } } } @@ -1171,9 +1211,11 @@ async fn collect_lock_quorum_observation_uncached() -> LockQuorumObservation { return LockQuorumObservation { status: LockQuorumStatus { ready: true, + read_ready: true, connected_clients: 1, total_clients: 1, required_quorum: 1, + required_read_quorum: 1, }, online_hosts: HashSet::new(), }; @@ -2254,9 +2296,11 @@ mod tests { ); assert!(status.ready); + assert!(status.read_ready); assert_eq!(status.connected_clients, 4); assert_eq!(status.total_clients, 4); assert_eq!(status.required_quorum, 4); + assert_eq!(status.required_read_quorum, 2); } #[test] @@ -2304,6 +2348,86 @@ mod tests { aggregate_lock_quorum_status(&pools, &["node1:9000"].into_iter().map(str::to_string).collect::>()); assert!(!status.ready); + assert!(!status.read_ready); + } + + #[test] + fn node_service_readiness_follows_read_quorum_not_write_or_pool_metadata() { + let blocked = StorageWriteReadinessStatus { + ready: false, + pool_metadata_reason: Some(ReadinessDegradedReason::PoolMetaWriteBlocked), + }; + let (readiness, storage, details) = apply_node_service_readiness( + blocked, + StorageReadinessDetails { + read_quorum_ready: true, + write_quorum_ready: false, + pool_metadata_write_ready: true, + }, + true, + true, + true, + ); + assert!(readiness.storage_ready); + assert!(readiness.lock_quorum_ready); + assert!(!details.write_quorum_ready); + assert!(!details.pool_metadata_write_ready); + assert!(storage.pool_metadata_reason.is_none()); + let report = dependency_readiness_report_from_write_status(readiness, storage); + assert!( + report.degraded_reasons.is_empty(), + "a readable node must stay ready when write quorum and the metadata writer are lost, got {:?}", + report.degraded_reasons + ); + + let (readiness, storage, _) = apply_node_service_readiness( + StorageWriteReadinessStatus { + ready: true, + pool_metadata_reason: None, + }, + StorageReadinessDetails { + read_quorum_ready: false, + write_quorum_ready: false, + pool_metadata_write_ready: false, + }, + false, + true, + true, + ); + assert_eq!( + dependency_readiness_report_from_write_status(readiness, storage).degraded_reasons, + vec![ReadinessDegradedReason::StorageAndLockUnavailable] + ); + } + + #[test] + fn shared_lock_quorum_stays_ready_when_two_of_four_clients_remain() { + let endpoints = (0..4) + .map(|disk_idx| Endpoint { + url: url::Url::parse(&format!("http://node{disk_idx}:9000/data")).expect("valid test endpoint"), + is_local: disk_idx == 0, + pool_idx: 0, + set_idx: 0, + disk_idx, + }) + .collect::>(); + let online = ["node0:9000", "node1:9000"] + .into_iter() + .map(str::to_string) + .collect::>(); + let status = set_lock_quorum_status(&online, &endpoints); + + assert_eq!(status.total_clients, 4); + assert_eq!(status.connected_clients, 2); + assert_eq!(status.required_quorum, 3); + assert_eq!(status.required_read_quorum, 2); + assert!(!status.ready, "exclusive locks need three of four clients"); + assert!(status.read_ready, "shared locks remain available at two of four clients"); + + let one = ["node0:9000"].into_iter().map(str::to_string).collect::>(); + let below_read = set_lock_quorum_status(&one, &endpoints); + assert!(!below_read.ready); + assert!(!below_read.read_ready); } #[test] @@ -2628,9 +2752,11 @@ mod tests { let observation = LockQuorumObservation { status: LockQuorumStatus { ready: true, + read_ready: true, connected_clients: 2, total_clients: 3, required_quorum: 2, + required_read_quorum: 2, }, online_hosts: HashSet::from(["node-a:9000".to_owned(), "node-b:9000".to_owned()]), };