mirror of
https://github.com/rustfs/rustfs.git
synced 2026-10-04 12:31:36 +00:00
fix: admit cold reads and readiness at read quorum (#8156)
EC 2+2 still meets read quorum with two of four nodes up. A survivor that had not cached a bucket returned 503 on GET, HEAD, and List, and /health/ready left the Service once write quorum was lost. Reads and Service membership now follow read quorum and shared locks. Writes and /minio/health/cluster still require write quorum and exclusive locks. Signed-off-by: loverustfs <155562731+loverustfs@users.noreply.github.com> Co-authored-by: Hauser <housemecn@gmail.com>
This commit is contained in:
@@ -462,7 +462,10 @@ async fn assert_node_readiness_tracks_quorum(cluster: &mut RustFSTestClusterEnvi
|
||||
}
|
||||
let write_ready = survivors >= 3;
|
||||
let read_quorum = survivors >= 2;
|
||||
let expected_status = if write_ready { 200 } else { 503 };
|
||||
// `/health/ready` keeps a node that can still serve reads in the Service.
|
||||
let node_ready = read_quorum;
|
||||
let node_status = if node_ready { 200 } else { 503 };
|
||||
let cluster_status = if write_ready { 200 } else { 503 };
|
||||
for (idx, client) in clients.iter().enumerate().take(survivors) {
|
||||
let url = &cluster.nodes[idx].url;
|
||||
let deadline = Instant::now() + Duration::from_secs(30);
|
||||
@@ -472,27 +475,27 @@ async fn assert_node_readiness_tracks_quorum(cluster: &mut RustFSTestClusterEnvi
|
||||
let response = http.get(format!("{url}/health/ready")).send().await?;
|
||||
let status = response.status().as_u16();
|
||||
let payload: serde_json::Value = response.json().await?;
|
||||
if status == expected_status
|
||||
&& payload["ready"] == write_ready
|
||||
&& payload["details"]["storage"]["ready"] == write_ready
|
||||
if status == node_status
|
||||
&& payload["ready"] == node_ready
|
||||
&& payload["details"]["storage"]["ready"] == node_ready
|
||||
&& payload["details"]["storage"]["readQuorum"] == read_quorum
|
||||
&& payload["details"]["storage"]["writeQuorum"] == write_ready
|
||||
&& payload["details"]["poolMetadata"]["ready"] == true
|
||||
&& payload["details"]["iam"]["ready"] == true
|
||||
&& payload["details"]["lock"]["ready"] == write_ready
|
||||
&& payload["details"]["lock"]["ready"] == read_quorum
|
||||
{
|
||||
break payload;
|
||||
}
|
||||
assert!(Instant::now() < deadline, "node {idx}, survivors={survivors}: HTTP {status}, {payload}");
|
||||
tokio::time::sleep(Duration::from_millis(200)).await;
|
||||
};
|
||||
assert_eq!(payload["details"]["storage"]["readinessScope"], "write_quorum_and_pool_metadata");
|
||||
assert_eq!(payload["details"]["storage"]["readinessScope"], "read_quorum");
|
||||
assert_eq!(payload["details"]["storage"]["source"], "local_runtime");
|
||||
assert_eq!(
|
||||
payload["details"]["storage"]["status"],
|
||||
if write_ready { "connected" } else { "disconnected" }
|
||||
if node_ready { "connected" } else { "disconnected" }
|
||||
);
|
||||
if !write_ready {
|
||||
if !node_ready {
|
||||
assert!(
|
||||
payload["degradedReasons"]
|
||||
.as_array()
|
||||
@@ -504,10 +507,10 @@ async fn assert_node_readiness_tracks_quorum(cluster: &mut RustFSTestClusterEnvi
|
||||
|
||||
for path in ["/health/ready", "/minio/health/ready"] {
|
||||
let head = http.head(format!("{url}{path}")).send().await?;
|
||||
assert_eq!(head.status().as_u16(), expected_status, "HEAD {path}, survivors={survivors}");
|
||||
assert_eq!(head.status().as_u16(), node_status, "HEAD {path}, survivors={survivors}");
|
||||
assert!(head.bytes().await?.is_empty());
|
||||
let response = http.get(format!("{url}{path}")).send().await?;
|
||||
assert_eq!(response.status().as_u16(), expected_status);
|
||||
assert_eq!(response.status().as_u16(), node_status);
|
||||
let body: serde_json::Value = response.json().await?;
|
||||
assert_eq!(body["details"]["storage"], payload["details"]["storage"]);
|
||||
assert_eq!(body["details"]["poolMetadata"], payload["details"]["poolMetadata"]);
|
||||
@@ -526,7 +529,7 @@ async fn assert_node_readiness_tracks_quorum(cluster: &mut RustFSTestClusterEnvi
|
||||
let response = http.get(format!("{url}{path}")).send().await?;
|
||||
let status = response.status().as_u16();
|
||||
let body: serde_json::Value = response.json().await?;
|
||||
if status == expected_status
|
||||
if status == cluster_status
|
||||
&& body["details"]["storage"]["ready"] == storage_ready
|
||||
&& body["details"]["lock"]["ready"] == write_ready
|
||||
{
|
||||
@@ -559,11 +562,12 @@ async fn assert_node_readiness_tracks_quorum(cluster: &mut RustFSTestClusterEnvi
|
||||
let get = client.get_object().bucket(BUCKET).key(seed_key).send().await;
|
||||
let get_status = match get {
|
||||
Ok(object) => {
|
||||
assert!(read_quorum, "GET must fail once read quorum is lost");
|
||||
assert_eq!(object.body.collect().await?.into_bytes().as_ref(), seed_body);
|
||||
200
|
||||
}
|
||||
Err(error) => {
|
||||
assert!(!write_ready, "GET must succeed on a ready cluster: {error:?}");
|
||||
assert!(!read_quorum, "GET must succeed while read quorum holds: {error:?}");
|
||||
let status = error
|
||||
.raw_response()
|
||||
.expect("GET should have an HTTP response")
|
||||
@@ -576,11 +580,12 @@ async fn assert_node_readiness_tracks_quorum(cluster: &mut RustFSTestClusterEnvi
|
||||
let list = client.list_objects_v2().bucket(BUCKET).send().await;
|
||||
let list_status = match list {
|
||||
Ok(result) => {
|
||||
assert!(read_quorum, "listing must fail once read quorum is lost");
|
||||
assert!(result.contents().iter().any(|object| object.key() == Some(seed_key)));
|
||||
200
|
||||
}
|
||||
Err(error) => {
|
||||
assert!(!write_ready, "listing must succeed on a ready cluster: {error:?}");
|
||||
assert!(!read_quorum, "listing must succeed while read quorum holds: {error:?}");
|
||||
let status = error
|
||||
.raw_response()
|
||||
.expect("LIST should have an HTTP response")
|
||||
|
||||
@@ -2430,6 +2430,99 @@ mod tests {
|
||||
restore_set_disks(&store, set, offline).await;
|
||||
}
|
||||
|
||||
/// EC 2+2 with two disks offline still has object read quorum and has lost
|
||||
/// write quorum. A node that has not cached the bucket must still serve
|
||||
/// HEAD bucket, GET, HEAD object, and List; a write must fail.
|
||||
#[tokio::test]
|
||||
#[serial]
|
||||
async fn cold_bucket_reads_succeed_below_write_quorum() {
|
||||
let (_temp_dir, store) = setup_bucket_quorum_test_env(&[4], Some(2)).await;
|
||||
metadata_sys::init_bucket_metadata_sys(store.clone(), Vec::new()).await;
|
||||
let bucket = format!("cold-read-quorum-{}", Uuid::new_v4().simple());
|
||||
let object = "seed-object";
|
||||
let body = b"cold bucket reads must survive lost write quorum".to_vec();
|
||||
store
|
||||
.make_bucket(&bucket, &MakeBucketOptions::default())
|
||||
.await
|
||||
.expect("healthy namespace should accept bucket creation");
|
||||
store
|
||||
.put_object(&bucket, object, &mut PutObjReader::from_vec(body.clone()), &ObjectOptions::default())
|
||||
.await
|
||||
.expect("healthy erasure set should accept the seed object");
|
||||
|
||||
metadata_sys::init_bucket_metadata_sys(store.clone(), Vec::new()).await;
|
||||
assert!(
|
||||
metadata_sys::get_in(&store.ctx, &bucket).await.is_err(),
|
||||
"the replacement metadata system must start without this bucket"
|
||||
);
|
||||
let set = &store.pools[0].disk_set[0];
|
||||
let lock = set
|
||||
.new_ns_lock(&bucket, object)
|
||||
.await
|
||||
.expect("seed namespace lock should resolve");
|
||||
drop(
|
||||
lock.get_write_lock(Duration::from_secs(30))
|
||||
.await
|
||||
.expect("seed physical fanout must finish before taking disks offline"),
|
||||
);
|
||||
let offline = take_set_disks_offline(&store, set, &[0, 1]).await;
|
||||
|
||||
let info = store
|
||||
.get_bucket_info(&bucket, &BucketOptions::default())
|
||||
.await
|
||||
.expect("HEAD bucket must succeed at read quorum before the bucket is cached");
|
||||
assert_eq!(info.name, bucket);
|
||||
|
||||
let mut reader = store
|
||||
.get_object_reader(&bucket, object, None, Default::default(), &ObjectOptions::default())
|
||||
.await
|
||||
.expect("GET must succeed for a bucket that was not loaded at startup");
|
||||
let mut restored = Vec::new();
|
||||
reader
|
||||
.stream
|
||||
.read_to_end(&mut restored)
|
||||
.await
|
||||
.expect("quorum read should reconstruct the body");
|
||||
assert_eq!(restored, body);
|
||||
drop(reader);
|
||||
|
||||
let object_info = store
|
||||
.get_object_info(&bucket, object, &ObjectOptions::default())
|
||||
.await
|
||||
.expect("HEAD object must succeed at read quorum");
|
||||
assert_eq!(object_info.name, object);
|
||||
assert_eq!(object_info.size, i64::try_from(body.len()).expect("body length fits i64"));
|
||||
|
||||
let listed = store
|
||||
.clone()
|
||||
.list_objects_v2(&bucket, "", None, None, 100, false, None, false)
|
||||
.await
|
||||
.expect("List must succeed at read quorum");
|
||||
assert!(
|
||||
listed.objects.iter().any(|item| item.name == object),
|
||||
"list should include the seed object, got {:?}",
|
||||
listed.objects.iter().map(|item| item.name.clone()).collect::<Vec<_>>()
|
||||
);
|
||||
assert!(
|
||||
metadata_sys::get_in(&store.ctx, &bucket).await.is_ok(),
|
||||
"the cold load should publish authoritative metadata"
|
||||
);
|
||||
|
||||
let write_error = store
|
||||
.put_object(&bucket, "rejected-object", &mut PutObjReader::from_vec(body), &ObjectOptions::default())
|
||||
.await
|
||||
.expect_err("writes must still fail without write quorum");
|
||||
assert!(
|
||||
matches!(
|
||||
write_error,
|
||||
StorageError::InsufficientWriteQuorum(_, _) | StorageError::ErasureWriteQuorum
|
||||
),
|
||||
"writes must fail with a write-quorum error, got {write_error}"
|
||||
);
|
||||
|
||||
restore_set_disks(&store, set, offline).await;
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
#[serial]
|
||||
async fn lazy_metadata_load_rejects_below_read_quorum() {
|
||||
|
||||
@@ -39,20 +39,21 @@ an unknown or unsupported peer-health snapshot degrades readiness with
|
||||
|
||||
- Liveness reports process availability and must not depend on storage, IAM,
|
||||
lock quorum, or peer health.
|
||||
- Node readiness reports the node's observed storage write quorum, local pool metadata write gate, IAM, and lock quorum. Runtime storage diagnostics do not change the existing startup `FullReady` publication gate or S3 request admission.
|
||||
- A blocked pool metadata writer degrades node and cluster-write readiness with
|
||||
`pool_meta_write_blocked`. Metadata save-gate inspection is bounded to 100 ms;
|
||||
contention reports `pool_metadata_check_timeout` without installing a block.
|
||||
Node readiness keeps the last confirmed writable observation for the normal
|
||||
readiness-cache TTL during a transient inspection timeout, while preserving a
|
||||
confirmed block and failing closed when no fresh observation exists.
|
||||
- Node readiness (`/health/ready`, `/minio/health/ready`) reports whether this process can serve reads: storage read quorum, IAM, and shared-lock quorum. It is the Kubernetes Service membership signal. Losing write quorum, the pool-metadata writer, or exclusive-lock quorum does not remove a node that can still read. The public health handler still returns 503 until startup publishes `FullReady`, so a process that has never opened S3 admission does not join the Service. Runtime storage diagnostics do not change that startup gate.
|
||||
- A blocked pool metadata writer degrades cluster-write readiness with
|
||||
`pool_meta_write_blocked`. It does not withdraw `/health/ready`, and the node
|
||||
probe omits that reason from `degradedReasons`. Metadata save-gate inspection
|
||||
is bounded to 100 ms; contention reports `pool_metadata_check_timeout` on the
|
||||
cluster-write probe without installing a block. The node probe keeps the last
|
||||
confirmed writable observation for `details.poolMetadata` during a transient
|
||||
inspection timeout, while preserving a confirmed block and failing that
|
||||
detail closed when no fresh observation exists.
|
||||
- The authenticated cluster snapshot extends its existing node-local metadata
|
||||
gate inspection with safe reason, failure phase, and original block time. It
|
||||
distinguishes timeout from a block and changes no admission or recovery
|
||||
decision. Runtime readiness and gate status are separate bounded observations.
|
||||
- Cluster write readiness requires write quorum and the runtime dependency
|
||||
readiness used by `FullReady`.
|
||||
- Cluster read readiness uses the storage read-quorum path and cluster-health timeout behavior. Its lock dependency still uses the per-set majority/write-lock health check. Actual shared namespace locks require `ceil(lock_clients / 2)`, so the cluster read probe is conservative: a four-client set can still admit some reads with two clients while the probe returns 503. This diagnostic change does not lower that probe's lock threshold.
|
||||
- Cluster write readiness (`/minio/health/cluster`) requires storage write quorum, the pool-metadata write gate, and exclusive-lock quorum. Use it when a caller must know that new writes can commit. It is not the Service membership probe.
|
||||
- Cluster read readiness uses the storage read-quorum path and cluster-health timeout behavior. Its lock dependency still uses the per-set majority/write-lock health check. Actual shared namespace locks require `n - n/2`, so the cluster read probe is conservative: a four-client set can still admit reads with two clients while that probe returns 503. Node `/health/ready` uses the shared-lock threshold instead, because that is what object reads acquire.
|
||||
- `HEAD` health probes keep header/status semantics and do not require response
|
||||
bodies.
|
||||
|
||||
@@ -62,22 +63,24 @@ The existing `details.storage.ready` boolean and `connected` / `disconnected` st
|
||||
|
||||
| Probe | `readinessScope` | `source` |
|
||||
| --- | --- | --- |
|
||||
| `/health/ready`, `/minio/health/ready` | `write_quorum_and_pool_metadata` | `local_runtime` |
|
||||
| `/health/ready`, `/minio/health/ready` | `read_quorum` | `local_runtime` |
|
||||
| `/minio/health/cluster` | `write_quorum_and_pool_metadata` | `storage_inventory` |
|
||||
| `/minio/health/cluster/read` | `read_quorum` | `storage_inventory` |
|
||||
|
||||
Node readiness additionally reports `details.storage.readQuorum`, `details.storage.writeQuorum`, and `details.poolMetadata.ready`. The metadata component's status is `writable` or `unavailable`; existing typed degradation reasons distinguish a write block from an inspection timeout. A healthy metadata writer alone no longer makes the storage component ready.
|
||||
Node readiness additionally reports `details.storage.readQuorum`, `details.storage.writeQuorum`, and `details.poolMetadata.ready`. `details.storage.ready` follows read quorum. `details.lock.ready` on this probe follows shared-lock quorum; on `/minio/health/cluster` it follows exclusive-lock quorum. The metadata component's status is `writable` or `unavailable`. A blocked metadata writer is visible there and on the cluster-write probe; it does not by itself make `/health/ready` return 503.
|
||||
|
||||
Node storage quorum uses configured drives per set, all configured pools/sets, and their Standard storage-class data/parity layout. Missing, duplicate, unreachable, or unhealthy disk observations cannot supply extra quorum votes. The read quorum is the data-drive count; the write quorum is that count plus one when data and parity counts are equal. These are observations of available storage slots, not guarantees that a particular object's metadata, shards, or required locks are available.
|
||||
|
||||
For a healthy IAM and metadata writer in a four-node, one-drive-per-node EC 2+2 set:
|
||||
For a healthy IAM and metadata writer in a four-node, one-drive-per-node EC 2+2 set, after startup has published `FullReady`:
|
||||
|
||||
| Surviving nodes | Storage read quorum | Storage write quorum | Pool metadata ready | Node HTTP / top-level ready |
|
||||
| --- | --- | --- | --- | --- |
|
||||
| 4 or 3 | true | true | true | 200 / true |
|
||||
| 2 | true | false | true | 503 / false |
|
||||
| 1 | false | false | true | 503 / false |
|
||||
| All restored | true | true | true | 200 / true |
|
||||
| Surviving nodes | Storage read quorum | Storage write quorum | Shared locks | Exclusive locks | Node `/health/ready` | `/minio/health/cluster` |
|
||||
| --- | --- | --- | --- | --- | --- | --- |
|
||||
| 4 or 3 | true | true | true | true | 200 | 200 |
|
||||
| 2 | true | false | true | false | 200 | 503 |
|
||||
| 1 | false | false | false | false | 503 | 503 |
|
||||
| All restored | true | true | true | true | 200 | 200 |
|
||||
|
||||
With two of four nodes up, GET, HEAD, and List of a bucket that this process has not yet cached still succeed. PUT and other mutations fail because write quorum and exclusive locks are gone.
|
||||
|
||||
The node path reads local disk-handle health and reuses the same reachable-host observation as its lock dependency, including the existing `RUSTFS_HEALTH_READINESS_CACHE_TTL_MS` cache. Only `Online` drives count; a reachable host with a `Returning` drive does not yet prove data I/O has recovered. It does not call cluster `storage_info`, local `disk_info`, or add disk-info RPCs. The entire storage inventory snapshot has a separate 100 ms wait budget; expiry reports `storage_readiness_check_timeout` and fails closed. Pool metadata inspection retains its own 100 ms budget; a timeout is counted by `rustfs_pool_metadata_check_timeouts_total` and uses the last confirmed node-local gate state only within the same cache TTL. These observations are not an atomic cluster snapshot and do not bypass the existing lock-probe timing or cache policy.
|
||||
|
||||
|
||||
@@ -56,8 +56,8 @@ When the whole cluster (or several nodes) went down and nodes are brought back o
|
||||
| `startup_finalization` | Last startup steps are being published. |
|
||||
|
||||
2. Logs say what the node waits for. The IAM recovery loop retries with backoff and logs `event="iam_bootstrap_retry_failed"` with an actionable `hint` field (for example, "storage read quorum not met yet; waiting for enough cluster nodes/disks to come online"). After repeated failures the level escalates from WARN to ERROR; this still does not kill the process.
|
||||
3. Recovery is automatic. Once storage read quorum is available, pending nodes can finish IAM bootstrap on the next retry. `/health/ready` returns `200` when storage write quorum, the metadata write gate, IAM, and lock readiness are satisfied. Restarting pending nodes does not speed this up.
|
||||
4. Check readiness detail while waiting. `/health/ready` (and `/minio/health/ready`) separate `storage` / `poolMetadata` / `iam` / `lock` readiness. `storage.ready` summarizes write quorum plus the metadata write gate; `storage.readQuorum` and `storage.writeQuorum` show the separate quorum observations. A healthy `poolMetadata.ready` does not imply storage quorum. `degradedReasons` lists machine-readable causes such as `storage_quorum_unavailable`, `storage_and_lock_unavailable`, or `pool_metadata_check_timeout`. See the [storage detail contract](../architecture/readiness-matrix.md#storage-detail-contract) for probe scopes and sampling limits:
|
||||
3. Recovery is automatic. Once storage read quorum is available, pending nodes can finish IAM bootstrap on the next retry. `/health/ready` returns `200` when startup has published `FullReady` and the node can still serve reads (storage read quorum, IAM, and shared-lock quorum). Write quorum is reported separately and is required by `/minio/health/cluster`, not by Service membership. Restarting pending nodes does not speed this up.
|
||||
4. Check readiness detail while waiting. `/health/ready` (and `/minio/health/ready`) separate `storage` / `poolMetadata` / `iam` / `lock` readiness. `storage.ready` follows read quorum (`readinessScope` is `read_quorum`); `storage.readQuorum` and `storage.writeQuorum` show the two observations. `poolMetadata.ready` is the metadata writer and does not by itself withdraw `/health/ready`. `degradedReasons` on this probe names read-path causes such as `storage_quorum_unavailable` and `storage_and_lock_unavailable`. `pool_meta_write_blocked` and `pool_metadata_check_timeout` stay on `/minio/health/cluster`. See the [storage detail contract](../architecture/readiness-matrix.md#storage-detail-contract) for probe scopes and sampling limits:
|
||||
|
||||
```bash
|
||||
curl -s http://<node>:9000/health/ready | jq
|
||||
|
||||
+12
-11
@@ -402,7 +402,9 @@ pub(crate) fn build_health_payload(ctx: HealthPayloadContext<'_>) -> Value {
|
||||
if ctx.include_dependency_details {
|
||||
payload["details"] = build_component_details(ctx.storage_ready, ctx.iam_ready, ctx.lock_quorum_ready, ctx.kms_ready);
|
||||
payload["details"]["storage"]["readinessScope"] = json!(match ctx.probe {
|
||||
HealthProbe::ClusterRead => "read_quorum",
|
||||
// Node readiness is the Service membership probe: it follows read
|
||||
// quorum. Cluster write keeps the write-quorum scope.
|
||||
HealthProbe::ClusterRead | HealthProbe::Readiness => "read_quorum",
|
||||
_ => "write_quorum_and_pool_metadata",
|
||||
});
|
||||
payload["details"]["storage"]["source"] = json!(match ctx.probe {
|
||||
@@ -461,13 +463,16 @@ mod tests {
|
||||
with_var(rustfs_config::ENV_HEALTH_MINIMAL_RESPONSE_ENABLE, Some("false"), || {
|
||||
for (read_quorum, write_quorum, metadata_ready, lock_ready) in [
|
||||
(true, true, true, true),
|
||||
(true, false, true, true),
|
||||
(true, false, false, true),
|
||||
(true, false, true, false),
|
||||
(false, false, true, false),
|
||||
(true, true, false, true),
|
||||
(true, true, true, true),
|
||||
] {
|
||||
let mut report = ready_report();
|
||||
report.readiness.storage_ready = write_quorum && metadata_ready;
|
||||
// The node collector stores read quorum here. Write quorum and the
|
||||
// pool-metadata writer remain detail fields and do not set HTTP status.
|
||||
report.readiness.storage_ready = read_quorum;
|
||||
report.readiness.lock_quorum_ready = lock_ready;
|
||||
report.storage_details = Some(crate::shared_types::StorageReadinessDetails {
|
||||
read_quorum_ready: read_quorum,
|
||||
@@ -475,7 +480,7 @@ mod tests {
|
||||
pool_metadata_write_ready: metadata_ready,
|
||||
});
|
||||
let parts = build_health_response_parts(Method::GET, HealthProbe::Readiness, Some(&report), "rustfs", None, None);
|
||||
let expected_ready = write_quorum && metadata_ready && lock_ready;
|
||||
let expected_ready = read_quorum && lock_ready;
|
||||
assert_eq!(
|
||||
parts.status_code,
|
||||
if expected_ready {
|
||||
@@ -486,20 +491,16 @@ mod tests {
|
||||
);
|
||||
let payload = parts.payload.expect("GET readiness body");
|
||||
assert_eq!(payload["ready"], expected_ready);
|
||||
assert_eq!(payload["details"]["storage"]["ready"], write_quorum && metadata_ready);
|
||||
assert_eq!(payload["details"]["storage"]["ready"], read_quorum);
|
||||
assert_eq!(
|
||||
payload["details"]["storage"]["status"],
|
||||
if write_quorum && metadata_ready {
|
||||
"connected"
|
||||
} else {
|
||||
"disconnected"
|
||||
}
|
||||
if read_quorum { "connected" } else { "disconnected" }
|
||||
);
|
||||
assert_eq!(payload["details"]["storage"]["readQuorum"], read_quorum);
|
||||
assert_eq!(payload["details"]["storage"]["writeQuorum"], write_quorum);
|
||||
assert_eq!(payload["details"]["poolMetadata"]["ready"], metadata_ready);
|
||||
assert_eq!(payload["details"]["storage"]["source"], "local_runtime");
|
||||
assert_eq!(payload["details"]["storage"]["readinessScope"], "write_quorum_and_pool_metadata");
|
||||
assert_eq!(payload["details"]["storage"]["readinessScope"], "read_quorum");
|
||||
assert!(
|
||||
build_health_response_parts(Method::HEAD, HealthProbe::Readiness, Some(&report), "rustfs", None, None)
|
||||
.payload
|
||||
|
||||
+150
-24
@@ -375,10 +375,15 @@ enum ClusterHealthProbeKind {
|
||||
|
||||
#[derive(Debug, Clone, Copy, Default, PartialEq, Eq)]
|
||||
pub struct LockQuorumStatus {
|
||||
/// Exclusive-lock quorum. Cluster write probes and startup publication use this.
|
||||
pub ready: bool,
|
||||
/// Shared-lock quorum. Node `/health/ready` uses this so a readable survivor
|
||||
/// stays in the Service after exclusive locks are no longer obtainable.
|
||||
pub read_ready: bool,
|
||||
pub connected_clients: usize,
|
||||
pub total_clients: usize,
|
||||
pub required_quorum: usize,
|
||||
pub required_read_quorum: usize,
|
||||
}
|
||||
|
||||
const DISK_STATE_OK: &str = "ok";
|
||||
@@ -635,6 +640,20 @@ fn pool_read_quorum(info: &StorageInfo, pool_idx: usize, set_drive_count: usize)
|
||||
pool_erasure_layout(info, pool_idx, set_drive_count).map(|(data_drives, _)| data_drives)
|
||||
}
|
||||
|
||||
/// Exclusive namespace locks use a strict majority of lock clients.
|
||||
fn exclusive_lock_quorum(total_clients: usize) -> usize {
|
||||
if total_clients > 1 { (total_clients / 2) + 1 } else { 1 }
|
||||
}
|
||||
|
||||
/// Shared namespace locks use the same client threshold as `DistributedLock::read_quorum`.
|
||||
fn shared_lock_quorum(total_clients: usize) -> usize {
|
||||
if total_clients > 1 {
|
||||
total_clients - (total_clients / 2)
|
||||
} else {
|
||||
1
|
||||
}
|
||||
}
|
||||
|
||||
fn configured_readiness_topology(info: &StorageInfo) -> Option<(&[usize], &[usize])> {
|
||||
if info.backend.total_sets.is_empty()
|
||||
|| info.backend.total_sets.len() != info.backend.drives_per_set.len()
|
||||
@@ -841,6 +860,27 @@ fn dependency_readiness_report_from_write_status(
|
||||
}
|
||||
}
|
||||
|
||||
/// `/health/ready` reports whether this process can serve reads. Write quorum,
|
||||
/// the pool-metadata writer, and exclusive locks stay on `/minio/health/cluster`.
|
||||
fn apply_node_service_readiness(
|
||||
mut storage: StorageWriteReadinessStatus,
|
||||
mut details: StorageReadinessDetails,
|
||||
lock_read_ready: bool,
|
||||
iam_ready: bool,
|
||||
peer_health_ready: bool,
|
||||
) -> (DependencyReadiness, StorageWriteReadinessStatus, StorageReadinessDetails) {
|
||||
details.pool_metadata_write_ready = storage.ready;
|
||||
storage.ready = details.read_quorum_ready;
|
||||
storage.pool_metadata_reason = None;
|
||||
let readiness = DependencyReadiness {
|
||||
storage_ready: storage.ready,
|
||||
iam_ready,
|
||||
lock_quorum_ready: lock_read_ready,
|
||||
peer_health_ready,
|
||||
};
|
||||
(readiness, storage, details)
|
||||
}
|
||||
|
||||
pub async fn collect_dependency_readiness_report() -> DependencyReadinessReport {
|
||||
let iam_ready_raw = runtime_sources::current_iam_ready();
|
||||
let storage = if let Some(cached) = load_cached_storage_readiness().await {
|
||||
@@ -879,7 +919,6 @@ pub async fn collect_node_readiness_report() -> DependencyReadinessReport {
|
||||
if let Some(store) = runtime_sources::current_object_store_handle() {
|
||||
let pool_metadata_status = store.pool_meta_write_status().await;
|
||||
storage = apply_node_pool_metadata_timeout_policy(pool_metadata_write_readiness(pool_metadata_status), Instant::now());
|
||||
details.pool_metadata_write_ready = storage.ready;
|
||||
match node_storage_snapshot(store.as_ref(), &lock_observation.online_hosts).await {
|
||||
Ok(info) => {
|
||||
details.read_quorum_ready = storage_read_ready_from_runtime_state(&info);
|
||||
@@ -889,13 +928,14 @@ pub async fn collect_node_readiness_report() -> DependencyReadinessReport {
|
||||
Err(_) => {}
|
||||
}
|
||||
}
|
||||
storage.ready &= details.write_quorum_ready;
|
||||
let readiness = DependencyReadiness {
|
||||
storage_ready: storage.ready,
|
||||
iam_ready: runtime_sources::current_iam_ready(),
|
||||
lock_quorum_ready: lock_observation.status.ready,
|
||||
peer_health_ready: collect_peer_health_readiness(),
|
||||
};
|
||||
// The HTTP layer still returns 503 until startup publishes `FullReady`.
|
||||
let (readiness, storage, details) = apply_node_service_readiness(
|
||||
storage,
|
||||
details,
|
||||
lock_observation.status.read_ready,
|
||||
runtime_sources::current_iam_ready(),
|
||||
collect_peer_health_readiness(),
|
||||
);
|
||||
let mut report = dependency_readiness_report_from_write_status(readiness, storage);
|
||||
report.storage_details = Some(details);
|
||||
if storage_check_timed_out {
|
||||
@@ -1101,17 +1141,16 @@ fn set_lock_quorum_status(online_hosts: &HashSet<String>, set_endpoints: &[Endpo
|
||||
}
|
||||
|
||||
let connected_clients = total_clients.iter().filter(|host| online_hosts.contains(*host)).count();
|
||||
let required_quorum = if total_clients_len > 1 {
|
||||
(total_clients_len / 2) + 1
|
||||
} else {
|
||||
1
|
||||
};
|
||||
let required_quorum = exclusive_lock_quorum(total_clients_len);
|
||||
let required_read_quorum = shared_lock_quorum(total_clients_len);
|
||||
|
||||
LockQuorumStatus {
|
||||
ready: connected_clients >= required_quorum,
|
||||
read_ready: connected_clients >= required_read_quorum,
|
||||
connected_clients,
|
||||
total_clients: total_clients_len,
|
||||
required_quorum,
|
||||
required_read_quorum,
|
||||
}
|
||||
}
|
||||
|
||||
@@ -1119,6 +1158,10 @@ fn aggregate_lock_quorum_status(pool_endpoints: &EndpointServerPools, online_hos
|
||||
let mut connected_clients = 0usize;
|
||||
let mut total_clients = 0usize;
|
||||
let mut required_quorum = 0usize;
|
||||
let mut required_read_quorum = 0usize;
|
||||
let mut write_ready = true;
|
||||
let mut read_ready = true;
|
||||
let mut saw_set = false;
|
||||
|
||||
for pool in pool_endpoints.as_ref() {
|
||||
for set_idx in 0..pool.set_count {
|
||||
@@ -1135,29 +1178,26 @@ fn aggregate_lock_quorum_status(pool_endpoints: &EndpointServerPools, online_hos
|
||||
return LockQuorumStatus::default();
|
||||
}
|
||||
|
||||
saw_set = true;
|
||||
connected_clients += status.connected_clients;
|
||||
total_clients += status.total_clients;
|
||||
required_quorum += status.required_quorum;
|
||||
|
||||
if !status.ready {
|
||||
return LockQuorumStatus {
|
||||
ready: false,
|
||||
connected_clients,
|
||||
total_clients,
|
||||
required_quorum,
|
||||
};
|
||||
}
|
||||
required_read_quorum += status.required_read_quorum;
|
||||
write_ready &= status.ready;
|
||||
read_ready &= status.read_ready;
|
||||
}
|
||||
}
|
||||
|
||||
if total_clients == 0 {
|
||||
if !saw_set {
|
||||
LockQuorumStatus::default()
|
||||
} else {
|
||||
LockQuorumStatus {
|
||||
ready: true,
|
||||
ready: write_ready,
|
||||
read_ready,
|
||||
connected_clients,
|
||||
total_clients,
|
||||
required_quorum,
|
||||
required_read_quorum,
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -1171,9 +1211,11 @@ async fn collect_lock_quorum_observation_uncached() -> LockQuorumObservation {
|
||||
return LockQuorumObservation {
|
||||
status: LockQuorumStatus {
|
||||
ready: true,
|
||||
read_ready: true,
|
||||
connected_clients: 1,
|
||||
total_clients: 1,
|
||||
required_quorum: 1,
|
||||
required_read_quorum: 1,
|
||||
},
|
||||
online_hosts: HashSet::new(),
|
||||
};
|
||||
@@ -2254,9 +2296,11 @@ mod tests {
|
||||
);
|
||||
|
||||
assert!(status.ready);
|
||||
assert!(status.read_ready);
|
||||
assert_eq!(status.connected_clients, 4);
|
||||
assert_eq!(status.total_clients, 4);
|
||||
assert_eq!(status.required_quorum, 4);
|
||||
assert_eq!(status.required_read_quorum, 2);
|
||||
}
|
||||
|
||||
#[test]
|
||||
@@ -2304,6 +2348,86 @@ mod tests {
|
||||
aggregate_lock_quorum_status(&pools, &["node1:9000"].into_iter().map(str::to_string).collect::<HashSet<_>>());
|
||||
|
||||
assert!(!status.ready);
|
||||
assert!(!status.read_ready);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn node_service_readiness_follows_read_quorum_not_write_or_pool_metadata() {
|
||||
let blocked = StorageWriteReadinessStatus {
|
||||
ready: false,
|
||||
pool_metadata_reason: Some(ReadinessDegradedReason::PoolMetaWriteBlocked),
|
||||
};
|
||||
let (readiness, storage, details) = apply_node_service_readiness(
|
||||
blocked,
|
||||
StorageReadinessDetails {
|
||||
read_quorum_ready: true,
|
||||
write_quorum_ready: false,
|
||||
pool_metadata_write_ready: true,
|
||||
},
|
||||
true,
|
||||
true,
|
||||
true,
|
||||
);
|
||||
assert!(readiness.storage_ready);
|
||||
assert!(readiness.lock_quorum_ready);
|
||||
assert!(!details.write_quorum_ready);
|
||||
assert!(!details.pool_metadata_write_ready);
|
||||
assert!(storage.pool_metadata_reason.is_none());
|
||||
let report = dependency_readiness_report_from_write_status(readiness, storage);
|
||||
assert!(
|
||||
report.degraded_reasons.is_empty(),
|
||||
"a readable node must stay ready when write quorum and the metadata writer are lost, got {:?}",
|
||||
report.degraded_reasons
|
||||
);
|
||||
|
||||
let (readiness, storage, _) = apply_node_service_readiness(
|
||||
StorageWriteReadinessStatus {
|
||||
ready: true,
|
||||
pool_metadata_reason: None,
|
||||
},
|
||||
StorageReadinessDetails {
|
||||
read_quorum_ready: false,
|
||||
write_quorum_ready: false,
|
||||
pool_metadata_write_ready: false,
|
||||
},
|
||||
false,
|
||||
true,
|
||||
true,
|
||||
);
|
||||
assert_eq!(
|
||||
dependency_readiness_report_from_write_status(readiness, storage).degraded_reasons,
|
||||
vec![ReadinessDegradedReason::StorageAndLockUnavailable]
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn shared_lock_quorum_stays_ready_when_two_of_four_clients_remain() {
|
||||
let endpoints = (0..4)
|
||||
.map(|disk_idx| Endpoint {
|
||||
url: url::Url::parse(&format!("http://node{disk_idx}:9000/data")).expect("valid test endpoint"),
|
||||
is_local: disk_idx == 0,
|
||||
pool_idx: 0,
|
||||
set_idx: 0,
|
||||
disk_idx,
|
||||
})
|
||||
.collect::<Vec<_>>();
|
||||
let online = ["node0:9000", "node1:9000"]
|
||||
.into_iter()
|
||||
.map(str::to_string)
|
||||
.collect::<HashSet<_>>();
|
||||
let status = set_lock_quorum_status(&online, &endpoints);
|
||||
|
||||
assert_eq!(status.total_clients, 4);
|
||||
assert_eq!(status.connected_clients, 2);
|
||||
assert_eq!(status.required_quorum, 3);
|
||||
assert_eq!(status.required_read_quorum, 2);
|
||||
assert!(!status.ready, "exclusive locks need three of four clients");
|
||||
assert!(status.read_ready, "shared locks remain available at two of four clients");
|
||||
|
||||
let one = ["node0:9000"].into_iter().map(str::to_string).collect::<HashSet<_>>();
|
||||
let below_read = set_lock_quorum_status(&one, &endpoints);
|
||||
assert!(!below_read.ready);
|
||||
assert!(!below_read.read_ready);
|
||||
}
|
||||
|
||||
#[test]
|
||||
@@ -2628,9 +2752,11 @@ mod tests {
|
||||
let observation = LockQuorumObservation {
|
||||
status: LockQuorumStatus {
|
||||
ready: true,
|
||||
read_ready: true,
|
||||
connected_clients: 2,
|
||||
total_clients: 3,
|
||||
required_quorum: 2,
|
||||
required_read_quorum: 2,
|
||||
},
|
||||
online_hosts: HashSet::from(["node-a:9000".to_owned(), "node-b:9000".to_owned()]),
|
||||
};
|
||||
|
||||
Reference in New Issue
Block a user