mirror of
https://github.com/rustfs/rustfs.git
synced 2026-09-06 12:09:12 +00:00
test(e2e): tighten movement refusal and cluster start fail-fast
Treat only product strings, 501, and opaque 500 as a refused move. Fail cluster start promptly when a node exits, and overlap GETs with peer kill. Drop the unused TwoPoolFourDrive live-start layout. Co-authored-by: RustFS <hello@rustfs.com>
This commit is contained in:
@@ -1,2 +1,2 @@
|
||||
sha256-linux=52a426e7fe4d41497bbe44db0dc450edbcb133dd966d7a41601cdae1e6788741
|
||||
sha256-darwin=52a426e7fe4d41497bbe44db0dc450edbcb133dd966d7a41601cdae1e6788741
|
||||
sha256-linux=47c690dc23add078813d44a78b6ab4a4347a5ecef9cf072bb71f08bd5ce1efe6
|
||||
sha256-darwin=47c690dc23add078813d44a78b6ab4a4347a5ecef9cf072bb71f08bd5ce1efe6
|
||||
|
||||
@@ -1770,11 +1770,6 @@ impl RustFSTestClusterEnvironment {
|
||||
self.node_capture_log_paths.push(None);
|
||||
self.volume_proxy_addresses.push(None);
|
||||
|
||||
if !self.extra_env.iter().any(|(key, _)| key == "RUSTFS_UNSAFE_BYPASS_DISK_CHECK") {
|
||||
self.extra_env
|
||||
.push(("RUSTFS_UNSAFE_BYPASS_DISK_CHECK".to_string(), "true".to_string()));
|
||||
}
|
||||
|
||||
Ok(new_idx)
|
||||
}
|
||||
|
||||
|
||||
@@ -17,7 +17,9 @@ use super::harness::{
|
||||
take_drive_offline, unique_bucket, wait_for_ready,
|
||||
};
|
||||
use crate::common::init_logging;
|
||||
use std::sync::Arc;
|
||||
use std::time::Duration;
|
||||
use tokio::sync::Barrier;
|
||||
|
||||
#[tokio::test]
|
||||
async fn kill_and_restart_node_preserves_objects() -> TestResult {
|
||||
@@ -80,18 +82,21 @@ async fn concurrent_gets_survive_peer_node_kill() -> TestResult {
|
||||
let body = vec![0x7Au8; 96 * 1024];
|
||||
put_object(&dist.client(0)?, &bucket, "steady.bin", body.clone()).await?;
|
||||
|
||||
dist.cluster.stop_node(3)?;
|
||||
|
||||
let live: Vec<_> = (0..3).map(|idx| dist.client(idx)).collect::<Result<Vec<_>, _>>()?;
|
||||
let start = Arc::new(Barrier::new(13));
|
||||
let mut handles = Vec::new();
|
||||
for idx in 0..12 {
|
||||
let client = live[idx % live.len()].clone();
|
||||
let bucket = bucket.clone();
|
||||
let body = body.clone();
|
||||
let start = start.clone();
|
||||
handles.push(tokio::spawn(async move {
|
||||
start.wait().await;
|
||||
retrying_get_equals(&client, &bucket, "steady.bin", &body, Duration::from_secs(20)).await
|
||||
}));
|
||||
}
|
||||
start.wait().await;
|
||||
dist.cluster.stop_node(3)?;
|
||||
for handle in handles {
|
||||
handle.await??;
|
||||
}
|
||||
|
||||
@@ -23,8 +23,8 @@
|
||||
//! tests. Live expand-then-restart currently hits `pool metadata recovery
|
||||
//! required` on localhost DistErasure. That is a production bootstrap-proof
|
||||
//! limitation this test lane does not change. Movement tests use 4×4 single
|
||||
//! pool and classify decommission/rebalance 4xx/5xx as a refused move while
|
||||
//! still asserting object bytes.
|
||||
//! pool and classify decommission/rebalance product refusals (and opaque
|
||||
//! 500 InternalError) as a refused move while still asserting object bytes.
|
||||
//!
|
||||
//! Genuine multi-node *striped* pools still need multi-host CI (backlog
|
||||
//! #1313 / #1314). Site replication uses two 4-node 1-drive clusters so the
|
||||
@@ -57,8 +57,6 @@ pub(crate) enum DistLayout {
|
||||
FourByFour,
|
||||
/// 4 nodes × 1 drive, one erasure pool (minimum 4-node 4-disk layout).
|
||||
FourNodeFourDisk,
|
||||
/// 2 single-node pools, 4 drives each (expansion seed).
|
||||
TwoPoolFourDrive,
|
||||
}
|
||||
|
||||
pub(crate) struct DistCluster {
|
||||
@@ -88,7 +86,6 @@ impl DistCluster {
|
||||
let topology = match layout {
|
||||
DistLayout::FourByFour => ClusterTopology::single_pool_multidrive(NODE_COUNT, DRIVES_PER_NODE),
|
||||
DistLayout::FourNodeFourDisk => ClusterTopology::single_pool(NODE_COUNT),
|
||||
DistLayout::TwoPoolFourDrive => ClusterTopology::per_node_pools(DRIVES_PER_NODE, vec![vec![0], vec![1]]),
|
||||
};
|
||||
let mut cluster = RustFSTestClusterEnvironment::with_topology(topology).await?;
|
||||
cluster.set_env("NO_PROXY", "127.0.0.1,localhost");
|
||||
@@ -516,9 +513,10 @@ pub(crate) fn is_pool_meta_write_fence(body: &str) -> bool {
|
||||
|| body.contains("live fleet capability proof")
|
||||
}
|
||||
|
||||
/// Product refusals that movement tests observe. Opaque 5xx stays in
|
||||
/// [`classify_data_movement_http`] because admin often wraps the fence as
|
||||
/// InternalError XML without the inner string. Auth failures are not refusals.
|
||||
/// Product refusals that movement tests observe. Opaque 500 InternalError stays
|
||||
/// in [`classify_data_movement_http`] because admin often wraps the fence as
|
||||
/// InternalError XML without the inner string. 502/503 and auth failures are
|
||||
/// not refusals.
|
||||
pub(crate) fn is_known_data_movement_refusal(body: &str) -> bool {
|
||||
is_pool_meta_write_fence(body)
|
||||
|| body.contains("NotImplemented")
|
||||
@@ -536,7 +534,7 @@ pub(crate) fn classify_data_movement_http(status: StatusCode, body: &str) -> Res
|
||||
if status.is_success() {
|
||||
return Ok(DataMovementStart::Started);
|
||||
}
|
||||
if is_known_data_movement_refusal(body) || status.as_u16() == 501 || status.is_server_error() {
|
||||
if is_known_data_movement_refusal(body) || status.as_u16() == 501 || status == StatusCode::INTERNAL_SERVER_ERROR {
|
||||
return Ok(DataMovementStart::Refused(format!("{status} {body}")));
|
||||
}
|
||||
Err(format!("{status} {body}"))
|
||||
@@ -552,7 +550,7 @@ pub(crate) async fn try_start_decommission(
|
||||
}
|
||||
|
||||
/// Returns whether decommission actually started. A product refusal or opaque
|
||||
/// 5xx is not a test failure: callers still assert object bytes.
|
||||
/// 500 InternalError is not a test failure: callers still assert object bytes.
|
||||
pub(crate) async fn decommission_started_or_refused(cluster: &RustFSTestClusterEnvironment, pool_id: usize) -> TestResult<bool> {
|
||||
match try_start_decommission(cluster, pool_id).await? {
|
||||
DataMovementStart::Started => Ok(true),
|
||||
@@ -778,17 +776,18 @@ pub(crate) async fn retrying_get_equals(
|
||||
|
||||
#[tokio::test]
|
||||
async fn append_single_node_pool_extends_ellipses_volumes() {
|
||||
let mut dist = DistCluster::new_stopped(DistLayout::TwoPoolFourDrive)
|
||||
.await
|
||||
.expect("two-pool seed topology");
|
||||
assert_eq!(dist.cluster.rustfs_volumes_arg().split(' ').count(), 2);
|
||||
let mut env =
|
||||
RustFSTestClusterEnvironment::with_topology(ClusterTopology::per_node_pools(DRIVES_PER_NODE, vec![vec![0], vec![1]]))
|
||||
.await
|
||||
.expect("two-pool seed topology");
|
||||
assert_eq!(env.rustfs_volumes_arg().split(' ').count(), 2);
|
||||
|
||||
let added = dist.cluster.append_single_node_pool().await.expect("append third pool");
|
||||
let added = env.append_single_node_pool().await.expect("append third pool");
|
||||
assert_eq!(added, 2);
|
||||
assert_eq!(dist.cluster.nodes.len(), 3);
|
||||
assert_eq!(dist.cluster.nodes[2].pool_idx, 2);
|
||||
assert_eq!(dist.cluster.nodes[2].data_dirs.len(), DRIVES_PER_NODE);
|
||||
let volumes = dist.cluster.rustfs_volumes_arg();
|
||||
assert_eq!(env.nodes.len(), 3);
|
||||
assert_eq!(env.nodes[2].pool_idx, 2);
|
||||
assert_eq!(env.nodes[2].data_dirs.len(), DRIVES_PER_NODE);
|
||||
let volumes = env.rustfs_volumes_arg();
|
||||
assert_eq!(volumes.split(' ').count(), 3, "expected three pool arguments, got: {volumes}");
|
||||
assert!(volumes.contains("/drive{0...3}"), "expanded layout must keep drive ellipses: {volumes}");
|
||||
}
|
||||
@@ -807,6 +806,35 @@ async fn append_single_node_pool_rejects_striped_single_pool() {
|
||||
);
|
||||
}
|
||||
|
||||
#[cfg(unix)]
|
||||
#[tokio::test]
|
||||
async fn cluster_start_fails_fast_when_node_process_exits() {
|
||||
let mut dist = DistCluster::new_stopped(DistLayout::FourNodeFourDisk)
|
||||
.await
|
||||
.expect("stopped 4-node cluster");
|
||||
let script = format!("{}/immediate-exit.sh", dist.cluster.temp_dir);
|
||||
std::fs::write(&script, "#!/bin/sh\nexit 1\n").expect("write exit stub");
|
||||
let mut perms = std::fs::metadata(&script).expect("stat exit stub").permissions();
|
||||
std::os::unix::fs::PermissionsExt::set_mode(&mut perms, 0o755);
|
||||
std::fs::set_permissions(&script, perms).expect("chmod exit stub");
|
||||
|
||||
let started = Instant::now();
|
||||
let err = dist
|
||||
.start_from_binary(Path::new(&script))
|
||||
.await
|
||||
.expect_err("a node that exits immediately must fail start");
|
||||
let elapsed = started.elapsed();
|
||||
let message = err.to_string();
|
||||
assert!(
|
||||
message.contains("exited before TCP ready") || message.contains("exited before S3 ready"),
|
||||
"unexpected start error: {message}"
|
||||
);
|
||||
assert!(
|
||||
elapsed < Duration::from_secs(30),
|
||||
"cluster start must fail fast when a node exits, took {elapsed:?}"
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn decommission_complete_reads_pool_status_and_info_flag() {
|
||||
let status = serde_json::json!({
|
||||
@@ -896,4 +924,10 @@ fn classify_data_movement_http_observes_product_refusals_not_auth_failures() {
|
||||
));
|
||||
let denied = classify_data_movement_http(StatusCode::FORBIDDEN, "AccessDenied").expect_err("auth failure is not a refusal");
|
||||
assert!(denied.contains("AccessDenied"), "{denied}");
|
||||
let unavailable = classify_data_movement_http(StatusCode::SERVICE_UNAVAILABLE, "ServiceUnavailable")
|
||||
.expect_err("503 is not a product refusal");
|
||||
assert!(unavailable.contains("ServiceUnavailable"), "{unavailable}");
|
||||
let bad_gateway =
|
||||
classify_data_movement_http(StatusCode::BAD_GATEWAY, "Bad Gateway").expect_err("502 is not a product refusal");
|
||||
assert!(bad_gateway.contains("502") || bad_gateway.contains("Bad Gateway"), "{bad_gateway}");
|
||||
}
|
||||
|
||||
@@ -15,7 +15,7 @@ The in-tree harness runs every node on `127.0.0.1` with a distinct port. That ma
|
||||
|
||||
A pool striped across several localhost ports is not expressible (`RUSTFS_VOLUMES` host ellipses would collide on disk paths). Multi-host striped pools remain the hardware functional-chain / backlog #1313 / #1314 lane.
|
||||
|
||||
Decommission and rebalance POST on the 4×4 single-pool layout is refused by the current product (`single pool deployments do not support decommission`, NotImplemented, or an opaque admin 5xx when the inner pool-meta fence is wrapped as InternalError). Those cases still assert object bytes and SHA-256; when the API starts they wait for completion and assert post-move integrity. They do not treat a refusal as a successful move. This lane does not change production pool-meta bootstrap, write-fence, or decommission policy; it only observes the current server behavior.
|
||||
Decommission and rebalance POST on the 4×4 single-pool layout is refused by the current product (`single pool deployments do not support decommission`, NotImplemented, or opaque 500 InternalError when the inner pool-meta fence is wrapped). 502/503 are not treated as a product refusal. Those cases still assert object bytes and SHA-256; when the API starts they wait for completion and assert post-move integrity. They do not treat a refusal as a successful move. This lane does not change production pool-meta bootstrap, write-fence, or decommission policy; it only observes the current server behavior.
|
||||
|
||||
## What this lane covers
|
||||
|
||||
|
||||
Reference in New Issue
Block a user