mirror of
https://github.com/rustfs/rustfs.git
synced 2026-10-04 04:21:35 +00:00
fix(ci): repair pool budgets and Connect test failures (#8155)
* ci: give the pool run a real wall-clock budget The pool suite budgets up to 24h each for rebalance and decommission (REBALANCE_TIMEOUT / DECOMMISSION_TIMEOUT in rustfs_pool_expand.sh), but the workflow step capped it at 45 minutes. Four consecutive runs died identically: steps 1-7 all PASS, then the rebalance wait was killed at exactly 45:12 - chain 35680308150, standalone 35702803577, chain 35757773373, chain 36283120750 - with rebalance at completed=1/4 (~21 minutes in), so a full pass has never been observed. Make the budget an input (default 240 minutes: covers the observed rebalance pace plus one decommission pass) and document why. The suite keeps failing the job through its [POOL-STEP] marker adjudication. * fix(ci): align pool job budget and timeout contract tests * fix(connect): preserve legacy heartbeats and isolate I/O tests --------- Signed-off-by: Hauser <housemecn@gmail.com> Co-authored-by: overtrue <anzhengchao@gmail.com> Co-authored-by: Hauser <housemecn@gmail.com> Co-authored-by: RustFS <hello@rustfs.com>
This commit is contained in:
@@ -132,10 +132,11 @@ test-group = 'ecstore-serial-flaky'
|
||||
filter = 'package(rustfs) & (binary(/^embedded.*_test$/) | binary(admin_diagnostic_capability_e2e))'
|
||||
test-group = 'embedded-test-ports'
|
||||
|
||||
# The real object probe includes bucket creation and cleanup in its short
|
||||
# measurement budget. Keep competing storage fixtures outside that budget.
|
||||
# Inventory delivery and real drive/object probes include durable filesystem IO
|
||||
# in short deadlines. Reserve capacity so unrelated storage fixtures cannot
|
||||
# exhaust those budgets. Keep the deadlines and assertions unchanged.
|
||||
[[profile.default.overrides]]
|
||||
filter = 'package(rustfs) & binary(connect_perf_object) & test(=real_rustfs_endpoint_and_production_cli_support_bounded_get_and_put)'
|
||||
filter = 'package(rustfs) & (binary(connect_inventory) | binary(connect_perf_drive) | (binary(connect_perf_object) & test(=real_rustfs_endpoint_and_production_cli_support_bounded_get_and_put)))'
|
||||
threads-required = "num-test-threads"
|
||||
|
||||
# Serialize the durable manual-transition checkpoint test across nextest's
|
||||
@@ -334,7 +335,7 @@ filter = 'package(rustfs) & (binary(/^embedded.*_test$/) | binary(admin_diagnost
|
||||
test-group = 'embedded-test-ports'
|
||||
|
||||
[[profile.ci.overrides]]
|
||||
filter = 'package(rustfs) & binary(connect_perf_object) & test(=real_rustfs_endpoint_and_production_cli_support_bounded_get_and_put)'
|
||||
filter = 'package(rustfs) & (binary(connect_inventory) | binary(connect_perf_drive) | (binary(connect_perf_object) & test(=real_rustfs_endpoint_and_production_cli_support_bounded_get_and_put)))'
|
||||
threads-required = "num-test-threads"
|
||||
|
||||
# Serialize the durable manual-transition checkpoint test under the ci profile
|
||||
|
||||
@@ -7,6 +7,11 @@ on:
|
||||
description: Verified candidate and chain attempt from the chain driver
|
||||
type: string
|
||||
required: true
|
||||
pool_timeout_minutes:
|
||||
description: Wall-clock budget (minutes) for the pool run step
|
||||
type: string
|
||||
required: false
|
||||
default: '240'
|
||||
workflow_dispatch:
|
||||
inputs:
|
||||
rustfs_version:
|
||||
@@ -39,6 +44,10 @@ on:
|
||||
description: 'Run the pool decommission step (3-pool topology only)'
|
||||
type: boolean
|
||||
default: true
|
||||
pool_timeout_minutes:
|
||||
description: 'Wall-clock budget (minutes) for the whole pool run: fill + rebalance wait + decommission'
|
||||
required: false
|
||||
default: '240'
|
||||
cleanup_before:
|
||||
description: 'Reset the nodes before the test (DESTROYS existing data/config)'
|
||||
type: boolean
|
||||
@@ -82,7 +91,8 @@ jobs:
|
||||
pool-expansion-test:
|
||||
name: Pool expansion / decommission test
|
||||
runs-on: smoke-testing
|
||||
timeout-minutes: 60
|
||||
# Leave setup and finalizer headroom beyond the default 240-minute test.
|
||||
timeout-minutes: 360
|
||||
if: ${{ inputs.chain_manifest != '' || github.event_name == 'workflow_dispatch' || github.event_name == 'repository_dispatch' }}
|
||||
env:
|
||||
RUSTFS_POOL_ADMIN_ENDPOINT: ${{ secrets.RUSTFS_POOL_ADMIN_ENDPOINT || vars.RUSTFS_POOL_ADMIN_ENDPOINT || 'http://rustfs-node1:9000' }}
|
||||
@@ -251,7 +261,12 @@ jobs:
|
||||
|
||||
- name: Run pool expansion & decommission test
|
||||
id: pool_test
|
||||
timeout-minutes: 45
|
||||
# The suite budgets up to 24h each for rebalance and decommission
|
||||
# (REBALANCE_TIMEOUT / DECOMMISSION_TIMEOUT in rustfs_pool_expand.sh);
|
||||
# the previous 45m step cap killed four consecutive runs mid-rebalance
|
||||
# with steps 1-7 green. 240m covers the observed pace (rebalance 1/4
|
||||
# in ~21m) plus one decommission pass; override per dispatch.
|
||||
timeout-minutes: ${{ inputs.pool_timeout_minutes || 240 }}
|
||||
# Case failures keep the run green: the report and the backlog
|
||||
# issue manager carry the product signal.
|
||||
continue-on-error: true
|
||||
|
||||
@@ -22,13 +22,13 @@ The final job checks all twelve expected suite results and all twelve proof file
|
||||
|
||||
## Stalled suites and time limits
|
||||
|
||||
The nine non-performance reusable suite jobs, and the standalone table suite, have a **60-minute job limit**. The primary test step has a **45-minute limit** so a stalled test can fail before the hard job cancellation. Cleanup steps are limited to five minutes; report generation, dashboard/backlog operations and ordinary artifact uploads are limited to two minutes each. Separate installer and preflight steps are limited to five minutes. Performance is explicitly excluded from these limits.
|
||||
Non-performance suites other than pool expansion have a **60-minute job limit** and a **45-minute primary test limit** so a stalled test can fail before the hard job cancellation. Pool expansion has a **360-minute job limit** and a **240-minute default test limit** to allow rebalance and decommission to finish. Its `pool_timeout_minutes` input overrides the test limit for manual and reusable calls; repository dispatch uses the default. Overrides must leave room within the fixed job limit for setup and finalizers. Cleanup steps are limited to five minutes; report generation, dashboard/backlog operations and ordinary artifact uploads are limited to two minutes each. Separate installer and preflight steps are limited to five minutes. Performance is explicitly excluded from these limits.
|
||||
|
||||
The limits use GitHub Actions native timeouts. They are wall-clock limits, not log-idle detection: a process printing progress forever is still stopped. A step timeout is not a passing result. Existing `always()` finalizers attempt reporting and cleanup, and the chain driver's `always()` plus successful-prepare condition allows the next suite after a failed or cancelled suite. Missing or partial evidence still fails the complete-success gate; it cannot authorize closing historical issues.
|
||||
|
||||
The 15-minute difference between test and job limits is **headroom, not a reserved cleanup window**: checkout and setup also consume the 60-minute job budget, and several failing finalizers may exhaust it. Reaching the hard limit, losing a runner, or terminating an SSH connection does not guarantee remote processes have stopped or cleanup has completed. The next suite must retain its pre-test cleanup. Inspect the runner and remote VMs after a hard timeout before trusting subsequent results; do not interpret contaminated-environment failures as independent product regressions.
|
||||
The difference between test and job limits (15 minutes for ordinary suites, 120 minutes for the default pool run) is **headroom, not a reserved cleanup window**: checkout and setup also consume the job budget, and several failing finalizers may exhaust it. Reaching the hard limit, losing a runner, or terminating an SSH connection does not guarantee remote processes have stopped or cleanup has completed. The next suite must retain its pre-test cleanup. Inspect the runner and remote VMs after a hard timeout before trusting subsequent results; do not interpret contaminated-environment failures as independent product regressions.
|
||||
|
||||
The performance workflow remains unchanged: its job limit is 900 minutes, the default duration is five minutes per round, and the external script retains its default sixty-second pauses. Its methods, sizes, manual overrides and step limits are not modified by the functional timeout policy. A long performance run can therefore still occupy the final lane and delay chain completion; it is not covered by the one-hour guarantee for non-performance suites.
|
||||
The performance workflow remains unchanged: its job limit is 900 minutes, the default duration is five minutes per round, and the external script retains its default sixty-second pauses. Its methods, sizes, manual overrides and step limits are not modified by the functional timeout policy. A long performance run can therefore still occupy the final lane and delay chain completion; like pool expansion, it is not covered by the one-hour limit for ordinary functional suites.
|
||||
|
||||
Job timeouts start when execution starts; they do **not** bound runner or concurrency queue time. Runner preflight checks detect an already-offline runner but are not reservations. The reusable chain continues after a job timeout without cancelling the whole Actions run. The legacy `repository_dispatch` path relies on an in-job handoff and cannot guarantee continuation after hard cancellation; use the reusable driver for bounded nightly chains. A queue watchdog would need an external dispatcher and environment recovery, not cancellation of the whole parent run (which would also cancel the remaining suites).
|
||||
|
||||
|
||||
@@ -112,13 +112,15 @@ impl PendingHeartbeat {
|
||||
]
|
||||
|| self.capabilities == heartbeat_capabilities(false)
|
||||
|| self.capabilities == heartbeat_capabilities(true)
|
||||
// RUSTFS_COMPAT_TODO(connect-894) Remove after upgrades from the pre-service-memory capability set are unsupported.
|
||||
// Preserve exact pending requests; never add the capability to a retry.
|
||||
// RUSTFS_COMPAT_TODO(connect-894) Remove after upgrades from the pre-health and pre-service-memory sets are unsupported.
|
||||
// Compare frozen historical advertisements, not subsets of today's capabilities.
|
||||
|| [false, true].into_iter().any(|job_capable| {
|
||||
heartbeat_capabilities(job_capable)
|
||||
.iter()
|
||||
.filter(|capability| capability.as_str() != "profile.memory.service@1")
|
||||
.eq(self.capabilities.iter())
|
||||
let legacy = pre_health_heartbeat_capabilities(job_capable);
|
||||
self.capabilities == legacy
|
||||
|| legacy
|
||||
.iter()
|
||||
.filter(|capability| capability.as_str() != "profile.memory.service@1")
|
||||
.eq(self.capabilities.iter())
|
||||
}))
|
||||
&& self.sequence <= MAX_SEQUENCE
|
||||
&& self.coarse_node_summary.is_valid()
|
||||
@@ -331,6 +333,40 @@ impl HeartbeatStateStore {
|
||||
}
|
||||
}
|
||||
|
||||
fn pre_health_heartbeat_capabilities(job_capable: bool) -> Vec<String> {
|
||||
let mut capabilities = [
|
||||
"heartbeat",
|
||||
"diagnostics.policy.v1",
|
||||
"inventory.environment@1",
|
||||
"performance.client@1",
|
||||
"performance.drive@1",
|
||||
"performance.network@1",
|
||||
"performance.object@1",
|
||||
"performance.siteReplication@1",
|
||||
"logs.capture@1",
|
||||
"profile.cpu@1",
|
||||
"profile.memory@1",
|
||||
"profile.memory.service@1",
|
||||
"profile.threads@1",
|
||||
"telemetry.record@1",
|
||||
"telemetry.otlp@1",
|
||||
"telemetry.replay@1",
|
||||
"top.api@1",
|
||||
"top.disk@1",
|
||||
"top.locks@1",
|
||||
"top.net@1",
|
||||
"top.rpc@1",
|
||||
"inspect.object@1",
|
||||
]
|
||||
.into_iter()
|
||||
.map(str::to_owned)
|
||||
.collect::<Vec<_>>();
|
||||
if job_capable {
|
||||
capabilities.insert(3, "jobs".to_owned());
|
||||
}
|
||||
capabilities
|
||||
}
|
||||
|
||||
fn heartbeat_capabilities(job_capable: bool) -> Vec<String> {
|
||||
let mut capabilities = vec![
|
||||
"heartbeat".to_owned(),
|
||||
|
||||
@@ -644,7 +644,7 @@ async fn restart_replays_pending_request_then_advances_sequence() {
|
||||
assert_eq!(seen[1]["sequence"].as_u64(), seen[0]["sequence"].as_u64().map(|value| value + 1));
|
||||
}
|
||||
|
||||
fn pre_service_memory_capabilities(job_capable: bool) -> Vec<&'static str> {
|
||||
fn legacy_capabilities(job_capable: bool, service_memory: bool) -> Vec<&'static str> {
|
||||
let mut capabilities = vec!["heartbeat", "diagnostics.policy.v1", "inventory.environment@1"];
|
||||
if job_capable {
|
||||
capabilities.push("jobs");
|
||||
@@ -671,6 +671,13 @@ fn pre_service_memory_capabilities(job_capable: bool) -> Vec<&'static str> {
|
||||
"top.rpc@1",
|
||||
"inspect.object@1",
|
||||
]);
|
||||
if service_memory {
|
||||
let memory = capabilities
|
||||
.iter()
|
||||
.position(|capability| *capability == "profile.memory@1")
|
||||
.unwrap();
|
||||
capabilities.insert(memory + 1, "profile.memory.service@1");
|
||||
}
|
||||
capabilities
|
||||
}
|
||||
|
||||
@@ -687,15 +694,15 @@ fn pending_heartbeat_with_capabilities(capabilities: &[&str]) -> Value {
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
async fn restart_replays_exact_pre_service_memory_capabilities_with_and_without_jobs() {
|
||||
for job_capable in [false, true] {
|
||||
async fn restart_replays_exact_legacy_capabilities_with_and_without_jobs() {
|
||||
for (job_capable, service_memory) in [(false, false), (true, false), (false, true), (true, true)] {
|
||||
let pki = TestPki::new();
|
||||
let server = server(&pki, vec![Reply::ok("2026-08-22T01:02:03Z")]).await;
|
||||
let temp = tempfile::tempdir().expect("tempdir");
|
||||
let shutdown = CancellationToken::new();
|
||||
let config = config(&temp, &pki, &server);
|
||||
fs::create_dir_all(config.state_path.parent().expect("state directory")).expect("create state directory");
|
||||
let pending = pending_heartbeat_with_capabilities(&pre_service_memory_capabilities(job_capable));
|
||||
let pending = pending_heartbeat_with_capabilities(&legacy_capabilities(job_capable, service_memory));
|
||||
let state = json!({"nextSequence": 0, "pending": pending});
|
||||
fs::write(&config.state_path, serde_json::to_vec(&state).expect("heartbeat state JSON")).expect("write heartbeat state");
|
||||
private_mode(&config.state_path);
|
||||
@@ -715,8 +722,8 @@ async fn restart_replays_exact_pre_service_memory_capabilities_with_and_without_
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
async fn pre_service_memory_compatibility_does_not_accept_changed_capabilities() {
|
||||
for job_capable in [false, true] {
|
||||
async fn legacy_compatibility_does_not_accept_changed_capabilities() {
|
||||
for (job_capable, service_memory) in [(false, false), (true, false), (false, true), (true, true)] {
|
||||
for mutation in ["unknown", "missing", "duplicate", "reordered"] {
|
||||
let pki = TestPki::new();
|
||||
let server = server(&pki, vec![]).await;
|
||||
@@ -724,7 +731,7 @@ async fn pre_service_memory_compatibility_does_not_accept_changed_capabilities()
|
||||
let shutdown = CancellationToken::new();
|
||||
let config = config(&temp, &pki, &server);
|
||||
fs::create_dir_all(config.state_path.parent().expect("state directory")).expect("create state directory");
|
||||
let mut capabilities = pre_service_memory_capabilities(job_capable);
|
||||
let mut capabilities = legacy_capabilities(job_capable, service_memory);
|
||||
match mutation {
|
||||
"unknown" => capabilities.push("shell.exec@1"),
|
||||
"missing" => {
|
||||
@@ -747,7 +754,7 @@ async fn pre_service_memory_compatibility_does_not_accept_changed_capabilities()
|
||||
wait_for(&mut status, |status| matches!(status, HeartbeatStatus::Failed { .. })).await,
|
||||
HeartbeatStatus::Failed { reason } if reason.contains("violates the protocol invariants")
|
||||
),
|
||||
"accepted altered persisted capabilities: {mutation}, jobs={job_capable}"
|
||||
"accepted altered persisted capabilities: {mutation}, jobs={job_capable}, service_memory={service_memory}"
|
||||
);
|
||||
assert!(server.seen.lock().expect("seen lock").is_empty());
|
||||
runtime.shutdown().await;
|
||||
|
||||
@@ -403,12 +403,22 @@ class WorkflowTimeoutTests(unittest.TestCase):
|
||||
with self.subTest(suite=suite):
|
||||
source = (candidate.ROOT / f".github/workflows/rustfs-{suite}-test.yml").read_text()
|
||||
job = yaml_block(source.splitlines(), job_id, 2)
|
||||
self.assertIn(" timeout-minutes: 60", job)
|
||||
job_timeout = 360 if suite == "pool-expand" else 60
|
||||
self.assertIn(f" timeout-minutes: {job_timeout}", job)
|
||||
steps = named_steps(job)
|
||||
primary = [step for step in steps.values() if any(
|
||||
line in (" id: test", " id: pool_test") for line in step)]
|
||||
self.assertEqual(len(primary), 1)
|
||||
self.assertIn(" timeout-minutes: 45", primary[0])
|
||||
if suite == "pool-expand":
|
||||
self.assertIn(" timeout-minutes: ${{ inputs.pool_timeout_minutes || 240 }}", primary[0])
|
||||
for event in ("workflow_call", "workflow_dispatch"):
|
||||
event_block = yaml_block(source.splitlines(), event, 2)
|
||||
timeout_input = yaml_block(event_block, "pool_timeout_minutes", 6)
|
||||
self.assertIsNotNone(timeout_input, event)
|
||||
self.assertIn(" default: '240'", timeout_input)
|
||||
self.assertIn(" required: false", timeout_input)
|
||||
else:
|
||||
self.assertIn(" timeout-minutes: 45", primary[0])
|
||||
cleanup = "Cleanup environment"
|
||||
for phase in ("before", "after"):
|
||||
self.assertIn(" timeout-minutes: 5", steps[f"{cleanup} ({phase})"])
|
||||
|
||||
Reference in New Issue
Block a user