diff --git a/.config/nextest.toml b/.config/nextest.toml index 5e5b5cd33..9e7351fa3 100644 --- a/.config/nextest.toml +++ b/.config/nextest.toml @@ -132,10 +132,11 @@ test-group = 'ecstore-serial-flaky' filter = 'package(rustfs) & (binary(/^embedded.*_test$/) | binary(admin_diagnostic_capability_e2e))' test-group = 'embedded-test-ports' -# The real object probe includes bucket creation and cleanup in its short -# measurement budget. Keep competing storage fixtures outside that budget. +# Inventory delivery and real drive/object probes include durable filesystem IO +# in short deadlines. Reserve capacity so unrelated storage fixtures cannot +# exhaust those budgets. Keep the deadlines and assertions unchanged. [[profile.default.overrides]] -filter = 'package(rustfs) & binary(connect_perf_object) & test(=real_rustfs_endpoint_and_production_cli_support_bounded_get_and_put)' +filter = 'package(rustfs) & (binary(connect_inventory) | binary(connect_perf_drive) | (binary(connect_perf_object) & test(=real_rustfs_endpoint_and_production_cli_support_bounded_get_and_put)))' threads-required = "num-test-threads" # Serialize the durable manual-transition checkpoint test across nextest's @@ -334,7 +335,7 @@ filter = 'package(rustfs) & (binary(/^embedded.*_test$/) | binary(admin_diagnost test-group = 'embedded-test-ports' [[profile.ci.overrides]] -filter = 'package(rustfs) & binary(connect_perf_object) & test(=real_rustfs_endpoint_and_production_cli_support_bounded_get_and_put)' +filter = 'package(rustfs) & (binary(connect_inventory) | binary(connect_perf_drive) | (binary(connect_perf_object) & test(=real_rustfs_endpoint_and_production_cli_support_bounded_get_and_put)))' threads-required = "num-test-threads" # Serialize the durable manual-transition checkpoint test under the ci profile diff --git a/.github/workflows/rustfs-pool-expand-test.yml b/.github/workflows/rustfs-pool-expand-test.yml index eff8b5330..ee2a31d06 100644 --- a/.github/workflows/rustfs-pool-expand-test.yml +++ b/.github/workflows/rustfs-pool-expand-test.yml @@ -7,6 +7,11 @@ on: description: Verified candidate and chain attempt from the chain driver type: string required: true + pool_timeout_minutes: + description: Wall-clock budget (minutes) for the pool run step + type: string + required: false + default: '240' workflow_dispatch: inputs: rustfs_version: @@ -39,6 +44,10 @@ on: description: 'Run the pool decommission step (3-pool topology only)' type: boolean default: true + pool_timeout_minutes: + description: 'Wall-clock budget (minutes) for the whole pool run: fill + rebalance wait + decommission' + required: false + default: '240' cleanup_before: description: 'Reset the nodes before the test (DESTROYS existing data/config)' type: boolean @@ -82,7 +91,8 @@ jobs: pool-expansion-test: name: Pool expansion / decommission test runs-on: smoke-testing - timeout-minutes: 60 + # Leave setup and finalizer headroom beyond the default 240-minute test. + timeout-minutes: 360 if: ${{ inputs.chain_manifest != '' || github.event_name == 'workflow_dispatch' || github.event_name == 'repository_dispatch' }} env: RUSTFS_POOL_ADMIN_ENDPOINT: ${{ secrets.RUSTFS_POOL_ADMIN_ENDPOINT || vars.RUSTFS_POOL_ADMIN_ENDPOINT || 'http://rustfs-node1:9000' }} @@ -251,7 +261,12 @@ jobs: - name: Run pool expansion & decommission test id: pool_test - timeout-minutes: 45 + # The suite budgets up to 24h each for rebalance and decommission + # (REBALANCE_TIMEOUT / DECOMMISSION_TIMEOUT in rustfs_pool_expand.sh); + # the previous 45m step cap killed four consecutive runs mid-rebalance + # with steps 1-7 green. 240m covers the observed pace (rebalance 1/4 + # in ~21m) plus one decommission pass; override per dispatch. + timeout-minutes: ${{ inputs.pool_timeout_minutes || 240 }} # Case failures keep the run green: the report and the backlog # issue manager carry the product signal. continue-on-error: true diff --git a/docs/testing/functional-chain.md b/docs/testing/functional-chain.md index c27692267..4e686be24 100644 --- a/docs/testing/functional-chain.md +++ b/docs/testing/functional-chain.md @@ -22,13 +22,13 @@ The final job checks all twelve expected suite results and all twelve proof file ## Stalled suites and time limits -The nine non-performance reusable suite jobs, and the standalone table suite, have a **60-minute job limit**. The primary test step has a **45-minute limit** so a stalled test can fail before the hard job cancellation. Cleanup steps are limited to five minutes; report generation, dashboard/backlog operations and ordinary artifact uploads are limited to two minutes each. Separate installer and preflight steps are limited to five minutes. Performance is explicitly excluded from these limits. +Non-performance suites other than pool expansion have a **60-minute job limit** and a **45-minute primary test limit** so a stalled test can fail before the hard job cancellation. Pool expansion has a **360-minute job limit** and a **240-minute default test limit** to allow rebalance and decommission to finish. Its `pool_timeout_minutes` input overrides the test limit for manual and reusable calls; repository dispatch uses the default. Overrides must leave room within the fixed job limit for setup and finalizers. Cleanup steps are limited to five minutes; report generation, dashboard/backlog operations and ordinary artifact uploads are limited to two minutes each. Separate installer and preflight steps are limited to five minutes. Performance is explicitly excluded from these limits. The limits use GitHub Actions native timeouts. They are wall-clock limits, not log-idle detection: a process printing progress forever is still stopped. A step timeout is not a passing result. Existing `always()` finalizers attempt reporting and cleanup, and the chain driver's `always()` plus successful-prepare condition allows the next suite after a failed or cancelled suite. Missing or partial evidence still fails the complete-success gate; it cannot authorize closing historical issues. -The 15-minute difference between test and job limits is **headroom, not a reserved cleanup window**: checkout and setup also consume the 60-minute job budget, and several failing finalizers may exhaust it. Reaching the hard limit, losing a runner, or terminating an SSH connection does not guarantee remote processes have stopped or cleanup has completed. The next suite must retain its pre-test cleanup. Inspect the runner and remote VMs after a hard timeout before trusting subsequent results; do not interpret contaminated-environment failures as independent product regressions. +The difference between test and job limits (15 minutes for ordinary suites, 120 minutes for the default pool run) is **headroom, not a reserved cleanup window**: checkout and setup also consume the job budget, and several failing finalizers may exhaust it. Reaching the hard limit, losing a runner, or terminating an SSH connection does not guarantee remote processes have stopped or cleanup has completed. The next suite must retain its pre-test cleanup. Inspect the runner and remote VMs after a hard timeout before trusting subsequent results; do not interpret contaminated-environment failures as independent product regressions. -The performance workflow remains unchanged: its job limit is 900 minutes, the default duration is five minutes per round, and the external script retains its default sixty-second pauses. Its methods, sizes, manual overrides and step limits are not modified by the functional timeout policy. A long performance run can therefore still occupy the final lane and delay chain completion; it is not covered by the one-hour guarantee for non-performance suites. +The performance workflow remains unchanged: its job limit is 900 minutes, the default duration is five minutes per round, and the external script retains its default sixty-second pauses. Its methods, sizes, manual overrides and step limits are not modified by the functional timeout policy. A long performance run can therefore still occupy the final lane and delay chain completion; like pool expansion, it is not covered by the one-hour limit for ordinary functional suites. Job timeouts start when execution starts; they do **not** bound runner or concurrency queue time. Runner preflight checks detect an already-offline runner but are not reservations. The reusable chain continues after a job timeout without cancelling the whole Actions run. The legacy `repository_dispatch` path relies on an in-job handoff and cannot guarantee continuation after hard cancellation; use the reusable driver for bounded nightly chains. A queue watchdog would need an external dispatcher and environment recovery, not cancellation of the whole parent run (which would also cancel the remaining suites). diff --git a/rustfs/src/connect/heartbeat.rs b/rustfs/src/connect/heartbeat.rs index 27d70a625..105f0f60b 100644 --- a/rustfs/src/connect/heartbeat.rs +++ b/rustfs/src/connect/heartbeat.rs @@ -112,13 +112,15 @@ impl PendingHeartbeat { ] || self.capabilities == heartbeat_capabilities(false) || self.capabilities == heartbeat_capabilities(true) - // RUSTFS_COMPAT_TODO(connect-894) Remove after upgrades from the pre-service-memory capability set are unsupported. - // Preserve exact pending requests; never add the capability to a retry. + // RUSTFS_COMPAT_TODO(connect-894) Remove after upgrades from the pre-health and pre-service-memory sets are unsupported. + // Compare frozen historical advertisements, not subsets of today's capabilities. || [false, true].into_iter().any(|job_capable| { - heartbeat_capabilities(job_capable) - .iter() - .filter(|capability| capability.as_str() != "profile.memory.service@1") - .eq(self.capabilities.iter()) + let legacy = pre_health_heartbeat_capabilities(job_capable); + self.capabilities == legacy + || legacy + .iter() + .filter(|capability| capability.as_str() != "profile.memory.service@1") + .eq(self.capabilities.iter()) })) && self.sequence <= MAX_SEQUENCE && self.coarse_node_summary.is_valid() @@ -331,6 +333,40 @@ impl HeartbeatStateStore { } } +fn pre_health_heartbeat_capabilities(job_capable: bool) -> Vec { + let mut capabilities = [ + "heartbeat", + "diagnostics.policy.v1", + "inventory.environment@1", + "performance.client@1", + "performance.drive@1", + "performance.network@1", + "performance.object@1", + "performance.siteReplication@1", + "logs.capture@1", + "profile.cpu@1", + "profile.memory@1", + "profile.memory.service@1", + "profile.threads@1", + "telemetry.record@1", + "telemetry.otlp@1", + "telemetry.replay@1", + "top.api@1", + "top.disk@1", + "top.locks@1", + "top.net@1", + "top.rpc@1", + "inspect.object@1", + ] + .into_iter() + .map(str::to_owned) + .collect::>(); + if job_capable { + capabilities.insert(3, "jobs".to_owned()); + } + capabilities +} + fn heartbeat_capabilities(job_capable: bool) -> Vec { let mut capabilities = vec![ "heartbeat".to_owned(), diff --git a/rustfs/tests/connect_heartbeat.rs b/rustfs/tests/connect_heartbeat.rs index cb079ee30..2fb7147a0 100644 --- a/rustfs/tests/connect_heartbeat.rs +++ b/rustfs/tests/connect_heartbeat.rs @@ -644,7 +644,7 @@ async fn restart_replays_pending_request_then_advances_sequence() { assert_eq!(seen[1]["sequence"].as_u64(), seen[0]["sequence"].as_u64().map(|value| value + 1)); } -fn pre_service_memory_capabilities(job_capable: bool) -> Vec<&'static str> { +fn legacy_capabilities(job_capable: bool, service_memory: bool) -> Vec<&'static str> { let mut capabilities = vec!["heartbeat", "diagnostics.policy.v1", "inventory.environment@1"]; if job_capable { capabilities.push("jobs"); @@ -671,6 +671,13 @@ fn pre_service_memory_capabilities(job_capable: bool) -> Vec<&'static str> { "top.rpc@1", "inspect.object@1", ]); + if service_memory { + let memory = capabilities + .iter() + .position(|capability| *capability == "profile.memory@1") + .unwrap(); + capabilities.insert(memory + 1, "profile.memory.service@1"); + } capabilities } @@ -687,15 +694,15 @@ fn pending_heartbeat_with_capabilities(capabilities: &[&str]) -> Value { } #[tokio::test] -async fn restart_replays_exact_pre_service_memory_capabilities_with_and_without_jobs() { - for job_capable in [false, true] { +async fn restart_replays_exact_legacy_capabilities_with_and_without_jobs() { + for (job_capable, service_memory) in [(false, false), (true, false), (false, true), (true, true)] { let pki = TestPki::new(); let server = server(&pki, vec![Reply::ok("2026-08-22T01:02:03Z")]).await; let temp = tempfile::tempdir().expect("tempdir"); let shutdown = CancellationToken::new(); let config = config(&temp, &pki, &server); fs::create_dir_all(config.state_path.parent().expect("state directory")).expect("create state directory"); - let pending = pending_heartbeat_with_capabilities(&pre_service_memory_capabilities(job_capable)); + let pending = pending_heartbeat_with_capabilities(&legacy_capabilities(job_capable, service_memory)); let state = json!({"nextSequence": 0, "pending": pending}); fs::write(&config.state_path, serde_json::to_vec(&state).expect("heartbeat state JSON")).expect("write heartbeat state"); private_mode(&config.state_path); @@ -715,8 +722,8 @@ async fn restart_replays_exact_pre_service_memory_capabilities_with_and_without_ } #[tokio::test] -async fn pre_service_memory_compatibility_does_not_accept_changed_capabilities() { - for job_capable in [false, true] { +async fn legacy_compatibility_does_not_accept_changed_capabilities() { + for (job_capable, service_memory) in [(false, false), (true, false), (false, true), (true, true)] { for mutation in ["unknown", "missing", "duplicate", "reordered"] { let pki = TestPki::new(); let server = server(&pki, vec![]).await; @@ -724,7 +731,7 @@ async fn pre_service_memory_compatibility_does_not_accept_changed_capabilities() let shutdown = CancellationToken::new(); let config = config(&temp, &pki, &server); fs::create_dir_all(config.state_path.parent().expect("state directory")).expect("create state directory"); - let mut capabilities = pre_service_memory_capabilities(job_capable); + let mut capabilities = legacy_capabilities(job_capable, service_memory); match mutation { "unknown" => capabilities.push("shell.exec@1"), "missing" => { @@ -747,7 +754,7 @@ async fn pre_service_memory_compatibility_does_not_accept_changed_capabilities() wait_for(&mut status, |status| matches!(status, HeartbeatStatus::Failed { .. })).await, HeartbeatStatus::Failed { reason } if reason.contains("violates the protocol invariants") ), - "accepted altered persisted capabilities: {mutation}, jobs={job_capable}" + "accepted altered persisted capabilities: {mutation}, jobs={job_capable}, service_memory={service_memory}" ); assert!(server.seen.lock().expect("seen lock").is_empty()); runtime.shutdown().await; diff --git a/scripts/test_functional_chain.py b/scripts/test_functional_chain.py index b7fde514a..a69987377 100644 --- a/scripts/test_functional_chain.py +++ b/scripts/test_functional_chain.py @@ -403,12 +403,22 @@ class WorkflowTimeoutTests(unittest.TestCase): with self.subTest(suite=suite): source = (candidate.ROOT / f".github/workflows/rustfs-{suite}-test.yml").read_text() job = yaml_block(source.splitlines(), job_id, 2) - self.assertIn(" timeout-minutes: 60", job) + job_timeout = 360 if suite == "pool-expand" else 60 + self.assertIn(f" timeout-minutes: {job_timeout}", job) steps = named_steps(job) primary = [step for step in steps.values() if any( line in (" id: test", " id: pool_test") for line in step)] self.assertEqual(len(primary), 1) - self.assertIn(" timeout-minutes: 45", primary[0]) + if suite == "pool-expand": + self.assertIn(" timeout-minutes: ${{ inputs.pool_timeout_minutes || 240 }}", primary[0]) + for event in ("workflow_call", "workflow_dispatch"): + event_block = yaml_block(source.splitlines(), event, 2) + timeout_input = yaml_block(event_block, "pool_timeout_minutes", 6) + self.assertIsNotNone(timeout_input, event) + self.assertIn(" default: '240'", timeout_input) + self.assertIn(" required: false", timeout_input) + else: + self.assertIn(" timeout-minutes: 45", primary[0]) cleanup = "Cleanup environment" for phase in ("before", "after"): self.assertIn(" timeout-minutes: 5", steps[f"{cleanup} ({phase})"])