# nextest configuration for RustFS. # # Serialize two known load-sensitive / global-state-sharing ecstore test groups # so the full parallel nextest suite stops producing spurious failures # (backlog #937). These tests pass in isolation but flake under the loaded # parallel run for two distinct reasons: # # * store::bucket::tests::bucket_delete_* share process/global state (disk # registry, lock client) and race make_bucket into InsufficientWriteQuorum # when run concurrently with other ecstore tests. # * bucket_lifecycle_ops::tests::concurrent_resend_same_part_commits_one_generation # asserts a lock-acquire correctness property whose serialized cross-disk # commits exceed the (already max'd, 60s) acquire deadline only when the # suite saturates disk I/O. # # serial_test's #[serial] attribute does NOT serialize these across runs: # nextest executes each test in its own process, where the in-process # serial_test mutex has no effect. A nextest test-group with max-threads = 1 is # the mechanism that actually serializes across nextest's process boundary. # # --------------------------------------------------------------------------- # Profiles # --------------------------------------------------------------------------- # The `default` profile is what local `cargo nextest run` uses. It NEVER # retries: a red test locally means a real failure to investigate, not noise to # paper over. The `ci` profile (below) is the strict CI gate: global # retries = 0 so a new race's first occurrence is never masked, plus a # narrowly-scoped quarantine list (retries = 2) for tests with a tracked OPEN # flake issue. Flake policy lives in docs/testing/README.md. [test-groups] ecstore-serial-flaky = { max-threads = 1 } # Reliability / fault-injection e2e tests each spawn a single-node 4-disk RustFS # server and manipulate its disk directories at runtime (crates/e2e_test: # reliability_disk_fault_test, degraded_read_eof_regression_test / dist-13). They # are correct in isolation but resource-heavy; serialize them under nextest's # process boundary (serial_test's #[serial] does not cross it) so several 4-disk # servers never run at once. ci-7's nightly picks these up via the e2e suite; # they are deliberately NOT in the fast PR `e2e-smoke` filter. e2e-reliability = { max-threads = 1 } e2e-inline-boundaries = { max-threads = 1 } # --- default profile (local): serialize the flaky groups, never retry -------- [[profile.default.overrides]] filter = 'package(rustfs-ecstore) & (test(concurrent_resend_same_part_commits_one_generation) | test(/^store::bucket::tests::bucket_delete_(mark_delete|purge_removes|default_s3_delete)/))' test-group = 'ecstore-serial-flaky' # Serialize the multipart crash-consistency scenarios (dist-2, backlog#1150): # each spawns a 4-disk hermetic erasure set and drives full staged-upload + # commit + GET cycles — the same cross-disk-commit IO shape that made # concurrent_resend load-sensitive. Preventive serialization only, no retries. # The matching ci-profile override is after [profile.ci]. [[profile.default.overrides]] filter = 'package(rustfs-ecstore) & test(/^set_disk::ops::multipart::tests::crash_consistency::/)' test-group = 'ecstore-serial-flaky' # Serialize the 4-disk reliability / degraded-read e2e tests (see the # e2e-reliability test-group note above). The matching ci-profile override is at # the end of the file, after [profile.ci] is declared. [[profile.default.overrides]] filter = 'package(e2e_test) & test(/^(reliability_disk_fault|degraded_read_eof_regression)_test::/)' test-group = 'e2e-reliability' [[profile.default.overrides]] filter = 'package(e2e_test) & test(/^inline_fast_path_cluster_test::/)' test-group = 'e2e-inline-boundaries' # --------------------------------------------------------------------------- # ci profile — the strict CI gate (ci.yml `cargo nextest run --profile ci`) # --------------------------------------------------------------------------- [profile.ci] # Strict: a new race must fail on its first occurrence, never be retried away. retries = 0 # Report every failure in one run instead of bailing on the first. fail-fast = false [profile.ci.junit] # Emitted to target/nextest/ci/junit.xml; uploaded as a CI artifact. # Tests that pass only after a quarantine retry are marked `flaky` here — that # marker is the observable signal the flake policy is built around. path = "junit.xml" # =========================================================================== # QUARANTINE — flaky tests granted retries = 2 under the ci profile ONLY. # # RULES (enforced by review, see docs/testing/README.md): # * Every entry MUST link exactly one OPEN issue tracking the flake. # * An entry stays until the issue is fixed (test made robust) or the test is # deleted — 30-day policy. No entry may exist without a live issue link. # # Each entry also re-declares the `ecstore-serial-flaky` test-group so the # serialization holds under the ci profile (nextest evaluates a named # profile's own overrides list, not the default profile's). # =========================================================================== # QUARANTINE: OPEN backlog#937 — concurrent_resend lock-acquire deadline flakes # under saturated disk I/O in the full parallel suite. [[profile.ci.overrides]] filter = 'package(rustfs-ecstore) & test(concurrent_resend_same_part_commits_one_generation)' test-group = 'ecstore-serial-flaky' retries = 2 # QUARANTINE: OPEN backlog#937 — store::bucket::tests::bucket_delete_* race # make_bucket into InsufficientWriteQuorum via shared global state under load. [[profile.ci.overrides]] filter = 'package(rustfs-ecstore) & test(/^store::bucket::tests::bucket_delete_(mark_delete|purge_removes|default_s3_delete)/)' test-group = 'ecstore-serial-flaky' retries = 2 # QUARANTINE: OPEN rustfs#4690 — walk_dir stall-budget accounting test depends # on producer/consumer timing windows that stretch past the budget on loaded # CI runners (regression test for rustfs#4644; failed on a zero-Rust-diff PR). [[profile.ci.overrides]] filter = 'package(rustfs-ecstore) & test(walk_dir_does_not_charge_consumer_backpressure_to_the_stall_budget)' retries = 2 # Serialize the 4-disk reliability / degraded-read e2e tests under the ci # profile too (see the e2e-reliability test-group note near the top). Not a # quarantine: no retries, just single-threaded so several 4-disk servers never # run concurrently when ci-7's nightly runs the full e2e suite. [[profile.ci.overrides]] filter = 'package(e2e_test) & test(/^(reliability_disk_fault|degraded_read_eof_regression)_test::/)' test-group = 'e2e-reliability' # Serialize the multipart crash-consistency scenarios under the ci profile too # (see the matching default-profile override near the top). Not a quarantine: # no retries, just serialized 4-disk cross-disk-commit IO. [[profile.ci.overrides]] filter = 'package(rustfs-ecstore) & test(/^set_disk::ops::multipart::tests::crash_consistency::/)' test-group = 'ecstore-serial-flaky' # --------------------------------------------------------------------------- # e2e-smoke profile — PR smoke subset of the e2e_test crate (backlog#1149 ci-4) # --------------------------------------------------------------------------- # PR smoke subset of the e2e_test crate (backlog#1149 ci-4). This profile is # the single wiring mechanism for e2e tests in CI: other suites join by # extending this filter (or a sibling profile), never by adding ad-hoc e2e # jobs to ci.yml. Admission criteria (see crates/e2e_test/README.md): fast, # single-node topology, no external dependencies (no awscurl / Vault / fixed # ports / pre-started server), no #[ignore]. # # Each e2e test spawns its own rustfs server on a random port with an isolated # temp dir (crates/e2e_test/src/common.rs), so the subset is parallel-safe. # # Replication failure harness (backlog#1147 repl-8): the first clause admits # its four in-process fake-target self-tests. They bind random loopback ports, # use no external service, and finish in under a second. # # Replication PR subset (backlog#1147 repl-1): the second clause admits the 20 # FAST bucket-replication tests from replication_extension_test — the # target-registration / replication-check / list / remove / delete admin paths # that validate config synchronously and never wait for asynchronous # replication convergence. Each spawns its own single-node rustfs server(s) on # random ports (source, plus an independent single-node target for the pair # checks — NOT a cluster), so the subset stays parallel-safe and single-digit # seconds. The data-plane tests that poll for convergence and all # `_real_dual_node` / `_real_single_node` # site-replication tests run in the [profile.e2e-repl-nightly] lane below, NOT # here. This allowlist is the single source of truth for the PR/nightly split: # the nightly profile derives its set as "the replication module MINUS this # allowlist", so any new replication test lands in nightly by default (never # silently unrun) until it is explicitly blessed as fast here. Keep the two # regexes byte-identical. Count invariant: 20 here + 27 nightly = 47 total # (authority: `cargo nextest list`; docs/testing/e2e-suite-inventory.md). # HISTORY (2026-07-11): the 20 fast tests were briefly pulled out of this lane # (#4724) because they set a loopback (127.0.0.1) replication target that the # SSRF egress guard rejected on every PR after repl-1 (#4712). That is fixed — # the guard now honours an off-by-default opt-in and this suite's source servers # set it (RUSTFS_REPLICATION_ALLOW_LOOPBACK_TARGET) — so the allowlist below is # restored. # # Security negative-auth subset (backlog#1151 sec-5): the three attacker-facing # S3 auth-rejection suites join the first clause above by module name — # presigned_negative (sec-2), negative_sigv4 (sec-1, header SigV4), and # admin_auth (sec-4, admin gate + root-credential lifecycle). All three use # RustFSTestEnvironment on a random port and are parallel-safe, so they meet the # smoke admission criteria unchanged. This is the wiring step that makes those # merged suites actually execute on every PR (they were dead until listed here). # A rename that drops any of them out of this filter would silently thin the # security gate with no CI signal, so scripts/check_security_smoke_count.sh owns # a count-floor guard over exactly this subset (infra-12 mechanism, floor in # .config/security-smoke-floor.txt), invoked from the e2e-tests job in ci.yml. # NOT here by topology: the GHSA-3p3x FTPS/WebDAV constant-time e2e # (protocols::test_protocol_core_suite) binds fixed ports and needs the # ftps,webdav features, so it cannot join this random-port, default-feature # profile; its GHSA-r5qv sibling is a unit test that already runs in the # test-and-lint `--all --exclude e2e_test` pass. See # docs/testing/security-regressions.md for the full CI-execution map. # # ILM tiering main path (backlog#1148 ilm-7): the `reliant::tiering::` clause # admits the hermetic transition e2e. Like the fast replication pair checks it # spawns a second independent single-node server (the cold RustFS tier), not a # cluster, so it keeps the lane's parallel-safe / no-external-dependency # properties. The RustFS warm backend has no loopback guard (that guard is # replication-only), so it needs no opt-in env for its 127.0.0.1 tier target. [profile.e2e-smoke] default-filter = """ package(e2e_test) & ( test(/^(delete_marker_migration_semantics|version_id_regression|list_objects_v2_pagination|list_object_versions_regression|list_objects_duplicates|list_buckets_double_slash|leading_slash_key|special_chars|create_bucket_region|delete_objects_versioning|head_object_consistency|head_object_range|copy_object_metadata|copy_object_tagging|copy_source_invalid_date|content_encoding|multipart_storage_class|storage_class_capability|ssec_copy|anonymous_access|bucket_policy_check|presigned_negative|negative_sigv4|admin_auth|notification_webhook|tls_hot_reload|console_smoke|admin_iam_crud|admin_pools)_test::|^fake_s3_target::/) | test(/^replication_extension_test::(test_replication_check_succeeds_with_remote_target|test_replication_check_rejects_target_without_object_lock|test_set_remote_target_rejects_unversioned_source_bucket|test_replication_check_rejects_unversioned_source_bucket|test_replication_check_rejects_missing_replication_config|test_replication_check_rejects_invalid_bucket|test_set_remote_target_rejects_same_bucket_on_same_deployment|test_set_remote_target_rejects_unversioned_target_bucket|test_set_remote_target_update_requires_arn|test_set_remote_target_update_rejects_missing_target|test_set_remote_target_rejects_invalid_target_url|test_set_remote_target_rejects_self_signed_https_target_without_skip_tls_verify|test_set_remote_target_rejects_private_ca_https_target_without_ca_cert_pem|test_list_remote_targets_rejects_empty_bucket|test_list_remote_targets_rejects_invalid_bucket|test_remove_remote_target_rejects_missing_target|test_remove_remote_target_rejects_missing_arn|test_remove_remote_target_rejects_invalid_bucket|test_remove_remote_target_rejects_target_used_by_replication|test_delete_bucket_replication_removes_remote_target)$/) | test(/^reliant::lifecycle::/) | test(/^reliant::tiering::/) ) """ fail-fast = false # --------------------------------------------------------------------------- # e2e-repl-nightly profile — scheduled full replication e2e lane (repl-1) # --------------------------------------------------------------------------- # backlog#1147 repl-1 (deps: ci-4). Runs the SLOW / cross-process replication # tests that are unfit for the per-PR e2e-smoke gate: # # * 2 remote-target TLS validation tests. # * 12 bucket-replication data-plane/helper tests — they PUT/delete objects # and poll until source and target converge; two replicate over HTTPS, two # pin active SSE failure contracts, and one guards event/history observers. # The SSE-S3 contract remains ignored under backlog#1291. # * 11 `_real_dual_node` site-replication tests — each spawns TWO full rustfs # servers and drives the cross-process site-replication control plane. # * 1 `_real_three_node` site-replication test. # * 1 `_real_single_node` service-account round-trip test. # # The set is defined as "everything in replication_extension_test that is NOT # in the e2e-smoke PR allowlist above" (the negated clause is byte-identical to # the allowlist), so a newly added replication test automatically runs here # until it is explicitly promoted to the fast PR subset — no replication test # is ever silently left out of CI. # # #[serial] does NOT serialize under nextest (process-per-test; see the file # header). These tests need no cross-test serialization: each spawns its own # server(s) on random ports with isolated temp dirs, so they are parallel-safe # by construction — the same property the e2e-smoke subset relies on. If load # on the runner surfaces a real flake, quarantine the specific test with an # OPEN issue link (ci-10 / backlog#937 policy), never blanket-retry or exclude. # # Wired by .github/workflows/e2e-replication-nightly.yml (schedule + # workflow_dispatch), which builds the rustfs binary once, installs awscurl so # the STS dual-node test actually exercises its path (it skips gracefully with # a visible log line when awscurl is absent), and routes scheduled failures # through .github/actions/schedule-failure-issue (ci-8). Explicit division of # labor with ci-5's future e2e-full merge gate: these tests run ONLY here, not # double-run there. TODO(ci-7): fold this interim repl-owned lane into the ci # domain's consolidated scheduled e2e workflow once it exists. [profile.e2e-repl-nightly] default-filter = """ package(e2e_test) & test(/^replication_extension_test::/) & !test(/^replication_extension_test::(test_replication_check_succeeds_with_remote_target|test_replication_check_rejects_target_without_object_lock|test_set_remote_target_rejects_unversioned_source_bucket|test_replication_check_rejects_unversioned_source_bucket|test_replication_check_rejects_missing_replication_config|test_replication_check_rejects_invalid_bucket|test_set_remote_target_rejects_same_bucket_on_same_deployment|test_set_remote_target_rejects_unversioned_target_bucket|test_set_remote_target_update_requires_arn|test_set_remote_target_update_rejects_missing_target|test_set_remote_target_rejects_invalid_target_url|test_set_remote_target_rejects_self_signed_https_target_without_skip_tls_verify|test_set_remote_target_rejects_private_ca_https_target_without_ca_cert_pem|test_list_remote_targets_rejects_empty_bucket|test_list_remote_targets_rejects_invalid_bucket|test_remove_remote_target_rejects_missing_target|test_remove_remote_target_rejects_missing_arn|test_remove_remote_target_rejects_invalid_bucket|test_remove_remote_target_rejects_target_used_by_replication|test_delete_bucket_replication_removes_remote_target)$/) """ fail-fast = false [profile.e2e-repl-nightly.junit] # Emitted to target/nextest/e2e-repl-nightly/junit.xml; uploaded by the nightly # workflow as the failure-triage artifact. path = "junit.xml" # --------------------------------------------------------------------------- # e2e-full profile — merge-gate full single-node e2e lane (backlog#1149 ci-5) # --------------------------------------------------------------------------- # The merge gate (ci.yml `e2e-full` job: push main + merge_group + # workflow_dispatch). Runs the never-automated user-visible suites — KMS (40), # object_lock (33), multipart_auth (109), quota, checksum, encryption, # security-boundary, ... — that the fast PR `e2e-smoke` subset deliberately # skips. Budget <= 45 min; authority for the suite count is `cargo nextest list # --profile e2e-full` (see docs/testing/e2e-suite-inventory.md). # # The filter is "the whole e2e_test crate MINUS the sets owned by other lanes": # * protocols:: — FTPS/SFTP/WebDAV, still pinned to --test-threads=1 by fixed # ports; they join a scheduled lane once ci-6 randomises the ports (ci-7). # * the 7 cluster suites that spin up a RustFSTestClusterEnvironment # (cluster_concurrency, cluster_multidrive_pool, stale_multipart_cleanup_cluster, # namespace_lock_quorum, heal_erasure_disk_rebuild, admin_timeout_regression, # object_lambda) — too heavy for the merge budget; they run in ci-7's # nightly 4-node lane. # * replication_extension_test — repl-1 already splits it into the PR # `e2e-smoke` (20 fast) and `e2e-repl-nightly` (27 slow) lanes and reserves # it for those, so e2e-full does not double-run it. # * #[ignore]d tests — nextest skips them by default (no --run-ignored); the # manual-localhost:9000 reliant/policy tests are ci-13's migration. # # Each e2e test spawns its own single-node rustfs server on a random port with # an isolated temp dir (crates/e2e_test/src/common.rs), so the set is # parallel-safe — the same property e2e-smoke relies on. The exception is the # 4-disk reliability / degraded-read fault-injection tests, serialized below # (identical to the ci profile) so several 4-disk servers never run at once. # KNOWN-FAILURE EXCLUSIONS (characterization run 29381309848, 2026-07-15: # 341 ran / 32 failed on the suites' first automated run ever). Deterministic # product failures cannot be quarantined away with retries, so each family is # excluded here with its tracking issue, under the same discipline as the # ci-profile quarantine (docs/testing/README.md): every entry MUST cite one # OPEN issue, and the fixing PR MUST delete the exclusion. The passing # negative-path siblings of each family stay in as regression guards. # * rustfs#4843 — over-limit archive entry paths hard-reject the whole # archive even under ignore-errors semantics. # * rustfs#4846 — distributed-lock quorum tests misclassify as timeout # under parallel load (multi-node in-process clusters; natural home is # ci-7's nightly cluster lane). [profile.e2e-full] default-filter = """ package(e2e_test) & !test(/^protocols::/) & !test(/^(admin_timeout_regression_test|cluster_concurrency_test|cluster_multidrive_pool_test|heal_erasure_disk_rebuild_test|namespace_lock_quorum_test|object_lambda_test|stale_multipart_cleanup_cluster_test)::/) & !test(/^replication_extension_test::/) & !test(/^multipart_auth_test::test_signed_put_object_extract_skips_invalid_entry_when_ignore_errors_enabled$/) & !test(/^snowball_auto_extract_test::tests::snowball_auto_extract_(ignores_invalid_entries_when_requested|supports_standard_headers_with_combined_extract_options)$/) & !test(/^reliant::lock::test_distributed_lock_(2_nodes_grpc_read_survives_failed_node|4_nodes_grpc_read_write_quorum_split_with_two_failed_nodes)$/) """ fail-fast = false [profile.e2e-full.junit] # Emitted to target/nextest/e2e-full/junit.xml; uploaded by the e2e-full job. path = "junit.xml" # Serialize the 4-disk reliability / degraded-read e2e tests under e2e-full too # (see the e2e-reliability test-group note near the top of this file). Not a # quarantine: no retries, just single-threaded so several 4-disk servers never # run concurrently. [[profile.e2e-full.overrides]] filter = 'package(e2e_test) & test(/^(reliability_disk_fault|degraded_read_eof_regression)_test::/)' test-group = 'e2e-reliability' [[profile.e2e-full.overrides]] filter = 'package(e2e_test) & test(/^inline_fast_path_cluster_test::/)' test-group = 'e2e-inline-boundaries'