From e4dcc21206c2b36c93d507919bc3089f5cfa8664 Mon Sep 17 00:00:00 2001 From: Cursor Agent Date: Fri, 4 Sep 2026 14:18:12 +0000 Subject: [PATCH] test(e2e): add 4-node upgrade coverage for history and IAM Seed a 4-node cluster on the pinned previous release, then prove direct and rolling upgrades keep historical objects and IAM AK/SK working. The distributed Actions lane now downloads that binary. Co-authored-by: RustFS --- .config/e2e-distributed-selection.txt | 4 +- .config/nextest.toml | 5 +- .github/workflows/e2e-distributed.yml | 22 +- crates/e2e_test/README.md | 5 +- crates/e2e_test/src/distributed/harness.rs | 57 ++- crates/e2e_test/src/distributed/mod.rs | 1 + .../e2e_test/src/distributed/upgrade_test.rs | 346 ++++++++++++++++++ docs/testing/README.md | 2 +- docs/testing/ci-gates.md | 2 +- docs/testing/distributed-e2e.md | 16 +- 10 files changed, 446 insertions(+), 14 deletions(-) create mode 100644 crates/e2e_test/src/distributed/upgrade_test.rs diff --git a/.config/e2e-distributed-selection.txt b/.config/e2e-distributed-selection.txt index 2b9c0b098..cbf18ef75 100644 --- a/.config/e2e-distributed-selection.txt +++ b/.config/e2e-distributed-selection.txt @@ -1,2 +1,2 @@ -sha256-linux=8ad80ba33862f5457d7597c843750cebb13c1a6550f5ca810908f7f3a76e9d7f -sha256-darwin=8ad80ba33862f5457d7597c843750cebb13c1a6550f5ca810908f7f3a76e9d7f +sha256-linux=3a905602a68459b9f1dc0924b1fcc5b1aade9341999f2254ba986d2fd85ddde9 +sha256-darwin=3a905602a68459b9f1dc0924b1fcc5b1aade9341999f2254ba986d2fd85ddde9 diff --git a/.config/nextest.toml b/.config/nextest.toml index b365e0cfc..a6a84d403 100644 --- a/.config/nextest.toml +++ b/.config/nextest.toml @@ -538,7 +538,8 @@ test-group = 'e2e-cluster-nightly' # --------------------------------------------------------------------------- # Nightly / dispatch lane owned by .github/workflows/e2e-distributed.yml. # Each case starts four rustfs processes (and for site replication, two -# clusters). Serialized via e2e-cluster-nightly. Not a PR merge gate. +# clusters). Upgrade cases also require RUSTFS_UPGRADE_SOURCE_BINARY. +# Serialized via e2e-cluster-nightly. Not a PR merge gate. [profile.e2e-distributed] default-filter = 'package(e2e_test) & test(/^distributed::/)' fail-fast = false @@ -614,7 +615,7 @@ path = "junit.xml" # this merge/main lane while retaining nightly coverage. # * distributed:: — 4-node 4-disk Actions suite (S3, lock, versioning, # replication, quota, observability, expand/decommission/rebalance, site -# replication, chaos). Owns [profile.e2e-distributed] and +# replication, chaos, upgrade history/IAM). Owns [profile.e2e-distributed] and # .github/workflows/e2e-distributed.yml. # * on_demand_migration::interop_test — the ODM-20 provider interoperability # cases, which are meaningless without a source: they run in the dedicated diff --git a/.github/workflows/e2e-distributed.yml b/.github/workflows/e2e-distributed.yml index 55fb6a9df..d95102f5a 100644 --- a/.github/workflows/e2e-distributed.yml +++ b/.github/workflows/e2e-distributed.yml @@ -16,10 +16,11 @@ # # Each selected test starts a real localhost cluster via # `RustFSTestClusterEnvironment` (4 processes; 4 drives per node unless the -# case is a two-site 4-node 1-drive pair). Membership is +# case is a two-site 4-node 1-drive pair or a 4-node upgrade). Membership is # `[profile.e2e-distributed]` in `.config/nextest.toml`. This is not a required # merge check: it is the scheduled/dispatch counterpart to the hardware # functional chain that currently clones rustfs/auto-testing onto three VMs. +# Upgrade cases download the same pinned previous release as e2e-upgrade.yml. name: e2e-distributed @@ -51,6 +52,10 @@ jobs: NO_PROXY: 127.0.0.1,localhost HTTP_PROXY: "" HTTPS_PROXY: "" + # Pinned previous release used by distributed::upgrade_test (same pin as e2e-upgrade.yml). + UPGRADE_SOURCE_VERSION: 1.0.0-rc.2 + UPGRADE_SOURCE_ASSET: rustfs-linux-x86_64-gnu-v1.0.0-rc.2.zip + UPGRADE_SOURCE_SHA256: 7c789386bf85278f865b8e0d359bf4edb84d5aa408cc3fa54a18c25ca74cd6e7 steps: - name: Checkout repository uses: actions/checkout@9c091bb21b7c1c1d1991bb908d89e4e9dddfe3e0 # v7 @@ -65,6 +70,21 @@ jobs: cache-save-if: ${{ github.ref == 'refs/heads/main' }} install-build-packaging-tools: 'false' + - name: Download pinned previous release + env: + SOURCE_DIR: ${{ runner.temp }}/rustfs-upgrade-source + run: | + set -euo pipefail + mkdir -p "$SOURCE_DIR" + archive="$SOURCE_DIR/$UPGRADE_SOURCE_ASSET" + curl --fail --location --retry 3 --output "$archive" \ + "https://github.com/${GITHUB_REPOSITORY}/releases/download/${UPGRADE_SOURCE_VERSION}/${UPGRADE_SOURCE_ASSET}" + echo "$UPGRADE_SOURCE_SHA256 $archive" | sha256sum --check --strict + unzip -q "$archive" -d "$SOURCE_DIR" + chmod +x "$SOURCE_DIR/rustfs" + test -x "$SOURCE_DIR/rustfs" + echo "RUSTFS_UPGRADE_SOURCE_BINARY=$SOURCE_DIR/rustfs" >> "$GITHUB_ENV" + - name: Build rustfs binary run: | cargo build -p rustfs --bins diff --git a/crates/e2e_test/README.md b/crates/e2e_test/README.md index 0587fccf3..e5fce8f76 100644 --- a/crates/e2e_test/README.md +++ b/crates/e2e_test/README.md @@ -26,7 +26,7 @@ Registered in [`src/lib.rs`](src/lib.rs). Grouped by concern: | **protocols** | [`src/protocols/`](src/protocols) | FTPS, WebDAV, SFTP compliance. Fixed ports, own guide: [`src/protocols/README.md`](src/protocols/README.md) | | **reliant** | [`src/reliant/`](src/reliant) | Tests that reuse an **externally started** server (SQL/select, conditional writes, lifecycle, deleted-object reads, node-interact). Run via [`scripts/run_e2e_tests.sh`](../../scripts/run_e2e_tests.sh); see [`src/reliant/README.md`](src/reliant/README.md) | | **cluster** | `cluster_concurrency_test`, `stale_multipart_cleanup_cluster_test`, `namespace_lock_quorum_test`, `admin_timeout_regression_test`, `object_lambda_test`, `replication_extension_test`, `tier_stats_cluster_test` | Multi-node scenarios via `RustFSTestClusterEnvironment` | -| **distributed 4×4** | [`src/distributed/`](src/distributed) | Nightly `e2e-distributed` lane: S3, object lock/WORM, versioning, bucket/site replication, quota, expand/decommission/rebalance, concurrency, chaos. Map: [`docs/testing/distributed-e2e.md`](../../docs/testing/distributed-e2e.md) | +| **distributed 4×4** | [`src/distributed/`](src/distributed) | Nightly `e2e-distributed` lane: S3, object lock/WORM, versioning, bucket/site replication, quota, expand/decommission/rebalance, concurrency, chaos, 4-node upgrade of historical data and IAM AK/SK. Map: [`docs/testing/distributed-e2e.md`](../../docs/testing/distributed-e2e.md) | | **chaos / reliability** | [`src/chaos.rs`](src/chaos.rs), `reliability_disk_fault_test`, `heal_erasure_disk_rebuild_test`, `server_startup_failfast_test` | Disk offline/replace/corrupt, EC rebuild, heal, fail-fast startup | | **upgrade compatibility** | `upgrade_compatibility_test` | Pinned previous-release writes followed by current-build reads on the same data directory | @@ -193,7 +193,8 @@ cargo nextest run --profile e2e-smoke -p e2e_test cargo nextest run --profile e2e-full -p e2e_test # Cluster fault nightly lane cargo nextest run --profile e2e-nightly -p e2e_test -# 4-node 4-disk distributed lane (S3 / lock / versioning / replication / decommission / chaos) +# 4-node 4-disk distributed lane (S3 / lock / versioning / replication / decommission / chaos / upgrade) +# Upgrade cases need RUSTFS_UPGRADE_SOURCE_BINARY; without it they fail closed. cargo nextest run --profile e2e-distributed -p e2e_test # Replication nightly lane; awscurl is required for STS paths cargo nextest run --profile e2e-repl-nightly -p e2e_test diff --git a/crates/e2e_test/src/distributed/harness.rs b/crates/e2e_test/src/distributed/harness.rs index 346968cab..870dd6613 100644 --- a/crates/e2e_test/src/distributed/harness.rs +++ b/crates/e2e_test/src/distributed/harness.rs @@ -29,8 +29,8 @@ //! process count stays at eight rather than sixteen. use crate::common::{ - ClusterTopology, FAST_DATA_USAGE_SCANNER_ENV, RustFSTestClusterEnvironment, admin_request, local_http_client, - replication_fast_env, signed_request, + ClusterTopology, FAST_DATA_USAGE_SCANNER_ENV, RustFSTestClusterEnvironment, admin_request, build_test_s3_config, + local_http_client, replication_fast_env, signed_request, }; use crate::replication_extension_test::LOOPBACK_REPLICATION_TARGET_ENV; use aws_sdk_s3::Client; @@ -69,6 +69,20 @@ impl DistCluster { } pub async fn start_with_env(layout: DistLayout, extra_env: &[(&str, &str)]) -> TestResult { + let mut dist = Self::new_stopped_with_env(layout, extra_env).await?; + dist.cluster.start().await?; + Ok(dist) + } + + /// Allocate ports and data dirs without spawning processes. + /// + /// Upgrade tests configure capture logs, then start a pinned previous + /// binary against the same directories. + pub async fn new_stopped(layout: DistLayout) -> TestResult { + Self::new_stopped_with_env(layout, &[]).await + } + + pub async fn new_stopped_with_env(layout: DistLayout, extra_env: &[(&str, &str)]) -> TestResult { let topology = match layout { DistLayout::FourByFour => ClusterTopology::single_pool_multidrive(NODE_COUNT, DRIVES_PER_NODE), DistLayout::FourNodeFourDisk => ClusterTopology::single_pool(NODE_COUNT), @@ -81,10 +95,47 @@ impl DistCluster { for &(key, value) in extra_env { cluster.set_env(key, value); } - cluster.start().await?; Ok(Self { cluster }) } + /// Start every node with a specific `rustfs` binary, keeping the allocated + /// data directories. Used to seed an old on-disk format before upgrading. + pub async fn start_from_binary(&mut self, binary: &Path) -> TestResult { + self.cluster.start_with_binary(binary).await?; + wait_for_ready(&self.cluster).await?; + Ok(()) + } + + /// Stop every node and bring the same data directories up on the workspace + /// binary (direct upgrade). + pub async fn restart_with_current_binary(&mut self) -> TestResult { + self.cluster.stop(); + self.cluster.start().await?; + wait_for_ready(&self.cluster).await?; + Ok(()) + } + + /// Replace one running node with the workspace binary (rolling upgrade). + pub async fn replace_node_with_current_binary(&mut self, node_idx: usize) -> TestResult { + self.cluster.stop_node(node_idx)?; + self.cluster.start_node(node_idx).await?; + wait_for_ready(&self.cluster).await?; + Ok(()) + } + + pub fn client_with_credentials(&self, node_idx: usize, access_key: &str, secret_key: &str) -> TestResult { + if node_idx >= self.cluster.nodes.len() { + return Err("node_idx is invalid".into()); + } + Ok(Client::from_conf(build_test_s3_config( + &self.cluster.nodes[node_idx].url, + access_key, + secret_key, + None, + "cluster-iam-test", + ))) + } + /// Four single-node pools, started as two pools then expanded. Cold-start /// four-pool DistErasure can 500 the first PUT while pool.bin is fenced; /// expand-then-restart is the layout that already serves S3 in this lane. diff --git a/crates/e2e_test/src/distributed/mod.rs b/crates/e2e_test/src/distributed/mod.rs index fa1ba1875..108721f4c 100644 --- a/crates/e2e_test/src/distributed/mod.rs +++ b/crates/e2e_test/src/distributed/mod.rs @@ -31,4 +31,5 @@ mod replication_quota_test; mod s3_basic_test; mod s3_during_data_movement_test; mod site_replication_test; +mod upgrade_test; mod versioning_test; diff --git a/crates/e2e_test/src/distributed/upgrade_test.rs b/crates/e2e_test/src/distributed/upgrade_test.rs new file mode 100644 index 000000000..c3e184cb5 --- /dev/null +++ b/crates/e2e_test/src/distributed/upgrade_test.rs @@ -0,0 +1,346 @@ +// Copyright 2026 RustFS Team +// +// Licensed under the Apache License, Version 2.0 (the "License"); +// you may not use this file except in compliance with the License. +// You may obtain a copy of the License at +// +// http://www.apache.org/LICENSE-2.0 +// +// Unless required by applicable law or agreed to in writing, software +// distributed under the License is distributed on an "AS IS" BASIS, +// WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +// See the License for the specific language governing permissions and +// limitations under the License. + +//! 4-node upgrade coverage for historical objects and IAM AK/SK. +//! +//! Complements `upgrade_compatibility_test` (single-node SSE/multipart and +//! mixed-version listing). This module pins the distributed contract the +//! hardware upgrade chain is meant to catch: after a 4-node upgrade, objects +//! written on the previous release still read back, and IAM user credentials +//! created before the upgrade still authenticate. +//! +//! Requires `RUSTFS_UPGRADE_SOURCE_BINARY` pointing at the pinned previous +//! release. The `e2e-distributed` workflow downloads that binary; a local run +//! without it fails closed rather than skipping. + +use super::harness::{ + DistCluster, DistLayout, TestResult, assert_object_bytes, cluster_admin_ok, enable_versioning, get_object_bytes, put_object, + unique_bucket, wait_until, +}; +use crate::common::{ + AdminTransport, admin_add_canned_policy_via, admin_attach_user_policy_via, admin_create_user_via, init_logging, +}; +use aws_sdk_s3::Client; +use aws_sdk_s3::error::ProvideErrorMetadata; +use std::path::{Path, PathBuf}; +use std::time::Duration; +use uuid::Uuid; + +const SOURCE_BINARY_ENV: &str = "RUSTFS_UPGRADE_SOURCE_BINARY"; +const IAM_SECRET: &str = "UpgradeTestSecretKey1"; +const WRONG_SECRET: &str = "WrongSecretKey000000"; +const CREDENTIAL_TIMEOUT: Duration = Duration::from_secs(30); + +struct UpgradeSeed { + history_bucket: String, + history_key: &'static str, + history_body: Vec, + versioned_bucket: String, + versioned_key: &'static str, + version1: String, + version1_body: Vec, + version2: String, + version2_body: Vec, + iam_bucket: String, + iam_key: &'static str, + iam_body: Vec, + iam_user: String, + iam_secret: &'static str, +} + +fn source_binary() -> TestResult { + let path = std::env::var_os(SOURCE_BINARY_ENV).map(PathBuf::from).ok_or_else(|| { + format!( + "{SOURCE_BINARY_ENV} must point to the pinned previous release binary (the e2e-distributed workflow downloads it)" + ) + })?; + if !path.is_file() { + return Err(format!("upgrade source binary does not exist: {}", path.display()).into()); + } + Ok(path) +} + +fn capture_upgrade_logs(cluster: &mut DistCluster, label: &str) -> TestResult { + let Some(log_dir) = std::env::var_os("RUSTFS_E2E_LOG_DIR") else { + return Ok(()); + }; + std::fs::create_dir_all(&log_dir)?; + for node_idx in 0..cluster.cluster.nodes.len() { + let path = Path::new(&log_dir).join(format!("{label}-node-{node_idx}.log")); + cluster + .cluster + .set_node_capture_log_path(node_idx, path.to_string_lossy().into_owned())?; + } + Ok(()) +} + +fn iam_rw_policy(bucket: &str) -> String { + serde_json::json!({ + "Version": "2012-10-17", + "Statement": [{ + "Effect": "Allow", + "Action": ["s3:*"], + "Resource": [ + format!("arn:aws:s3:::{bucket}"), + format!("arn:aws:s3:::{bucket}/*") + ] + }] + }) + .to_string() +} + +async fn create_iam_user(dist: &DistCluster, user: &str, secret: &str, policy_name: &str, bucket: &str) -> TestResult { + let url = &dist.cluster.nodes[0].url; + let access = &dist.cluster.access_key; + let admin_secret = &dist.cluster.secret_key; + admin_create_user_via(AdminTransport::Signed, url, access, admin_secret, user, secret).await?; + admin_add_canned_policy_via(AdminTransport::Signed, url, access, admin_secret, policy_name, &iam_rw_policy(bucket)).await?; + admin_attach_user_policy_via(AdminTransport::Signed, url, access, admin_secret, policy_name, user).await?; + Ok(()) +} + +async fn wait_for_put(client: &Client, bucket: &str, key: &str, body: Vec, label: &str) -> TestResult { + wait_until( + CREDENTIAL_TIMEOUT, + || { + let client = client.clone(); + let bucket = bucket.to_string(); + let key = key.to_string(); + let body = body.clone(); + async move { + put_object(&client, &bucket, &key, body).await?; + Ok(true) + } + }, + label, + ) + .await +} + +async fn wait_for_bytes(client: &Client, bucket: &str, key: &str, expected: &[u8], label: &str) -> TestResult { + wait_until( + CREDENTIAL_TIMEOUT, + || { + let client = client.clone(); + let bucket = bucket.to_string(); + let key = key.to_string(); + let expected = expected.to_vec(); + async move { + let got = get_object_bytes(&client, &bucket, &key).await?; + Ok(got == expected) + } + }, + label, + ) + .await +} + +async fn seed_history_and_iam(dist: &DistCluster) -> TestResult { + let history_bucket = unique_bucket("upg-hist"); + let versioned_bucket = unique_bucket("upg-ver"); + let iam_bucket = unique_bucket("upg-iam"); + dist.create_bucket(&history_bucket).await?; + dist.create_bucket(&versioned_bucket).await?; + dist.create_bucket(&iam_bucket).await?; + + let root = dist.client(0)?; + enable_versioning(&root, &versioned_bucket).await?; + + let history_key = "plain-history.bin"; + let history_body = b"written by the previous 4-node release".to_vec(); + put_object(&root, &history_bucket, history_key, history_body.clone()).await?; + + let versioned_key = "versioned-history.txt"; + let version1_body = b"version-one-before-upgrade".to_vec(); + let version1 = root + .put_object() + .bucket(&versioned_bucket) + .key(versioned_key) + .body(aws_sdk_s3::primitives::ByteStream::from(version1_body.clone())) + .send() + .await? + .version_id() + .ok_or("first versioned PUT omitted version ID")? + .to_string(); + let version2_body = b"version-two-before-upgrade".to_vec(); + let version2 = root + .put_object() + .bucket(&versioned_bucket) + .key(versioned_key) + .body(aws_sdk_s3::primitives::ByteStream::from(version2_body.clone())) + .send() + .await? + .version_id() + .ok_or("second versioned PUT omitted version ID")? + .to_string(); + + let iam_user = format!("upg{}", &Uuid::new_v4().simple().to_string()[..8]); + let policy_name = format!("upgpol{}", &Uuid::new_v4().simple().to_string()[..8]); + create_iam_user(dist, &iam_user, IAM_SECRET, &policy_name, &iam_bucket).await?; + + let iam_key = "iam-history.bin"; + let iam_body = b"written with pre-upgrade IAM AK/SK".to_vec(); + let iam_client = dist.client_with_credentials(1, &iam_user, IAM_SECRET)?; + wait_for_put(&iam_client, &iam_bucket, iam_key, iam_body.clone(), "IAM user PUT before upgrade").await?; + + Ok(UpgradeSeed { + history_bucket, + history_key, + history_body, + versioned_bucket, + versioned_key, + version1, + version1_body, + version2, + version2_body, + iam_bucket, + iam_key, + iam_body, + iam_user, + iam_secret: IAM_SECRET, + }) +} + +async fn assert_history_and_iam(dist: &DistCluster, seed: &UpgradeSeed, context: &str) -> TestResult { + let root_a = dist.client(0)?; + let root_b = dist.client(3)?; + wait_for_bytes( + &root_b, + &seed.history_bucket, + seed.history_key, + &seed.history_body, + &format!("{context}: root GET historical object"), + ) + .await?; + assert_object_bytes(&root_a, &seed.history_bucket, seed.history_key, &seed.history_body).await?; + + let v1 = root_b + .get_object() + .bucket(&seed.versioned_bucket) + .key(seed.versioned_key) + .version_id(&seed.version1) + .send() + .await?; + let v1_body = v1.body.collect().await?.into_bytes(); + if v1_body.as_ref() != seed.version1_body.as_slice() { + return Err(format!("{context}: version 1 bytes changed after upgrade").into()); + } + let v2 = root_a + .get_object() + .bucket(&seed.versioned_bucket) + .key(seed.versioned_key) + .version_id(&seed.version2) + .send() + .await?; + let v2_body = v2.body.collect().await?.into_bytes(); + if v2_body.as_ref() != seed.version2_body.as_slice() { + return Err(format!("{context}: version 2 bytes changed after upgrade").into()); + } + + let users = cluster_admin_ok(&dist.cluster, http::Method::GET, "/rustfs/admin/v3/list-users", None).await?; + if !users.contains(&seed.iam_user) { + return Err(format!("{context}: list-users lost IAM user {}: {users}", seed.iam_user).into()); + } + + let iam_on_upgraded = dist.client_with_credentials(0, &seed.iam_user, seed.iam_secret)?; + let iam_on_peer = dist.client_with_credentials(3, &seed.iam_user, seed.iam_secret)?; + wait_for_bytes( + &iam_on_upgraded, + &seed.iam_bucket, + seed.iam_key, + &seed.iam_body, + &format!("{context}: IAM GET historical object on node 0"), + ) + .await?; + wait_for_bytes( + &iam_on_peer, + &seed.iam_bucket, + seed.iam_key, + &seed.iam_body, + &format!("{context}: IAM GET historical object on node 3"), + ) + .await?; + + let post_key = format!("after-upgrade-{context}.txt"); + let post_body = format!("{context}: written with the same IAM AK/SK after upgrade").into_bytes(); + wait_for_put( + &iam_on_peer, + &seed.iam_bucket, + &post_key, + post_body.clone(), + &format!("{context}: IAM PUT after upgrade"), + ) + .await?; + assert_object_bytes(&iam_on_upgraded, &seed.iam_bucket, &post_key, &post_body).await?; + + let bad = dist.client_with_credentials(1, &seed.iam_user, WRONG_SECRET)?; + match bad.get_object().bucket(&seed.iam_bucket).key(seed.iam_key).send().await { + Ok(_) => return Err(format!("{context}: wrong secret must not read the IAM object").into()), + Err(error) => { + let code = error.as_service_error().and_then(ProvideErrorMetadata::code); + if code == Some("SignatureDoesNotMatch") + || code == Some("InvalidAccessKeyId") + || code == Some("AccessDenied") + || code == Some("InvalidArgument") + { + } else if error.raw_response().is_some_and(|response| response.status().as_u16() == 403) { + } else { + return Err(format!("{context}: wrong secret failed with unexpected error {error:?}").into()); + } + } + } + + let post_root_key = format!("root-after-{context}.bin"); + let post_root_body = format!("{context}: root write after upgrade").into_bytes(); + put_object(&root_a, &seed.history_bucket, &post_root_key, post_root_body.clone()).await?; + assert_object_bytes(&root_b, &seed.history_bucket, &post_root_key, &post_root_body).await?; + Ok(()) +} + +#[tokio::test] +async fn four_node_direct_upgrade_preserves_history_and_iam_credentials() -> TestResult { + init_logging(); + let previous = source_binary()?; + let mut dist = DistCluster::new_stopped(DistLayout::FourNodeFourDisk).await?; + capture_upgrade_logs(&mut dist, "direct-upgrade")?; + dist.start_from_binary(&previous).await?; + + let seed = seed_history_and_iam(&dist).await?; + dist.restart_with_current_binary().await?; + assert_history_and_iam(&dist, &seed, "direct").await?; + Ok(()) +} + +#[tokio::test] +async fn four_node_rolling_upgrade_preserves_history_and_iam_credentials() -> TestResult { + init_logging(); + let previous = source_binary()?; + let mut dist = DistCluster::new_stopped(DistLayout::FourNodeFourDisk).await?; + capture_upgrade_logs(&mut dist, "rolling-upgrade")?; + dist.start_from_binary(&previous).await?; + + let seed = seed_history_and_iam(&dist).await?; + + dist.replace_node_with_current_binary(0).await?; + assert_history_and_iam(&dist, &seed, "one-current-node").await?; + + for node_idx in [1, 2] { + dist.replace_node_with_current_binary(node_idx).await?; + } + assert_history_and_iam(&dist, &seed, "one-previous-node").await?; + + dist.replace_node_with_current_binary(3).await?; + assert_history_and_iam(&dist, &seed, "homogeneous-current").await?; + Ok(()) +} diff --git a/docs/testing/README.md b/docs/testing/README.md index 53bab77bf..111aaf892 100644 --- a/docs/testing/README.md +++ b/docs/testing/README.md @@ -61,7 +61,7 @@ All profiles are defined in `.config/nextest.toml`; its block comments hold the | `e2e-full` | Merge-queue / main-push single-node e2e lane | | `e2e-repl-nightly` | Nightly slow / cross-process replication lane | | `e2e-nightly` | Nightly serial multi-process cluster fault lane | -| `e2e-distributed` | Nightly 4-node 4-disk S3 / lock / versioning / replication / quota / expand / decommission / rebalance / site-replication / chaos lane | +| `e2e-distributed` | Nightly 4-node 4-disk S3 / lock / versioning / replication / quota / expand / decommission / rebalance / site-replication / chaos / upgrade (history + IAM AK/SK) lane | | `e2e-protocols` | Nightly fixed-port FTPS/SFTP/WebDAV lane, run with `-j 1` | Membership of each e2e profile is pinned by a digest in `.config/e2e--selection.txt` and checked by `scripts/check_test_wiring.py --check-profile ` before the lane runs. To list what a profile selects on your platform (the result is platform-dependent because some modules are linux-only): diff --git a/docs/testing/ci-gates.md b/docs/testing/ci-gates.md index 522b753d7..0b6fa2d32 100644 --- a/docs/testing/ci-gates.md +++ b/docs/testing/ci-gates.md @@ -74,7 +74,7 @@ Scheduled lanes never block a PR. Their workflow-local gate fails the run, sched | `ci.yml` (weekly) | full matrix, including the schedule/dispatch-only rio-v2 jobs `build-rustfs-debug-binary-rio-v2` and `e2e-tests-rio-v2` | per-job | yes | dispatch `ci.yml` | | `build.yml` (weekly) | `build-rustfs` over the six-target platform matrix in `prepare-platform-matrix` (four Linux, macOS aarch64, Windows x86_64) | build/package integrity | yes | dispatch `build.yml` with an exact platform set | | `e2e-replication-nightly.yml` (nightly) | `repl-nightly`, `cluster-nightly`, `protocols-nightly` | three independent gates; JUnit, membership listing, server logs | yes | `cargo nextest run --profile e2e-repl-nightly -p e2e_test`; `--profile e2e-nightly`; `-j 1 --profile e2e-protocols` | -| `e2e-distributed.yml` (nightly) | `distributed` | 4-node 4-disk e2e gate; JUnit, membership listing, server logs | yes, with `never_ran_grace_until` | `cargo nextest run --profile e2e-distributed -p e2e_test` | +| `e2e-distributed.yml` (nightly) | `distributed` | 4-node 4-disk e2e gate including direct/rolling upgrade of historical objects and IAM AK/SK; JUnit, membership listing, server logs | yes, with `never_ran_grace_until` | download the pinned previous release as in the workflow, export `RUSTFS_UPGRADE_SOURCE_BINARY`, then `cargo nextest run --profile e2e-distributed -p e2e_test` | | `e2e-s3tests.yml` (weekly) | `s3tests` (single and distributed, four shards each), `upstream-head-canary` | compatibility gate; report, JUnit, node IDs, server logs | yes | `scripts/s3-tests/run.sh` against an existing single or distributed target | | `fuzz.yml` (nightly) | `nightly-fuzz-corpus` per target | gate; corpus and crash artifacts | yes | `MAX_TOTAL_TIME= ./scripts/fuzz/run.sh` | | `minio-interop.yml` (nightly) | `minio-interop` | EC + SSE read-parity gate | yes, with `never_ran_grace_until` | pinned Docker fixture steps in the workflow | diff --git a/docs/testing/distributed-e2e.md b/docs/testing/distributed-e2e.md index b09e7db2b..cdced9f25 100644 --- a/docs/testing/distributed-e2e.md +++ b/docs/testing/distributed-e2e.md @@ -10,7 +10,7 @@ The in-tree harness runs every node on `127.0.0.1` with a distinct port. That ma | Layout | Constructor | Use | |---|---|---| | 4 nodes × 4 drives, one pool | `ClusterTopology::single_pool_multidrive(4, 4)` | S3, object lock, versioning, quota, observability, concurrency, chaos | -| 4 nodes × 1 drive, one pool | `ClusterTopology::single_pool(4)` | Two-site replication (8 processes total) | +| 4 nodes × 1 drive, one pool | `ClusterTopology::single_pool(4)` | Two-site replication (8 processes total); direct/rolling upgrade from the pinned previous release | | 2 single-node pools × 4 drives, then `append_single_node_pool` twice | expansion seed | Pool expand, then decommission / rebalance / integrity | A pool striped across several localhost ports is not expressible (`RUSTFS_VOLUMES` host ellipses would collide on disk paths). Multi-host striped pools remain the hardware functional-chain / backlog #1313 / #1314 lane. @@ -32,6 +32,7 @@ Decommission and rebalance POST currently 500 on localhost DistErasure multi-poo - Node kill/restart, full process restart, drive offline (4×4). Volume-proxy blackhole stays in `cluster_volume_fault_proxy_pass_smoke` (2×2); a 4-node volume proxy cannot format because RPC audience is the listen port - Multipart, cross-node listing, list-buckets agreement - Concurrent GET while a peer node is killed +- Direct and rolling upgrade from the pinned previous release: historical objects, versioned history, and IAM user AK/SK still work afterwards ## Existing Actions gaps this lane does not replace @@ -39,7 +40,8 @@ Those suites stay in place; this lane fills the in-tree 4×4 hole they leave. | Existing lane | Gap | |---|---| -| `rustfs-*-test.yml` functional chain | Clones private `rustfs/auto-testing`, runs on three shared VMs (`vm000`–`vm002`), `continue-on-error: true`, not a merge signal, not 4 nodes | +| `rustfs-*-test.yml` functional chain | Clones private `rustfs/auto-testing`, runs on three shared VMs (`vm000`–`vm002`), `continue-on-error: true`, not a merge signal, not 4 nodes. Hardware `rustfs-upgrade-test.yml` stays there | +| `e2e-upgrade.yml` | Single-node SSE/multipart/delete-marker contracts plus mixed-version listing; does not pin IAM user AK/SK on a 4-node cluster | | `e2e-smoke` / `e2e-full` | Almost all cases are single-node | | `e2e-nightly` | 4-node cluster faults and heal, not S3/lock/versioning/quota/decommission matrix | | `e2e-repl-nightly` | Site and bucket replication on 1–3 *single-node* processes | @@ -52,7 +54,17 @@ Hardware power-loss, NIC pull, and real disk replacement still belong on the smo ```bash cargo build -p rustfs --bins +# Upgrade cases require the pinned previous binary (CI downloads it). +export RUSTFS_UPGRADE_SOURCE_BINARY=/path/to/rustfs-1.0.0-rc.2 cargo nextest run --profile e2e-distributed -p e2e_test ``` +Without `RUSTFS_UPGRADE_SOURCE_BINARY` the two `distributed::upgrade_test::*` cases fail closed. Filter them out for a local run that is not checking upgrade: + +```bash +cargo nextest run --profile e2e-distributed -p e2e_test -E 'not test(/^distributed::upgrade_test::/)' +``` + +The upgrade topology is `ClusterTopology::single_pool(4)` (4 nodes × 1 drive). That matches the proven mixed-version fixture in `upgrade_compatibility_test`; 4×4 localhost drives are rejected by the previous release's same-device disk check. + Membership is pinned by `.config/e2e-distributed-selection.txt`. Update it with `python3 ./scripts/check_test_wiring.py --update-profile e2e-distributed linux` after adding or renaming a case.