diff --git a/.config/e2e-nightly-selection.txt b/.config/e2e-nightly-selection.txt index bec86f799..07dfe2811 100644 --- a/.config/e2e-nightly-selection.txt +++ b/.config/e2e-nightly-selection.txt @@ -1,2 +1,2 @@ -sha256-darwin=83a7dcaffd5a789517ae9f02a224f66a9713937885cff96fca2ad7e216f197ae -sha256-linux=626c10f8c964507ff987b6c86069e9019dc6d2ae7fb02db9be5df5aa8cc5145b +sha256-darwin=efe68675f4676a0d2901242c5815b1f7af5cdec8231f7cccd408f6dacf5c75db +sha256-linux=003ed6480935a0a95680cc061be143722fbd37bd477b8ff98607b1a064f3aef1 diff --git a/.github/workflows/audit.yml b/.github/workflows/audit.yml index 3c4f371aa..45513d603 100644 --- a/.github/workflows/audit.yml +++ b/.github/workflows/audit.yml @@ -204,7 +204,7 @@ jobs: # open the Actions tab. Same ci-8 mechanism coverage.yml and # e2e-replication-nightly.yml already use. if: always() && github.event_name == 'schedule' && contains(needs.*.result, 'failure') - runs-on: sm-standard-2 + runs-on: ubuntu-latest timeout-minutes: 10 permissions: contents: read diff --git a/.github/workflows/build.yml b/.github/workflows/build.yml index cf0e7e449..c7c442afa 100644 --- a/.github/workflows/build.yml +++ b/.github/workflows/build.yml @@ -1167,7 +1167,7 @@ jobs: if: >- always() && github.event_name == 'schedule' && (contains(needs.*.result, 'failure') || contains(needs.*.result, 'cancelled')) - runs-on: sm-standard-2 + runs-on: ubuntu-latest timeout-minutes: 10 permissions: contents: read diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml index 08e386509..7d1171500 100644 --- a/.github/workflows/ci.yml +++ b/.github/workflows/ci.yml @@ -1270,7 +1270,7 @@ jobs: if: >- always() && github.event_name == 'schedule' && (contains(needs.*.result, 'failure') || contains(needs.*.result, 'cancelled')) - runs-on: sm-standard-2 + runs-on: ubuntu-latest timeout-minutes: 10 permissions: contents: read diff --git a/.github/workflows/coverage.yml b/.github/workflows/coverage.yml index 2f661dcc6..6ceefefad 100644 --- a/.github/workflows/coverage.yml +++ b/.github/workflows/coverage.yml @@ -128,7 +128,7 @@ jobs: # manual workflow_dispatch runs stay quiet so a debugging run never files a # spurious alert. if: always() && github.event_name == 'schedule' && contains(needs.*.result, 'failure') - runs-on: sm-standard-2 + runs-on: ubuntu-latest timeout-minutes: 10 permissions: contents: read diff --git a/.github/workflows/e2e-distributed.yml b/.github/workflows/e2e-distributed.yml index 7dfb84a29..889d84733 100644 --- a/.github/workflows/e2e-distributed.yml +++ b/.github/workflows/e2e-distributed.yml @@ -200,7 +200,7 @@ jobs: name: Alert on scheduled failure needs: [distributed] if: always() && github.event_name == 'schedule' && contains(needs.*.result, 'failure') - runs-on: sm-standard-2 + runs-on: ubuntu-latest timeout-minutes: 10 permissions: contents: read diff --git a/.github/workflows/e2e-replication-nightly.yml b/.github/workflows/e2e-replication-nightly.yml index ed63c21a8..b2f5b33ba 100644 --- a/.github/workflows/e2e-replication-nightly.yml +++ b/.github/workflows/e2e-replication-nightly.yml @@ -230,7 +230,7 @@ jobs: # manual workflow_dispatch runs stay quiet so a debugging run never files a # spurious alert. if: always() && github.event_name == 'schedule' && contains(needs.*.result, 'failure') - runs-on: sm-standard-2 + runs-on: ubuntu-latest timeout-minutes: 10 permissions: contents: read diff --git a/.github/workflows/e2e-s3tests.yml b/.github/workflows/e2e-s3tests.yml index 859e0e0ed..37a678273 100644 --- a/.github/workflows/e2e-s3tests.yml +++ b/.github/workflows/e2e-s3tests.yml @@ -453,7 +453,7 @@ jobs: # job fails. Alerts only for scheduled runs (backlog#1149 ci-8) — manual # dispatch failures are already watched by a human. if: always() && github.event_name == 'schedule' && contains(needs.*.result, 'failure') - runs-on: sm-standard-2 + runs-on: ubuntu-latest timeout-minutes: 10 permissions: contents: read diff --git a/.github/workflows/fuzz.yml b/.github/workflows/fuzz.yml index 0ad25ce1d..2e6d8d7b3 100644 --- a/.github/workflows/fuzz.yml +++ b/.github/workflows/fuzz.yml @@ -240,7 +240,7 @@ jobs: # job fails. Alerts only for scheduled (nightly) runs (backlog#1149 # ci-8); PR and manual dispatch failures are already watched by a human. if: always() && github.event_name == 'schedule' && contains(needs.*.result, 'failure') - runs-on: sm-standard-2 + runs-on: ubuntu-latest timeout-minutes: 10 permissions: contents: read diff --git a/.github/workflows/minio-interop.yml b/.github/workflows/minio-interop.yml index c6af0b2f2..e22b42048 100644 --- a/.github/workflows/minio-interop.yml +++ b/.github/workflows/minio-interop.yml @@ -60,9 +60,9 @@ permissions: jobs: minio-interop: name: MinIO interop (EC + SSE read parity) - # Skip on forks: needs the repo's runners and is not a contributor gate. + # Keep this scheduled interoperability lane scoped to the upstream repo. if: github.repository == 'rustfs/rustfs' - runs-on: sm-standard-2 + runs-on: ubuntu-latest timeout-minutes: 40 env: # Fixed 32-byte test KMS key baked into the fixture lab; not a secret. @@ -131,7 +131,7 @@ jobs: name: Alert on scheduled failure needs: [minio-interop] if: always() && github.event_name == 'schedule' && contains(needs.*.result, 'failure') - runs-on: sm-standard-2 + runs-on: ubuntu-latest timeout-minutes: 10 permissions: contents: read diff --git a/.github/workflows/mint.yml b/.github/workflows/mint.yml index 5849c1de2..d01c9d8c7 100644 --- a/.github/workflows/mint.yml +++ b/.github/workflows/mint.yml @@ -258,7 +258,7 @@ jobs: # job fails. Alerts only for scheduled runs (backlog#1149 ci-8) — manual # dispatch failures are already watched by a human. if: always() && github.event_name == 'schedule' && contains(needs.*.result, 'failure') - runs-on: sm-standard-2 + runs-on: ubuntu-latest timeout-minutes: 10 permissions: contents: read diff --git a/.github/workflows/nightly-gnu.yml b/.github/workflows/nightly-gnu.yml index e1a7a996f..739033ddb 100644 --- a/.github/workflows/nightly-gnu.yml +++ b/.github/workflows/nightly-gnu.yml @@ -498,7 +498,7 @@ jobs: if: >- always() && github.event_name == 'schedule' && (contains(needs.*.result, 'failure') || contains(needs.*.result, 'cancelled')) - runs-on: sm-standard-2 + runs-on: ubuntu-latest timeout-minutes: 10 permissions: contents: read diff --git a/.github/workflows/oidc-keycloak.yml b/.github/workflows/oidc-keycloak.yml index 94cbe3f1c..16b736521 100644 --- a/.github/workflows/oidc-keycloak.yml +++ b/.github/workflows/oidc-keycloak.yml @@ -51,7 +51,7 @@ concurrency: jobs: oidc-keycloak-live: name: OIDC Keycloak live gate - runs-on: sm-standard-2 + runs-on: ubuntu-latest timeout-minutes: 60 env: FORCE_JAVASCRIPT_ACTIONS_TO_NODE24: "true" @@ -72,6 +72,11 @@ jobs: - name: Build RustFS run: cargo build --locked -p rustfs --bin rustfs + - name: Set up Python + uses: actions/setup-python@5fda3b95a4ea91299a34e894583c3862153e4b97 # v7.0.0 + with: + python-version: "3.12" + - name: Install pinned request signer run: | python3 -m pip install --user --upgrade pip "awscurl==0.44" @@ -95,7 +100,7 @@ jobs: if: >- always() && github.event_name == 'schedule' && (needs.oidc-keycloak-live.result == 'failure' || needs.oidc-keycloak-live.result == 'cancelled') - runs-on: sm-standard-2 + runs-on: ubuntu-latest timeout-minutes: 10 permissions: contents: read diff --git a/.github/workflows/on-demand-migration-interop.yml b/.github/workflows/on-demand-migration-interop.yml index a4eb9fe4e..172607285 100644 --- a/.github/workflows/on-demand-migration-interop.yml +++ b/.github/workflows/on-demand-migration-interop.yml @@ -69,10 +69,9 @@ env: jobs: minio-source: name: MinIO source (read-through, list-through, backfill) - # Skip on forks: needs this repository's runners and is not a contributor - # gate. + # Keep this scheduled interoperability lane scoped to the upstream repo. if: github.repository == 'rustfs/rustfs' - runs-on: sm-standard-2 + runs-on: ubuntu-latest timeout-minutes: 90 env: NO_PROXY: 127.0.0.1,localhost @@ -108,11 +107,15 @@ jobs: run: | set -euo pipefail mkdir -p artifacts/odm-interop/minio + docker build -t rustfs-odm-interop-minio \ + -f crates/rio-v2/tests/minio_fixture_lab/Dockerfile \ + crates/rio-v2/tests/minio_fixture_lab docker run -d --name rustfs-odm-interop-minio \ + --entrypoint /usr/local/bin/minio \ -e "MINIO_ROOT_USER=${MINIO_ROOT_USER}" \ -e "MINIO_ROOT_PASSWORD=${MINIO_ROOT_PASSWORD}" \ -p 9100:9000 \ - minio/minio:RELEASE.2025-09-07T16-13-09Z server /data + rustfs-odm-interop-minio server /data for _ in $(seq 1 120); do curl -fsS http://127.0.0.1:9100/minio/health/live >/dev/null 2>&1 && break sleep 1 @@ -300,7 +303,7 @@ jobs: if: >- always() && github.event_name == 'schedule' && (contains(needs.*.result, 'failure') || contains(needs.*.result, 'cancelled')) - runs-on: sm-standard-2 + runs-on: ubuntu-latest timeout-minutes: 10 permissions: contents: read diff --git a/.github/workflows/performance-ab.yml b/.github/workflows/performance-ab.yml index 886e83674..d402ca636 100644 --- a/.github/workflows/performance-ab.yml +++ b/.github/workflows/performance-ab.yml @@ -366,7 +366,7 @@ jobs: if: >- always() && github.event_name == 'schedule' && (contains(needs.*.result, 'failure') || contains(needs.*.result, 'cancelled')) - runs-on: sm-standard-2 + runs-on: ubuntu-latest timeout-minutes: 10 permissions: contents: read diff --git a/.github/workflows/runner-hygiene.yml b/.github/workflows/runner-hygiene.yml index 3bd9b76f4..06c73e030 100644 --- a/.github/workflows/runner-hygiene.yml +++ b/.github/workflows/runner-hygiene.yml @@ -66,7 +66,7 @@ jobs: # scheduled runs file a tracking issue, manual dispatch stays quiet so # debugging never produces a spurious alert. if: always() && github.event_name == 'schedule' && contains(needs.*.result, 'failure') - runs-on: sm-standard-2 + runs-on: ubuntu-latest timeout-minutes: 10 permissions: contents: read diff --git a/.github/workflows/schedule-failure-alert-drill.yml b/.github/workflows/schedule-failure-alert-drill.yml index e00d2ddf3..735806e3a 100644 --- a/.github/workflows/schedule-failure-alert-drill.yml +++ b/.github/workflows/schedule-failure-alert-drill.yml @@ -58,7 +58,7 @@ jobs: # Mirrors the consumer wiring, minus the schedule-event guard (this # workflow is dispatch-only by design). if: always() && contains(needs.*.result, 'failure') - runs-on: sm-standard-2 + runs-on: ubuntu-latest timeout-minutes: 10 permissions: contents: read diff --git a/.github/workflows/targets-integration.yml b/.github/workflows/targets-integration.yml index fdefb7d7b..0e0255a9d 100644 --- a/.github/workflows/targets-integration.yml +++ b/.github/workflows/targets-integration.yml @@ -177,7 +177,7 @@ jobs: if: >- always() && github.event_name == 'schedule' && (contains(needs.*.result, 'failure') || contains(needs.*.result, 'cancelled')) - runs-on: sm-standard-2 + runs-on: ubuntu-latest timeout-minutes: 10 permissions: contents: read diff --git a/crates/e2e_test/src/chaos.rs b/crates/e2e_test/src/chaos.rs index 26092beca..ed343b893 100644 --- a/crates/e2e_test/src/chaos.rs +++ b/crates/e2e_test/src/chaos.rs @@ -384,17 +384,17 @@ pub(crate) fn is_cluster_heal_coordination_unavailable(error: &(dyn std::error:: message.contains("500 Internal Server Error") && message.contains("cluster heal coordination unavailable") } -/// Wait for the restarted cluster to admit the first root heal request. +/// Wait for the restarted cluster to admit a root heal request and return its response. pub(crate) async fn start_root_heal_when_control_ready( heal_url: &str, heal_body: &str, access_key: &str, secret_key: &str, -) -> ChaosResult<()> { +) -> ChaosResult { let deadline = tokio::time::Instant::now() + std::time::Duration::from_secs(45); loop { match signed_admin_post(heal_url, Some(heal_body), access_key, secret_key).await { - Ok(_) => return Ok(()), + Ok(body) => return Ok(body), Err(error) if is_cluster_heal_coordination_unavailable(error.as_ref()) && tokio::time::Instant::now() < deadline => { tokio::time::sleep(std::time::Duration::from_millis(250)).await; } diff --git a/crates/e2e_test/src/distributed_startup_regression_test.rs b/crates/e2e_test/src/distributed_startup_regression_test.rs index 57674fbdf..5a6f08579 100644 --- a/crates/e2e_test/src/distributed_startup_regression_test.rs +++ b/crates/e2e_test/src/distributed_startup_regression_test.rs @@ -215,6 +215,8 @@ mod tests { probe_startup_cas_binary(&binary, &nonce, &artifact).await?; let mut cluster = RustFSTestClusterEnvironment::new(4).await?; + // Isolate bootstrap CAS and its S3 readback from periodic usage publication. + cluster.set_env("RUSTFS_SCANNER_ENABLED", "false"); let mut logs = Vec::new(); let mut disks = Vec::new(); let mut endpoints = Vec::new(); diff --git a/crates/e2e_test/src/heal_erasure_disk_rebuild_test.rs b/crates/e2e_test/src/heal_erasure_disk_rebuild_test.rs index ebe36c089..c1a2c6af2 100644 --- a/crates/e2e_test/src/heal_erasure_disk_rebuild_test.rs +++ b/crates/e2e_test/src/heal_erasure_disk_rebuild_test.rs @@ -1799,7 +1799,8 @@ mod tests { let (mut client_token, mut task_status_url) = if background_rejoin_heal_evidence { (String::new(), String::new()) } else { - let heal_start_body = signed_admin_post(&heal_url, Some(heal_body), &cluster.access_key, &cluster.secret_key).await?; + let heal_start_body = + start_root_heal_when_control_ready(&heal_url, heal_body, &cluster.access_key, &cluster.secret_key).await?; let heal_start: serde_json::Value = serde_json::from_str(&heal_start_body) .map_err(|err| format!("heal start response is not JSON ({err}): {heal_start_body}"))?; let client_token = heal_start["clientToken"] @@ -2140,7 +2141,8 @@ mod tests { && status["summary"].as_str() == Some("failed") { let heal_start_body = - signed_admin_post(&heal_url, Some(heal_body), &cluster.access_key, &cluster.secret_key).await?; + start_root_heal_when_control_ready(&heal_url, heal_body, &cluster.access_key, &cluster.secret_key) + .await?; let heal_start: serde_json::Value = serde_json::from_str(&heal_start_body) .map_err(|err| format!("recovery heal start response is not JSON ({err}): {heal_start_body}"))?; client_token = heal_start["clientToken"] diff --git a/crates/e2e_test/src/replication_extension_test.rs b/crates/e2e_test/src/replication_extension_test.rs index a22c2e399..ab98d0cb0 100644 --- a/crates/e2e_test/src/replication_extension_test.rs +++ b/crates/e2e_test/src/replication_extension_test.rs @@ -7194,13 +7194,30 @@ async fn test_site_replication_state_edit_fresh_and_stale_real_dual_node() -> Re assert!(source_info.sites.iter().all(|peer| !peer.replicate_ilm_expiry)); assert!(target_info.sites.iter().all(|peer| !peer.replicate_ilm_expiry)); + let source_deployment_id = source_info + .sites + .iter() + .find(|peer| peer.endpoint == source_env.url) + .map(|peer| peer.deployment_id.clone()) + .ok_or("source site missing from its own replication info")?; + let target_deployment_id = target_info + .sites + .iter() + .find(|peer| peer.endpoint == target_env.url) + .map(|peer| peer.deployment_id.clone()) + .ok_or("target site missing from its own replication info")?; let target_status = wait_for_site_replication_status(&target_env, "peer-state=true", |status| status.peer_states.len() == 2).await?; let current_updated_at = target_status .peer_states - .values() - .find_map(|state| state.updated_at) + .get(&target_deployment_id) + .and_then(|state| state.updated_at) .ok_or("missing target site replication updated_at")?; + let source_updated_at = target_status + .peer_states + .get(&source_deployment_id) + .and_then(|state| state.updated_at) + .ok_or("missing source site replication updated_at")?; let mut stale_peers = BTreeMap::new(); for peer in target_info.sites { @@ -7244,16 +7261,23 @@ async fn test_site_replication_state_edit_fresh_and_stale_real_dual_node() -> Re .await?; assert!(target_after_fresh.sites.iter().all(|peer| peer.replicate_ilm_expiry)); + // State edits apply only to the addressed site; status reports each site's own state. let target_status_after_fresh = wait_for_site_replication_status(&target_env, "peer-state=true", |status| { status.peer_states.len() == 2 - && status.peer_states.values().all(|state| { - state.updated_at == Some(fresh_updated_at) && state.peers.values().all(|peer| peer.replicate_ilm_expiry) + && status.peer_states.get(&target_deployment_id).is_some_and(|state| { + state.updated_at == Some(fresh_updated_at) + && state.peers.len() == 2 + && state.peers.values().all(|peer| peer.replicate_ilm_expiry) }) }) .await?; - assert!(target_status_after_fresh.peer_states.values().all(|state| { - state.updated_at == Some(fresh_updated_at) && state.peers.values().all(|peer| peer.replicate_ilm_expiry) - })); + let source_state_after_fresh = target_status_after_fresh + .peer_states + .get(&source_deployment_id) + .ok_or("missing source site replication state after target edit")?; + assert_eq!(source_state_after_fresh.updated_at, Some(source_updated_at)); + assert_eq!(source_state_after_fresh.peers.len(), 2); + assert!(source_state_after_fresh.peers.values().all(|peer| !peer.replicate_ilm_expiry)); let source_after_fresh = site_replication_info(&source_env).await?; assert!(source_after_fresh.sites.iter().all(|peer| !peer.replicate_ilm_expiry)); diff --git a/crates/rio-v2/tests/minio_fixture_lab/Dockerfile b/crates/rio-v2/tests/minio_fixture_lab/Dockerfile index 98c02c829..89bc2d6ce 100644 --- a/crates/rio-v2/tests/minio_fixture_lab/Dockerfile +++ b/crates/rio-v2/tests/minio_fixture_lab/Dockerfile @@ -1,20 +1,34 @@ # Throwaway image for generating real MinIO on-disk fixtures without installing -# a MinIO server on the host. Multi-stage: pull the official MinIO server binary -# from the published image (linux, no dl.min.io download), then drop it into a +# a MinIO server on the host. Multi-stage: extract the official MinIO server +# binary from its checksum-pinned GitHub release package, then drop it into a # small Python image that runs the fixture lab (lab.py needs only python3 + # openssl; it drives MinIO's S3 API directly, so no `mc` is required). # # The MinIO release is pinned so the captured fixture format is reproducible; # this is the release the interop tests were validated against. # -# Both base images are build args so a network that cannot reach Docker Hub can -# point them at a mirror carrying the same content — quay.io publishes the MinIO -# releases, and public.ecr.aws mirrors the official Python images. CI keeps the -# Docker Hub defaults. Override with: -# --build-arg MINIO_IMAGE=quay.io/minio/minio:RELEASE.2025-09-07T16-13-09Z \ -# --build-arg PYTHON_IMAGE=public.ecr.aws/docker/library/python:3.12-slim -ARG MINIO_IMAGE=minio/minio:RELEASE.2025-09-07T16-13-09Z +# MINIO_IMAGE can override the package stage with an accessible image carrying +# the same release at /usr/bin/minio. PYTHON_IMAGE can point at a Python mirror. +ARG MINIO_IMAGE=minio-release ARG PYTHON_IMAGE=python:3.12-slim + +FROM scratch AS minio-package-amd64 +ADD --checksum=sha256:eeda08f699f6592d1b868ac8bda864ae2cacdb5ee1b888663366e8c8ff566249 https://github.com/minio/minio/releases/download/RELEASE.2025-09-07T16-13-09Z/minio_20250907161309.0.0_amd64.deb /minio.deb + +FROM scratch AS minio-package-arm64 +ADD --checksum=sha256:f9466ed832ac4926c8870b3b8378d77861a87b41529fecae7789b9bc54acd78b https://github.com/minio/minio/releases/download/RELEASE.2025-09-07T16-13-09Z/minio_20250907161309.0.0_arm64.deb /minio.deb + +FROM scratch AS minio-package-ppc64le +ADD --checksum=sha256:39eec77281c552dfa0a92b5221c657dc58306ea8c4e92b79e9562cee61b59caa https://github.com/minio/minio/releases/download/RELEASE.2025-09-07T16-13-09Z/minio_20250907161309.0.0_ppc64el.deb /minio.deb + +FROM minio-package-${TARGETARCH} AS minio-package + +FROM ${PYTHON_IMAGE} AS minio-release +COPY --from=minio-package /minio.deb /tmp/minio.deb +RUN dpkg-deb --extract /tmp/minio.deb /tmp/minio-package \ + && install -D /tmp/minio-package/usr/local/bin/minio /usr/bin/minio \ + && rm -rf /tmp/minio.deb /tmp/minio-package + FROM ${MINIO_IMAGE} AS minio FROM ${PYTHON_IMAGE} diff --git a/crates/rio-v2/tests/minio_fixture_lab/README.md b/crates/rio-v2/tests/minio_fixture_lab/README.md index a003bab26..a1634fd25 100644 --- a/crates/rio-v2/tests/minio_fixture_lab/README.md +++ b/crates/rio-v2/tests/minio_fixture_lab/README.md @@ -24,15 +24,14 @@ Use the automated path when you want the lab to: ## Networks without Docker Hub access -`capture_via_docker.sh` pulls its two base images from Docker Hub by default. Where that registry is unreachable, point the build at mirrors carrying the same content — quay.io publishes the MinIO releases and public.ecr.aws mirrors the official Python images: +`capture_via_docker.sh` extracts MinIO from the official GitHub release package after verifying its pinned SHA256. The Dockerfile selects the matching amd64, arm64, or ppc64le package; the MinIO release remains `RELEASE.2025-09-07T16-13-09Z`. Python comes from Docker Hub by default. Where that registry is unreachable, public.ecr.aws mirrors the official Python images: ```bash -MINIO_LAB_MINIO_IMAGE=quay.io/minio/minio:RELEASE.2025-09-07T16-13-09Z \ MINIO_LAB_PYTHON_IMAGE=public.ecr.aws/docker/library/python:3.12-slim \ ./capture_via_docker.sh ``` -Pin the MinIO tag to the same release the Dockerfile names; an unpinned `:latest` captures whatever format that day's build writes, which is not what the interop tests were validated against. +`MINIO_LAB_MINIO_IMAGE` can override the package stage with an accessible image containing the same MinIO release at `/usr/bin/minio`. Pin that image to the same release; an unpinned `:latest` captures whatever format that day's build writes, which is not what the interop tests were validated against. ## Layout diff --git a/crates/rio-v2/tests/minio_fixture_lab/capture_via_docker.sh b/crates/rio-v2/tests/minio_fixture_lab/capture_via_docker.sh index 6c1bec2f1..24e57f66b 100755 --- a/crates/rio-v2/tests/minio_fixture_lab/capture_via_docker.sh +++ b/crates/rio-v2/tests/minio_fixture_lab/capture_via_docker.sh @@ -34,9 +34,9 @@ if [ "${cases[0]}" != "all" ]; then done fi -# Base images are overridable so a network without Docker Hub access can point -# them at a mirror (see the Dockerfile header). Unset by default, which keeps the -# Dockerfile's Docker Hub defaults for CI. +# The Dockerfile defaults to a checksum-pinned MinIO release package and a +# Python base image. Overrides can select accessible images carrying the same +# content (see the Dockerfile header). build_args=() if [ -n "${MINIO_LAB_MINIO_IMAGE:-}" ]; then build_args+=(--build-arg "MINIO_IMAGE=${MINIO_LAB_MINIO_IMAGE}") @@ -49,7 +49,8 @@ echo ">> building ${IMAGE}" docker build -f "${SCRIPT_DIR}/Dockerfile" -t "${IMAGE}" "${build_args[@]}" "${SCRIPT_DIR}" echo ">> capturing fixtures into ${FIXTURE_REL}" -docker run --rm -v "${REPO_ROOT}:/repo" "${IMAGE}" \ +# The host reader initializes disk metadata inside the captured fixture tree. +docker run --rm --user "$(id -u):$(id -g)" -v "${REPO_ROOT}:/repo" "${IMAGE}" \ python3 "/repo/${LAB_REL}/lab.py" capture-matrix \ --root "/repo/${FIXTURE_REL}" \ --work-root /tmp/minio-lab-work \ diff --git a/rustfs/src/connect/diagnostics/job.rs b/rustfs/src/connect/diagnostics/job.rs index 576f62ce0..dac920e83 100644 --- a/rustfs/src/connect/diagnostics/job.rs +++ b/rustfs/src/connect/diagnostics/job.rs @@ -1161,7 +1161,8 @@ async fn execute_top_disk_job( }, }, limits: TopCaptureLimits { - max_duration_millis: envelope.parameters.duration_millis, + // Measured sampling time includes scheduling overhead within the signed job timeout. + max_duration_millis: envelope.limits.timeout_seconds * 1_000, max_working_memory_bytes: envelope.limits.max_memory_bytes, max_cpu_millis: envelope.limits.max_cpu_millis, ..TopCaptureLimits::default() @@ -1830,6 +1831,10 @@ mod tests { let result: serde_json::Value = serde_json::from_reader(archive.by_name("result.json").unwrap()).unwrap(); assert_eq!(result["toolId"], "top.disk"); assert_eq!(result["data"]["resourceAlias"], "resource-1"); + let duration_millis = result["durationMillis"].as_u64().unwrap(); + assert!(duration_millis >= top.parameters.duration_millis); + assert!(duration_millis <= top.limits.timeout_seconds * 1_000); + assert_eq!(result["data"]["windowMillis"], duration_millis); assert!(result["data"]["writeBytes"].as_u64().unwrap() >= 65_536); assert!(result["data"]["ioCount"].as_u64().unwrap() > 0); use std::io::Read as _; diff --git a/rustfs/src/server/readiness.rs b/rustfs/src/server/readiness.rs index 127d7b7e4..56fa9b885 100644 --- a/rustfs/src/server/readiness.rs +++ b/rustfs/src/server/readiness.rs @@ -2406,6 +2406,7 @@ mod tests { read_quorum_ready: true, write_quorum_ready: false, pool_metadata_write_ready: true, + unavailable_drives: Vec::new(), }, true, true, @@ -2432,6 +2433,7 @@ mod tests { read_quorum_ready: false, write_quorum_ready: false, pool_metadata_write_ready: false, + unavailable_drives: Vec::new(), }, false, true, diff --git a/rustfs/tests/connect_heartbeat.rs b/rustfs/tests/connect_heartbeat.rs index 2fb7147a0..814b8ad71 100644 --- a/rustfs/tests/connect_heartbeat.rs +++ b/rustfs/tests/connect_heartbeat.rs @@ -649,7 +649,6 @@ fn legacy_capabilities(job_capable: bool, service_memory: bool) -> Vec<&'static if job_capable { capabilities.push("jobs"); } - capabilities.push("health.check.service@1"); // Freeze the preceding release, independently of today's advertisement. capabilities.extend([ "performance.client@1", diff --git a/rustfs/tests/connect_perf_object.rs b/rustfs/tests/connect_perf_object.rs index d9b6deb91..4ed6fb45a 100644 --- a/rustfs/tests/connect_perf_object.rs +++ b/rustfs/tests/connect_perf_object.rs @@ -612,7 +612,12 @@ async fn real_rustfs_endpoint_and_production_cli_support_bounded_get_and_put_bod .expect("CLI task") .expect("run production rustfs binary"); - assert!(result.status.success(), "stderr: {}", String::from_utf8_lossy(&result.stderr)); + assert!( + result.status.success(), + "stdout: {}\nstderr: {}", + String::from_utf8_lossy(&result.stdout), + String::from_utf8_lossy(&result.stderr) + ); let stdout = String::from_utf8(result.stdout).expect("UTF-8 stdout"); assert!(stdout.contains("tool=performance.object outcome=SUCCEEDED reason=COMPLETE\n")); assert!(stdout.contains("upload=not-performed\n")); diff --git a/rustfs/tests/connect_top_disk.rs b/rustfs/tests/connect_top_disk.rs index 4332eef2f..05a28b3d7 100644 --- a/rustfs/tests/connect_top_disk.rs +++ b/rustfs/tests/connect_top_disk.rs @@ -15,8 +15,8 @@ use std::time::Duration; use rustfs::connect::diagnostics::{ - DiskCounterSnapshot, LocalTopConsent, MAX_SAFE_INTEGER, TopCaptureLimits, TopCaptureRequest, TopCaptureScope, TopOutcome, - TopReasonCode, evaluate_disk_window, + DiskCounterSnapshot, LocalTopConsent, MAX_SAFE_INTEGER, TopCaptureError, TopCaptureLimits, TopCaptureRequest, + TopCaptureScope, TopOutcome, TopReasonCode, evaluate_disk_window, }; use time::OffsetDateTime; @@ -104,3 +104,32 @@ fn top_disk_counter_reset_and_safe_integer_overflow_fail_without_data() { assert!(result.data.is_none()); } } + +#[test] +fn top_disk_measured_window_obeys_the_duration_limit() { + let mut request = request(); + request.window = Duration::from_millis(1_000); + request.limits.max_duration_millis = 2_000; + let before = DiskCounterSnapshot { + read_bytes: 100, + write_bytes: 200, + io_count: 10, + }; + let after = DiskCounterSnapshot { + read_bytes: 110, + write_bytes: 220, + io_count: 12, + }; + + for window_millis in [1_001, 2_000] { + let result = evaluate_disk_window(&request, before, after, window_millis).expect("window within limit"); + assert_eq!(result.duration_millis, window_millis); + assert_eq!(result.data.expect("disk counters").window_millis, window_millis); + } + for window_millis in [0, 2_001] { + assert!(matches!( + evaluate_disk_window(&request, before, after, window_millis), + Err(TopCaptureError::Limits) + )); + } +}