mirror of
https://github.com/rustfs/rustfs.git
synced 2026-10-04 04:21:35 +00:00
fix(ci): restore mainline and scheduled test reliability (#8187)
* fix(ci): restore mainline and scheduled test reliability * fix(ci): provide GitHub CLI for CPU acceptance * ci: provide Docker for CPU service acceptance * ci: provision Python and Docker for OIDC validation * ci: restore hosted runners for Docker validation * ci: use verified MinIO release packages for interop * ci: preserve host ownership of MinIO fixtures * fix(ci): correct diagnostic limits and isolate startup checks * test(connect): include object CLI failure details * test(readiness): initialize unavailable drive diagnostics
This commit is contained in:
@@ -1,2 +1,2 @@
|
||||
sha256-darwin=83a7dcaffd5a789517ae9f02a224f66a9713937885cff96fca2ad7e216f197ae
|
||||
sha256-linux=626c10f8c964507ff987b6c86069e9019dc6d2ae7fb02db9be5df5aa8cc5145b
|
||||
sha256-darwin=efe68675f4676a0d2901242c5815b1f7af5cdec8231f7cccd408f6dacf5c75db
|
||||
sha256-linux=003ed6480935a0a95680cc061be143722fbd37bd477b8ff98607b1a064f3aef1
|
||||
|
||||
@@ -204,7 +204,7 @@ jobs:
|
||||
# open the Actions tab. Same ci-8 mechanism coverage.yml and
|
||||
# e2e-replication-nightly.yml already use.
|
||||
if: always() && github.event_name == 'schedule' && contains(needs.*.result, 'failure')
|
||||
runs-on: sm-standard-2
|
||||
runs-on: ubuntu-latest
|
||||
timeout-minutes: 10
|
||||
permissions:
|
||||
contents: read
|
||||
|
||||
@@ -1167,7 +1167,7 @@ jobs:
|
||||
if: >-
|
||||
always() && github.event_name == 'schedule' &&
|
||||
(contains(needs.*.result, 'failure') || contains(needs.*.result, 'cancelled'))
|
||||
runs-on: sm-standard-2
|
||||
runs-on: ubuntu-latest
|
||||
timeout-minutes: 10
|
||||
permissions:
|
||||
contents: read
|
||||
|
||||
@@ -1270,7 +1270,7 @@ jobs:
|
||||
if: >-
|
||||
always() && github.event_name == 'schedule' &&
|
||||
(contains(needs.*.result, 'failure') || contains(needs.*.result, 'cancelled'))
|
||||
runs-on: sm-standard-2
|
||||
runs-on: ubuntu-latest
|
||||
timeout-minutes: 10
|
||||
permissions:
|
||||
contents: read
|
||||
|
||||
@@ -128,7 +128,7 @@ jobs:
|
||||
# manual workflow_dispatch runs stay quiet so a debugging run never files a
|
||||
# spurious alert.
|
||||
if: always() && github.event_name == 'schedule' && contains(needs.*.result, 'failure')
|
||||
runs-on: sm-standard-2
|
||||
runs-on: ubuntu-latest
|
||||
timeout-minutes: 10
|
||||
permissions:
|
||||
contents: read
|
||||
|
||||
@@ -200,7 +200,7 @@ jobs:
|
||||
name: Alert on scheduled failure
|
||||
needs: [distributed]
|
||||
if: always() && github.event_name == 'schedule' && contains(needs.*.result, 'failure')
|
||||
runs-on: sm-standard-2
|
||||
runs-on: ubuntu-latest
|
||||
timeout-minutes: 10
|
||||
permissions:
|
||||
contents: read
|
||||
|
||||
@@ -230,7 +230,7 @@ jobs:
|
||||
# manual workflow_dispatch runs stay quiet so a debugging run never files a
|
||||
# spurious alert.
|
||||
if: always() && github.event_name == 'schedule' && contains(needs.*.result, 'failure')
|
||||
runs-on: sm-standard-2
|
||||
runs-on: ubuntu-latest
|
||||
timeout-minutes: 10
|
||||
permissions:
|
||||
contents: read
|
||||
|
||||
@@ -453,7 +453,7 @@ jobs:
|
||||
# job fails. Alerts only for scheduled runs (backlog#1149 ci-8) — manual
|
||||
# dispatch failures are already watched by a human.
|
||||
if: always() && github.event_name == 'schedule' && contains(needs.*.result, 'failure')
|
||||
runs-on: sm-standard-2
|
||||
runs-on: ubuntu-latest
|
||||
timeout-minutes: 10
|
||||
permissions:
|
||||
contents: read
|
||||
|
||||
@@ -240,7 +240,7 @@ jobs:
|
||||
# job fails. Alerts only for scheduled (nightly) runs (backlog#1149
|
||||
# ci-8); PR and manual dispatch failures are already watched by a human.
|
||||
if: always() && github.event_name == 'schedule' && contains(needs.*.result, 'failure')
|
||||
runs-on: sm-standard-2
|
||||
runs-on: ubuntu-latest
|
||||
timeout-minutes: 10
|
||||
permissions:
|
||||
contents: read
|
||||
|
||||
@@ -60,9 +60,9 @@ permissions:
|
||||
jobs:
|
||||
minio-interop:
|
||||
name: MinIO interop (EC + SSE read parity)
|
||||
# Skip on forks: needs the repo's runners and is not a contributor gate.
|
||||
# Keep this scheduled interoperability lane scoped to the upstream repo.
|
||||
if: github.repository == 'rustfs/rustfs'
|
||||
runs-on: sm-standard-2
|
||||
runs-on: ubuntu-latest
|
||||
timeout-minutes: 40
|
||||
env:
|
||||
# Fixed 32-byte test KMS key baked into the fixture lab; not a secret.
|
||||
@@ -131,7 +131,7 @@ jobs:
|
||||
name: Alert on scheduled failure
|
||||
needs: [minio-interop]
|
||||
if: always() && github.event_name == 'schedule' && contains(needs.*.result, 'failure')
|
||||
runs-on: sm-standard-2
|
||||
runs-on: ubuntu-latest
|
||||
timeout-minutes: 10
|
||||
permissions:
|
||||
contents: read
|
||||
|
||||
@@ -258,7 +258,7 @@ jobs:
|
||||
# job fails. Alerts only for scheduled runs (backlog#1149 ci-8) — manual
|
||||
# dispatch failures are already watched by a human.
|
||||
if: always() && github.event_name == 'schedule' && contains(needs.*.result, 'failure')
|
||||
runs-on: sm-standard-2
|
||||
runs-on: ubuntu-latest
|
||||
timeout-minutes: 10
|
||||
permissions:
|
||||
contents: read
|
||||
|
||||
@@ -498,7 +498,7 @@ jobs:
|
||||
if: >-
|
||||
always() && github.event_name == 'schedule' &&
|
||||
(contains(needs.*.result, 'failure') || contains(needs.*.result, 'cancelled'))
|
||||
runs-on: sm-standard-2
|
||||
runs-on: ubuntu-latest
|
||||
timeout-minutes: 10
|
||||
permissions:
|
||||
contents: read
|
||||
|
||||
@@ -51,7 +51,7 @@ concurrency:
|
||||
jobs:
|
||||
oidc-keycloak-live:
|
||||
name: OIDC Keycloak live gate
|
||||
runs-on: sm-standard-2
|
||||
runs-on: ubuntu-latest
|
||||
timeout-minutes: 60
|
||||
env:
|
||||
FORCE_JAVASCRIPT_ACTIONS_TO_NODE24: "true"
|
||||
@@ -72,6 +72,11 @@ jobs:
|
||||
- name: Build RustFS
|
||||
run: cargo build --locked -p rustfs --bin rustfs
|
||||
|
||||
- name: Set up Python
|
||||
uses: actions/setup-python@5fda3b95a4ea91299a34e894583c3862153e4b97 # v7.0.0
|
||||
with:
|
||||
python-version: "3.12"
|
||||
|
||||
- name: Install pinned request signer
|
||||
run: |
|
||||
python3 -m pip install --user --upgrade pip "awscurl==0.44"
|
||||
@@ -95,7 +100,7 @@ jobs:
|
||||
if: >-
|
||||
always() && github.event_name == 'schedule' &&
|
||||
(needs.oidc-keycloak-live.result == 'failure' || needs.oidc-keycloak-live.result == 'cancelled')
|
||||
runs-on: sm-standard-2
|
||||
runs-on: ubuntu-latest
|
||||
timeout-minutes: 10
|
||||
permissions:
|
||||
contents: read
|
||||
|
||||
@@ -69,10 +69,9 @@ env:
|
||||
jobs:
|
||||
minio-source:
|
||||
name: MinIO source (read-through, list-through, backfill)
|
||||
# Skip on forks: needs this repository's runners and is not a contributor
|
||||
# gate.
|
||||
# Keep this scheduled interoperability lane scoped to the upstream repo.
|
||||
if: github.repository == 'rustfs/rustfs'
|
||||
runs-on: sm-standard-2
|
||||
runs-on: ubuntu-latest
|
||||
timeout-minutes: 90
|
||||
env:
|
||||
NO_PROXY: 127.0.0.1,localhost
|
||||
@@ -108,11 +107,15 @@ jobs:
|
||||
run: |
|
||||
set -euo pipefail
|
||||
mkdir -p artifacts/odm-interop/minio
|
||||
docker build -t rustfs-odm-interop-minio \
|
||||
-f crates/rio-v2/tests/minio_fixture_lab/Dockerfile \
|
||||
crates/rio-v2/tests/minio_fixture_lab
|
||||
docker run -d --name rustfs-odm-interop-minio \
|
||||
--entrypoint /usr/local/bin/minio \
|
||||
-e "MINIO_ROOT_USER=${MINIO_ROOT_USER}" \
|
||||
-e "MINIO_ROOT_PASSWORD=${MINIO_ROOT_PASSWORD}" \
|
||||
-p 9100:9000 \
|
||||
minio/minio:RELEASE.2025-09-07T16-13-09Z server /data
|
||||
rustfs-odm-interop-minio server /data
|
||||
for _ in $(seq 1 120); do
|
||||
curl -fsS http://127.0.0.1:9100/minio/health/live >/dev/null 2>&1 && break
|
||||
sleep 1
|
||||
@@ -300,7 +303,7 @@ jobs:
|
||||
if: >-
|
||||
always() && github.event_name == 'schedule' &&
|
||||
(contains(needs.*.result, 'failure') || contains(needs.*.result, 'cancelled'))
|
||||
runs-on: sm-standard-2
|
||||
runs-on: ubuntu-latest
|
||||
timeout-minutes: 10
|
||||
permissions:
|
||||
contents: read
|
||||
|
||||
@@ -366,7 +366,7 @@ jobs:
|
||||
if: >-
|
||||
always() && github.event_name == 'schedule' &&
|
||||
(contains(needs.*.result, 'failure') || contains(needs.*.result, 'cancelled'))
|
||||
runs-on: sm-standard-2
|
||||
runs-on: ubuntu-latest
|
||||
timeout-minutes: 10
|
||||
permissions:
|
||||
contents: read
|
||||
|
||||
@@ -66,7 +66,7 @@ jobs:
|
||||
# scheduled runs file a tracking issue, manual dispatch stays quiet so
|
||||
# debugging never produces a spurious alert.
|
||||
if: always() && github.event_name == 'schedule' && contains(needs.*.result, 'failure')
|
||||
runs-on: sm-standard-2
|
||||
runs-on: ubuntu-latest
|
||||
timeout-minutes: 10
|
||||
permissions:
|
||||
contents: read
|
||||
|
||||
@@ -58,7 +58,7 @@ jobs:
|
||||
# Mirrors the consumer wiring, minus the schedule-event guard (this
|
||||
# workflow is dispatch-only by design).
|
||||
if: always() && contains(needs.*.result, 'failure')
|
||||
runs-on: sm-standard-2
|
||||
runs-on: ubuntu-latest
|
||||
timeout-minutes: 10
|
||||
permissions:
|
||||
contents: read
|
||||
|
||||
@@ -177,7 +177,7 @@ jobs:
|
||||
if: >-
|
||||
always() && github.event_name == 'schedule' &&
|
||||
(contains(needs.*.result, 'failure') || contains(needs.*.result, 'cancelled'))
|
||||
runs-on: sm-standard-2
|
||||
runs-on: ubuntu-latest
|
||||
timeout-minutes: 10
|
||||
permissions:
|
||||
contents: read
|
||||
|
||||
@@ -384,17 +384,17 @@ pub(crate) fn is_cluster_heal_coordination_unavailable(error: &(dyn std::error::
|
||||
message.contains("500 Internal Server Error") && message.contains("cluster heal coordination unavailable")
|
||||
}
|
||||
|
||||
/// Wait for the restarted cluster to admit the first root heal request.
|
||||
/// Wait for the restarted cluster to admit a root heal request and return its response.
|
||||
pub(crate) async fn start_root_heal_when_control_ready(
|
||||
heal_url: &str,
|
||||
heal_body: &str,
|
||||
access_key: &str,
|
||||
secret_key: &str,
|
||||
) -> ChaosResult<()> {
|
||||
) -> ChaosResult<String> {
|
||||
let deadline = tokio::time::Instant::now() + std::time::Duration::from_secs(45);
|
||||
loop {
|
||||
match signed_admin_post(heal_url, Some(heal_body), access_key, secret_key).await {
|
||||
Ok(_) => return Ok(()),
|
||||
Ok(body) => return Ok(body),
|
||||
Err(error) if is_cluster_heal_coordination_unavailable(error.as_ref()) && tokio::time::Instant::now() < deadline => {
|
||||
tokio::time::sleep(std::time::Duration::from_millis(250)).await;
|
||||
}
|
||||
|
||||
@@ -215,6 +215,8 @@ mod tests {
|
||||
probe_startup_cas_binary(&binary, &nonce, &artifact).await?;
|
||||
|
||||
let mut cluster = RustFSTestClusterEnvironment::new(4).await?;
|
||||
// Isolate bootstrap CAS and its S3 readback from periodic usage publication.
|
||||
cluster.set_env("RUSTFS_SCANNER_ENABLED", "false");
|
||||
let mut logs = Vec::new();
|
||||
let mut disks = Vec::new();
|
||||
let mut endpoints = Vec::new();
|
||||
|
||||
@@ -1799,7 +1799,8 @@ mod tests {
|
||||
let (mut client_token, mut task_status_url) = if background_rejoin_heal_evidence {
|
||||
(String::new(), String::new())
|
||||
} else {
|
||||
let heal_start_body = signed_admin_post(&heal_url, Some(heal_body), &cluster.access_key, &cluster.secret_key).await?;
|
||||
let heal_start_body =
|
||||
start_root_heal_when_control_ready(&heal_url, heal_body, &cluster.access_key, &cluster.secret_key).await?;
|
||||
let heal_start: serde_json::Value = serde_json::from_str(&heal_start_body)
|
||||
.map_err(|err| format!("heal start response is not JSON ({err}): {heal_start_body}"))?;
|
||||
let client_token = heal_start["clientToken"]
|
||||
@@ -2140,7 +2141,8 @@ mod tests {
|
||||
&& status["summary"].as_str() == Some("failed")
|
||||
{
|
||||
let heal_start_body =
|
||||
signed_admin_post(&heal_url, Some(heal_body), &cluster.access_key, &cluster.secret_key).await?;
|
||||
start_root_heal_when_control_ready(&heal_url, heal_body, &cluster.access_key, &cluster.secret_key)
|
||||
.await?;
|
||||
let heal_start: serde_json::Value = serde_json::from_str(&heal_start_body)
|
||||
.map_err(|err| format!("recovery heal start response is not JSON ({err}): {heal_start_body}"))?;
|
||||
client_token = heal_start["clientToken"]
|
||||
|
||||
@@ -7194,13 +7194,30 @@ async fn test_site_replication_state_edit_fresh_and_stale_real_dual_node() -> Re
|
||||
assert!(source_info.sites.iter().all(|peer| !peer.replicate_ilm_expiry));
|
||||
assert!(target_info.sites.iter().all(|peer| !peer.replicate_ilm_expiry));
|
||||
|
||||
let source_deployment_id = source_info
|
||||
.sites
|
||||
.iter()
|
||||
.find(|peer| peer.endpoint == source_env.url)
|
||||
.map(|peer| peer.deployment_id.clone())
|
||||
.ok_or("source site missing from its own replication info")?;
|
||||
let target_deployment_id = target_info
|
||||
.sites
|
||||
.iter()
|
||||
.find(|peer| peer.endpoint == target_env.url)
|
||||
.map(|peer| peer.deployment_id.clone())
|
||||
.ok_or("target site missing from its own replication info")?;
|
||||
let target_status =
|
||||
wait_for_site_replication_status(&target_env, "peer-state=true", |status| status.peer_states.len() == 2).await?;
|
||||
let current_updated_at = target_status
|
||||
.peer_states
|
||||
.values()
|
||||
.find_map(|state| state.updated_at)
|
||||
.get(&target_deployment_id)
|
||||
.and_then(|state| state.updated_at)
|
||||
.ok_or("missing target site replication updated_at")?;
|
||||
let source_updated_at = target_status
|
||||
.peer_states
|
||||
.get(&source_deployment_id)
|
||||
.and_then(|state| state.updated_at)
|
||||
.ok_or("missing source site replication updated_at")?;
|
||||
|
||||
let mut stale_peers = BTreeMap::new();
|
||||
for peer in target_info.sites {
|
||||
@@ -7244,16 +7261,23 @@ async fn test_site_replication_state_edit_fresh_and_stale_real_dual_node() -> Re
|
||||
.await?;
|
||||
assert!(target_after_fresh.sites.iter().all(|peer| peer.replicate_ilm_expiry));
|
||||
|
||||
// State edits apply only to the addressed site; status reports each site's own state.
|
||||
let target_status_after_fresh = wait_for_site_replication_status(&target_env, "peer-state=true", |status| {
|
||||
status.peer_states.len() == 2
|
||||
&& status.peer_states.values().all(|state| {
|
||||
state.updated_at == Some(fresh_updated_at) && state.peers.values().all(|peer| peer.replicate_ilm_expiry)
|
||||
&& status.peer_states.get(&target_deployment_id).is_some_and(|state| {
|
||||
state.updated_at == Some(fresh_updated_at)
|
||||
&& state.peers.len() == 2
|
||||
&& state.peers.values().all(|peer| peer.replicate_ilm_expiry)
|
||||
})
|
||||
})
|
||||
.await?;
|
||||
assert!(target_status_after_fresh.peer_states.values().all(|state| {
|
||||
state.updated_at == Some(fresh_updated_at) && state.peers.values().all(|peer| peer.replicate_ilm_expiry)
|
||||
}));
|
||||
let source_state_after_fresh = target_status_after_fresh
|
||||
.peer_states
|
||||
.get(&source_deployment_id)
|
||||
.ok_or("missing source site replication state after target edit")?;
|
||||
assert_eq!(source_state_after_fresh.updated_at, Some(source_updated_at));
|
||||
assert_eq!(source_state_after_fresh.peers.len(), 2);
|
||||
assert!(source_state_after_fresh.peers.values().all(|peer| !peer.replicate_ilm_expiry));
|
||||
|
||||
let source_after_fresh = site_replication_info(&source_env).await?;
|
||||
assert!(source_after_fresh.sites.iter().all(|peer| !peer.replicate_ilm_expiry));
|
||||
|
||||
@@ -1,20 +1,34 @@
|
||||
# Throwaway image for generating real MinIO on-disk fixtures without installing
|
||||
# a MinIO server on the host. Multi-stage: pull the official MinIO server binary
|
||||
# from the published image (linux, no dl.min.io download), then drop it into a
|
||||
# a MinIO server on the host. Multi-stage: extract the official MinIO server
|
||||
# binary from its checksum-pinned GitHub release package, then drop it into a
|
||||
# small Python image that runs the fixture lab (lab.py needs only python3 +
|
||||
# openssl; it drives MinIO's S3 API directly, so no `mc` is required).
|
||||
#
|
||||
# The MinIO release is pinned so the captured fixture format is reproducible;
|
||||
# this is the release the interop tests were validated against.
|
||||
#
|
||||
# Both base images are build args so a network that cannot reach Docker Hub can
|
||||
# point them at a mirror carrying the same content — quay.io publishes the MinIO
|
||||
# releases, and public.ecr.aws mirrors the official Python images. CI keeps the
|
||||
# Docker Hub defaults. Override with:
|
||||
# --build-arg MINIO_IMAGE=quay.io/minio/minio:RELEASE.2025-09-07T16-13-09Z \
|
||||
# --build-arg PYTHON_IMAGE=public.ecr.aws/docker/library/python:3.12-slim
|
||||
ARG MINIO_IMAGE=minio/minio:RELEASE.2025-09-07T16-13-09Z
|
||||
# MINIO_IMAGE can override the package stage with an accessible image carrying
|
||||
# the same release at /usr/bin/minio. PYTHON_IMAGE can point at a Python mirror.
|
||||
ARG MINIO_IMAGE=minio-release
|
||||
ARG PYTHON_IMAGE=python:3.12-slim
|
||||
|
||||
FROM scratch AS minio-package-amd64
|
||||
ADD --checksum=sha256:eeda08f699f6592d1b868ac8bda864ae2cacdb5ee1b888663366e8c8ff566249 https://github.com/minio/minio/releases/download/RELEASE.2025-09-07T16-13-09Z/minio_20250907161309.0.0_amd64.deb /minio.deb
|
||||
|
||||
FROM scratch AS minio-package-arm64
|
||||
ADD --checksum=sha256:f9466ed832ac4926c8870b3b8378d77861a87b41529fecae7789b9bc54acd78b https://github.com/minio/minio/releases/download/RELEASE.2025-09-07T16-13-09Z/minio_20250907161309.0.0_arm64.deb /minio.deb
|
||||
|
||||
FROM scratch AS minio-package-ppc64le
|
||||
ADD --checksum=sha256:39eec77281c552dfa0a92b5221c657dc58306ea8c4e92b79e9562cee61b59caa https://github.com/minio/minio/releases/download/RELEASE.2025-09-07T16-13-09Z/minio_20250907161309.0.0_ppc64el.deb /minio.deb
|
||||
|
||||
FROM minio-package-${TARGETARCH} AS minio-package
|
||||
|
||||
FROM ${PYTHON_IMAGE} AS minio-release
|
||||
COPY --from=minio-package /minio.deb /tmp/minio.deb
|
||||
RUN dpkg-deb --extract /tmp/minio.deb /tmp/minio-package \
|
||||
&& install -D /tmp/minio-package/usr/local/bin/minio /usr/bin/minio \
|
||||
&& rm -rf /tmp/minio.deb /tmp/minio-package
|
||||
|
||||
FROM ${MINIO_IMAGE} AS minio
|
||||
|
||||
FROM ${PYTHON_IMAGE}
|
||||
|
||||
@@ -24,15 +24,14 @@ Use the automated path when you want the lab to:
|
||||
|
||||
## Networks without Docker Hub access
|
||||
|
||||
`capture_via_docker.sh` pulls its two base images from Docker Hub by default. Where that registry is unreachable, point the build at mirrors carrying the same content — quay.io publishes the MinIO releases and public.ecr.aws mirrors the official Python images:
|
||||
`capture_via_docker.sh` extracts MinIO from the official GitHub release package after verifying its pinned SHA256. The Dockerfile selects the matching amd64, arm64, or ppc64le package; the MinIO release remains `RELEASE.2025-09-07T16-13-09Z`. Python comes from Docker Hub by default. Where that registry is unreachable, public.ecr.aws mirrors the official Python images:
|
||||
|
||||
```bash
|
||||
MINIO_LAB_MINIO_IMAGE=quay.io/minio/minio:RELEASE.2025-09-07T16-13-09Z \
|
||||
MINIO_LAB_PYTHON_IMAGE=public.ecr.aws/docker/library/python:3.12-slim \
|
||||
./capture_via_docker.sh
|
||||
```
|
||||
|
||||
Pin the MinIO tag to the same release the Dockerfile names; an unpinned `:latest` captures whatever format that day's build writes, which is not what the interop tests were validated against.
|
||||
`MINIO_LAB_MINIO_IMAGE` can override the package stage with an accessible image containing the same MinIO release at `/usr/bin/minio`. Pin that image to the same release; an unpinned `:latest` captures whatever format that day's build writes, which is not what the interop tests were validated against.
|
||||
|
||||
## Layout
|
||||
|
||||
|
||||
@@ -34,9 +34,9 @@ if [ "${cases[0]}" != "all" ]; then
|
||||
done
|
||||
fi
|
||||
|
||||
# Base images are overridable so a network without Docker Hub access can point
|
||||
# them at a mirror (see the Dockerfile header). Unset by default, which keeps the
|
||||
# Dockerfile's Docker Hub defaults for CI.
|
||||
# The Dockerfile defaults to a checksum-pinned MinIO release package and a
|
||||
# Python base image. Overrides can select accessible images carrying the same
|
||||
# content (see the Dockerfile header).
|
||||
build_args=()
|
||||
if [ -n "${MINIO_LAB_MINIO_IMAGE:-}" ]; then
|
||||
build_args+=(--build-arg "MINIO_IMAGE=${MINIO_LAB_MINIO_IMAGE}")
|
||||
@@ -49,7 +49,8 @@ echo ">> building ${IMAGE}"
|
||||
docker build -f "${SCRIPT_DIR}/Dockerfile" -t "${IMAGE}" "${build_args[@]}" "${SCRIPT_DIR}"
|
||||
|
||||
echo ">> capturing fixtures into ${FIXTURE_REL}"
|
||||
docker run --rm -v "${REPO_ROOT}:/repo" "${IMAGE}" \
|
||||
# The host reader initializes disk metadata inside the captured fixture tree.
|
||||
docker run --rm --user "$(id -u):$(id -g)" -v "${REPO_ROOT}:/repo" "${IMAGE}" \
|
||||
python3 "/repo/${LAB_REL}/lab.py" capture-matrix \
|
||||
--root "/repo/${FIXTURE_REL}" \
|
||||
--work-root /tmp/minio-lab-work \
|
||||
|
||||
@@ -1161,7 +1161,8 @@ async fn execute_top_disk_job(
|
||||
},
|
||||
},
|
||||
limits: TopCaptureLimits {
|
||||
max_duration_millis: envelope.parameters.duration_millis,
|
||||
// Measured sampling time includes scheduling overhead within the signed job timeout.
|
||||
max_duration_millis: envelope.limits.timeout_seconds * 1_000,
|
||||
max_working_memory_bytes: envelope.limits.max_memory_bytes,
|
||||
max_cpu_millis: envelope.limits.max_cpu_millis,
|
||||
..TopCaptureLimits::default()
|
||||
@@ -1830,6 +1831,10 @@ mod tests {
|
||||
let result: serde_json::Value = serde_json::from_reader(archive.by_name("result.json").unwrap()).unwrap();
|
||||
assert_eq!(result["toolId"], "top.disk");
|
||||
assert_eq!(result["data"]["resourceAlias"], "resource-1");
|
||||
let duration_millis = result["durationMillis"].as_u64().unwrap();
|
||||
assert!(duration_millis >= top.parameters.duration_millis);
|
||||
assert!(duration_millis <= top.limits.timeout_seconds * 1_000);
|
||||
assert_eq!(result["data"]["windowMillis"], duration_millis);
|
||||
assert!(result["data"]["writeBytes"].as_u64().unwrap() >= 65_536);
|
||||
assert!(result["data"]["ioCount"].as_u64().unwrap() > 0);
|
||||
use std::io::Read as _;
|
||||
|
||||
@@ -2406,6 +2406,7 @@ mod tests {
|
||||
read_quorum_ready: true,
|
||||
write_quorum_ready: false,
|
||||
pool_metadata_write_ready: true,
|
||||
unavailable_drives: Vec::new(),
|
||||
},
|
||||
true,
|
||||
true,
|
||||
@@ -2432,6 +2433,7 @@ mod tests {
|
||||
read_quorum_ready: false,
|
||||
write_quorum_ready: false,
|
||||
pool_metadata_write_ready: false,
|
||||
unavailable_drives: Vec::new(),
|
||||
},
|
||||
false,
|
||||
true,
|
||||
|
||||
@@ -649,7 +649,6 @@ fn legacy_capabilities(job_capable: bool, service_memory: bool) -> Vec<&'static
|
||||
if job_capable {
|
||||
capabilities.push("jobs");
|
||||
}
|
||||
capabilities.push("health.check.service@1");
|
||||
// Freeze the preceding release, independently of today's advertisement.
|
||||
capabilities.extend([
|
||||
"performance.client@1",
|
||||
|
||||
@@ -612,7 +612,12 @@ async fn real_rustfs_endpoint_and_production_cli_support_bounded_get_and_put_bod
|
||||
.expect("CLI task")
|
||||
.expect("run production rustfs binary");
|
||||
|
||||
assert!(result.status.success(), "stderr: {}", String::from_utf8_lossy(&result.stderr));
|
||||
assert!(
|
||||
result.status.success(),
|
||||
"stdout: {}\nstderr: {}",
|
||||
String::from_utf8_lossy(&result.stdout),
|
||||
String::from_utf8_lossy(&result.stderr)
|
||||
);
|
||||
let stdout = String::from_utf8(result.stdout).expect("UTF-8 stdout");
|
||||
assert!(stdout.contains("tool=performance.object outcome=SUCCEEDED reason=COMPLETE\n"));
|
||||
assert!(stdout.contains("upload=not-performed\n"));
|
||||
|
||||
@@ -15,8 +15,8 @@
|
||||
use std::time::Duration;
|
||||
|
||||
use rustfs::connect::diagnostics::{
|
||||
DiskCounterSnapshot, LocalTopConsent, MAX_SAFE_INTEGER, TopCaptureLimits, TopCaptureRequest, TopCaptureScope, TopOutcome,
|
||||
TopReasonCode, evaluate_disk_window,
|
||||
DiskCounterSnapshot, LocalTopConsent, MAX_SAFE_INTEGER, TopCaptureError, TopCaptureLimits, TopCaptureRequest,
|
||||
TopCaptureScope, TopOutcome, TopReasonCode, evaluate_disk_window,
|
||||
};
|
||||
use time::OffsetDateTime;
|
||||
|
||||
@@ -104,3 +104,32 @@ fn top_disk_counter_reset_and_safe_integer_overflow_fail_without_data() {
|
||||
assert!(result.data.is_none());
|
||||
}
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn top_disk_measured_window_obeys_the_duration_limit() {
|
||||
let mut request = request();
|
||||
request.window = Duration::from_millis(1_000);
|
||||
request.limits.max_duration_millis = 2_000;
|
||||
let before = DiskCounterSnapshot {
|
||||
read_bytes: 100,
|
||||
write_bytes: 200,
|
||||
io_count: 10,
|
||||
};
|
||||
let after = DiskCounterSnapshot {
|
||||
read_bytes: 110,
|
||||
write_bytes: 220,
|
||||
io_count: 12,
|
||||
};
|
||||
|
||||
for window_millis in [1_001, 2_000] {
|
||||
let result = evaluate_disk_window(&request, before, after, window_millis).expect("window within limit");
|
||||
assert_eq!(result.duration_millis, window_millis);
|
||||
assert_eq!(result.data.expect("disk counters").window_millis, window_millis);
|
||||
}
|
||||
for window_millis in [0, 2_001] {
|
||||
assert!(matches!(
|
||||
evaluate_disk_window(&request, before, after, window_millis),
|
||||
Err(TopCaptureError::Limits)
|
||||
));
|
||||
}
|
||||
}
|
||||
|
||||
Reference in New Issue
Block a user