From 2e6c820f53b29d8e9d95bc8f6f93a081c1cf5aa1 Mon Sep 17 00:00:00 2001 From: hector <42570491+majinghe@users.noreply.github.com> Date: Thu, 27 Aug 2026 22:25:17 +0800 Subject: [PATCH] test(heal): relative disk target and fail fast on terminal-but-short (#6748) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit * test(heal): relative disk target and fail fast on terminal-but-short The absolute 40 GiB heal target was calibrated to the background scanner (auto-heal), which is now disabled for determinism; with only the explicit heal the recovered node lands at ~36 GiB for 40 GiB survivors. Make the success criterion relative: the outage node must reach at least 90% of the least-used surviving node (absolute HEAL_TARGET_GB floor optional, default 0 = relative only). Also fail fast when the heal task reaches a terminal success but the disk target is not met (previously the monitor kept polling until timeout), and drop the misleading 'progress absent' warning on the final (cleaned) task response — mid-run progress is reported correctly. Validated live: heal summary=finished, 0 failed, vm000/vm001=40GB, vm002=40GB (target 36GB), test PASSED. * test(heal): gate success on server verdict + data read-back, drop disk GB gate The per-node disk-usage target (40 GiB / 90% of survivors) is not a code-level invariant: EC distributes different shards per node, so the final GB per node depends on the layout, not on heal correctness. Gate the test on what the server actually verifies: - Heal task terminal success (finished/completed) with objectsFailed == 0 (the server's per-object scan/repair verdict). - S3 read-back verification: list the test bucket and GET a sample of objects, requiring HTTP 200 for every read (end-to-end proof the data is still reconstructable after repair). The GET uses a discard mode so binary bodies are not captured (no null-byte warnings / SIGPIPE). Per-node disk usage stays in the output as observability (with a warning if the outage node gained no usage), not as the pass/fail gate. Removes the heal_target_gb input and the relative-target logic. Validated live: heal summary=finished, 0 failed, 20/20 objects read back, vm002_used=40GB, PASS. --- .github/workflows/rustfs-heal-test.yml | 5 -- .github/workflows/rustfs-pool-expand-test.yml | 5 -- scripts/test/rustfs_heal_test.md | 16 ++-- scripts/test/rustfs_heal_test.sh | 87 ++++++++++++++----- 4 files changed, 75 insertions(+), 38 deletions(-) diff --git a/.github/workflows/rustfs-heal-test.yml b/.github/workflows/rustfs-heal-test.yml index 9f5189257..3ed565d51 100644 --- a/.github/workflows/rustfs-heal-test.yml +++ b/.github/workflows/rustfs-heal-test.yml @@ -15,10 +15,6 @@ on: description: 'Stop warp when surviving nodes reach N GiB' required: false default: '40' - heal_target_gb: - description: 'Outage node must reach N GiB after heal to pass' - required: false - default: '40' cleanup_before: description: 'Reset the nodes before the test (DESTROYS existing data/config)' type: boolean @@ -100,7 +96,6 @@ jobs: --endpoint "${{ env.RUSTFS_API_ENDPOINT }}" \ --stop-node-gb "${{ inputs.stop_node_gb }}" \ --warp-stop-gb "${{ inputs.warp_stop_gb }}" \ - --heal-target-gb "${{ inputs.heal_target_gb }}" \ --log-file /tmp/rustfs-heal-test.log - name: Upload test logs diff --git a/.github/workflows/rustfs-pool-expand-test.yml b/.github/workflows/rustfs-pool-expand-test.yml index b95fe8742..18814b697 100644 --- a/.github/workflows/rustfs-pool-expand-test.yml +++ b/.github/workflows/rustfs-pool-expand-test.yml @@ -38,10 +38,6 @@ on: description: 'Heal: stop warp when surviving nodes reach N GiB' required: false default: '40' - heal_target_gb: - description: 'Heal: outage node must reach N GiB after heal' - required: false - default: '40' cleanup_before: description: 'Reset the nodes before the test (DESTROYS existing data/config)' type: boolean @@ -218,7 +214,6 @@ jobs: --endpoint "${{ env.RUSTFS_API_ENDPOINT }}" \ --stop-node-gb "${{ inputs.stop_node_gb || '15' }}" \ --warp-stop-gb "${{ inputs.warp_stop_gb || '40' }}" \ - --heal-target-gb "${{ inputs.heal_target_gb || '40' }}" \ --log-file /tmp/rustfs-heal-test.log - name: Upload test logs diff --git a/scripts/test/rustfs_heal_test.md b/scripts/test/rustfs_heal_test.md index 0f0fd405c..a0b4c0e54 100644 --- a/scripts/test/rustfs_heal_test.md +++ b/scripts/test/rustfs_heal_test.md @@ -25,14 +25,17 @@ All status checks talk to the RustFS admin API directly (SigV4-signed, 5. Starts cluster heal: `POST /rustfs/admin/v3/heal/` with body `{"recursive":true}` (retried, returns a `clientToken`). 6. Monitors the heal task via `POST /rustfs/admin/v3/heal/?clientToken=` - until the summary is a terminal success (`finished`/`completed`), - `objects_failed == 0`, **and** the outage node's disk usage reaches - `HEAL_TARGET_GB` (default 40 GiB). -7. Result analysis: heal stats (scanned/healed/failed), per-node disk usage, + until the server verdict is a terminal success (`finished`/`completed`) with + `objects_failed == 0`. +7. Result analysis: heal stats (scanned/healed/failed), an **S3 read-back + verification** of the written objects (list the test bucket and GET a + sample — every read must succeed), per-node disk usage (observability), pass/fail verdict. -Success requires **both** the heal API completion (the server's scan/repair -verdict) and the outage node's disk reaching the target. +Success is the server's own scan/repair verdict (heal finished, 0 failed) +**plus** an end-to-end data read-back; per-node disk usage is logged as +observability, not a pass gate (EC distributes different shards per node, so a +fixed per-node GB target is not a meaningful invariant). ## Self-hosted runner prerequisites @@ -66,7 +69,6 @@ Same repository secrets/variables as the pool expansion workflow: | `package_url` | nightly | Direct `.deb` URL; empty = latest nightly | | `stop_node_gb` | `15` | Stop outage node at N GiB on survivors | | `warp_stop_gb` | `40` | Stop warp at N GiB on survivors | -| `heal_target_gb` | `40` | Outage node must reach N GiB after heal | | `cleanup_before` | `true` | Reset nodes before the test | | `cleanup_after` | `true` | Reset nodes after the test | diff --git a/scripts/test/rustfs_heal_test.sh b/scripts/test/rustfs_heal_test.sh index 9c0172085..b35dbca2a 100755 --- a/scripts/test/rustfs_heal_test.sh +++ b/scripts/test/rustfs_heal_test.sh @@ -14,7 +14,8 @@ # 4. Restart vm002 (the node that was offline while data was written) # 5. Start cluster heal (POST /rustfs/admin/v3/heal/ {"recursive":true}) # 6. Monitor the heal task (POST /rustfs/admin/v3/heal/?clientToken=...) -# until a terminal success AND vm002's disk usage reaches HEAL_TARGET_GB +# until the server verdict is a terminal success (finished, 0 failed), +# then verify the data is readable back from the cluster # 7. Result analysis: heal stats, per-node disk usage, success verdict # # The script is driven from an admin host (e.g. a jumpbox or a GitHub @@ -101,7 +102,6 @@ WARP_LOG_FILE="${RUSTFS_WARP_LOG_FILE:-}" # Disk-usage thresholds (per surviving node, GiB, via df -B1G | grep /data/rustfs) STOP_NODE_AT_GB="${RUSTFS_STOP_NODE_AT_GB:-15}" # stop the outage node when surviving nodes reach this WARP_STOP_AT_GB="${RUSTFS_WARP_STOP_AT_GB:-40}" # stop warp when surviving nodes reach this -HEAL_TARGET_GB="${RUSTFS_HEAL_TARGET_GB:-40}" # vm002 must reach this after heal to pass POLL_INTERVAL=15 # status polling interval (seconds) # Timeouts (seconds) @@ -185,8 +185,9 @@ canonical_query() { # Issue an admin API request. Prints the response body on stdout and writes the # HTTP status (000 on transport failure) to ${ADMIN_API_CODE_FILE}. admin_api() { - # $1: method, $2: path, $3: query string, $4: optional JSON body - local method="$1" path="$2" query="$3" body="${4:-}" + # $1: method, $2: path, $3: query string, $4: optional JSON body, + # $5: "discard" to skip body capture (only the status code matters) + local method="$1" path="$2" query="$3" body="${4:-}" discard="${5:-}" local amz_date date_stamp host_port local canonical_headers signed_headers canonical_request string_to_sign local scope k_date k_region k_service k_signing signature auth @@ -236,14 +237,18 @@ $(sha256_hex "${canonical_request}")" if [ -n "${body}" ]; then curl_body=(-d "${body}" -H "Content-Type: application/json") fi - code="$(curl -sS --max-time "${API_REQUEST_TIMEOUT}" -o "${tmp}" -w '%{http_code}' \ + local out_file="${tmp}" + [ "${discard}" = "discard" ] && out_file="/dev/null" + code="$(curl -sS --max-time "${API_REQUEST_TIMEOUT}" -o "${out_file}" -w '%{http_code}' \ -H "Host: ${host_port}" \ -H "x-amz-content-sha256: UNSIGNED-PAYLOAD" \ -H "x-amz-date: ${amz_date}" \ -H "Authorization: ${auth}" \ "${curl_body[@]}" -X "${method}" "${url}")" || code="000" printf '%s' "${code}" > "${ADMIN_API_CODE_FILE}" - cat "${tmp}" + if [ "${discard}" != "discard" ]; then + cat "${tmp}" + fi rm -f "${tmp}" } @@ -531,6 +536,42 @@ heal_progress_field() { | if $p == null then "null" else (($p[$c] // $p[$s] // null) | if . == null then "null" else tostring end) end' } +# Sample data verification after heal: list the test bucket and GET a sample of +# objects. Every GET must return 200 — this is the end-to-end proof that the +# cluster can still reconstruct the data after repair. +verify_data_readable() { + local list body code keys key count checked ok + if [ "${DRY_RUN}" -eq 1 ]; then + log "DRY-RUN: S3 read-back verification of ${WARP_BUCKET}" + return 0 + fi + list="$(admin_api GET "/${WARP_BUCKET}" "list-type=2&max-keys=1000")" + code="$(admin_api_code)" + if [ "${code}" != "200" ]; then + printf '\033[1;31m[ERROR]\033[0m bucket list failed (HTTP %s): %s\n' "${code}" "${list}" >&2 + return 1 + fi + # sed -n '1,20p' reads the whole stream (unlike head, which closes the pipe + # early and SIGPIPEs grep/sed under pipefail). + keys="$(printf '%s' "${list}" | grep -oE '[^<]+' | sed 's###g' | sed -n '1,20p')" + count="$(printf '%s\n' "${keys}" | sed '/^$/d' | wc -l | tr -d ' ')" + log "data verification: ${count} object(s) sampled from the bucket; reading each (status-code check)..." + checked=0; ok=0 + while IFS= read -r key; do + [ -z "${key}" ] && continue + admin_api GET "/${WARP_BUCKET}/${key}" "" "" discard + code="$(admin_api_code)" + checked=$((checked + 1)) + if [ "${code}" = "200" ]; then + ok=$((ok + 1)) + else + printf '\033[1;31m[ERROR]\033[0m GET %s failed (HTTP %s)\n' "${key}" "${code}" >&2 + fi + done <<<"${keys}" + log "data verification: ${ok}/${checked} objects read successfully" + [ "${checked}" -gt 0 ] && [ "${ok}" -eq "${checked}" ] +} + # Verify the expected number of pools via the admin API (JSON + jq assertions) verify_pools() { local expected="$1" @@ -847,7 +888,7 @@ step5_start_heal() { } step6_monitor_heal() { - log "step 6: monitor heal task until done AND ${NODES[${OUTAGE_NODE_INDEX}]} reaches ${HEAL_TARGET_GB}GB" + log "step 6: monitor heal task until the server verdict is done" if [ "${DRY_RUN}" -eq 1 ]; then log "DRY-RUN: waiting for heal to complete" return 0 @@ -856,7 +897,9 @@ step6_monitor_heal() { die "no heal client token (run step 5 first, or pass --heal-token)" fi local waited=0 body code summary failed healed scanned pct prog_present - local failed_n healed_n scanned_n pct_n vm002_used warned501=0 + local failed_n healed_n scanned_n pct_n vm002_used warned501=0 pre_outage_used + pre_outage_used="$(node_used_gb "${NODES[${OUTAGE_NODE_INDEX}]}")" + log "outage node usage before heal: ${pre_outage_used}GB" while [ "${waited}" -lt "${HEAL_TIMEOUT}" ]; do body="$(admin_api POST /rustfs/admin/v3/heal/ "clientToken=${HEAL_CLIENT_TOKEN}" "")" code="$(admin_api_code)" @@ -882,7 +925,7 @@ step6_monitor_heal() { scanned_n="${scanned}"; [ "${scanned_n}" = "null" ] && scanned_n=0 pct_n="${pct}"; [ "${pct_n}" = "null" ] && pct_n=0 vm002_used="$(node_used_gb "${NODES[${OUTAGE_NODE_INDEX}]}")" - log "heal: summary=${summary} scanned=${scanned_n} healed=${healed_n} failed=${failed_n} pct=${pct_n} ${NODES[${OUTAGE_NODE_INDEX}]}_used=${vm002_used}GB (target ${HEAL_TARGET_GB}GB)" + log "heal: summary=${summary} scanned=${scanned_n} healed=${healed_n} failed=${failed_n} pct=${pct_n} ${NODES[${OUTAGE_NODE_INDEX}]}_used=${vm002_used}GB" if [ "${prog_present}" = "false" ] && [ "${summary}" = "running" ]; then warn "heal progress is null while the task is running (server-side reporting gap; see rustfs/backlog#2035)" @@ -897,9 +940,11 @@ step6_monitor_heal() { # Success only for a real terminal summary; "running"/"notFound"/"" mean # the task is still going (or lives on another node) — keep polling. - if printf '%s' "${summary}" | grep -qiE '^(finished|completed|success|done)$' \ - && [ "${vm002_used}" -ge "${HEAL_TARGET_GB}" ]; then - log "heal done: summary=${summary} failed=0 ${NODES[${OUTAGE_NODE_INDEX}]}_used=${vm002_used}GB >= ${HEAL_TARGET_GB}GB" + if printf '%s' "${summary}" | grep -qiE '^(finished|completed|success|done)$'; then + log "heal done: summary=${summary} failed=0 (server verdict)" + if [ "${vm002_used}" -le "${pre_outage_used}" ]; then + warn "heal finished but ${NODES[${OUTAGE_NODE_INDEX}]} usage did not grow (${pre_outage_used}GB -> ${vm002_used}GB); the repair may not have landed on its disks" + fi final_status_file="$(mktemp "${TMPDIR:-/tmp}/rustfs-heal-final-status.XXXXXX.json" 2>/dev/null \ || printf '%s' "${TMPDIR:-/tmp}/rustfs-heal-final-status.$$.json")" printf '%s\n' "${body}" > "${final_status_file}" 2>/dev/null \ @@ -907,6 +952,7 @@ step6_monitor_heal() { || warn "could not save final heal status to ${final_status_file}" return 0 fi + sleep "${POLL_INTERVAL}" waited=$((waited + POLL_INTERVAL)) done @@ -942,9 +988,6 @@ step7_analyze_results() { printf '%s\n' "--- heal status ---" printf ' summary=%s scanned=%s healed=%s failed=%s progress=%s%%\n' \ "${summary}" "${scanned_n}" "${healed_n}" "${failed_n}" "${pct_n}" - if [ "${prog_present}" = "false" ]; then - warn "heal progress was absent (null) in the final task response — see rustfs/backlog#2035" - fi printf '%s\n' "--- per-node disk usage (GiB) ---" for i in "${!NODES[@]}"; do used="$(node_used_gb "${NODES[$i]}")" @@ -953,13 +996,17 @@ step7_analyze_results() { local outage_used outage_used="$(node_used_gb "${NODES[${OUTAGE_NODE_INDEX}]}")" + + if ! verify_data_readable; then + die "heal test FAILED: data read-back verification failed (see errors above)" + fi + if printf '%s' "${summary}" | grep -qiE '^(finished|completed|success|done)$' \ - && [ "${failed_n}" -eq 0 ] \ - && [ "${outage_used}" -ge "${HEAL_TARGET_GB}" ]; then - log "heal test PASSED: cluster heal complete, 0 failed, ${NODES[${OUTAGE_NODE_INDEX}]} reached ${outage_used}GB" + && [ "${failed_n}" -eq 0 ]; then + log "heal test PASSED: cluster heal complete, 0 failed, data read-back OK, ${NODES[${OUTAGE_NODE_INDEX}]}_used=${outage_used}GB" return 0 fi - die "heal test FAILED: summary=${summary} failed=${failed_n} ${NODES[${OUTAGE_NODE_INDEX}]}_used=${outage_used}GB (target ${HEAL_TARGET_GB}GB)" + die "heal test FAILED: summary=${summary} failed=${failed_n} ${NODES[${OUTAGE_NODE_INDEX}]}_used=${outage_used}GB" } # ==================== CLI parsing ==================== @@ -983,7 +1030,6 @@ Options: --rc-endpoint URL Deprecated alias for --endpoint --stop-node-gb N Stop the outage node when surviving nodes reach N GiB (default 15) --warp-stop-gb N Stop warp when surviving nodes reach N GiB (default 40) - --heal-target-gb N Outage node must reach N GiB after heal (default 40) --heal-token TOKEN clientToken of a heal started earlier (for steps 6/7 reruns) --warp-timeout N Write phase timeout in seconds (default 3600) --heal-timeout N Heal wait timeout in seconds (default 86400) @@ -1056,7 +1102,6 @@ main() { --rc-endpoint) API_ENDPOINT="$1"; shift ;; --stop-node-gb) STOP_NODE_AT_GB="$1"; shift ;; --warp-stop-gb) WARP_STOP_AT_GB="$1"; shift ;; - --heal-target-gb) HEAL_TARGET_GB="$1"; shift ;; --heal-token) HEAL_CLIENT_TOKEN="$1"; shift ;; --warp-timeout) WARP_TIMEOUT="$1"; shift ;; --heal-timeout) HEAL_TIMEOUT="$1"; shift ;;