mirror of
https://github.com/rustfs/rustfs.git
synced 2026-08-29 16:37:07 +00:00
2e6c820f53
* test(heal): relative disk target and fail fast on terminal-but-short The absolute 40 GiB heal target was calibrated to the background scanner (auto-heal), which is now disabled for determinism; with only the explicit heal the recovered node lands at ~36 GiB for 40 GiB survivors. Make the success criterion relative: the outage node must reach at least 90% of the least-used surviving node (absolute HEAL_TARGET_GB floor optional, default 0 = relative only). Also fail fast when the heal task reaches a terminal success but the disk target is not met (previously the monitor kept polling until timeout), and drop the misleading 'progress absent' warning on the final (cleaned) task response — mid-run progress is reported correctly. Validated live: heal summary=finished, 0 failed, vm000/vm001=40GB, vm002=40GB (target 36GB), test PASSED. * test(heal): gate success on server verdict + data read-back, drop disk GB gate The per-node disk-usage target (40 GiB / 90% of survivors) is not a code-level invariant: EC distributes different shards per node, so the final GB per node depends on the layout, not on heal correctness. Gate the test on what the server actually verifies: - Heal task terminal success (finished/completed) with objectsFailed == 0 (the server's per-object scan/repair verdict). - S3 read-back verification: list the test bucket and GET a sample of objects, requiring HTTP 200 for every read (end-to-end proof the data is still reconstructable after repair). The GET uses a discard mode so binary bodies are not captured (no null-byte warnings / SIGPIPE). Per-node disk usage stays in the output as observability (with a warning if the outage node gained no usage), not as the pass/fail gate. Removes the heal_target_gb input and the relative-target logic. Validated live: heal summary=finished, 0 failed, 20/20 objects read back, vm002_used=40GB, PASS.
122 lines
4.1 KiB
YAML
122 lines
4.1 KiB
YAML
name: RustFS Heal Test
|
|
|
|
on:
|
|
workflow_dispatch:
|
|
inputs:
|
|
package_url:
|
|
description: 'Direct .deb URL (nightly/R2). Defaults to the latest nightly deb.'
|
|
required: false
|
|
type: string
|
|
stop_node_gb:
|
|
description: 'Stop the outage node when surviving nodes reach N GiB'
|
|
required: false
|
|
default: '15'
|
|
warp_stop_gb:
|
|
description: 'Stop warp when surviving nodes reach N GiB'
|
|
required: false
|
|
default: '40'
|
|
cleanup_before:
|
|
description: 'Reset the nodes before the test (DESTROYS existing data/config)'
|
|
type: boolean
|
|
default: true
|
|
cleanup_after:
|
|
description: 'Reset the nodes after the test (DESTROYS test data/config)'
|
|
type: boolean
|
|
default: true
|
|
|
|
permissions:
|
|
contents: read
|
|
|
|
# Only one test at a time: both this and the pool-expansion workflow mutate
|
|
# the same test environment, so they share one concurrency group.
|
|
concurrency:
|
|
group: rustfs-pool-expansion-test
|
|
cancel-in-progress: false
|
|
|
|
defaults:
|
|
run:
|
|
shell: bash
|
|
|
|
env:
|
|
RUSTFS_ACCESS_KEY: ${{ secrets.RUSTFS_ACCESS_KEY }}
|
|
RUSTFS_SECRET_KEY: ${{ secrets.RUSTFS_SECRET_KEY }}
|
|
RUSTFS_API_ENDPOINT: ${{ secrets.RUSTFS_API_ENDPOINT || vars.RUSTFS_API_ENDPOINT || vars.RUSTFS_RC_ENDPOINT }}
|
|
RUSTFS_NODES: ${{ secrets.RUSTFS_NODES || vars.RUSTFS_NODES }}
|
|
RUSTFS_SSH_USER: ${{ secrets.RUSTFS_SSH_USER || vars.RUSTFS_SSH_USER }}
|
|
RUSTFS_NIGHTLY_PACKAGE_URL: ${{ vars.RUSTFS_NIGHTLY_PACKAGE_URL || 'https://dl.rustfs.com/artifacts/rustfs/packages/nightly/rustfs-nightly-latest.deb' }}
|
|
|
|
jobs:
|
|
heal-test:
|
|
runs-on: smoke-testing
|
|
timeout-minutes: 480
|
|
steps:
|
|
- name: Checkout
|
|
uses: actions/checkout@9c091bb21b7c1c1d1991bb908d89e4e9dddfe3e0 # v7
|
|
with:
|
|
persist-credentials: false
|
|
|
|
- name: Show environment
|
|
run: |
|
|
uname -a
|
|
jq --version
|
|
openssl version
|
|
warp --version || true
|
|
df -h /data | tail -1
|
|
|
|
- name: Reset test environment (before)
|
|
if: ${{ inputs.cleanup_before != 'false' }}
|
|
run: |
|
|
chmod +x scripts/test/rustfs_heal_test.sh
|
|
./scripts/test/rustfs_heal_test.sh --reset -y
|
|
|
|
- name: Install RustFS package & start cluster
|
|
run: |
|
|
ARGS=(--steps "1,2" -y --endpoint "${{ env.RUSTFS_API_ENDPOINT }}")
|
|
if [ -n "${{ inputs.package_url }}" ]; then
|
|
ARGS+=(--package-url "${{ inputs.package_url }}")
|
|
else
|
|
ARGS+=(--package-url "${{ env.RUSTFS_NIGHTLY_PACKAGE_URL }}")
|
|
fi
|
|
./scripts/test/rustfs_heal_test.sh "${ARGS[@]}"
|
|
|
|
- name: Preflight checks
|
|
run: |
|
|
ARGS=(--preflight --endpoint "${{ env.RUSTFS_API_ENDPOINT }}")
|
|
if [ -n "${{ inputs.package_url }}" ]; then
|
|
ARGS+=(--package-url "${{ inputs.package_url }}")
|
|
else
|
|
ARGS+=(--package-url "${{ env.RUSTFS_NIGHTLY_PACKAGE_URL }}")
|
|
fi
|
|
./scripts/test/rustfs_heal_test.sh "${ARGS[@]}"
|
|
|
|
- name: Run heal test (write -> outage -> heal -> verify)
|
|
run: |
|
|
./scripts/test/rustfs_heal_test.sh \
|
|
--steps "3,4,5,6,7" -y \
|
|
--endpoint "${{ env.RUSTFS_API_ENDPOINT }}" \
|
|
--stop-node-gb "${{ inputs.stop_node_gb }}" \
|
|
--warp-stop-gb "${{ inputs.warp_stop_gb }}" \
|
|
--log-file /tmp/rustfs-heal-test.log
|
|
|
|
- name: Upload test logs
|
|
if: always()
|
|
uses: actions/upload-artifact@b7c566a772e6b6bfb58ed0dc250532a479d7789f # v6
|
|
with:
|
|
name: rustfs-heal-test-${{ github.run_id }}
|
|
path: |
|
|
/tmp/rustfs-heal-test*.log
|
|
/tmp/rustfs-warp.*.log
|
|
if-no-files-found: warn
|
|
|
|
- name: Reset test environment (after)
|
|
if: ${{ always() && inputs.cleanup_after != 'false' }}
|
|
run: |
|
|
./scripts/test/rustfs_heal_test.sh --reset -y
|
|
|
|
- name: Notify on failure
|
|
if: failure()
|
|
run: |
|
|
echo "RustFS heal test failed"
|
|
echo "Package source: ${{ inputs.package_url || 'nightly (R2 latest)' }}"
|
|
echo "See the uploaded log artifact for details."
|