# Copyright 2024 RustFS Team # # Licensed under the Apache License, Version 2.0 (the "License"); # you may not use this file except in compliance with the License. # You may obtain a copy of the License at # # http://www.apache.org/licenses/LICENSE-2.0 # # Unless required by applicable law or agreed to in writing, software # distributed under the License is distributed on an "AS IS" BASIS, # WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. # See the License for the specific language governing permissions and # limitations under the License. # Warp A/B budget gate for the hotpath series (rustfs/backlog#935 HP-14, item 4). # # Two entry points, honestly scoped: # * schedule (nightly, on main): post-merge detection — catches a regression # within 24h of landing, not before merge. # * workflow_dispatch: an explicitly selected trusted ref. # The dispatch input can run the gate with --allow-regression so a deliberate # correctness cost (e.g. the #4221 fsync durability fix) is recorded, not # blocked (rustfs/backlog#935 correction 1). name: Performance A/B on: schedule: - cron: "31 6 * * *" # 06:31 UTC nightly, against main workflow_dispatch: inputs: duration: description: "warp duration per round" required: false default: "12s" type: string allow_regression: description: "Pass the gate despite a FAIL (deliberate tradeoff)" required: false default: false type: boolean permissions: actions: read contents: read env: CARGO_TERM_COLOR: always RUST_BACKTRACE: 1 jobs: warp-ab: name: Warp A/B budget gate runs-on: sm-standard-2 # A normal nightly restores the last successful binary and builds only the # candidate; daily access keeps that cache warm. A cache miss may build both # and needs room for the A/B run plus artifact and cache publication. timeout-minutes: 180 steps: - name: Checkout repository uses: actions/checkout@9c091bb21b7c1c1d1991bb908d89e4e9dddfe3e0 # v7 with: persist-credentials: false fetch-depth: 0 # baseline may be an earlier successful scheduled head - name: Setup Rust environment uses: ./.github/actions/setup with: rust-version: stable cache-shared-key: warp-ab-${{ hashFiles('**/Cargo.lock') }} cache-save-if: ${{ github.ref == 'refs/heads/main' }} - name: Install warp run: | set -euo pipefail WARP_VERSION="1.0.0" curl -fsSL "https://github.com/minio/warp/releases/download/v${WARP_VERSION}/warp_Linux_x86_64.tar.gz" \ | sudo tar -xz -C /usr/local/bin warp warp --version - name: Decide exemption id: exempt env: INPUT_ALLOW_REGRESSION: ${{ github.event.inputs.allow_regression }} run: | allow="false" if [[ "$INPUT_ALLOW_REGRESSION" == "true" ]]; then allow="true" fi echo "allow_regression=$allow" >> "$GITHUB_OUTPUT" # A failed regression run must keep comparing against the last known-good # scheduled head. Otherwise the next nightly would absorb the regression # into its baseline and turn green without a fix. - name: Find last successful scheduled baseline id: scheduled_baseline if: github.event_name == 'schedule' uses: actions/github-script@ed597411d8f924073f98dfc5c65a23a2325f34cd # v8 with: result-encoding: string script: | const { data } = await github.rest.actions.listWorkflowRuns({ owner: context.repo.owner, repo: context.repo.repo, workflow_id: "performance-ab.yml", event: "schedule", status: "success", per_page: 1, }); return data.workflow_runs[0]?.head_sha ?? ""; # Manual runs compare a selected ref with current main. Scheduled runs # compare current main with the last successful scheduled head. With no # history, the first run measures the candidate against itself and seeds # that head only if the complete rig succeeds. - name: Resolve baseline / candidate commits id: commits env: SCHEDULED_BASELINE_SHA: ${{ steps.scheduled_baseline.outputs.result }} run: | set -euo pipefail candidate_sha="$(git rev-parse HEAD)" if [[ "${{ github.event_name }}" == "schedule" ]]; then baseline_sha="${SCHEDULED_BASELINE_SHA:-$candidate_sha}" if ! git merge-base --is-ancestor "$baseline_sha" "$candidate_sha"; then echo "::error::scheduled baseline $baseline_sha is not an ancestor of candidate $candidate_sha" >&2 exit 1 fi else baseline_sha="$(git rev-parse origin/main)" fi git cat-file -e "${baseline_sha}^{commit}" echo "baseline_sha=$baseline_sha" >> "$GITHUB_OUTPUT" echo "candidate_sha=$candidate_sha" >> "$GITHUB_OUTPUT" echo "baseline commit: $baseline_sha" echo "candidate commit: $candidate_sha" # Exact-key restore of the candidate binary saved by its successful # scheduled run. A miss leaves cache-hit unset and falls back to a source # build of that known-good head. - name: Restore cached baseline binary id: baseline_cache uses: actions/cache/restore@55cc8345863c7cc4c66a329aec7e433d2d1c52a9 # v6 with: path: baseline-bin/rustfs key: rustfs-baseline-${{ steps.commits.outputs.baseline_sha }} # Self-heal: on a nightly/dispatch run where the candidate commit IS the # baseline commit, a cache miss would make the rig build the same commit # twice (~65min per side with the post-#4806 LTO profile — no job budget # fits that). Build it once here, reuse it for both phases, and save it # back to the cache so the next run hits. - name: Build baseline on cache miss (same-commit self-heal) id: selfheal if: >- steps.baseline_cache.outputs.cache-hit != 'true' && steps.commits.outputs.baseline_sha == steps.commits.outputs.candidate_sha run: | set -euo pipefail cargo build --release --bin rustfs mkdir -p baseline-bin cp target/release/rustfs baseline-bin/rustfs echo "built=true" >> "$GITHUB_OUTPUT" - name: Build baseline on cache miss (different candidate) id: baseline_build if: >- steps.baseline_cache.outputs.cache-hit != 'true' && steps.commits.outputs.baseline_sha != steps.commits.outputs.candidate_sha run: | set -euo pipefail baseline_root="$RUNNER_TEMP/rustfs-baseline-${{ github.run_id }}" baseline_target="$RUNNER_TEMP/rustfs-baseline-target-${{ github.run_id }}" git worktree add --detach "$baseline_root" "${{ steps.commits.outputs.baseline_sha }}" cargo build --release --manifest-path "$baseline_root/Cargo.toml" --bin rustfs --target-dir "$baseline_target" mkdir -p baseline-bin cp "$baseline_target/release/rustfs" baseline-bin/rustfs git worktree remove --force "$baseline_root" echo "built=true" >> "$GITHUB_OUTPUT" - name: Build candidate binary id: candidate_build if: steps.commits.outputs.baseline_sha != steps.commits.outputs.candidate_sha run: | set -euo pipefail cargo build --release --bin rustfs mkdir -p candidate-bin cp target/release/rustfs candidate-bin/rustfs echo "built=true" >> "$GITHUB_OUTPUT" - name: Save self-healed baseline to cache if: steps.selfheal.outputs.built == 'true' || steps.baseline_build.outputs.built == 'true' uses: actions/cache/save@55cc8345863c7cc4c66a329aec7e433d2d1c52a9 # v6 with: path: baseline-bin/rustfs key: rustfs-baseline-${{ steps.commits.outputs.baseline_sha }} - name: Run warp A/B and gate id: ab env: INPUT_DURATION: ${{ github.event.inputs.duration }} run: | set -euo pipefail # The formal runner executes A1 baseline -> B1 candidate -> B2 candidate # -> A2 baseline for each workload and drive-sync cell. It requires three # rounds per leg to emit tail latency and error-rate evidence. # --health-timeout 180 outlasts the server's own 120s startup-readiness # budget, which the rig's previous 60s health poll undershot (the first # two nightly failures). perf-6 recalibrates these once the noise study # lands. duration="${INPUT_DURATION:-12s}" baseline_sha="${{ steps.commits.outputs.baseline_sha }}" candidate_sha="${{ steps.commits.outputs.candidate_sha }}" baseline_hit="${{ steps.baseline_cache.outputs.cache-hit }}" selfheal_built="${{ steps.selfheal.outputs.built }}" baseline_built="${{ steps.baseline_build.outputs.built }}" candidate_built="${{ steps.candidate_build.outputs.built }}" args=(--duration "$duration" --rounds 3 --cooldown 5 --health-timeout 180 --baseline-revision "$baseline_sha" --candidate-revision "$candidate_sha") if [[ "$baseline_hit" == "true" || "$selfheal_built" == "true" || "$baseline_built" == "true" ]]; then chmod +x baseline-bin/rustfs base_bin="$PWD/baseline-bin/rustfs" args+=(--baseline-bin "$base_bin") if [[ "$baseline_hit" == "true" ]]; then base_src="actions-cache (rustfs-baseline-$baseline_sha)" elif [[ "$selfheal_built" == "true" ]]; then base_src="source build (cache self-heal, saved as rustfs-baseline-$baseline_sha)" else base_src="isolated baseline source build (saved as rustfs-baseline-$baseline_sha)" fi if [[ "$candidate_sha" == "$baseline_sha" ]]; then # No commits landed since the last successful baseline, so reuse # the one binary for both phases and measure only rig drift. args+=(--candidate-bin "$base_bin") cand_src="same binary as baseline (same commit)" elif [[ "$candidate_built" == "true" ]]; then chmod +x candidate-bin/rustfs args+=(--candidate-bin "$PWD/candidate-bin/rustfs") cand_src="source build of the checked-out ref" else echo "::error::candidate binary was not built" >&2 exit 2 fi else echo "::error::baseline binary was not restored or built" >&2 exit 2 fi echo "baseline binary: $base_src" echo "candidate binary: $cand_src" if [[ "${{ steps.exempt.outputs.allow_regression }}" == "true" ]]; then args+=(--allow-regression --exemption-reason "workflow dispatch override") fi # Do not let a gate FAIL abort the job here; capture status and surface # it after the step summary is written. set +e bash scripts/run_hotpath_warp_abba.sh "${args[@]}" echo "status=$?" >> "$GITHUB_OUTPUT" set -e # Locate the newest run dir + gate.md for the summary/comment/artifact # steps. On a startup failure there is no gate.md, but the run dir still # holds server-logs/ for diagnosis. # Run dirs are UTC-timestamp names (no special chars); ls is safe here. # shellcheck disable=SC2012 run_dir="$(ls -td target/hotpath-abba/*/ 2>/dev/null | head -n1 || true)" echo "run_dir=${run_dir%/}" >> "$GITHUB_OUTPUT" # shellcheck disable=SC2012 gate_md="$(ls -t target/hotpath-abba/*/candidate_gate.md 2>/dev/null | head -n1 || true)" echo "gate_md=$gate_md" >> "$GITHUB_OUTPUT" - name: Upload A/B results if: always() uses: actions/upload-artifact@b7c566a772e6b6bfb58ed0dc250532a479d7789f # v6 with: name: hotpath-warp-ab-${{ github.run_number }} # Includes per-cell median_summary.csv / baseline_compare.csv, both gates, # and server-logs/ (rustfs.log + startup env per phase) so a failed run # is diagnosable. Short retention: this is churny nightly debug data. path: target/hotpath-abba/ if-no-files-found: warn retention-days: 14 - name: Write gate summary if: always() run: | set -euo pipefail status="${{ steps.ab.outputs.status }}" gate_md="${{ steps.ab.outputs.gate_md }}" run_dir="${{ steps.ab.outputs.run_dir }}" { echo "## Hotpath warp A/B — run ${{ github.run_number }}" echo if [[ "$status" == "0" ]]; then echo "Rig/gate exit: \`0\` (pass or warn)." else echo "Rig/gate exit: \`${status:-unknown}\` — **FAILED**." fi echo if [[ -n "$gate_md" && -f "$gate_md" ]]; then cat "$gate_md" else echo "No \`gate.md\` produced — the rig failed **before** the gate" echo "(most likely server startup / health). Failing phase(s) below;" echo "full logs in the \`hotpath-warp-ab-${{ github.run_number }}\` artifact." if [[ -n "$run_dir" && -d "$run_dir/server-logs" ]]; then for f in "$run_dir"/server-logs/*.log; do [[ -f "$f" ]] || continue echo echo "
$(basename "$f")" echo echo '```' tail -n 30 "$f" echo '```' echo "
" done fi fi } >> "$GITHUB_STEP_SUMMARY" - name: Stage successful candidate baseline if: >- steps.ab.outputs.status == '0' && steps.commits.outputs.baseline_sha != steps.commits.outputs.candidate_sha run: | set -euo pipefail cp candidate-bin/rustfs baseline-bin/rustfs - name: Cache successful candidate baseline if: >- steps.ab.outputs.status == '0' && steps.commits.outputs.baseline_sha != steps.commits.outputs.candidate_sha uses: actions/cache/save@55cc8345863c7cc4c66a329aec7e433d2d1c52a9 # v6 with: path: baseline-bin/rustfs key: rustfs-baseline-${{ steps.commits.outputs.candidate_sha }} # Scheduled failure alerting is handled by the alert-on-failure job below # (perf-2 consuming ci-8's schedule-failure-issue composite action). - name: Enforce gate if: always() run: | status="${{ steps.ab.outputs.status }}" if [[ "$status" != "0" ]]; then echo "::error::warp A/B budget gate failed (exit $status). See the step summary / gate.md artifact." >&2 exit "$status" fi echo "warp A/B budget gate passed." alert-on-failure: name: Alert on scheduled failure needs: [warp-ab] # `always()` is required: without it this job is skipped when a needed # job fails. Alerts only for scheduled (nightly) runs (backlog#1149 # ci-8); manual dispatch failures are already watched by a human. # `cancelled` is included alongside `failure` on purpose: a job that hits # timeout-minutes ends as `cancelled`, and the 2026-07-11..07-14 nightly # timeouts went silent precisely because the guard was failure-only. The # composite action already reports cancelled/timed-out jobs in the issue # body. if: >- always() && github.event_name == 'schedule' && (contains(needs.*.result, 'failure') || contains(needs.*.result, 'cancelled')) runs-on: ubuntu-latest timeout-minutes: 10 permissions: contents: read issues: write steps: - uses: actions/checkout@9c091bb21b7c1c1d1991bb908d89e4e9dddfe3e0 # v7 with: persist-credentials: false - name: Open or update failure-tracking issue uses: ./.github/actions/schedule-failure-issue with: github-token: ${{ secrets.GITHUB_TOKEN }}