# Copyright 2024 RustFS Team # # Licensed under the Apache License, Version 2.0 (the "License"); # you may not use this file except in compliance with the License. # You may obtain a copy of the License at # # http://www.apache.org/licenses/LICENSE-2.0 # # Unless required by applicable law or agreed to in writing, software # distributed under the License is distributed on an "AS IS" BASIS, # WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. # See the License for the specific language governing permissions and # limitations under the License. # Warp A/B budget gate for the hotpath series (rustfs/backlog#935 HP-14, item 4). # # Two entry points, honestly scoped: # * schedule (nightly, on main): post-merge detection — catches a regression # within 24h of landing, not before merge. # * pull_request labeled `perf-ab`: opt-in pre-merge gate for a specific PR. # The `perf-deliberate-tradeoff` label runs the gate with --allow-regression so # a deliberate correctness cost (e.g. the #4221 fsync durability fix) is # recorded but does not block (rustfs/backlog#935 correction 1). name: Performance A/B on: schedule: - cron: "0 6 * * *" # 06:00 UTC nightly, against main workflow_dispatch: inputs: duration: description: "warp duration per round (short by default to fit the double-build budget)" required: false default: "12s" type: string allow_regression: description: "Pass the gate despite a FAIL (deliberate tradeoff)" required: false default: false type: boolean pull_request: types: [labeled, synchronize, reopened] permissions: contents: read pull-requests: write # Per-PR: a new push cancels the previous (up to 90-minute) A/B run instead of # stacking them. Nightly schedule and manual dispatch get a unique group and # always run to completion. concurrency: group: ${{ github.workflow }}-${{ github.event.pull_request.number || github.run_id }} cancel-in-progress: ${{ github.event_name == 'pull_request' }} env: CARGO_TERM_COLOR: always RUST_BACKTRACE: 1 jobs: warp-ab: name: Warp A/B budget gate # Opt-in on PRs: only run when the `perf-ab` label is present, and for # `labeled` events only when the label being added is `perf-ab` itself — # adding an unrelated label to an opted-in PR must not re-run the gate. # Always run on schedule / manual dispatch. if: >- github.event_name != 'pull_request' || (contains(github.event.pull_request.labels.*.name, 'perf-ab') && (github.event.action != 'labeled' || github.event.label.name == 'perf-ab')) runs-on: sm-standard-2 # Phase-0 stopgap: the baseline+candidate release double-build alone is # ~65min on this runner, so 90min left no room for a real full-matrix # measurement (the earlier nightly runs only ever failed *before* # measuring). 120min gives the 24-cell short matrix headroom until perf-3 # caches the baseline binary and restores a tighter budget. timeout-minutes: 120 steps: - name: Checkout repository uses: actions/checkout@9c091bb21b7c1c1d1991bb908d89e4e9dddfe3e0 # v7 with: fetch-depth: 0 # baseline is built from origin/main - name: Setup Rust environment uses: ./.github/actions/setup with: rust-version: stable cache-shared-key: warp-ab-${{ hashFiles('**/Cargo.lock') }} cache-save-if: ${{ github.ref == 'refs/heads/main' }} github-token: ${{ secrets.GITHUB_TOKEN }} - name: Install warp run: | set -euo pipefail WARP_VERSION="1.0.0" curl -fsSL "https://github.com/minio/warp/releases/download/v${WARP_VERSION}/warp_Linux_x86_64.tar.gz" \ | sudo tar -xz -C /usr/local/bin warp warp --version - name: Decide exemption id: exempt run: | allow="false" if [[ "${{ github.event_name }}" == "pull_request" ]] \ && ${{ contains(github.event.pull_request.labels.*.name, 'perf-deliberate-tradeoff') }}; then allow="true" fi if [[ "${{ github.event.inputs.allow_regression }}" == "true" ]]; then allow="true" fi echo "allow_regression=$allow" >> "$GITHUB_OUTPUT" - name: Run warp A/B and gate id: ab run: | set -euo pipefail # Budget note: the baseline+candidate release double-build (~65 min, # cached away later by perf-3) dominates the 90-min job, so the # measurement runs a short warp matrix — duration/rounds/cooldown are # kept small to fit all 24 cells (6 workloads x 2 phases x 2 drive-sync) # under budget rather than dropping cells. --health-timeout 180 outlasts # the server's own 120s startup-readiness budget, which is what the # rig's previous 60s health poll undershot (the first two nightly # failures). perf-6 will recalibrate these once the pipeline is green. duration="${{ github.event.inputs.duration || '12s' }}" args=(--baseline-ref origin/main --duration "$duration" --rounds 2 --cooldown 5 --health-timeout 180) if [[ "${{ steps.exempt.outputs.allow_regression }}" == "true" ]]; then args+=(--allow-regression --exemption-reason "labeled perf-deliberate-tradeoff / dispatch override") fi # Do not let a gate FAIL abort the job here; capture status and surface # it after the PR comment is posted. set +e bash scripts/run_hotpath_warp_ab.sh "${args[@]}" echo "status=$?" >> "$GITHUB_OUTPUT" set -e # Locate the newest run dir + gate.md for the summary/comment/artifact # steps. On a startup failure there is no gate.md, but the run dir still # holds server-logs/ for diagnosis. # Run dirs are UTC-timestamp names (no special chars); ls is safe here. # shellcheck disable=SC2012 run_dir="$(ls -td target/hotpath-ab/*/ 2>/dev/null | head -n1 || true)" echo "run_dir=${run_dir%/}" >> "$GITHUB_OUTPUT" # shellcheck disable=SC2012 gate_md="$(ls -t target/hotpath-ab/*/gate.md 2>/dev/null | head -n1 || true)" echo "gate_md=$gate_md" >> "$GITHUB_OUTPUT" - name: Upload A/B results if: always() uses: actions/upload-artifact@b7c566a772e6b6bfb58ed0dc250532a479d7789f # v6 with: name: hotpath-warp-ab-${{ github.run_number }} # Includes per-cell median_summary.csv / baseline_compare.csv, gate.md, # and server-logs/ (rustfs.log + startup env per phase) so a failed run # is diagnosable. Short retention: this is churny nightly debug data. path: target/hotpath-ab/ if-no-files-found: warn retention-days: 14 - name: Write gate summary if: always() run: | set -euo pipefail status="${{ steps.ab.outputs.status }}" gate_md="${{ steps.ab.outputs.gate_md }}" run_dir="${{ steps.ab.outputs.run_dir }}" { echo "## Hotpath warp A/B — run ${{ github.run_number }}" echo if [[ "$status" == "0" ]]; then echo "Rig/gate exit: \`0\` (pass or warn)." else echo "Rig/gate exit: \`${status:-unknown}\` — **FAILED**." fi echo if [[ -n "$gate_md" && -f "$gate_md" ]]; then cat "$gate_md" else echo "No \`gate.md\` produced — the rig failed **before** the gate" echo "(most likely server startup / health). Failing phase(s) below;" echo "full logs in the \`hotpath-warp-ab-${{ github.run_number }}\` artifact." if [[ -n "$run_dir" && -d "$run_dir/server-logs" ]]; then for f in "$run_dir"/server-logs/*.log; do [[ -f "$f" ]] || continue echo echo "
$(basename "$f")" echo echo '```' tail -n 30 "$f" echo '```' echo "
" done fi fi } >> "$GITHUB_STEP_SUMMARY" - name: Comment gate result on PR if: always() && github.event_name == 'pull_request' && steps.ab.outputs.gate_md != '' env: GH_TOKEN: ${{ secrets.GITHUB_TOKEN }} run: | gh pr comment "${{ github.event.pull_request.number }}" --body-file "${{ steps.ab.outputs.gate_md }}" # TODO(ci-8/perf-2): scheduled/dispatch failures are currently silent. Once # ci-8 lands the .github/actions/schedule-failure-issue composite action, # perf-2 adds a step here guarded by # if: failure() && github.event_name != 'pull_request' # that calls it (label perf-nightly-failure, append to an existing open # issue) instead of hand-rolling gh CLI dedup. Do not implement it here. - name: Enforce gate if: always() run: | status="${{ steps.ab.outputs.status }}" if [[ "$status" != "0" ]]; then echo "::error::warp A/B budget gate failed (exit $status). See the step summary / PR comment / gate.md artifact." >&2 exit "$status" fi echo "warp A/B budget gate passed." alert-on-failure: name: Alert on scheduled failure needs: [warp-ab] # `always()` is required: without it this job is skipped when a needed # job fails. Alerts only for scheduled (nightly) runs (backlog#1149 # ci-8); PR and manual dispatch failures are already watched by a human. if: always() && github.event_name == 'schedule' && contains(needs.*.result, 'failure') runs-on: ubuntu-latest timeout-minutes: 10 permissions: contents: read issues: write steps: - uses: actions/checkout@9c091bb21b7c1c1d1991bb908d89e4e9dddfe3e0 # v7 - name: Open or update failure-tracking issue uses: ./.github/actions/schedule-failure-issue with: github-token: ${{ secrets.GITHUB_TOKEN }}