mirror of
https://github.com/rustfs/rustfs.git
synced 2026-08-23 04:39:04 +00:00
378 lines
16 KiB
YAML
378 lines
16 KiB
YAML
# Copyright 2024 RustFS Team
|
|
#
|
|
# Licensed under the Apache License, Version 2.0 (the "License");
|
|
# you may not use this file except in compliance with the License.
|
|
# You may obtain a copy of the License at
|
|
#
|
|
# http://www.apache.org/licenses/LICENSE-2.0
|
|
#
|
|
# Unless required by applicable law or agreed to in writing, software
|
|
# distributed under the License is distributed on an "AS IS" BASIS,
|
|
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
|
# See the License for the specific language governing permissions and
|
|
# limitations under the License.
|
|
|
|
# Warp A/B budget gate for the hotpath series (rustfs/backlog#935 HP-14, item 4).
|
|
#
|
|
# Two entry points, honestly scoped:
|
|
# * schedule (nightly, on main): post-merge detection — catches a regression
|
|
# within 24h of landing, not before merge.
|
|
# * workflow_dispatch: an explicitly selected trusted ref.
|
|
# The dispatch input can run the gate with --allow-regression so a deliberate
|
|
# correctness cost (e.g. the #4221 fsync durability fix) is recorded, not
|
|
# blocked (rustfs/backlog#935 correction 1).
|
|
|
|
name: Performance A/B
|
|
|
|
on:
|
|
schedule:
|
|
- cron: "31 6 * * *" # 06:31 UTC nightly, against main
|
|
workflow_dispatch:
|
|
inputs:
|
|
duration:
|
|
description: "warp duration per round"
|
|
required: false
|
|
default: "12s"
|
|
type: string
|
|
allow_regression:
|
|
description: "Pass the gate despite a FAIL (deliberate tradeoff)"
|
|
required: false
|
|
default: false
|
|
type: boolean
|
|
permissions:
|
|
actions: read
|
|
contents: read
|
|
|
|
env:
|
|
CARGO_TERM_COLOR: always
|
|
RUST_BACKTRACE: 1
|
|
|
|
jobs:
|
|
warp-ab:
|
|
name: Warp A/B budget gate
|
|
runs-on: sm-standard-2
|
|
# A normal nightly restores the last successful binary and builds only the
|
|
# candidate; daily access keeps that cache warm. A cache miss may build both
|
|
# and needs room for the A/B run plus artifact and cache publication.
|
|
timeout-minutes: 180
|
|
steps:
|
|
- name: Checkout repository
|
|
uses: actions/checkout@9c091bb21b7c1c1d1991bb908d89e4e9dddfe3e0 # v7
|
|
with:
|
|
persist-credentials: false
|
|
fetch-depth: 0 # baseline may be an earlier successful scheduled head
|
|
|
|
- name: Setup Rust environment
|
|
uses: ./.github/actions/setup
|
|
with:
|
|
rust-version: stable
|
|
cache-shared-key: warp-ab-${{ hashFiles('**/Cargo.lock') }}
|
|
cache-save-if: ${{ github.ref == 'refs/heads/main' }}
|
|
|
|
- name: Install warp
|
|
run: |
|
|
set -euo pipefail
|
|
WARP_VERSION="1.0.0"
|
|
curl -fsSL "https://github.com/minio/warp/releases/download/v${WARP_VERSION}/warp_Linux_x86_64.tar.gz" \
|
|
| sudo tar -xz -C /usr/local/bin warp
|
|
warp --version
|
|
|
|
- name: Decide exemption
|
|
id: exempt
|
|
env:
|
|
INPUT_ALLOW_REGRESSION: ${{ github.event.inputs.allow_regression }}
|
|
run: |
|
|
allow="false"
|
|
if [[ "$INPUT_ALLOW_REGRESSION" == "true" ]]; then
|
|
allow="true"
|
|
fi
|
|
echo "allow_regression=$allow" >> "$GITHUB_OUTPUT"
|
|
|
|
# A failed regression run must keep comparing against the last known-good
|
|
# scheduled head. Otherwise the next nightly would absorb the regression
|
|
# into its baseline and turn green without a fix.
|
|
- name: Find last successful scheduled baseline
|
|
id: scheduled_baseline
|
|
if: github.event_name == 'schedule'
|
|
uses: actions/github-script@ed597411d8f924073f98dfc5c65a23a2325f34cd # v8
|
|
with:
|
|
result-encoding: string
|
|
script: |
|
|
const { data } = await github.rest.actions.listWorkflowRuns({
|
|
owner: context.repo.owner,
|
|
repo: context.repo.repo,
|
|
workflow_id: "performance-ab.yml",
|
|
event: "schedule",
|
|
status: "success",
|
|
per_page: 1,
|
|
});
|
|
return data.workflow_runs[0]?.head_sha ?? "";
|
|
|
|
# Manual runs compare a selected ref with current main. Scheduled runs
|
|
# compare current main with the last successful scheduled head. With no
|
|
# history, the first run measures the candidate against itself and seeds
|
|
# that head only if the complete rig succeeds.
|
|
- name: Resolve baseline / candidate commits
|
|
id: commits
|
|
env:
|
|
SCHEDULED_BASELINE_SHA: ${{ steps.scheduled_baseline.outputs.result }}
|
|
run: |
|
|
set -euo pipefail
|
|
candidate_sha="$(git rev-parse HEAD)"
|
|
if [[ "${{ github.event_name }}" == "schedule" ]]; then
|
|
baseline_sha="${SCHEDULED_BASELINE_SHA:-$candidate_sha}"
|
|
if ! git merge-base --is-ancestor "$baseline_sha" "$candidate_sha"; then
|
|
echo "::error::scheduled baseline $baseline_sha is not an ancestor of candidate $candidate_sha" >&2
|
|
exit 1
|
|
fi
|
|
else
|
|
baseline_sha="$(git rev-parse origin/main)"
|
|
fi
|
|
git cat-file -e "${baseline_sha}^{commit}"
|
|
echo "baseline_sha=$baseline_sha" >> "$GITHUB_OUTPUT"
|
|
echo "candidate_sha=$candidate_sha" >> "$GITHUB_OUTPUT"
|
|
echo "baseline commit: $baseline_sha"
|
|
echo "candidate commit: $candidate_sha"
|
|
|
|
# Exact-key restore of the candidate binary saved by its successful
|
|
# scheduled run. A miss leaves cache-hit unset and falls back to a source
|
|
# build of that known-good head.
|
|
- name: Restore cached baseline binary
|
|
id: baseline_cache
|
|
uses: actions/cache/restore@55cc8345863c7cc4c66a329aec7e433d2d1c52a9 # v6
|
|
with:
|
|
path: baseline-bin/rustfs
|
|
key: rustfs-baseline-${{ steps.commits.outputs.baseline_sha }}
|
|
|
|
# Self-heal: on a nightly/dispatch run where the candidate commit IS the
|
|
# baseline commit, a cache miss would make the rig build the same commit
|
|
# twice (~65min per side with the post-#4806 LTO profile — no job budget
|
|
# fits that). Build it once here, reuse it for both phases, and save it
|
|
# back to the cache so the next run hits.
|
|
- name: Build baseline on cache miss (same-commit self-heal)
|
|
id: selfheal
|
|
if: >-
|
|
steps.baseline_cache.outputs.cache-hit != 'true' &&
|
|
steps.commits.outputs.baseline_sha == steps.commits.outputs.candidate_sha
|
|
run: |
|
|
set -euo pipefail
|
|
cargo build --release --bin rustfs
|
|
mkdir -p baseline-bin
|
|
cp target/release/rustfs baseline-bin/rustfs
|
|
echo "built=true" >> "$GITHUB_OUTPUT"
|
|
|
|
- name: Build baseline on cache miss (different candidate)
|
|
id: baseline_build
|
|
if: >-
|
|
steps.baseline_cache.outputs.cache-hit != 'true' &&
|
|
steps.commits.outputs.baseline_sha != steps.commits.outputs.candidate_sha
|
|
run: |
|
|
set -euo pipefail
|
|
baseline_root="$RUNNER_TEMP/rustfs-baseline-${{ github.run_id }}"
|
|
baseline_target="$RUNNER_TEMP/rustfs-baseline-target-${{ github.run_id }}"
|
|
git worktree add --detach "$baseline_root" "${{ steps.commits.outputs.baseline_sha }}"
|
|
cargo build --release --manifest-path "$baseline_root/Cargo.toml" --bin rustfs --target-dir "$baseline_target"
|
|
mkdir -p baseline-bin
|
|
cp "$baseline_target/release/rustfs" baseline-bin/rustfs
|
|
git worktree remove --force "$baseline_root"
|
|
echo "built=true" >> "$GITHUB_OUTPUT"
|
|
|
|
- name: Build candidate binary
|
|
id: candidate_build
|
|
if: steps.commits.outputs.baseline_sha != steps.commits.outputs.candidate_sha
|
|
run: |
|
|
set -euo pipefail
|
|
cargo build --release --bin rustfs
|
|
mkdir -p candidate-bin
|
|
cp target/release/rustfs candidate-bin/rustfs
|
|
echo "built=true" >> "$GITHUB_OUTPUT"
|
|
|
|
- name: Save self-healed baseline to cache
|
|
if: steps.selfheal.outputs.built == 'true' || steps.baseline_build.outputs.built == 'true'
|
|
uses: actions/cache/save@55cc8345863c7cc4c66a329aec7e433d2d1c52a9 # v6
|
|
with:
|
|
path: baseline-bin/rustfs
|
|
key: rustfs-baseline-${{ steps.commits.outputs.baseline_sha }}
|
|
|
|
- name: Run warp A/B and gate
|
|
id: ab
|
|
env:
|
|
INPUT_DURATION: ${{ github.event.inputs.duration }}
|
|
run: |
|
|
set -euo pipefail
|
|
# The formal runner executes A1 baseline -> B1 candidate -> B2 candidate
|
|
# -> A2 baseline for each workload and drive-sync cell. It requires three
|
|
# rounds per leg to emit tail latency and error-rate evidence.
|
|
# --health-timeout 180 outlasts the server's own 120s startup-readiness
|
|
# budget, which the rig's previous 60s health poll undershot (the first
|
|
# two nightly failures). perf-6 recalibrates these once the noise study
|
|
# lands.
|
|
duration="${INPUT_DURATION:-12s}"
|
|
baseline_sha="${{ steps.commits.outputs.baseline_sha }}"
|
|
candidate_sha="${{ steps.commits.outputs.candidate_sha }}"
|
|
baseline_hit="${{ steps.baseline_cache.outputs.cache-hit }}"
|
|
selfheal_built="${{ steps.selfheal.outputs.built }}"
|
|
baseline_built="${{ steps.baseline_build.outputs.built }}"
|
|
candidate_built="${{ steps.candidate_build.outputs.built }}"
|
|
|
|
args=(--duration "$duration" --rounds 3 --cooldown 5 --health-timeout 180 --baseline-revision "$baseline_sha" --candidate-revision "$candidate_sha")
|
|
|
|
if [[ "$baseline_hit" == "true" || "$selfheal_built" == "true" || "$baseline_built" == "true" ]]; then
|
|
chmod +x baseline-bin/rustfs
|
|
base_bin="$PWD/baseline-bin/rustfs"
|
|
args+=(--baseline-bin "$base_bin")
|
|
if [[ "$baseline_hit" == "true" ]]; then
|
|
base_src="actions-cache (rustfs-baseline-$baseline_sha)"
|
|
elif [[ "$selfheal_built" == "true" ]]; then
|
|
base_src="source build (cache self-heal, saved as rustfs-baseline-$baseline_sha)"
|
|
else
|
|
base_src="isolated baseline source build (saved as rustfs-baseline-$baseline_sha)"
|
|
fi
|
|
if [[ "$candidate_sha" == "$baseline_sha" ]]; then
|
|
# No commits landed since the last successful baseline, so reuse
|
|
# the one binary for both phases and measure only rig drift.
|
|
args+=(--candidate-bin "$base_bin")
|
|
cand_src="same binary as baseline (same commit)"
|
|
elif [[ "$candidate_built" == "true" ]]; then
|
|
chmod +x candidate-bin/rustfs
|
|
args+=(--candidate-bin "$PWD/candidate-bin/rustfs")
|
|
cand_src="source build of the checked-out ref"
|
|
else
|
|
echo "::error::candidate binary was not built" >&2
|
|
exit 2
|
|
fi
|
|
else
|
|
echo "::error::baseline binary was not restored or built" >&2
|
|
exit 2
|
|
fi
|
|
|
|
echo "baseline binary: $base_src"
|
|
echo "candidate binary: $cand_src"
|
|
|
|
if [[ "${{ steps.exempt.outputs.allow_regression }}" == "true" ]]; then
|
|
args+=(--allow-regression --exemption-reason "workflow dispatch override")
|
|
fi
|
|
# Do not let a gate FAIL abort the job here; capture status and surface
|
|
# it after the step summary is written.
|
|
set +e
|
|
bash scripts/run_hotpath_warp_abba.sh "${args[@]}"
|
|
echo "status=$?" >> "$GITHUB_OUTPUT"
|
|
set -e
|
|
# Locate the newest run dir + gate.md for the summary/comment/artifact
|
|
# steps. On a startup failure there is no gate.md, but the run dir still
|
|
# holds server-logs/ for diagnosis.
|
|
# Run dirs are UTC-timestamp names (no special chars); ls is safe here.
|
|
# shellcheck disable=SC2012
|
|
run_dir="$(ls -td target/hotpath-abba/*/ 2>/dev/null | head -n1 || true)"
|
|
echo "run_dir=${run_dir%/}" >> "$GITHUB_OUTPUT"
|
|
# shellcheck disable=SC2012
|
|
gate_md="$(ls -t target/hotpath-abba/*/candidate_gate.md 2>/dev/null | head -n1 || true)"
|
|
echo "gate_md=$gate_md" >> "$GITHUB_OUTPUT"
|
|
|
|
- name: Upload A/B results
|
|
if: always()
|
|
uses: actions/upload-artifact@b7c566a772e6b6bfb58ed0dc250532a479d7789f # v6
|
|
with:
|
|
name: hotpath-warp-ab-${{ github.run_number }}
|
|
# Includes per-cell median_summary.csv / baseline_compare.csv, both gates,
|
|
# and server-logs/ (rustfs.log + startup env per phase) so a failed run
|
|
# is diagnosable. Short retention: this is churny nightly debug data.
|
|
path: target/hotpath-abba/
|
|
if-no-files-found: warn
|
|
retention-days: 14
|
|
|
|
- name: Write gate summary
|
|
if: always()
|
|
run: |
|
|
set -euo pipefail
|
|
status="${{ steps.ab.outputs.status }}"
|
|
gate_md="${{ steps.ab.outputs.gate_md }}"
|
|
run_dir="${{ steps.ab.outputs.run_dir }}"
|
|
{
|
|
echo "## Hotpath warp A/B — run ${{ github.run_number }}"
|
|
echo
|
|
if [[ "$status" == "0" ]]; then
|
|
echo "Rig/gate exit: \`0\` (pass or warn)."
|
|
else
|
|
echo "Rig/gate exit: \`${status:-unknown}\` — **FAILED**."
|
|
fi
|
|
echo
|
|
if [[ -n "$gate_md" && -f "$gate_md" ]]; then
|
|
cat "$gate_md"
|
|
else
|
|
echo "No \`gate.md\` produced — the rig failed **before** the gate"
|
|
echo "(most likely server startup / health). Failing phase(s) below;"
|
|
echo "full logs in the \`hotpath-warp-ab-${{ github.run_number }}\` artifact."
|
|
if [[ -n "$run_dir" && -d "$run_dir/server-logs" ]]; then
|
|
for f in "$run_dir"/server-logs/*.log; do
|
|
[[ -f "$f" ]] || continue
|
|
echo
|
|
echo "<details><summary>$(basename "$f")</summary>"
|
|
echo
|
|
echo '```'
|
|
tail -n 30 "$f"
|
|
echo '```'
|
|
echo "</details>"
|
|
done
|
|
fi
|
|
fi
|
|
} >> "$GITHUB_STEP_SUMMARY"
|
|
|
|
- name: Stage successful candidate baseline
|
|
if: >-
|
|
steps.ab.outputs.status == '0' &&
|
|
steps.commits.outputs.baseline_sha != steps.commits.outputs.candidate_sha
|
|
run: |
|
|
set -euo pipefail
|
|
cp candidate-bin/rustfs baseline-bin/rustfs
|
|
|
|
- name: Cache successful candidate baseline
|
|
if: >-
|
|
steps.ab.outputs.status == '0' &&
|
|
steps.commits.outputs.baseline_sha != steps.commits.outputs.candidate_sha
|
|
uses: actions/cache/save@55cc8345863c7cc4c66a329aec7e433d2d1c52a9 # v6
|
|
with:
|
|
path: baseline-bin/rustfs
|
|
key: rustfs-baseline-${{ steps.commits.outputs.candidate_sha }}
|
|
|
|
# Scheduled failure alerting is handled by the alert-on-failure job below
|
|
# (perf-2 consuming ci-8's schedule-failure-issue composite action).
|
|
|
|
- name: Enforce gate
|
|
if: always()
|
|
run: |
|
|
status="${{ steps.ab.outputs.status }}"
|
|
if [[ "$status" != "0" ]]; then
|
|
echo "::error::warp A/B budget gate failed (exit $status). See the step summary / gate.md artifact." >&2
|
|
exit "$status"
|
|
fi
|
|
echo "warp A/B budget gate passed."
|
|
|
|
alert-on-failure:
|
|
name: Alert on scheduled failure
|
|
needs: [warp-ab]
|
|
# `always()` is required: without it this job is skipped when a needed
|
|
# job fails. Alerts only for scheduled (nightly) runs (backlog#1149
|
|
# ci-8); manual dispatch failures are already watched by a human.
|
|
# `cancelled` is included alongside `failure` on purpose: a job that hits
|
|
# timeout-minutes ends as `cancelled`, and the 2026-07-11..07-14 nightly
|
|
# timeouts went silent precisely because the guard was failure-only. The
|
|
# composite action already reports cancelled/timed-out jobs in the issue
|
|
# body.
|
|
if: >-
|
|
always() && github.event_name == 'schedule' &&
|
|
(contains(needs.*.result, 'failure') || contains(needs.*.result, 'cancelled'))
|
|
runs-on: ubuntu-latest
|
|
timeout-minutes: 10
|
|
permissions:
|
|
contents: read
|
|
issues: write
|
|
steps:
|
|
- uses: actions/checkout@9c091bb21b7c1c1d1991bb908d89e4e9dddfe3e0 # v7
|
|
with:
|
|
persist-credentials: false
|
|
- name: Open or update failure-tracking issue
|
|
uses: ./.github/actions/schedule-failure-issue
|
|
with:
|
|
github-token: ${{ secrets.GITHUB_TOKEN }}
|