From 40eee6177a74470b3cb554f39fabe7f30daeb554 Mon Sep 17 00:00:00 2001 From: houseme Date: Fri, 31 Jul 2026 19:33:10 +0800 Subject: [PATCH] test: add formal hotpath ABBA runner (#5507) Co-authored-by: heihutu --- crates/kms/tests/vault_fault_injection.rs | 39 +- docs/operations/hotpath-warp-abba-runbook.md | 277 ++++++++++++ rustfs/src/admin/router.rs | 12 +- scripts/run_hotpath_warp_abba.sh | 428 +++++++++++++++++++ 4 files changed, 733 insertions(+), 23 deletions(-) create mode 100644 docs/operations/hotpath-warp-abba-runbook.md create mode 100755 scripts/run_hotpath_warp_abba.sh diff --git a/crates/kms/tests/vault_fault_injection.rs b/crates/kms/tests/vault_fault_injection.rs index 14176fdb9..f7f86113a 100644 --- a/crates/kms/tests/vault_fault_injection.rs +++ b/crates/kms/tests/vault_fault_injection.rs @@ -33,9 +33,11 @@ use std::time::Duration; use metrics_util::MetricKind; use metrics_util::debugging::{DebugValue, DebuggingRecorder}; -use rustfs_kms::backends::KmsClient; -use rustfs_kms::backends::vault::VaultKmsClient; -use rustfs_kms::{KmsConfig, KmsError, VaultAuthMethod, VaultConfig}; +use rustfs_kms::backends::KmsBackend as KmsBackendTrait; +use rustfs_kms::backends::vault::VaultKmsBackend; +use rustfs_kms::{ + BackendConfig, DescribeKeyRequest, KmsBackend as KmsBackendKind, KmsConfig, KmsError, VaultAuthMethod, VaultConfig, +}; const OPERATIONS_TOTAL: &str = "rustfs_kms_backend_operations_total"; const ATTEMPT_FAILURES_TOTAL: &str = "rustfs_kms_backend_attempt_failures_total"; @@ -54,14 +56,23 @@ fn vault_config(address: &str, token: &str) -> VaultConfig { } } -fn kms_config(attempt_timeout: Duration, retry_attempts: u32) -> KmsConfig { +fn kms_config(vault_config: VaultConfig, attempt_timeout: Duration, retry_attempts: u32) -> KmsConfig { KmsConfig { + backend: KmsBackendKind::VaultKv2, + backend_config: BackendConfig::VaultKv2(Box::new(vault_config)), + allow_insecure_dev_defaults: true, timeout: attempt_timeout, retry_attempts, ..KmsConfig::default() } } +fn describe_key_request(key_id: &str) -> DescribeKeyRequest { + DescribeKeyRequest { + key_id: key_id.to_string(), + } +} + type MetricEntry = ( metrics_util::CompositeKey, Option, @@ -118,11 +129,10 @@ fn connection_refused_is_retried_within_budget() { let snapshot = record_metrics(|| { Box::pin(async move { - let client = VaultKmsClient::new(vault_config(&address, "unused"), &kms_config(Duration::from_secs(2), 2)) + let client = VaultKmsBackend::new(kms_config(vault_config(&address, "unused"), Duration::from_secs(2), 2)) .await .expect("client construction performs no network calls"); - let error = client - .describe_key("fault-injection-refused", None) + let error = KmsBackendTrait::describe_key(&client, describe_key_request("fault-injection-refused")) .await .expect_err("a refused connection must fail the operation"); assert!(matches!(error, KmsError::BackendError { .. }), "got {error:?}"); @@ -166,11 +176,10 @@ fn stalled_connection_is_cut_off_by_the_attempt_timeout() { } }); - let client = VaultKmsClient::new(vault_config(&address, "unused"), &kms_config(Duration::from_millis(250), 1)) + let client = VaultKmsBackend::new(kms_config(vault_config(&address, "unused"), Duration::from_millis(250), 1)) .await .expect("client construction performs no network calls"); - let error = client - .describe_key("fault-injection-stalled", None) + let error = KmsBackendTrait::describe_key(&client, describe_key_request("fault-injection-stalled")) .await .expect_err("a stalled request must be cut off by the attempt timeout"); assert!( @@ -201,11 +210,10 @@ fn real_vault_invalid_token_is_fatal_and_never_retried() { let snapshot = record_metrics(|| { Box::pin(async { let config = vault_config(&real_vault_address(), "fault-injection-invalid-token"); - let client = VaultKmsClient::new(config, &kms_config(Duration::from_secs(5), 3)) + let client = VaultKmsBackend::new(kms_config(config, Duration::from_secs(5), 3)) .await .expect("client construction performs no network calls"); - let error = client - .describe_key("fault-injection-forbidden", None) + let error = KmsBackendTrait::describe_key(&client, describe_key_request("fault-injection-forbidden")) .await .expect_err("an invalid token must be rejected"); assert!(matches!(error, KmsError::BackendError { .. }), "got {error:?}"); @@ -235,11 +243,10 @@ fn real_vault_missing_key_is_resolved_in_one_attempt() { let snapshot = record_metrics(|| { Box::pin(async move { let config = vault_config(&real_vault_address(), &token); - let client = VaultKmsClient::new(config, &kms_config(Duration::from_secs(5), 3)) + let client = VaultKmsBackend::new(kms_config(config, Duration::from_secs(5), 3)) .await .expect("client construction performs no network calls"); - let error = client - .describe_key("fault-injection-definitely-missing", None) + let error = KmsBackendTrait::describe_key(&client, describe_key_request("fault-injection-definitely-missing")) .await .expect_err("a missing key must resolve to key-not-found"); assert!(matches!(error, KmsError::KeyNotFound { .. }), "got {error:?}"); diff --git a/docs/operations/hotpath-warp-abba-runbook.md b/docs/operations/hotpath-warp-abba-runbook.md new file mode 100644 index 000000000..5d73b59d1 --- /dev/null +++ b/docs/operations/hotpath-warp-abba-runbook.md @@ -0,0 +1,277 @@ +# Hotpath warp ABBA validation runbook + +This runbook describes how to collect formal Linux or production-cluster +evidence for hotpath performance changes. Use it when a short local A/B smoke +run is too noisy to decide whether a regression is real. + +The ABBA runner executes each workload and drive-sync cell as: + +```text +A1 baseline -> B1 candidate -> B2 candidate -> A2 baseline +``` + +`B1` and `B2` are compared with `A1` to measure the candidate delta. `A2` is +also compared with `A1` to measure baseline drift. Treat a candidate regression +as actionable only when the `A2` drift is passing or materially smaller than +the `B1` and `B2` delta for the same workload. + +## Scope + +Use this runbook for hotpath profiling and performance validation of RustFS +object I/O changes, especially when CPU, memory allocation, lock/channel wait +time, request throughput, or tail latency is the review question. + +The script validates the same workload matrix as the hotpath warp A/B gate: + +| Workload | mode | size | +| --- | --- | --- | +| `put-4kib` | put | 4KiB | +| `put-4mib` | put | 4MiB | +| `get-4kib` | get | 4KiB | +| `get-4mib` | get | 4MiB | +| `get-10mib` | get | 10MiB | +| `mixed-256k` | mixed | 256KiB | + +Each workload runs with `RUSTFS_DRIVE_SYNC_ENABLE=true` and +`RUSTFS_DRIVE_SYNC_ENABLE=false`, so a full ABBA pass produces 48 measurement +cells: 6 workloads x 2 drive-sync modes x 4 ABBA legs. + +## Prerequisites + +Run the formal pass on Linux, not on a laptop smoke environment. + +Required tools on the bench host: + +- `bash`, `curl`, `git`, and core GNU userland. +- `warp` on `PATH`, or pass `--warp-bin`. +- Two RustFS Linux binaries: one baseline and one candidate. +- Enough isolated disks or directories for the local runner, or an externally + managed RustFS cluster for production-like validation. +- Stable host telemetry collection such as `pidstat`, `mpstat`, `iostat`, + `sar`, `perf`, `heaptrack`, or the platform's equivalent observability stack. + +Cluster-mode requirements: + +- A deploy hook that can replace the RustFS binary on every node. +- The hook must apply `RUSTFS_DRIVE_SYNC_ENABLE` for the current ABBA leg. +- The hook must restart RustFS and return only after the rollout command has + been accepted. The ABBA script performs the HTTP readiness wait. +- The benchmark client should run outside the RustFS nodes when possible. +- Do not run against a production data set unless the workload bucket and test + credentials are isolated and approved for destructive benchmark traffic. + +## Build the binaries + +Build the baseline from the comparison commit, usually `origin/main` or the +previous accepted release: + +```bash +git fetch origin main +git switch --detach origin/main +cargo build --release -p rustfs --bins +cp target/release/rustfs /tmp/rustfs-baseline +``` + +Build the candidate from the PR commit: + +```bash +git switch +cargo build --release -p rustfs --bins +cp target/release/rustfs /tmp/rustfs-candidate +``` + +For cross-compiled cluster binaries, keep both outputs on the bench host and +make the deploy hook copy the selected binary to the cluster. The ABBA runner +passes the selected binary path through `HOTPATH_ABBA_BINARY`. + +## Local Linux runner + +Use local mode for a dedicated Linux runner with disposable data paths. This is +not a substitute for a production-like cluster, but it is useful before spending +cluster time. + +```bash +scripts/run_hotpath_warp_abba.sh \ + --baseline-bin /tmp/rustfs-baseline \ + --candidate-bin /tmp/rustfs-candidate \ + --address 127.0.0.1:9000 \ + --data-root /var/tmp/rustfs-hotpath-abba \ + --disks 4 \ + --duration 120s \ + --rounds 3 \ + --cooldown 30 \ + --concurrency 16 \ + --out-dir target/hotpath-abba/linux-local +``` + +The script starts and stops RustFS for each ABBA leg. The data root is +throwaway and should not contain important data. + +## Production-like cluster runner + +Use external mode when RustFS lifecycle is managed by ansible, systemd, a +cluster scheduler, or a dedicated deployment harness. In this mode the ABBA +script does not start RustFS directly; it calls `--deploy-hook` before each leg +and then waits for `http://`. + +The deploy hook receives: + +| Environment variable | Value | +| --- | --- | +| `HOTPATH_ABBA_LEG` | `A1`, `B1`, `B2`, or `A2` | +| `HOTPATH_ABBA_PHASE` | `baseline` or `candidate` | +| `HOTPATH_ABBA_BINARY` | selected baseline or candidate binary path | +| `HOTPATH_ABBA_DRIVE_SYNC` | `true` or `false` | + +Example ansible-shaped command: + +```bash +scripts/run_hotpath_warp_abba.sh \ + --baseline-bin /srv/rustfs-binaries/rustfs-baseline \ + --candidate-bin /srv/rustfs-binaries/rustfs-candidate \ + --endpoint rustfs-bench.example.internal:9000 \ + --deploy-hook ' + set -euo pipefail + cd /srv/rustfs-ansible + cp "${HOTPATH_ABBA_BINARY:?}" roles/rustfs/files/rustfs + export RUSTFS_DRIVE_SYNC_ENABLE="${HOTPATH_ABBA_DRIVE_SYNC:?}" + ansible-playbook -f 4 -l bench rustfs-manage.yml --tags stop + ansible-playbook -f 4 -l bench rustfs-manage.yml --tags config + ansible-playbook -f 4 -l bench rustfs-manage.yml --tags binary-copy + ansible-playbook -f 4 -l bench rustfs-manage.yml --tags start + ' \ + --duration 180s \ + --rounds 5 \ + --cooldown 45 \ + --concurrency 32 \ + --out-dir target/hotpath-abba/cluster-pr-XXXX +``` + +For formal evidence, prefer `--rounds 5` or higher when the cluster budget +allows it. The script enforces `--rounds >= 3`. + +## CPU and memory evidence + +ABBA warp output answers whether the candidate changed throughput or latency. +Collect host telemetry at the same time to explain why. + +Recommended minimum: + +```bash +mkdir -p target/hotpath-abba/cluster-pr-XXXX/telemetry + +pidstat -durh 5 > target/hotpath-abba/cluster-pr-XXXX/telemetry/pidstat.txt & +PIDSTAT_PID=$! + +mpstat 5 > target/hotpath-abba/cluster-pr-XXXX/telemetry/mpstat.txt & +MPSTAT_PID=$! + +iostat -xz 5 > target/hotpath-abba/cluster-pr-XXXX/telemetry/iostat.txt & +IOSTAT_PID=$! +``` + +Stop the collectors after the ABBA script exits: + +```bash +kill "$PIDSTAT_PID" "$MPSTAT_PID" "$IOSTAT_PID" +``` + +For deeper CPU attribution, run `perf record` around one representative +workload after the ABBA gate identifies a candidate regression or improvement: + +```bash +perf record -F 99 -g -- sleep 180 +perf report --stdio > target/hotpath-abba/cluster-pr-XXXX/telemetry/perf-report.txt +``` + +For allocation profiling, build the candidate with: + +```bash +cargo build --release -p rustfs --bins --features hotpath-alloc +``` + +Then run the same ABBA command with that binary. Compare allocation-heavy +function sections only within the same build mode. Do not compare +`hotpath-alloc` binaries directly with default release binaries for throughput +acceptance, because allocation instrumentation intentionally changes what is +measured. + +For CPU hotpath sections emitted by hotpath, build with: + +```bash +cargo build --release -p rustfs --bins --features hotpath-cpu +``` + +Use the CPU-enabled report to explain hotspots after the default or plain +`hotpath` ABBA gate shows a real effect. + +## Output layout + +The ABBA runner writes: + +```text +/ + manifest.env + abba_schedule.csv + candidate_gate.md + baseline_drift_gate.md + summary.md + ///median_summary.csv + ///baseline_compare.csv +``` + +Attach or link at least these files in the issue or PR: + +- `summary.md` +- `candidate_gate.md` +- `baseline_drift_gate.md` +- `abba_schedule.csv` +- every `median_summary.csv` and `baseline_compare.csv` for a failed or + borderline workload +- host telemetry files used to explain CPU, memory, or disk saturation + +## Interpretation + +Use this decision table: + +| Candidate gate | A2 drift gate | Interpretation | +| --- | --- | --- | +| PASS | PASS | Candidate is acceptable for the measured matrix. | +| WARN | PASS | Candidate has a small measurable signal; inspect telemetry and decide if it is expected. | +| FAIL | PASS | Candidate likely regressed the affected workload; investigate before merge. | +| FAIL | FAIL on the same workload | Environment drift is high; rerun on a quieter runner or increase duration and rounds. | +| PASS | FAIL | Candidate did not exceed the budget, but the rig was unstable; avoid using the numbers as proof of improvement. | + +When `B1` and `B2` disagree, treat the result as inconclusive even if the gate +passes. Increase duration, rounds, cooldown, or runner isolation before drawing +a conclusion. + +## AI execution checklist + +When delegating the run to an AI agent or an automation runner, provide these +inputs explicitly: + +- repository checkout and candidate branch or commit; +- baseline commit or binary path; +- candidate binary path; +- runner type: local Linux or external cluster; +- endpoint, access key, secret key source, and region; +- deploy hook path or exact command for cluster mode; +- output directory; +- required duration, rounds, cooldown, concurrency, and fail/warn budgets; +- where to upload artifacts after the run. + +The AI agent should execute this sequence: + +1. Confirm `uname -a`, RustFS commits, binary SHA256 sums, `warp --version`, + CPU model, memory size, disk layout, and whether the run is local or cluster. +2. Run `scripts/run_hotpath_warp_abba.sh --dry-run` with the final arguments. +3. Run the real ABBA command with `--rounds >= 3`. +4. Preserve the full output directory without editing generated CSV files. +5. Read `summary.md`, `candidate_gate.md`, and `baseline_drift_gate.md`. +6. Summarize only measured facts: candidate deltas, baseline drift, CPU or + memory saturation, and any failed workloads. +7. Post the summary and artifact location to the tracking issue or PR. + +Do not report a performance win or loss when the baseline drift gate failed on +the same workload and no rerun was collected. diff --git a/rustfs/src/admin/router.rs b/rustfs/src/admin/router.rs index a72858a8a..7bec6c88a 100644 --- a/rustfs/src/admin/router.rs +++ b/rustfs/src/admin/router.rs @@ -2708,18 +2708,16 @@ impl S3Router { pub fn insert(&mut self, method: Method, path: &str, operation: T) -> std::io::Result<()> { let path = Self::make_route_str(method, path); + #[cfg(test)] + let registered_path = path.clone(); // warn!("set uri {}", &path); - #[cfg(test)] - { - self.router.insert(path.clone(), operation).map_err(std::io::Error::other)?; - self.registered_routes.push(path); - } - - #[cfg(not(test))] self.router.insert(path, operation).map_err(std::io::Error::other)?; + #[cfg(test)] + self.registered_routes.push(registered_path); + Ok(()) } diff --git a/scripts/run_hotpath_warp_abba.sh b/scripts/run_hotpath_warp_abba.sh new file mode 100755 index 000000000..ccaff476d --- /dev/null +++ b/scripts/run_hotpath_warp_abba.sh @@ -0,0 +1,428 @@ +#!/usr/bin/env bash +# Formal Linux / production-cluster ABBA runner for the hotpath warp matrix. +# +# This script is intentionally a thin orchestrator around the existing +# run_object_batch_bench_enhanced.sh load driver and hotpath_warp_ab_gate.sh +# relative-budget gate. It runs each durability/workload cell as: +# +# A1 baseline -> B1 candidate -> B2 candidate -> A2 baseline +# +# Candidate legs are compared against A1. The final A2 leg is also compared +# against A1 to quantify baseline drift separately from candidate deltas. + +set -euo pipefail + +PROJECT_ROOT="$(git rev-parse --show-toplevel)" +ENHANCED_BENCH="${PROJECT_ROOT}/scripts/run_object_batch_bench_enhanced.sh" +GATE="${PROJECT_ROOT}/scripts/hotpath_warp_ab_gate.sh" + +BASELINE_BIN="" +CANDIDATE_BIN="" +ENDPOINT="" +DEPLOY_HOOK="" +HEALTH_PATH="/health" +ADDRESS="127.0.0.1:9000" +DATA_ROOT="/tmp/rustfs-hotpath-abba" +DISKS=4 +ACCESS_KEY="rustfsadmin" +SECRET_KEY="rustfsadmin" +REGION="us-east-1" +WARP_BIN="warp" +CONCURRENCY=8 +DURATION="60s" +ROUNDS=3 +COOLDOWN_SECS=20 +HEALTH_TIMEOUT_SECS=180 +FAIL_PCT=10 +WARN_PCT=5 +ALLOW_REGRESSION=false +EXEMPTION_REASON="deliberate correctness tradeoff" +OUT_DIR="${PROJECT_ROOT}/target/hotpath-abba/$(date -u +%Y%m%dT%H%M%SZ 2>/dev/null || echo run)" +DRY_RUN=false + +WORKLOADS=( + "put-4kib|put|4KiB" + "put-4mib|put|4MiB" + "get-4kib|get|4KiB" + "get-4mib|get|4MiB" + "get-10mib|get|10MiB" + "mixed-256k|mixed|256KiB" +) +DRIVE_SYNC_MATRIX=("sync-on|true" "sync-off|false") + +usage() { + cat <<'USAGE' +Usage: scripts/run_hotpath_warp_abba.sh --baseline-bin --candidate-bin [options] + +Formal ABBA mode for Linux runners or production-like clusters. The schedule is +A1 baseline -> B1 candidate -> B2 candidate -> A2 baseline for every workload +and drive-sync cell. + +Required: + --baseline-bin Baseline RustFS binary. + --candidate-bin Candidate RustFS binary. + +Local Linux runner mode: + --address Local RustFS address (default 127.0.0.1:9000). + --disks Throwaway local disks per node (default 4). + --data-root Local disk root (default /tmp/rustfs-hotpath-abba). + +Production / cluster mode: + --endpoint Existing cluster endpoint. Enables external mode. + --deploy-hook Command run before each ABBA leg. It receives: + HOTPATH_ABBA_LEG=A1|B1|B2|A2 + HOTPATH_ABBA_PHASE=baseline|candidate + HOTPATH_ABBA_BINARY= + HOTPATH_ABBA_DRIVE_SYNC=true|false + --health-path Readiness path (default /health). + +Benchmark: + --duration warp duration per cell (default 60s). + --rounds rounds per cell; must be >= 3 (default 3). + --cooldown cooldown seconds between rounds/sizes (default 20). + --concurrency warp concurrency (default 8). + --warp-bin warp binary (default warp). + +Credentials: + --access-key S3 access key (default rustfsadmin). + --secret-key S3 secret key (default rustfsadmin). + --region S3 region (default us-east-1). + +Gate: + --fail-pct Regression budget that fails gate (default 10). + --warn-pct Regression budget that warns (default 5). + --allow-regression Downgrade candidate gate FAIL to WARN. + --exemption-reason Reason recorded when allow-regression is used. + +Output: + --out-dir Output dir (default target/hotpath-abba/). + --dry-run Print commands without starting servers or warp. + -h, --help + +Outputs: + /abba_schedule.csv + /candidate_gate.md + /baseline_drift_gate.md + /summary.md + ////{median_summary.csv,baseline_compare.csv} +USAGE +} + +die() { + echo "error: $*" >&2 + exit 2 +} + +log() { + printf '[hotpath-abba] %s\n' "$*" >&2 +} + +run() { + if [[ "$DRY_RUN" == "true" ]]; then + { printf 'DRY-RUN:'; printf ' %q' "$@"; printf '\n'; } >&2 + return 0 + fi + "$@" +} + +validate_positive_int() { + local value="$1" name="$2" + [[ "$value" =~ ^[0-9]+$ && "$value" -gt 0 ]] || die "$name must be a positive integer" +} + +while [[ $# -gt 0 ]]; do + case "$1" in + --baseline-bin) BASELINE_BIN="$2"; shift 2 ;; + --candidate-bin) CANDIDATE_BIN="$2"; shift 2 ;; + --endpoint) ENDPOINT="$2"; shift 2 ;; + --deploy-hook) DEPLOY_HOOK="$2"; shift 2 ;; + --health-path) HEALTH_PATH="$2"; shift 2 ;; + --address) ADDRESS="$2"; shift 2 ;; + --data-root) DATA_ROOT="$2"; shift 2 ;; + --disks) DISKS="$2"; shift 2 ;; + --access-key) ACCESS_KEY="$2"; shift 2 ;; + --secret-key) SECRET_KEY="$2"; shift 2 ;; + --region) REGION="$2"; shift 2 ;; + --warp-bin) WARP_BIN="$2"; shift 2 ;; + --concurrency) CONCURRENCY="$2"; shift 2 ;; + --duration) DURATION="$2"; shift 2 ;; + --rounds) ROUNDS="$2"; shift 2 ;; + --cooldown) COOLDOWN_SECS="$2"; shift 2 ;; + --health-timeout) HEALTH_TIMEOUT_SECS="$2"; shift 2 ;; + --fail-pct) FAIL_PCT="$2"; shift 2 ;; + --warn-pct) WARN_PCT="$2"; shift 2 ;; + --allow-regression) ALLOW_REGRESSION=true; shift ;; + --exemption-reason) EXEMPTION_REASON="$2"; shift 2 ;; + --out-dir) OUT_DIR="$2"; shift 2 ;; + --dry-run) DRY_RUN=true; shift ;; + -h|--help) usage; exit 0 ;; + *) die "unknown argument: $1" ;; + esac +done + +validate_positive_int "$DISKS" "--disks" +validate_positive_int "$CONCURRENCY" "--concurrency" +validate_positive_int "$ROUNDS" "--rounds" +validate_positive_int "$COOLDOWN_SECS" "--cooldown" +validate_positive_int "$HEALTH_TIMEOUT_SECS" "--health-timeout" +validate_positive_int "$FAIL_PCT" "--fail-pct" +validate_positive_int "$WARN_PCT" "--warn-pct" +[[ "$ROUNDS" -ge 3 ]] || die "--rounds must be >= 3 for formal ABBA evidence" +[[ -n "$BASELINE_BIN" ]] || die "--baseline-bin is required" +[[ -n "$CANDIDATE_BIN" ]] || die "--candidate-bin is required" +[[ "$DRY_RUN" == "true" || -x "$BASELINE_BIN" ]] || die "baseline binary is not executable: $BASELINE_BIN" +[[ "$DRY_RUN" == "true" || -x "$CANDIDATE_BIN" ]] || die "candidate binary is not executable: $CANDIDATE_BIN" +[[ -x "$ENHANCED_BENCH" ]] || die "missing load driver: $ENHANCED_BENCH" +[[ -x "$GATE" ]] || die "missing gate: $GATE" +if [[ "$DRY_RUN" != "true" ]] && ! command -v "$WARP_BIN" >/dev/null 2>&1; then + die "warp not found on PATH; install warp or pass --warp-bin" +fi + +EXTERNAL=false +if [[ -n "$ENDPOINT" ]]; then + EXTERNAL=true + ADDRESS="$ENDPOINT" + [[ -n "$DEPLOY_HOOK" ]] || log "warning: external mode without --deploy-hook; binaries must be swapped out of band" +fi + +mkdir -p "$OUT_DIR" +SERVER_LOG_DIR="$OUT_DIR/server-logs" +run mkdir -p "$SERVER_LOG_DIR" + +SERVER_PID="" +SERVER_LOG="" +tear_down() { + [[ "$EXTERNAL" == "true" ]] && return 0 + [[ -n "$SERVER_PID" ]] || return 0 + run kill "$SERVER_PID" 2>/dev/null || true + SERVER_PID="" +} +trap tear_down EXIT INT TERM + +dump_server_log() { + [[ -n "$SERVER_LOG" && -f "$SERVER_LOG" ]] || return 0 + echo "----- last 80 lines of $SERVER_LOG -----" >&2 + tail -n 80 "$SERVER_LOG" >&2 || true + echo "----------------------------------------" >&2 +} + +wait_health() { + [[ "$DRY_RUN" == "true" ]] && return 0 + local i + for ((i = 0; i < HEALTH_TIMEOUT_SECS; i++)); do + if [[ "$EXTERNAL" != "true" && -n "$SERVER_PID" ]] && ! kill -0 "$SERVER_PID" 2>/dev/null; then + echo "error: rustfs server (pid $SERVER_PID) exited before becoming healthy after ${i}s" >&2 + dump_server_log + return 1 + fi + if curl -fsS "http://${ADDRESS}${HEALTH_PATH}" >/dev/null 2>&1; then + return 0 + fi + sleep 1 + done + echo "error: endpoint http://${ADDRESS}${HEALTH_PATH} did not become healthy within ${HEALTH_TIMEOUT_SECS}s" >&2 + dump_server_log + return 1 +} + +binary_for_leg() { + case "$1" in + A1|A2) echo "$BASELINE_BIN" ;; + B1|B2) echo "$CANDIDATE_BIN" ;; + *) die "unknown ABBA leg: $1" ;; + esac +} + +phase_for_leg() { + case "$1" in + A1|A2) echo "baseline" ;; + B1|B2) echo "candidate" ;; + *) die "unknown ABBA leg: $1" ;; + esac +} + +bring_up() { + local leg="$1" drive_sync="$2" + local phase bin + phase="$(phase_for_leg "$leg")" + bin="$(binary_for_leg "$leg")" + + if [[ "$EXTERNAL" == "true" ]]; then + if [[ -n "$DEPLOY_HOOK" ]]; then + log "deploy hook: leg=$leg phase=$phase drive_sync=$drive_sync" + HOTPATH_ABBA_LEG="$leg" HOTPATH_ABBA_PHASE="$phase" HOTPATH_ABBA_BINARY="$bin" HOTPATH_ABBA_DRIVE_SYNC="$drive_sync" \ + run bash -c "$DEPLOY_HOOK" + fi + wait_health + return 0 + fi + + local node_dir="$DATA_ROOT/$leg-sync-$drive_sync" + local disks=() d + for ((d = 1; d <= DISKS; d++)); do + disks+=("$node_dir/d$d") + done + run mkdir -p "${disks[@]}" + SERVER_LOG="$SERVER_LOG_DIR/$leg-sync-$drive_sync.log" + + if [[ "$DRY_RUN" == "true" ]]; then + { printf 'DRY-RUN: RUSTFS_DRIVE_SYNC_ENABLE=%s %q server' "$drive_sync" "$bin" + printf ' %q' "${disks[@]}"; printf '\n'; } >&2 + SERVER_PID="dry-run" + return 0 + fi + + cat >"$SERVER_LOG_DIR/$leg-sync-$drive_sync.env" </dev/null || echo unknown) +warp_version=$("$WARP_BIN" --version 2>/dev/null | head -n1 || echo unknown) +EOF + RUSTFS_UNSAFE_BYPASS_DISK_CHECK=true \ + RUSTFS_ADDRESS="$ADDRESS" \ + RUSTFS_ACCESS_KEY="$ACCESS_KEY" \ + RUSTFS_SECRET_KEY="$SECRET_KEY" \ + RUSTFS_REGION="$REGION" \ + RUSTFS_CONSOLE_ENABLE=false \ + RUSTFS_DRIVE_SYNC_ENABLE="$drive_sync" \ + "$bin" server "${disks[@]}" >"$SERVER_LOG" 2>&1 & + SERVER_PID=$! + wait_health +} + +measure() { + local leg="$1" workload="$2" mode="$3" size="$4" sync_label="$5" baseline_csv="${6:-}" + local cell="$OUT_DIR/$workload/$sync_label/$leg" + local args=( + --tool warp --warp-bin "$WARP_BIN" --warp-mode "$mode" + --endpoint "$ADDRESS" --access-key "$ACCESS_KEY" --secret-key "$SECRET_KEY" + --region "$REGION" --sizes "$size" --concurrency "$CONCURRENCY" + --duration "$DURATION" --rounds "$ROUNDS" --cooldown-secs "$COOLDOWN_SECS" + --out-dir "$cell" + ) + [[ -n "$baseline_csv" ]] && args+=(--baseline-csv "$baseline_csv") + run "$ENHANCED_BENCH" "${args[@]}" >&2 + echo "$cell" +} + +write_schedule_header() { + echo "sync_label,drive_sync,workload,mode,size,leg,phase,binary,out_dir" >"$OUT_DIR/abba_schedule.csv" +} + +append_schedule() { + local sync_label="$1" drive_sync="$2" workload="$3" mode="$4" size="$5" leg="$6" + local phase bin + phase="$(phase_for_leg "$leg")" + bin="$(binary_for_leg "$leg")" + echo "$sync_label,$drive_sync,$workload,$mode,$size,$leg,$phase,$bin,$OUT_DIR/$workload/$sync_label/$leg" >>"$OUT_DIR/abba_schedule.csv" +} + +write_manifest() { + cat >"$OUT_DIR/manifest.env" </dev/null || echo unknown) +runner=$(uname -srm 2>/dev/null || echo unknown) +schedule=ABBA +rounds=$ROUNDS +duration=$DURATION +cooldown_secs=$COOLDOWN_SECS +concurrency=$CONCURRENCY +baseline_bin=$BASELINE_BIN +candidate_bin=$CANDIDATE_BIN +external=$EXTERNAL +endpoint=$ADDRESS +warp_version=$("$WARP_BIN" --version 2>/dev/null | head -n1 || echo unknown) +EOF +} + +declare -a CANDIDATE_COMPARE_CSVS=() +declare -a DRIFT_COMPARE_CSVS=() + +write_manifest +write_schedule_header + +for ds_spec in "${DRIVE_SYNC_MATRIX[@]}"; do + IFS='|' read -r sync_label drive_sync <<<"$ds_spec" + + for leg in A1 B1 B2 A2; do + log "=== $sync_label leg $leg ($(phase_for_leg "$leg")) ===" + bring_up "$leg" "$drive_sync" + + for wl_spec in "${WORKLOADS[@]}"; do + IFS='|' read -r workload mode size <<<"$wl_spec" + append_schedule "$sync_label" "$drive_sync" "$workload" "$mode" "$size" "$leg" + + baseline_csv="" + if [[ "$leg" != "A1" ]]; then + baseline_csv="$OUT_DIR/$workload/$sync_label/A1/median_summary.csv" + fi + + cell="$(measure "$leg" "$workload" "$mode" "$size" "$sync_label" "$baseline_csv")" + case "$leg" in + B1|B2) CANDIDATE_COMPARE_CSVS+=("$cell/baseline_compare.csv") ;; + A2) DRIFT_COMPARE_CSVS+=("$cell/baseline_compare.csv") ;; + esac + done + + tear_down + done +done + +gate_args=(--fail-pct "$FAIL_PCT" --warn-pct "$WARN_PCT" --markdown "$OUT_DIR/candidate_gate.md") +for csv in "${CANDIDATE_COMPARE_CSVS[@]}"; do + gate_args+=(--compare-csv "$csv") +done +[[ "$ALLOW_REGRESSION" == "true" ]] && gate_args+=(--allow-regression --exemption-reason "$EXEMPTION_REASON") + +drift_gate_args=(--fail-pct "$FAIL_PCT" --warn-pct "$WARN_PCT" --markdown "$OUT_DIR/baseline_drift_gate.md") +for csv in "${DRIFT_COMPARE_CSVS[@]}"; do + drift_gate_args+=(--compare-csv "$csv") +done + +if [[ "$DRY_RUN" == "true" ]]; then + log "dry-run complete; candidate compare CSVs=${#CANDIDATE_COMPARE_CSVS[@]} baseline drift CSVs=${#DRIFT_COMPARE_CSVS[@]}" + { printf 'DRY-RUN:'; printf ' %q' "$GATE" "${gate_args[@]}"; printf '\n'; } >&2 + { printf 'DRY-RUN:'; printf ' %q' "$GATE" "${drift_gate_args[@]}"; printf '\n'; } >&2 + exit 0 +fi + +log "applying candidate relative-budget gate" +set +e +"$GATE" "${gate_args[@]}" +candidate_status=$? +set -e + +log "applying A2-vs-A1 baseline drift gate" +set +e +"$GATE" "${drift_gate_args[@]}" +drift_status=$? +set -e + +cat >"$OUT_DIR/summary.md" < B1 candidate -> B2 candidate -> A2 baseline +- runner: $(uname -srm 2>/dev/null || echo unknown) +- warp: $("$WARP_BIN" --version 2>/dev/null | head -n1 || echo unknown) +- matrix: duration=$DURATION rounds=$ROUNDS cooldown=$COOLDOWN_SECS disks=$DISKS concurrency=$CONCURRENCY +- endpoint: $ADDRESS +- baseline binary: $BASELINE_BIN +- candidate binary: $CANDIDATE_BIN +- candidate gate: $OUT_DIR/candidate_gate.md (exit $candidate_status) +- baseline drift gate: $OUT_DIR/baseline_drift_gate.md (exit $drift_status) + +Interpretation: + +- Treat candidate gate failures as actionable only when the A2-vs-A1 drift gate is PASS or the affected workload's A2 drift is materially smaller than the B1/B2 candidate delta. +- If both candidate and baseline drift fail on the same workload, rerun with longer duration, more rounds, or a quieter runner before assigning causality. +EOF + +log "summary written to $OUT_DIR/summary.md" +if [[ "$candidate_status" -ne 0 || "$drift_status" -ne 0 ]]; then + exit 1 +fi