mirror of
https://github.com/rustfs/rustfs.git
synced 2026-07-26 08:18:18 +00:00
46d7f9e1f2
* feat(get): consolidate GET performance optimization Consolidated implementation of all GET performance optimizations into a single, well-organized commit replacing the previous patch-on-patch approach. ## Changes ### Configuration (set_disk/mod.rs) - Consolidated all GET optimization flags into a single organized section - Enabled by default: codec streaming, metadata early-stop, page cache reclaim - Added codec streaming multipart flag (default: disabled) - Added version-aware early-stop flag (default: disabled) - Added adaptive duplex buffer sizing based on object size - All flags use OnceLock caching with rollout percentage support ### Metadata Early-Stop (set_disk/read.rs) - Delete marker early-stop when quorum agrees - Version-aware early-stop for versioned GET requests - MetadataQuorumAccumulator enhanced with: - delete_marker_votes tracking - requested_version_id and matching_version_votes tracking - version_early_stop_decision() method - 6 new tests for version early-stop scenarios ### Codec Streaming (erasure/coding/decode_reader.rs) - DualInFlight (2-stripe lookahead) enabled by default ### Decode Pipeline (erasure/coding/decode.rs) - Stripe prefetch count configuration - Bitrot-decode overlap configuration ### Disk Layer (disk/local.rs) - O_DIRECT read configuration constants (preparation) ### Metrics (io-metrics/lib.rs) - BytesPool acquisition/return metrics - Metadata phase duration with early-stop label - Total duration with reader_path label ### Diagnostics (diagnostics/) - Early-stop reason constants - Pool tier/outcome label constants ### Observability (.docker/observability/) - 3 Grafana dashboards for GET optimization monitoring - Prometheus alert rules (6 alerts: 3 critical, 3 warning) - Updated README.md and README_ZH.md with usage docs ### Config (config/src/constants/runtime.rs) - Page cache reclaim read enabled by default ## Environment Variables | Variable | Default | Description | |----------|---------|-------------| | RUSTFS_GET_CODEC_STREAMING_ENABLE | true | Codec streaming base flag | | RUSTFS_GET_CODEC_STREAMING_ROLLOUT_PCT | 100 | Codec streaming rollout % | | RUSTFS_GET_CODEC_STREAMING_MULTIPART_ENABLE | false | Multipart codec streaming | | RUSTFS_GET_METADATA_EARLY_STOP_ENABLE | true | Early-stop base flag | | RUSTFS_GET_METADATA_EARLY_STOP_ROLLOUT_PCT | 100 | Early-stop rollout % | | RUSTFS_GET_METADATA_VERSION_EARLY_STOP_ENABLE | false | Version-aware early-stop | | RUSTFS_OBJECT_FILE_CACHE_RECLAIM_READ_ENABLE | true | Page cache reclaim | | RUSTFS_OBJECT_DIRECT_IO_READ_ENABLE | false | O_DIRECT (preparation) | | RUSTFS_GET_DECODE_STRIPE_PREFETCH_COUNT | 1 | Stripe prefetch | | RUSTFS_GET_BITROT_DECODE_OVERLAP_ENABLE | false | Bitrot-decode overlap | | RUSTFS_GET_CODEC_STREAMING_MAX_INFLIGHT | 2 | DualInFlight stripes | ## Rollback All optimizations can be disabled via environment variables: RUSTFS_GET_CODEC_STREAMING_ENABLE=false RUSTFS_GET_METADATA_EARLY_STOP_ENABLE=false RUSTFS_OBJECT_FILE_CACHE_RECLAIM_READ_ENABLE=false Co-Authored-By: heihutu <heihutu@gmail.com> * test(get): add stress test scripts for GET optimization validation - quick-validate-get-optimization.sh: Quick 5-minute validation - stress-test-get-optimization.sh: Full 30+ minute stress test - README-stress-test.md: Usage documentation Co-Authored-By: heihutu <heihutu@gmail.com> * test(ecstore): align file cache reclaim defaults * chore(deps): update redis and erasure codec * test(ecstore): align decode fill policy default * fix(get): wire codec streaming rollout gate * perf(get): skip metrics-off codec timers * test(get): capture codec streaming diagnostics * test(get): add multipart fallback probe * test(get): add encrypted fallback probe * test(get): add compressed fallback probe * test(get): add degraded read fallback probe * test(get): cover remote fallback probe * test(get): report warp request p99 * test(get): capture OTLP metric deltas * perf(get): align codec streaming inflight default * perf(get): reuse codec reader output buffers * test(get): count codec reader fill starts * perf(get): reuse codec reader fill worker * perf(get): lazy init rustfs codec reconstruct * test(get): cover rustfs codec source faults * docs(get): record rustfs codec fallback scope * feat(get): add multipart codec reader opt-in * test(get): add multipart codec smoke option * test(get): cover multipart codec degraded fallback * perf(get): bound multipart codec eager setup * test(get): satisfy codec hardening PR gate --------- Co-authored-by: heihutu <heihutu@gmail.com>
622 lines
18 KiB
Bash
Executable File
622 lines
18 KiB
Bash
Executable File
#!/usr/bin/env bash
|
|
set -euo pipefail
|
|
|
|
# Enhanced batch object benchmark runner for warp/s3bench:
|
|
# - Multi-round execution (default 3 rounds)
|
|
# - Retry failed round attempts automatically
|
|
# - Median aggregation per object size
|
|
# - Optional baseline CSV comparison
|
|
|
|
DEFAULT_SIZES="1KiB,4KiB,8KiB,16KiB,32KiB,100KiB,512KiB,1MiB,2MiB,5MiB,10MiB"
|
|
|
|
TOOL="warp"
|
|
ENDPOINT=""
|
|
ACCESS_KEY=""
|
|
SECRET_KEY=""
|
|
BUCKET="rustfs-bench"
|
|
REGION="us-east-1"
|
|
CONCURRENCY=128
|
|
SIZES="$DEFAULT_SIZES"
|
|
OUT_DIR=""
|
|
INSECURE=false
|
|
DRY_RUN=false
|
|
|
|
# warp options
|
|
WARP_BIN="warp"
|
|
WARP_MODE="mixed"
|
|
DURATION="60s"
|
|
|
|
# s3bench options
|
|
S3BENCH_BIN="s3bench"
|
|
SAMPLES=20000
|
|
|
|
# enhancement options
|
|
ROUNDS=3
|
|
RETRY_PER_ROUND=2
|
|
RETRY_SLEEP_SECS=2
|
|
COOLDOWN_SECS=0
|
|
BASELINE_CSV=""
|
|
EXTRA_ARGS=()
|
|
FAILED_FINAL_ROUNDS=0
|
|
|
|
usage() {
|
|
cat <<'USAGE'
|
|
Usage:
|
|
scripts/run_object_batch_bench_enhanced.sh --tool <warp|s3bench> --endpoint <host:port|url> \
|
|
--access-key <ak> --secret-key <sk> [options]
|
|
|
|
Required:
|
|
--tool warp | s3bench
|
|
--endpoint S3 endpoint
|
|
--access-key S3 access key
|
|
--secret-key S3 secret key
|
|
|
|
Core options:
|
|
--bucket Bucket name (default: rustfs-bench)
|
|
--region Region (default: us-east-1)
|
|
--concurrency Concurrency for all sizes (default: 128)
|
|
--sizes Comma-separated sizes (default: 1KiB..10MiB matrix)
|
|
--out-dir Output directory (default: target/bench/object-batch-enhanced-<timestamp>)
|
|
--insecure Allow insecure TLS
|
|
--dry-run Print commands only, do not execute
|
|
|
|
Warp options:
|
|
--warp-bin warp binary (default: warp)
|
|
--warp-mode get|put|mixed (default: mixed)
|
|
--duration e.g. 60s/2m (default: 60s)
|
|
|
|
s3bench options:
|
|
--s3bench-bin s3bench binary (default: s3bench)
|
|
--samples numSamples (default: 20000)
|
|
|
|
Enhanced options:
|
|
--rounds Benchmark rounds per size (default: 3)
|
|
--retry-per-round Retry count per failed round (default: 2)
|
|
--retry-sleep-secs Sleep seconds between retries (default: 2)
|
|
--cooldown-secs Sleep seconds between rounds/sizes (default: 0)
|
|
--round-cooldown-secs Compatibility alias for --cooldown-secs
|
|
--baseline-csv Baseline median CSV to compare
|
|
--extra-args Extra args appended to tool command, quoted as one string
|
|
|
|
Output files:
|
|
round_results.csv One row per round attempt (with retry trace)
|
|
median_summary.csv Median metrics per object size
|
|
baseline_compare.csv Delta vs baseline (if --baseline-csv is set)
|
|
|
|
Example:
|
|
scripts/run_object_batch_bench_enhanced.sh \
|
|
--tool warp --endpoint http://127.0.0.1:9000 \
|
|
--access-key minioadmin --secret-key minioadmin \
|
|
--bucket bench-obj --concurrency 128 --duration 90s \
|
|
--rounds 3 --retry-per-round 2 --baseline-csv old/median_summary.csv
|
|
USAGE
|
|
}
|
|
|
|
require_cmd() {
|
|
if ! command -v "$1" >/dev/null 2>&1; then
|
|
echo "ERROR: command not found: $1" >&2
|
|
exit 1
|
|
fi
|
|
}
|
|
|
|
normalize_warp_host() {
|
|
local raw="$1"
|
|
raw="${raw#http://}"
|
|
raw="${raw#https://}"
|
|
raw="${raw%%/*}"
|
|
raw="${raw%%\?*}"
|
|
raw="${raw%%\#*}"
|
|
echo "$raw"
|
|
}
|
|
|
|
parse_args() {
|
|
while [[ $# -gt 0 ]]; do
|
|
case "$1" in
|
|
--tool) TOOL="$2"; shift 2 ;;
|
|
--endpoint) ENDPOINT="$2"; shift 2 ;;
|
|
--access-key) ACCESS_KEY="$2"; shift 2 ;;
|
|
--secret-key) SECRET_KEY="$2"; shift 2 ;;
|
|
--bucket) BUCKET="$2"; shift 2 ;;
|
|
--region) REGION="$2"; shift 2 ;;
|
|
--concurrency) CONCURRENCY="$2"; shift 2 ;;
|
|
--sizes) SIZES="$2"; shift 2 ;;
|
|
--out-dir) OUT_DIR="$2"; shift 2 ;;
|
|
--insecure) INSECURE=true; shift ;;
|
|
--dry-run) DRY_RUN=true; shift ;;
|
|
--warp-bin) WARP_BIN="$2"; shift 2 ;;
|
|
--warp-mode) WARP_MODE="$2"; shift 2 ;;
|
|
--duration) DURATION="$2"; shift 2 ;;
|
|
--s3bench-bin) S3BENCH_BIN="$2"; shift 2 ;;
|
|
--samples) SAMPLES="$2"; shift 2 ;;
|
|
--rounds) ROUNDS="$2"; shift 2 ;;
|
|
--retry-per-round) RETRY_PER_ROUND="$2"; shift 2 ;;
|
|
--retry-sleep-secs) RETRY_SLEEP_SECS="$2"; shift 2 ;;
|
|
--cooldown-secs) COOLDOWN_SECS="$2"; shift 2 ;;
|
|
--round-cooldown-secs) COOLDOWN_SECS="$2"; shift 2 ;;
|
|
--baseline-csv) BASELINE_CSV="$2"; shift 2 ;;
|
|
--extra-args)
|
|
# shellcheck disable=SC2206
|
|
EXTRA_ARGS=($2)
|
|
shift 2
|
|
;;
|
|
-h|--help) usage; exit 0 ;;
|
|
*)
|
|
echo "ERROR: unknown arg: $1" >&2
|
|
usage
|
|
exit 1
|
|
;;
|
|
esac
|
|
done
|
|
}
|
|
|
|
validate_positive_int() {
|
|
local v="$1"
|
|
local n="$2"
|
|
if ! [[ "$v" =~ ^[0-9]+$ ]] || [[ "$v" -le 0 ]]; then
|
|
echo "ERROR: $n must be a positive integer, got: $v" >&2
|
|
exit 1
|
|
fi
|
|
}
|
|
|
|
validate_nonnegative_int() {
|
|
local v="$1"
|
|
local n="$2"
|
|
if ! [[ "$v" =~ ^[0-9]+$ ]]; then
|
|
echo "ERROR: $n must be a nonnegative integer, got: $v" >&2
|
|
exit 1
|
|
fi
|
|
}
|
|
|
|
cooldown_sleep() {
|
|
local secs="$1"
|
|
if (( secs <= 0 )); then
|
|
return 0
|
|
fi
|
|
if [[ "$DRY_RUN" == "true" ]]; then
|
|
echo "[DRY-RUN] sleep ${secs}"
|
|
else
|
|
sleep "$secs"
|
|
fi
|
|
}
|
|
|
|
validate_args() {
|
|
if [[ "$TOOL" != "warp" && "$TOOL" != "s3bench" ]]; then
|
|
echo "ERROR: --tool must be warp or s3bench" >&2
|
|
exit 1
|
|
fi
|
|
if [[ -z "$ENDPOINT" || -z "$ACCESS_KEY" || -z "$SECRET_KEY" ]]; then
|
|
echo "ERROR: --endpoint/--access-key/--secret-key are required" >&2
|
|
exit 1
|
|
fi
|
|
validate_positive_int "$CONCURRENCY" "--concurrency"
|
|
validate_positive_int "$ROUNDS" "--rounds"
|
|
validate_positive_int "$RETRY_PER_ROUND" "--retry-per-round"
|
|
validate_positive_int "$RETRY_SLEEP_SECS" "--retry-sleep-secs"
|
|
validate_nonnegative_int "$COOLDOWN_SECS" "--cooldown-secs"
|
|
if [[ "$TOOL" == "s3bench" ]]; then
|
|
validate_positive_int "$SAMPLES" "--samples"
|
|
fi
|
|
if [[ -n "$BASELINE_CSV" && ! -f "$BASELINE_CSV" ]]; then
|
|
echo "ERROR: --baseline-csv does not exist: $BASELINE_CSV" >&2
|
|
exit 1
|
|
fi
|
|
if [[ "$TOOL" == "warp" ]]; then
|
|
local warp_host
|
|
warp_host="$(normalize_warp_host "$ENDPOINT")"
|
|
if [[ -z "$warp_host" ]]; then
|
|
echo "ERROR: invalid --endpoint for warp: $ENDPOINT" >&2
|
|
exit 1
|
|
fi
|
|
fi
|
|
}
|
|
|
|
setup_output() {
|
|
if [[ -z "$OUT_DIR" ]]; then
|
|
OUT_DIR="target/bench/object-batch-enhanced-$(date +%Y%m%d-%H%M%S)"
|
|
fi
|
|
mkdir -p "$OUT_DIR/logs"
|
|
|
|
ROUND_CSV="$OUT_DIR/round_results.csv"
|
|
MEDIAN_CSV="$OUT_DIR/median_summary.csv"
|
|
COMPARE_CSV="$OUT_DIR/baseline_compare.csv"
|
|
|
|
echo "size,tool,round,attempt,concurrency,status,exit_code,round_started_at_utc,round_finished_at_utc,throughput_human,throughput_bps,reqps,latency_human,latency_ms,log_file,req_p90_human,req_p90_ms,req_p99_human,req_p99_ms" > "$ROUND_CSV"
|
|
echo "size,tool,concurrency,successful_rounds,failed_rounds,median_throughput_bps,median_reqps,median_latency_ms,median_req_p90_ms,median_req_p99_ms" > "$MEDIAN_CSV"
|
|
}
|
|
|
|
trim() {
|
|
echo "$1" | awk '{$1=$1;print}'
|
|
}
|
|
|
|
to_bps() {
|
|
local human="$1"
|
|
local number unit factor
|
|
if [[ "$human" == "N/A" || -z "$human" ]]; then
|
|
echo "N/A"
|
|
return
|
|
fi
|
|
|
|
# Normalize "123MiB/s" → "123 MiB/s" when no space separator exists.
|
|
human="$(echo "$human" | sed -E 's/^([0-9]+(\.[0-9]+)?)([A-Za-zµ])/\1 \3/')"
|
|
number="$(echo "$human" | awk '{print $1}')"
|
|
unit="$(echo "$human" | awk '{print $2}')"
|
|
case "$unit" in
|
|
GiB/s) factor=1073741824 ;;
|
|
MiB/s) factor=1048576 ;;
|
|
KiB/s) factor=1024 ;;
|
|
GB/s) factor=1000000000 ;;
|
|
MB/s) factor=1000000 ;;
|
|
KB/s) factor=1000 ;;
|
|
B/s) factor=1 ;;
|
|
*)
|
|
echo "N/A"
|
|
return
|
|
;;
|
|
esac
|
|
|
|
awk -v n="$number" -v f="$factor" 'BEGIN { printf "%.6f\n", n * f }'
|
|
}
|
|
|
|
to_ms() {
|
|
local human="$1"
|
|
local number unit factor
|
|
if [[ "$human" == "N/A" || -z "$human" ]]; then
|
|
echo "N/A"
|
|
return
|
|
fi
|
|
|
|
# Normalize "50ms" → "50 ms" when no space separator exists.
|
|
human="$(echo "$human" | sed -E 's/^([0-9]+(\.[0-9]+)?)([A-Za-zµ])/\1 \3/')"
|
|
number="$(echo "$human" | awk '{print $1}')"
|
|
unit="$(echo "$human" | awk '{print $2}')"
|
|
case "$unit" in
|
|
s) factor=1000 ;;
|
|
ms) factor=1 ;;
|
|
us|µs) factor=0.001 ;;
|
|
*)
|
|
echo "N/A"
|
|
return
|
|
;;
|
|
esac
|
|
|
|
awk -v n="$number" -v f="$factor" 'BEGIN { printf "%.6f\n", n * f }'
|
|
}
|
|
|
|
extract_first() {
|
|
local regex="$1"
|
|
local file="$2"
|
|
rg -o "$regex" "$file" | head -n1 || true
|
|
}
|
|
|
|
normalize_duration_metric() {
|
|
local value="$1"
|
|
value="$(trim "${value:-N/A}")"
|
|
|
|
if [[ -z "$value" || "$value" == "N/A" ]]; then
|
|
echo "N/A"
|
|
return
|
|
fi
|
|
|
|
if [[ "$value" =~ ^([0-9]+(\.[0-9]+)?)(ms|us|µs|s)$ ]]; then
|
|
echo "${BASH_REMATCH[1]} ${BASH_REMATCH[3]}"
|
|
return
|
|
fi
|
|
|
|
echo "$value" | awk '{ if ($1 == "" || $1 == "N/A") print "N/A"; else print $1" "$2 }'
|
|
}
|
|
|
|
extract_metrics() {
|
|
local log_file="$1"
|
|
|
|
local throughput reqps latency req_p90 req_p99
|
|
throughput="$(extract_first '[0-9]+(\.[0-9]+)?[[:space:]]*(GiB/s|MiB/s|KiB/s|GB/s|MB/s|KB/s|B/s)' "$log_file")"
|
|
reqps="$(extract_first '[0-9]+(\.[0-9]+)?[[:space:]]*(obj/s|req/s|ops/s|requests/s)' "$log_file")"
|
|
latency="$(rg -o 'Reqs:[[:space:]]+Avg:[[:space:]]+[0-9]+(\.[0-9]+)?(ms|us|µs|s)' "$log_file" | head -n1 | sed -E 's/^Reqs:[[:space:]]+Avg:[[:space:]]+//')"
|
|
req_p90="$(rg -o '90%:[[:space:]]+[0-9]+(\.[0-9]+)?(ms|us|µs|s)' "$log_file" | head -n1 | sed -E 's/^90%:[[:space:]]+//')"
|
|
req_p99="$(rg -o '99%:[[:space:]]+[0-9]+(\.[0-9]+)?(ms|us|µs|s)' "$log_file" | head -n1 | sed -E 's/^99%:[[:space:]]+//')"
|
|
|
|
if [[ -z "$latency" ]]; then
|
|
latency="$(extract_first '[0-9]+(\.[0-9]+)?[[:space:]]*(ms|us|µs|s)' "$log_file")"
|
|
fi
|
|
|
|
throughput="$(trim "${throughput:-N/A}")"
|
|
reqps="$(trim "${reqps:-N/A}")"
|
|
latency="$(trim "${latency:-N/A}")"
|
|
req_p90="$(trim "${req_p90:-N/A}")"
|
|
req_p99="$(trim "${req_p99:-N/A}")"
|
|
|
|
# Normalize to "<num> <unit>" shape for downstream conversion helpers.
|
|
latency="$(normalize_duration_metric "$latency")"
|
|
req_p90="$(normalize_duration_metric "$req_p90")"
|
|
req_p99="$(normalize_duration_metric "$req_p99")"
|
|
reqps_num="$(echo "$reqps" | awk '{print $1}')"
|
|
|
|
echo "$throughput,${reqps_num:-N/A},$latency,$req_p90,$req_p99"
|
|
}
|
|
|
|
median_from_numbers() {
|
|
local values="$1"
|
|
local count
|
|
count="$(printf '%s\n' "$values" | awk 'NF{c++} END{print c+0}')"
|
|
if [[ "$count" -eq 0 ]]; then
|
|
echo "N/A"
|
|
return
|
|
fi
|
|
|
|
printf '%s\n' "$values" | awk 'NF' | sort -n | awk '
|
|
{a[NR]=$1}
|
|
END{
|
|
n=NR
|
|
if (n==0) { print "N/A"; exit }
|
|
if (n%2==1) {
|
|
printf "%.6f\n", a[(n+1)/2]
|
|
} else {
|
|
printf "%.6f\n", (a[n/2]+a[n/2+1])/2
|
|
}
|
|
}'
|
|
}
|
|
|
|
run_one_attempt() {
|
|
local size="$1"
|
|
local round="$2"
|
|
local attempt="$3"
|
|
local log_file="$OUT_DIR/logs/${TOOL}_${size}_r${round}_a${attempt}.log"
|
|
local status="ok"
|
|
local exit_code=0
|
|
local started_at_utc finished_at_utc
|
|
started_at_utc="$(date -u +%Y-%m-%dT%H:%M:%SZ)"
|
|
|
|
if [[ "$TOOL" == "warp" ]]; then
|
|
local warp_host
|
|
warp_host="$(normalize_warp_host "$ENDPOINT")"
|
|
local cmd=(
|
|
"$WARP_BIN" "$WARP_MODE"
|
|
"--host" "$warp_host"
|
|
"--access-key" "$ACCESS_KEY"
|
|
"--secret-key" "$SECRET_KEY"
|
|
"--bucket" "$BUCKET"
|
|
"--obj.size" "$size"
|
|
"--concurrent" "$CONCURRENCY"
|
|
"--duration" "$DURATION"
|
|
"--region" "$REGION"
|
|
)
|
|
if [[ "$INSECURE" == "true" ]]; then
|
|
cmd+=("--insecure")
|
|
fi
|
|
if [[ ${EXTRA_ARGS[@]+_} ]]; then
|
|
cmd+=("${EXTRA_ARGS[@]}")
|
|
fi
|
|
|
|
if [[ "$DRY_RUN" == "true" ]]; then
|
|
printf '[DRY-RUN] %q ' "${cmd[@]}"
|
|
printf '\n'
|
|
echo "dry run" > "$log_file"
|
|
else
|
|
if "${cmd[@]}" 2>&1 | tee "$log_file" >&2; then
|
|
:
|
|
else
|
|
exit_code=$?
|
|
status="failed"
|
|
fi
|
|
fi
|
|
else
|
|
local cmd=(
|
|
"$S3BENCH_BIN"
|
|
"-accessKey=$ACCESS_KEY"
|
|
"-secretKey=$SECRET_KEY"
|
|
"-bucket=$BUCKET"
|
|
"-endpoint=$ENDPOINT"
|
|
"-region=$REGION"
|
|
"-numClients=$CONCURRENCY"
|
|
"-numSamples=$SAMPLES"
|
|
"-objectSize=$size"
|
|
)
|
|
if [[ "$INSECURE" == "true" ]]; then
|
|
cmd+=("-insecure")
|
|
fi
|
|
if [[ ${EXTRA_ARGS[@]+_} ]]; then
|
|
cmd+=("${EXTRA_ARGS[@]}")
|
|
fi
|
|
|
|
if [[ "$DRY_RUN" == "true" ]]; then
|
|
printf '[DRY-RUN] %q ' "${cmd[@]}"
|
|
printf '\n'
|
|
echo "dry run" > "$log_file"
|
|
else
|
|
if "${cmd[@]}" 2>&1 | tee "$log_file" >&2; then
|
|
:
|
|
else
|
|
exit_code=$?
|
|
status="failed"
|
|
fi
|
|
fi
|
|
fi
|
|
finished_at_utc="$(date -u +%Y-%m-%dT%H:%M:%SZ)"
|
|
|
|
local metrics throughput_human reqps latency_human throughput_bps latency_ms req_p90_human req_p90_ms req_p99_human req_p99_ms
|
|
if [[ "$DRY_RUN" == "true" ]]; then
|
|
throughput_human="N/A"
|
|
reqps="N/A"
|
|
latency_human="N/A"
|
|
throughput_bps="N/A"
|
|
latency_ms="N/A"
|
|
req_p90_human="N/A"
|
|
req_p90_ms="N/A"
|
|
req_p99_human="N/A"
|
|
req_p99_ms="N/A"
|
|
else
|
|
metrics="$(extract_metrics "$log_file")"
|
|
throughput_human="$(echo "$metrics" | cut -d',' -f1)"
|
|
reqps="$(echo "$metrics" | cut -d',' -f2)"
|
|
latency_human="$(echo "$metrics" | cut -d',' -f3)"
|
|
req_p90_human="$(echo "$metrics" | cut -d',' -f4)"
|
|
req_p99_human="$(echo "$metrics" | cut -d',' -f5)"
|
|
throughput_bps="$(to_bps "$throughput_human")"
|
|
latency_ms="$(to_ms "$latency_human")"
|
|
req_p90_ms="$(to_ms "$req_p90_human")"
|
|
req_p99_ms="$(to_ms "$req_p99_human")"
|
|
fi
|
|
|
|
if [[ "$DRY_RUN" != "true" && "$status" == "ok" ]]; then
|
|
if [[ "$throughput_bps" == "N/A" && "$reqps" == "N/A" ]]; then
|
|
status="failed"
|
|
exit_code=1
|
|
fi
|
|
fi
|
|
|
|
echo "$size,$TOOL,$round,$attempt,$CONCURRENCY,$status,$exit_code,$started_at_utc,$finished_at_utc,$throughput_human,$throughput_bps,$reqps,$latency_human,$latency_ms,$log_file,$req_p90_human,$req_p90_ms,$req_p99_human,$req_p99_ms" >> "$ROUND_CSV"
|
|
echo "$status"
|
|
}
|
|
|
|
run_size() {
|
|
local size="$1"
|
|
local round success attempt rc max_attempts
|
|
max_attempts=$((RETRY_PER_ROUND + 1))
|
|
|
|
for ((round=1; round<=ROUNDS; round++)); do
|
|
success="no"
|
|
for ((attempt=1; attempt<=max_attempts; attempt++)); do
|
|
echo "==== size=$size round=$round attempt=$attempt/$max_attempts ===="
|
|
rc="$(run_one_attempt "$size" "$round" "$attempt")"
|
|
if [[ "$rc" == "ok" || "$DRY_RUN" == "true" ]]; then
|
|
success="yes"
|
|
break
|
|
fi
|
|
if (( attempt < RETRY_PER_ROUND+1 )); then
|
|
echo "Round failed, retry in ${RETRY_SLEEP_SECS}s..."
|
|
cooldown_sleep "$RETRY_SLEEP_SECS"
|
|
fi
|
|
done
|
|
|
|
if [[ "$success" == "no" ]]; then
|
|
echo "WARN: size=$size round=$round failed after retries."
|
|
FAILED_FINAL_ROUNDS=$((FAILED_FINAL_ROUNDS + 1))
|
|
fi
|
|
|
|
if (( COOLDOWN_SECS > 0 && round < ROUNDS )); then
|
|
echo "Cooldown after size=$size round=$round: ${COOLDOWN_SECS}s"
|
|
cooldown_sleep "$COOLDOWN_SECS"
|
|
fi
|
|
done
|
|
}
|
|
|
|
build_median_summary() {
|
|
local sizes_arr size
|
|
IFS=',' read -r -a sizes_arr <<< "$SIZES"
|
|
|
|
for raw in "${sizes_arr[@]}"; do
|
|
size="$(trim "$raw")"
|
|
[[ -z "$size" ]] && continue
|
|
|
|
local ok_rounds fail_rounds t_vals r_vals l_vals p90_vals p99_vals
|
|
ok_rounds="$(awk -F',' -v s="$size" 'NR>1 && $1==s && $6=="ok" {c++} END{print c+0}' "$ROUND_CSV")"
|
|
fail_rounds="$(awk -F',' -v s="$size" 'NR>1 && $1==s && $6!="ok" {c++} END{print c+0}' "$ROUND_CSV")"
|
|
|
|
t_vals="$(awk -F',' -v s="$size" 'NR>1 && $1==s && $6=="ok" && $11!="N/A" {print $11}' "$ROUND_CSV")"
|
|
r_vals="$(awk -F',' -v s="$size" 'NR>1 && $1==s && $6=="ok" && $12!="N/A" {print $12}' "$ROUND_CSV")"
|
|
l_vals="$(awk -F',' -v s="$size" 'NR>1 && $1==s && $6=="ok" && $14!="N/A" {print $14}' "$ROUND_CSV")"
|
|
p90_vals="$(awk -F',' -v s="$size" 'NR>1 && $1==s && $6=="ok" && $17!="N/A" {print $17}' "$ROUND_CSV")"
|
|
p99_vals="$(awk -F',' -v s="$size" 'NR>1 && $1==s && $6=="ok" && $19!="N/A" {print $19}' "$ROUND_CSV")"
|
|
|
|
local m_t m_r m_l m_p90 m_p99
|
|
m_t="$(median_from_numbers "$t_vals")"
|
|
m_r="$(median_from_numbers "$r_vals")"
|
|
m_l="$(median_from_numbers "$l_vals")"
|
|
m_p90="$(median_from_numbers "$p90_vals")"
|
|
m_p99="$(median_from_numbers "$p99_vals")"
|
|
|
|
echo "$size,$TOOL,$CONCURRENCY,$ok_rounds,$fail_rounds,$m_t,$m_r,$m_l,$m_p90,$m_p99" >> "$MEDIAN_CSV"
|
|
done
|
|
}
|
|
|
|
compare_baseline() {
|
|
if [[ -z "$BASELINE_CSV" ]]; then
|
|
return
|
|
fi
|
|
|
|
echo "size,tool,concurrency,new_median_reqps,baseline_median_reqps,delta_reqps_pct,new_median_latency_ms,baseline_median_latency_ms,delta_latency_pct,new_median_throughput_bps,baseline_median_throughput_bps,delta_throughput_pct" > "$COMPARE_CSV"
|
|
|
|
awk -F',' '
|
|
NR==FNR {
|
|
if (FNR==1) next
|
|
key=$1
|
|
b_req[key]=$7
|
|
b_lat[key]=$8
|
|
b_thr[key]=$6
|
|
next
|
|
}
|
|
FNR==1 {next}
|
|
{
|
|
key=$1
|
|
n_thr=$6; n_req=$7; n_lat=$8
|
|
br=(key in b_req)?b_req[key]:"N/A"
|
|
bl=(key in b_lat)?b_lat[key]:"N/A"
|
|
bt=(key in b_thr)?b_thr[key]:"N/A"
|
|
|
|
dr="N/A"; dl="N/A"; dt="N/A"
|
|
if (br!="N/A" && n_req!="N/A" && br+0!=0) dr=sprintf("%.2f", ((n_req-br)/br)*100)
|
|
if (bl!="N/A" && n_lat!="N/A" && bl+0!=0) dl=sprintf("%.2f", ((n_lat-bl)/bl)*100)
|
|
if (bt!="N/A" && n_thr!="N/A" && bt+0!=0) dt=sprintf("%.2f", ((n_thr-bt)/bt)*100)
|
|
|
|
print key "," $2 "," $3 "," n_req "," br "," dr "," n_lat "," bl "," dl "," n_thr "," bt "," dt
|
|
}
|
|
' "$BASELINE_CSV" "$MEDIAN_CSV" >> "$COMPARE_CSV"
|
|
}
|
|
|
|
main() {
|
|
parse_args "$@"
|
|
validate_args
|
|
require_cmd rg
|
|
require_cmd awk
|
|
require_cmd sort
|
|
if [[ "$TOOL" == "warp" ]]; then
|
|
require_cmd "$WARP_BIN"
|
|
else
|
|
require_cmd "$S3BENCH_BIN"
|
|
fi
|
|
|
|
setup_output
|
|
|
|
echo "Output dir: $OUT_DIR"
|
|
echo "Tool: $TOOL"
|
|
echo "Sizes: $SIZES"
|
|
echo "Concurrency: $CONCURRENCY"
|
|
echo "Rounds: $ROUNDS"
|
|
echo "Retry per round: $RETRY_PER_ROUND"
|
|
echo "Cooldown secs: $COOLDOWN_SECS"
|
|
|
|
IFS=',' read -r -a size_arr <<< "$SIZES"
|
|
local total_sizes="${#size_arr[@]}"
|
|
local size_index=0
|
|
for raw in "${size_arr[@]}"; do
|
|
size="$(trim "$raw")"
|
|
[[ -z "$size" ]] && continue
|
|
size_index=$(( size_index + 1 ))
|
|
run_size "$size"
|
|
if (( COOLDOWN_SECS > 0 && size_index < total_sizes )); then
|
|
echo "Cooldown after size=$size: ${COOLDOWN_SECS}s"
|
|
cooldown_sleep "$COOLDOWN_SECS"
|
|
fi
|
|
done
|
|
|
|
build_median_summary
|
|
compare_baseline
|
|
|
|
echo
|
|
echo "=== Median Summary ==="
|
|
cat "$MEDIAN_CSV"
|
|
|
|
if [[ -n "$BASELINE_CSV" ]]; then
|
|
echo
|
|
echo "=== Baseline Compare ==="
|
|
cat "$COMPARE_CSV"
|
|
fi
|
|
|
|
if [[ "$DRY_RUN" != "true" && "$FAILED_FINAL_ROUNDS" -gt 0 ]]; then
|
|
echo "ERROR: ${FAILED_FINAL_ROUNDS} benchmark rounds failed after retries." >&2
|
|
exit 1
|
|
fi
|
|
}
|
|
|
|
main "$@"
|