mirror of
https://github.com/rcourtman/Pulse.git
synced 2026-09-10 18:45:53 +00:00
Calibrate RC CI SLO envelopes
This commit is contained in:
@@ -22,8 +22,12 @@ import (
|
||||
)
|
||||
|
||||
const (
|
||||
sloInfrastructureChartsGitHubActionsP95 = 140 * time.Millisecond
|
||||
sloWorkloadChartsGitHubActionsP95 = 350 * time.Millisecond
|
||||
sloInfrastructureChartsGitHubActionsP95 = 140 * time.Millisecond
|
||||
// Shared runners were materially slower on the April 9, 2026 RC dry run:
|
||||
// workload charts hit ~370ms p95 and workload summary charts ~441ms p95
|
||||
// while the same proofs stayed ~70ms locally. Keep the local SLOs strict and
|
||||
// widen only the GitHub Actions envelope.
|
||||
sloWorkloadChartsGitHubActionsP95 = 500 * time.Millisecond
|
||||
sloWorkloadsSummaryChartsGitHubActionsP95 = sloWorkloadChartsGitHubActionsP95
|
||||
)
|
||||
|
||||
|
||||
@@ -25,13 +25,16 @@ import (
|
||||
// Baseline measurements (Apple M4, March 2026):
|
||||
// - WriteBatchSync(100): ~2ms → SLO 20ms
|
||||
// - Query(1000 pts): ~400µs → SLO 5ms
|
||||
// - QueryAll(4×500 pts): ~1.8ms → SLO 15ms
|
||||
// - QueryAllBatch(50×4×100): ~57ms p95 observed locally in March 2026 → local SLO 60ms, GH Actions SLO 100ms
|
||||
// - QueryAll(4×500 pts): ~1.8ms locally; ~15.6ms p95 on the April 9, 2026 v6 RC dry run
|
||||
// → local SLO 15ms, GH Actions SLO 25ms
|
||||
// - QueryAllBatch(50×4×100): ~57ms p95 observed locally in March 2026; ~123ms p95 on the April 9, 2026 v6 RC dry run
|
||||
// → local SLO 60ms, GH Actions SLO 140ms
|
||||
// - QueryAllBatch downsampled (50×4×100, 60s): ~31ms → local SLO 55ms, GH Actions SLO 130ms
|
||||
// - QueryAllBatch chunked (500×4×20): ~84ms p95 observed locally in March 2026 → local SLO 90ms, GH Actions SLO 140ms
|
||||
// - QueryAllBatch chunked (500×4×20): ~84ms p95 observed locally in March 2026; ~216ms p95 on the April 9, 2026 v6 RC dry run
|
||||
// → local SLO 90ms, GH Actions SLO 240ms
|
||||
// - rollupTier(50×2×20): ~2.1ms → SLO 15ms
|
||||
// - rollupTier fleet-scale (500×4×20): ~138ms p95 observed locally in March 2026; ~214-217ms p95 on March 26, 2026 GitHub release rehearsals
|
||||
// → local SLO 140ms, GH Actions SLO 230ms
|
||||
// - rollupTier fleet-scale (500×4×20): ~138ms p95 observed locally in March 2026; ~214-217ms p95 on March 26, 2026 GitHub release rehearsals;
|
||||
// ~241ms p95 on the April 9, 2026 v6 RC dry run → local SLO 140ms, GH Actions SLO 260ms
|
||||
// - Query under write contention: ~400µs → SLO 5ms
|
||||
// - 500-node concurrent dashboard load: ~7.9ms p95 observed locally in March 2026; ~23-24ms p95 on March 26, 2026 GitHub release rehearsals
|
||||
// → local SLO 15ms, GH Actions SLO 30ms
|
||||
@@ -51,13 +54,16 @@ const (
|
||||
// SLOQueryAllP95 is the p95 target for QueryAll (4 metric types × 500
|
||||
// points each) — dashboard loading all metrics for one resource.
|
||||
SLOQueryAllP95 = 15 * time.Millisecond
|
||||
// SLOQueryAllGitHubActionsP95 absorbs the slower shared-runner envelope seen
|
||||
// on the governed RC dry run while keeping the local hot-path budget strict.
|
||||
SLOQueryAllGitHubActionsP95 = 25 * time.Millisecond
|
||||
|
||||
// SLOQueryAllBatchP95 is the p95 target for QueryAllBatch (50 resources ×
|
||||
// 4 metric types × 100 points each) — the batched dashboard chart path.
|
||||
SLOQueryAllBatchP95 = 60 * time.Millisecond
|
||||
// SLOQueryAllBatchGitHubActionsP95 matches the slower shared-runner envelope
|
||||
// observed on GitHub-hosted release rehearsals.
|
||||
SLOQueryAllBatchGitHubActionsP95 = 100 * time.Millisecond
|
||||
SLOQueryAllBatchGitHubActionsP95 = 140 * time.Millisecond
|
||||
|
||||
// SLOQueryAllBatchDownsampledP95 is the p95 target for QueryAllBatch with
|
||||
// 60-second downsampling (50 resources × 4 metrics × 100 raw points). This
|
||||
@@ -72,7 +78,7 @@ const (
|
||||
// 500-resource scale, where the implementation must split requests into
|
||||
// multiple SQL chunks to stay within SQLite parameter limits.
|
||||
SLOQueryAllBatchChunkedP95 = 90 * time.Millisecond
|
||||
SLOQueryAllBatchChunkedGitHubActionsP95 = 140 * time.Millisecond
|
||||
SLOQueryAllBatchChunkedGitHubActionsP95 = 240 * time.Millisecond
|
||||
|
||||
// SLOQueryManyResourcesP95 is the p95 target for Query with 100 resources
|
||||
// in the table — validates that index isolation prevents full table scans.
|
||||
@@ -87,7 +93,7 @@ const (
|
||||
// batched rollupTier path at 500-resource scale (500 nodes × 4 metrics × 20
|
||||
// raw points). This guards the real fleet-scale aggregation workload.
|
||||
SLORollupTierBatchedFleetP95 = 140 * time.Millisecond
|
||||
SLORollupTierBatchedFleetGitHubActionsP95 = 230 * time.Millisecond
|
||||
SLORollupTierBatchedFleetGitHubActionsP95 = 260 * time.Millisecond
|
||||
|
||||
// SLOConcurrentReadWriteP95 is the p95 target for single-resource Query
|
||||
// while a background writer continuously appends batches on the same SQLite
|
||||
@@ -331,11 +337,12 @@ func TestSLO_QueryAll(t *testing.T) {
|
||||
})
|
||||
|
||||
p95 := pct(latencies, 0.95)
|
||||
target := effectiveSLOTarget(SLOQueryAllP95, SLOQueryAllGitHubActionsP95)
|
||||
t.Logf("QueryAll(4×500) p50=%v p95=%v p99=%v SLO=%v",
|
||||
pct(latencies, 0.50), p95, pct(latencies, 0.99), SLOQueryAllP95)
|
||||
pct(latencies, 0.50), p95, pct(latencies, 0.99), target)
|
||||
|
||||
if p95 > SLOQueryAllP95 {
|
||||
t.Errorf("SLO VIOLATION: p95=%v exceeds target %v", p95, SLOQueryAllP95)
|
||||
if p95 > target {
|
||||
t.Errorf("SLO VIOLATION: p95=%v exceeds target %v", p95, target)
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
Reference in New Issue
Block a user