From 265f4293580d0cfdd6a41a66527983bc79fb0400 Mon Sep 17 00:00:00 2001 From: houseme Date: Mon, 7 Sep 2026 19:54:48 +0800 Subject: [PATCH] test(scanner): require heal pressure evidence metrics (#7392) Require scanner/heal ABBA adapters to emit foreground pressure, heal lock wait p99, and heal attempt counters before a measured W10/W11 run can be accepted. Report per-leg pressure ratios, lock p99 samples, and attempt cost per healed object so synthetic harness runs remain evidence-contract validation rather than performance proof. Co-authored-by: zhi22915 --- scripts/scanner_abba.py | 40 +++++++++++++++++++++++++++++++++++- scripts/test_scanner_abba.py | 16 ++++++++++++++- 2 files changed, 54 insertions(+), 2 deletions(-) diff --git a/scripts/scanner_abba.py b/scripts/scanner_abba.py index 5edbd3a08..9e6c33dbb 100644 --- a/scripts/scanner_abba.py +++ b/scripts/scanner_abba.py @@ -22,6 +22,9 @@ METRICS = ( "p99_ms", "throughput_ops", "rss_bytes", "cpu_seconds", "iops", "rpc_count", "cache_clone_bytes", "encode_bytes", "save_bytes", "oldest_age_seconds", "walk_objects", "cold_walk_objects", "healed_objects", "errors", "requests", + "foreground_pressure_samples", "foreground_pressure_high_samples", + "heal_lock_wait_p99_ms", "heal_attempts", "heal_attempt_failures", + "heal_retry_attempts", ) REPEATABILITY_LIMIT = Decimal("0.05") P2_WORK_MULTIPLE_LIMIT = Decimal("1.2") @@ -252,6 +255,11 @@ def validate_result(result, request, expected): number(metrics.get(key), key) for key in ("requests", "p99_ms", "throughput_ops"): require(metrics[key] > 0, f"zero {key}") + require(metrics["foreground_pressure_samples"] > 0, "zero foreground pressure samples") + require(metrics["foreground_pressure_high_samples"] <= metrics["foreground_pressure_samples"], + "foreground pressure high samples exceed samples") + require(metrics["heal_attempt_failures"] <= metrics["heal_attempts"], "heal failures exceed attempts") + require(metrics["heal_retry_attempts"] <= metrics["heal_attempts"], "heal retries exceed attempts") require(metrics["errors"] == 0, "workload request errors") require(metrics["cold_walk_objects"] <= metrics["walk_objects"], "cold walk exceeds total walk") require(result.get("oracle") == expected, "object/version/byte oracle mismatch") @@ -263,6 +271,18 @@ def validate_result(result, request, expected): return result +def attempt_cost(metrics): + healed = decimal_number(metrics["healed_objects"], "healed_objects") + if healed == 0: + return None + return ratio(metrics["heal_attempts"], healed, "heal attempt cost") + + +def pressure_high_ratio(metrics): + return ratio(metrics["foreground_pressure_high_samples"], + metrics["foreground_pressure_samples"], "foreground pressure high samples") + + def convergence(result): window = result.get("convergence") if not window or window.get("writes_stopped") is not True or window.get("last_mutation_observed") is not True or window.get("first_complete_publication") is not True: @@ -318,6 +338,10 @@ def evaluate(cells): p2_pending = any(value is None for value in candidate_p2) passed &= all(ratio(value, 1, "p2 work multiple") <= P2_WORK_MULTIPLE_LIMIT for value in candidate_p2 if value is not None) p2_report = [None if value is None else float(value) for value in p2] + attempt_costs = [attempt_cost(cell["result"]["metrics"]) if cell["background"] == "on" else None for cell in group] + candidate_attempt_costs = [ + value for cell, value in zip(group, attempt_costs) if cell["leg"].startswith("B") and value is not None + ] inconclusive |= noise or p2_pending if not noise and not passed: failed = True @@ -327,7 +351,21 @@ def evaluate(cells): "p99_regression": float(p99), "throughput_change": float(throughput), "thresholds": {key: float(value) for key, value in thresholds.items()}, "p1": p1, "p2_max_work_multiple": float(P2_WORK_MULTIPLE_LIMIT), - "p2_post_stop_work_multiples": p2_report}) + "p2_post_stop_work_multiples": p2_report, + "w10_w11": { + "foreground_pressure_high_sample_ratios": [ + float(pressure_high_ratio(cell["result"]["metrics"])) for cell in group + ], + "heal_lock_wait_p99_ms": [ + cell["result"]["metrics"]["heal_lock_wait_p99_ms"] for cell in group + ], + "attempt_cost_per_healed_object": [ + None if value is None else float(value) for value in attempt_costs + ], + "candidate_attempt_cost_per_healed_object": ( + None if not candidate_attempt_costs else float(max(candidate_attempt_costs)) + ), + }}) return ("fail" if failed else "inconclusive" if inconclusive else "pass"), comparisons diff --git a/scripts/test_scanner_abba.py b/scripts/test_scanner_abba.py index 5408eb44c..c982589b7 100755 --- a/scripts/test_scanner_abba.py +++ b/scripts/test_scanner_abba.py @@ -96,6 +96,12 @@ def fake_adapter(): del result["metrics"]["save_bytes"] elif fault == "incomplete-repair": result["metrics"]["healed_objects"] = 0 + elif fault == "zero-pressure-samples": + result["metrics"]["foreground_pressure_samples"] = 0 + elif fault == "pressure-sample-order": + result["metrics"]["foreground_pressure_high_samples"] = result["metrics"]["foreground_pressure_samples"] + 1 + elif fault == "attempt-accounting": + result["metrics"]["heal_attempt_failures"] = result["metrics"]["heal_attempts"] + 1 harness.write_json(Path(output_path), result) return 0 @@ -271,10 +277,18 @@ class ScannerAbbaTest(unittest.TestCase): legs = [r for r in requests if (r["scenario"], r["comparison"], r["round"]) == (scenario, comparison, round_id)] self.assertEqual({r["leg"] for r in legs}, set(harness.LEGS)) self.assertTrue(all(c["p2_max_work_multiple"] == 1.2 for c in report["comparisons"])) + for comparison in report["comparisons"]: + w10_w11 = comparison["w10_w11"] + self.assertEqual(w10_w11["foreground_pressure_high_sample_ratios"], [1.0, 1.0, 1.0, 1.0]) + self.assertEqual(w10_w11["heal_lock_wait_p99_ms"], [10, 10, 10, 10]) + expected_attempt_cost = [None, 1.0, 1.0, None] if comparison["comparison"] == "background" else [1.0, 1.0, 1.0, 1.0] + self.assertEqual(w10_w11["attempt_cost_per_healed_object"], expected_attempt_cost) + self.assertEqual(w10_w11["candidate_attempt_cost_per_healed_object"], 1.0) def test_fail_closed_adapter_and_data_errors(self): for fault in ("measure-exit", "oracle-exit", "missing-oracle", "oracle-mismatch", "zero-samples", - "zero-requests", "request-errors", "load-drift", "missing-metric", "incomplete-repair"): + "zero-requests", "request-errors", "load-drift", "missing-metric", "incomplete-repair", + "zero-pressure-samples", "pressure-sample-order", "attempt-accounting"): with self.subTest(fault=fault), tempfile.TemporaryDirectory() as directory: self.root = Path(directory) with self.assertRaises((ValueError, OSError, subprocess.SubprocessError)):