mirror of
https://github.com/rustfs/rustfs.git
synced 2026-09-06 12:09:12 +00:00
test(scanner): verify restart evidence against tested builds (#7232)
* chore(deps): refresh scanner heal batch dependency baseline Regenerate compatible lockfile selections before the next implementation batch. Cargo upgrade leaves direct requirements unchanged. Co-Authored-By: heihutu <heihutu@gmail.com> Co-Authored-By: zhi22915 <qiuzgang@gmail.com> * fix(ecstore): remove duplicate local rename implementation Keep the canonical commit module after concurrent storage changes merged. The control-write and rollback changes are already present there. Co-Authored-By: heihutu <heihutu@gmail.com> Co-Authored-By: zhi22915 <qiuzgang@gmail.com> * chore(deps): refresh profiling dependencies for the next batch Update hotpath and its macro crate to the compatible patch release before the next dependency-ready implementation tasks. Co-Authored-By: heihutu <heihutu@gmail.com> Co-Authored-By: zhi22915 <qiuzgang@gmail.com> * fix(deps): preserve supported hotpath focus expressions Keep the profiler runtime before its regex-lite compatibility regression. Track the opt-in validation required to remove this constraint in backlog. Refs rustfs/backlog#2302. Co-Authored-By: heihutu <heihutu@gmail.com> Co-Authored-By: zhi22915 <qiuzgang@gmail.com> * test(scanner): add bounded ABBA validation harness Refs rustfs/backlog#2266 and rustfs/backlog#2240. Co-Authored-By: heihutu <heihutu@gmail.com> Co-Authored-By: zhi22915 <qiuzgang@gmail.com> * test(scanner): verify real restart evidence before release gates Co-Authored-By: heihutu <heihutu@gmail.com> Co-Authored-By: zhi22915 <qiuzgang@gmail.com> * fix(test): bind scanner evidence to execution and build identity Co-Authored-By: heihutu <heihutu@gmail.com> Co-Authored-By: zhi22915 <qiuzgang@gmail.com> * docs(test): use the nextest workspace report directory Co-Authored-By: heihutu <heihutu@gmail.com> Co-Authored-By: zhi22915 <qiuzgang@gmail.com> * fix(test): reap ABBA leaders only after process-group cleanup Co-Authored-By: heihutu <heihutu@gmail.com> Co-Authored-By: zhi22915 <qiuzgang@gmail.com> * fix(test): preserve inclusive ABBA thresholds Use decimal boundary comparisons for ABBA ratio checks and cover exact documented p99, throughput, and P1 limits. Co-Authored-By: heihutu <heihutu@gmail.com> Co-Authored-By: zhi22915 <qiuzgang@gmail.com> --------- Co-authored-by: heihutu <heihutu@gmail.com> Co-authored-by: zhi22915 <qiuzgang@gmail.com> Co-authored-by: overtrue <anzhengchao@gmail.com>
This commit is contained in:
@@ -12,11 +12,15 @@ import sys
|
||||
import tempfile
|
||||
import tomllib
|
||||
import unittest
|
||||
import uuid
|
||||
import xml.etree.ElementTree as ET
|
||||
from datetime import datetime, timezone
|
||||
from unittest import mock
|
||||
from pathlib import Path
|
||||
from zoneinfo import ZoneInfo, ZoneInfoNotFoundError
|
||||
|
||||
from scanner_abba import MAX_JSON_BYTES, digest, number, read_json, require, sha, write_json
|
||||
|
||||
|
||||
ROOT = Path(__file__).resolve().parents[1]
|
||||
SCHEDULED_ALERT_WORKFLOWS = tuple(
|
||||
@@ -873,6 +877,187 @@ def check_core_listing(root: Path, listing: Path) -> list[str]:
|
||||
return [f"cannot read core nextest listing: {error}"]
|
||||
|
||||
|
||||
def evidence_integer(value: object, name: str, minimum: int, maximum: int) -> int:
|
||||
require(type(value) is int and minimum <= value <= maximum, f"invalid integer {name}")
|
||||
return value
|
||||
|
||||
|
||||
def begin_scanner_heal_receipt(root: Path, directory: Path, binary: Path, test_binary: Path) -> None:
|
||||
"""Record an existing build; this command never builds or runs a test."""
|
||||
require(not directory.exists(), "scanner/heal run directory must be new")
|
||||
require(not subprocess.check_output(["git", "status", "--porcelain", "--untracked-files=no"], cwd=root, text=True).strip(),
|
||||
"commit tracked source changes before creating evidence")
|
||||
builds = {}
|
||||
for label, path in (("binary", binary), ("test_binary", test_binary)):
|
||||
path = path.resolve(strict=True)
|
||||
require(path.is_file() and os.access(path, os.X_OK), f"missing executable {label}")
|
||||
builds[label] = {"path": str(path), "sha256": digest(path)}
|
||||
revision = subprocess.check_output(["git", "rev-parse", "HEAD"], cwd=root, text=True).strip()
|
||||
require(re.fullmatch(r"[0-9a-f]{40}", revision), "invalid source revision")
|
||||
version = subprocess.check_output([builds["binary"]["path"], "--version"], text=True, timeout=30)
|
||||
embedded_revision = re.search(r"^git commit\s*:\s*([0-9a-f]{40})\s*$", version, re.MULTILINE)
|
||||
embedded_status = re.search(r"^git status\s*:\s*(.*)\Z", version, re.MULTILINE | re.DOTALL)
|
||||
require(embedded_revision is not None and embedded_revision[1] == revision, "server binary source revision mismatch")
|
||||
require(embedded_status is not None and not embedded_status[1].strip(), "server binary was built from dirty/unknown source")
|
||||
lock_blob = subprocess.check_output(["git", "hash-object", "Cargo.lock"], cwd=root, text=True).strip()
|
||||
require(re.fullmatch(r"[0-9a-f]{40}", lock_blob), "invalid Cargo.lock identity")
|
||||
features = os.environ.get("RUSTFS_E2E_EXPECTED_FEATURES")
|
||||
require(features is not None, "set RUSTFS_E2E_EXPECTED_FEATURES to the compiled e2e crate feature set")
|
||||
features = ",".join(sorted(set(filter(None, (feature.strip() for feature in features.split(","))))))
|
||||
require(all(re.fullmatch(r"[a-z0-9-]+", feature) for feature in features.split(",") if feature), "invalid expected features")
|
||||
directory.mkdir(parents=True)
|
||||
write_json(directory / "run.json", {"schema": 1, "run_id": uuid.uuid4().hex,
|
||||
"source_revision": revision,
|
||||
"binary_source_revision": embedded_revision[1],
|
||||
"test_build": {"source_revision": revision, "dirty": False,
|
||||
"lock_blob": lock_blob, "features": features},
|
||||
"started_at": datetime.now(timezone.utc).timestamp(), **builds})
|
||||
|
||||
|
||||
def finish_scanner_heal_receipt(directory: Path, exit_code: int) -> None:
|
||||
require(type(exit_code) is int and 0 <= exit_code <= 255, "invalid test exit code")
|
||||
require(not (directory / "execution.json").exists(), "execution receipt already exists")
|
||||
run = read_json(directory / "run.json")
|
||||
artifacts = {}
|
||||
for name in ("listing.json", "junit.xml", "background-target-restart.json"):
|
||||
path = directory / name
|
||||
if exit_code != 0 and not path.exists():
|
||||
continue
|
||||
require(path.is_file() and 0 < path.stat().st_size <= MAX_JSON_BYTES, f"missing/oversized {name}")
|
||||
require(path.stat().st_mtime >= run["started_at"], f"stale {name}")
|
||||
artifacts[name] = digest(path)
|
||||
write_json(directory / "execution.json", {"run_id": run["run_id"], "exit_code": exit_code,
|
||||
"finished_at": datetime.now(timezone.utc).timestamp(),
|
||||
"artifacts": artifacts})
|
||||
|
||||
|
||||
def check_scanner_heal_evidence(root: Path, directory: Path, case_id: str) -> list[str]:
|
||||
"""Validate one actual case, or fail the release while required lanes are pending."""
|
||||
try:
|
||||
registry = read_json(root / ".config/scanner-heal-required-tests.json")
|
||||
evidence_integer(registry.get("schema"), "registry schema", 1, 1)
|
||||
require(registry.get("cases"), "invalid scanner/heal registry")
|
||||
selected = registry["cases"] if case_id == "release" else {case_id: registry["cases"][case_id]}
|
||||
run = read_json(directory / "run.json")
|
||||
execution = read_json(directory / "execution.json")
|
||||
evidence_integer(run.get("schema"), "run schema", 1, 1)
|
||||
require(re.fullmatch(r"[0-9a-f]{32}", run["run_id"]), "invalid run identity")
|
||||
require(re.fullmatch(r"[0-9a-f]{40}", run["source_revision"]), "invalid source revision")
|
||||
require(run.get("binary_source_revision") == run["source_revision"], "server source provenance missing")
|
||||
expected_build = run["test_build"]
|
||||
require(expected_build["source_revision"] == run["source_revision"] and expected_build["dirty"] is False,
|
||||
"test source provenance missing")
|
||||
require(re.fullmatch(r"[0-9a-f]{40}", expected_build["lock_blob"]), "invalid test lockfile identity")
|
||||
require(isinstance(expected_build["features"], str), "missing test features")
|
||||
require(execution.get("run_id") == run["run_id"], "execution belongs to another run")
|
||||
require(type(execution.get("exit_code")) is int and execution["exit_code"] == 0, "test command failed or did not run")
|
||||
number(run["started_at"], "started_at", 1)
|
||||
number(execution["finished_at"], "finished_at", run["started_at"])
|
||||
for label in ("binary", "test_binary"):
|
||||
require(sha(run[label]["sha256"]) and digest(Path(run[label]["path"])) == run[label]["sha256"],
|
||||
f"{label} changed or missing")
|
||||
for name in ("listing.json", "junit.xml"):
|
||||
path = directory / name
|
||||
require(0 < path.stat().st_size <= MAX_JSON_BYTES, f"missing/oversized {name}")
|
||||
require(run["started_at"] <= path.stat().st_mtime <= execution["finished_at"], f"{name} outside run window")
|
||||
require(digest(path) == execution["artifacts"][name], f"{name} hash mismatch")
|
||||
suites = read_json(directory / "listing.json")["rust-suites"]
|
||||
xml = (directory / "junit.xml").read_bytes()
|
||||
require(b"<!DOCTYPE" not in xml and b"<!ENTITY" not in xml, "JUnit entities are forbidden")
|
||||
junit = ET.fromstring(xml)
|
||||
cases = list(junit.iter("testcase"))
|
||||
require(bool(cases), "JUnit has zero testcases")
|
||||
for case in cases:
|
||||
require(not any(child.tag in ("failure", "error", "skipped", "rerunFailure", "rerunError", "flakyFailure", "flakyError")
|
||||
for child in case), "JUnit contains failed, skipped or retried tests")
|
||||
errors = []
|
||||
for name, requirement in selected.items():
|
||||
suite, test = requirement["suite"], requirement["name"]
|
||||
for key in ("nodes", "drives_per_node"):
|
||||
evidence_integer(requirement["topology"][key], f"required {key}", 1, 16)
|
||||
listing_suite = suites[suite]
|
||||
require(listing_suite["binary-id"] == suite, "nextest suite binary identity mismatch")
|
||||
listed_binary = Path(listing_suite["binary-path"]).resolve(strict=True)
|
||||
require(listed_binary == Path(run["test_binary"]["path"]).resolve(strict=True) and
|
||||
digest(listed_binary) == run["test_binary"]["sha256"], "nextest selected another test binary")
|
||||
require(listing_suite["package-name"] == "e2e_test" and listing_suite["build-platform"] in ("host", "target"),
|
||||
"unexpected nextest suite metadata")
|
||||
listed = listing_suite.get("testcases", {}).get(test, {})
|
||||
require(listed.get("ignored") is False and listed.get("filter-match", {}).get("status") == "matches",
|
||||
f"required test not selected: {suite}::{test}")
|
||||
matches = [case for case in cases if case.get("name") == test and case.get("classname") == suite]
|
||||
require(len(matches) == 1, f"missing/duplicate JUnit case: {suite}::{test}")
|
||||
started = datetime.fromisoformat(matches[0].attrib["timestamp"].replace("Z", "+00:00"))
|
||||
require(started.tzinfo is not None, "JUnit timestamp must include timezone")
|
||||
# quick-junit truncates timestamps to milliseconds.
|
||||
require(run["started_at"] - 0.001 <= started.timestamp() <= execution["finished_at"],
|
||||
"JUnit testcase executed outside this run")
|
||||
path = directory / requirement["oracle"]
|
||||
require(path.resolve().is_relative_to(directory.resolve()), "oracle path escapes run directory")
|
||||
require(run["started_at"] <= path.stat().st_mtime <= execution["finished_at"], "oracle outside run window")
|
||||
require(digest(path) == execution["artifacts"][requirement["oracle"]], "oracle hash mismatch")
|
||||
oracle = read_json(path)
|
||||
evidence_integer(oracle.get("schema"), "oracle schema", 1, 1)
|
||||
require(oracle.get("evidence") == "process-restart", "not real process-restart evidence")
|
||||
require(oracle.get("case") == name and oracle.get("run_id") == run["run_id"], "oracle belongs to another case/run")
|
||||
require(oracle.get("source_revision") == run["source_revision"], "oracle source mismatch")
|
||||
built = oracle["test_build"]
|
||||
for key in ("source_revision", "dirty", "lock_blob", "features"):
|
||||
require(built[key] == expected_build[key], f"compiled test {key} mismatch")
|
||||
require(built["dirty"] is False, "test binary was compiled from dirty source")
|
||||
require(all(isinstance(built[key], str) and built[key] and built[key] != "unknown" for key in ("target", "profile")),
|
||||
"missing compiled target/profile")
|
||||
require(isinstance(built["rustflags_hex"], str) and re.fullmatch(r"(?:[0-9a-f]{2})*", built["rustflags_hex"]) is not None,
|
||||
"invalid compiled rustflags")
|
||||
for label in ("binary", "test_binary"):
|
||||
require(oracle.get(f"{label}_sha256") == run[label]["sha256"], f"oracle {label} mismatch")
|
||||
require(oracle.get("topology") == requirement["topology"], "oracle topology mismatch")
|
||||
for key in ("nodes", "drives_per_node"):
|
||||
evidence_integer(oracle["topology"][key], f"observed {key}", 1, 16)
|
||||
evidence_integer(oracle.get("pid_before"), "pid_before", 1, 2**32 - 1)
|
||||
evidence_integer(oracle.get("pid_after"), "pid_after", 1, 2**32 - 1)
|
||||
require(oracle["pid_before"] != oracle["pid_after"], "no process restart witnessed")
|
||||
objects = oracle["objects"]
|
||||
require(isinstance(objects, list) and requirement["min_objects"] <= len(objects) <= requirement["max_objects"],
|
||||
"incomplete/oversized object oracle")
|
||||
require(len({obj["key"] for obj in objects}) == len(objects), "duplicate object identity")
|
||||
require(sum(obj["expected_physical"] is None for obj in objects) == 1,
|
||||
"only the outage object may lack a pre-fault target manifest")
|
||||
for obj in objects:
|
||||
require(isinstance(obj["key"], str) and 0 < len(obj["key"].encode()) <= 1024, "invalid object identity")
|
||||
require(obj["version_id"] is None, "this case only covers unversioned objects")
|
||||
require(type(obj["expected_bytes"]) is int and obj["expected_bytes"] > 0, "missing expected bytes")
|
||||
require(type(obj["actual_bytes"]) is int and obj["actual_bytes"] == obj["expected_bytes"], "S3 body length mismatch")
|
||||
require(sha(obj["expected_sha256"]) and obj["actual_sha256"] == obj["expected_sha256"], "S3 body digest mismatch")
|
||||
physical = obj["physical"]
|
||||
if obj["expected_physical"] is not None:
|
||||
require(physical == obj["expected_physical"], "target shard differs from pre-fault manifest")
|
||||
for geometry in [physical] + ([obj["expected_physical"]] if obj["expected_physical"] is not None else []):
|
||||
data = evidence_integer(geometry["data_blocks"], "EC data blocks", 1, 16)
|
||||
parity = evidence_integer(geometry["parity_blocks"], "EC parity blocks", 1, 16)
|
||||
require(data + parity == oracle["topology"]["nodes"] * oracle["topology"]["drives_per_node"],
|
||||
"EC geometry differs from this case's single set")
|
||||
evidence_integer(geometry["erasure_index"], "target erasure index", 1, data + parity)
|
||||
require(physical["has_xl_meta"] is True and physical["version_id"] is None, "missing target metadata")
|
||||
parts = physical["expected_part_numbers"]
|
||||
require(isinstance(parts, list) and 0 < len(parts) <= 10000, "no physical part coverage")
|
||||
require(all(type(part) is int and part > 0 for part in parts) and len(set(parts)) == len(parts),
|
||||
"invalid physical part identity")
|
||||
require({str(part) for part in parts} == set(physical["present_part_fingerprints"]), "target shard parts missing")
|
||||
for part in physical["present_part_fingerprints"].values():
|
||||
require(type(part["size"]) is int and part["size"] > 0 and sha(part["sha256"]), "invalid target shard fingerprint")
|
||||
node_listings = oracle["node_listings"]
|
||||
require(isinstance(node_listings, list) and len(node_listings) == requirement["topology"]["nodes"],
|
||||
"missing per-node S3 listing")
|
||||
require(all(keys == sorted(obj["key"] for obj in objects) for keys in node_listings),
|
||||
"S3 listing differs from object oracle")
|
||||
if case_id == "release":
|
||||
errors.extend(f"pending {gate}: {reason}" for gate, reason in registry["release_pending"].items())
|
||||
return errors
|
||||
except (OSError, KeyError, TypeError, ValueError, ET.ParseError) as error:
|
||||
return [f"scanner/heal evidence rejected: {error}"]
|
||||
|
||||
|
||||
def validate(root: Path) -> list[str]:
|
||||
errors: list[str] = []
|
||||
errors.extend(check_core_fixtures(root))
|
||||
@@ -1022,6 +1207,213 @@ class SelfTests(unittest.TestCase):
|
||||
with mock.patch(__name__ + ".check_quick_checks", return_value=[error]):
|
||||
self.assertIn(error, validate(ROOT))
|
||||
|
||||
def scanner_heal_fixture(self, directory: Path) -> tuple[Path, Path]:
|
||||
"""Parser fixtures only; these files are never runtime evidence."""
|
||||
root, run_dir = directory / "repo", directory / "run"
|
||||
(root / ".config").mkdir(parents=True)
|
||||
run_dir.mkdir()
|
||||
registry = read_json(ROOT / ".config/scanner-heal-required-tests.json")
|
||||
write_json(root / ".config/scanner-heal-required-tests.json", registry)
|
||||
requirement = registry["cases"]["background-target-restart"]
|
||||
binary = directory / "fake-binary"
|
||||
binary.write_bytes(b"parser fixture, not a real build")
|
||||
binary.chmod(0o700)
|
||||
build = {"path": str(binary), "sha256": digest(binary)}
|
||||
write_json(run_dir / "run.json", {"schema": 1, "run_id": "a" * 32, "source_revision": "b" * 40,
|
||||
"binary_source_revision": "b" * 40,
|
||||
"test_build": {"source_revision": "b" * 40, "dirty": False,
|
||||
"lock_blob": "c" * 40, "features": "default"},
|
||||
"started_at": datetime.now(timezone.utc).timestamp() - 1,
|
||||
"binary": build, "test_binary": build})
|
||||
write_json(run_dir / "listing.json", {"rust-suites": {requirement["suite"]: {
|
||||
"binary-id": requirement["suite"], "binary-path": str(binary), "package-name": "e2e_test", "build-platform": "target",
|
||||
"testcases": {
|
||||
requirement["name"]: {"ignored": False, "filter-match": {"status": "matches"}}
|
||||
}}}})
|
||||
(run_dir / "junit.xml").write_text(
|
||||
f'<testsuites><testsuite><testcase name="{requirement["name"]}" classname="{requirement["suite"]}" '
|
||||
f'timestamp="{datetime.now(timezone.utc).isoformat(timespec="milliseconds")}"/></testsuite></testsuites>')
|
||||
physical = {"has_xl_meta": True, "version_id": None, "data_dir": "data-generation",
|
||||
"erasure_index": 1, "data_blocks": 2, "parity_blocks": 2, "expected_part_numbers": [1],
|
||||
"present_part_fingerprints": {"1": {"size": 12, "sha256": "c" * 64}},
|
||||
"inline_data_fingerprint": None}
|
||||
obj = {"key": "object", "version_id": None, "expected_bytes": 16, "actual_bytes": 16,
|
||||
"expected_sha256": "d" * 64, "actual_sha256": "d" * 64,
|
||||
"expected_physical": physical, "physical": physical}
|
||||
objects = [dict(obj, key=f"object-{index}") for index in range(9)]
|
||||
objects[-1] = dict(objects[-1], expected_physical=None)
|
||||
write_json(run_dir / "background-target-restart.json", {
|
||||
"schema": 1, "evidence": "process-restart", "case": "background-target-restart",
|
||||
"run_id": "a" * 32, "source_revision": "b" * 40,
|
||||
"test_build": {"source_revision": "b" * 40, "dirty": False, "lock_blob": "c" * 40,
|
||||
"features": "default", "target": "aarch64-apple-darwin", "profile": "debug", "rustflags_hex": ""},
|
||||
"binary_sha256": build["sha256"], "test_binary_sha256": build["sha256"],
|
||||
"topology": {"nodes": 4, "drives_per_node": 1}, "pid_before": 10, "pid_after": 11,
|
||||
"objects": objects, "node_listings": [[item["key"] for item in objects]] * 4,
|
||||
})
|
||||
finish_scanner_heal_receipt(run_dir, 0)
|
||||
return root, run_dir
|
||||
|
||||
def test_scanner_heal_case_does_not_approve_pending_release(self) -> None:
|
||||
with tempfile.TemporaryDirectory() as tmp:
|
||||
root, run_dir = self.scanner_heal_fixture(Path(tmp))
|
||||
self.assertEqual(check_scanner_heal_evidence(root, run_dir, "background-target-restart"), [])
|
||||
errors = check_scanner_heal_evidence(root, run_dir, "release")
|
||||
self.assertEqual(len(errors), 21)
|
||||
self.assertTrue(any(error.startswith("pending R-E:") for error in errors))
|
||||
self.assertTrue(any(error.startswith("pending R-D:") for error in errors))
|
||||
self.assertTrue(any(error.startswith("pending R-L:") for error in errors))
|
||||
|
||||
def test_scanner_heal_rejects_broken_execution_and_artifacts(self) -> None:
|
||||
for fault in ("exit", "missing", "zero", "skipped", "failed", "retry", "filtered", "ignored", "stale",
|
||||
"hash", "binary", "synthetic", "wrong-run", "same-pid", "body", "parts", "listing", "topology"):
|
||||
with self.subTest(fault=fault), tempfile.TemporaryDirectory() as tmp:
|
||||
root, run_dir = self.scanner_heal_fixture(Path(tmp))
|
||||
path = run_dir / "background-target-restart.json"
|
||||
oracle = read_json(path)
|
||||
if fault == "exit":
|
||||
receipt = read_json(run_dir / "execution.json")
|
||||
receipt["exit_code"] = 42
|
||||
write_json(run_dir / "execution.json", receipt)
|
||||
elif fault == "missing":
|
||||
path.unlink()
|
||||
elif fault == "zero":
|
||||
(run_dir / "junit.xml").write_text("<testsuites/>")
|
||||
elif fault in ("skipped", "failed", "retry"):
|
||||
junit = run_dir / "junit.xml"
|
||||
tag = {"skipped": "skipped", "failed": "failure", "retry": "rerunFailure"}[fault]
|
||||
junit.write_text(junit.read_text().replace("/></testsuite>", f"><{tag}/></testcase></testsuite>"))
|
||||
elif fault in ("filtered", "ignored"):
|
||||
listing = read_json(run_dir / "listing.json")
|
||||
case = next(iter(listing["rust-suites"]["e2e_test"]["testcases"].values()))
|
||||
case["ignored"] = fault == "ignored"
|
||||
case["filter-match"]["status"] = "mismatch" if fault == "filtered" else "matches"
|
||||
write_json(run_dir / "listing.json", listing)
|
||||
elif fault == "stale":
|
||||
os.utime(path, (1, 1))
|
||||
elif fault == "hash":
|
||||
path.write_text(path.read_text() + " ")
|
||||
elif fault == "binary":
|
||||
Path(read_json(run_dir / "run.json")["binary"]["path"]).write_bytes(b"another build")
|
||||
else:
|
||||
if fault == "synthetic":
|
||||
oracle["evidence"] = "synthetic"
|
||||
elif fault == "wrong-run":
|
||||
oracle["run_id"] = "f" * 32
|
||||
elif fault == "same-pid":
|
||||
oracle["pid_after"] = oracle["pid_before"]
|
||||
elif fault == "body":
|
||||
oracle["objects"][0]["actual_sha256"] = "e" * 64
|
||||
elif fault == "parts":
|
||||
oracle["objects"][0]["physical"]["present_part_fingerprints"] = {}
|
||||
elif fault == "listing":
|
||||
oracle["node_listings"][0] = []
|
||||
elif fault == "topology":
|
||||
oracle["topology"] = {"nodes": 3, "drives_per_node": 4}
|
||||
write_json(path, oracle)
|
||||
if fault not in ("exit", "missing", "stale", "hash", "binary"):
|
||||
(run_dir / "execution.json").unlink()
|
||||
finish_scanner_heal_receipt(run_dir, 0)
|
||||
self.assertTrue(check_scanner_heal_evidence(root, run_dir, "background-target-restart"), fault)
|
||||
|
||||
def test_scanner_heal_receipts_reject_reuse_and_missing_builds(self) -> None:
|
||||
with tempfile.TemporaryDirectory() as tmp:
|
||||
root, run_dir = self.scanner_heal_fixture(Path(tmp))
|
||||
with self.assertRaisesRegex(ValueError, "already exists"):
|
||||
finish_scanner_heal_receipt(run_dir, 0)
|
||||
with self.assertRaisesRegex(ValueError, "must be new"):
|
||||
begin_scanner_heal_receipt(root, run_dir, Path("missing"), Path("missing"))
|
||||
with mock.patch("subprocess.check_output", side_effect=["", "b" * 40]):
|
||||
with self.assertRaises(FileNotFoundError):
|
||||
begin_scanner_heal_receipt(root, Path(tmp) / "new-run", Path(tmp) / "missing", Path(tmp) / "missing")
|
||||
|
||||
def test_scanner_heal_begin_requires_embedded_source_provenance(self) -> None:
|
||||
for kind in ("current", "stale", "dirty", "unknown"):
|
||||
with self.subTest(kind=kind), tempfile.TemporaryDirectory() as tmp:
|
||||
root, _ = self.scanner_heal_fixture(Path(tmp))
|
||||
sources = root / "crates/e2e_test/src"
|
||||
sources.mkdir(parents=True)
|
||||
(sources / "heal_erasure_disk_rebuild_test.rs").write_bytes(b"oracle source")
|
||||
(sources / "chaos.rs").write_bytes(b"census source")
|
||||
revision = "c" * 40 if kind == "stale" else "b" * 40
|
||||
version = f"rustfs\ngit commit : {revision}\ngit status :\n"
|
||||
if kind == "dirty":
|
||||
version += "modified source\n"
|
||||
if kind == "unknown":
|
||||
version = "rustfs without build provenance"
|
||||
with mock.patch("subprocess.check_output", side_effect=["", "b" * 40, version, "c" * 40]), \
|
||||
mock.patch.dict(os.environ, {"RUSTFS_E2E_EXPECTED_FEATURES": "default"}):
|
||||
directory = Path(tmp) / "fresh"
|
||||
binary = Path(tmp) / "fake-binary"
|
||||
if kind == "current":
|
||||
begin_scanner_heal_receipt(root, directory, binary, binary)
|
||||
self.assertEqual(read_json(directory / "run.json")["binary_source_revision"], "b" * 40)
|
||||
else:
|
||||
with self.assertRaisesRegex(ValueError, "server binary"):
|
||||
begin_scanner_heal_receipt(root, directory, binary, binary)
|
||||
self.assertFalse(directory.exists())
|
||||
|
||||
def test_scanner_heal_rejects_copied_junit_and_wrong_suite_build(self) -> None:
|
||||
for fault in ("old-junit", "missing-time", "wrong-binary", "common-source", "lockfile", "features", "dirty-build"):
|
||||
with self.subTest(fault=fault), tempfile.TemporaryDirectory() as tmp:
|
||||
root, run_dir = self.scanner_heal_fixture(Path(tmp))
|
||||
if fault in ("old-junit", "missing-time"):
|
||||
path = run_dir / "junit.xml"
|
||||
xml = ET.fromstring(path.read_bytes())
|
||||
testcase = next(xml.iter("testcase"))
|
||||
if fault == "old-junit":
|
||||
testcase.set("timestamp", "2000-01-01T00:00:00.000Z")
|
||||
else:
|
||||
del testcase.attrib["timestamp"]
|
||||
# Rewriting/copying gives an old execution a fresh mtime.
|
||||
path.write_bytes(ET.tostring(xml))
|
||||
elif fault == "wrong-binary":
|
||||
path = run_dir / "listing.json"
|
||||
listing = read_json(path)
|
||||
another = Path(tmp) / "another-binary"
|
||||
another.write_bytes(Path(tmp, "fake-binary").read_bytes())
|
||||
listing["rust-suites"]["e2e_test"]["binary-path"] = str(another)
|
||||
write_json(path, listing)
|
||||
else:
|
||||
path = run_dir / "background-target-restart.json"
|
||||
oracle = read_json(path)
|
||||
key, value = {"common-source": ("source_revision", "f" * 40), "lockfile": ("lock_blob", "f" * 40),
|
||||
"features": ("features", "default,sftp"), "dirty-build": ("dirty", True)}[fault]
|
||||
oracle["test_build"][key] = value
|
||||
write_json(path, oracle)
|
||||
(run_dir / "execution.json").unlink()
|
||||
finish_scanner_heal_receipt(run_dir, 0)
|
||||
self.assertTrue(check_scanner_heal_evidence(root, run_dir, "background-target-restart"), fault)
|
||||
|
||||
def test_scanner_heal_rejects_boolean_fractional_and_out_of_geometry_integers(self) -> None:
|
||||
valid = {"schema": 1, "nodes": 4, "drives_per_node": 1, "pid_before": 10, "pid_after": 11,
|
||||
"erasure_index": 1, "data_blocks": 2, "parity_blocks": 2}
|
||||
cases = [(field, value) for field, correct in valid.items() for value in (True, float(correct))]
|
||||
cases += [("erasure_index", 5), ("erasure_index", 0), ("pid_after", -1)]
|
||||
for field, value in cases:
|
||||
with self.subTest(field=field, value=value), tempfile.TemporaryDirectory() as tmp:
|
||||
root, run_dir = self.scanner_heal_fixture(Path(tmp))
|
||||
path = run_dir / "background-target-restart.json"
|
||||
oracle = read_json(path)
|
||||
if field in ("nodes", "drives_per_node"):
|
||||
oracle["topology"][field] = value
|
||||
elif field in ("erasure_index", "data_blocks", "parity_blocks"):
|
||||
oracle["objects"][-1]["physical"][field] = value
|
||||
else:
|
||||
oracle[field] = value
|
||||
write_json(path, oracle)
|
||||
(run_dir / "execution.json").unlink()
|
||||
finish_scanner_heal_receipt(run_dir, 0)
|
||||
self.assertTrue(check_scanner_heal_evidence(root, run_dir, "background-target-restart"))
|
||||
for filename in ("run.json", ".config/scanner-heal-required-tests.json"):
|
||||
with self.subTest(filename=filename), tempfile.TemporaryDirectory() as tmp:
|
||||
root, run_dir = self.scanner_heal_fixture(Path(tmp))
|
||||
path = (root if filename.startswith(".config") else run_dir) / filename
|
||||
content = read_json(path)
|
||||
content["schema"] = True
|
||||
write_json(path, content)
|
||||
self.assertTrue(check_scanner_heal_evidence(root, run_dir, "background-target-restart"))
|
||||
|
||||
def test_core_gate_rejects_missing_ignored_filtered_and_corrupt_inputs(self) -> None:
|
||||
with tempfile.TemporaryDirectory() as tmp:
|
||||
root = Path(tmp)
|
||||
@@ -1664,6 +2056,25 @@ def main() -> int:
|
||||
if sys.argv[1:] == ["--self-test"]:
|
||||
suite = unittest.defaultTestLoader.loadTestsFromTestCase(SelfTests)
|
||||
return 0 if unittest.TextTestRunner(verbosity=2).run(suite).wasSuccessful() else 1
|
||||
if sys.argv[1:2] in (["--begin-scanner-heal"], ["--finish-scanner-heal"], ["--check-scanner-heal"]):
|
||||
try:
|
||||
if len(sys.argv) == 5 and sys.argv[1] == "--begin-scanner-heal":
|
||||
begin_scanner_heal_receipt(ROOT, Path(sys.argv[2]), Path(sys.argv[3]), Path(sys.argv[4]))
|
||||
return 0
|
||||
if len(sys.argv) == 4 and sys.argv[1] == "--finish-scanner-heal":
|
||||
finish_scanner_heal_receipt(Path(sys.argv[2]), int(sys.argv[3]))
|
||||
return 0
|
||||
if len(sys.argv) == 4 and sys.argv[1] == "--check-scanner-heal":
|
||||
errors = check_scanner_heal_evidence(ROOT, Path(sys.argv[2]), sys.argv[3])
|
||||
for error in errors:
|
||||
print(f"ERROR: {error}", file=sys.stderr)
|
||||
if not errors:
|
||||
print(f"Case evidence verified: {sys.argv[3]}; this does not approve release")
|
||||
return 1 if errors else 0
|
||||
raise ValueError("expected --begin-scanner-heal DIR BINARY TEST_BINARY, --finish-scanner-heal DIR EXIT, or --check-scanner-heal DIR CASE|release")
|
||||
except (OSError, KeyError, TypeError, ValueError, subprocess.SubprocessError) as error:
|
||||
print(f"ERROR: {error}", file=sys.stderr)
|
||||
return 1
|
||||
if len(sys.argv) == 3 and sys.argv[1] == "--check-core":
|
||||
errors = check_core_listing(ROOT, Path(sys.argv[2]))
|
||||
for error in errors:
|
||||
|
||||
+104
-29
@@ -8,6 +8,7 @@ import json
|
||||
import math
|
||||
import os
|
||||
from pathlib import Path
|
||||
import select
|
||||
import shutil
|
||||
import signal
|
||||
import subprocess
|
||||
@@ -126,28 +127,110 @@ def validate_manifest(manifest):
|
||||
require(manifest["expected_healed_objects"][scenario] > 0, f"{scenario} requires repairs")
|
||||
|
||||
|
||||
class OwnedCommand:
|
||||
"""Keep the session leader unreaped until its group's last signal is sent."""
|
||||
|
||||
def __init__(self, args, log):
|
||||
require(sys.platform == "darwin" or hasattr(os, "waitid"), "non-reaping child observation is unavailable")
|
||||
self.args, self.status = args, None
|
||||
self.queue = select.kqueue() if sys.platform == "darwin" else None
|
||||
self.process = None
|
||||
read_gate, write_gate = os.pipe()
|
||||
try:
|
||||
# The shell has already exec'd when Popen returns. Gate the target
|
||||
# until kqueue is registered; preexec_fn would deadlock Popen here.
|
||||
gate = f'read -r _scanner_gate <&{read_gate} || exit 125; exec {read_gate}<&-; exec "$@"'
|
||||
self.process = subprocess.Popen(["bash", "-c", gate, "scanner-abba", *args], pass_fds=(read_gate,),
|
||||
stdout=log, stderr=subprocess.STDOUT, start_new_session=True)
|
||||
if self.queue is not None:
|
||||
# Darwin NOTE_EXITSTATUS is not exposed by Python's select constants.
|
||||
event = select.kevent(self.process.pid, filter=select.KQ_FILTER_PROC,
|
||||
flags=select.KQ_EV_ADD | select.KQ_EV_ONESHOT,
|
||||
fflags=select.KQ_NOTE_EXIT | 0x04000000)
|
||||
self.queue.control([event], 0, 0)
|
||||
os.write(write_gate, b"\n")
|
||||
except BaseException:
|
||||
try:
|
||||
if self.process is not None:
|
||||
try:
|
||||
self._signal_group(signal.SIGKILL)
|
||||
finally:
|
||||
self.process.wait(timeout=10)
|
||||
finally:
|
||||
if self.queue is not None:
|
||||
self.queue.close()
|
||||
raise
|
||||
finally:
|
||||
os.close(read_gate)
|
||||
os.close(write_gate)
|
||||
|
||||
def wait(self, timeout):
|
||||
if self.status is not None:
|
||||
return self.status
|
||||
deadline = time.monotonic() + timeout
|
||||
while True:
|
||||
remaining = deadline - time.monotonic()
|
||||
if remaining <= 0:
|
||||
raise subprocess.TimeoutExpired(self.args, timeout)
|
||||
if self.queue is not None:
|
||||
events = self.queue.control(None, 1, remaining)
|
||||
if events:
|
||||
self.status = os.waitstatus_to_exitcode(events[0].data)
|
||||
return self.status
|
||||
else:
|
||||
result = os.waitid(os.P_PID, self.process.pid, os.WEXITED | os.WNOWAIT | os.WNOHANG)
|
||||
if result is not None:
|
||||
self.status = result.si_status if result.si_code == os.CLD_EXITED else -result.si_status
|
||||
return self.status
|
||||
time.sleep(min(0.05, remaining))
|
||||
|
||||
def _signal_group(self, sig):
|
||||
try:
|
||||
os.killpg(self.process.pid, sig)
|
||||
return True
|
||||
except ProcessLookupError:
|
||||
return False
|
||||
|
||||
def finish(self, terminate=False):
|
||||
if self.process.returncode is not None:
|
||||
return self.process.returncode
|
||||
try:
|
||||
if terminate:
|
||||
try:
|
||||
self._signal_group(signal.SIGTERM)
|
||||
deadline = time.monotonic() + 10
|
||||
while time.monotonic() < deadline and self._signal_group(0):
|
||||
time.sleep(0.05)
|
||||
finally:
|
||||
# Keep the PID reserved through the last group signal, even
|
||||
# when the cleanup grace period itself is interrupted.
|
||||
self._signal_group(signal.SIGKILL)
|
||||
finally:
|
||||
try:
|
||||
returncode = self.process.wait(timeout=10)
|
||||
finally:
|
||||
if self.queue is not None:
|
||||
self.queue.close()
|
||||
return returncode
|
||||
|
||||
|
||||
def invoke(adapter, action, request, timeout):
|
||||
"""The adapter writes bounded JSON separately; stderr/stdout remain raw evidence."""
|
||||
output = request.parent / f"{action}.json"
|
||||
with (request.parent / f"{action}.log").open("wb") as log:
|
||||
process = subprocess.Popen([str(adapter), action, str(request), str(output)],
|
||||
stdout=log, stderr=subprocess.STDOUT, start_new_session=True)
|
||||
process = OwnedCommand([str(adapter), action, str(request), str(output)], log)
|
||||
try:
|
||||
returncode = process.wait(timeout=timeout)
|
||||
returncode = process.wait(timeout)
|
||||
if returncode:
|
||||
raise subprocess.CalledProcessError(returncode, [str(adapter), action])
|
||||
finally:
|
||||
if process.poll() != 0:
|
||||
try:
|
||||
os.killpg(process.pid, signal.SIGTERM)
|
||||
except ProcessLookupError:
|
||||
pass
|
||||
try:
|
||||
process.wait(timeout=10)
|
||||
except subprocess.TimeoutExpired:
|
||||
os.killpg(process.pid, signal.SIGKILL)
|
||||
process.wait()
|
||||
return read_json(output)
|
||||
result = read_json(output)
|
||||
except BaseException:
|
||||
process.finish(terminate=True)
|
||||
raise
|
||||
else:
|
||||
# Successful prepare may intentionally leave adapter-owned services.
|
||||
process.finish()
|
||||
return result
|
||||
|
||||
|
||||
def validate_result(result, request, expected):
|
||||
@@ -188,7 +271,7 @@ def convergence(result):
|
||||
require(window["full_walk_objects"] > 0, "zero full walk reference")
|
||||
require(0 < window["budget_available_seconds"] <= window["window_end"] - window["window_start"],
|
||||
"invalid convergence budget window")
|
||||
return window["walk_objects"] / window["full_walk_objects"]
|
||||
return ratio(window["walk_objects"], window["full_walk_objects"], "convergence work")
|
||||
|
||||
|
||||
def evaluate(cells):
|
||||
@@ -229,6 +312,7 @@ def evaluate(cells):
|
||||
candidate_p2 = [value for cell, value in zip(group, p2) if cell["leg"].startswith("B")]
|
||||
p2_pending = any(value is None for value in candidate_p2)
|
||||
passed &= all(ratio(value, 1, "p2 work multiple") <= P2_WORK_MULTIPLE_LIMIT for value in candidate_p2 if value is not None)
|
||||
p2_report = [None if value is None else float(value) for value in p2]
|
||||
inconclusive |= noise or p2_pending
|
||||
if not noise and not passed:
|
||||
failed = True
|
||||
@@ -238,7 +322,7 @@ def evaluate(cells):
|
||||
"p99_regression": float(p99), "throughput_change": float(throughput),
|
||||
"thresholds": {key: float(value) for key, value in thresholds.items()},
|
||||
"p1": p1, "p2_max_work_multiple": float(P2_WORK_MULTIPLE_LIMIT),
|
||||
"p2_post_stop_work_multiples": p2})
|
||||
"p2_post_stop_work_multiples": p2_report})
|
||||
return ("fail" if failed else "inconclusive" if inconclusive else "pass"), comparisons
|
||||
|
||||
|
||||
@@ -254,12 +338,12 @@ def collect_live(prepared, request, request_path, adapter):
|
||||
"--samples", str(request["duration_seconds"] // 60 + 1), "--interval-secs", "60",
|
||||
"--out-dir", str(output)]
|
||||
with (request_path.parent / "collector.log").open("wb") as log:
|
||||
process = subprocess.Popen(args, stdout=log, stderr=subprocess.STDOUT, start_new_session=True)
|
||||
process = OwnedCommand(args, log)
|
||||
try:
|
||||
started = time.monotonic()
|
||||
result = invoke(adapter, "measure", request_path, request["duration_seconds"] + 300)
|
||||
require(time.monotonic() - started >= request["duration_seconds"], "measurement ended before required window")
|
||||
require(process.wait(timeout=120) == 0, "scanner collector failed")
|
||||
require(process.wait(120) == 0, "scanner collector failed")
|
||||
require(output.joinpath("scanner-summary.csv").stat().st_size > 0, "missing collector samples")
|
||||
samples = list((output / "status").glob("scanner-status.*.json"))
|
||||
require(len(samples) == request["duration_seconds"] // 60 + 1, "missing scanner samples")
|
||||
@@ -286,16 +370,7 @@ def collect_live(prepared, request, request_path, adapter):
|
||||
"missing per-host scanner metrics")
|
||||
return result
|
||||
finally:
|
||||
# Stop telemetry children as well when measurement fails or times out.
|
||||
try:
|
||||
os.killpg(process.pid, signal.SIGTERM)
|
||||
except ProcessLookupError:
|
||||
pass
|
||||
try:
|
||||
process.wait(timeout=10)
|
||||
except subprocess.TimeoutExpired:
|
||||
os.killpg(process.pid, signal.SIGKILL)
|
||||
process.wait()
|
||||
process.finish(terminate=True)
|
||||
|
||||
|
||||
def run(manifest, adapter, output, data_root):
|
||||
|
||||
@@ -3,13 +3,17 @@
|
||||
|
||||
import contextlib
|
||||
import copy
|
||||
import fcntl
|
||||
import io
|
||||
import json
|
||||
import os
|
||||
from pathlib import Path
|
||||
import shlex
|
||||
import signal
|
||||
import subprocess
|
||||
import sys
|
||||
import tempfile
|
||||
import time
|
||||
import unittest
|
||||
from unittest.mock import Mock, patch
|
||||
|
||||
@@ -20,9 +24,18 @@ def fake_adapter():
|
||||
action, request_path, output_path = sys.argv[1:]
|
||||
request = harness.read_json(Path(request_path))
|
||||
fault = os.environ.get("SCANNER_ABBA_TEST_FAULT", "")
|
||||
if fault == "stubborn-child" and action in ("prepare", "measure"):
|
||||
marker = Path(request_path).parent / "stubborn.pid"
|
||||
if os.fork() == 0:
|
||||
os.execv(sys.executable, [sys.executable, str(Path(__file__).resolve()), "--stubborn-worker", str(marker)])
|
||||
wait_for_marker(marker)
|
||||
if action == "measure":
|
||||
time.sleep(60)
|
||||
if action == "prepare":
|
||||
result = {"ready": True}
|
||||
elif action == "stop":
|
||||
if fault == "stubborn-child":
|
||||
reap_fixture(Path(request_path).parent / "stubborn.pid")
|
||||
result = {"stopped": True}
|
||||
elif action == "oracle":
|
||||
if fault == "oracle-exit":
|
||||
@@ -87,6 +100,36 @@ def fake_adapter():
|
||||
return 0
|
||||
|
||||
|
||||
def wait_for_marker(marker):
|
||||
deadline = time.monotonic() + 5
|
||||
while time.monotonic() < deadline:
|
||||
if marker.exists() and marker.stat().st_size:
|
||||
return
|
||||
time.sleep(0.01)
|
||||
raise AssertionError("fixture child did not become ready")
|
||||
|
||||
|
||||
def child_released(marker, timeout=1):
|
||||
deadline = time.monotonic() + timeout
|
||||
with marker.open("r+") as stream:
|
||||
while True:
|
||||
try:
|
||||
fcntl.flock(stream, fcntl.LOCK_EX | fcntl.LOCK_NB)
|
||||
return True
|
||||
except BlockingIOError:
|
||||
if time.monotonic() >= deadline:
|
||||
return False
|
||||
time.sleep(0.01)
|
||||
|
||||
|
||||
def reap_fixture(marker):
|
||||
if marker.exists() and marker.stat().st_size and not child_released(marker, timeout=0):
|
||||
# The unique file lock proves the original fixture process still owns this PID.
|
||||
os.kill(int(marker.read_text()), signal.SIGKILL)
|
||||
if not child_released(marker, timeout=5):
|
||||
raise AssertionError("fixture child did not release its process-owned lock")
|
||||
|
||||
|
||||
class ScannerAbbaTest(unittest.TestCase):
|
||||
def setUp(self):
|
||||
self.temp = tempfile.TemporaryDirectory()
|
||||
@@ -110,6 +153,112 @@ class ScannerAbbaTest(unittest.TestCase):
|
||||
with patch.dict(os.environ, {"SCANNER_ABBA_TEST_FAULT": fault}), contextlib.redirect_stdout(io.StringIO()):
|
||||
return harness.run(copy.deepcopy(self.manifest), self.adapter, self.root / "out", self.root / "data")
|
||||
|
||||
def test_adapter_timeout_reaps_group_after_parent_exits_on_term(self):
|
||||
request = self.root / "request.json"
|
||||
harness.write_json(request, {})
|
||||
marker = self.root / "stubborn.pid"
|
||||
try:
|
||||
with patch.dict(os.environ, {"SCANNER_ABBA_TEST_FAULT": "stubborn-child"}):
|
||||
with self.assertRaises(subprocess.TimeoutExpired):
|
||||
harness.invoke(self.adapter, "measure", request, 3)
|
||||
wait_for_marker(marker)
|
||||
self.assertTrue(child_released(marker), "TERM-exited parent left its TERM-ignoring child alive")
|
||||
finally:
|
||||
reap_fixture(marker)
|
||||
|
||||
def test_collector_failure_reaps_group_after_parent_exits_on_term(self):
|
||||
request = self.root / "request.json"
|
||||
harness.write_json(request, {})
|
||||
marker = self.root / "stubborn.pid"
|
||||
collector = self.root / "run_scanner_validation_harness.sh"
|
||||
command = [sys.executable, str(self.adapter), "measure", str(request), str(self.root / "unused.json")]
|
||||
collector.write_text("#!/usr/bin/env bash\nexec " + shlex.join(command) + "\n")
|
||||
|
||||
def failed_measure(*_):
|
||||
wait_for_marker(marker)
|
||||
raise ValueError("injected measurement failure")
|
||||
|
||||
try:
|
||||
with patch.dict(os.environ, {"SCANNER_ABBA_TEST_FAULT": "stubborn-child"}), \
|
||||
patch.object(harness, "__file__", str(self.root / "scanner_abba.py")), \
|
||||
patch.object(harness, "invoke", side_effect=failed_measure):
|
||||
with self.assertRaisesRegex(ValueError, "injected measurement failure"):
|
||||
harness.collect_live({"collector": {"alias": "fixture", "endpoint": "fixture", "metrics_endpoints": "fixture"}},
|
||||
{"duration_seconds": 900}, request, self.adapter)
|
||||
self.assertTrue(child_released(marker), "collector parent exit did not end its telemetry child")
|
||||
finally:
|
||||
reap_fixture(marker)
|
||||
|
||||
def test_successful_prepare_keeps_service_alive(self):
|
||||
request = self.root / "request.json"
|
||||
harness.write_json(request, {})
|
||||
marker = self.root / "stubborn.pid"
|
||||
try:
|
||||
with patch.dict(os.environ, {"SCANNER_ABBA_TEST_FAULT": "stubborn-child"}):
|
||||
self.assertEqual(harness.invoke(self.adapter, "prepare", request, 5), {"ready": True})
|
||||
self.assertFalse(child_released(marker, timeout=0), "successful prepare must preserve its service")
|
||||
self.assertEqual(harness.invoke(self.adapter, "stop", request, 5), {"stopped": True})
|
||||
self.assertTrue(child_released(marker), "adapter stop must release its service")
|
||||
finally:
|
||||
reap_fixture(marker)
|
||||
|
||||
def test_reaped_owner_never_signals_a_reused_process_group(self):
|
||||
with (self.root / "owner.log").open("wb") as log:
|
||||
owner = harness.OwnedCommand([sys.executable, "-c", "pass"], log)
|
||||
self.assertEqual(owner.wait(5), 0)
|
||||
self.assertEqual(owner.finish(), 0)
|
||||
with patch.object(harness.os, "killpg", side_effect=AssertionError("released PGID must not be signalled")):
|
||||
self.assertEqual(owner.finish(terminate=True), 0)
|
||||
|
||||
def test_cleanup_interruption_still_kills_group_and_reaps_leader(self):
|
||||
request = self.root / "request.json"
|
||||
harness.write_json(request, {})
|
||||
marker = self.root / "stubborn.pid"
|
||||
original_sleep = time.sleep
|
||||
interrupted = False
|
||||
|
||||
def interrupt_once(delay):
|
||||
nonlocal interrupted
|
||||
if not interrupted:
|
||||
interrupted = True
|
||||
raise KeyboardInterrupt
|
||||
original_sleep(delay)
|
||||
|
||||
with (self.root / "interrupted.log").open("wb") as log:
|
||||
with patch.dict(os.environ, {"SCANNER_ABBA_TEST_FAULT": "stubborn-child"}):
|
||||
owner = harness.OwnedCommand([str(self.adapter), "measure", str(request), str(self.root / "unused.json")], log)
|
||||
try:
|
||||
wait_for_marker(marker)
|
||||
with patch.object(harness.time, "sleep", side_effect=interrupt_once):
|
||||
with self.assertRaises(KeyboardInterrupt):
|
||||
owner.finish(terminate=True)
|
||||
self.assertTrue(child_released(marker), "cleanup cancellation left its child alive")
|
||||
self.assertIsNotNone(owner.process.returncode, "cleanup cancellation must reap its leader")
|
||||
finally:
|
||||
reap_fixture(marker)
|
||||
owner.process.wait(timeout=5)
|
||||
|
||||
def test_constructor_failure_after_gate_release_kills_group(self):
|
||||
request = self.root / "request.json"
|
||||
harness.write_json(request, {})
|
||||
marker = self.root / "stubborn.pid"
|
||||
original_write = os.write
|
||||
|
||||
def release_then_fail(fd, data):
|
||||
original_write(fd, data)
|
||||
wait_for_marker(marker)
|
||||
raise OSError("injected failure after gate release")
|
||||
|
||||
try:
|
||||
with (self.root / "construction.log").open("wb") as log, \
|
||||
patch.dict(os.environ, {"SCANNER_ABBA_TEST_FAULT": "stubborn-child"}), \
|
||||
patch.object(harness.os, "write", side_effect=release_then_fail):
|
||||
with self.assertRaisesRegex(OSError, "injected failure after gate release"):
|
||||
harness.OwnedCommand([str(self.adapter), "measure", str(request), str(self.root / "unused.json")], log)
|
||||
self.assertTrue(child_released(marker), "initialization failure left its child alive")
|
||||
finally:
|
||||
reap_fixture(marker)
|
||||
|
||||
def test_complete_synthetic_matrix_is_not_performance_evidence(self):
|
||||
self.assertEqual(self.run_harness(), 0)
|
||||
report = harness.read_json(self.root / "out/report.json")
|
||||
@@ -201,10 +350,9 @@ class ScannerAbbaTest(unittest.TestCase):
|
||||
else:
|
||||
harness.write_json(sample, payload)
|
||||
process = Mock(pid=123, wait=Mock(return_value=1 if name == "collector-exit" else 0))
|
||||
with patch.object(harness.subprocess, "Popen", return_value=process), \
|
||||
with patch.object(harness, "OwnedCommand", return_value=process), \
|
||||
patch.object(harness, "invoke", return_value={"sample_count": 10}), \
|
||||
patch.object(harness.time, "monotonic", side_effect=(0, 900)), \
|
||||
patch.object(harness.os, "killpg"):
|
||||
patch.object(harness.time, "monotonic", side_effect=(0, 900)):
|
||||
if error:
|
||||
with self.assertRaisesRegex(ValueError, error):
|
||||
harness.collect_live(prepared, {"duration_seconds": 900}, self.root / "request.json", self.adapter)
|
||||
@@ -212,6 +360,8 @@ class ScannerAbbaTest(unittest.TestCase):
|
||||
self.assertEqual(harness.collect_live(prepared, {"duration_seconds": 900},
|
||||
self.root / "request.json", self.adapter), {"sample_count": 10})
|
||||
|
||||
process.finish.assert_called_once_with(terminate=True)
|
||||
|
||||
def test_unstable_p1_work_control_is_inconclusive(self):
|
||||
with patch.object(harness, "SCENARIOS", ("cold-hot",)):
|
||||
self.assertEqual(self.run_harness("unstable-p1-control"), 3)
|
||||
@@ -259,6 +409,14 @@ class ScannerAbbaTest(unittest.TestCase):
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
if len(sys.argv) == 3 and sys.argv[1] == "--stubborn-worker":
|
||||
signal.signal(signal.SIGTERM, signal.SIG_IGN)
|
||||
with Path(sys.argv[2]).open("w+") as marker:
|
||||
fcntl.flock(marker, fcntl.LOCK_EX)
|
||||
marker.write(str(os.getpid()))
|
||||
marker.flush()
|
||||
while True:
|
||||
time.sleep(1)
|
||||
if len(sys.argv) == 4 and sys.argv[1] in ("prepare", "measure", "oracle", "stop"):
|
||||
sys.exit(fake_adapter())
|
||||
unittest.main()
|
||||
|
||||
Reference in New Issue
Block a user